初始化项目,由ModelHub XC社区提供模型
Model: distil-labs/distil-qwen3-1.7b-customer-support-deferral Source: Original Platform
This commit is contained in:
36
.gitattributes
vendored
Normal file
36
.gitattributes
vendored
Normal file
@@ -0,0 +1,36 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
31
LICENSE
Normal file
31
LICENSE
Normal file
@@ -0,0 +1,31 @@
|
||||
GENERAL TERMS AND CONDITIONS
|
||||
|
||||
Note that if you want to use the Commercial licence, please contact us at contact@distillabs.ai
|
||||
|
||||
- Model License Terms -
|
||||
|
||||
R&D License
|
||||
|
||||
1. SERVICES, PRICES AND PAYMENT
|
||||
|
||||
1.1 The Customer pays a one-time license fee, as indicated in the check-out process, for running of one (1) training process of the selected Base Model using Customer Data (“License Fee”).
|
||||
|
||||
1.2 The License Fee shall be due for payment in advance. The Customer shall only be permitted to set off against payment claims of Distil Labs if the Customer’s claims are undisputed or have become res judicata.
|
||||
|
||||
2. MODEL LICENSE: R&D LICENSE
|
||||
|
||||
2.1 Subject to Customer’s payment of the license fee, Distil Labs grants to Customer the Model License (as defined below). For clarification, Distil Labs retains any other rights in its software or know- how, in particular in the codebase needed for the fine-tuning of the Trained Model.
|
||||
|
||||
2.2 Subject to the requirements of the Base Model License (cf. Section 2.5 below), Distil Labs transfers to the Customer the perpetual, non-exclusive usage right to the Trained Model for non-commercial purposes of prototyping and research & development. The Parties agree, that commercial purposes include deployment in production externally (to be used by Customer’s customers paid or free of charge) or internally (as a tool for Customer’s employees). The territorial scope of the license is limited to the use within the United States of America and the European Economic Area including all member states of the European Union (“Model License”).
|
||||
|
||||
2.3 The Model License for non-commercial purposes of prototyping and research & development shall include (i) the non-exclusive right to permanent or temporary reproduction, in whole or in part, by any means and in any form (e.g. permanent and/or volatile storage on electrical, electromagnetic, optical storage media, such as any type of SDD, HDD, DVD, memory cards, USB sticks), (ii) the non-exclusive right to distribution in any form, media and by any means regardless of whether the distribution is in tangible or intangible form, in particular to transmit the Trained Model via wired and wireless networks (e.g. for download from internet or intranet by wire or wireless means including broadband, cable, fiberglass, WIFI, LTE, 5G, satellite internet, other data networks), and (iii) the non-exclusive right of making available to the public in such a way that members of the public can access it from places and at times of their choice (e.g. by web or mobile app, virtual or augmented reality, cloud storage, cloud hosting, decentralized hosting, non-fungible token, application service providing, software as a service, or cloud computing). The license shall also contain, to the extent necessary for prototyping and research & development, the right to adapt and modify the Trained Model subject to the limitation in Section 2.4 and 2.5 below, to further develop the Trained Model including changes to functions or appearance, adapt to other software versions, to exchange parts of the Trained Model or combine the Trained Model with other results of work and to use the results in the same way as the original Trained Model. Any derived models from the Trained Model shall retain this model license.
|
||||
|
||||
2.4 The Customer shall not, without the prior written consent of Distil Labs:
|
||||
|
||||
2.4.1 train, fine-tune, re-train, or otherwise modify the Trained Model, unless for purpose of research & development;
|
||||
|
||||
2.4.2 use the Trained Model or any part thereof to create derivative models or services that compete with those of Distil Labs;
|
||||
|
||||
2.4.3 circumvent any technical restrictions embedded in the Trained Model or Base Model that are designed to enforce usage limitations.
|
||||
|
||||
2.5 The Parties acknowledge and agree that the Trained Model is developed from Base Models which are supplied by a third party. Therefore, the Model License is subject to the restrictions resulting from the open-source or any other applicable license of the Base Model (“Base Model License”) and the Customer must use the Trained Model in compliance with the Base Model License. In particular, the Customer must oblige their clients to compliance with the Base Model License in any case of transferring or sublicensing the rights to or making available in any way the Trained Model. The applicable Base Model License is defined in the Training Configuration and will be provided for download. The Customer agrees to indemnify Distil Labs for any and all claims brought by the Base Model provider for violations of the Base Model License.
|
||||
59
Modelfile
Normal file
59
Modelfile
Normal file
@@ -0,0 +1,59 @@
|
||||
|
||||
FROM ./model.gguf
|
||||
|
||||
TEMPLATE """{{- $lastUserIdx := -1 -}}
|
||||
{{- range $idx, $msg := .Messages -}}
|
||||
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
|
||||
{{- end }}
|
||||
{{- if or .System .Tools }}<|im_start|>system
|
||||
{{ if .System }}
|
||||
{{ .System }}
|
||||
{{- end }}
|
||||
{{- if .Tools }}
|
||||
|
||||
# Tools
|
||||
|
||||
You may call one or more functions to assist with the user query.
|
||||
|
||||
You are provided with function signatures within <tools></tools> XML tags:
|
||||
<tools>
|
||||
{{- range .Tools }}
|
||||
{"type": "function", "function": {{ .Function }}}
|
||||
{{- end }}
|
||||
</tools>
|
||||
|
||||
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
|
||||
<tool_call>
|
||||
{"name": <function-name>, "arguments": <args-json-object>}
|
||||
</tool_call>
|
||||
{{- end -}}
|
||||
<|im_end|>
|
||||
{{ end }}
|
||||
{{- range $i, $_ := .Messages }}
|
||||
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
|
||||
{{- if eq .Role "user" }}<|im_start|>user
|
||||
{{ .Content }}<|im_end|>
|
||||
{{ else if eq .Role "assistant" }}<|im_start|>assistant
|
||||
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
|
||||
<think>{{ .Thinking }}</think>
|
||||
{{ end -}}
|
||||
{{ if .Content }}{{ .Content }}
|
||||
{{- else if .ToolCalls }}<tool_call>
|
||||
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
|
||||
{{ end }}</tool_call>
|
||||
{{- end }}{{ if not $last }}<|im_end|>
|
||||
{{ end }}
|
||||
{{- else if eq .Role "tool" }}<|im_start|>user
|
||||
<tool_response>
|
||||
{{ .Content }}
|
||||
</tool_response><|im_end|>
|
||||
{{ end }}
|
||||
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
|
||||
{{ if and $.IsThinkSet (not $.Think) -}}
|
||||
<think>
|
||||
|
||||
</think>
|
||||
|
||||
{{ end -}}
|
||||
{{ end }}
|
||||
{{- end }}"""
|
||||
200
README.md
Normal file
200
README.md
Normal file
@@ -0,0 +1,200 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
base_model: Qwen/Qwen3-1.7B
|
||||
tags:
|
||||
- tool-calling
|
||||
- function-calling
|
||||
- customer-support
|
||||
- airline
|
||||
- model-cascade
|
||||
- deferral
|
||||
- distil-labs
|
||||
language:
|
||||
- en
|
||||
pipeline_tag: text-generation
|
||||
library_name: transformers
|
||||
---
|
||||
|
||||
# Distil-Qwen3-1.7B-Customer-Support-Deferral
|
||||
|
||||
A fine-tuned Qwen3-1.7B model for multi-turn **airline customer support** that runs as
|
||||
the small tier of a **two-model cascade**. It handles most support turns itself and
|
||||
**defers genuinely-hard turns to a larger model** by emitting a `defer_to_larger_model`
|
||||
tool call. Trained with knowledge distillation from a large teacher model (`zai.glm-5`).
|
||||
|
||||
Every assistant action is a single tool call, including talking to the customer via
|
||||
`respond_to_user`, so the model can be driven by a thin, deterministic orchestrator.
|
||||
|
||||
## Results
|
||||
|
||||
Evaluated on a held-out set of airline customer-support turns, scored by an independent
|
||||
GLM-5 judge (score = fraction of responses rated correct).
|
||||
|
||||
| System | Quality | Frontier-model calls |
|
||||
|---|:---:|:---:|
|
||||
| Frontier model alone (GLM-5) | 0.80 | 100% |
|
||||
| **This model + escalation (local)** | **~0.75** | **~4%** |
|
||||
| Untrained Qwen3-1.7B | 0.42 | 0% |
|
||||
|
||||
Fine-tuning lifts the local 1.7B from 0.42 to ~0.75 (closing roughly 85% of the gap to its
|
||||
frontier-scale teacher), while running ~96% of turns locally and escalating only the
|
||||
hardest ~4% to the larger model.
|
||||
|
||||
Score is a reference-free LLM-as-a-judge rating on held-out turns, not exact-match accuracy.
|
||||
The escalation (`defer_to_larger_model`) is a cost/safety mechanism that reserves the frontier
|
||||
model for the hard minority, not a quality boost over the small model alone.
|
||||
|
||||
## What the model does
|
||||
|
||||
Given the airline policy (as the system prompt), the available tools, and the
|
||||
conversation so far, the model produces the next single tool call:
|
||||
|
||||
- **Talk to the customer**: `respond_to_user(message=...)` (terminal, ends the turn).
|
||||
- **Act / look up**: `get_reservation_details`, `book_reservation`, `send_certificate`, and so on.
|
||||
- **Reason silently**: `think(thought=...)`.
|
||||
- **Escalate to a larger model**: `defer_to_larger_model(reason=...)` on turns whose
|
||||
correct action depends on non-obvious policy eligibility, combining several rules, a
|
||||
multi-step calculation, or a genuinely ambiguous judgement call.
|
||||
- **Hand off to a human**: `transfer_to_human_agents(summary=...)` for out-of-scope
|
||||
requests or explicit human requests (distinct from deferral, which stays automated).
|
||||
|
||||
### Deferral vs. human transfer
|
||||
|
||||
`defer_to_larger_model` is a **capability escalation**: a larger, more capable model takes
|
||||
over the same conversation with the same tools and policy, and the customer keeps being
|
||||
served automatically. `transfer_to_human_agents` is for requests outside the tools' scope
|
||||
or when the user asks for a person. Judging *when* to defer, by the absolute structure of
|
||||
the problem rather than the model's own confidence, is the core skill this model is
|
||||
distilled for.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Using Transformers
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
model_id = "distil-labs/distil-qwen3-1.7b-customer-support-deferral"
|
||||
model = AutoModelForCausalLM.from_pretrained(model_id)
|
||||
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
||||
|
||||
# The full airline policy (system prompt) and the 16 tool schemas ship with the demo
|
||||
# app as `job_description.json`. Wrap the policy in the distil tool-calling preamble:
|
||||
TASK_DESCRIPTION = "# Airline Agent Policy\n... (see job_description.json) ..."
|
||||
SYSTEM = (
|
||||
"You are a tool-calling model working on:\n"
|
||||
f"<task_description>{TASK_DESCRIPTION}</task_description>\n\n"
|
||||
"Respond to the conversation history by generating an appropriate tool call that "
|
||||
"satisfies the user request. Generate only the tool call according to the provided "
|
||||
"tool schema, do not generate anything else. Always respond with a tool call."
|
||||
)
|
||||
TOOLS = [ ... ] # 16 tools from job_description.json
|
||||
|
||||
messages = [
|
||||
{"role": "system", "content": SYSTEM},
|
||||
{"role": "user", "content": "Can I get a refund for reservation 8JX2WO?"},
|
||||
]
|
||||
text = tokenizer.apply_chat_template(
|
||||
messages, tools=TOOLS, tokenize=False, add_generation_prompt=True, enable_thinking=False,
|
||||
)
|
||||
inputs = tokenizer(text, return_tensors="pt")
|
||||
outputs = model.generate(**inputs, max_new_tokens=256, temperature=0)
|
||||
print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:], skip_special_tokens=True))
|
||||
# <tool_call>
|
||||
# {"name": "defer_to_larger_model", "arguments": {"reason": "refund eligibility depends on fare class + travel insurance"}}
|
||||
# </tool_call>
|
||||
```
|
||||
|
||||
### Using the Demo App
|
||||
|
||||
This model powers the [Flexible Customer Support Bot](https://github.com/distil-labs/distil-dual-size-customer-support-bot)
|
||||
demo, a terminal cascade where a local SLM handles most airline-support turns and defers
|
||||
hard turns to a larger, OpenAI-compatible model.
|
||||
|
||||
### Using llama.cpp
|
||||
|
||||
For local serving, use the GGUF build at
|
||||
[distil-labs/distil-qwen3-1.7b-customer-support-deferral-gguf](https://huggingface.co/distil-labs/distil-qwen3-1.7b-customer-support-deferral-gguf):
|
||||
|
||||
```bash
|
||||
llama-server --model distil-qwen3-1.7b-customer-support-deferral.gguf --port 8000 --jinja
|
||||
```
|
||||
|
||||
## Model Details
|
||||
|
||||
| Property | Value |
|
||||
|---|---|
|
||||
| Base Model | [Qwen/Qwen3-1.7B](https://huggingface.co/Qwen/Qwen3-1.7B) |
|
||||
| Parameters | 1.7 billion |
|
||||
| Architecture | Qwen3ForCausalLM |
|
||||
| Context Length | 40,960 tokens |
|
||||
| Precision | bfloat16 (merged) |
|
||||
| Teacher Model | GLM-5 (`zai.glm-5`) |
|
||||
| Task | Multi-turn tool calling (closed book) with model deferral |
|
||||
|
||||
## Training
|
||||
|
||||
The model is distilled with the [Distil Labs](https://distillabs.ai/) platform:
|
||||
|
||||
1. **Traces**: airline customer-support conversations (tau-bench airline tool set),
|
||||
processed and cleaned through the distil trace-processing pipeline.
|
||||
2. **Deferral signal**: a `defer_to_larger_model` tool and policy guidance, so the teacher
|
||||
marks genuinely-hard turns for escalation while the student learns the rest.
|
||||
3. **Synthetic expansion + fine-tuning**: distilled onto Qwen3-1.7B with GLM-5 as teacher.
|
||||
|
||||
### Supported Functions (16 tools)
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `book_reservation` | Book a new flight reservation |
|
||||
| `cancel_reservation` | Cancel an existing reservation |
|
||||
| `get_reservation_details` | Look up a reservation |
|
||||
| `get_user_details` | Look up a user / profile |
|
||||
| `list_all_airports` | List supported airports |
|
||||
| `search_direct_flight` | Search direct flights |
|
||||
| `search_onestop_flight` | Search one-stop flights |
|
||||
| `update_reservation_flights` | Change flights on a reservation |
|
||||
| `update_reservation_baggages` | Update baggage on a reservation |
|
||||
| `update_reservation_passengers` | Update passengers on a reservation |
|
||||
| `send_certificate` | Issue a travel certificate / compensation |
|
||||
| `calculate` | Perform an arithmetic calculation |
|
||||
| `think` | Private step-by-step reasoning (no side effects) |
|
||||
| `respond_to_user` | Send a natural-language message to the customer (ends the turn) |
|
||||
| `transfer_to_human_agents` | Hand off to a human agent (out-of-scope / explicit request) |
|
||||
| `defer_to_larger_model` | Escalate this turn to a larger model (capability escalation) |
|
||||
|
||||
## Use Cases
|
||||
|
||||
- Cost-efficient customer-support assistants: a small local model handles the bulk of
|
||||
traffic, a larger model is invoked only on the hard minority of turns.
|
||||
- Any multi-turn tool-calling task with a bounded tool catalog and a difficulty signal
|
||||
worth routing on.
|
||||
|
||||
## Limitations
|
||||
|
||||
- English airline customer-support only, not a general-purpose tool caller.
|
||||
- Deferral calibration depends on the policy and tool catalog it was trained with.
|
||||
|
||||
## License
|
||||
|
||||
Released under the Apache 2.0 license. See `STUDENT_LICENSE` (base model) and
|
||||
`TEACHER_LICENSE` (teacher model) for upstream terms.
|
||||
|
||||
## Links
|
||||
|
||||
- [GGUF model](https://huggingface.co/distil-labs/distil-qwen3-1.7b-customer-support-deferral-gguf)
|
||||
- [Demo app](https://github.com/distil-labs/distil-dual-size-customer-support-bot)
|
||||
- [Distil Labs Website](https://distillabs.ai/)
|
||||
- [Hugging Face](https://huggingface.co/distil-labs)
|
||||
|
||||
## Citation
|
||||
|
||||
```bibtex
|
||||
@misc{distil-qwen3-1.7b-customer-support-deferral,
|
||||
author = {Distil Labs},
|
||||
title = {Distil-Qwen3-1.7B-Customer-Support-Deferral: A Fine-tuned SLM for Airline Support with Model Deferral},
|
||||
year = {2026},
|
||||
publisher = {Hugging Face},
|
||||
url = {https://huggingface.co/distil-labs/distil-qwen3-1.7b-customer-support-deferral}
|
||||
}
|
||||
```
|
||||
13
STUDENT_LICENSE
Normal file
13
STUDENT_LICENSE
Normal file
@@ -0,0 +1,13 @@
|
||||
Copyright 2023 Qwen
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
9
TEACHER_LICENSE
Normal file
9
TEACHER_LICENSE
Normal file
@@ -0,0 +1,9 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2025 Zhipu AI
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
89
chat_template.jinja
Normal file
89
chat_template.jinja
Normal file
@@ -0,0 +1,89 @@
|
||||
{%- if tools %}
|
||||
{{- '<|im_start|>system\n' }}
|
||||
{%- if messages[0].role == 'system' %}
|
||||
{{- messages[0].content + '\n\n' }}
|
||||
{%- endif %}
|
||||
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||
{%- for tool in tools %}
|
||||
{{- "\n" }}
|
||||
{{- tool | tojson }}
|
||||
{%- endfor %}
|
||||
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||
{%- else %}
|
||||
{%- if messages[0].role == 'system' %}
|
||||
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||
{%- for message in messages[::-1] %}
|
||||
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||
{%- set ns.multi_step_tool = false %}
|
||||
{%- set ns.last_query_index = index %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- for message in messages %}
|
||||
{%- if message.content is string %}
|
||||
{%- set content = message.content %}
|
||||
{%- else %}
|
||||
{%- set content = '' %}
|
||||
{%- endif %}
|
||||
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
||||
{%- elif message.role == "assistant" %}
|
||||
{%- set reasoning_content = '' %}
|
||||
{%- if message.reasoning_content is string %}
|
||||
{%- set reasoning_content = message.reasoning_content %}
|
||||
{%- else %}
|
||||
{%- if '</think>' in content %}
|
||||
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- if loop.index0 > ns.last_query_index %}
|
||||
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||
{%- endif %}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||
{%- endif %}
|
||||
{%- if message.tool_calls %}
|
||||
{%- for tool_call in message.tool_calls %}
|
||||
{%- if (loop.first and content) or (not loop.first) %}
|
||||
{{- '\n' }}
|
||||
{%- endif %}
|
||||
{%- if tool_call.function %}
|
||||
{%- set tool_call = tool_call.function %}
|
||||
{%- endif %}
|
||||
{{- '<tool_call>\n{"name": "' }}
|
||||
{{- tool_call.name }}
|
||||
{{- '", "arguments": ' }}
|
||||
{%- if tool_call.arguments is string %}
|
||||
{{- tool_call.arguments }}
|
||||
{%- else %}
|
||||
{{- tool_call.arguments | tojson }}
|
||||
{%- endif %}
|
||||
{{- '}\n</tool_call>' }}
|
||||
{%- endfor %}
|
||||
{%- endif %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- elif message.role == "tool" %}
|
||||
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||
{{- '<|im_start|>user' }}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_response>\n' }}
|
||||
{{- content }}
|
||||
{{- '\n</tool_response>' }}
|
||||
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- if add_generation_prompt %}
|
||||
{{- '<|im_start|>assistant\n' }}
|
||||
{%- if enable_thinking is defined and enable_thinking is false %}
|
||||
{{- '<think>\n\n</think>\n\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
66
config.json
Normal file
66
config.json
Normal file
@@ -0,0 +1,66 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token": "<|endoftext|>",
|
||||
"bos_token_id": 151643,
|
||||
"dtype": "bfloat16",
|
||||
"eos_token": "<|im_end|>",
|
||||
"eos_token_id": 151645,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"layer_types": [
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention"
|
||||
],
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"model_type": "qwen3",
|
||||
"num_attention_heads": 16,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"pad_token_id": 151643,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_parameters": {
|
||||
"rope_theta": 1000000,
|
||||
"rope_type": "default"
|
||||
},
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"transformers_version": "5.10.2",
|
||||
"use_cache": false,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
13
generation_config.json
Normal file
13
generation_config.json
Normal file
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"bos_token_id": 151643,
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151645,
|
||||
151643
|
||||
],
|
||||
"pad_token_id": 151643,
|
||||
"temperature": 0.6,
|
||||
"top_k": 20,
|
||||
"top_p": 0.95,
|
||||
"transformers_version": "5.10.2"
|
||||
}
|
||||
151388
merges.txt
Normal file
151388
merges.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5a102236925d6dac96acffcb92265945f23f1cb8956adfb515e7313d0314bae9
|
||||
size 3441185608
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
|
||||
size 11422650
|
||||
74
tokenizer_config.json
Normal file
74
tokenizer_config.json
Normal file
@@ -0,0 +1,74 @@
|
||||
{
|
||||
"add_prefix_space": false,
|
||||
"backend": "tokenizers",
|
||||
"bos_token": "<|endoftext|>",
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|im_end|>",
|
||||
"errors": "replace",
|
||||
"extra_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"is_local": false,
|
||||
"local_files_only": false,
|
||||
"model_max_length": 131072,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"padding_side": "left",
|
||||
"response_schema": {
|
||||
"properties": {
|
||||
"content": {
|
||||
"type": "string"
|
||||
},
|
||||
"reasoning_content": {
|
||||
"type": "string"
|
||||
},
|
||||
"role": {
|
||||
"const": "assistant"
|
||||
},
|
||||
"tool_calls": {
|
||||
"items": {
|
||||
"properties": {
|
||||
"function": {
|
||||
"properties": {
|
||||
"arguments": {
|
||||
"additionalProperties": {},
|
||||
"type": "object"
|
||||
},
|
||||
"name": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"type": "object"
|
||||
},
|
||||
"type": {
|
||||
"const": "function"
|
||||
}
|
||||
},
|
||||
"type": "object",
|
||||
"x-parser": "json",
|
||||
"x-parser-args": {
|
||||
"transform": "{type: 'function', function: @}"
|
||||
}
|
||||
},
|
||||
"type": "array",
|
||||
"x-regex-iterator": "<tool_call>\\s*(.+?)\\s*</tool_call>"
|
||||
}
|
||||
},
|
||||
"type": "object",
|
||||
"x-regex": "^(?:<think>\\n?(?:(?P<reasoning_content>.*?\\S.*?)\\n?|[\\s]*)</think>\\s*)?(?P<content>(?:(?!<tool_call>)[\\s\\S])*?)(?:\\n(?=<tool_call>))?(?=(?:<tool_call>|<\\|im_end\\|>|$))(?P<tool_calls>(?:<tool_call>(?:(?!</tool_call>)[\\s\\S])+</tool_call>\\s*)+)?\\s*(?:<\\|im_end\\|>|$)"
|
||||
},
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"unk_token": null
|
||||
}
|
||||
1
vocab.json
Normal file
1
vocab.json
Normal file
File diff suppressed because one or more lines are too long
Reference in New Issue
Block a user