初始化项目,由ModelHub XC社区提供模型
Model: 0xSero/GLM-4.7-REAP-50-W4A16 Source: Original Platform
This commit is contained in:
70
.gitattributes
vendored
Normal file
70
.gitattributes
vendored
Normal file
@@ -0,0 +1,70 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zstandard filter=lfs diff=lfs merge=lfs -text
|
||||
*.tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
*.db* filter=lfs diff=lfs merge=lfs -text
|
||||
*.ark* filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.gguf* filter=lfs diff=lfs merge=lfs -text
|
||||
*.ggml filter=lfs diff=lfs merge=lfs -text
|
||||
*.llamafile* filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
model-00018-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00003-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00001-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
model-00006-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00019-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00004-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00011-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00008-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00009-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00017-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00015-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00020-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00012-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00016-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model.safetensors.index.json filter=lfs diff=lfs merge=lfs -text
|
||||
model-00014-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00013-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00002-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00007-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00005-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00010-of-00020.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
314
README.md
Normal file
314
README.md
Normal file
@@ -0,0 +1,314 @@
|
||||
---
|
||||
language:
|
||||
- en
|
||||
library_name: transformers
|
||||
tags:
|
||||
- 4bit
|
||||
- MOE
|
||||
- autoround
|
||||
- cerebras
|
||||
- code
|
||||
- compression
|
||||
- function-calling
|
||||
- glm
|
||||
- glm4
|
||||
- gptq
|
||||
- pruning
|
||||
- quantized
|
||||
- reap
|
||||
- w4a16
|
||||
license: apache-2.0
|
||||
pipeline_tag: text-generation
|
||||
base_model:
|
||||
- 0xSero/GLM-4.7-185B
|
||||
datasets:
|
||||
- NeelNanda/pile-10k
|
||||
base_model_relation: quantized
|
||||
---
|
||||
|
||||
> [!TIP]
|
||||
> **[Support this work →](https://donate.sybilsolutions.ai)** · [X](https://x.com/0xsero) · [GitHub](https://github.com/0xsero) · [REAP paper](https://arxiv.org/abs/2510.13999) · [Cerebras REAP](https://huggingface.co/collections/cerebras/cerebras-reap)
|
||||
|
||||
# GLM-4.7-185B-W4A16
|
||||
|
||||
W4A16 quantization of [0xSero/GLM-4.7-185B](https://huggingface.co/0xSero/GLM-4.7-185B).
|
||||
|
||||
## At a glance
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Base model | [0xSero/GLM-4.7-185B](https://huggingface.co/0xSero/GLM-4.7-185B) |
|
||||
| Format | W4A16 |
|
||||
| Total params | **185B** |
|
||||
| Active / token | — |
|
||||
| Experts / layer | 80 |
|
||||
| Layers | 92 |
|
||||
| Hidden size | 5120 |
|
||||
| Context | 202,752 |
|
||||
| On-disk size | 99 GB |
|
||||
|
||||
## Which variant should I pick?
|
||||
|
||||
| Variant | Format | Link |
|
||||
|---|---|---|
|
||||
| `GLM-4.7-185B` | BF16 | [link](https://huggingface.co/0xSero/GLM-4.7-185B) |
|
||||
| `GLM-4.7-185B-W4A16` **(this)** | W4A16 | [link](https://huggingface.co/0xSero/GLM-4.7-185B-W4A16) |
|
||||
| `GLM-4.7-202B` | BF16 | [link](https://huggingface.co/0xSero/GLM-4.7-202B) |
|
||||
| `GLM-4.7-218B-W4A16` | W4A16 | [link](https://huggingface.co/0xSero/GLM-4.7-218B-W4A16) |
|
||||
| `GLM-4.7-REAP-40-W4A16` | W4A16 | [link](https://huggingface.co/0xSero/GLM-4.7-REAP-40-W4A16) |
|
||||
|
||||
<p align="center">
|
||||
<em>𓌳 <strong>REAP</strong>𓌳 the Experts: Why Pruning Prevails for One-Shot MoE Compression</em><br>
|
||||
<a href="https://arxiv.org/abs/2510.13999">📄 Paper</a> • <a href="https://github.com/CerebrasResearch/reap">💻 Code</a> • <a href="https://www.cerebras.ai/blog/reap">📝 Blog</a>
|
||||
</p>
|
||||
|
||||
# GLM-4.7-REAP-50-W4A16
|
||||
|
||||
## ✨ Highlights
|
||||
|
||||
**50% Expert-Pruned + INT4 Quantized** — Double compression for efficient deployment.
|
||||
|
||||
- **~6.5x Total Compression**: 700GB → ~92GB
|
||||
- **REAP + AutoRound**: Expert pruning + weight quantization
|
||||
- **Optimized for Code & Tools**: Calibrated on code generation and function calling
|
||||
- **Lower VRAM**: Fits on 2-4x fewer GPUs than BF16
|
||||
|
||||
|
||||
## 📋 Model Specifications
|
||||
|
||||
| Property | Value |
|
||||
|----------|-------|
|
||||
| **Base Model** | [GLM-4.7-REAP-50](https://huggingface.co/0xSero/GLM-4.7-185B) |
|
||||
| **Original (GLM-4.7)** | 358B params, ~700GB |
|
||||
| **After REAP 50%** | 179B params |
|
||||
| **After W4A16 Quant** | ~92GB on disk |
|
||||
| **Quantization** | INT4 weights, FP16 activations |
|
||||
| **Group Size** | 128 |
|
||||
| **Format** | GPTQ (AutoRound) |
|
||||
| **Experts per Layer** | 80 (was 160) |
|
||||
| **VRAM Required** | ~100GB |
|
||||
|
||||
### Compression Pipeline
|
||||
|
||||
```
|
||||
GLM-4.7 (358B, 700GB)
|
||||
│
|
||||
▼ REAP 50% expert pruning
|
||||
│
|
||||
GLM-4.7-REAP-50 (179B)
|
||||
│
|
||||
▼ AutoRound W4A16 quantization
|
||||
│
|
||||
GLM-4.7-REAP-50-W4A16 (~92GB) ◀── This model
|
||||
|
||||
Total: ~6.5x compression
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
|
||||
## 🔬 Calibration Dataset: Deep Dive
|
||||
|
||||
REAP's effectiveness depends critically on **calibration data that represents the target use case**. We specifically optimized for **code generation**, **function/tool calling**, and **agentic workflows**.
|
||||
|
||||
### Why These 3 Datasets?
|
||||
|
||||
| Dataset | Samples | Purpose | Why It Matters |
|
||||
|---------|---------|---------|----------------|
|
||||
| [evol-codealpaca-v1](https://huggingface.co/datasets/theblackcat102/evol-codealpaca-v1) | 700 | Code generation | **51% of mix** — Code tasks activate specific expert pathways; pruning without code calibration destroys coding ability |
|
||||
| [xlam-function-calling-60k](https://huggingface.co/datasets/Salesforce/xlam-function-calling-60k) | 330 | Function/tool calling | **24% of mix** — Tool use requires structured JSON output; experts handling schema generation must be preserved |
|
||||
| [SWE-smith-trajectories](https://huggingface.co/datasets/SWE-bench/SWE-smith-trajectories) | 330 | Agentic multi-turn | **24% of mix** — Real SWE-bench trajectories with tool calls, file edits, and multi-step reasoning |
|
||||
|
||||
### The Science Behind Dataset Selection
|
||||
|
||||
```
|
||||
REAP Algorithm:
|
||||
1. Forward pass calibration samples through model
|
||||
2. Record which experts activate and their magnitudes
|
||||
3. Compute saliency = router_weight × activation_norm
|
||||
4. Prune lowest-saliency experts
|
||||
|
||||
Key Insight: Experts are TASK-SPECIFIC
|
||||
├── Some experts specialize in natural language
|
||||
├── Some experts specialize in code syntax
|
||||
├── Some experts specialize in JSON/structured output
|
||||
└── Some experts specialize in multi-turn context
|
||||
|
||||
If calibration lacks code → code-specialized experts appear "unused" → get pruned → model loses coding ability
|
||||
```
|
||||
|
||||
### Cerebras' Original Mix (from paper)
|
||||
|
||||
Cerebras used the same 3 datasets in their GLM-4.6 REAP experiments:
|
||||
- evol-codealpaca-v1 for code generation
|
||||
- xlam-function-calling-60k for tool calling
|
||||
- SWE-smith-trajectories for agentic tasks
|
||||
|
||||
We followed this exact recipe for reproducibility.
|
||||
|
||||
### Combined Dataset
|
||||
|
||||
Our calibration mix: [0xSero/glm47-reap-calibration-v2](https://huggingface.co/datasets/0xSero/glm47-reap-calibration-v2)
|
||||
|
||||
|
||||
---
|
||||
|
||||
## 🚀 Deployment
|
||||
|
||||
### vLLM (Recommended)
|
||||
|
||||
```bash
|
||||
vllm serve 0xSero/GLM-4.7-185B-W4A16 \
|
||||
--tensor-parallel-size 4 \
|
||||
--trust-remote-code \
|
||||
--quantization gptq
|
||||
```
|
||||
|
||||
### Transformers
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
"0xSero/GLM-4.7-185B-W4A16",
|
||||
device_map="auto",
|
||||
trust_remote_code=True
|
||||
)
|
||||
tokenizer = AutoTokenizer.from_pretrained("0xSero/GLM-4.7-185B-W4A16", trust_remote_code=True)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🧩 Reproduction
|
||||
|
||||
### Step 1: REAP Pruning
|
||||
|
||||
|
||||
```python
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
REAP Pruning Script for MoE Models
|
||||
Adapted from: https://github.com/CerebrasResearch/reap
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
def run_reap(
|
||||
model_path: str,
|
||||
compression_ratio: float,
|
||||
dataset: str = "0xSero/glm47-reap-calibration-v2",
|
||||
samples: int = 1360,
|
||||
seed: int = 42,
|
||||
distance: str = "angular",
|
||||
reuse_observations: str = None,
|
||||
):
|
||||
"""
|
||||
Run REAP expert pruning.
|
||||
|
||||
Args:
|
||||
model_path: Path to base model
|
||||
compression_ratio: 0.30 = prune 30%, keep 70%
|
||||
dataset: Calibration dataset (code + tools + agentic)
|
||||
samples: Number of calibration samples
|
||||
seed: Random seed for reproducibility
|
||||
distance: Distance metric for expert clustering
|
||||
reuse_observations: Path to pre-computed observations for instant pruning
|
||||
"""
|
||||
cmd = [
|
||||
sys.executable, "src/reap/prune.py",
|
||||
"--model-name", model_path,
|
||||
"--dataset-name", dataset,
|
||||
"--compression-ratio", str(compression_ratio),
|
||||
"--prune-method", "reap",
|
||||
"--seed", str(seed),
|
||||
"--samples_per_category", str(samples),
|
||||
"--model_max_length", "2048",
|
||||
"--distance_measure", distance,
|
||||
"--record_pruning_metrics_only", "true",
|
||||
]
|
||||
|
||||
if reuse_observations:
|
||||
# Instant pruning: skip calibration, reuse precomputed expert scores
|
||||
cmd.extend(["--load_observations", reuse_observations])
|
||||
|
||||
subprocess.run(cmd, check=True)
|
||||
|
||||
# Example: Create 40% pruned model
|
||||
run_reap(
|
||||
model_path="/path/to/GLM-4.7",
|
||||
compression_ratio=0.40, # Prune 40% of experts
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
### Step 2: AutoRound Quantization
|
||||
|
||||
|
||||
```python
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
AutoRound W4A16 Quantization
|
||||
Intel's state-of-the-art weight quantization using signed gradient descent.
|
||||
"""
|
||||
|
||||
from auto_round import AutoRound
|
||||
|
||||
def quantize_w4a16(
|
||||
model_path: str,
|
||||
output_dir: str,
|
||||
bits: int = 4,
|
||||
group_size: int = 128,
|
||||
format: str = "auto_gptq",
|
||||
):
|
||||
"""
|
||||
Quantize model to INT4 weights with FP16 activations.
|
||||
|
||||
Args:
|
||||
model_path: Path to REAP-pruned model
|
||||
output_dir: Output directory
|
||||
bits: Weight bit width (4 for W4A16)
|
||||
group_size: Quantization group size (128 is optimal)
|
||||
format: Output format (auto_gptq for vLLM compatibility)
|
||||
"""
|
||||
ar = AutoRound(
|
||||
model_path,
|
||||
scheme="W4A16",
|
||||
device="cuda",
|
||||
device_map="auto",
|
||||
trust_remote_code=True,
|
||||
batch_size=1,
|
||||
seqlen=512,
|
||||
nsamples=64,
|
||||
)
|
||||
ar.quantize_and_save(output_dir, format=format)
|
||||
|
||||
# Example: Quantize REAP-40 to W4A16
|
||||
quantize_w4a16(
|
||||
model_path="./GLM-4.7-REAP-40",
|
||||
output_dir="./GLM-4.7-REAP-40-W4A16",
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
---
|
||||
|
||||
## ⚖️ License
|
||||
|
||||
Apache 2.0
|
||||
|
||||
---
|
||||
|
||||
## License & citation
|
||||
License inherited from the base model.
|
||||
|
||||
```bibtex
|
||||
@misc{lasby2025reap,
|
||||
title = {REAP the Experts: Why Pruning Prevails for One-Shot MoE Compression},
|
||||
author = {Mike Lasby and Ivan Lazarevich and Nish Sinnadurai and Sean Lie and Yani Ioannou and Vithursan Thangarasa},
|
||||
year = {2025}, eprint = {2510.13999}, archivePrefix = {arXiv}
|
||||
}
|
||||
```
|
||||
|
||||
## Sponsors
|
||||
Made possible by **NVIDIA · TNG Technology · Lambda · Prime Intellect · Hot Aisle**.
|
||||
86
chat_template.jinja
Normal file
86
chat_template.jinja
Normal file
@@ -0,0 +1,86 @@
|
||||
[gMASK]<sop>
|
||||
{%- if tools -%}
|
||||
<|system|>
|
||||
# Tools
|
||||
|
||||
You may call one or more functions to assist with the user query.
|
||||
|
||||
You are provided with function signatures within <tools></tools> XML tags:
|
||||
<tools>
|
||||
{% for tool in tools %}
|
||||
{{ tool | tojson(ensure_ascii=False) }}
|
||||
{% endfor %}
|
||||
</tools>
|
||||
|
||||
For each function call, output the function name and arguments within the following XML format:
|
||||
<tool_call>{function-name}<arg_key>{arg-key-1}</arg_key><arg_value>{arg-value-1}</arg_value><arg_key>{arg-key-2}</arg_key><arg_value>{arg-value-2}</arg_value>...</tool_call>{%- endif -%}
|
||||
{%- macro visible_text(content) -%}
|
||||
{%- if content is string -%}
|
||||
{{- content }}
|
||||
{%- elif content is iterable and content is not mapping -%}
|
||||
{%- for item in content -%}
|
||||
{%- if item is mapping and item.type == 'text' -%}
|
||||
{{- item.text }}
|
||||
{%- elif item is string -%}
|
||||
{{- item }}
|
||||
{%- endif -%}
|
||||
{%- endfor -%}
|
||||
{%- else -%}
|
||||
{{- content }}
|
||||
{%- endif -%}
|
||||
{%- endmacro -%}
|
||||
{%- set ns = namespace(last_user_index=-1) %}
|
||||
{%- for m in messages %}
|
||||
{%- if m.role == 'user' %}
|
||||
{% set ns.last_user_index = loop.index0 -%}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{% for m in messages %}
|
||||
{%- if m.role == 'user' -%}<|user|>{{ visible_text(m.content) }}
|
||||
{%- elif m.role == 'assistant' -%}
|
||||
<|assistant|>
|
||||
{%- set reasoning_content = '' %}
|
||||
{%- set content = visible_text(m.content) %}
|
||||
{%- if m.reasoning_content is string %}
|
||||
{%- set reasoning_content = m.reasoning_content %}
|
||||
{%- else %}
|
||||
{%- if '</think>' in content %}
|
||||
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- if ((clear_thinking is defined and not clear_thinking) or loop.index0 > ns.last_user_index) and reasoning_content -%}
|
||||
{{ '<think>' + reasoning_content.strip() + '</think>'}}
|
||||
{%- else -%}
|
||||
{{ '</think>' }}
|
||||
{%- endif -%}
|
||||
{%- if content.strip() -%}
|
||||
{{ content.strip() }}
|
||||
{%- endif -%}
|
||||
{% if m.tool_calls %}
|
||||
{% for tc in m.tool_calls %}
|
||||
{%- if tc.function %}
|
||||
{%- set tc = tc.function %}
|
||||
{%- endif %}
|
||||
{{- '<tool_call>' + tc.name -}}
|
||||
{% set _args = tc.arguments %}{% for k, v in _args.items() %}<arg_key>{{ k }}</arg_key><arg_value>{{ v | tojson(ensure_ascii=False) if v is not string else v }}</arg_value>{% endfor %}</tool_call>{% endfor %}
|
||||
{% endif %}
|
||||
{%- elif m.role == 'tool' -%}
|
||||
{%- if m.content is string -%}
|
||||
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||
{{- '<|observation|>' }}
|
||||
{%- endif %}
|
||||
{{- '<tool_response>' }}
|
||||
{{- m.content }}
|
||||
{{- '</tool_response>' }}
|
||||
{%- else -%}
|
||||
<|observation|>{% for tr in m.content %}
|
||||
<tool_response>{{ tr.output if tr.output is defined else tr }}</tool_response>{% endfor -%}
|
||||
{% endif -%}
|
||||
{%- elif m.role == 'system' -%}
|
||||
<|system|>{{ visible_text(m.content) }}
|
||||
{%- endif -%}
|
||||
{%- endfor -%}
|
||||
{%- if add_generation_prompt -%}
|
||||
<|assistant|>{{- '</think>' if (enable_thinking is defined and not enable_thinking) else '<think>' -}}
|
||||
{%- endif -%}
|
||||
57
config.json
Normal file
57
config.json
Normal file
@@ -0,0 +1,57 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Glm4MoeForCausalLM"
|
||||
],
|
||||
"attention_bias": true,
|
||||
"attention_dropout": 0.0,
|
||||
"dtype": "bfloat16",
|
||||
"eos_token_id": [
|
||||
151329,
|
||||
151336,
|
||||
151338
|
||||
],
|
||||
"first_k_dense_replace": 3,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 5120,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 12288,
|
||||
"max_position_embeddings": 202752,
|
||||
"model_type": "glm4_moe",
|
||||
"moe_intermediate_size": 1536,
|
||||
"n_group": 1,
|
||||
"n_routed_experts": 80,
|
||||
"n_shared_experts": 1,
|
||||
"norm_topk_prob": true,
|
||||
"num_attention_heads": 96,
|
||||
"num_experts_per_tok": 8,
|
||||
"num_hidden_layers": 92,
|
||||
"num_key_value_heads": 8,
|
||||
"num_nextn_predict_layers": 1,
|
||||
"pad_token_id": 151329,
|
||||
"partial_rotary_factor": 0.5,
|
||||
"quantization_config": {
|
||||
"autoround_version": "0.9.4",
|
||||
"bits": 4,
|
||||
"damp_percent": 0.01,
|
||||
"data_type": "int",
|
||||
"desc_act": false,
|
||||
"group_size": 128,
|
||||
"lm_head": false,
|
||||
"provider": "auto-round",
|
||||
"quant_method": "gptq",
|
||||
"sym": true,
|
||||
"true_sequential": false
|
||||
},
|
||||
"rms_norm_eps": 1e-05,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"routed_scaling_factor": 2.5,
|
||||
"tie_word_embeddings": false,
|
||||
"topk_group": 1,
|
||||
"transformers_version": "4.57.3",
|
||||
"use_cache": true,
|
||||
"use_qk_norm": true,
|
||||
"vocab_size": 151552,
|
||||
"torch_dtype": "float16"
|
||||
}
|
||||
1
configuration.json
Normal file
1
configuration.json
Normal file
@@ -0,0 +1 @@
|
||||
{"framework": "pytorch", "task": "text-generation", "allow_remote": true}
|
||||
10
generation_config.json
Normal file
10
generation_config.json
Normal file
@@ -0,0 +1,10 @@
|
||||
{
|
||||
"_from_model_config": true,
|
||||
"eos_token_id": [
|
||||
151329,
|
||||
151336,
|
||||
151338
|
||||
],
|
||||
"pad_token_id": 151329,
|
||||
"transformers_version": "4.57.3"
|
||||
}
|
||||
3
model-00001-of-00020.safetensors
Normal file
3
model-00001-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:51b9a429c7021f4c4ff720be8a38050b6f9c692f6136816cd1585813deead7e3
|
||||
size 4997550752
|
||||
3
model-00002-of-00020.safetensors
Normal file
3
model-00002-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:743bf1b2865e872528fa8e652bcea59db46743a91fbd0f6498bfcc4832cee4e6
|
||||
size 4998308888
|
||||
3
model-00003-of-00020.safetensors
Normal file
3
model-00003-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:269df6ce1691a82f775dde26bc81fe6ace8ee985b01ecc4602e70c71efa19144
|
||||
size 4998313136
|
||||
3
model-00004-of-00020.safetensors
Normal file
3
model-00004-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3452412d33626598e1f7fd7abea70272cfca5d485c47cc3b2533264f3f2771e3
|
||||
size 5000345656
|
||||
3
model-00005-of-00020.safetensors
Normal file
3
model-00005-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:7651dd4050ce4339f7dc18f8ac75a1ceb6770c33151f1ebfcf61a862e57ecee4
|
||||
size 4998313056
|
||||
3
model-00006-of-00020.safetensors
Normal file
3
model-00006-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f19b6f7d13c50f232ba96cb0b4b02d91173eb3fbb49fc905e13d422bcf5913dd
|
||||
size 4998313104
|
||||
3
model-00007-of-00020.safetensors
Normal file
3
model-00007-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:260de5c6e4f3930e05d53a3a87abce8caa0cf0b6245c20932961425a1826c79c
|
||||
size 5000345696
|
||||
3
model-00008-of-00020.safetensors
Normal file
3
model-00008-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f86faaa1c6a8064359812684af415b4a68bf9e7f84fbc0ad21e85cab703dd6b9
|
||||
size 4998313056
|
||||
3
model-00009-of-00020.safetensors
Normal file
3
model-00009-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:40657f83dc2199e9ef2114172cb812da958f38ff108f820ba2ffd6f8ceafd50f
|
||||
size 4998313064
|
||||
3
model-00010-of-00020.safetensors
Normal file
3
model-00010-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c12d8dd06c01ff4a8a972037db7cf745bd58b0b1f3318d1844666a135568e389
|
||||
size 5000345728
|
||||
3
model-00011-of-00020.safetensors
Normal file
3
model-00011-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:bd40ccd7c643a07a8c1cda0e5781b7ffba3eb5e458f7c3d6b168cc838efa0501
|
||||
size 4998313056
|
||||
3
model-00012-of-00020.safetensors
Normal file
3
model-00012-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:95630e748f0821e41c35fa323e527d37cd871675197958318e86fce970bf855e
|
||||
size 4998313056
|
||||
3
model-00013-of-00020.safetensors
Normal file
3
model-00013-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cfc33aed9aa5cc28aadfc78d14dd51a3eb42301272770f74fa82b7048a035586
|
||||
size 5000345736
|
||||
3
model-00014-of-00020.safetensors
Normal file
3
model-00014-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:dbd69ee08379ae6b4b9294efe4cd9452e06b6fc6fc761e82d08b20976e3a2e43
|
||||
size 4998313056
|
||||
3
model-00015-of-00020.safetensors
Normal file
3
model-00015-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:564194e7c14da860d3d3d633e34a8e5182b1bfd8778cc80b443375cd4ea74ee1
|
||||
size 4998313056
|
||||
3
model-00016-of-00020.safetensors
Normal file
3
model-00016-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:36b5fed9cedcd99a9db14e67487a7c0c0b363f93d2e980e97bd7bb7583f92df8
|
||||
size 5000345736
|
||||
3
model-00017-of-00020.safetensors
Normal file
3
model-00017-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:936139c8bb51a38c5961d60e115f16779d378f632f832effc606bb4ac31aa3ff
|
||||
size 4998313056
|
||||
3
model-00018-of-00020.safetensors
Normal file
3
model-00018-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4856541d3c6d5a00258d19ee7243aaaecdac3082425b538564e5e5480a5d6cd7
|
||||
size 4998313056
|
||||
3
model-00019-of-00020.safetensors
Normal file
3
model-00019-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c00ae49f0b57380e2198c95647a30b2fdab8cd3aebd11d7f8cb00c9d0f8fbc28
|
||||
size 5000345736
|
||||
3
model-00020-of-00020.safetensors
Normal file
3
model-00020-of-00020.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:16c1fc16cf433ecce99f8b24651125466cad91e821495a383f3f637237b308f3
|
||||
size 3735009720
|
||||
3
model.safetensors.index.json
Normal file
3
model.safetensors.index.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8427f2a2c04c4fc8e759ca6b52a6510222f2c98ae710b8f4f48e5fd6a46723e4
|
||||
size 7994803
|
||||
13
quantization_config.json
Normal file
13
quantization_config.json
Normal file
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"bits": 4,
|
||||
"group_size": 128,
|
||||
"sym": true,
|
||||
"data_type": "int",
|
||||
"autoround_version": "0.9.4",
|
||||
"lm_head": false,
|
||||
"provider": "auto-round",
|
||||
"quant_method": "gptq",
|
||||
"desc_act": false,
|
||||
"true_sequential": false,
|
||||
"damp_percent": 0.01
|
||||
}
|
||||
95
scripts/run_autoround.py
Normal file
95
scripts/run_autoround.py
Normal file
@@ -0,0 +1,95 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
AutoRound W4A16 Quantization for GLM-4.7 REAP models
|
||||
|
||||
This script quantizes a REAP-pruned GLM-4.7 model to INT4 weights using Intel's AutoRound.
|
||||
Reduces model size by ~4x while maintaining quality.
|
||||
|
||||
Requirements:
|
||||
pip install auto-round
|
||||
|
||||
Usage:
|
||||
python run_autoround.py --model-path ./GLM-4.7-REAP-50 --output-dir ./GLM-4.7-REAP-50-W4A16
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="AutoRound W4A16 quantization")
|
||||
parser.add_argument("--model-path", type=str, required=True,
|
||||
help="Path to REAP-pruned model")
|
||||
parser.add_argument("--output-dir", type=str, default=None,
|
||||
help="Output directory (default: {model-path}-W4A16)")
|
||||
parser.add_argument("--bits", type=int, default=4,
|
||||
help="Weight bit width (default: 4)")
|
||||
parser.add_argument("--group-size", type=int, default=128,
|
||||
help="Quantization group size (default: 128)")
|
||||
parser.add_argument("--format", type=str, default="auto_gptq",
|
||||
choices=["auto_gptq", "auto_awq", "auto_round"],
|
||||
help="Output format (default: auto_gptq)")
|
||||
parser.add_argument("--iters", type=int, default=200,
|
||||
help="Optimization iterations (default: 200)")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Validate
|
||||
if not Path(args.model_path).exists():
|
||||
print(f"ERROR: Model path not found: {args.model_path}")
|
||||
sys.exit(1)
|
||||
|
||||
# Build output directory
|
||||
if args.output_dir is None:
|
||||
args.output_dir = f"{args.model_path}-W{args.bits}A16"
|
||||
|
||||
Path(args.output_dir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Get model size info
|
||||
model_size_gb = sum(f.stat().st_size for f in Path(args.model_path).rglob("*.safetensors")) / (1024**3)
|
||||
expected_output_gb = model_size_gb / 4 # ~4x compression for W4
|
||||
|
||||
print("=" * 60)
|
||||
print(f"AutoRound W{args.bits}A16 Quantization")
|
||||
print("=" * 60)
|
||||
print(f"Input Model: {args.model_path}")
|
||||
print(f"Input Size: {model_size_gb:.1f} GB")
|
||||
print(f"Output: {args.output_dir}")
|
||||
print(f"Expected Output Size: ~{expected_output_gb:.1f} GB")
|
||||
print(f"Config: {args.bits}-bit, group_size={args.group_size}, format={args.format}")
|
||||
print("=" * 60)
|
||||
print("\nThis will take ~2-3 hours for a 92-layer MoE model...")
|
||||
print()
|
||||
|
||||
# Build command
|
||||
cmd = [
|
||||
"auto-round",
|
||||
"--model", args.model_path,
|
||||
"--bits", str(args.bits),
|
||||
"--group_size", str(args.group_size),
|
||||
"--format", args.format,
|
||||
"--output_dir", args.output_dir,
|
||||
"--iters", str(args.iters),
|
||||
]
|
||||
|
||||
result = subprocess.run(cmd)
|
||||
|
||||
if result.returncode == 0:
|
||||
# Calculate actual output size
|
||||
output_size_gb = sum(f.stat().st_size for f in Path(args.output_dir).rglob("*.safetensors")) / (1024**3)
|
||||
compression = model_size_gb / output_size_gb if output_size_gb > 0 else 0
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("AutoRound quantization complete!")
|
||||
print(f"Output: {args.output_dir}")
|
||||
print(f"Output Size: {output_size_gb:.1f} GB ({compression:.1f}x compression)")
|
||||
print("=" * 60)
|
||||
else:
|
||||
print(f"\nERROR: AutoRound failed with code {result.returncode}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
115
scripts/run_reap.py
Normal file
115
scripts/run_reap.py
Normal file
@@ -0,0 +1,115 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
REAP (Router-Experts Activation Pruning) for GLM-4.7 MoE
|
||||
|
||||
This script prunes MoE experts from GLM-4.7 using the REAP methodology from Cerebras.
|
||||
Requires: https://github.com/Cerebras/reap (or fork with GLM support)
|
||||
|
||||
Usage:
|
||||
python run_reap.py --compression-ratio 0.50 --model-path /path/to/GLM-4.7
|
||||
|
||||
For observation reuse (instant pruning at different ratios):
|
||||
python run_reap.py --compression-ratio 0.35 --reuse-observations observations_1360_angular-seed_42.pt
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="REAP pruning for GLM-4.7")
|
||||
parser.add_argument("--model-path", type=str, required=True,
|
||||
help="Path to GLM-4.7 model")
|
||||
parser.add_argument("--compression-ratio", type=float, required=True,
|
||||
help="Compression ratio (0.30 = keep 70%, 0.50 = keep 50%)")
|
||||
parser.add_argument("--output-dir", type=str, default=None,
|
||||
help="Output directory (default: auto-generated)")
|
||||
parser.add_argument("--dataset", type=str,
|
||||
default="0xSero/glm47-reap-calibration-v2",
|
||||
help="Calibration dataset")
|
||||
parser.add_argument("--samples", type=int, default=1360,
|
||||
help="Number of calibration samples")
|
||||
parser.add_argument("--seed", type=int, default=42,
|
||||
help="Random seed")
|
||||
parser.add_argument("--distance", type=str, default="angular",
|
||||
choices=["angular", "cosine", "euclidean"],
|
||||
help="Distance measure for expert clustering")
|
||||
parser.add_argument("--reuse-observations", type=str, default=None,
|
||||
help="Path to pre-computed observations file for instant pruning")
|
||||
parser.add_argument("--reap-repo", type=str, default="./reap",
|
||||
help="Path to REAP repository")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Validate
|
||||
if not Path(args.model_path).exists():
|
||||
print(f"ERROR: Model path not found: {args.model_path}")
|
||||
sys.exit(1)
|
||||
|
||||
reap_script = Path(args.reap_repo) / "src" / "reap" / "prune.py"
|
||||
if not reap_script.exists():
|
||||
print(f"ERROR: REAP prune.py not found at: {reap_script}")
|
||||
print("Clone the REAP repo: git clone https://github.com/Cerebras/reap")
|
||||
sys.exit(1)
|
||||
|
||||
# Build output directory name
|
||||
if args.output_dir is None:
|
||||
ratio_pct = int(args.compression_ratio * 100)
|
||||
args.output_dir = f"./GLM-4.7-REAP-{ratio_pct}"
|
||||
|
||||
Path(args.output_dir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Build command
|
||||
cmd = [
|
||||
sys.executable, str(reap_script),
|
||||
"--model-name", args.model_path,
|
||||
"--dataset-name", args.dataset,
|
||||
"--compression-ratio", str(args.compression_ratio),
|
||||
"--prune-method", "reap",
|
||||
"--seed", str(args.seed),
|
||||
"--do-eval", "false",
|
||||
"--profile", "false",
|
||||
"--samples_per_category", str(args.samples),
|
||||
"--model_max_length", "2048",
|
||||
"--distance_measure", args.distance,
|
||||
"--record_pruning_metrics_only", "true",
|
||||
"--output_file_name", f"observations_{args.samples}_{args.distance}-seed_{args.seed}.pt",
|
||||
]
|
||||
|
||||
if args.reuse_observations:
|
||||
cmd.extend(["--load_observations", args.reuse_observations])
|
||||
print(f"Reusing observations from: {args.reuse_observations}")
|
||||
print("This enables instant pruning without re-running calibration!")
|
||||
|
||||
print("=" * 60)
|
||||
print(f"REAP Pruning: GLM-4.7 @ {args.compression_ratio*100:.0f}% compression")
|
||||
print("=" * 60)
|
||||
print(f"Model: {args.model_path}")
|
||||
print(f"Output: {args.output_dir}")
|
||||
print(f"Dataset: {args.dataset} ({args.samples} samples)")
|
||||
print(f"Distance: {args.distance}")
|
||||
print("=" * 60)
|
||||
|
||||
# Run REAP
|
||||
env = {
|
||||
**dict(__import__('os').environ),
|
||||
"CUDA_VISIBLE_DEVICES": "0,1,2,3,4,5,6,7",
|
||||
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
|
||||
}
|
||||
|
||||
result = subprocess.run(cmd, env=env)
|
||||
|
||||
if result.returncode == 0:
|
||||
print("\n" + "=" * 60)
|
||||
print("REAP pruning complete!")
|
||||
print(f"Pruned model saved to: {args.output_dir}")
|
||||
print("=" * 60)
|
||||
else:
|
||||
print(f"\nERROR: REAP failed with code {result.returncode}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
40
special_tokens_map.json
Normal file
40
special_tokens_map.json
Normal file
@@ -0,0 +1,40 @@
|
||||
{
|
||||
"additional_special_tokens": [
|
||||
"<|endoftext|>",
|
||||
"[MASK]",
|
||||
"[gMASK]",
|
||||
"[sMASK]",
|
||||
"<sop>",
|
||||
"<eop>",
|
||||
"<|system|>",
|
||||
"<|user|>",
|
||||
"<|assistant|>",
|
||||
"<|observation|>",
|
||||
"<|begin_of_image|>",
|
||||
"<|end_of_image|>",
|
||||
"<|begin_of_video|>",
|
||||
"<|end_of_video|>",
|
||||
"<|begin_of_audio|>",
|
||||
"<|end_of_audio|>",
|
||||
"<|begin_of_transcription|>",
|
||||
"<|end_of_transcription|>",
|
||||
"<|code_prefix|>",
|
||||
"<|code_middle|>",
|
||||
"<|code_suffix|>",
|
||||
"/nothink"
|
||||
],
|
||||
"eos_token": {
|
||||
"content": "<|endoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"pad_token": {
|
||||
"content": "<|endoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
}
|
||||
}
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:bda8e2146c3bb7b7e0fc96dcc4f0aeff041c6c27952e3ace0665663ebff346ba
|
||||
size 19970700
|
||||
325
tokenizer_config.json
Normal file
325
tokenizer_config.json
Normal file
@@ -0,0 +1,325 @@
|
||||
{
|
||||
"added_tokens_decoder": {
|
||||
"151329": {
|
||||
"content": "<|endoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151330": {
|
||||
"content": "[MASK]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151331": {
|
||||
"content": "[gMASK]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151332": {
|
||||
"content": "[sMASK]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151333": {
|
||||
"content": "<sop>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151334": {
|
||||
"content": "<eop>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151335": {
|
||||
"content": "<|system|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151336": {
|
||||
"content": "<|user|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151337": {
|
||||
"content": "<|assistant|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151338": {
|
||||
"content": "<|observation|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151339": {
|
||||
"content": "<|begin_of_image|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151340": {
|
||||
"content": "<|end_of_image|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151341": {
|
||||
"content": "<|begin_of_video|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151342": {
|
||||
"content": "<|end_of_video|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151343": {
|
||||
"content": "<|begin_of_audio|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151344": {
|
||||
"content": "<|end_of_audio|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151345": {
|
||||
"content": "<|begin_of_transcription|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151346": {
|
||||
"content": "<|end_of_transcription|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151347": {
|
||||
"content": "<|code_prefix|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151348": {
|
||||
"content": "<|code_middle|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151349": {
|
||||
"content": "<|code_suffix|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151350": {
|
||||
"content": "<think>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151351": {
|
||||
"content": "</think>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151352": {
|
||||
"content": "<tool_call>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151353": {
|
||||
"content": "</tool_call>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151354": {
|
||||
"content": "<tool_response>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151355": {
|
||||
"content": "</tool_response>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151356": {
|
||||
"content": "<arg_key>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151357": {
|
||||
"content": "</arg_key>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151358": {
|
||||
"content": "<arg_value>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151359": {
|
||||
"content": "</arg_value>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151360": {
|
||||
"content": "/nothink",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151361": {
|
||||
"content": "<|begin_of_box|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151362": {
|
||||
"content": "<|end_of_box|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151363": {
|
||||
"content": "<|image|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151364": {
|
||||
"content": "<|video|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
}
|
||||
},
|
||||
"additional_special_tokens": [
|
||||
"<|endoftext|>",
|
||||
"[MASK]",
|
||||
"[gMASK]",
|
||||
"[sMASK]",
|
||||
"<sop>",
|
||||
"<eop>",
|
||||
"<|system|>",
|
||||
"<|user|>",
|
||||
"<|assistant|>",
|
||||
"<|observation|>",
|
||||
"<|begin_of_image|>",
|
||||
"<|end_of_image|>",
|
||||
"<|begin_of_video|>",
|
||||
"<|end_of_video|>",
|
||||
"<|begin_of_audio|>",
|
||||
"<|end_of_audio|>",
|
||||
"<|begin_of_transcription|>",
|
||||
"<|end_of_transcription|>",
|
||||
"<|code_prefix|>",
|
||||
"<|code_middle|>",
|
||||
"<|code_suffix|>",
|
||||
"/nothink"
|
||||
],
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"do_lower_case": false,
|
||||
"eos_token": "<|endoftext|>",
|
||||
"extra_special_tokens": {},
|
||||
"model_max_length": 128000,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"padding_side": "left",
|
||||
"remove_space": false,
|
||||
"tokenizer_class": "PreTrainedTokenizerFast"
|
||||
}
|
||||
Reference in New Issue
Block a user