commit 5b07c289be6179331e76687f5b25105b96823406 Author: ModelHub XC Date: Tue Sep 29 03:31:14 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: yujianboisme/qwen2.5-1.5b-darkhorse-code-fine-tuning Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..c1700c2 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,50 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bin.* filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zstandard filter=lfs diff=lfs merge=lfs -text +*.tfevents* filter=lfs diff=lfs merge=lfs -text +*.db* filter=lfs diff=lfs merge=lfs -text +*.ark* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.gguf* filter=lfs diff=lfs merge=lfs -text +*.ggml filter=lfs diff=lfs merge=lfs -text +*.llamafile* filter=lfs diff=lfs merge=lfs -text +*.pt2 filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +"/merges.txt" filter=lfs diff=lfs merge=lfs -text +"/tokenizer.json" filter=lfs diff=lfs merge=lfs -text +"/vocab.json" filter=lfs diff=lfs merge=lfs -text \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..dedb182 --- /dev/null +++ b/README.md @@ -0,0 +1,319 @@ +--- +license: apache-2.0 +license_link: https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct/blob/main/LICENSE +language: +- zh +pipeline_tag: text-generation +library_name: transformers +base_model: Qwen/Qwen2.5-1.5B-Instruct +tags: +- text-generation +- qwen2.5 +- lora +- json +- instruction-following +- chinese +- command-translation +--- + +# darkhorse-code-instruct-1.5b + +> **把一个微调过的 1.5B 小模型,做成编辑器里的「自然语言 → 私域命令」翻译器。** +> 输入一句中文口语,输出**一行严格 JSON**,不是聊天回复。 + +``` +输入:关掉倒数第5个文件 +输出:{"intent": "command", "cmd": "close -5"} +``` + +```json +{"intent": "command", "cmd": "del src/P05-es/pdf2es.py"} // 删除类 +{"intent": "reject", "reply": "您没有权限!"} // 越权/不合法 +{"intent": "chat", "reply": "你还是好好工作吧,房贷还清了吗?车贷还清了吗?"} // 非命令输入 +``` + +本模型是 **Qwen2.5-1.5B-Instruct + LoRA(r=16)** 指令微调后**合并**(merged)得到的自包含权重, +不需要 `peft`、不需要 adapter,`transformers` / vLLM 直接加载即可。 + +--- + +## 1. 它解决什么问题 + +`darkhorse-code` 是一个编辑器项目,它有一套**私域命令语法**(`open project` / `close -3` / +`config add -r k=v` …),命令语法固定、可校验,但用户不想背。于是把「用户口语 → 一条标准命令」 +这个窄任务交给一个 1.5B 小模型:**小、快、可完全本地跑,且输出必须能被程序解析**。 + +它不是通用聊天模型,**不会**回答知识问题、不会写代码、不会陪你聊天 —— +遇到非命令输入,它的正确行为是回一句 20 字以内的短回复(或一句固定的「劝退」话术)。 + +## 2. 输出契约(三种信封) + +模型**只输出一行 JSON**,`intent` 决定信封形状: + +| intent | 形状 | 含义 | +|---|---|---| +| `command` | `{"intent":"command","cmd":"<标准命令>"}` | 翻译成功,`cmd` 可直接执行 | +| `reject` | `{"intent":"reject","reply":"<拒绝原因>"}` | 命中硬性规则,如路径越权 | +| `chat` | `{"intent":"chat","reply":"<≤20字短回复>"}` | 不是私域命令 | + +两条固定话术(评测按**逐字**判定): + +- 绝对路径出现在 `open file` / `new` / `del` / `rename` → `您没有权限!` +- 要求把运行目标存成全局配置 → `运行目标不能保存为全局` +- 闲聊超过 20 字说不完 → `你还是好好工作吧,房贷还清了吗?车贷还清了吗?` + +## 3. 命令语法(模型学到的目标空间) + +``` +open project <绝对路径> | open file <相对路径> +close project|all|other|left|right|<序号> # 序号 0 起,负数从右往左,-1 是最后一个 +config add|remove|update|get [-g|-p|-r] = # 未指定层级默认 -p +new file <相对路径> | new folder <相对路径> | new py|rs|md|c <名称> +del <相对路径> | run <名称>=<命令> | run del <名称> +rename <旧相对路径> <新相对路径> +project lang <语言> <绝对路径> | project delete <绝对路径> | project migrate +project edit "<路径>" "<名称>" <语言> +``` + +语言取值:`unknown/mix/java/c/python/rust/web/golang/document/kotlin`(中文表达做了映射:前端→web、Go→golang、文档→document…)。 + +## 4. 快速开始(transformers) + +⚠️ **两条铁律,违反任一条效果都会明显掉:** + +1. **必须使用本模型内置的 system prompt**(下面 `SYSTEM_PROMPT` 原文,逐字), + 模型是在这份 prompt 下微调的;换 prompt = 训练/推理不一致。 +2. **必须走模型自带的 chat template**(本仓库的 `chat_template.jinja`), + `tokenizer.apply_chat_template` 会自动读到它。 + +```python +import json +from transformers import AutoModelForCausalLM, AutoTokenizer + +MODEL = "darkhorse-code-instruct-1.5b" # 换成你的 ModelScope 仓库 id +tok = AutoTokenizer.from_pretrained(MODEL) +model = AutoModelForCausalLM.from_pretrained(MODEL, torch_dtype="bfloat16", device_map="auto") + +SYSTEM_PROMPT = """你是 darkhorse-code 编辑器的命令翻译器。把用户的话翻译成一条标准命令,只输出一行 JSON。 + +命令语法: +open project <绝对路径> | open file <相对路径> +close project|all|other|left|right|<序号> +close 的序号从 0 开始,负数从右往左,-1 是最后一个 +config add|remove|update|get [-g|-p|-r] =,未指定层级默认 -p +new file <相对路径> | new folder <相对路径> | new py|rs|md|c <名称> +del <相对路径> | run <名称>=<命令> | run del <名称> +rename <旧相对路径> <新相对路径> +project lang <语言> <绝对路径> | project delete <绝对路径> | project migrate +project edit "<路径>" "<名称>" <语言> + +规则: +1. 语言取值 unknown/mix/java/c/python/rust/web/golang/document/kotlin;中文表达要映射:Python→python、Go→golang、前端→web、文档→document、混合→mix、未知→unknown。 +2. 绝对路径(以 / 或 C:/ D:/ 开头)只允许用在 open project、project lang/edit/delete 和 config 的值里;出现在 open file、new、del、rename 时一律拒绝,reply 固定为「您没有权限!」。 +3. 路径里的正斜杠输出一律转成反斜杠。 +4. 运行目标只能存项目配置,不能存全局;用户要求存全局时 reply 固定为「运行目标不能保存为全局」。 +5. config 的值含空格时用双引号包裹。上下文给出已有运行目标时,新增运行目标索引 = 最大索引 + 1。 +6. 不是私域命令的输入(闲聊、问知识、要你写文档)用 chat:reply 不超过 20 字;一句话说不完 20 字时 reply 固定为「你还是好好工作吧,房贷还清了吗?车贷还清了吗?」。 +7. 同义命令输出规范形式:删除类写 del,关闭其他文件写 close other,重命名写 rename,新建目录写 new folder,run 的删除写 run del。 + +输出格式(只输出一行 JSON,不要解释、不要代码块): +{"intent":"command","cmd":"<标准命令>"} +{"intent":"reject","reply":"<拒绝原因>"} +{"intent":"chat","reply":"<短回复>"}""" + + +def translate(text: str, context: str | None = None) -> dict: + # 上下文按训练时的约定另起一行:'[上下文] <内容>' + user = f"[上下文] {context}\n{text}" if context else text + msgs = [{"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": user}] + prompt = tok.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True) + ids = tok(prompt, return_tensors="pt", add_special_tokens=False).to(model.device) + out = model.generate(**ids, max_new_tokens=128, do_sample=False, # 贪心,别采样 + pad_token_id=tok.pad_token_id or tok.eos_token_id) + raw = tok.decode(out[0][ids["input_ids"].shape[1]:], skip_special_tokens=True).strip() + return json.loads(raw) # 建议外面再包一层 try:小模型偶尔会多写一句解释 + + +print(translate("把倒数第三个文件关掉")) +# {'intent': 'command', 'cmd': 'close -3'} +print(translate("加一个运行目标 测试=pytest -q", context="当前项目: D:/Projects/demo")) +# {'intent': 'command', 'cmd': 'run 测试=pytest -q'} +``` + +## 5. vLLM 部署(推荐,OpenAI 兼容) + +8 GB 显存(如 RTX 5060 Laptop)即可跑 bf16: + +```bash +vllm serve ./qwen2.5-1.5b-cmd-merged \ + --served-model-name darkhorse-code-instruct \ + --chat-template ./qwen2.5-1.5b-cmd-merged/chat_template.jinja \ + --dtype bfloat16 --max-model-len 32768 \ + --gpu-memory-utilization 0.85 --max-num-seqs 32 \ + --enable-prefix-caching --port 8001 +``` + +```bash +curl http://127.0.0.1:8001/v1/chat/completions -H "Content-Type: application/json" -d '{ + "model": "darkhorse-code-instruct", + "messages": [ + {"role": "system", "content": "<上面那份 SYSTEM_PROMPT 原文>"}, + {"role": "user", "content": "关掉倒数第5个文件"} + ], + "temperature": 0, "max_tokens": 128 +}' +``` + +要点: + +- `temperature=0`(训练与评测都是贪心解码);**不要**用 `generation_config.json` 里的默认采样参数。 +- 请求里的 `system` 必须是上面那份原文;本项目服务端做薄封装时会把 system 强制替换成它。 +- `choices[0].message.content` 是**契约 JSON 字符串**,调用方需要 `json.loads`。 +- `--enable-prefix-caching` 对这类场景收益明显:所有请求共享同一份 553 token 的 system prompt 前缀。 +- 需要「输出结构绝对合法」时,可以上受约束解码(JSON Schema 限死三种信封), + 代价是采样空间被约束,个别内容可能与贪心输出不同。 + +### 实测性能(RTX 5060 Laptop 8G / sm_120,bf16) + +同一批 15 条真实指令压测,`--max-model-len 32768 --max-num-seqs 32 --enable-prefix-caching`: + +| 并发 | 请求吞吐 | 生成吞吐 | 延迟 p50 | 延迟 p95 | +|---|---|---|---|---| +| 1 | 3.6 req/s | 72 tok/s | 251 ms | 389 ms | +| 8 | 23.0 req/s | 469 tok/s | 278 ms | 425 ms | +| 32 | **49.4 req/s** | **1002 tok/s** | 337 ms | 480 ms | + +流式 TTFT(并发 1)**87 ms**;显存占用:权重 3.1 G + KV cache 3.46 GiB(129,680 token ⇒ 32K 上下文可并发 3.96 路)。 +参考:同一模型用 CPU + transformers 逐条生成是 **4.1–6.4 s/条**(约 0.2 req/s)。 + +### 在 WSL2 上部署的三个额外环境变量 + +vLLM 官方只发 Linux 轮子,Windows 用户通常跑在 WSL2 里;此时还需要: + +```bash +# 1) V2 Model Runner 需要 pinned memory/UVA,而 vLLM 在 WSL2 上默认关掉 pin memory +export VLLM_WSL2_ENABLE_PIN_MEMORY=1 +# 2) WSL 里若没有 gcc,Triton 运行时编译 kernel launcher 会报 "Failed to find C compiler" +# 可装 gcc(sudo apt-get install -y gcc g++),或用一个 zig cc 包装脚本当 CC +export CC=/path/to/your/c-compiler +# 3) 默认的 FlashInfer 采样器在缺 cubin 时会 JIT(需要 nvcc);无 CUDA toolkit 时关掉它 +export VLLM_USE_FLASHINFER_SAMPLER=0 +``` + +另外 WSL2 无 CUDA toolkit 时建议 `--compilation-config '{"mode":0}'`(跳过 torch.compile,启动更快、不需要 C++ 编译器)。 + +## 6. 评测 + +评测集 **131 条**,全部**未出现在训练集模板**中的说法(含 27 条刻意构造的硬样本): +中文序数词+负索引、口语化包装(「帮我/麻烦/能不能」)、上下文注入(`[上下文] 当前项目: …`)、 +同义命令归一(删除/删掉/移除 → `del`)、绝对路径拒绝、运行目标索引推断… + +| 指标 | 基座 Qwen2.5-1.5B-Instruct | **本模型(LoRA 微调后)** | +|---|---|---| +| 输出是合法 JSON | 131/131 = 100% | **131/131 = 100%** | +| 与标准命令**完全一致**(exact) | 39/131 = 29.8% | **122/131 = 93.1%** | + +> 基座 JSON 合法率也是 100%,是因为语法和输出格式就写在 system prompt 里 —— +> 但「知道语法」≠「会用语法」:55 个命令族里有 39 个族的准确率发生变化, +> 基座在序号、配置、闲聊、拒绝这些地方大面积翻车。 + +几族的对比(体现微调到底学到了什么): + +| 命令族 | 基座 | 本模型 | 说明 | +|---|---|---|---| +| `close_index_neg`(倒数第 N 个) | 0/7 | **7/7** | 中文序数词 → 负索引 | +| `del`(删除/删掉/移除归一) | 5/9 | **9/9** | 同义命令规范形式 | +| `new_typed`(`new py/rs/md/c` 省后缀) | 0/4 | **4/4** | 类型子命令 | +| `run_add_ctx`(上下文推索引) | 0/5 | **5/5** | `max+1` 索引推断 | +| `reject_abs_dir` / `reject_abs_file` | 0/4 / 0/3 | **4/4 / 3/3** | 固定拒绝话术 | +| `open_project_fwd`(正斜杠归一) | 0/3 | **3/3** | `/` → `\` | +| `chat_long`(超 20 字固定话术) | 0/3 | **3/3** | 逐字背下话术 | +| `config_*`(add/get/update/remove) | 0/1~0/3 | **多数转正** | 仍是最弱的一族,见局限 | + +**未满分的族(诚实列出,共 8 个)**:`config_get` 2/3、`config_remove` 1/3、 +`hard/config` 2/3、`hard/close_index` 1/2、`hard/open_project` 1/2、`hard/reject_abs` 1/2、 +`hard/reject_run_global` 0/1、`chat_short` 0/1。 +典型错误是**配置层级判断**:用户说「去掉配置项 X」时,模型输出 `config remove -p X`, +而标准答案是 `config remove -r X`(这类「默认层级到底该取谁」的语义在数据里信号不够强)。 + +## 7. 输入输出示例(**真实推理输出**,取自上述评测集) + +``` +输入:关掉倒数第5个文件 +输出:{"intent": "command", "cmd": "close -5"} + +输入:打开 C:/Users/yujia/PycharmProjects/ai-notes +输出:{"intent": "command", "cmd": "open project C:\\Users\\yujia\\PycharmProjects\\ai-notes"} + (正斜杠已按规则归一为反斜杠) + +输入:删除 C:/Users/yujia/PycharmProjects/tmf/src/utils.py +输出:{"intent": "reject", "reply": "您没有权限!"} + (绝对路径出现在 del 上 → 固定拒绝) + +输入:[上下文] 已有运行目标: target3 + 加个运行目标 测试=.venv\Scripts\python.exe -m uvicorn app:api --port 8000 +输出:{"intent": "command", "cmd": "run 测试=.venv\\Scripts\\python.exe -m uvicorn app:api --port 8000"} + +输入:把 darkhorse.code.editor.tab_size 设置成 4,存到运行时 +输出:{"intent": "command", "cmd": "config add -r darkhorse.code.editor.tab_size=4"} + +输入:帮我建一个 rust 的模块,叫 order_service +输出:{"intent": "command", "cmd": "new rs order_service"} + +输入:D:\Projects\Python\Code-Rag 是Kotlin项目 +输出:{"intent": "command", "cmd": "project lang kotlin D:\\Projects\\Python\\Code-Rag"} + +输入:你觉得我该不该辞职去创业,帮我分析一下利弊 +输出:{"intent": "chat", "reply": "你还是好好工作吧,房贷还清了吗?车贷还清了吗?"} +``` + +## 8. 训练细节 + +| 项 | 取值 | +|---|---| +| 基座 | `Qwen/Qwen2.5-1.5B-Instruct` | +| 微调方式 | LoRA(`r=16`、`alpha=32`、`dropout=0.05`),训练后**合并**成自包含权重 | +| 可训练参数 | 约 1.18%(其余冻结) | +| 训练数据 | **1638 条**(模板 + 槽位合成,`--seed 42` 完全可复现,不用 LLM 生成) | +| 超参 | `lr=2e-4`、`warmup_ratio=0.03`、2 epoch、`max_len=704`、`micro_bs=1` × `grad_accum=8` → 410 步 | +| 耗时 / 显存 | 45.1 分钟 / 峰值 **7.68 GB**(单卡 8 GB 笔记本 GPU) | +| loss | train 0.6818 → 0.0079(末轮均值 0.0319);eval 0.039 → 0.0092 | +| 精度 | bf16 | + +**数据为什么是合成的**:这个任务要学的是「**照抄**」而不是「记住」—— 命令里大量是路径、命令串、 +文件名。生成器把它们做成随机槽位,模型必须学会把输入里的路径原样搬进输出,而不是背样本; +同时保证可复现、零 API 成本、不会把外部 LLM 的错误学进去。 + +**三个刻意设计的难点**:① 约 15% 样本带上下文行,既要会用上下文补全、也要会忽略无关上下文; +② 随机加「帮我/麻烦/能不能/顺手」前缀与「吧/,谢谢/呀」后缀,放大表达多样性; +③ 27 条硬样本**只进评测集**,用来量真实泛化而不是量记忆。 + +## 9. 已知局限(请务必先读) + +1. **只懂 darkhorse-code 的命令语**。换个 CLI/编辑器的命令体系,必须重新造数据微调。 +2. **不是聊天模型**:闲聊会被压成 ≤20 字短回复,甚至回那句固定的「劝退」话术 —— 这是设计目标,不是 bug。 +3. **`config` 族最弱**(见 §6):配置层级(`-g/-p/-r`)的默认值判断会出错,接入方应对 + `config` 类输出做二次校验(层级缺失时按自己的业务规则补默认值)。 +4. **输出偶尔会被解析层救回来**:小模型可能把 JSON 包在 ```json 代码块里、或路径里写单反斜杠 + (非法 JSON 转义)。建议接入方保留「抠 JSON + 修非法转义」的解析兜底,而不是直接 `json.loads`。 +5. **上下文格式是约定**:必须写成 `[上下文] <内容>` 并**另起一行**再接用户原话, + 这是训练时的格式,别自由发挥。 +6. **1.5B 的常识/推理上限**:超出命令翻译的语义理解(比如反讽、多轮澄清)不要指望它。 + +## 10. 文件说明 + +| 文件 | 说明 | +|---|---| +| `model.safetensors` | bf16 合并权重(约 2.9 GB),自包含、无需 adapter | +| `chat_template.jinja` | **模型配套的 chat template**,请务必用它(vLLM `--chat-template` 指向它) | +| `config.json` / `generation_config.json` | Qwen2 结构(28 层、2 KV head、`max_position_embeddings=32768`) | +| `tokenizer.json` / `tokenizer_config.json` / `vocab.json` / `merges.txt` / `added_tokens.json` | tokenizer 全套 | +| `configuration.json` | ModelScope 标记文件 | + +## 11. 许可与致谢 + +- 本模型以 **Apache-2.0** 许可发布,与基座 `Qwen/Qwen2.5-1.5B-Instruct` 一致; + 使用前请同时遵守基座模型的许可条款。 +- 感谢 Qwen 团队开源的 Qwen2.5 系列。 +- 训练所用数据为由模板与槽位**程序生成**的合成数据,不含任何真实用户数据或第三方隐私内容。 diff --git a/added_tokens.json b/added_tokens.json new file mode 100644 index 0000000..06135f3 --- /dev/null +++ b/added_tokens.json @@ -0,0 +1,24 @@ +{ + "": 151658, + "": 151657, + "<|box_end|>": 151649, + "<|box_start|>": 151648, + "<|endoftext|>": 151643, + "<|file_sep|>": 151664, + "<|fim_middle|>": 151660, + "<|fim_pad|>": 151662, + "<|fim_prefix|>": 151659, + "<|fim_suffix|>": 151661, + "<|im_end|>": 151645, + "<|im_start|>": 151644, + "<|image_pad|>": 151655, + "<|object_ref_end|>": 151647, + "<|object_ref_start|>": 151646, + "<|quad_end|>": 151651, + "<|quad_start|>": 151650, + "<|repo_name|>": 151663, + "<|video_pad|>": 151656, + "<|vision_end|>": 151653, + "<|vision_pad|>": 151654, + "<|vision_start|>": 151652 +} diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..9840d40 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/config.json b/config.json new file mode 100644 index 0000000..8337f2e --- /dev/null +++ b/config.json @@ -0,0 +1,58 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "hidden_act": "silu", + "hidden_size": 1536, + "initializer_range": 0.02, + "intermediate_size": 8960, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 32768, + "max_window_layers": 21, + "model_type": "qwen2", + "num_attention_heads": 12, + "num_hidden_layers": 28, + "num_key_value_heads": 2, + "rms_norm_eps": 1e-06, + "rope_scaling": null, + "rope_theta": 1000000.0, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "4.57.6", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..9e26dfe --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..dfc1107 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,14 @@ +{ + "bos_token_id": 151643, + "pad_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "repetition_penalty": 1.1, + "temperature": 0.7, + "top_p": 0.8, + "top_k": 20, + "transformers_version": "4.37.0" +} diff --git a/merges.txt b/merges.txt new file mode 100644 index 0000000..80c1a19 --- /dev/null +++ b/merges.txt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5 +size 1671853 diff --git a/model-00001-of-00057.safetensors b/model-00001-of-00057.safetensors new file mode 100644 index 0000000..8f64769 --- /dev/null +++ b/model-00001-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:40b7e6a365b661ef9a18e28c26e71c942a28c7eda6ded217cc6641333e026a73 +size 466747528 diff --git a/model-00002-of-00057.safetensors b/model-00002-of-00057.safetensors new file mode 100644 index 0000000..59eb8e0 --- /dev/null +++ b/model-00002-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0e424dec445006aa68ba120d04ff5fde7137ffd7ebfaa694ef9a79aba9a7d0be +size 38540176 diff --git a/model-00003-of-00057.safetensors b/model-00003-of-00057.safetensors new file mode 100644 index 0000000..1c4bc96 --- /dev/null +++ b/model-00003-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:107f4d225c81534f5ca05c78d577ecf0de1532bcb0c1a68c83365f82efed1794 +size 61353056 diff --git a/model-00004-of-00057.safetensors b/model-00004-of-00057.safetensors new file mode 100644 index 0000000..b436c94 --- /dev/null +++ b/model-00004-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1e72c454caa0c0f9c5241a00a7bdb29a28d39cb1cb46105ea5e80e9935d48099 +size 59769200 diff --git a/model-00005-of-00057.safetensors b/model-00005-of-00057.safetensors new file mode 100644 index 0000000..b8603e4 --- /dev/null +++ b/model-00005-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5b422b8db4bb1e9b513924ca9e8970baf1fb0e2d28a07a143116aec50ebf9e0c +size 38546536 diff --git a/model-00006-of-00057.safetensors b/model-00006-of-00057.safetensors new file mode 100644 index 0000000..2f91691 --- /dev/null +++ b/model-00006-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27c8e431125b19bf2193c62e9acf40b2003ea674a891442d83f6c701e92cd96c +size 55050496 diff --git a/model-00007-of-00057.safetensors b/model-00007-of-00057.safetensors new file mode 100644 index 0000000..fc839bb --- /dev/null +++ b/model-00007-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:43a1bb4fddd96c3936958f20f900e9374f7216bd7f0f1a138d94fc91840494a0 +size 38546536 diff --git a/model-00008-of-00057.safetensors b/model-00008-of-00057.safetensors new file mode 100644 index 0000000..9c79709 --- /dev/null +++ b/model-00008-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de579e160842610e4bf68ddf214d9cbb16ceb7964c01559fdd2abb09321cf387 +size 55050496 diff --git a/model-00009-of-00057.safetensors b/model-00009-of-00057.safetensors new file mode 100644 index 0000000..59e8505 --- /dev/null +++ b/model-00009-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f5502ae20af0b18fb010125ef28364ec6479adc747e24b9e9139c6f8f3952d0f +size 38546536 diff --git a/model-00010-of-00057.safetensors b/model-00010-of-00057.safetensors new file mode 100644 index 0000000..ef98875 --- /dev/null +++ b/model-00010-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:960196906e2f08ed4047e0373c36679d4693f8e6fabbef4c9b0115d27be0400f +size 55050496 diff --git a/model-00011-of-00057.safetensors b/model-00011-of-00057.safetensors new file mode 100644 index 0000000..579383b --- /dev/null +++ b/model-00011-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d5511820e9e5f5e2f22336d11034e74a90aeae4a5c9fd45dab1d742f991bc59b +size 38546536 diff --git a/model-00012-of-00057.safetensors b/model-00012-of-00057.safetensors new file mode 100644 index 0000000..65c79ab --- /dev/null +++ b/model-00012-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1d626fb78d605e8ded76ebed21c5d56093444039e4f033e4be8c5792da388003 +size 55050496 diff --git a/model-00013-of-00057.safetensors b/model-00013-of-00057.safetensors new file mode 100644 index 0000000..c80144f --- /dev/null +++ b/model-00013-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:aa0597fb6748edd075389a3bf03d5ff26e45277e46c589b26b4169a3e1e6c300 +size 38546536 diff --git a/model-00014-of-00057.safetensors b/model-00014-of-00057.safetensors new file mode 100644 index 0000000..d7e8852 --- /dev/null +++ b/model-00014-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc5713f7f58e4471dc6024d398fd583ab87ccd4d924073dae498cd290afbcb78 +size 55050496 diff --git a/model-00015-of-00057.safetensors b/model-00015-of-00057.safetensors new file mode 100644 index 0000000..b649eb6 --- /dev/null +++ b/model-00015-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3bf78a5882022a5690824dee00f2b622057d62d88b9edb5a7bb158689eff9ee0 +size 38546536 diff --git a/model-00016-of-00057.safetensors b/model-00016-of-00057.safetensors new file mode 100644 index 0000000..ccced08 --- /dev/null +++ b/model-00016-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:94d03f1a2bf7d4432a16aaa42be2ea0c133b49092d768a7142a927761eb47937 +size 55050496 diff --git a/model-00017-of-00057.safetensors b/model-00017-of-00057.safetensors new file mode 100644 index 0000000..6fb38b7 --- /dev/null +++ b/model-00017-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:51b819f0ecf1e11c8615723717faf65bc97bf62fd5f6dcde0223675b9d47a268 +size 38546536 diff --git a/model-00018-of-00057.safetensors b/model-00018-of-00057.safetensors new file mode 100644 index 0000000..600d084 --- /dev/null +++ b/model-00018-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dba6f1d847b5c9b6fb81504a8e69e98677013835110ecfeb11ec0e4cdbf3bfb2 +size 55050496 diff --git a/model-00019-of-00057.safetensors b/model-00019-of-00057.safetensors new file mode 100644 index 0000000..cffd8d1 --- /dev/null +++ b/model-00019-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7214926cb491e9ce4fdb493e97210531d7e52e9c9a2beedecbab2bf2cd2f1f12 +size 38546536 diff --git a/model-00020-of-00057.safetensors b/model-00020-of-00057.safetensors new file mode 100644 index 0000000..138444b --- /dev/null +++ b/model-00020-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b74b2b946b5809eb8ad4a4797d9bf5f2b50b7867d1c5a7deaab410e2cd983be8 +size 55050496 diff --git a/model-00021-of-00057.safetensors b/model-00021-of-00057.safetensors new file mode 100644 index 0000000..ca80f3d --- /dev/null +++ b/model-00021-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:260746ed008a026b63945f0883b8cceb1f1c0d8f1a05ed47be4fe7af36061d7b +size 38546536 diff --git a/model-00022-of-00057.safetensors b/model-00022-of-00057.safetensors new file mode 100644 index 0000000..5e78109 --- /dev/null +++ b/model-00022-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e19822c7fc679bb79f53353b1136748d6b4ad7126393a1adbde7b1012764fcb1 +size 55050496 diff --git a/model-00023-of-00057.safetensors b/model-00023-of-00057.safetensors new file mode 100644 index 0000000..db67c50 --- /dev/null +++ b/model-00023-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7cdbb586278e92f657d33c1a202dcf4dd6b5e258db6437bdfbea6ac38289e4be +size 38546544 diff --git a/model-00024-of-00057.safetensors b/model-00024-of-00057.safetensors new file mode 100644 index 0000000..65f846a --- /dev/null +++ b/model-00024-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:29cf52078e1a6ef26b017f6ea0f410b38973dd1ff6be5870c8ad807a137c1af7 +size 55050496 diff --git a/model-00025-of-00057.safetensors b/model-00025-of-00057.safetensors new file mode 100644 index 0000000..c6dffc1 --- /dev/null +++ b/model-00025-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9057960ad72e5b8f6c417500efe829840b1d209953c7588c0c01697d486e38d +size 38546544 diff --git a/model-00026-of-00057.safetensors b/model-00026-of-00057.safetensors new file mode 100644 index 0000000..8a60e50 --- /dev/null +++ b/model-00026-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:26cedc91f5b3e8219448f9b34ccfd56a9f8aeb0616582d5d8837d5d8086330fc +size 55050496 diff --git a/model-00027-of-00057.safetensors b/model-00027-of-00057.safetensors new file mode 100644 index 0000000..1beb86f --- /dev/null +++ b/model-00027-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b71d9ad7856d40b601f5a03860545f2b43df13c52e154ca41491cf08f2777af0 +size 38546544 diff --git a/model-00028-of-00057.safetensors b/model-00028-of-00057.safetensors new file mode 100644 index 0000000..8509d11 --- /dev/null +++ b/model-00028-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8f712681a834c3f680acbe11f7b35dce89bd3763c486a71f28bdabc0345692e +size 55050496 diff --git a/model-00029-of-00057.safetensors b/model-00029-of-00057.safetensors new file mode 100644 index 0000000..4924fc5 --- /dev/null +++ b/model-00029-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc08b6f95bd1235edf0d2e1ee8e8c787133906494d68127b0a6256286696ee4f +size 38546544 diff --git a/model-00030-of-00057.safetensors b/model-00030-of-00057.safetensors new file mode 100644 index 0000000..6a6c6d9 --- /dev/null +++ b/model-00030-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4436dcae1abb5721cd1740583c8fdd5d6917577e6e57c6a61f22e361a6031a7d +size 55050496 diff --git a/model-00031-of-00057.safetensors b/model-00031-of-00057.safetensors new file mode 100644 index 0000000..7574193 --- /dev/null +++ b/model-00031-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b5165e80fc0fb8fd8e781a33e586535281cdba313766326b2c8a01eca1b51cd +size 38546544 diff --git a/model-00032-of-00057.safetensors b/model-00032-of-00057.safetensors new file mode 100644 index 0000000..7acb1c9 --- /dev/null +++ b/model-00032-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7c4c097b5664f5e4f0fa6240ad22a0e728a92dfffc9e4e7f8c438d6f4f7088de +size 55050496 diff --git a/model-00033-of-00057.safetensors b/model-00033-of-00057.safetensors new file mode 100644 index 0000000..467ca0a --- /dev/null +++ b/model-00033-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:507e36dd372754a19cf75c14baecaa44594b3a2090577ab31aaf9d86b289e6e0 +size 38546544 diff --git a/model-00034-of-00057.safetensors b/model-00034-of-00057.safetensors new file mode 100644 index 0000000..7e33e4f --- /dev/null +++ b/model-00034-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4623bf9fc7f2bcf68cdf5bf3f88ee61505a41fcc74e28c90d448b3f6ccdefc8 +size 55050496 diff --git a/model-00035-of-00057.safetensors b/model-00035-of-00057.safetensors new file mode 100644 index 0000000..9777c35 --- /dev/null +++ b/model-00035-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:50a2fc54c87bc5f00275c6aeabe443e32295f95c1fe638968be0f4015543d059 +size 38546544 diff --git a/model-00036-of-00057.safetensors b/model-00036-of-00057.safetensors new file mode 100644 index 0000000..eb73c3e --- /dev/null +++ b/model-00036-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6cd4da12a345f292720ac49729277613070051715201ff60001b6069995ea7e2 +size 55050496 diff --git a/model-00037-of-00057.safetensors b/model-00037-of-00057.safetensors new file mode 100644 index 0000000..853f491 --- /dev/null +++ b/model-00037-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fadf5efa6c63de94ac68823a8d8b94eb37063d3fd92058271f55bbef41bc5952 +size 38546544 diff --git a/model-00038-of-00057.safetensors b/model-00038-of-00057.safetensors new file mode 100644 index 0000000..8d9acaa --- /dev/null +++ b/model-00038-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd4881fe3cd500ed5cf7f4d75c6986430586f107daa796072721a2491aadd0a4 +size 55050496 diff --git a/model-00039-of-00057.safetensors b/model-00039-of-00057.safetensors new file mode 100644 index 0000000..0643644 --- /dev/null +++ b/model-00039-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:478372df4b06c239a57eb2fd0a56703a6cebf5da36a7ea86ee6360b0450bc22f +size 38546544 diff --git a/model-00040-of-00057.safetensors b/model-00040-of-00057.safetensors new file mode 100644 index 0000000..bd70d47 --- /dev/null +++ b/model-00040-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3ea8051e5ff9bd279e92ad1506d866c2ecbdd945e3700cfd793c9caf7c9fda24 +size 55050496 diff --git a/model-00041-of-00057.safetensors b/model-00041-of-00057.safetensors new file mode 100644 index 0000000..899f549 --- /dev/null +++ b/model-00041-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:17b9e9feb1763522dd2e6cdb7d494e1718af1100ac64b306f6936c31d89217ae +size 38546544 diff --git a/model-00042-of-00057.safetensors b/model-00042-of-00057.safetensors new file mode 100644 index 0000000..89f4ca9 --- /dev/null +++ b/model-00042-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:575563b5969455602b208be41fbc8bedd27f5a0fdf8ae5cfc156ec557b09a0d7 +size 55050496 diff --git a/model-00043-of-00057.safetensors b/model-00043-of-00057.safetensors new file mode 100644 index 0000000..51b13fe --- /dev/null +++ b/model-00043-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:651673ab04ecae55e7c558c99c5e5c5a9f09b4744695095b8cd1c911c659c8db +size 38546544 diff --git a/model-00044-of-00057.safetensors b/model-00044-of-00057.safetensors new file mode 100644 index 0000000..3d4d9bc --- /dev/null +++ b/model-00044-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:70f4830f07fe2791b5e1f62ae129b8d7ec2d9fd77ed212281d529a3cedb4e237 +size 55050496 diff --git a/model-00045-of-00057.safetensors b/model-00045-of-00057.safetensors new file mode 100644 index 0000000..2cf8364 --- /dev/null +++ b/model-00045-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4098ae1e96480159035b629ffb7b80b31fff216c1ca966280d7d673e5672d8fa +size 38546544 diff --git a/model-00046-of-00057.safetensors b/model-00046-of-00057.safetensors new file mode 100644 index 0000000..fcd1e48 --- /dev/null +++ b/model-00046-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ef1c309e528f7a231880fed9558834f8cf0404084712c892551cc8168a884aa +size 55050496 diff --git a/model-00047-of-00057.safetensors b/model-00047-of-00057.safetensors new file mode 100644 index 0000000..21322da --- /dev/null +++ b/model-00047-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1e992519bcb9cadbd852193798259ad9710acb7d16e1afeec139ff1572f06fef +size 38546544 diff --git a/model-00048-of-00057.safetensors b/model-00048-of-00057.safetensors new file mode 100644 index 0000000..0444a72 --- /dev/null +++ b/model-00048-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6e937439fe4c443b25d753422fa7ac838aaa34ca5ce0952af2d17e8b5ac014b8 +size 55050496 diff --git a/model-00049-of-00057.safetensors b/model-00049-of-00057.safetensors new file mode 100644 index 0000000..b674961 --- /dev/null +++ b/model-00049-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc34d6d302f7ed475e4937054136c72ceace861b3a309b4d1eaded8693fc1fec +size 38546544 diff --git a/model-00050-of-00057.safetensors b/model-00050-of-00057.safetensors new file mode 100644 index 0000000..2e2b400 --- /dev/null +++ b/model-00050-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b86abe7ee395a1dbbf16dc34ce2e217daedb9c5d406667f154937b0d0dcceea6 +size 55050496 diff --git a/model-00051-of-00057.safetensors b/model-00051-of-00057.safetensors new file mode 100644 index 0000000..85b9da3 --- /dev/null +++ b/model-00051-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:089b7b9ba149df121622a7a87fcebef93658960f67aa398974382e09bf0eec89 +size 38546544 diff --git a/model-00052-of-00057.safetensors b/model-00052-of-00057.safetensors new file mode 100644 index 0000000..ac548c5 --- /dev/null +++ b/model-00052-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:71c1362e4063164535d363adb831f9bfe5dc69016ebdc0115b6c152decd912ec +size 55050496 diff --git a/model-00053-of-00057.safetensors b/model-00053-of-00057.safetensors new file mode 100644 index 0000000..b497e1c --- /dev/null +++ b/model-00053-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2f7172e683cd9661414244c6e12bdef297373966e152345ca360cbdcc5f939e6 +size 38546544 diff --git a/model-00054-of-00057.safetensors b/model-00054-of-00057.safetensors new file mode 100644 index 0000000..6190f0b --- /dev/null +++ b/model-00054-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0fd66b78de9e51288dc84f6d4ea435586bfc6dbc17f0393d61043bbf4cb7ae71 +size 55050496 diff --git a/model-00055-of-00057.safetensors b/model-00055-of-00057.safetensors new file mode 100644 index 0000000..aea4ad4 --- /dev/null +++ b/model-00055-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10d83ed67dafeeb78c5508b58c6ed34a0557257402d87957ffe7dcec949aa17a +size 38546544 diff --git a/model-00056-of-00057.safetensors b/model-00056-of-00057.safetensors new file mode 100644 index 0000000..c906a57 --- /dev/null +++ b/model-00056-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:17cb8f0e6a2f4914e5fced6e69b6a65cda80d077e754754e6e9d63603ed3caec +size 55050496 diff --git a/model-00057-of-00057.safetensors b/model-00057-of-00057.safetensors new file mode 100644 index 0000000..3cac837 --- /dev/null +++ b/model-00057-of-00057.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3aa220232e481fd80a3472c157de1bd936cb479716e2fd8f9a42c50b7da90e49 +size 27534784 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..52796db --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,346 @@ +{ + "metadata": { + "total_parameters": 1543714304, + "total_size": 3087428608 + }, + "weight_map": { + "model.embed_tokens.weight": "model-00001-of-00057.safetensors", + "model.layers.0.input_layernorm.weight": "model-00003-of-00057.safetensors", + "model.layers.0.mlp.down_proj.weight": "model-00003-of-00057.safetensors", + "model.layers.0.mlp.gate_proj.weight": "model-00002-of-00057.safetensors", + "model.layers.0.mlp.up_proj.weight": "model-00003-of-00057.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model-00003-of-00057.safetensors", + "model.layers.0.self_attn.k_proj.bias": "model-00002-of-00057.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model-00002-of-00057.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model-00002-of-00057.safetensors", + "model.layers.0.self_attn.q_proj.bias": "model-00002-of-00057.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model-00002-of-00057.safetensors", + "model.layers.0.self_attn.v_proj.bias": "model-00002-of-00057.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model-00002-of-00057.safetensors", + "model.layers.1.input_layernorm.weight": "model-00005-of-00057.safetensors", + "model.layers.1.mlp.down_proj.weight": "model-00005-of-00057.safetensors", + "model.layers.1.mlp.gate_proj.weight": "model-00004-of-00057.safetensors", + "model.layers.1.mlp.up_proj.weight": "model-00004-of-00057.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model-00005-of-00057.safetensors", + "model.layers.1.self_attn.k_proj.bias": "model-00003-of-00057.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model-00003-of-00057.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model-00004-of-00057.safetensors", + "model.layers.1.self_attn.q_proj.bias": "model-00003-of-00057.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model-00003-of-00057.safetensors", + "model.layers.1.self_attn.v_proj.bias": "model-00003-of-00057.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model-00003-of-00057.safetensors", + "model.layers.10.input_layernorm.weight": "model-00023-of-00057.safetensors", + "model.layers.10.mlp.down_proj.weight": "model-00023-of-00057.safetensors", + "model.layers.10.mlp.gate_proj.weight": "model-00022-of-00057.safetensors", + "model.layers.10.mlp.up_proj.weight": "model-00022-of-00057.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model-00023-of-00057.safetensors", + "model.layers.10.self_attn.k_proj.bias": "model-00021-of-00057.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model-00021-of-00057.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model-00021-of-00057.safetensors", + "model.layers.10.self_attn.q_proj.bias": "model-00021-of-00057.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model-00021-of-00057.safetensors", + "model.layers.10.self_attn.v_proj.bias": "model-00021-of-00057.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model-00021-of-00057.safetensors", + "model.layers.11.input_layernorm.weight": "model-00025-of-00057.safetensors", + "model.layers.11.mlp.down_proj.weight": "model-00025-of-00057.safetensors", + "model.layers.11.mlp.gate_proj.weight": "model-00024-of-00057.safetensors", + "model.layers.11.mlp.up_proj.weight": "model-00024-of-00057.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model-00025-of-00057.safetensors", + "model.layers.11.self_attn.k_proj.bias": "model-00023-of-00057.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00023-of-00057.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00023-of-00057.safetensors", + "model.layers.11.self_attn.q_proj.bias": "model-00023-of-00057.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00023-of-00057.safetensors", + "model.layers.11.self_attn.v_proj.bias": "model-00023-of-00057.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00023-of-00057.safetensors", + "model.layers.12.input_layernorm.weight": "model-00027-of-00057.safetensors", + "model.layers.12.mlp.down_proj.weight": "model-00027-of-00057.safetensors", + "model.layers.12.mlp.gate_proj.weight": "model-00026-of-00057.safetensors", + "model.layers.12.mlp.up_proj.weight": "model-00026-of-00057.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model-00027-of-00057.safetensors", + "model.layers.12.self_attn.k_proj.bias": "model-00025-of-00057.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model-00025-of-00057.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model-00025-of-00057.safetensors", + "model.layers.12.self_attn.q_proj.bias": "model-00025-of-00057.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model-00025-of-00057.safetensors", + "model.layers.12.self_attn.v_proj.bias": "model-00025-of-00057.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model-00025-of-00057.safetensors", + "model.layers.13.input_layernorm.weight": "model-00029-of-00057.safetensors", + "model.layers.13.mlp.down_proj.weight": "model-00029-of-00057.safetensors", + "model.layers.13.mlp.gate_proj.weight": "model-00028-of-00057.safetensors", + "model.layers.13.mlp.up_proj.weight": "model-00028-of-00057.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model-00029-of-00057.safetensors", + "model.layers.13.self_attn.k_proj.bias": "model-00027-of-00057.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model-00027-of-00057.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model-00027-of-00057.safetensors", + "model.layers.13.self_attn.q_proj.bias": "model-00027-of-00057.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model-00027-of-00057.safetensors", + "model.layers.13.self_attn.v_proj.bias": "model-00027-of-00057.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model-00027-of-00057.safetensors", + "model.layers.14.input_layernorm.weight": "model-00031-of-00057.safetensors", + "model.layers.14.mlp.down_proj.weight": "model-00031-of-00057.safetensors", + "model.layers.14.mlp.gate_proj.weight": "model-00030-of-00057.safetensors", + "model.layers.14.mlp.up_proj.weight": "model-00030-of-00057.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model-00031-of-00057.safetensors", + "model.layers.14.self_attn.k_proj.bias": "model-00029-of-00057.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model-00029-of-00057.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model-00029-of-00057.safetensors", + "model.layers.14.self_attn.q_proj.bias": "model-00029-of-00057.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model-00029-of-00057.safetensors", + "model.layers.14.self_attn.v_proj.bias": "model-00029-of-00057.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model-00029-of-00057.safetensors", + "model.layers.15.input_layernorm.weight": "model-00033-of-00057.safetensors", + "model.layers.15.mlp.down_proj.weight": "model-00033-of-00057.safetensors", + "model.layers.15.mlp.gate_proj.weight": "model-00032-of-00057.safetensors", + "model.layers.15.mlp.up_proj.weight": "model-00032-of-00057.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model-00033-of-00057.safetensors", + "model.layers.15.self_attn.k_proj.bias": "model-00031-of-00057.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00031-of-00057.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00031-of-00057.safetensors", + "model.layers.15.self_attn.q_proj.bias": "model-00031-of-00057.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00031-of-00057.safetensors", + "model.layers.15.self_attn.v_proj.bias": "model-00031-of-00057.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00031-of-00057.safetensors", + "model.layers.16.input_layernorm.weight": "model-00035-of-00057.safetensors", + "model.layers.16.mlp.down_proj.weight": "model-00035-of-00057.safetensors", + "model.layers.16.mlp.gate_proj.weight": "model-00034-of-00057.safetensors", + "model.layers.16.mlp.up_proj.weight": "model-00034-of-00057.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model-00035-of-00057.safetensors", + "model.layers.16.self_attn.k_proj.bias": "model-00033-of-00057.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model-00033-of-00057.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model-00033-of-00057.safetensors", + "model.layers.16.self_attn.q_proj.bias": "model-00033-of-00057.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model-00033-of-00057.safetensors", + "model.layers.16.self_attn.v_proj.bias": "model-00033-of-00057.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model-00033-of-00057.safetensors", + "model.layers.17.input_layernorm.weight": "model-00037-of-00057.safetensors", + "model.layers.17.mlp.down_proj.weight": "model-00037-of-00057.safetensors", + "model.layers.17.mlp.gate_proj.weight": "model-00036-of-00057.safetensors", + "model.layers.17.mlp.up_proj.weight": "model-00036-of-00057.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model-00037-of-00057.safetensors", + "model.layers.17.self_attn.k_proj.bias": "model-00035-of-00057.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model-00035-of-00057.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model-00035-of-00057.safetensors", + "model.layers.17.self_attn.q_proj.bias": "model-00035-of-00057.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model-00035-of-00057.safetensors", + "model.layers.17.self_attn.v_proj.bias": "model-00035-of-00057.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model-00035-of-00057.safetensors", + "model.layers.18.input_layernorm.weight": "model-00039-of-00057.safetensors", + "model.layers.18.mlp.down_proj.weight": "model-00039-of-00057.safetensors", + "model.layers.18.mlp.gate_proj.weight": "model-00038-of-00057.safetensors", + "model.layers.18.mlp.up_proj.weight": "model-00038-of-00057.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model-00039-of-00057.safetensors", + "model.layers.18.self_attn.k_proj.bias": "model-00037-of-00057.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model-00037-of-00057.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model-00037-of-00057.safetensors", + "model.layers.18.self_attn.q_proj.bias": "model-00037-of-00057.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model-00037-of-00057.safetensors", + "model.layers.18.self_attn.v_proj.bias": "model-00037-of-00057.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model-00037-of-00057.safetensors", + "model.layers.19.input_layernorm.weight": "model-00041-of-00057.safetensors", + "model.layers.19.mlp.down_proj.weight": "model-00041-of-00057.safetensors", + "model.layers.19.mlp.gate_proj.weight": "model-00040-of-00057.safetensors", + "model.layers.19.mlp.up_proj.weight": "model-00040-of-00057.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model-00041-of-00057.safetensors", + "model.layers.19.self_attn.k_proj.bias": "model-00039-of-00057.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00039-of-00057.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00039-of-00057.safetensors", + "model.layers.19.self_attn.q_proj.bias": "model-00039-of-00057.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00039-of-00057.safetensors", + "model.layers.19.self_attn.v_proj.bias": "model-00039-of-00057.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00039-of-00057.safetensors", + "model.layers.2.input_layernorm.weight": "model-00007-of-00057.safetensors", + "model.layers.2.mlp.down_proj.weight": "model-00007-of-00057.safetensors", + "model.layers.2.mlp.gate_proj.weight": "model-00006-of-00057.safetensors", + "model.layers.2.mlp.up_proj.weight": "model-00006-of-00057.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model-00007-of-00057.safetensors", + "model.layers.2.self_attn.k_proj.bias": "model-00005-of-00057.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model-00005-of-00057.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model-00005-of-00057.safetensors", + "model.layers.2.self_attn.q_proj.bias": "model-00005-of-00057.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model-00005-of-00057.safetensors", + "model.layers.2.self_attn.v_proj.bias": "model-00005-of-00057.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model-00005-of-00057.safetensors", + "model.layers.20.input_layernorm.weight": "model-00043-of-00057.safetensors", + "model.layers.20.mlp.down_proj.weight": "model-00043-of-00057.safetensors", + "model.layers.20.mlp.gate_proj.weight": "model-00042-of-00057.safetensors", + "model.layers.20.mlp.up_proj.weight": "model-00042-of-00057.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model-00043-of-00057.safetensors", + "model.layers.20.self_attn.k_proj.bias": "model-00041-of-00057.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model-00041-of-00057.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model-00041-of-00057.safetensors", + "model.layers.20.self_attn.q_proj.bias": "model-00041-of-00057.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model-00041-of-00057.safetensors", + "model.layers.20.self_attn.v_proj.bias": "model-00041-of-00057.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model-00041-of-00057.safetensors", + "model.layers.21.input_layernorm.weight": "model-00045-of-00057.safetensors", + "model.layers.21.mlp.down_proj.weight": "model-00045-of-00057.safetensors", + "model.layers.21.mlp.gate_proj.weight": "model-00044-of-00057.safetensors", + "model.layers.21.mlp.up_proj.weight": "model-00044-of-00057.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model-00045-of-00057.safetensors", + "model.layers.21.self_attn.k_proj.bias": "model-00043-of-00057.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model-00043-of-00057.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model-00043-of-00057.safetensors", + "model.layers.21.self_attn.q_proj.bias": "model-00043-of-00057.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model-00043-of-00057.safetensors", + "model.layers.21.self_attn.v_proj.bias": "model-00043-of-00057.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model-00043-of-00057.safetensors", + "model.layers.22.input_layernorm.weight": "model-00047-of-00057.safetensors", + "model.layers.22.mlp.down_proj.weight": "model-00047-of-00057.safetensors", + "model.layers.22.mlp.gate_proj.weight": "model-00046-of-00057.safetensors", + "model.layers.22.mlp.up_proj.weight": "model-00046-of-00057.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model-00047-of-00057.safetensors", + "model.layers.22.self_attn.k_proj.bias": "model-00045-of-00057.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model-00045-of-00057.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model-00045-of-00057.safetensors", + "model.layers.22.self_attn.q_proj.bias": "model-00045-of-00057.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model-00045-of-00057.safetensors", + "model.layers.22.self_attn.v_proj.bias": "model-00045-of-00057.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model-00045-of-00057.safetensors", + "model.layers.23.input_layernorm.weight": "model-00049-of-00057.safetensors", + "model.layers.23.mlp.down_proj.weight": "model-00049-of-00057.safetensors", + "model.layers.23.mlp.gate_proj.weight": "model-00048-of-00057.safetensors", + "model.layers.23.mlp.up_proj.weight": "model-00048-of-00057.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model-00049-of-00057.safetensors", + "model.layers.23.self_attn.k_proj.bias": "model-00047-of-00057.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00047-of-00057.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00047-of-00057.safetensors", + "model.layers.23.self_attn.q_proj.bias": "model-00047-of-00057.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00047-of-00057.safetensors", + "model.layers.23.self_attn.v_proj.bias": "model-00047-of-00057.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00047-of-00057.safetensors", + "model.layers.24.input_layernorm.weight": "model-00051-of-00057.safetensors", + "model.layers.24.mlp.down_proj.weight": "model-00051-of-00057.safetensors", + "model.layers.24.mlp.gate_proj.weight": "model-00050-of-00057.safetensors", + "model.layers.24.mlp.up_proj.weight": "model-00050-of-00057.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model-00051-of-00057.safetensors", + "model.layers.24.self_attn.k_proj.bias": "model-00049-of-00057.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model-00049-of-00057.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model-00049-of-00057.safetensors", + "model.layers.24.self_attn.q_proj.bias": "model-00049-of-00057.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model-00049-of-00057.safetensors", + "model.layers.24.self_attn.v_proj.bias": "model-00049-of-00057.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model-00049-of-00057.safetensors", + "model.layers.25.input_layernorm.weight": "model-00053-of-00057.safetensors", + "model.layers.25.mlp.down_proj.weight": "model-00053-of-00057.safetensors", + "model.layers.25.mlp.gate_proj.weight": "model-00052-of-00057.safetensors", + "model.layers.25.mlp.up_proj.weight": "model-00052-of-00057.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model-00053-of-00057.safetensors", + "model.layers.25.self_attn.k_proj.bias": "model-00051-of-00057.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model-00051-of-00057.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model-00051-of-00057.safetensors", + "model.layers.25.self_attn.q_proj.bias": "model-00051-of-00057.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model-00051-of-00057.safetensors", + "model.layers.25.self_attn.v_proj.bias": "model-00051-of-00057.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model-00051-of-00057.safetensors", + "model.layers.26.input_layernorm.weight": "model-00055-of-00057.safetensors", + "model.layers.26.mlp.down_proj.weight": "model-00055-of-00057.safetensors", + "model.layers.26.mlp.gate_proj.weight": "model-00054-of-00057.safetensors", + "model.layers.26.mlp.up_proj.weight": "model-00054-of-00057.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model-00055-of-00057.safetensors", + "model.layers.26.self_attn.k_proj.bias": "model-00053-of-00057.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model-00053-of-00057.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model-00053-of-00057.safetensors", + "model.layers.26.self_attn.q_proj.bias": "model-00053-of-00057.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model-00053-of-00057.safetensors", + "model.layers.26.self_attn.v_proj.bias": "model-00053-of-00057.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model-00053-of-00057.safetensors", + "model.layers.27.input_layernorm.weight": "model-00057-of-00057.safetensors", + "model.layers.27.mlp.down_proj.weight": "model-00057-of-00057.safetensors", + "model.layers.27.mlp.gate_proj.weight": "model-00056-of-00057.safetensors", + "model.layers.27.mlp.up_proj.weight": "model-00056-of-00057.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model-00057-of-00057.safetensors", + "model.layers.27.self_attn.k_proj.bias": "model-00055-of-00057.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00055-of-00057.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00055-of-00057.safetensors", + "model.layers.27.self_attn.q_proj.bias": "model-00055-of-00057.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00055-of-00057.safetensors", + "model.layers.27.self_attn.v_proj.bias": "model-00055-of-00057.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00055-of-00057.safetensors", + "model.layers.3.input_layernorm.weight": "model-00009-of-00057.safetensors", + "model.layers.3.mlp.down_proj.weight": "model-00009-of-00057.safetensors", + "model.layers.3.mlp.gate_proj.weight": "model-00008-of-00057.safetensors", + "model.layers.3.mlp.up_proj.weight": "model-00008-of-00057.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model-00009-of-00057.safetensors", + "model.layers.3.self_attn.k_proj.bias": "model-00007-of-00057.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00007-of-00057.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00007-of-00057.safetensors", + "model.layers.3.self_attn.q_proj.bias": "model-00007-of-00057.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00007-of-00057.safetensors", + "model.layers.3.self_attn.v_proj.bias": "model-00007-of-00057.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00007-of-00057.safetensors", + "model.layers.4.input_layernorm.weight": "model-00011-of-00057.safetensors", + "model.layers.4.mlp.down_proj.weight": "model-00011-of-00057.safetensors", + "model.layers.4.mlp.gate_proj.weight": "model-00010-of-00057.safetensors", + "model.layers.4.mlp.up_proj.weight": "model-00010-of-00057.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model-00011-of-00057.safetensors", + "model.layers.4.self_attn.k_proj.bias": "model-00009-of-00057.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model-00009-of-00057.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model-00009-of-00057.safetensors", + "model.layers.4.self_attn.q_proj.bias": "model-00009-of-00057.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model-00009-of-00057.safetensors", + "model.layers.4.self_attn.v_proj.bias": "model-00009-of-00057.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model-00009-of-00057.safetensors", + "model.layers.5.input_layernorm.weight": "model-00013-of-00057.safetensors", + "model.layers.5.mlp.down_proj.weight": "model-00013-of-00057.safetensors", + "model.layers.5.mlp.gate_proj.weight": "model-00012-of-00057.safetensors", + "model.layers.5.mlp.up_proj.weight": "model-00012-of-00057.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model-00013-of-00057.safetensors", + "model.layers.5.self_attn.k_proj.bias": "model-00011-of-00057.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model-00011-of-00057.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model-00011-of-00057.safetensors", + "model.layers.5.self_attn.q_proj.bias": "model-00011-of-00057.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model-00011-of-00057.safetensors", + "model.layers.5.self_attn.v_proj.bias": "model-00011-of-00057.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model-00011-of-00057.safetensors", + "model.layers.6.input_layernorm.weight": "model-00015-of-00057.safetensors", + "model.layers.6.mlp.down_proj.weight": "model-00015-of-00057.safetensors", + "model.layers.6.mlp.gate_proj.weight": "model-00014-of-00057.safetensors", + "model.layers.6.mlp.up_proj.weight": "model-00014-of-00057.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model-00015-of-00057.safetensors", + "model.layers.6.self_attn.k_proj.bias": "model-00013-of-00057.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model-00013-of-00057.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model-00013-of-00057.safetensors", + "model.layers.6.self_attn.q_proj.bias": "model-00013-of-00057.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model-00013-of-00057.safetensors", + "model.layers.6.self_attn.v_proj.bias": "model-00013-of-00057.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model-00013-of-00057.safetensors", + "model.layers.7.input_layernorm.weight": "model-00017-of-00057.safetensors", + "model.layers.7.mlp.down_proj.weight": "model-00017-of-00057.safetensors", + "model.layers.7.mlp.gate_proj.weight": "model-00016-of-00057.safetensors", + "model.layers.7.mlp.up_proj.weight": "model-00016-of-00057.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model-00017-of-00057.safetensors", + "model.layers.7.self_attn.k_proj.bias": "model-00015-of-00057.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00015-of-00057.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00015-of-00057.safetensors", + "model.layers.7.self_attn.q_proj.bias": "model-00015-of-00057.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00015-of-00057.safetensors", + "model.layers.7.self_attn.v_proj.bias": "model-00015-of-00057.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00015-of-00057.safetensors", + "model.layers.8.input_layernorm.weight": "model-00019-of-00057.safetensors", + "model.layers.8.mlp.down_proj.weight": "model-00019-of-00057.safetensors", + "model.layers.8.mlp.gate_proj.weight": "model-00018-of-00057.safetensors", + "model.layers.8.mlp.up_proj.weight": "model-00018-of-00057.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model-00019-of-00057.safetensors", + "model.layers.8.self_attn.k_proj.bias": "model-00017-of-00057.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model-00017-of-00057.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model-00017-of-00057.safetensors", + "model.layers.8.self_attn.q_proj.bias": "model-00017-of-00057.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model-00017-of-00057.safetensors", + "model.layers.8.self_attn.v_proj.bias": "model-00017-of-00057.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model-00017-of-00057.safetensors", + "model.layers.9.input_layernorm.weight": "model-00021-of-00057.safetensors", + "model.layers.9.mlp.down_proj.weight": "model-00021-of-00057.safetensors", + "model.layers.9.mlp.gate_proj.weight": "model-00020-of-00057.safetensors", + "model.layers.9.mlp.up_proj.weight": "model-00020-of-00057.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model-00021-of-00057.safetensors", + "model.layers.9.self_attn.k_proj.bias": "model-00019-of-00057.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model-00019-of-00057.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model-00019-of-00057.safetensors", + "model.layers.9.self_attn.q_proj.bias": "model-00019-of-00057.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model-00019-of-00057.safetensors", + "model.layers.9.self_attn.v_proj.bias": "model-00019-of-00057.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model-00019-of-00057.safetensors", + "model.norm.weight": "model-00057-of-00057.safetensors" + } +} diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..3a78403 --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,31 @@ +{ + "additional_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "eos_token": { + "content": "<|im_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "<|endoftext|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..51ebb3b --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa +size 11421896 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..d7792fd --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,207 @@ +{ + "add_bos_token": false, + "add_prefix_space": false, + "added_tokens_decoder": { + "151643": { + "content": "<|endoftext|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151644": { + "content": "<|im_start|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151645": { + "content": "<|im_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151646": { + "content": "<|object_ref_start|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151647": { + "content": "<|object_ref_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151648": { + "content": "<|box_start|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151649": { + "content": "<|box_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151650": { + "content": "<|quad_start|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151651": { + "content": "<|quad_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151652": { + "content": "<|vision_start|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151653": { + "content": "<|vision_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151654": { + "content": "<|vision_pad|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151655": { + "content": "<|image_pad|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151656": { + "content": "<|video_pad|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + }, + "151657": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151658": { + "content": "", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151659": { + "content": "<|fim_prefix|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151660": { + "content": "<|fim_middle|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151661": { + "content": "<|fim_suffix|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151662": { + "content": "<|fim_pad|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151663": { + "content": "<|repo_name|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + }, + "151664": { + "content": "<|file_sep|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": false + } + }, + "additional_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": {}, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/vocab.json b/vocab.json new file mode 100644 index 0000000..6c49fc6 --- /dev/null +++ b/vocab.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910 +size 2776833