From 4bd118f2b32cc85fc0f8722139849c006e35eee5 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sat, 12 Sep 2026 14:50:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: HamoAI/hamo-score-0.6b Source: Original Platform --- .gitattributes | 39 +++ LICENSE | 52 ++++ README.md | 464 +++++++++++++++++++++++++++++++ chat_template.jinja | 85 ++++++ config.json | 30 ++ gguf/hamo-score-0.6b-v61.q8.gguf | 3 + gguf/hamo-score-0.6b-v7.q8.gguf | 3 + model.safetensors | 3 + model.safetensors.index.json | 318 +++++++++++++++++++++ tokenizer.json | 3 + tokenizer_config.json | 16 ++ 11 files changed, 1016 insertions(+) create mode 100644 .gitattributes create mode 100644 LICENSE create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 gguf/hamo-score-0.6b-v61.q8.gguf create mode 100644 gguf/hamo-score-0.6b-v7.q8.gguf create mode 100644 model.safetensors create mode 100644 model.safetensors.index.json create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..04d6ee4 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,39 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +gguf/hamo-score-0.6b-v4.q8.gguf filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +gguf/hamo-score-0.6b-v61.q8.gguf filter=lfs diff=lfs merge=lfs -text +gguf/hamo-score-0.6b-v7.q8.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..db9015e --- /dev/null +++ b/LICENSE @@ -0,0 +1,52 @@ +HAMO-RAIL-S License, Version 1.0 (2026-08-05) + +Copyright (c) 2026 Hamo AI (Chris Cheng) + +This is a Responsible AI License ("RAIL") for the model weights known as +"hamo-score-0.6b" (the "Model"). It grants broad, free rights of use, +modification, and redistribution, subject to the Use Restrictions below. +It is intentionally NOT an OSI-approved open-source license: the +restrictions are a deliberate trade-off for a model that reads +psychological state signals. + +1. GRANT. Subject to Section 3, you are granted a perpetual, worldwide, + non-exclusive, royalty-free license to use, reproduce, modify, create + derivative works of, and redistribute the Model, for commercial and + non-commercial purposes. + +2. ATTRIBUTION. Redistributions of the Model or derivatives must retain + this LICENSE file and a reference to the source repository. The Model + is derived from Qwen3-0.6B (Apache License 2.0, Copyright Alibaba + Cloud); that license and its notices continue to apply to the base + weights. + +3. USE RESTRICTIONS. You may NOT use the Model or its derivatives: + a) as a standalone basis for clinical diagnosis, treatment decisions, + or any healthcare determination, without review by a licensed + professional who retains decision authority; + b) as the sole or primary basis for consequential decisions about an + identifiable person — including employment, insurance, credit, + education, immigration, or law-enforcement screening — or for + covert monitoring or surveillance of a person's psychological + state; + c) in consumer-facing mental-wellness deployments UNLESS crisis and + self-harm content is handled by an independent mechanism upstream + of the Model (the Model is not a crisis detector and must never be + the safety net), and the deployment discloses that an AI system is + in use; + d) to attempt to re-identify individuals from scores, or to link + scores to identities beyond what your lawful, consented purpose + requires. + These restrictions must be passed on, in substance, to any recipient + of the Model or of derivative weights. + +4. NO WARRANTY; LIMITATION OF LIABILITY. THE MODEL IS PROVIDED "AS IS", + WITHOUT WARRANTY OF ANY KIND. SCORES ARE PROBABILISTIC SIGNALS, NOT + FACTS ABOUT A PERSON. IN NO EVENT SHALL THE COPYRIGHT HOLDERS BE + LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY ARISING FROM THE USE + OF THE MODEL. This license does not reduce your own obligations under + applicable law (including privacy, consumer-protection, and + professional-practice regulation in your jurisdiction). + +5. TERMINATION. Your rights under this license terminate automatically + upon material breach of Section 3. diff --git a/README.md b/README.md new file mode 100644 index 0000000..5726981 --- /dev/null +++ b/README.md @@ -0,0 +1,464 @@ +--- +license: other +license_name: hamo-rail-s-1.0 +license_link: LICENSE +language: + - zh + - en +base_model: Qwen/Qwen3-0.6B +pipeline_tag: text-generation +tags: + - mental-wellness + - affect-scoring + - psychology + - knowledge-distillation + - qwen3 + - gguf + - mlx + - hamo +--- + +# hamo-score-0.6b — the little model that takes your pulse + +*给每句话把脉的小模型(中文版说明见下半部分)* + +> ### 👉 Start with the toolkit, not the weights +> `pip install hamo-score` → [**hamo-score-toolkit**](https://github.com/HamoAI/hamo-score-toolkit) (Apache-2.0) +> is the other half of this model: the exact prompt format, the crisis gate this model +> *requires upstream of it*, the smoothing math its scores are designed to feed, a +> one-command Docker server, and a 195-question self-check exam. Weights alone invite the +> one deployment shape this model was designed against. +> **Disagree with a score?** That is the single most useful thing you can send us — +> [open a disagreement report](https://github.com/HamoAI/hamo-score-toolkit/issues/new?template=score_disagreement.md); +> they feed the human gold-label program that steers future versions. + +**hamo-score-0.6b** reads one message from a mental-wellness conversation and scores five +psychological pulse signals. It never writes replies. It is the first production-distilled +component of [Hamo AI](https://www.hamo.ai)'s closed-loop wellness engine, released so that +practitioner-supervised tools can run state scoring **locally** — no API, no data leaving the room. + +> ⚠️ **What this model is NOT.** It is not a chatbot, not a diagnostic instrument, and +> **not a crisis detector**. In Hamo's own production system, crisis and self-harm content is +> short-circuited by an independent deterministic mechanism *upstream* of this model — it never +> reaches the scorer. Any deployment must reproduce that pattern (see LICENSE §3c). + +## The five pulses (AWEHB) + +Each user message gets five scores on a 0.0–3.0 scale (0.5 grid): + +| Dim | Name | Plain reading | +|---|---|---| +| **A** | Agency | Is the person doing something for themselves? (incl. small plans, coping statements) | +| **W** | Withdrawal | Giving up, avoiding, disengaging? | +| **E** | Extremity | Catastrophizing chains, all-or-nothing thinking? (bounded realistic worry stays LOW) | +| **H** | Hostility | Attacking someone? (venting frustration without a target is NOT hostility) | +| **B** | Boundary | Can they speak from an "I" position — needs, limits, clear stance? | + +**A note on B.** Its theoretical root is *differentiation of self* (family-systems sense: +a bounded two-person relationship vs. an enmeshed, undifferentiated one). A per-message scorer +cannot see the relationship — it sees language. So B measures the **linguistic footprint** of +boundaries: "I need… / I'm not willing… / this is my limit" scores high; panicked venting +(self dissolved in affect) scores low; insults are H, not B. B is a per-message signal, +not a relationship diagnosis. + +The scores are designed to feed **deterministic downstream code** (stress update, state +buckets, action gating) — in Hamo, an exponential blend `0.8 × history + 0.2 × message` +smooths per-message noise 5× before any decision is taken. We recommend the same pattern. + +## Quickstart + +> 🚀 **What the toolkit gives you**, in detail: +> - **Library** — prompt format, parsing, the smoothing math, and the license-required +> crisis gate in `pip install` + a few lines of code; +> - **Reference server** — `docker compose up` fetches the GGUF, warms the model, and +> exposes the full gate → score → smooth → bucket pipeline as `POST /score`; +> - **Self-check exam** — 195 synthetic teacher-labeled questions + 10 handwritten gate +> cases, with an official reference band (JSON 100% · dimension-level 84.0% · gate 10/10) +> so you can verify your wiring reproduces the official numbers; +> - **Fine-tuning guide** — [`docs/finetune.md`](https://github.com/HamoAI/hamo-score-toolkit/blob/main/docs/finetune.md), +> the six-generation playbook (including the two rejected generations and why) for +> adapting the scorer to your own population with your own consented data. +> +> Release notes: [EN](https://www.hamoai.tech/blog/open-sourcing-hamo-score-toolkit/) · +> [中文](https://www.hamo.ai/blog/open-sourcing-hamo-score-toolkit/). + +The model was trained on **exactly one prompt format** (its rubric is baked into the weights — +do not add scoring instructions): + +``` +给来访者最新消息打分(AWEHB,0.0-3.0)。 +此前对话: +user: +assistant: +最新消息: +``` + +The context block (`此前对话:`) is optional; up to 5 turns are accepted, and the official +toolkit trims to the production-validated guard — last 3 turns × 200 chars, message capped +at 500 chars. Apply the Qwen3 +chat template with thinking disabled, temperature 0. Output is a single JSON object. + +**transformers** + +```python +from transformers import AutoModelForCausalLM, AutoTokenizer +import torch, json, re + +m = AutoModelForCausalLM.from_pretrained("HamoAI/hamo-score-0.6b", torch_dtype=torch.bfloat16) +tok = AutoTokenizer.from_pretrained("HamoAI/hamo-score-0.6b") + +prompt = "给来访者最新消息打分(AWEHB,0.0-3.0)。\n此前对话:\nassistant: 这周过得怎么样?\n最新消息: 今天试着出门散了个步" +text = tok.apply_chat_template([{"role": "user", "content": prompt}], + add_generation_prompt=True, tokenize=False, enable_thinking=False) +out = m.generate(**tok(text, return_tensors="pt"), max_new_tokens=80, do_sample=False) +print(re.search(r"\{[^{}]*\}", tok.decode(out[0])).group()) +# {"A": 1.5, "W": 0.0, "E": 0.0, "H": 0.0, "B": 1.0} +``` + +**ollama / llama.cpp** — a ready `q8_0` GGUF is in [`gguf/`](https://huggingface.co/HamoAI/hamo-score-0.6b/tree/main/gguf). Modelfile: + +``` +FROM ./hamo-score-0.6b-v7.q8.gguf +TEMPLATE """<|im_start|>user +{{ .Prompt }}<|im_end|> +<|im_start|>assistant + + + + +""" +PARAMETER temperature 0 +PARAMETER num_predict 80 +PARAMETER stop <|im_end|> +PARAMETER repeat_penalty 1.0 +PARAMETER top_k 0 +PARAMETER top_p 1.0 +``` + +> ⚠️ **The last three parameters are not optional.** ollama defaults to +> `repeat_penalty 1.1`. This model's output — `{"A": 0.0, "W": 0.0, "E": 0.0, +> "H": 0.0, "B": 0.0}` — is deliberately repetitive, so penalising repeated +> tokens pushes every score *away from zero*, fabricating signal that isn't +> there. Measured on our 300-question boundary-discrimination exam, same +> weights, sampling as the only variable: fabrication rate **13.5% → 25.0%**, +> boundary sign-flips (a true `0.0` scored `≥2.0`) **6 → 10**. Earlier +> revisions of this card omitted them; if you deployed from those +> instructions, add them and re-create the model. + +Parse the first `{...}` in the response (the model may emit an empty `` block first). + +## Evaluation + +Held-out exam: **758 real, pseudonymised production turns** (labels = the production-scale +LLM scorer this model replaces; the exam turns are **never trained on** — the only real data +in training is the separately disclosed 440 consented staff turns, see "How it was trained"). + +| Metric | Better is | **v7 (this release)** | v6.1 (previous) | Teacher (DeepSeek, qualification paper) | Reference scorer self-consistency* | +|---|---|---|---|---|---| +| Dimension-level, within ±0.5 | ↑ higher | **85.1%** (A84 / W87 / E87 / H94 / B74) | 85.6% | 88.7% | 94–98% | +| Decision-level (state bucket after deterministic stress calc) | ↑ higher | **97.1%** | 96.2% | 97.5% | — | +| Crisis-phrase W-recall misses (count, not %) | ↓ **lower** | **3** | 4 | — | — | +| JSON validity | ↑ higher | ~100% | ~100% | — | — | + +**Read the two headline rows together.** v7 trades ~0.5 pt of raw agreement with the +reference scorer's labels for +0.9 pt on the decision the product actually consumes, and one +fewer crisis miss. The dimension-level dip is expected rather than a regression: v7 was +trained on a re-labelled corpus (the teacher re-scored every row at temperature 0, removing +label noise), so the student is now more faithful to the *denoised* labels and correspondingly +slightly less aligned with the noisier reference it was originally distilled from. Measured +directly: agreement with the temperature-0 teacher on boundary scores rose **57.3% → 69.3%** +while agreement with the original reference scorer on the same dimension stayed flat. + +Evaluated on an untouched 453-turn final split of the real-conversation exam (never trained +on, never used for checkpoint selection; gold labels include 8 human corrections). + +**Residual textual overlap, measured rather than assumed.** The exam and training splits are +disjoint by sample and by conversation, but the consented staff contributors repeat themselves +across sessions: 42 of the 453 final-exam turns (9.3%) carry a message text that also occurs +somewhere in the 440 consented training turns, 11 of them (2.4%) with the same recent context. +We scored those rows separately. For v7: on the 42 overlapping turns the model reaches 88.1% +dimension-level and 100% decision-level; on the 411 with no textual overlap, **84.8% and +96.8%**. So the headline figures carry roughly **+0.3 points** of optimism from this +effect — small, and stated here rather than left for someone else to find. (v6.1 measured the +same way: 89.0%/100% on the overlap, 85.2%/95.9% clean.) (Only 5 of the 42 +also share the training row's label vector: the same sentence usually earns different scores in +a different turn, so these are not free points, merely easier ones.) + +### Boundary discrimination — what v7 was actually built to fix + +Agreement metrics hide the failure that mattered most. On a 300-question exam built +specifically to probe the B (Boundary) decision surface — 60 matched pairs plus 180 singletons +across nine trap cells — v6.1 was reading **self-erasure as strong boundary**: "行,我全听你的, +你说哪天去就哪天去" ("fine, I'll do whatever you say, you decide") scored B=2.5 where the truth +is 0.0. That is a sign error on B's negative pole, not a calibration wobble, and B carries the +largest single weight in the downstream stress formula. + +| 300-question boundary exam | Better is | v6.1 | **v7** | +|---|---|---|---| +| Fabrication — a true 0 scored ≥1.0 (sees boundary that isn't there) | ↓ **lower** | 13.5% | **2.9%** | +| Miss — a true high scored ≤0.5 (misses a real boundary) | ↓ **lower** | 16.1% | **12.8%** | +| **Sign flips** — a true 0.0 scored ≥2.0 (count; reads self-erasure as strong boundary) | ↓ **lower** | 6 | **0** | +| Paired direction accuracy — ranks the higher-boundary arm above the lower | ↑ higher | 81.7% | **98.3%** | + +Three of the four rows are error rates, so **lower is better on the first three and higher on +the last**; v7 improves on all four. + +Both columns measured on the shipped q8 GGUF with neutral sampling, so they describe the +artifact you download rather than an internal checkpoint. Direction accuracy is the cleanest +of the four — it asks only whether the model ranks the higher-boundary arm of a matched pair +above the lower one, so it is immune to absolute calibration. + +\* Self-consistency = the same messages scored twice by the reference scorer in two live +environments; its own agreement is only 94–98% at dimension level — the practical ceiling. + +Latency (single message, warm): ~0.8 s on Apple M1 Pro (MLX bf16); 1.5–2.9 s on a +2-vCPU ARM server (q8 GGUF, CPU-only). Crisis-phrase W-recall improved 2× in the v4 +generation and edged further down in v6.1 (final-exam misses 11 → 5 → 4) — but see the +crisis disclaimer above: recall here is defense-in-depth, not the defense. + +## Community quantizations — and what we measured on them + +[**mradermacher/hamo-score-0.6b-GGUF**](https://huggingface.co/mradermacher/hamo-score-0.6b-GGUF) +provides static GGUF quants of this model from Q2_K to f16 — twelve build targets we never +shipped ourselves. Thanks to mradermacher for the work, and for carrying the RAIL-S license +terms through redistribution. + +Our published metrics were measured on **Q8_0**. Because this model's read-outs gate how deep +a conversation may go, quantization damage here is a clinical question rather than a +perplexity number — so we ran two community builds through the same 453-turn final exam, same +prompt, same parser: + +| | Q8_0 (ours, published) | Q6_K (community) | Q4_K_M (community) | +|---|---|---|---| +| Dimension-level ±0.5 | 85.5% | 84.4% | 84.2% | +| Decision-level (state bucket) | 96.2% | 95.8% | 95.6% | +| Crisis W-misses (gold ≥2.5 → pred <0.5, n=37) | 4 | 3 | **5** | +| Mean W on those 37 crisis-adjacent turns (gold 2.84) | 2.39 | 2.26 | **2.01** | +| JSON validity | 100% | 100% | 100% | +| File size | 0.64 GB | 0.50 GB | 0.40 GB | +| P50 latency (M-series, Metal) | 0.50 s | 0.44 s | 0.42 s | + +Head to head against Q8_0, both builds land in the same state bucket ~98% of the time +(mean |Δstress| 0.09). The headline numbers are nearly indistinguishable — which is exactly +why we looked underneath them. + +**What the headline numbers hide: low-bit builds attenuate, and Q4_K_M attenuates most where +it matters least forgivingly.** Every dimension drifts downward relative to Q8_0, and the +drift concentrates on Agency (mean −0.19 for Q4_K_M, −0.11 for Q6_K). On the 37 +crisis-adjacent turns of the exam — gold W ≥ 2.5 — **Q4_K_M scores lower than Q8_0 on 20 of +them and higher on exactly 1**, pulling that subset's mean withdrawal signal from 2.39 down to +2.01 and costing one additional missed crisis signal. Q6_K's withdrawal signal survives +essentially intact across the full exam (mean shift +0.01 vs Q4_K_M's −0.04). Note that +bucket agreement moves only 0.4–0.6 points across all three builds: **the state buckets are +coarse enough to absorb a damped signal, so bucket agreement alone would never have surfaced +this.** (On crisis misses specifically, 3 vs 4 out of 37 is within noise — we read Q6_K as +matching Q8_0 there, not beating it.) + +**What we recommend.** + +- **Q8_0** — the reference build. Use it when the read-outs gate behaviour and you have the 0.64 GB. +- **Q6_K** — the lowest build we would validate for gating use. It costs ~1 point of + dimension-level agreement and preserves the withdrawal signal; it saves 22% of the size. +- **Q4_K_M** — fine for research, offline analysis, and any use where a human reads the scores + rather than a system acting on them. If memory forces it into a gating deployment, lower + your withdrawal thresholds to compensate for the documented damping, and keep deterministic + crisis detection upstream where it belongs (LICENSE §3c requires that pattern at any + quantization). + +The other nine builds remain unvalidated by us; bit-widths below Q4_K_M should be assumed +worse until measured. The evaluation harness used here is in the +[toolkit](https://github.com/HamoAI/hamo-score-toolkit) — if you validate a build we haven't, +we would be glad to link your numbers. + +## How it was trained + +A three-stage distillation chain — the full story is in the companion write-up +[*Distilling hamo-score-0.6b: A Plateau, Three Bugs, and Why Data Beat Model Size*](https://www.hamoai.tech/blog/distilling-hamo-score-0-6b/): + +1. **Exam by the incumbent**: 1,198 pseudonymised production turns with reference scores, + split by session hash — 440 calibration turns and a 758-turn evaluation set, the latter + splitting again into 305 selection and the **453-turn final** that grades this release. + The teacher was qualified on the calibration turns; from v6.1 those same 440 turns, all + consented staff data, also enter training. Elsewhere in this card **440 always means that + consented set** — the qualification paper is described by its role, not its size, so the + two uses cannot be mistaken for unrelated numbers that happen to coincide. + + *On "pseudonymised" rather than "anonymised" or "de-identified" — the weaker word is the + honest one.* The salt is a fixed hard-coded string, so anyone holding the script can + recompute the mapping; full timestamps are kept; message text is preserved verbatim; and + the redaction patterns cover mainland-China formats only — Hong Kong 8-digit numbers, + North American +1 numbers, personal names, WeChat IDs and street addresses were all + measured passing through. **This data therefore remains personal data, and a deletion + request still reaches it.** A separate and non-substituting fact: the payload that reaches + the weights is only the prompt plus five scores, carrying no identifier and no timestamp. + Both statements are true; neither one covers for the other. +2. **Affordable teacher**: `deepseek-chat` running the exact production rubric, qualified at + **88.7% dimension-level / 97.5% decision-level** agreement before being allowed to label + anything. +3. **Synthetic textbook**: ~20,000 admitted dialogue windows across 6 data generations + (40+ scenario cells with per-cell label-band admission gates, style quotas for short/ + fragmented/code-switched messages, crisis and boundary contrast pairs). + **No external-client message has ever entered training — by construction.** Starting + with v6.1, the corpus additionally includes 440 real conversation turns contributed by + three company-internal staff members (the founder and two staff counselors), with their + explicit consent, upsampled ×3 (~8% of the corpus). +4. **Student**: Qwen3-0.6B, LoRA on a single MacBook (MLX; prompt-masked loss, cosine decay, + grad-checkpointing). Total API cost of the whole project: ~US$7. + +Key lessons the hard way (kept as disciplines): gradient-mask the prompt (72% of gradient was +being wasted); halve batch size when doubling sequence length (a silent fp16 explosion taught us); +verify every deploy down to a landed row. + +## Limitations & known residuals + +- **Chinese-primary** (zh 60% / mixed 22% / en 18% in training); English works but is less tested. +- Message-level footprint, not a person-level or relationship-level assessment. +- Mid-band calibration is coarse (0.5 grid; mid-band usage 7.4% vs reference 21–32%). +- Known residuals: a small set of highly implicit severe-distress phrasings remains hard + (shared across all versions and the reference scorer); occasional over-scoring of bounded + multi-step worries on E. The conversational-action gap on A was substantially closed in v6.1 + by real-conversation training data (A 81% → 85%). +- Trained against one specific rubric; scores are **relative to that rubric**, not universal + psychological ground truth. + +## Versions + +| Version | Change | Dim-level | Decision-level | +|---|---|---|---| +| v2 | first distillation (7.5k synthetic) | 81% | 95.4% | +| v3.x | rebalance + defect repair | 81% | 96.8% | +| v4 | 8-agent data audit → 15k corpus, masked loss | 84% | 96.3% | +| v5 | synthetic patch cells — **rejected** (crisis-recall regression; kept as a negative result) | — | — | +| v6 | + real turns with incumbent labels — **rejected** (3 crisis-artifact rows rode into training, crisis misses 5 → 9; kept as a negative result) | — | — | +| v6.1 | + 440 consented internal-staff turns (teacher labels) | 85.6% | 96.2% | +| **v7 (this release)** | corpus re-labelled at temperature 0 (label denoising) + 2,713-row boundary-discrimination patch | 85.1% | **97.1%** | + +## License + +**HAMO-RAIL-S 1.0** (see [LICENSE](https://huggingface.co/HamoAI/hamo-score-0.6b/blob/main/LICENSE)): free commercial and non-commercial use, +modification and redistribution, with four use restrictions — no standalone clinical +determinations, no consequential decisions about individuals (employment / insurance / +surveillance screening), consumer mental-wellness deployments must keep independent upstream +crisis handling + AI disclosure, no re-identification. Base model Qwen3-0.6B remains Apache-2.0. + +--- + +# 中文说明 + +> ### 👉 请从工具包开始,而不是从权重开始 +> `pip install hamo-score` → [**hamo-score-toolkit**](https://github.com/HamoAI/hamo-score-toolkit)(Apache-2.0) +> 是这个模型的另一半:唯一正确的提示词格式、**必须置于模型上游的危机闸门**、分数该喂进去的 +> 平滑折算、一条命令起的 Docker 服务器,以及 195 题自检考卷。只拿权重,恰恰会走成这个模型 +> 设计上要防住的那种部署。 +> **对某个评分不服?** 那是你能给我们的最有价值的东西—— +> [提一条分歧报告](https://github.com/HamoAI/hamo-score-toolkit/issues/new?template=score_disagreement.md), +> 它会直接进入引导后续版本的人类金标计划。 + +**hamo-score-0.6b** 是 Hamo AI 闭环疗愈引擎里第一个蒸馏进生产的组件:给心理支持对话中 +来访者的每一句话「把脉」,输出五路 0–3 分的脉象(A 行动力 / W 退缩 / E 极端化 / H 敌意 / +B 边界感)。**它从不写回复,也不是危机检测器**——在 Hamo 生产系统里,危机内容在更上游被 +独立的确定性机制短路,永远到不了把脉师面前;任何部署都必须复刻这个模式(见 LICENSE §3c)。 + +**关于 B(边界感)**:它的理论本源是家庭治疗中的「自我分化」——是「我是我、你是你」的二元 +关系,还是彼此淹没的混沌一元。逐句评分器看不见关系,只看得见语言,所以 B 测的是边界感的 +**语言足迹**:「我需要…」「这是我的底线」得高分;惊慌的倾泻(自我淹没在情绪里)得低分; +骂人算 H 不算 B。B 是逐句信号,不是关系诊断。 + +**成绩单(v7,本次发布)**:真实假名化生产对话终评(453 条未动用终评切分,金标含 8 处人工 +修正;评分真值来自被替换的大模型评分器;考卷数据从未参与训练):维度级 ±0.5 一致率 **85.1%**、 +决策级(经确定性压力折算后的状态桶判定)**97.1%**、危机语漏检 **3 条**(v6.1 对应为 85.6% / 96.2% / 4 条)。 +其中前两项是一致率、**越高越好**;「危机语漏检」是条数、**越低越好**——v7 由 4 条降到 3 条。 +维度级那 0.5 个点的回落不是退步:v7 的语料由教师在温度 0 下全量重标(去标签噪声),学生因此 +更忠于**去噪后**的标签,对当初那个含噪参照的一致率自然略降——实测其与温度 0 教师在 B 维的一致率 +由 57.3% 升到 69.3%,而对原参照的 B 一致率纹丝不动。 + +**v7 真正修好的是边界判别**:在一份专为探测 B 决策面而造的 300 题考卷上(60 组配对 + 180 条单题, +覆盖九类陷阱格子),v6.1 会把**自我消融**读成强边界——「行,我全听你的,你说哪天去就哪天去」 +真值 B=0.0,它给 2.5。那是 B 负极上的符号错误,而 B 在下游压力公式里权重最大。 + +| 300 题边界判别卷 | 越好方向 | v6.1 | **v7** | +|---|---|---|---| +| 造分——真值 0 却给 ≥1.0(看见并不存在的边界) | ↓ **越低越好** | 13.5% | **2.9%** | +| 漏判——真值高却给 ≤0.5(漏掉真实的边界) | ↓ **越低越好** | 16.1% | **12.8%** | +| **符号翻转**——真值 0.0 却给 ≥2.0(条数;把自我消融读成强边界) | ↓ **越低越好** | 6 条 | **0 条** | +| 配对方向正确率——高边界那一臂是否排在低的之上 | ↑ 越高越好 | 81.7% | **98.3%** | + +前三行是错误率、第四行是正确率,所以**前三行越低越好、最后一行越高越好**;v7 四项全部改善。 + +两列均测于随包发布的 q8 GGUF + 中性采样,描述的是你下载到的产物本身。 + +参照系——同一批消息让原评分器自己打两遍,维度级自洽也只有 94–98%。单条延迟:M1 Pro 约 +0.8 秒;2 vCPU ARM 服务器(纯 CPU,q8 GGUF)1.5–2.9 秒。 + +**残余重叠:我们量了,没有假设掉。** 考卷与训练集按样本、按会话完全不相交,但授权供数的内部 +员工会在不同会话里重复说同样的话:终评 453 题中有 42 题(9.3%)的正文在那 440 条授权训练数据 +里出现过,其中 11 题(2.4%)连最近上下文也相同。分开判卷的结果——这 42 题维度级 89.0%、 +决策级 100%;其余 411 题无任何正文重叠,**维度级 85.2%、决策级 95.9%**。也就是说,上面两个 +成绩含约 **+0.4 / +0.3 个百分点**的乐观。幅度不大,但我们选择自己写出来。(42 题里只有 5 题 +连标签也相同——同一句话换个轮次通常拿到不同分数,所以它们不是白送的分,只是更容易的分。) + +**训练方式**:三级师徒链——生产历史评分出考卷(1,198 条假名化真题——用「假名化」而非「匿名化」 +是因为弱的那个词才是诚实的:盐是硬编码固定字符串、完整时间戳保留、正文逐字保留,且脱敏正则只覆盖 +大陆格式,香港 8 位号码、北美 +1 号码、人名、微信号与住址实测全部穿过,**故这批数据仍属个人信息, +删除权仍及于它**;另一件必须单独陈述、不可用来顶替上一条的事实是:进入权重的载荷只有提示词与五个 +分数,不含任何标识符与时间戳。按会话哈希切分为 440 条 +校准集与 758 条评测集,后者再切成 305 条选型集与判定本次成绩的 453 条终评集)→ DeepSeek 在 +校准集上通过资格考(维度级 88.7% / 决策级 97.5%)后当教师;本卡中「440」始终指那批经授权的 +内部员工轮次,资格考卷按用途称呼、不按题量称呼,以免两处用法被误读成两个撞车的数字 → 约 2 万段合成对话当教材(40+ 场景格子、逐格标签准入闸门、短句/ +碎片/中英混杂风格配额)→ Qwen3-0.6B 学生在一台 MacBook 上 LoRA 学成。全项目 API 成本 +约 7 美元。**训练语料从不包含任何外部来访者消息(构造上保证);自 v6.1 起额外加入 440 条公司内部员工(创始人与两位咨询师)明示授权的真实对话轮次(×3 上采样,约占语料 8%)。** + +**社区量化档位(我们实测过其中一档)**:社区志愿者 mradermacher 制作了 +[**Q2_K→f16 共 12 个静态 GGUF 量化档**](https://huggingface.co/mradermacher/hamo-score-0.6b-GGUF) +——感谢他的工作,也感谢他在再分发中完整保留了 RAIL-S 许可条款。我们公布的指标测于 **Q8_0**; +由于这个模型的读数要门控对话能走多深,低比特量化掉了多少不是困惑度数字而是临床问题,所以我们 +用同一套 453 题终评、同一段提示词、同一个解析器,实测了社区的 **Q6_K** 与 **Q4_K_M**: + +| | Q8_0(我们的基线) | Q6_K(社区) | Q4_K_M(社区) | +|---|---|---|---| +| 维度级 ±0.5 一致率 | 85.5% | 84.4% | 84.2% | +| 决策级(状态桶) | 96.2% | 95.8% | 95.6% | +| 危机 W 漏检(金标 ≥2.5 → 预测 <0.5,n=37) | 4 | 3 | **5** | +| 那 37 条危机相邻样本的 W 均值(金标 2.84) | 2.39 | 2.26 | **2.01** | +| JSON 合法率 | 100% | 100% | 100% | +| 体积 | 0.64 GB | 0.50 GB | 0.40 GB | +| P50 延迟(M 系列,Metal) | 0.50 秒 | 0.44 秒 | 0.42 秒 | + +与 Q8_0 直接对比,两个社区档位落在同一状态桶的比例都约 98%(压力更新偏差均值 0.09)。总分 +几乎分不出高下——正因如此,我们才往下面看了一层。 + +**总分掩盖了什么:低比特档位会「衰减」信号,而 Q4_K_M 恰恰在最不容有失的地方衰减最厉害。** +所有维度相对 Q8_0 系统性下移,且集中在行动力上(Q4_K_M 均值 −0.19,Q6_K −0.11)。在终评集 +的 37 条危机相邻样本(金标 W≥2.5)上,**Q4_K_M 有 20 条打得比 Q8_0 更低、仅 1 条更高**,把 +该子集的退缩信号均值从 2.39 拉到 2.01,并因此多漏检一条危机信号;而 Q6_K 的退缩信号在全卷上 +基本无损(均值偏移 +0.01,Q4_K_M 为 −0.04)。请注意三个档位的状态桶一致率只相差 0.4–0.6 个 +百分点:**桶的边界粗到足以吸收一个被压扁的信号,所以只看桶一致率永远发现不了这件事。** +(就危机漏检本身而言,37 条里的 3 与 4 属于噪声范围——我们把 Q6_K 读作「与 Q8_0 持平」, +而不是「优于」。) + +**我们的建议**: + +- **Q8_0** —— 参考档。读数用于门控行为、且你付得起 0.64 GB 时,用它。 +- **Q6_K** —— 我们愿意为门控用途背书的最低档。代价是约 1 个百分点的维度级一致率,退缩信号 + 保持完好,体积省 22%。 +- **Q4_K_M** —— 适合研究、离线分析,以及分数由人来读而不是由系统据以行动的场景。若内存迫使 + 它进入门控部署,请相应下调退缩维度的阈值以补偿上述衰减,并把确定性危机检测保持在上游 + (无论用哪个量化档,LICENSE §3c 都要求这个模式)。 + +其余九个档位我们未做验证,比 Q4_K_M 更低的比特应默认更差,直到有人量过。评测脚本在 +[工具包](https://github.com/HamoAI/hamo-score-toolkit)里——如果你验证了我们没验证过的档位, +我们很乐意把你的数据链上来。 + +**官方工具包(已开源到 GitHub)**:[hamo-score-toolkit](https://github.com/HamoAI/hamo-score-toolkit) +(Apache-2.0)是这个模型的「另一半」——一条 `pip install` 装上唯一正确的提示词格式、容错 +解析、参考版压力折算与许可证要求的危机闸门;一条 `docker compose up` 跑起参考服务器 +(`POST /score` 走完整的 闸门→评分→平滑→状态桶 管线);一份 195 题合成自检考卷 + 10 条 +手写闸门用例,对照官方参考带(JSON 100%、维度级 84.0%、闸门 10/10)验证你的部署接线; +还有一份微调指南([docs/finetune.md](https://github.com/HamoAI/hamo-score-toolkit/blob/main/docs/finetune.md), +六代打法,含两代拒收的完整原因)。发布文:《[开源 hamo-score-toolkit:把模型的另一半也交出去](https://www.hamo.ai/blog/open-sourcing-hamo-score-toolkit/)》。 + +**许可证**:HAMO-RAIL-S 1.0——自由商用与修改,但有四条使用限制:不得独立做临床判定、 +不得用于对个人的重大决定(雇佣/保险/监控筛查)、面向消费者的心理健康部署必须保留独立的 +上游危机处理与 AI 身份披露、不得试图重识别个人。 + +配套阅读(背景与方法论):《[hamo-score-0.6b 是怎么蒸出来的:一段平台期、三个坑,以及数据为什么赢了参数量](https://www.hamo.ai/blog/distilling-hamo-score-0-6b/)》。 diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..699ff8d --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,85 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" and not(message.content.startswith('') and message.content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} +{%- endfor %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set content = message.content %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is defined and message.reasoning_content is not none %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in message.content %} + {%- set content = message.content.split('')[-1].lstrip('\n') %} + {%- set reasoning_content = message.content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if loop.index0 > ns.last_query_index %} + {%- if loop.last or (not loop.last and reasoning_content) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content.strip('\n') + '\n\n\n' + content.lstrip('\n') }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..ee621e8 --- /dev/null +++ b/config.json @@ -0,0 +1,30 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 1024, + "initializer_range": 0.02, + "intermediate_size": 3072, + "max_position_embeddings": 40960, + "max_window_layers": 28, + "model_type": "qwen3", + "num_attention_heads": 16, + "num_hidden_layers": 28, + "num_key_value_heads": 8, + "rms_norm_eps": 1e-06, + "rope_scaling": null, + "rope_theta": 1000000, + "sliding_window": null, + "tie_word_embeddings": true, + "torch_dtype": "bfloat16", + "transformers_version": "4.51.0", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} \ No newline at end of file diff --git a/gguf/hamo-score-0.6b-v61.q8.gguf b/gguf/hamo-score-0.6b-v61.q8.gguf new file mode 100644 index 0000000..b261cae --- /dev/null +++ b/gguf/hamo-score-0.6b-v61.q8.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:114a0ccba6e13b35170425fbdbdad8908b741f0ea66360c422f26ee633398561 +size 639446912 diff --git a/gguf/hamo-score-0.6b-v7.q8.gguf b/gguf/hamo-score-0.6b-v7.q8.gguf new file mode 100644 index 0000000..4a57f20 --- /dev/null +++ b/gguf/hamo-score-0.6b-v7.q8.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fb1018c8a38ad0da573544b827e42f1685e051c688f8de587cbb81ae1275e3ea +size 639446912 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..6e16238 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:710672d4ffbfcdb5790b0001bee4d4e67414fa5ce9b2cd1be693ab203b90d1ad +size 1192134935 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..c8156ac --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,318 @@ +{ + "metadata": { + "total_size": 1192099840, + "total_parameters": 596049920 + }, + "weight_map": { + "model.embed_tokens.weight": "model.safetensors", + "model.layers.0.input_layernorm.weight": "model.safetensors", + "model.layers.0.mlp.down_proj.weight": "model.safetensors", + "model.layers.0.mlp.gate_proj.weight": "model.safetensors", + "model.layers.0.mlp.up_proj.weight": "model.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model.safetensors", + "model.layers.0.self_attn.k_norm.weight": "model.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model.safetensors", + "model.layers.0.self_attn.q_norm.weight": "model.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model.safetensors", + "model.layers.1.input_layernorm.weight": "model.safetensors", + "model.layers.1.mlp.down_proj.weight": "model.safetensors", + "model.layers.1.mlp.gate_proj.weight": "model.safetensors", + "model.layers.1.mlp.up_proj.weight": "model.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model.safetensors", + "model.layers.1.self_attn.k_norm.weight": "model.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model.safetensors", + "model.layers.1.self_attn.q_norm.weight": "model.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model.safetensors", + "model.layers.10.input_layernorm.weight": "model.safetensors", + "model.layers.10.mlp.down_proj.weight": "model.safetensors", + "model.layers.10.mlp.gate_proj.weight": "model.safetensors", + "model.layers.10.mlp.up_proj.weight": "model.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model.safetensors", + "model.layers.10.self_attn.k_norm.weight": "model.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model.safetensors", + "model.layers.10.self_attn.q_norm.weight": "model.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model.safetensors", + "model.layers.11.input_layernorm.weight": "model.safetensors", + "model.layers.11.mlp.down_proj.weight": "model.safetensors", + "model.layers.11.mlp.gate_proj.weight": "model.safetensors", + "model.layers.11.mlp.up_proj.weight": "model.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model.safetensors", + "model.layers.11.self_attn.k_norm.weight": "model.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model.safetensors", + "model.layers.11.self_attn.q_norm.weight": "model.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model.safetensors", + "model.layers.12.input_layernorm.weight": "model.safetensors", + "model.layers.12.mlp.down_proj.weight": "model.safetensors", + "model.layers.12.mlp.gate_proj.weight": "model.safetensors", + "model.layers.12.mlp.up_proj.weight": "model.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model.safetensors", + "model.layers.12.self_attn.k_norm.weight": "model.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model.safetensors", + "model.layers.12.self_attn.q_norm.weight": "model.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model.safetensors", + "model.layers.13.input_layernorm.weight": "model.safetensors", + "model.layers.13.mlp.down_proj.weight": "model.safetensors", + "model.layers.13.mlp.gate_proj.weight": "model.safetensors", + "model.layers.13.mlp.up_proj.weight": "model.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model.safetensors", + "model.layers.13.self_attn.k_norm.weight": "model.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model.safetensors", + "model.layers.13.self_attn.q_norm.weight": "model.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model.safetensors", + "model.layers.14.input_layernorm.weight": "model.safetensors", + "model.layers.14.mlp.down_proj.weight": "model.safetensors", + "model.layers.14.mlp.gate_proj.weight": "model.safetensors", + "model.layers.14.mlp.up_proj.weight": "model.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model.safetensors", + "model.layers.14.self_attn.k_norm.weight": "model.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model.safetensors", + "model.layers.14.self_attn.q_norm.weight": "model.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model.safetensors", + "model.layers.15.input_layernorm.weight": "model.safetensors", + "model.layers.15.mlp.down_proj.weight": "model.safetensors", + "model.layers.15.mlp.gate_proj.weight": "model.safetensors", + "model.layers.15.mlp.up_proj.weight": "model.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model.safetensors", + "model.layers.15.self_attn.k_norm.weight": "model.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model.safetensors", + "model.layers.15.self_attn.q_norm.weight": "model.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model.safetensors", + "model.layers.16.input_layernorm.weight": "model.safetensors", + "model.layers.16.mlp.down_proj.weight": "model.safetensors", + "model.layers.16.mlp.gate_proj.weight": "model.safetensors", + "model.layers.16.mlp.up_proj.weight": "model.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model.safetensors", + "model.layers.16.self_attn.k_norm.weight": "model.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model.safetensors", + "model.layers.16.self_attn.q_norm.weight": "model.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model.safetensors", + "model.layers.17.input_layernorm.weight": "model.safetensors", + "model.layers.17.mlp.down_proj.weight": "model.safetensors", + "model.layers.17.mlp.gate_proj.weight": "model.safetensors", + "model.layers.17.mlp.up_proj.weight": "model.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model.safetensors", + "model.layers.17.self_attn.k_norm.weight": "model.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model.safetensors", + "model.layers.17.self_attn.q_norm.weight": "model.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model.safetensors", + "model.layers.18.input_layernorm.weight": "model.safetensors", + "model.layers.18.mlp.down_proj.weight": "model.safetensors", + "model.layers.18.mlp.gate_proj.weight": "model.safetensors", + "model.layers.18.mlp.up_proj.weight": "model.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model.safetensors", + "model.layers.18.self_attn.k_norm.weight": "model.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model.safetensors", + "model.layers.18.self_attn.q_norm.weight": "model.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model.safetensors", + "model.layers.19.input_layernorm.weight": "model.safetensors", + "model.layers.19.mlp.down_proj.weight": "model.safetensors", + "model.layers.19.mlp.gate_proj.weight": "model.safetensors", + "model.layers.19.mlp.up_proj.weight": "model.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model.safetensors", + "model.layers.19.self_attn.k_norm.weight": "model.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model.safetensors", + "model.layers.19.self_attn.q_norm.weight": "model.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model.safetensors", + "model.layers.2.input_layernorm.weight": "model.safetensors", + "model.layers.2.mlp.down_proj.weight": "model.safetensors", + "model.layers.2.mlp.gate_proj.weight": "model.safetensors", + "model.layers.2.mlp.up_proj.weight": "model.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model.safetensors", + "model.layers.2.self_attn.k_norm.weight": "model.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model.safetensors", + "model.layers.2.self_attn.q_norm.weight": "model.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model.safetensors", + "model.layers.20.input_layernorm.weight": "model.safetensors", + "model.layers.20.mlp.down_proj.weight": "model.safetensors", + "model.layers.20.mlp.gate_proj.weight": "model.safetensors", + "model.layers.20.mlp.up_proj.weight": "model.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model.safetensors", + "model.layers.20.self_attn.k_norm.weight": "model.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model.safetensors", + "model.layers.20.self_attn.q_norm.weight": "model.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model.safetensors", + "model.layers.21.input_layernorm.weight": "model.safetensors", + "model.layers.21.mlp.down_proj.weight": "model.safetensors", + "model.layers.21.mlp.gate_proj.weight": "model.safetensors", + "model.layers.21.mlp.up_proj.weight": "model.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model.safetensors", + "model.layers.21.self_attn.k_norm.weight": "model.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model.safetensors", + "model.layers.21.self_attn.q_norm.weight": "model.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model.safetensors", + "model.layers.22.input_layernorm.weight": "model.safetensors", + "model.layers.22.mlp.down_proj.weight": "model.safetensors", + "model.layers.22.mlp.gate_proj.weight": "model.safetensors", + "model.layers.22.mlp.up_proj.weight": "model.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model.safetensors", + "model.layers.22.self_attn.k_norm.weight": "model.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model.safetensors", + "model.layers.22.self_attn.q_norm.weight": "model.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model.safetensors", + "model.layers.23.input_layernorm.weight": "model.safetensors", + "model.layers.23.mlp.down_proj.weight": "model.safetensors", + "model.layers.23.mlp.gate_proj.weight": "model.safetensors", + "model.layers.23.mlp.up_proj.weight": "model.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model.safetensors", + "model.layers.23.self_attn.k_norm.weight": "model.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model.safetensors", + "model.layers.23.self_attn.q_norm.weight": "model.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model.safetensors", + "model.layers.24.input_layernorm.weight": "model.safetensors", + "model.layers.24.mlp.down_proj.weight": "model.safetensors", + "model.layers.24.mlp.gate_proj.weight": "model.safetensors", + "model.layers.24.mlp.up_proj.weight": "model.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model.safetensors", + "model.layers.24.self_attn.k_norm.weight": "model.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model.safetensors", + "model.layers.24.self_attn.q_norm.weight": "model.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model.safetensors", + "model.layers.25.input_layernorm.weight": "model.safetensors", + "model.layers.25.mlp.down_proj.weight": "model.safetensors", + "model.layers.25.mlp.gate_proj.weight": "model.safetensors", + "model.layers.25.mlp.up_proj.weight": "model.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model.safetensors", + "model.layers.25.self_attn.k_norm.weight": "model.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model.safetensors", + "model.layers.25.self_attn.q_norm.weight": "model.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model.safetensors", + "model.layers.26.input_layernorm.weight": "model.safetensors", + "model.layers.26.mlp.down_proj.weight": "model.safetensors", + "model.layers.26.mlp.gate_proj.weight": "model.safetensors", + "model.layers.26.mlp.up_proj.weight": "model.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model.safetensors", + "model.layers.26.self_attn.k_norm.weight": "model.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model.safetensors", + "model.layers.26.self_attn.q_norm.weight": "model.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model.safetensors", + "model.layers.27.input_layernorm.weight": "model.safetensors", + "model.layers.27.mlp.down_proj.weight": "model.safetensors", + "model.layers.27.mlp.gate_proj.weight": "model.safetensors", + "model.layers.27.mlp.up_proj.weight": "model.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model.safetensors", + "model.layers.27.self_attn.k_norm.weight": "model.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model.safetensors", + "model.layers.27.self_attn.q_norm.weight": "model.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model.safetensors", + "model.layers.3.input_layernorm.weight": "model.safetensors", + "model.layers.3.mlp.down_proj.weight": "model.safetensors", + "model.layers.3.mlp.gate_proj.weight": "model.safetensors", + "model.layers.3.mlp.up_proj.weight": "model.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model.safetensors", + "model.layers.3.self_attn.k_norm.weight": "model.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model.safetensors", + "model.layers.3.self_attn.q_norm.weight": "model.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model.safetensors", + "model.layers.4.input_layernorm.weight": "model.safetensors", + "model.layers.4.mlp.down_proj.weight": "model.safetensors", + "model.layers.4.mlp.gate_proj.weight": "model.safetensors", + "model.layers.4.mlp.up_proj.weight": "model.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model.safetensors", + "model.layers.4.self_attn.k_norm.weight": "model.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model.safetensors", + "model.layers.4.self_attn.q_norm.weight": "model.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model.safetensors", + "model.layers.5.input_layernorm.weight": "model.safetensors", + "model.layers.5.mlp.down_proj.weight": "model.safetensors", + "model.layers.5.mlp.gate_proj.weight": "model.safetensors", + "model.layers.5.mlp.up_proj.weight": "model.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model.safetensors", + "model.layers.5.self_attn.k_norm.weight": "model.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model.safetensors", + "model.layers.5.self_attn.q_norm.weight": "model.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model.safetensors", + "model.layers.6.input_layernorm.weight": "model.safetensors", + "model.layers.6.mlp.down_proj.weight": "model.safetensors", + "model.layers.6.mlp.gate_proj.weight": "model.safetensors", + "model.layers.6.mlp.up_proj.weight": "model.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model.safetensors", + "model.layers.6.self_attn.k_norm.weight": "model.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model.safetensors", + "model.layers.6.self_attn.q_norm.weight": "model.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model.safetensors", + "model.layers.7.input_layernorm.weight": "model.safetensors", + "model.layers.7.mlp.down_proj.weight": "model.safetensors", + "model.layers.7.mlp.gate_proj.weight": "model.safetensors", + "model.layers.7.mlp.up_proj.weight": "model.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model.safetensors", + "model.layers.7.self_attn.k_norm.weight": "model.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model.safetensors", + "model.layers.7.self_attn.q_norm.weight": "model.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model.safetensors", + "model.layers.8.input_layernorm.weight": "model.safetensors", + "model.layers.8.mlp.down_proj.weight": "model.safetensors", + "model.layers.8.mlp.gate_proj.weight": "model.safetensors", + "model.layers.8.mlp.up_proj.weight": "model.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model.safetensors", + "model.layers.8.self_attn.k_norm.weight": "model.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model.safetensors", + "model.layers.8.self_attn.q_norm.weight": "model.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model.safetensors", + "model.layers.9.input_layernorm.weight": "model.safetensors", + "model.layers.9.mlp.down_proj.weight": "model.safetensors", + "model.layers.9.mlp.gate_proj.weight": "model.safetensors", + "model.layers.9.mlp.up_proj.weight": "model.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model.safetensors", + "model.layers.9.self_attn.k_norm.weight": "model.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model.safetensors", + "model.layers.9.self_attn.q_norm.weight": "model.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model.safetensors", + "model.norm.weight": "model.safetensors" + } +} \ No newline at end of file diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..d89bb6f --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,16 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "is_local": true, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "tool_parser_type": "json_tools", + "unk_token": null +}