初始化项目,由ModelHub XC社区提供模型
Model: HamoAI/hamo-score-0.6b Source: Original Platform
This commit is contained in:
39
.gitattributes
vendored
Normal file
39
.gitattributes
vendored
Normal file
@@ -0,0 +1,39 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
gguf/hamo-score-0.6b-v4.q8.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
gguf/hamo-score-0.6b-v61.q8.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
gguf/hamo-score-0.6b-v7.q8.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
52
LICENSE
Normal file
52
LICENSE
Normal file
@@ -0,0 +1,52 @@
|
||||
HAMO-RAIL-S License, Version 1.0 (2026-08-05)
|
||||
|
||||
Copyright (c) 2026 Hamo AI (Chris Cheng)
|
||||
|
||||
This is a Responsible AI License ("RAIL") for the model weights known as
|
||||
"hamo-score-0.6b" (the "Model"). It grants broad, free rights of use,
|
||||
modification, and redistribution, subject to the Use Restrictions below.
|
||||
It is intentionally NOT an OSI-approved open-source license: the
|
||||
restrictions are a deliberate trade-off for a model that reads
|
||||
psychological state signals.
|
||||
|
||||
1. GRANT. Subject to Section 3, you are granted a perpetual, worldwide,
|
||||
non-exclusive, royalty-free license to use, reproduce, modify, create
|
||||
derivative works of, and redistribute the Model, for commercial and
|
||||
non-commercial purposes.
|
||||
|
||||
2. ATTRIBUTION. Redistributions of the Model or derivatives must retain
|
||||
this LICENSE file and a reference to the source repository. The Model
|
||||
is derived from Qwen3-0.6B (Apache License 2.0, Copyright Alibaba
|
||||
Cloud); that license and its notices continue to apply to the base
|
||||
weights.
|
||||
|
||||
3. USE RESTRICTIONS. You may NOT use the Model or its derivatives:
|
||||
a) as a standalone basis for clinical diagnosis, treatment decisions,
|
||||
or any healthcare determination, without review by a licensed
|
||||
professional who retains decision authority;
|
||||
b) as the sole or primary basis for consequential decisions about an
|
||||
identifiable person — including employment, insurance, credit,
|
||||
education, immigration, or law-enforcement screening — or for
|
||||
covert monitoring or surveillance of a person's psychological
|
||||
state;
|
||||
c) in consumer-facing mental-wellness deployments UNLESS crisis and
|
||||
self-harm content is handled by an independent mechanism upstream
|
||||
of the Model (the Model is not a crisis detector and must never be
|
||||
the safety net), and the deployment discloses that an AI system is
|
||||
in use;
|
||||
d) to attempt to re-identify individuals from scores, or to link
|
||||
scores to identities beyond what your lawful, consented purpose
|
||||
requires.
|
||||
These restrictions must be passed on, in substance, to any recipient
|
||||
of the Model or of derivative weights.
|
||||
|
||||
4. NO WARRANTY; LIMITATION OF LIABILITY. THE MODEL IS PROVIDED "AS IS",
|
||||
WITHOUT WARRANTY OF ANY KIND. SCORES ARE PROBABILISTIC SIGNALS, NOT
|
||||
FACTS ABOUT A PERSON. IN NO EVENT SHALL THE COPYRIGHT HOLDERS BE
|
||||
LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY ARISING FROM THE USE
|
||||
OF THE MODEL. This license does not reduce your own obligations under
|
||||
applicable law (including privacy, consumer-protection, and
|
||||
professional-practice regulation in your jurisdiction).
|
||||
|
||||
5. TERMINATION. Your rights under this license terminate automatically
|
||||
upon material breach of Section 3.
|
||||
464
README.md
Normal file
464
README.md
Normal file
@@ -0,0 +1,464 @@
|
||||
---
|
||||
license: other
|
||||
license_name: hamo-rail-s-1.0
|
||||
license_link: LICENSE
|
||||
language:
|
||||
- zh
|
||||
- en
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
pipeline_tag: text-generation
|
||||
tags:
|
||||
- mental-wellness
|
||||
- affect-scoring
|
||||
- psychology
|
||||
- knowledge-distillation
|
||||
- qwen3
|
||||
- gguf
|
||||
- mlx
|
||||
- hamo
|
||||
---
|
||||
|
||||
# hamo-score-0.6b — the little model that takes your pulse
|
||||
|
||||
*给每句话把脉的小模型(中文版说明见下半部分)*
|
||||
|
||||
> ### 👉 Start with the toolkit, not the weights
|
||||
> `pip install hamo-score` → [**hamo-score-toolkit**](https://github.com/HamoAI/hamo-score-toolkit) (Apache-2.0)
|
||||
> is the other half of this model: the exact prompt format, the crisis gate this model
|
||||
> *requires upstream of it*, the smoothing math its scores are designed to feed, a
|
||||
> one-command Docker server, and a 195-question self-check exam. Weights alone invite the
|
||||
> one deployment shape this model was designed against.
|
||||
> **Disagree with a score?** That is the single most useful thing you can send us —
|
||||
> [open a disagreement report](https://github.com/HamoAI/hamo-score-toolkit/issues/new?template=score_disagreement.md);
|
||||
> they feed the human gold-label program that steers future versions.
|
||||
|
||||
**hamo-score-0.6b** reads one message from a mental-wellness conversation and scores five
|
||||
psychological pulse signals. It never writes replies. It is the first production-distilled
|
||||
component of [Hamo AI](https://www.hamo.ai)'s closed-loop wellness engine, released so that
|
||||
practitioner-supervised tools can run state scoring **locally** — no API, no data leaving the room.
|
||||
|
||||
> ⚠️ **What this model is NOT.** It is not a chatbot, not a diagnostic instrument, and
|
||||
> **not a crisis detector**. In Hamo's own production system, crisis and self-harm content is
|
||||
> short-circuited by an independent deterministic mechanism *upstream* of this model — it never
|
||||
> reaches the scorer. Any deployment must reproduce that pattern (see LICENSE §3c).
|
||||
|
||||
## The five pulses (AWEHB)
|
||||
|
||||
Each user message gets five scores on a 0.0–3.0 scale (0.5 grid):
|
||||
|
||||
| Dim | Name | Plain reading |
|
||||
|---|---|---|
|
||||
| **A** | Agency | Is the person doing something for themselves? (incl. small plans, coping statements) |
|
||||
| **W** | Withdrawal | Giving up, avoiding, disengaging? |
|
||||
| **E** | Extremity | Catastrophizing chains, all-or-nothing thinking? (bounded realistic worry stays LOW) |
|
||||
| **H** | Hostility | Attacking someone? (venting frustration without a target is NOT hostility) |
|
||||
| **B** | Boundary | Can they speak from an "I" position — needs, limits, clear stance? |
|
||||
|
||||
**A note on B.** Its theoretical root is *differentiation of self* (family-systems sense:
|
||||
a bounded two-person relationship vs. an enmeshed, undifferentiated one). A per-message scorer
|
||||
cannot see the relationship — it sees language. So B measures the **linguistic footprint** of
|
||||
boundaries: "I need… / I'm not willing… / this is my limit" scores high; panicked venting
|
||||
(self dissolved in affect) scores low; insults are H, not B. B is a per-message signal,
|
||||
not a relationship diagnosis.
|
||||
|
||||
The scores are designed to feed **deterministic downstream code** (stress update, state
|
||||
buckets, action gating) — in Hamo, an exponential blend `0.8 × history + 0.2 × message`
|
||||
smooths per-message noise 5× before any decision is taken. We recommend the same pattern.
|
||||
|
||||
## Quickstart
|
||||
|
||||
> 🚀 **What the toolkit gives you**, in detail:
|
||||
> - **Library** — prompt format, parsing, the smoothing math, and the license-required
|
||||
> crisis gate in `pip install` + a few lines of code;
|
||||
> - **Reference server** — `docker compose up` fetches the GGUF, warms the model, and
|
||||
> exposes the full gate → score → smooth → bucket pipeline as `POST /score`;
|
||||
> - **Self-check exam** — 195 synthetic teacher-labeled questions + 10 handwritten gate
|
||||
> cases, with an official reference band (JSON 100% · dimension-level 84.0% · gate 10/10)
|
||||
> so you can verify your wiring reproduces the official numbers;
|
||||
> - **Fine-tuning guide** — [`docs/finetune.md`](https://github.com/HamoAI/hamo-score-toolkit/blob/main/docs/finetune.md),
|
||||
> the six-generation playbook (including the two rejected generations and why) for
|
||||
> adapting the scorer to your own population with your own consented data.
|
||||
>
|
||||
> Release notes: [EN](https://www.hamoai.tech/blog/open-sourcing-hamo-score-toolkit/) ·
|
||||
> [中文](https://www.hamo.ai/blog/open-sourcing-hamo-score-toolkit/).
|
||||
|
||||
The model was trained on **exactly one prompt format** (its rubric is baked into the weights —
|
||||
do not add scoring instructions):
|
||||
|
||||
```
|
||||
给来访者最新消息打分(AWEHB,0.0-3.0)。
|
||||
此前对话:
|
||||
user: <turn>
|
||||
assistant: <turn>
|
||||
最新消息: <message to score>
|
||||
```
|
||||
|
||||
The context block (`此前对话:`) is optional; up to 5 turns are accepted, and the official
|
||||
toolkit trims to the production-validated guard — last 3 turns × 200 chars, message capped
|
||||
at 500 chars. Apply the Qwen3
|
||||
chat template with thinking disabled, temperature 0. Output is a single JSON object.
|
||||
|
||||
**transformers**
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
import torch, json, re
|
||||
|
||||
m = AutoModelForCausalLM.from_pretrained("HamoAI/hamo-score-0.6b", torch_dtype=torch.bfloat16)
|
||||
tok = AutoTokenizer.from_pretrained("HamoAI/hamo-score-0.6b")
|
||||
|
||||
prompt = "给来访者最新消息打分(AWEHB,0.0-3.0)。\n此前对话:\nassistant: 这周过得怎么样?\n最新消息: 今天试着出门散了个步"
|
||||
text = tok.apply_chat_template([{"role": "user", "content": prompt}],
|
||||
add_generation_prompt=True, tokenize=False, enable_thinking=False)
|
||||
out = m.generate(**tok(text, return_tensors="pt"), max_new_tokens=80, do_sample=False)
|
||||
print(re.search(r"\{[^{}]*\}", tok.decode(out[0])).group())
|
||||
# {"A": 1.5, "W": 0.0, "E": 0.0, "H": 0.0, "B": 1.0}
|
||||
```
|
||||
|
||||
**ollama / llama.cpp** — a ready `q8_0` GGUF is in [`gguf/`](https://huggingface.co/HamoAI/hamo-score-0.6b/tree/main/gguf). Modelfile:
|
||||
|
||||
```
|
||||
FROM ./hamo-score-0.6b-v7.q8.gguf
|
||||
TEMPLATE """<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
<think>
|
||||
|
||||
</think>
|
||||
|
||||
"""
|
||||
PARAMETER temperature 0
|
||||
PARAMETER num_predict 80
|
||||
PARAMETER stop <|im_end|>
|
||||
PARAMETER repeat_penalty 1.0
|
||||
PARAMETER top_k 0
|
||||
PARAMETER top_p 1.0
|
||||
```
|
||||
|
||||
> ⚠️ **The last three parameters are not optional.** ollama defaults to
|
||||
> `repeat_penalty 1.1`. This model's output — `{"A": 0.0, "W": 0.0, "E": 0.0,
|
||||
> "H": 0.0, "B": 0.0}` — is deliberately repetitive, so penalising repeated
|
||||
> tokens pushes every score *away from zero*, fabricating signal that isn't
|
||||
> there. Measured on our 300-question boundary-discrimination exam, same
|
||||
> weights, sampling as the only variable: fabrication rate **13.5% → 25.0%**,
|
||||
> boundary sign-flips (a true `0.0` scored `≥2.0`) **6 → 10**. Earlier
|
||||
> revisions of this card omitted them; if you deployed from those
|
||||
> instructions, add them and re-create the model.
|
||||
|
||||
Parse the first `{...}` in the response (the model may emit an empty `<think>` block first).
|
||||
|
||||
## Evaluation
|
||||
|
||||
Held-out exam: **758 real, pseudonymised production turns** (labels = the production-scale
|
||||
LLM scorer this model replaces; the exam turns are **never trained on** — the only real data
|
||||
in training is the separately disclosed 440 consented staff turns, see "How it was trained").
|
||||
|
||||
| Metric | Better is | **v7 (this release)** | v6.1 (previous) | Teacher (DeepSeek, qualification paper) | Reference scorer self-consistency* |
|
||||
|---|---|---|---|---|---|
|
||||
| Dimension-level, within ±0.5 | ↑ higher | **85.1%** (A84 / W87 / E87 / H94 / B74) | 85.6% | 88.7% | 94–98% |
|
||||
| Decision-level (state bucket after deterministic stress calc) | ↑ higher | **97.1%** | 96.2% | 97.5% | — |
|
||||
| Crisis-phrase W-recall misses (count, not %) | ↓ **lower** | **3** | 4 | — | — |
|
||||
| JSON validity | ↑ higher | ~100% | ~100% | — | — |
|
||||
|
||||
**Read the two headline rows together.** v7 trades ~0.5 pt of raw agreement with the
|
||||
reference scorer's labels for +0.9 pt on the decision the product actually consumes, and one
|
||||
fewer crisis miss. The dimension-level dip is expected rather than a regression: v7 was
|
||||
trained on a re-labelled corpus (the teacher re-scored every row at temperature 0, removing
|
||||
label noise), so the student is now more faithful to the *denoised* labels and correspondingly
|
||||
slightly less aligned with the noisier reference it was originally distilled from. Measured
|
||||
directly: agreement with the temperature-0 teacher on boundary scores rose **57.3% → 69.3%**
|
||||
while agreement with the original reference scorer on the same dimension stayed flat.
|
||||
|
||||
Evaluated on an untouched 453-turn final split of the real-conversation exam (never trained
|
||||
on, never used for checkpoint selection; gold labels include 8 human corrections).
|
||||
|
||||
**Residual textual overlap, measured rather than assumed.** The exam and training splits are
|
||||
disjoint by sample and by conversation, but the consented staff contributors repeat themselves
|
||||
across sessions: 42 of the 453 final-exam turns (9.3%) carry a message text that also occurs
|
||||
somewhere in the 440 consented training turns, 11 of them (2.4%) with the same recent context.
|
||||
We scored those rows separately. For v7: on the 42 overlapping turns the model reaches 88.1%
|
||||
dimension-level and 100% decision-level; on the 411 with no textual overlap, **84.8% and
|
||||
96.8%**. So the headline figures carry roughly **+0.3 points** of optimism from this
|
||||
effect — small, and stated here rather than left for someone else to find. (v6.1 measured the
|
||||
same way: 89.0%/100% on the overlap, 85.2%/95.9% clean.) (Only 5 of the 42
|
||||
also share the training row's label vector: the same sentence usually earns different scores in
|
||||
a different turn, so these are not free points, merely easier ones.)
|
||||
|
||||
### Boundary discrimination — what v7 was actually built to fix
|
||||
|
||||
Agreement metrics hide the failure that mattered most. On a 300-question exam built
|
||||
specifically to probe the B (Boundary) decision surface — 60 matched pairs plus 180 singletons
|
||||
across nine trap cells — v6.1 was reading **self-erasure as strong boundary**: "行,我全听你的,
|
||||
你说哪天去就哪天去" ("fine, I'll do whatever you say, you decide") scored B=2.5 where the truth
|
||||
is 0.0. That is a sign error on B's negative pole, not a calibration wobble, and B carries the
|
||||
largest single weight in the downstream stress formula.
|
||||
|
||||
| 300-question boundary exam | Better is | v6.1 | **v7** |
|
||||
|---|---|---|---|
|
||||
| Fabrication — a true 0 scored ≥1.0 (sees boundary that isn't there) | ↓ **lower** | 13.5% | **2.9%** |
|
||||
| Miss — a true high scored ≤0.5 (misses a real boundary) | ↓ **lower** | 16.1% | **12.8%** |
|
||||
| **Sign flips** — a true 0.0 scored ≥2.0 (count; reads self-erasure as strong boundary) | ↓ **lower** | 6 | **0** |
|
||||
| Paired direction accuracy — ranks the higher-boundary arm above the lower | ↑ higher | 81.7% | **98.3%** |
|
||||
|
||||
Three of the four rows are error rates, so **lower is better on the first three and higher on
|
||||
the last**; v7 improves on all four.
|
||||
|
||||
Both columns measured on the shipped q8 GGUF with neutral sampling, so they describe the
|
||||
artifact you download rather than an internal checkpoint. Direction accuracy is the cleanest
|
||||
of the four — it asks only whether the model ranks the higher-boundary arm of a matched pair
|
||||
above the lower one, so it is immune to absolute calibration.
|
||||
|
||||
\* Self-consistency = the same messages scored twice by the reference scorer in two live
|
||||
environments; its own agreement is only 94–98% at dimension level — the practical ceiling.
|
||||
|
||||
Latency (single message, warm): ~0.8 s on Apple M1 Pro (MLX bf16); 1.5–2.9 s on a
|
||||
2-vCPU ARM server (q8 GGUF, CPU-only). Crisis-phrase W-recall improved 2× in the v4
|
||||
generation and edged further down in v6.1 (final-exam misses 11 → 5 → 4) — but see the
|
||||
crisis disclaimer above: recall here is defense-in-depth, not the defense.
|
||||
|
||||
## Community quantizations — and what we measured on them
|
||||
|
||||
[**mradermacher/hamo-score-0.6b-GGUF**](https://huggingface.co/mradermacher/hamo-score-0.6b-GGUF)
|
||||
provides static GGUF quants of this model from Q2_K to f16 — twelve build targets we never
|
||||
shipped ourselves. Thanks to mradermacher for the work, and for carrying the RAIL-S license
|
||||
terms through redistribution.
|
||||
|
||||
Our published metrics were measured on **Q8_0**. Because this model's read-outs gate how deep
|
||||
a conversation may go, quantization damage here is a clinical question rather than a
|
||||
perplexity number — so we ran two community builds through the same 453-turn final exam, same
|
||||
prompt, same parser:
|
||||
|
||||
| | Q8_0 (ours, published) | Q6_K (community) | Q4_K_M (community) |
|
||||
|---|---|---|---|
|
||||
| Dimension-level ±0.5 | 85.5% | 84.4% | 84.2% |
|
||||
| Decision-level (state bucket) | 96.2% | 95.8% | 95.6% |
|
||||
| Crisis W-misses (gold ≥2.5 → pred <0.5, n=37) | 4 | 3 | **5** |
|
||||
| Mean W on those 37 crisis-adjacent turns (gold 2.84) | 2.39 | 2.26 | **2.01** |
|
||||
| JSON validity | 100% | 100% | 100% |
|
||||
| File size | 0.64 GB | 0.50 GB | 0.40 GB |
|
||||
| P50 latency (M-series, Metal) | 0.50 s | 0.44 s | 0.42 s |
|
||||
|
||||
Head to head against Q8_0, both builds land in the same state bucket ~98% of the time
|
||||
(mean |Δstress| 0.09). The headline numbers are nearly indistinguishable — which is exactly
|
||||
why we looked underneath them.
|
||||
|
||||
**What the headline numbers hide: low-bit builds attenuate, and Q4_K_M attenuates most where
|
||||
it matters least forgivingly.** Every dimension drifts downward relative to Q8_0, and the
|
||||
drift concentrates on Agency (mean −0.19 for Q4_K_M, −0.11 for Q6_K). On the 37
|
||||
crisis-adjacent turns of the exam — gold W ≥ 2.5 — **Q4_K_M scores lower than Q8_0 on 20 of
|
||||
them and higher on exactly 1**, pulling that subset's mean withdrawal signal from 2.39 down to
|
||||
2.01 and costing one additional missed crisis signal. Q6_K's withdrawal signal survives
|
||||
essentially intact across the full exam (mean shift +0.01 vs Q4_K_M's −0.04). Note that
|
||||
bucket agreement moves only 0.4–0.6 points across all three builds: **the state buckets are
|
||||
coarse enough to absorb a damped signal, so bucket agreement alone would never have surfaced
|
||||
this.** (On crisis misses specifically, 3 vs 4 out of 37 is within noise — we read Q6_K as
|
||||
matching Q8_0 there, not beating it.)
|
||||
|
||||
**What we recommend.**
|
||||
|
||||
- **Q8_0** — the reference build. Use it when the read-outs gate behaviour and you have the 0.64 GB.
|
||||
- **Q6_K** — the lowest build we would validate for gating use. It costs ~1 point of
|
||||
dimension-level agreement and preserves the withdrawal signal; it saves 22% of the size.
|
||||
- **Q4_K_M** — fine for research, offline analysis, and any use where a human reads the scores
|
||||
rather than a system acting on them. If memory forces it into a gating deployment, lower
|
||||
your withdrawal thresholds to compensate for the documented damping, and keep deterministic
|
||||
crisis detection upstream where it belongs (LICENSE §3c requires that pattern at any
|
||||
quantization).
|
||||
|
||||
The other nine builds remain unvalidated by us; bit-widths below Q4_K_M should be assumed
|
||||
worse until measured. The evaluation harness used here is in the
|
||||
[toolkit](https://github.com/HamoAI/hamo-score-toolkit) — if you validate a build we haven't,
|
||||
we would be glad to link your numbers.
|
||||
|
||||
## How it was trained
|
||||
|
||||
A three-stage distillation chain — the full story is in the companion write-up
|
||||
[*Distilling hamo-score-0.6b: A Plateau, Three Bugs, and Why Data Beat Model Size*](https://www.hamoai.tech/blog/distilling-hamo-score-0-6b/):
|
||||
|
||||
1. **Exam by the incumbent**: 1,198 pseudonymised production turns with reference scores,
|
||||
split by session hash — 440 calibration turns and a 758-turn evaluation set, the latter
|
||||
splitting again into 305 selection and the **453-turn final** that grades this release.
|
||||
The teacher was qualified on the calibration turns; from v6.1 those same 440 turns, all
|
||||
consented staff data, also enter training. Elsewhere in this card **440 always means that
|
||||
consented set** — the qualification paper is described by its role, not its size, so the
|
||||
two uses cannot be mistaken for unrelated numbers that happen to coincide.
|
||||
|
||||
*On "pseudonymised" rather than "anonymised" or "de-identified" — the weaker word is the
|
||||
honest one.* The salt is a fixed hard-coded string, so anyone holding the script can
|
||||
recompute the mapping; full timestamps are kept; message text is preserved verbatim; and
|
||||
the redaction patterns cover mainland-China formats only — Hong Kong 8-digit numbers,
|
||||
North American +1 numbers, personal names, WeChat IDs and street addresses were all
|
||||
measured passing through. **This data therefore remains personal data, and a deletion
|
||||
request still reaches it.** A separate and non-substituting fact: the payload that reaches
|
||||
the weights is only the prompt plus five scores, carrying no identifier and no timestamp.
|
||||
Both statements are true; neither one covers for the other.
|
||||
2. **Affordable teacher**: `deepseek-chat` running the exact production rubric, qualified at
|
||||
**88.7% dimension-level / 97.5% decision-level** agreement before being allowed to label
|
||||
anything.
|
||||
3. **Synthetic textbook**: ~20,000 admitted dialogue windows across 6 data generations
|
||||
(40+ scenario cells with per-cell label-band admission gates, style quotas for short/
|
||||
fragmented/code-switched messages, crisis and boundary contrast pairs).
|
||||
**No external-client message has ever entered training — by construction.** Starting
|
||||
with v6.1, the corpus additionally includes 440 real conversation turns contributed by
|
||||
three company-internal staff members (the founder and two staff counselors), with their
|
||||
explicit consent, upsampled ×3 (~8% of the corpus).
|
||||
4. **Student**: Qwen3-0.6B, LoRA on a single MacBook (MLX; prompt-masked loss, cosine decay,
|
||||
grad-checkpointing). Total API cost of the whole project: ~US$7.
|
||||
|
||||
Key lessons the hard way (kept as disciplines): gradient-mask the prompt (72% of gradient was
|
||||
being wasted); halve batch size when doubling sequence length (a silent fp16 explosion taught us);
|
||||
verify every deploy down to a landed row.
|
||||
|
||||
## Limitations & known residuals
|
||||
|
||||
- **Chinese-primary** (zh 60% / mixed 22% / en 18% in training); English works but is less tested.
|
||||
- Message-level footprint, not a person-level or relationship-level assessment.
|
||||
- Mid-band calibration is coarse (0.5 grid; mid-band usage 7.4% vs reference 21–32%).
|
||||
- Known residuals: a small set of highly implicit severe-distress phrasings remains hard
|
||||
(shared across all versions and the reference scorer); occasional over-scoring of bounded
|
||||
multi-step worries on E. The conversational-action gap on A was substantially closed in v6.1
|
||||
by real-conversation training data (A 81% → 85%).
|
||||
- Trained against one specific rubric; scores are **relative to that rubric**, not universal
|
||||
psychological ground truth.
|
||||
|
||||
## Versions
|
||||
|
||||
| Version | Change | Dim-level | Decision-level |
|
||||
|---|---|---|---|
|
||||
| v2 | first distillation (7.5k synthetic) | 81% | 95.4% |
|
||||
| v3.x | rebalance + defect repair | 81% | 96.8% |
|
||||
| v4 | 8-agent data audit → 15k corpus, masked loss | 84% | 96.3% |
|
||||
| v5 | synthetic patch cells — **rejected** (crisis-recall regression; kept as a negative result) | — | — |
|
||||
| v6 | + real turns with incumbent labels — **rejected** (3 crisis-artifact rows rode into training, crisis misses 5 → 9; kept as a negative result) | — | — |
|
||||
| v6.1 | + 440 consented internal-staff turns (teacher labels) | 85.6% | 96.2% |
|
||||
| **v7 (this release)** | corpus re-labelled at temperature 0 (label denoising) + 2,713-row boundary-discrimination patch | 85.1% | **97.1%** |
|
||||
|
||||
## License
|
||||
|
||||
**HAMO-RAIL-S 1.0** (see [LICENSE](https://huggingface.co/HamoAI/hamo-score-0.6b/blob/main/LICENSE)): free commercial and non-commercial use,
|
||||
modification and redistribution, with four use restrictions — no standalone clinical
|
||||
determinations, no consequential decisions about individuals (employment / insurance /
|
||||
surveillance screening), consumer mental-wellness deployments must keep independent upstream
|
||||
crisis handling + AI disclosure, no re-identification. Base model Qwen3-0.6B remains Apache-2.0.
|
||||
|
||||
---
|
||||
|
||||
# 中文说明
|
||||
|
||||
> ### 👉 请从工具包开始,而不是从权重开始
|
||||
> `pip install hamo-score` → [**hamo-score-toolkit**](https://github.com/HamoAI/hamo-score-toolkit)(Apache-2.0)
|
||||
> 是这个模型的另一半:唯一正确的提示词格式、**必须置于模型上游的危机闸门**、分数该喂进去的
|
||||
> 平滑折算、一条命令起的 Docker 服务器,以及 195 题自检考卷。只拿权重,恰恰会走成这个模型
|
||||
> 设计上要防住的那种部署。
|
||||
> **对某个评分不服?** 那是你能给我们的最有价值的东西——
|
||||
> [提一条分歧报告](https://github.com/HamoAI/hamo-score-toolkit/issues/new?template=score_disagreement.md),
|
||||
> 它会直接进入引导后续版本的人类金标计划。
|
||||
|
||||
**hamo-score-0.6b** 是 Hamo AI 闭环疗愈引擎里第一个蒸馏进生产的组件:给心理支持对话中
|
||||
来访者的每一句话「把脉」,输出五路 0–3 分的脉象(A 行动力 / W 退缩 / E 极端化 / H 敌意 /
|
||||
B 边界感)。**它从不写回复,也不是危机检测器**——在 Hamo 生产系统里,危机内容在更上游被
|
||||
独立的确定性机制短路,永远到不了把脉师面前;任何部署都必须复刻这个模式(见 LICENSE §3c)。
|
||||
|
||||
**关于 B(边界感)**:它的理论本源是家庭治疗中的「自我分化」——是「我是我、你是你」的二元
|
||||
关系,还是彼此淹没的混沌一元。逐句评分器看不见关系,只看得见语言,所以 B 测的是边界感的
|
||||
**语言足迹**:「我需要…」「这是我的底线」得高分;惊慌的倾泻(自我淹没在情绪里)得低分;
|
||||
骂人算 H 不算 B。B 是逐句信号,不是关系诊断。
|
||||
|
||||
**成绩单(v7,本次发布)**:真实假名化生产对话终评(453 条未动用终评切分,金标含 8 处人工
|
||||
修正;评分真值来自被替换的大模型评分器;考卷数据从未参与训练):维度级 ±0.5 一致率 **85.1%**、
|
||||
决策级(经确定性压力折算后的状态桶判定)**97.1%**、危机语漏检 **3 条**(v6.1 对应为 85.6% / 96.2% / 4 条)。
|
||||
其中前两项是一致率、**越高越好**;「危机语漏检」是条数、**越低越好**——v7 由 4 条降到 3 条。
|
||||
维度级那 0.5 个点的回落不是退步:v7 的语料由教师在温度 0 下全量重标(去标签噪声),学生因此
|
||||
更忠于**去噪后**的标签,对当初那个含噪参照的一致率自然略降——实测其与温度 0 教师在 B 维的一致率
|
||||
由 57.3% 升到 69.3%,而对原参照的 B 一致率纹丝不动。
|
||||
|
||||
**v7 真正修好的是边界判别**:在一份专为探测 B 决策面而造的 300 题考卷上(60 组配对 + 180 条单题,
|
||||
覆盖九类陷阱格子),v6.1 会把**自我消融**读成强边界——「行,我全听你的,你说哪天去就哪天去」
|
||||
真值 B=0.0,它给 2.5。那是 B 负极上的符号错误,而 B 在下游压力公式里权重最大。
|
||||
|
||||
| 300 题边界判别卷 | 越好方向 | v6.1 | **v7** |
|
||||
|---|---|---|---|
|
||||
| 造分——真值 0 却给 ≥1.0(看见并不存在的边界) | ↓ **越低越好** | 13.5% | **2.9%** |
|
||||
| 漏判——真值高却给 ≤0.5(漏掉真实的边界) | ↓ **越低越好** | 16.1% | **12.8%** |
|
||||
| **符号翻转**——真值 0.0 却给 ≥2.0(条数;把自我消融读成强边界) | ↓ **越低越好** | 6 条 | **0 条** |
|
||||
| 配对方向正确率——高边界那一臂是否排在低的之上 | ↑ 越高越好 | 81.7% | **98.3%** |
|
||||
|
||||
前三行是错误率、第四行是正确率,所以**前三行越低越好、最后一行越高越好**;v7 四项全部改善。
|
||||
|
||||
两列均测于随包发布的 q8 GGUF + 中性采样,描述的是你下载到的产物本身。
|
||||
|
||||
参照系——同一批消息让原评分器自己打两遍,维度级自洽也只有 94–98%。单条延迟:M1 Pro 约
|
||||
0.8 秒;2 vCPU ARM 服务器(纯 CPU,q8 GGUF)1.5–2.9 秒。
|
||||
|
||||
**残余重叠:我们量了,没有假设掉。** 考卷与训练集按样本、按会话完全不相交,但授权供数的内部
|
||||
员工会在不同会话里重复说同样的话:终评 453 题中有 42 题(9.3%)的正文在那 440 条授权训练数据
|
||||
里出现过,其中 11 题(2.4%)连最近上下文也相同。分开判卷的结果——这 42 题维度级 89.0%、
|
||||
决策级 100%;其余 411 题无任何正文重叠,**维度级 85.2%、决策级 95.9%**。也就是说,上面两个
|
||||
成绩含约 **+0.4 / +0.3 个百分点**的乐观。幅度不大,但我们选择自己写出来。(42 题里只有 5 题
|
||||
连标签也相同——同一句话换个轮次通常拿到不同分数,所以它们不是白送的分,只是更容易的分。)
|
||||
|
||||
**训练方式**:三级师徒链——生产历史评分出考卷(1,198 条假名化真题——用「假名化」而非「匿名化」
|
||||
是因为弱的那个词才是诚实的:盐是硬编码固定字符串、完整时间戳保留、正文逐字保留,且脱敏正则只覆盖
|
||||
大陆格式,香港 8 位号码、北美 +1 号码、人名、微信号与住址实测全部穿过,**故这批数据仍属个人信息,
|
||||
删除权仍及于它**;另一件必须单独陈述、不可用来顶替上一条的事实是:进入权重的载荷只有提示词与五个
|
||||
分数,不含任何标识符与时间戳。按会话哈希切分为 440 条
|
||||
校准集与 758 条评测集,后者再切成 305 条选型集与判定本次成绩的 453 条终评集)→ DeepSeek 在
|
||||
校准集上通过资格考(维度级 88.7% / 决策级 97.5%)后当教师;本卡中「440」始终指那批经授权的
|
||||
内部员工轮次,资格考卷按用途称呼、不按题量称呼,以免两处用法被误读成两个撞车的数字 → 约 2 万段合成对话当教材(40+ 场景格子、逐格标签准入闸门、短句/
|
||||
碎片/中英混杂风格配额)→ Qwen3-0.6B 学生在一台 MacBook 上 LoRA 学成。全项目 API 成本
|
||||
约 7 美元。**训练语料从不包含任何外部来访者消息(构造上保证);自 v6.1 起额外加入 440 条公司内部员工(创始人与两位咨询师)明示授权的真实对话轮次(×3 上采样,约占语料 8%)。**
|
||||
|
||||
**社区量化档位(我们实测过其中一档)**:社区志愿者 mradermacher 制作了
|
||||
[**Q2_K→f16 共 12 个静态 GGUF 量化档**](https://huggingface.co/mradermacher/hamo-score-0.6b-GGUF)
|
||||
——感谢他的工作,也感谢他在再分发中完整保留了 RAIL-S 许可条款。我们公布的指标测于 **Q8_0**;
|
||||
由于这个模型的读数要门控对话能走多深,低比特量化掉了多少不是困惑度数字而是临床问题,所以我们
|
||||
用同一套 453 题终评、同一段提示词、同一个解析器,实测了社区的 **Q6_K** 与 **Q4_K_M**:
|
||||
|
||||
| | Q8_0(我们的基线) | Q6_K(社区) | Q4_K_M(社区) |
|
||||
|---|---|---|---|
|
||||
| 维度级 ±0.5 一致率 | 85.5% | 84.4% | 84.2% |
|
||||
| 决策级(状态桶) | 96.2% | 95.8% | 95.6% |
|
||||
| 危机 W 漏检(金标 ≥2.5 → 预测 <0.5,n=37) | 4 | 3 | **5** |
|
||||
| 那 37 条危机相邻样本的 W 均值(金标 2.84) | 2.39 | 2.26 | **2.01** |
|
||||
| JSON 合法率 | 100% | 100% | 100% |
|
||||
| 体积 | 0.64 GB | 0.50 GB | 0.40 GB |
|
||||
| P50 延迟(M 系列,Metal) | 0.50 秒 | 0.44 秒 | 0.42 秒 |
|
||||
|
||||
与 Q8_0 直接对比,两个社区档位落在同一状态桶的比例都约 98%(压力更新偏差均值 0.09)。总分
|
||||
几乎分不出高下——正因如此,我们才往下面看了一层。
|
||||
|
||||
**总分掩盖了什么:低比特档位会「衰减」信号,而 Q4_K_M 恰恰在最不容有失的地方衰减最厉害。**
|
||||
所有维度相对 Q8_0 系统性下移,且集中在行动力上(Q4_K_M 均值 −0.19,Q6_K −0.11)。在终评集
|
||||
的 37 条危机相邻样本(金标 W≥2.5)上,**Q4_K_M 有 20 条打得比 Q8_0 更低、仅 1 条更高**,把
|
||||
该子集的退缩信号均值从 2.39 拉到 2.01,并因此多漏检一条危机信号;而 Q6_K 的退缩信号在全卷上
|
||||
基本无损(均值偏移 +0.01,Q4_K_M 为 −0.04)。请注意三个档位的状态桶一致率只相差 0.4–0.6 个
|
||||
百分点:**桶的边界粗到足以吸收一个被压扁的信号,所以只看桶一致率永远发现不了这件事。**
|
||||
(就危机漏检本身而言,37 条里的 3 与 4 属于噪声范围——我们把 Q6_K 读作「与 Q8_0 持平」,
|
||||
而不是「优于」。)
|
||||
|
||||
**我们的建议**:
|
||||
|
||||
- **Q8_0** —— 参考档。读数用于门控行为、且你付得起 0.64 GB 时,用它。
|
||||
- **Q6_K** —— 我们愿意为门控用途背书的最低档。代价是约 1 个百分点的维度级一致率,退缩信号
|
||||
保持完好,体积省 22%。
|
||||
- **Q4_K_M** —— 适合研究、离线分析,以及分数由人来读而不是由系统据以行动的场景。若内存迫使
|
||||
它进入门控部署,请相应下调退缩维度的阈值以补偿上述衰减,并把确定性危机检测保持在上游
|
||||
(无论用哪个量化档,LICENSE §3c 都要求这个模式)。
|
||||
|
||||
其余九个档位我们未做验证,比 Q4_K_M 更低的比特应默认更差,直到有人量过。评测脚本在
|
||||
[工具包](https://github.com/HamoAI/hamo-score-toolkit)里——如果你验证了我们没验证过的档位,
|
||||
我们很乐意把你的数据链上来。
|
||||
|
||||
**官方工具包(已开源到 GitHub)**:[hamo-score-toolkit](https://github.com/HamoAI/hamo-score-toolkit)
|
||||
(Apache-2.0)是这个模型的「另一半」——一条 `pip install` 装上唯一正确的提示词格式、容错
|
||||
解析、参考版压力折算与许可证要求的危机闸门;一条 `docker compose up` 跑起参考服务器
|
||||
(`POST /score` 走完整的 闸门→评分→平滑→状态桶 管线);一份 195 题合成自检考卷 + 10 条
|
||||
手写闸门用例,对照官方参考带(JSON 100%、维度级 84.0%、闸门 10/10)验证你的部署接线;
|
||||
还有一份微调指南([docs/finetune.md](https://github.com/HamoAI/hamo-score-toolkit/blob/main/docs/finetune.md),
|
||||
六代打法,含两代拒收的完整原因)。发布文:《[开源 hamo-score-toolkit:把模型的另一半也交出去](https://www.hamo.ai/blog/open-sourcing-hamo-score-toolkit/)》。
|
||||
|
||||
**许可证**:HAMO-RAIL-S 1.0——自由商用与修改,但有四条使用限制:不得独立做临床判定、
|
||||
不得用于对个人的重大决定(雇佣/保险/监控筛查)、面向消费者的心理健康部署必须保留独立的
|
||||
上游危机处理与 AI 身份披露、不得试图重识别个人。
|
||||
|
||||
配套阅读(背景与方法论):《[hamo-score-0.6b 是怎么蒸出来的:一段平台期、三个坑,以及数据为什么赢了参数量](https://www.hamo.ai/blog/distilling-hamo-score-0-6b/)》。
|
||||
85
chat_template.jinja
Normal file
85
chat_template.jinja
Normal file
@@ -0,0 +1,85 @@
|
||||
{%- if tools %}
|
||||
{{- '<|im_start|>system\n' }}
|
||||
{%- if messages[0].role == 'system' %}
|
||||
{{- messages[0].content + '\n\n' }}
|
||||
{%- endif %}
|
||||
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||
{%- for tool in tools %}
|
||||
{{- "\n" }}
|
||||
{{- tool | tojson }}
|
||||
{%- endfor %}
|
||||
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||
{%- else %}
|
||||
{%- if messages[0].role == 'system' %}
|
||||
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||
{%- for message in messages[::-1] %}
|
||||
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||
{%- if ns.multi_step_tool and message.role == "user" and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||
{%- set ns.multi_step_tool = false %}
|
||||
{%- set ns.last_query_index = index %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- for message in messages %}
|
||||
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
|
||||
{%- elif message.role == "assistant" %}
|
||||
{%- set content = message.content %}
|
||||
{%- set reasoning_content = '' %}
|
||||
{%- if message.reasoning_content is defined and message.reasoning_content is not none %}
|
||||
{%- set reasoning_content = message.reasoning_content %}
|
||||
{%- else %}
|
||||
{%- if '</think>' in message.content %}
|
||||
{%- set content = message.content.split('</think>')[-1].lstrip('\n') %}
|
||||
{%- set reasoning_content = message.content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- if loop.index0 > ns.last_query_index %}
|
||||
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||
{%- endif %}
|
||||
{%- else %}
|
||||
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||
{%- endif %}
|
||||
{%- if message.tool_calls %}
|
||||
{%- for tool_call in message.tool_calls %}
|
||||
{%- if (loop.first and content) or (not loop.first) %}
|
||||
{{- '\n' }}
|
||||
{%- endif %}
|
||||
{%- if tool_call.function %}
|
||||
{%- set tool_call = tool_call.function %}
|
||||
{%- endif %}
|
||||
{{- '<tool_call>\n{"name": "' }}
|
||||
{{- tool_call.name }}
|
||||
{{- '", "arguments": ' }}
|
||||
{%- if tool_call.arguments is string %}
|
||||
{{- tool_call.arguments }}
|
||||
{%- else %}
|
||||
{{- tool_call.arguments | tojson }}
|
||||
{%- endif %}
|
||||
{{- '}\n</tool_call>' }}
|
||||
{%- endfor %}
|
||||
{%- endif %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- elif message.role == "tool" %}
|
||||
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||
{{- '<|im_start|>user' }}
|
||||
{%- endif %}
|
||||
{{- '\n<tool_response>\n' }}
|
||||
{{- message.content }}
|
||||
{{- '\n</tool_response>' }}
|
||||
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||
{{- '<|im_end|>\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- if add_generation_prompt %}
|
||||
{{- '<|im_start|>assistant\n' }}
|
||||
{%- if enable_thinking is defined and enable_thinking is false %}
|
||||
{{- '<think>\n\n</think>\n\n' }}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
30
config.json
Normal file
30
config.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 151643,
|
||||
"eos_token_id": 151645,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 1024,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 3072,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"model_type": "qwen3",
|
||||
"num_attention_heads": 16,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.51.0",
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
3
gguf/hamo-score-0.6b-v61.q8.gguf
Normal file
3
gguf/hamo-score-0.6b-v61.q8.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:114a0ccba6e13b35170425fbdbdad8908b741f0ea66360c422f26ee633398561
|
||||
size 639446912
|
||||
3
gguf/hamo-score-0.6b-v7.q8.gguf
Normal file
3
gguf/hamo-score-0.6b-v7.q8.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:fb1018c8a38ad0da573544b827e42f1685e051c688f8de587cbb81ae1275e3ea
|
||||
size 639446912
|
||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:710672d4ffbfcdb5790b0001bee4d4e67414fa5ce9b2cd1be693ab203b90d1ad
|
||||
size 1192134935
|
||||
318
model.safetensors.index.json
Normal file
318
model.safetensors.index.json
Normal file
@@ -0,0 +1,318 @@
|
||||
{
|
||||
"metadata": {
|
||||
"total_size": 1192099840,
|
||||
"total_parameters": 596049920
|
||||
},
|
||||
"weight_map": {
|
||||
"model.embed_tokens.weight": "model.safetensors",
|
||||
"model.layers.0.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.0.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.0.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.0.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.0.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.0.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.0.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.0.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.0.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.0.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.0.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.1.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.1.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.1.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.1.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.1.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.1.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.1.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.1.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.1.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.1.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.1.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.10.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.10.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.10.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.10.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.10.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.10.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.10.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.10.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.10.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.10.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.10.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.11.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.11.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.11.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.11.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.11.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.11.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.11.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.11.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.11.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.11.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.11.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.12.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.12.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.12.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.12.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.12.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.12.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.12.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.12.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.12.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.12.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.12.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.13.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.13.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.13.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.13.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.13.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.13.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.13.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.13.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.13.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.13.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.13.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.14.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.14.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.14.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.14.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.14.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.14.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.14.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.14.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.14.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.14.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.14.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.15.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.15.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.15.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.15.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.15.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.15.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.15.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.15.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.15.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.15.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.15.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.16.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.16.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.16.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.16.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.16.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.16.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.16.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.16.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.16.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.16.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.16.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.17.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.17.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.17.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.17.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.17.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.17.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.17.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.17.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.17.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.17.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.17.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.18.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.18.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.18.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.18.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.18.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.18.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.18.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.18.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.18.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.18.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.18.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.19.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.19.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.19.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.19.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.19.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.19.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.19.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.19.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.19.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.19.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.19.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.2.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.2.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.2.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.2.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.2.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.2.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.2.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.2.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.2.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.2.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.2.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.20.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.20.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.20.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.20.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.20.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.20.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.20.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.20.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.20.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.20.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.20.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.21.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.21.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.21.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.21.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.21.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.21.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.21.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.21.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.21.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.21.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.21.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.22.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.22.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.22.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.22.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.22.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.22.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.22.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.22.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.22.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.22.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.22.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.23.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.23.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.23.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.23.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.23.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.23.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.23.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.23.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.23.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.23.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.23.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.24.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.24.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.24.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.24.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.24.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.24.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.24.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.24.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.24.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.24.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.24.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.25.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.25.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.25.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.25.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.25.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.25.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.25.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.25.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.25.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.25.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.25.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.26.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.26.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.26.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.26.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.26.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.26.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.26.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.26.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.26.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.26.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.26.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.27.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.27.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.27.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.27.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.27.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.27.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.27.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.27.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.27.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.27.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.27.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.3.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.3.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.3.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.3.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.3.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.3.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.3.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.3.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.3.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.3.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.3.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.4.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.4.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.4.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.4.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.4.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.4.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.4.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.4.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.4.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.4.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.4.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.5.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.5.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.5.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.5.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.5.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.5.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.5.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.5.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.5.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.5.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.5.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.6.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.6.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.6.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.6.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.6.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.6.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.6.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.6.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.6.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.6.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.6.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.7.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.7.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.7.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.7.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.7.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.7.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.7.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.7.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.7.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.7.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.7.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.8.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.8.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.8.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.8.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.8.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.8.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.8.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.8.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.8.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.8.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.8.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.layers.9.input_layernorm.weight": "model.safetensors",
|
||||
"model.layers.9.mlp.down_proj.weight": "model.safetensors",
|
||||
"model.layers.9.mlp.gate_proj.weight": "model.safetensors",
|
||||
"model.layers.9.mlp.up_proj.weight": "model.safetensors",
|
||||
"model.layers.9.post_attention_layernorm.weight": "model.safetensors",
|
||||
"model.layers.9.self_attn.k_norm.weight": "model.safetensors",
|
||||
"model.layers.9.self_attn.k_proj.weight": "model.safetensors",
|
||||
"model.layers.9.self_attn.o_proj.weight": "model.safetensors",
|
||||
"model.layers.9.self_attn.q_norm.weight": "model.safetensors",
|
||||
"model.layers.9.self_attn.q_proj.weight": "model.safetensors",
|
||||
"model.layers.9.self_attn.v_proj.weight": "model.safetensors",
|
||||
"model.norm.weight": "model.safetensors"
|
||||
}
|
||||
}
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
|
||||
size 11422650
|
||||
16
tokenizer_config.json
Normal file
16
tokenizer_config.json
Normal file
@@ -0,0 +1,16 @@
|
||||
{
|
||||
"add_prefix_space": false,
|
||||
"backend": "tokenizers",
|
||||
"bos_token": null,
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|im_end|>",
|
||||
"errors": "replace",
|
||||
"is_local": true,
|
||||
"local_files_only": false,
|
||||
"model_max_length": 131072,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"tool_parser_type": "json_tools",
|
||||
"unk_token": null
|
||||
}
|
||||
Reference in New Issue
Block a user