初始化项目,由ModelHub XC社区提供模型
Model: flowxai/scam-guard-qwen06b Source: Original Platform
This commit is contained in:
37
.gitattributes
vendored
Normal file
37
.gitattributes
vendored
Normal file
@@ -0,0 +1,37 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||||
|
gguf/scam-guard-qwen06b-cuda-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
637
README.md
Normal file
637
README.md
Normal file
@@ -0,0 +1,637 @@
|
|||||||
|
---
|
||||||
|
license: apache-2.0
|
||||||
|
language:
|
||||||
|
- en
|
||||||
|
- ro
|
||||||
|
library_name: mlx
|
||||||
|
pipeline_tag: text-classification
|
||||||
|
tags:
|
||||||
|
- scam-detection
|
||||||
|
- fraud-detection
|
||||||
|
- smishing
|
||||||
|
- phishing
|
||||||
|
- on-device
|
||||||
|
- mlx
|
||||||
|
- gguf
|
||||||
|
- qwen3
|
||||||
|
- safety
|
||||||
|
- romanian
|
||||||
|
base_model: Qwen/Qwen3-0.6B
|
||||||
|
datasets:
|
||||||
|
- flowxai/scamguardbench
|
||||||
|
model-index:
|
||||||
|
- name: scam-guard-qwen06b
|
||||||
|
results:
|
||||||
|
- task:
|
||||||
|
type: text-classification
|
||||||
|
name: Scam verdict classification (ScamGuardBench v0.2, 120-item slice)
|
||||||
|
dataset:
|
||||||
|
name: ScamGuardBench v0.2
|
||||||
|
type: flowxai/scamguardbench
|
||||||
|
metrics:
|
||||||
|
- type: f1
|
||||||
|
name: verdict micro-F1
|
||||||
|
value: 0.958
|
||||||
|
- type: f1
|
||||||
|
name: verdict macro-F1
|
||||||
|
value: 0.926
|
||||||
|
- type: f1
|
||||||
|
name: tactic macro-F1
|
||||||
|
value: 0.951
|
||||||
|
- type: recall
|
||||||
|
name: evidence pass rate
|
||||||
|
value: 0.980
|
||||||
|
- type: false_positive_rate
|
||||||
|
name: legit-confusable FP-rate (scam_likely on legit)
|
||||||
|
value: 0.000
|
||||||
|
- task:
|
||||||
|
type: text-classification
|
||||||
|
name: Out-of-distribution fresh CERT-pattern messages (20-item hand-authored set)
|
||||||
|
dataset:
|
||||||
|
name: ScamGuardBench v0.2
|
||||||
|
type: flowxai/scamguardbench
|
||||||
|
metrics:
|
||||||
|
- type: accuracy
|
||||||
|
name: verdict accuracy (correct / 20)
|
||||||
|
value: 0.90
|
||||||
|
- type: f1
|
||||||
|
name: verdict macro-F1 (OOD)
|
||||||
|
value: 0.614
|
||||||
|
- type: false_positive_rate
|
||||||
|
name: OOD legit false-alarm rate
|
||||||
|
value: 0.000
|
||||||
|
---
|
||||||
|
|
||||||
|
## Inference contract
|
||||||
|
|
||||||
|
Running this model correctly requires its **frozen inference contract** — the exact
|
||||||
|
system prompt, output JSON schema, user-turn format, and constrained-decode spec it
|
||||||
|
was trained against. See [`inference_contract/`](./inference_contract):
|
||||||
|
|
||||||
|
- [`INFERENCE.md`](./inference_contract/INFERENCE.md) — wiring guide: system prompt, user turn `[channel: <tag>]\n<message>` (tag `sms`/`email`/`chat`), **constrained JSON decoding** (required — pins the enums), and the verbatim-evidence check.
|
||||||
|
- [`prompt_scamguard_sys_v1.txt`](./inference_contract/prompt_scamguard_sys_v1.txt) — the system prompt, verbatim.
|
||||||
|
- [`schema_scamguard_v1.json`](./inference_contract/schema_scamguard_v1.json) — output JSON Schema for constrained decoding.
|
||||||
|
|
||||||
|
Prompt version `scamguard_sys_v1`. Do not edit the prompt/schema; the weights are trained against them.
|
||||||
|
|
||||||
|
<!--
|
||||||
|
PREPARED FOR HUGGING FACE — NOT YET UPLOADED. Nothing here has been published.
|
||||||
|
|
||||||
|
Working NAME: "scam-guard". A final release name is chosen by a human at the
|
||||||
|
Phase-7 review STOP; search-replace "scam-guard" to rename in one pass. The ORG
|
||||||
|
is flowxai (fixed): dataset flowxai/scamguardbench, this model
|
||||||
|
flowxai/<name>-qwen06b, its sibling flowxai/<name>-qwen17b.
|
||||||
|
|
||||||
|
This is the 0.6B (on-device) card. The 1.7B (quality) card is a sibling repo:
|
||||||
|
release/README_qwen17b.md -> flowxai/scam-guard-qwen17b.
|
||||||
|
|
||||||
|
The fine-tuned synthetic-bench metrics are PROVISIONAL-on-synthetic-bench; the
|
||||||
|
out-of-distribution fresh-message results are measured and included. All fine-tuned
|
||||||
|
numbers on this card are the FINAL CUDA 3-epoch run (superseding the MLX first pass).
|
||||||
|
-->
|
||||||
|
|
||||||
|
# scam-guard 0.6B (working name) — the on-device pick
|
||||||
|
|
||||||
|
**An on-device scam & fraud message detector for SMS, email, and chat text (English + Romanian).**
|
||||||
|
|
||||||
|
This is the **0.6B** model — the **smallest and fastest** of the two scam-guard
|
||||||
|
sizes, and the **on-device target** (~0.4 GB at int4). For higher out-of-distribution
|
||||||
|
accuracy at a larger footprint, see the sibling **[1.7B quality
|
||||||
|
pick](https://huggingface.co/flowxai/scam-guard-qwen17b)**.
|
||||||
|
|
||||||
|
Given one message, scam-guard returns a **3-level verdict**, the **manipulation
|
||||||
|
tactics** it found (each with a verbatim evidence span quoted from the message), a
|
||||||
|
**calm plain-language explanation**, and a **recommended safe action** from a fixed
|
||||||
|
list. It is built for everyday people — explicitly including **elderly and
|
||||||
|
non-technical users**, who are the most targeted.
|
||||||
|
|
||||||
|
The product insight that shapes everything: **the `suspicious` middle level exists
|
||||||
|
to be honest about uncertainty rather than force a binary.** A consumer safety tool
|
||||||
|
that must answer "scam or not" will either cry wolf or wave real scams through; a
|
||||||
|
third honest verdict — "this might be fine, verify through your own channel first"
|
||||||
|
— lets the model say *I'm not sure* instead of guessing. Relatedly, scam-guard
|
||||||
|
**never emits a probability**: an uncalibrated confidence number on a consumer
|
||||||
|
safety tool is worse than none, so we give you honest per-class behaviour instead
|
||||||
|
(see [Calibration](#calibration--we-dont-give-you-a-probability)).
|
||||||
|
|
||||||
|
- **`flowxai/scam-guard-qwen06b`** (this card, 0.6B) — smallest and fastest; the on-device target.
|
||||||
|
- **`flowxai/scam-guard-qwen17b`** (1.7B, sibling) — the more robust choice out-of-distribution (see [Size decision](#size-decision)).
|
||||||
|
|
||||||
|
Both are LoRA-fine-tuned from Apache-2.0 Qwen3 base models. On-device formats:
|
||||||
|
**GGUF** (llama.cpp, int8 `Q8_0` + int4 `Q4_K_M`) and **MLX-quantized** (int4 + int8;
|
||||||
|
mlx-swift runs these on iOS too).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## How do I use it?
|
||||||
|
|
||||||
|
Three copy-pasteable ways to turn a message into a verdict. All run **fully
|
||||||
|
on-device** — no network at inference, ever.
|
||||||
|
|
||||||
|
Real example input (a fresh Romanian courier-fee smishing message):
|
||||||
|
|
||||||
|
```
|
||||||
|
Coletul dumneavoastra nu a putut fi livrat. Pentru reprogramare achitati taxa
|
||||||
|
vamala de 3,20 lei aici: http://colet-reprogramare.example.net/plata Livrarea se
|
||||||
|
anuleaza in 48h.
|
||||||
|
```
|
||||||
|
|
||||||
|
### (a) llama.cpp / GGUF
|
||||||
|
|
||||||
|
Download a GGUF (int8 `Q8_0` recommended) and run the message through it. The model
|
||||||
|
emits a single strict JSON object.
|
||||||
|
|
||||||
|
**`llama-cli` (CPU-only, `-ngl 0`):**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
llama-cli -m scam-guard-qwen06b-Q8_0.gguf -ngl 0 --temp 0 -no-cnv \
|
||||||
|
-p "$(cat <<'EOF'
|
||||||
|
<system prompt: see src/scamguard/schema.py::SYSTEM_PROMPT>
|
||||||
|
[channel: sms]
|
||||||
|
Coletul dumneavoastra nu a putut fi livrat. Pentru reprogramare achitati taxa vamala de 3,20 lei aici: http://colet-reprogramare.example.net/plata Livrarea se anuleaza in 48h.
|
||||||
|
EOF
|
||||||
|
)"
|
||||||
|
```
|
||||||
|
|
||||||
|
**`llama-cpp-python`:**
|
||||||
|
|
||||||
|
```python
|
||||||
|
from llama_cpp import Llama
|
||||||
|
from scamguard.schema import SYSTEM_PROMPT, ScamGuardOutput # the fixed task prompt + schema
|
||||||
|
|
||||||
|
llm = Llama(model_path="scam-guard-qwen06b-Q8_0.gguf", n_gpu_layers=0)
|
||||||
|
msg = ("Coletul dumneavoastra nu a putut fi livrat. Pentru reprogramare achitati "
|
||||||
|
"taxa vamala de 3,20 lei aici: http://colet-reprogramare.example.net/plata "
|
||||||
|
"Livrarea se anuleaza in 48h.")
|
||||||
|
|
||||||
|
out = llm.create_chat_completion(
|
||||||
|
messages=[
|
||||||
|
{"role": "system", "content": SYSTEM_PROMPT},
|
||||||
|
{"role": "user", "content": f"[channel: sms]\n{msg}"},
|
||||||
|
],
|
||||||
|
temperature=0.0,
|
||||||
|
)
|
||||||
|
raw = out["choices"][0]["message"]["content"]
|
||||||
|
verdict = ScamGuardOutput.model_validate_json(raw) # strict, extra="forbid"
|
||||||
|
```
|
||||||
|
|
||||||
|
### (b) MLX (Apple Silicon)
|
||||||
|
|
||||||
|
Off the MLX-quantized weights (int4/int8):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
mlx_lm.generate --model scam-guard-qwen06b-mlx-int4 --temp 0 \
|
||||||
|
--prompt "$(printf '[channel: sms]\nColetul dumneavoastra nu a putut fi livrat. Pentru reprogramare achitati taxa vamala de 3,20 lei aici: http://colet-reprogramare.example.net/plata Livrarea se anuleaza in 48h.')"
|
||||||
|
```
|
||||||
|
|
||||||
|
```python
|
||||||
|
from mlx_lm import load, generate
|
||||||
|
from scamguard.schema import SYSTEM_PROMPT
|
||||||
|
|
||||||
|
model, tok = load("scam-guard-qwen06b-mlx-int4")
|
||||||
|
msg = ("Coletul dumneavoastra nu a putut fi livrat. Pentru reprogramare achitati "
|
||||||
|
"taxa vamala de 3,20 lei aici: http://colet-reprogramare.example.net/plata "
|
||||||
|
"Livrarea se anuleaza in 48h.")
|
||||||
|
prompt = tok.apply_chat_template(
|
||||||
|
[{"role": "system", "content": SYSTEM_PROMPT},
|
||||||
|
{"role": "user", "content": f"[channel: sms]\n{msg}"}],
|
||||||
|
add_generation_prompt=True, enable_thinking=False,
|
||||||
|
)
|
||||||
|
raw = generate(model, tok, prompt=prompt, max_tokens=256, verbose=False)
|
||||||
|
```
|
||||||
|
|
||||||
|
Either backend returns the **same strict JSON** for this message:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"verdict": "scam_likely",
|
||||||
|
"tactics": [
|
||||||
|
{
|
||||||
|
"tactic": "subscription_trap",
|
||||||
|
"evidence": "achitati taxa vamala de 3,20 lei",
|
||||||
|
"explanation": "It asks you to pay a small fee to release a parcel."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"tactic": "urgency_pressure",
|
||||||
|
"evidence": "se anuleaza in 48h",
|
||||||
|
"explanation": "It invents a 48-hour deadline to rush you."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"explanation": "This looks like a scam because it uses a fake fee to prompt a payment and pressures you with an artificial deadline; do not act on it, and check with the real organisation through a channel you already trust.",
|
||||||
|
"recommended_action": "verify_via_official_app_or_site"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### (c) The demo verdict card (`demo/check.py`)
|
||||||
|
|
||||||
|
The reference demo renders that JSON as a card a family member can read. It
|
||||||
|
hard-enforces the no-network privacy promise:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
echo "Coletul dumneavoastra nu a putut fi livrat. Pentru reprogramare achitati taxa vamala de 3,20 lei aici: http://colet-reprogramare.example.net/plata Livrarea se anuleaza in 48h." | python demo/check.py
|
||||||
|
```
|
||||||
|
|
||||||
|
Rendered output (the actual card, from `reports/ood_fresh_demo.md`):
|
||||||
|
|
||||||
|
```
|
||||||
|
====================================================================
|
||||||
|
scam-guard — message safety check
|
||||||
|
====================================================================
|
||||||
|
[!] VERDICT: Likely a scam
|
||||||
|
|
||||||
|
This message shows clear signs of a scam. You do not need to do
|
||||||
|
anything it asks. Take your time — real organisations are fine
|
||||||
|
with you checking first.
|
||||||
|
--------------------------------------------------------------------
|
||||||
|
What we noticed:
|
||||||
|
- Sending a fake renewal or invoice to make you call or click
|
||||||
|
seen in: "achitati taxa vamala de 3,20 lei"
|
||||||
|
- Rushing you with a deadline or threat
|
||||||
|
seen in: "se anuleaza in 48h"
|
||||||
|
--------------------------------------------------------------------
|
||||||
|
What to do:
|
||||||
|
Check directly using the company's official app or website
|
||||||
|
that you open yourself — not the link here.
|
||||||
|
--------------------------------------------------------------------
|
||||||
|
In plain words:
|
||||||
|
This looks like a scam because it uses a fake renewal invoice to
|
||||||
|
prompt a call and it pressures you with an artificial deadline;
|
||||||
|
do not act on it, and check with the real organisation through a
|
||||||
|
channel you already trust.
|
||||||
|
====================================================================
|
||||||
|
scam-guard is a helper, not a guarantee. When in doubt, verify
|
||||||
|
through a channel you already trust. It never opens links.
|
||||||
|
====================================================================
|
||||||
|
```
|
||||||
|
|
||||||
|
`demo/check.py` defaults to the 0.6B MLX model (this one); `--backend gguf` and
|
||||||
|
`--model <path>` switch weights/backend, and `--size 1.7b` selects the sibling.
|
||||||
|
`--channel {sms,email,chat}` sets the channel tag.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## How it works
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
flowchart LR
|
||||||
|
SMS --> SG["scam-guard 0.6B"]
|
||||||
|
Email --> SG
|
||||||
|
Chat --> SG
|
||||||
|
SG --> V[verdict]
|
||||||
|
SG --> T[tactics]
|
||||||
|
SG --> E[evidence]
|
||||||
|
SG --> X[explanation]
|
||||||
|
SG --> A[action]
|
||||||
|
```
|
||||||
|
|
||||||
|
Compact ASCII flow — three channels in, one local model, five fields out:
|
||||||
|
|
||||||
|
```
|
||||||
|
SMS
|
||||||
|
\
|
||||||
|
Email ---> scam-guard 0.6B
|
||||||
|
Chat /
|
||||||
|
|
|
||||||
|
+ verdict
|
||||||
|
+ tactics
|
||||||
|
+ evidence
|
||||||
|
+ explanation
|
||||||
|
+ action
|
||||||
|
```
|
||||||
|
|
||||||
|
Under the hood, `scam-guard` is: `tokenizer → Qwen3-0.6B (LoRA fine-tuned) →
|
||||||
|
constrained JSON decode → evidence verifier (verbatim-substring kill-switch:
|
||||||
|
drops fabricated spans)`.
|
||||||
|
|
||||||
|
**The evidence kill-switch is the safety-critical stage.** Every tactic must cite a
|
||||||
|
span that is a **verbatim substring** of the input message (whitespace-normalized
|
||||||
|
only — no case/diacritic folding). A tactic whose evidence is not found verbatim is
|
||||||
|
**dropped and counted** as fabricated, so the model can never hallucinate a quote to
|
||||||
|
justify a warning.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Output schema
|
||||||
|
|
||||||
|
A single strict JSON object (`extra="forbid"`, frozen — a spurious field like
|
||||||
|
`confidence` is rejected):
|
||||||
|
|
||||||
|
- `verdict` — `scam_likely` | `suspicious` | `no_indicators` (**never a probability**).
|
||||||
|
- `scam_likely` — a clear scam mechanism is present and driven by tactics.
|
||||||
|
- `suspicious` — the honest middle: signals present but plausibly legitimate, or
|
||||||
|
only weak indicators (urgency alone, a link alone, an authority claim with no
|
||||||
|
ask). "Verify through your own channel first."
|
||||||
|
- `no_indicators` — no scam mechanism; legitimate messages can be urgent and contain links.
|
||||||
|
- `tactics[]` — each `{tactic, evidence, explanation}`, where `tactic` is one of 13
|
||||||
|
fixed ids and `evidence` is a **verbatim substring** (see the kill-switch above).
|
||||||
|
- `explanation` — one or two calm, actionable sentences.
|
||||||
|
- `recommended_action` — one id from a fixed list of 10 safe actions; the model can
|
||||||
|
never compose free-text advice that points back at the scammer's own channel.
|
||||||
|
|
||||||
|
The 13 tactics: `urgency_pressure`, `authority_impersonation`, `payment_redirect`,
|
||||||
|
`credential_phishing`, `courier_customs_fee`, `prize_lottery`, `investment_too_good`,
|
||||||
|
`romance_advance_fee`, `family_emergency_impersonation`, `tech_support`,
|
||||||
|
`link_obfuscation`, `refund_overpayment`, `subscription_trap`.
|
||||||
|
|
||||||
|
The 10 safe actions: `call_bank_official_number`, `do_not_click_link`,
|
||||||
|
`verify_via_official_app_or_site`, `call_family_member_known_number`,
|
||||||
|
`do_not_share_codes_or_credentials`, `do_not_send_money`, `ignore_and_delete`,
|
||||||
|
`report_to_authorities`, `check_sender_address`, `no_action_needed`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## On-device privacy promise
|
||||||
|
|
||||||
|
**scam-guard makes no network calls at inference, ever.** It reasons over the
|
||||||
|
message text only, fully locally. URL handling is **lexical only** — it inspects the
|
||||||
|
visible URL string (lookalike domains, userinfo tricks, shorteners, punycode hints)
|
||||||
|
and **never fetches anything**. This is the whole point: it works on messages people
|
||||||
|
would never upload to a cloud service. The reference demo (`demo/check.py`)
|
||||||
|
hard-enforces this with a no-network guard. At ~0.4 GB (int4), the 0.6B is the
|
||||||
|
smallest footprint of the two sizes — the one most comfortable on a phone.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Evaluation (0.6B)
|
||||||
|
|
||||||
|
Two evaluations, in order of what they tell you:
|
||||||
|
|
||||||
|
1. **Out-of-distribution (OOD) fresh messages** — the honest real-world signal.
|
||||||
|
2. **ScamGuardBench v0.2 synthetic bench** — a large in-distribution slice the model has
|
||||||
|
a home-field advantage on (read the caveat).
|
||||||
|
3. **Calibration** — class-level behaviour, since there is no probability to calibrate.
|
||||||
|
|
||||||
|
All fine-tuned numbers below are the **FINAL CUDA 3-epoch** run (`qwen06b_cuda`),
|
||||||
|
which supersedes the MLX first pass.
|
||||||
|
|
||||||
|
### Out-of-distribution results (fresh CERT-pattern messages)
|
||||||
|
|
||||||
|
20 **fresh** messages — 10 realistic scam patterns modeled on current CERT/DNSC-style
|
||||||
|
alerts (RO+EN) and 10 genuinely legit messages — **hand-authored from public
|
||||||
|
alert-pattern descriptions, NOT run through the synthetic generator** the model
|
||||||
|
trained on, sanitized, and asserted text-disjoint from training + every ScamGuardBench
|
||||||
|
version (`tests/test_ood_fresh.py`). This is the honest generalization test. Full
|
||||||
|
write-up: `reports/ood_fresh_demo.md`.
|
||||||
|
|
||||||
|
| Model | Correct / 20 | Dangerous MISSES (scam→no_indicators) | FALSE ALARMS (legit→scam_likely) | verdict macro-F1 |
|
||||||
|
| --- | --- | --- | --- | --- |
|
||||||
|
| `flowxai/scam-guard-qwen06b` (0.6B, on-device) | **18/20 (90%)** | 1 | **0** | 0.614 |
|
||||||
|
| claude-haiku-4-5 (OOD reference) | **19/20 (95%)** | 0 | 0 | 0.649 |
|
||||||
|
| `flowxai/scam-guard-qwen17b` (1.7B, sibling) | **19/20 (95%)** | 1 | **0** | 0.967 |
|
||||||
|
|
||||||
|
**Bench → OOD gap (the home-field advantage, quantified):** the 0.6B macro-F1 drops
|
||||||
|
**0.926 → 0.614 (−0.31)** on fresh messages. On a 20-item set macro-F1 is noisy (one
|
||||||
|
slip on a rare class moves it a lot), so the plain-verdict accuracy (18/20) is the more
|
||||||
|
stable read. Legit-FP stays **0.000**.
|
||||||
|
|
||||||
|
Honest reading:
|
||||||
|
|
||||||
|
- **The one dangerous MISS:** the RO WhatsApp family-emergency scam *"Mama, am
|
||||||
|
pierdut telefonul … poti sa imi trimiti 850 lei"*, waved through as
|
||||||
|
`no_indicators`. Family-emergency framing without an explicit money-transfer
|
||||||
|
keyword can slip past the model. This is a known RO-dominant pattern and the top
|
||||||
|
recall gap to fix; the frontier reference (haiku) catches it. It **persists on the
|
||||||
|
1.7B at 3 epochs too**, confirming it is a training-data gap, not a size/compute gap.
|
||||||
|
- **Zero false alarms** — no legit message was flagged `scam_likely`. The
|
||||||
|
"disable-in-a-week" failure mode did not appear on fresh legit traffic. (The 0.6B
|
||||||
|
soft-hedged one legit-adjacent scam to `suspicious` — the RO energy-subsidy scam —
|
||||||
|
reported, not counted as a false alarm.)
|
||||||
|
- **Format robustness held:** JSON validity 1.000 and evidence pass 1.000 on fresh
|
||||||
|
text (no repair/fallback needed).
|
||||||
|
|
||||||
|
**RO vs EN, OOD (plain per-language accuracy over the 20-item OOD set — 9 RO / 11 EN;
|
||||||
|
a per-language F1 is not computed in the reports):**
|
||||||
|
|
||||||
|
| Model | RO correct | EN correct |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `flowxai/scam-guard-qwen06b` | 7 / 9 | 10 / 11 |
|
||||||
|
|
||||||
|
RO is the first-class target language: the RO miss is the single family-emergency
|
||||||
|
scam (the 0.6B also soft-hedges the RO energy-subsidy scam to `suspicious`). On the
|
||||||
|
in-distribution bench the RO-dominant tactics score at ceiling
|
||||||
|
(`family_emergency_impersonation` F1 **1.000**, `courier_customs_fee` **1.000**,
|
||||||
|
`refund_overpayment` **0.889**), which is exactly why the OOD RO family-emergency
|
||||||
|
miss is the honest gap to close.
|
||||||
|
|
||||||
|
### ScamGuardBench v0.2 (synthetic, in-distribution) — with the home-field caveat
|
||||||
|
|
||||||
|
> **Honesty caveat — home-field advantage (read before citing these numbers).** The
|
||||||
|
> fine-tuned numbers **beat the frontier reference on this benchmark, and that does NOT
|
||||||
|
> mean the model is a better real-world scam detector.** ScamGuardBench v0.2 is built from the
|
||||||
|
> **same synthetic generator** the model trained on (held-out split,
|
||||||
|
> contamination-verified — no leakage of specific messages, but the **same
|
||||||
|
> distribution**: same output format, phrasing, tactic-to-message style). The
|
||||||
|
> fine-tune learned exactly that style. **The frontier reference is the honest upper
|
||||||
|
> reference for a cold-prompted generalist** — a better OOD proxy than the fine-tuned
|
||||||
|
> column — and the OOD results above are the real-world signal these bench numbers
|
||||||
|
> flatter.
|
||||||
|
|
||||||
|
Seeded **120-message** slice (18 `suspicious`, 39 legit-confusable). A false positive
|
||||||
|
is a `scam_likely` verdict on a legitimate message; `suspicious` is reported but
|
||||||
|
**not** counted as an FP. FP-rate on the legit-confusable subset is the **release gate**.
|
||||||
|
|
||||||
|
| Metric | **qwen06b (CUDA)** | claude-haiku-4-5 (frontier ref) | keyword (lower ref) |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| JSON validity (deployed decoder) | 1.000 | 1.000 | 1.000 |
|
||||||
|
| **raw** JSON validity (no repair) | 1.000 | n/a | n/a |
|
||||||
|
| verdict micro-F1 | 0.958 | 0.900 | 0.483 |
|
||||||
|
| verdict macro-F1 | **0.926** | 0.829 | 0.482 |
|
||||||
|
| tactic macro-F1 | 0.951 | 0.770 | 0.319 |
|
||||||
|
| evidence pass rate | 0.980 | 0.969 | 1.000 |
|
||||||
|
| **legit-confusable FP-rate** | **0.000** | 0.051 | 0.026 |
|
||||||
|
|
||||||
|
The reference `claude-haiku-4-5` **fails the FP gate** (legit-confusable FP 0.051,
|
||||||
|
2/39 — it over-flags genuine family money requests), while the fine-tune holds
|
||||||
|
**0.000**. That is the home-field-advantage signal, not a claim of superior
|
||||||
|
real-world judgment. `family_money_request` (a genuine "send me money" from family)
|
||||||
|
is the hardest legit class — every frontier model over-flags it — while this
|
||||||
|
fine-tune holds **0.000**.
|
||||||
|
|
||||||
|
> **Full-budget note (MLX → CUDA).** The shipped 0.6B is the CUDA 3-epoch run. It
|
||||||
|
> supersedes the earlier MLX ~1-epoch first pass (macro-F1 0.903 → **0.926**,
|
||||||
|
> micro 0.942 → **0.958**), while **holding legit-FP at 0.000**. The full budget
|
||||||
|
> bought +2.3 macro-F1 points in-distribution.
|
||||||
|
|
||||||
|
### Calibration — we don't give you a probability
|
||||||
|
|
||||||
|
scam-guard **deliberately emits no confidence percentage**. An uncalibrated number
|
||||||
|
on a consumer safety tool is worse than none: it invites false precision on a
|
||||||
|
judgement that is genuinely uncertain. Instead of a probability to calibrate, we give
|
||||||
|
you **honest per-class behaviour** so you know where the model is weak and where it is
|
||||||
|
safe.
|
||||||
|
|
||||||
|
**Confusion matrix** (`qwen06b` CUDA, ScamGuardBench v0.2, 120-item slice; rows = gold,
|
||||||
|
cols = predicted):
|
||||||
|
|
||||||
|
| gold \ pred | scam_likely | suspicious | no_indicators | recall |
|
||||||
|
| --- | --- | --- | --- | --- |
|
||||||
|
| scam_likely | 63 | 0 | 0 | 1.000 (n=63) |
|
||||||
|
| suspicious | 0 | 13 | 5 | 0.722 (n=18) |
|
||||||
|
| no_indicators | 0 | 0 | 39 | 1.000 (n=39) |
|
||||||
|
|
||||||
|
**Per-class precision / recall / F1** (`reports/eval_frontier.md`) — micro-F1 0.958,
|
||||||
|
macro-F1 0.926:
|
||||||
|
|
||||||
|
| Verdict | Precision | Recall | F1 | Support |
|
||||||
|
| --- | --- | --- | --- | --- |
|
||||||
|
| scam_likely | 1.000 | 1.000 | 1.000 | 63 |
|
||||||
|
| suspicious | 1.000 | 0.722 | 0.839 | 18 |
|
||||||
|
| no_indicators | 0.886 | 1.000 | 0.940 | 39 |
|
||||||
|
|
||||||
|
**Where the model is weak, and where it is safe.** The weak class is `suspicious`
|
||||||
|
recall (0.722, 13/18) — the honest middle is the hardest to catch — but `suspicious`
|
||||||
|
**precision is 1.000** (it never over-calls the middle). **Crucially, every
|
||||||
|
`suspicious` miss bleeds into `no_indicators` (the safe direction), never into
|
||||||
|
`scam_likely`, and no legit item is ever flipped to `scam_likely`** — which is why
|
||||||
|
the legit-confusable FP-rate is **0.000**. The model under-warns on ambiguous
|
||||||
|
messages rather than over-warning on real ones; for a consumer guard that is the
|
||||||
|
failure mode you want.
|
||||||
|
|
||||||
|
### Size decision
|
||||||
|
|
||||||
|
- **This model (`flowxai/scam-guard-qwen06b`, 0.6B) — smallest and fastest, weaker
|
||||||
|
OOD.** Passes all three release gates on the bench (JSON >99%, evidence >95%,
|
||||||
|
legit-FP <3%) and is a genuinely defensible on-device ship at ~0.4 GB (int4), but
|
||||||
|
drops hardest out-of-distribution (macro-F1 −0.31, 18/20).
|
||||||
|
- **The sibling `flowxai/scam-guard-qwen17b` (1.7B) — the more robust choice.** On
|
||||||
|
fresh messages the extra capacity generalizes materially better (19/20, no macro-F1
|
||||||
|
drop), a gap the in-distribution bench (both ~0.9+) did not surface. Where
|
||||||
|
robustness matters more than size/latency, [ship the
|
||||||
|
1.7B](https://huggingface.co/flowxai/scam-guard-qwen17b).
|
||||||
|
- **Both sizes share the RO family-emergency recall gap** (the one dangerous OOD
|
||||||
|
miss) and both hold legit-FP at 0.000. That shared recall gap is the honest thing
|
||||||
|
to fix before any release claim.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Formats & on-device latency (0.6B)
|
||||||
|
|
||||||
|
**We targeted 150 ms. We measured ~1.1 s (best path). This target is currently not
|
||||||
|
met.** scam-guard emits a full multi-field JSON card (~138 tokens), not a single
|
||||||
|
label, so decode dominates latency.
|
||||||
|
|
||||||
|
The two headline configs for the 0.6B on the representative 300-char SMS (median,
|
||||||
|
M3 Max):
|
||||||
|
|
||||||
|
| Config | median | meets 150 ms? |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| MLX int4, Apple-Silicon GPU (Metal) | ~1.1 s | no |
|
||||||
|
| GGUF int8 (`Q8_0`), CPU-only (`-ngl 0`) | ~3.3 s | no |
|
||||||
|
|
||||||
|
int4 GGUF (`Q4_K_M`) is a poor trade on this small model — on CPU it is *slower*
|
||||||
|
than int8 and degrades quality (a spot-check sample became invalid JSON) — so
|
||||||
|
**`Q8_0` is the recommended GGUF quant**, and MLX int4 is the fastest quality-holding
|
||||||
|
path.
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary>Full per-quant numbers (0.6B)</summary>
|
||||||
|
|
||||||
|
GGUF (llama.cpp), CPU-only (`-ngl 0`), M3 Max — full JSON card, median over N=12
|
||||||
|
(`reports/benchmark_gguf.json`):
|
||||||
|
|
||||||
|
| quant | file size | median | JSON spot-check |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| int8 (Q8_0) | 639 MB | 3204 ms | 4/4 valid |
|
||||||
|
| int4 (Q4_K_M) | 397 MB | 4000 ms | 3/4 (degraded) |
|
||||||
|
|
||||||
|
MLX-quantized, Apple-Silicon GPU (Metal), M3 Max (`reports/benchmark_mlx_quant.json`):
|
||||||
|
|
||||||
|
| quant | weights size | median | JSON spot-check |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| int4 | 335 MB | 1153 ms | 4/4 valid |
|
||||||
|
| int8 | 633 MB | 1281 ms | 4/4 valid |
|
||||||
|
|
||||||
|
bf16 MLX-Metal latency/memory and per-input-length detail are in
|
||||||
|
`reports/benchmark.md`. **Core ML** (.mlpackage) was attempted; the LLM→Core ML
|
||||||
|
conversion is finicky (stateful KV-cache handling) and the attempt is documented in
|
||||||
|
`PROGRESS.md` rather than shipped as a fabricated artifact. GGUF and MLX are the
|
||||||
|
recommended on-device paths today.
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
A human decision is needed at the release STOP: accept the ~1–3 s latency, ship the
|
||||||
|
~1.1 s GPU/MLX path, or shrink the output contract to approach 150 ms.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Intended use & limitations
|
||||||
|
|
||||||
|
**Intended use.** A **consumer triage aid** that explains *why* a message looks risky
|
||||||
|
and points you to *your own* trusted channel to verify. Runs on-device. Languages: EN
|
||||||
|
and RO at v1 (Romanian is a first-class citizen, not an afterthought); PL/HU planned
|
||||||
|
fast-follow through the same pipeline.
|
||||||
|
|
||||||
|
**Out of scope & limitations.**
|
||||||
|
|
||||||
|
- **Not a guarantee.** A verdict is a signal, not proof. Scammers adapt continuously;
|
||||||
|
the benchmark is versioned because patterns rotate.
|
||||||
|
- **Verdicts can be wrong in both directions** — a real scam may score
|
||||||
|
`no_indicators` (the OOD RO family-emergency miss is a documented example), and a
|
||||||
|
legitimate message may score `suspicious`. The `suspicious` middle level exists to
|
||||||
|
be honest about uncertainty rather than force a binary.
|
||||||
|
- **Known recall gap:** the RO family-emergency pattern (framing without an explicit
|
||||||
|
money-transfer keyword) can slip past this model (and the 1.7B); it is a confirmed
|
||||||
|
training-data gap, fixable with a data addition before any release claim.
|
||||||
|
- **The model never fetches URLs.** It cannot tell you where a link *actually*
|
||||||
|
resolves, only what the visible string suggests. A lexically-clean URL can still be
|
||||||
|
malicious.
|
||||||
|
- Not a replacement for a bank's fraud line, a national anti-fraud service, or human
|
||||||
|
judgment. The recommended action always routes to *your own* channel.
|
||||||
|
- **Text only** at v1: no image/OCR, no audio, no attachment parsing, no
|
||||||
|
email-header/routing analysis.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Training data
|
||||||
|
|
||||||
|
- **Public seed layer (relabeled):** UCI SMS Spam Collection (CC BY 4.0), enron_ham
|
||||||
|
(SetFit/enron_spam ham slice; no explicit license → reference/research-use),
|
||||||
|
phishing_email (zefang-liu; LGPL-3.0). Relabeled into the verdict+tactic scheme with
|
||||||
|
the evidence kill-switch; per-source human spot-check kept relabel disagreement
|
||||||
|
under the 10% gate.
|
||||||
|
- **Synthetic layer:** generated from specs (tactic × channel × language × register),
|
||||||
|
including a `suspicious` middle-ground tier and adversarial keyword-evasion
|
||||||
|
paraphrases (the `hard` subset). Every bench-destined message passes a sanitizer
|
||||||
|
audit (reserved domains, non-dialable phones).
|
||||||
|
- **Balance:** ≥45% legitimate messages, RO ≥35%, ~40% of RO diacritic-free,
|
||||||
|
`suspicious` ~14.5% of the SFT train set.
|
||||||
|
- Split discipline: synthetic by spec-family, public by source-text hash; paraphrases
|
||||||
|
follow their parent's split. **Train ∩ ScamGuardBench = ∅** (contamination-verified).
|
||||||
|
|
||||||
|
Both sizes trained from the **same data**; the size decision is evidence-based (above).
|
||||||
|
|
||||||
|
## Fine-tuning
|
||||||
|
|
||||||
|
**CUDA full-budget 3-epoch LoRA** (`training/train_cuda.py`, transformers + PEFT +
|
||||||
|
trl `SFTTrainer`), from `Qwen/Qwen3-0.6B`. LoRA rank 16 / alpha 32 / dropout 0.05,
|
||||||
|
all-linear modules, adamw, cosine 2e-5→2e-6 with 60-step warmup, effective batch 4,
|
||||||
|
seq 1280, bf16, gradient-checkpointing, seed 20260703; **3 real epochs** on an
|
||||||
|
on-demand NVIDIA L4 (~2h16m, 0 OOM), thinking disabled. This **supersedes** the MLX
|
||||||
|
first pass (~1 epoch, bounded by Metal stalls); the full budget bought +2.3 macro-F1
|
||||||
|
points in-distribution.
|
||||||
|
|
||||||
|
> **Base-model note.** The exact repo `Qwen3-0.6B-Instruct` does **not** exist on
|
||||||
|
> Hugging Face — Qwen3 merged instruct+thinking into the single base repo
|
||||||
|
> `Qwen/Qwen3-0.6B` (instruction-capable, Apache-2.0). We fine-tune it with thinking
|
||||||
|
> disabled.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Dual-use statement
|
||||||
|
|
||||||
|
We release a **detector**, a **benchmark**, and **pattern-level** explanations. We do
|
||||||
|
**not** release the scam-variant generation prompts as a standalone tool. Public
|
||||||
|
benchmark scam texts carry no real dialable phone numbers and no working URLs
|
||||||
|
(reserved/example domains and clearly-fake numbers only). Output explanations describe
|
||||||
|
the manipulation *pattern*, never instructions for constructing one.
|
||||||
|
|
||||||
|
## Links
|
||||||
|
|
||||||
|
- **Sibling model (quality pick, 1.7B):** [`flowxai/scam-guard-qwen17b`](https://huggingface.co/flowxai/scam-guard-qwen17b) — higher OOD accuracy, larger footprint (~1.1 GB int4).
|
||||||
|
- **Benchmark / dataset:** [`flowxai/scamguardbench`](https://huggingface.co/datasets/flowxai/scamguardbench).
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
Apache-2.0 (weights, code, and benchmark). Base model `Qwen/Qwen3-0.6B` is Apache-2.0.
|
||||||
|
</content>
|
||||||
|
</invoke>
|
||||||
28
added_tokens.json
Normal file
28
added_tokens.json
Normal file
@@ -0,0 +1,28 @@
|
|||||||
|
{
|
||||||
|
"</think>": 151668,
|
||||||
|
"</tool_call>": 151658,
|
||||||
|
"</tool_response>": 151666,
|
||||||
|
"<think>": 151667,
|
||||||
|
"<tool_call>": 151657,
|
||||||
|
"<tool_response>": 151665,
|
||||||
|
"<|box_end|>": 151649,
|
||||||
|
"<|box_start|>": 151648,
|
||||||
|
"<|endoftext|>": 151643,
|
||||||
|
"<|file_sep|>": 151664,
|
||||||
|
"<|fim_middle|>": 151660,
|
||||||
|
"<|fim_pad|>": 151662,
|
||||||
|
"<|fim_prefix|>": 151659,
|
||||||
|
"<|fim_suffix|>": 151661,
|
||||||
|
"<|im_end|>": 151645,
|
||||||
|
"<|im_start|>": 151644,
|
||||||
|
"<|image_pad|>": 151655,
|
||||||
|
"<|object_ref_end|>": 151647,
|
||||||
|
"<|object_ref_start|>": 151646,
|
||||||
|
"<|quad_end|>": 151651,
|
||||||
|
"<|quad_start|>": 151650,
|
||||||
|
"<|repo_name|>": 151663,
|
||||||
|
"<|video_pad|>": 151656,
|
||||||
|
"<|vision_end|>": 151653,
|
||||||
|
"<|vision_pad|>": 151654,
|
||||||
|
"<|vision_start|>": 151652
|
||||||
|
}
|
||||||
89
chat_template.jinja
Normal file
89
chat_template.jinja
Normal file
@@ -0,0 +1,89 @@
|
|||||||
|
{%- if tools %}
|
||||||
|
{{- '<|im_start|>system\n' }}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- messages[0].content + '\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||||
|
{%- for tool in tools %}
|
||||||
|
{{- "\n" }}
|
||||||
|
{{- tool | tojson }}
|
||||||
|
{%- endfor %}
|
||||||
|
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||||
|
{%- else %}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||||
|
{%- for message in messages[::-1] %}
|
||||||
|
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||||
|
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||||
|
{%- set ns.multi_step_tool = false %}
|
||||||
|
{%- set ns.last_query_index = index %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- for message in messages %}
|
||||||
|
{%- if message.content is string %}
|
||||||
|
{%- set content = message.content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- set content = '' %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
||||||
|
{%- elif message.role == "assistant" %}
|
||||||
|
{%- set reasoning_content = '' %}
|
||||||
|
{%- if message.reasoning_content is string %}
|
||||||
|
{%- set reasoning_content = message.reasoning_content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- if '</think>' in content %}
|
||||||
|
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||||
|
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if loop.index0 > ns.last_query_index %}
|
||||||
|
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if message.tool_calls %}
|
||||||
|
{%- for tool_call in message.tool_calls %}
|
||||||
|
{%- if (loop.first and content) or (not loop.first) %}
|
||||||
|
{{- '\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if tool_call.function %}
|
||||||
|
{%- set tool_call = tool_call.function %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<tool_call>\n{"name": "' }}
|
||||||
|
{{- tool_call.name }}
|
||||||
|
{{- '", "arguments": ' }}
|
||||||
|
{%- if tool_call.arguments is string %}
|
||||||
|
{{- tool_call.arguments }}
|
||||||
|
{%- else %}
|
||||||
|
{{- tool_call.arguments | tojson }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '}\n</tool_call>' }}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- elif message.role == "tool" %}
|
||||||
|
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||||
|
{{- '<|im_start|>user' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '\n<tool_response>\n' }}
|
||||||
|
{{- content }}
|
||||||
|
{{- '\n</tool_response>' }}
|
||||||
|
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- if add_generation_prompt %}
|
||||||
|
{{- '<|im_start|>assistant\n' }}
|
||||||
|
{%- if enable_thinking is defined and enable_thinking is false %}
|
||||||
|
{{- '<think>\n\n</think>\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
60
config.json
Normal file
60
config.json
Normal file
@@ -0,0 +1,60 @@
|
|||||||
|
{
|
||||||
|
"architectures": [
|
||||||
|
"Qwen3ForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_bias": false,
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"bos_token_id": 151643,
|
||||||
|
"eos_token_id": 151645,
|
||||||
|
"head_dim": 128,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 1024,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 3072,
|
||||||
|
"layer_types": [
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention"
|
||||||
|
],
|
||||||
|
"max_position_embeddings": 40960,
|
||||||
|
"max_window_layers": 28,
|
||||||
|
"model_type": "qwen3",
|
||||||
|
"num_attention_heads": 16,
|
||||||
|
"num_hidden_layers": 28,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"rms_norm_eps": 1e-06,
|
||||||
|
"rope_scaling": null,
|
||||||
|
"rope_theta": 1000000,
|
||||||
|
"sliding_window": null,
|
||||||
|
"tie_word_embeddings": true,
|
||||||
|
"torch_dtype": "bfloat16",
|
||||||
|
"transformers_version": "4.55.2",
|
||||||
|
"use_cache": true,
|
||||||
|
"use_sliding_window": false,
|
||||||
|
"vocab_size": 151936
|
||||||
|
}
|
||||||
13
generation_config.json
Normal file
13
generation_config.json
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
{
|
||||||
|
"bos_token_id": 151643,
|
||||||
|
"do_sample": true,
|
||||||
|
"eos_token_id": [
|
||||||
|
151645,
|
||||||
|
151643
|
||||||
|
],
|
||||||
|
"pad_token_id": 151643,
|
||||||
|
"temperature": 0.6,
|
||||||
|
"top_k": 20,
|
||||||
|
"top_p": 0.95,
|
||||||
|
"transformers_version": "4.55.2"
|
||||||
|
}
|
||||||
3
gguf/scam-guard-qwen06b-cuda-Q8_0.gguf
Normal file
3
gguf/scam-guard-qwen06b-cuda-Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:3d57376ae9e91c11fe8842373246c5eeac5b0fe9f755a7556f8b0f27214f91f6
|
||||||
|
size 639446752
|
||||||
193
inference_contract/INFERENCE.md
Normal file
193
inference_contract/INFERENCE.md
Normal file
@@ -0,0 +1,193 @@
|
|||||||
|
# scam-guard — Inference Contract (`scamguard_sys_v1`)
|
||||||
|
|
||||||
|
**For the iOS / ScamGuardMLX engine.** This is the frozen inference contract for the
|
||||||
|
published `scam-guard-qwen06b` (and `-qwen17b`) weights. It is the exact spec the
|
||||||
|
model was trained and evaluated against — the thing the M0 spike correctly
|
||||||
|
identified as missing. Wire the engine to *this* and the enum typos and the
|
||||||
|
wrong-verdict-on-the-reference-message go away.
|
||||||
|
|
||||||
|
Three files in this folder:
|
||||||
|
- `INFERENCE.md` — this document
|
||||||
|
- `prompt_scamguard_sys_v1.txt` — the system prompt, verbatim (source of truth)
|
||||||
|
- `schema_scamguard_v1.json` — the output JSON Schema, for constrained decoding
|
||||||
|
|
||||||
|
Great M0 work — feasibility (0.5–2 s) matches the card, and your schema correction
|
||||||
|
is **correct**: the contract is `tactics[] = {tactic, evidence, explanation}`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## TL;DR — what fixes the blocker
|
||||||
|
|
||||||
|
The model is stochastic over an *exact* training-time contract. Three things must
|
||||||
|
match it byte-for-byte; miss any one and you get enum typos / wrong verdicts:
|
||||||
|
|
||||||
|
1. **System prompt** = the frozen `scamguard_sys_v1` string below. Never edit it.
|
||||||
|
2. **User turn** = `[channel: <tag>]\n<message text>` where `<tag>` ∈ `sms | email | chat`.
|
||||||
|
3. **Constrained JSON decoding** against `schema_scamguard_v1.json`. This is what
|
||||||
|
kills the enum typos — the enums are decoded as a closed set, not free text.
|
||||||
|
|
||||||
|
Then a **post-decode evidence check** (verbatim-substring) is part of the contract,
|
||||||
|
not optional.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. System prompt — `scamguard_sys_v1` (verbatim, do not edit)
|
||||||
|
|
||||||
|
Load `prompt_scamguard_sys_v1.txt` as the `system` turn exactly. Bump the version
|
||||||
|
id and re-coordinate with the model team if a single character changes — the model
|
||||||
|
is trained against this string.
|
||||||
|
|
||||||
|
```
|
||||||
|
You are an on-device scam and fraud detector. You read ONE message (SMS, email,
|
||||||
|
or chat text, possibly containing visible URLs) and judge whether it is a scam.
|
||||||
|
You never fetch URLs or use any network; you reason ONLY over the visible text.
|
||||||
|
|
||||||
|
Return a single JSON object with these fields:
|
||||||
|
|
||||||
|
1. verdict — exactly one of three levels (never a probability or percentage):
|
||||||
|
- scam_likely: clear scam mechanism present (a request for money, credentials,
|
||||||
|
card data, OTP relay, a fee to release a parcel/prize, payment redirection,
|
||||||
|
remote access, etc.) driven by manipulation tactics. Reserve this level for
|
||||||
|
messages that would cause real harm if acted on.
|
||||||
|
- suspicious: manipulation signals are present but the message could plausibly
|
||||||
|
be legitimate, OR the only indicators are weak (urgency alone, an unfamiliar
|
||||||
|
link alone, an authority claim with no money/data ask). The honest middle:
|
||||||
|
"verify through your own channel before acting".
|
||||||
|
- no_indicators: no scam mechanism; a genuine-looking OTP, transaction alert,
|
||||||
|
courier notice, receipt, promo, or ordinary message. Legitimate messages can
|
||||||
|
be urgent and can contain links — do NOT flag them for that alone.
|
||||||
|
|
||||||
|
2. tactics — a list of the manipulation tactics you detected, each an object with:
|
||||||
|
- tactic: one of the fixed tactic ids below,
|
||||||
|
- evidence: a VERBATIM substring copied character-for-character from the input
|
||||||
|
message (it will be checked by exact substring match; any tactic whose
|
||||||
|
evidence is not found verbatim in the message is dropped and counted as
|
||||||
|
fabricated, so never paraphrase, truncate mid-word, or invent evidence),
|
||||||
|
- explanation: one calm, plain sentence a non-technical person (including an
|
||||||
|
elderly person) can understand — describe the manipulation pattern, never
|
||||||
|
instructions for constructing it, and never panic language.
|
||||||
|
When verdict is no_indicators, tactics is an empty list.
|
||||||
|
|
||||||
|
Fixed tactic ids (use these exact strings, nothing else):
|
||||||
|
- urgency_pressure
|
||||||
|
- authority_impersonation
|
||||||
|
- payment_redirect
|
||||||
|
- credential_phishing
|
||||||
|
- courier_customs_fee
|
||||||
|
- prize_lottery
|
||||||
|
- investment_too_good
|
||||||
|
- romance_advance_fee
|
||||||
|
- family_emergency_impersonation
|
||||||
|
- tech_support
|
||||||
|
- link_obfuscation
|
||||||
|
- refund_overpayment
|
||||||
|
- subscription_trap
|
||||||
|
|
||||||
|
3. explanation — one or two calm, actionable sentences summarizing the verdict
|
||||||
|
for a frightened non-technical reader. No alarmist tone; state what is going on
|
||||||
|
and, implicitly, that they can verify safely.
|
||||||
|
|
||||||
|
4. recommended_action — one safe action id from the fixed schema enum. Prefer the
|
||||||
|
action that routes the reader to THEIR OWN trusted channel (their bank's
|
||||||
|
official number, the courier's own app) rather than any contact in the message.
|
||||||
|
Use no_action_needed only for no_indicators.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Evidence must be a verbatim substring of the message. This is non-negotiable.
|
||||||
|
- A tactic label means "this manipulation pattern is present", not "this is a
|
||||||
|
scam" — the verdict is a separate judgment. Multiple tactics per message are
|
||||||
|
expected.
|
||||||
|
- Urgency alone, a shortener/unfamiliar link alone, or an authority claim with no
|
||||||
|
money/data ask should rarely exceed suspicious.
|
||||||
|
```
|
||||||
|
|
||||||
|
## 2. User-turn format (exact)
|
||||||
|
|
||||||
|
```
|
||||||
|
[channel: sms]
|
||||||
|
Contul tău a fost blocat. Confirmă datele aici: http://example-bank.invalid/verify
|
||||||
|
```
|
||||||
|
|
||||||
|
- The prefix is literally `[channel: ` + tag + `]` then a newline then the raw
|
||||||
|
message text. Tag is one of `sms`, `email`, `chat` (lowercase).
|
||||||
|
- Apply the model's chat template around system+user as usual (mlx-swift
|
||||||
|
`apply_chat_template` with `add_generation_prompt=true`). No few-shot, no extra
|
||||||
|
preamble — the system prompt is the whole instruction.
|
||||||
|
|
||||||
|
## 3. Constrained JSON decoding (the enum-typo fix)
|
||||||
|
|
||||||
|
The reference implementation decodes with the output constrained to
|
||||||
|
`schema_scamguard_v1.json` (a JSON-schema grammar). Do the same on device — this is
|
||||||
|
non-negotiable for a shippable engine:
|
||||||
|
|
||||||
|
- Minimum bar: constrain the three enum fields to their closed sets — `verdict`,
|
||||||
|
each `tactics[].tactic`, and `recommended_action`. That alone removes the enum
|
||||||
|
typos you saw.
|
||||||
|
- Better: constrain the whole object shape (a GBNF grammar compiled from the JSON
|
||||||
|
schema; llama.cpp/`mlx`-side grammar or a logit mask). `additionalProperties` is
|
||||||
|
false — reject unknown keys.
|
||||||
|
- Free-text fields (`evidence`, `explanation`) stay unconstrained strings.
|
||||||
|
|
||||||
|
If mlx-swift lacks grammar decoding today: implement a logit mask over the enum
|
||||||
|
tokens at minimum, and keep your strict validator as the backstop (re-sample on
|
||||||
|
invalid). But plan for real grammar-constrained decode — the validity gate depends
|
||||||
|
on it.
|
||||||
|
|
||||||
|
## 4. Output contract (schema summary)
|
||||||
|
|
||||||
|
Machine-readable: `schema_scamguard_v1.json`. Shape:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"verdict": "scam_likely | suspicious | no_indicators",
|
||||||
|
"tactics": [
|
||||||
|
{ "tactic": "<one of the 13 ids>", "evidence": "<verbatim substring>", "explanation": "<plain sentence>" }
|
||||||
|
],
|
||||||
|
"explanation": "<overall plain-language explanation>",
|
||||||
|
"recommended_action": "<one of the 10 safe-action ids>"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
- `verdict` (required, enum, 3 values above)
|
||||||
|
- `tactics` (required list; **empty** when `verdict == no_indicators`). Each item is
|
||||||
|
`{tactic, evidence, explanation}` — all three required, `evidence`/`explanation`
|
||||||
|
min length 1.
|
||||||
|
- `explanation` (required, non-empty)
|
||||||
|
- `recommended_action` (required, enum, 10 values):
|
||||||
|
`call_bank_official_number`, `do_not_click_link`, `verify_via_official_app_or_site`,
|
||||||
|
`call_family_member_known_number`, `do_not_share_codes_or_credentials`,
|
||||||
|
`do_not_send_money`, `ignore_and_delete`, `report_to_authorities`,
|
||||||
|
`check_sender_address`, `no_action_needed`.
|
||||||
|
- `additionalProperties: false` at every level.
|
||||||
|
|
||||||
|
The 13 tactic ids are the exact strings in the prompt above (taxonomy v1).
|
||||||
|
|
||||||
|
## 5. Post-decode evidence kill-switch (part of the contract)
|
||||||
|
|
||||||
|
After decode, for each detected tactic, verify `evidence` is a **verbatim
|
||||||
|
substring** of the input message. If it is not found character-for-character, DROP
|
||||||
|
that tactic (do not display it) and count it as fabricated. This mirrors the
|
||||||
|
training/eval verifier (`scamguard.verify`); the model is trained expecting it.
|
||||||
|
Keep your strict validator — just add this substring check to it.
|
||||||
|
|
||||||
|
## 6. The reference Romanian message / int4 note
|
||||||
|
|
||||||
|
With the exact prompt (§1) + user format (§2) + constrained decode (§3), the model
|
||||||
|
card's reference RO message should classify correctly. If it still misfires **after
|
||||||
|
all three are matched**, it's most likely quantization sensitivity — try the **int8**
|
||||||
|
weights instead of int4 for that case and tell us; we'll co-debug (it may be a known
|
||||||
|
int4 edge we document, not an integration bug). Don't treat a single reference-case
|
||||||
|
miss as a contract failure until §1–§3 are byte-exact.
|
||||||
|
|
||||||
|
## 7. Versioning
|
||||||
|
|
||||||
|
`PROMPT_VERSION = scamguard_sys_v1`. The prompt + schema are frozen together and the
|
||||||
|
weights are trained against them. Any change = a new version id + a retrain; never
|
||||||
|
silently edit the prompt on the client. Pin the version string in the app and log it
|
||||||
|
with each verdict so a future model swap is traceable.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
*Delivered by the scam-guard model team to unblock the M0 spike. Questions on the
|
||||||
|
prompt/schema/decode → model team; this contract is the source of truth over any
|
||||||
|
value inferred from the PRD.*
|
||||||
61
inference_contract/prompt_scamguard_sys_v1.txt
Normal file
61
inference_contract/prompt_scamguard_sys_v1.txt
Normal file
@@ -0,0 +1,61 @@
|
|||||||
|
You are an on-device scam and fraud detector. You read ONE message (SMS, email,
|
||||||
|
or chat text, possibly containing visible URLs) and judge whether it is a scam.
|
||||||
|
You never fetch URLs or use any network; you reason ONLY over the visible text.
|
||||||
|
|
||||||
|
Return a single JSON object with these fields:
|
||||||
|
|
||||||
|
1. verdict — exactly one of three levels (never a probability or percentage):
|
||||||
|
- scam_likely: clear scam mechanism present (a request for money, credentials,
|
||||||
|
card data, OTP relay, a fee to release a parcel/prize, payment redirection,
|
||||||
|
remote access, etc.) driven by manipulation tactics. Reserve this level for
|
||||||
|
messages that would cause real harm if acted on.
|
||||||
|
- suspicious: manipulation signals are present but the message could plausibly
|
||||||
|
be legitimate, OR the only indicators are weak (urgency alone, an unfamiliar
|
||||||
|
link alone, an authority claim with no money/data ask). The honest middle:
|
||||||
|
"verify through your own channel before acting".
|
||||||
|
- no_indicators: no scam mechanism; a genuine-looking OTP, transaction alert,
|
||||||
|
courier notice, receipt, promo, or ordinary message. Legitimate messages can
|
||||||
|
be urgent and can contain links — do NOT flag them for that alone.
|
||||||
|
|
||||||
|
2. tactics — a list of the manipulation tactics you detected, each an object with:
|
||||||
|
- tactic: one of the fixed tactic ids below,
|
||||||
|
- evidence: a VERBATIM substring copied character-for-character from the input
|
||||||
|
message (it will be checked by exact substring match; any tactic whose
|
||||||
|
evidence is not found verbatim in the message is dropped and counted as
|
||||||
|
fabricated, so never paraphrase, truncate mid-word, or invent evidence),
|
||||||
|
- explanation: one calm, plain sentence a non-technical person (including an
|
||||||
|
elderly person) can understand — describe the manipulation pattern, never
|
||||||
|
instructions for constructing it, and never panic language.
|
||||||
|
When verdict is no_indicators, tactics is an empty list.
|
||||||
|
|
||||||
|
Fixed tactic ids (use these exact strings, nothing else):
|
||||||
|
- urgency_pressure
|
||||||
|
- authority_impersonation
|
||||||
|
- payment_redirect
|
||||||
|
- credential_phishing
|
||||||
|
- courier_customs_fee
|
||||||
|
- prize_lottery
|
||||||
|
- investment_too_good
|
||||||
|
- romance_advance_fee
|
||||||
|
- family_emergency_impersonation
|
||||||
|
- tech_support
|
||||||
|
- link_obfuscation
|
||||||
|
- refund_overpayment
|
||||||
|
- subscription_trap
|
||||||
|
|
||||||
|
3. explanation — one or two calm, actionable sentences summarizing the verdict
|
||||||
|
for a frightened non-technical reader. No alarmist tone; state what is going on
|
||||||
|
and, implicitly, that they can verify safely.
|
||||||
|
|
||||||
|
4. recommended_action — one safe action id from the fixed schema enum. Prefer the
|
||||||
|
action that routes the reader to THEIR OWN trusted channel (their bank's
|
||||||
|
official number, the courier's own app) rather than any contact in the message.
|
||||||
|
Use no_action_needed only for no_indicators.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Evidence must be a verbatim substring of the message. This is non-negotiable.
|
||||||
|
- A tactic label means "this manipulation pattern is present", not "this is a
|
||||||
|
scam" — the verdict is a separate judgment. Multiple tactics per message are
|
||||||
|
expected.
|
||||||
|
- Urgency alone, a shortener/unfamiliar link alone, or an authority claim with no
|
||||||
|
money/data ask should rarely exceed suspicious.
|
||||||
110
inference_contract/schema_scamguard_v1.json
Normal file
110
inference_contract/schema_scamguard_v1.json
Normal file
@@ -0,0 +1,110 @@
|
|||||||
|
{
|
||||||
|
"$defs": {
|
||||||
|
"DetectedTactic": {
|
||||||
|
"additionalProperties": false,
|
||||||
|
"description": "One detected manipulation tactic with its verbatim evidence span.",
|
||||||
|
"properties": {
|
||||||
|
"tactic": {
|
||||||
|
"$ref": "#/$defs/TacticId"
|
||||||
|
},
|
||||||
|
"evidence": {
|
||||||
|
"description": "Verbatim substring of the input message that triggered this tactic.",
|
||||||
|
"minLength": 1,
|
||||||
|
"title": "Evidence",
|
||||||
|
"type": "string"
|
||||||
|
},
|
||||||
|
"explanation": {
|
||||||
|
"description": "Calm, plain-language sentence(s) a non-technical person can act on.",
|
||||||
|
"minLength": 1,
|
||||||
|
"title": "Explanation",
|
||||||
|
"type": "string"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"required": [
|
||||||
|
"tactic",
|
||||||
|
"evidence",
|
||||||
|
"explanation"
|
||||||
|
],
|
||||||
|
"title": "DetectedTactic",
|
||||||
|
"type": "object"
|
||||||
|
},
|
||||||
|
"SafeAction": {
|
||||||
|
"description": "Fixed list of recommended safe actions.\n\nThe model must pick one of these; it can never compose free-text advice\nthat points back at the scammer's own channel. End-user wording lives in\nthe demo/report layer; these are stable machine-readable ids.",
|
||||||
|
"enum": [
|
||||||
|
"call_bank_official_number",
|
||||||
|
"do_not_click_link",
|
||||||
|
"verify_via_official_app_or_site",
|
||||||
|
"call_family_member_known_number",
|
||||||
|
"do_not_share_codes_or_credentials",
|
||||||
|
"do_not_send_money",
|
||||||
|
"ignore_and_delete",
|
||||||
|
"report_to_authorities",
|
||||||
|
"check_sender_address",
|
||||||
|
"no_action_needed"
|
||||||
|
],
|
||||||
|
"title": "SafeAction",
|
||||||
|
"type": "string"
|
||||||
|
},
|
||||||
|
"TacticId": {
|
||||||
|
"description": "Fixed tactic taxonomy ids (taxonomy.yaml v1).",
|
||||||
|
"enum": [
|
||||||
|
"urgency_pressure",
|
||||||
|
"authority_impersonation",
|
||||||
|
"payment_redirect",
|
||||||
|
"credential_phishing",
|
||||||
|
"courier_customs_fee",
|
||||||
|
"prize_lottery",
|
||||||
|
"investment_too_good",
|
||||||
|
"romance_advance_fee",
|
||||||
|
"family_emergency_impersonation",
|
||||||
|
"tech_support",
|
||||||
|
"link_obfuscation",
|
||||||
|
"refund_overpayment",
|
||||||
|
"subscription_trap"
|
||||||
|
],
|
||||||
|
"title": "TacticId",
|
||||||
|
"type": "string"
|
||||||
|
},
|
||||||
|
"Verdict": {
|
||||||
|
"description": "Three-level verdict. Deliberately not a probability (CLAUDE.md section 2).",
|
||||||
|
"enum": [
|
||||||
|
"scam_likely",
|
||||||
|
"suspicious",
|
||||||
|
"no_indicators"
|
||||||
|
],
|
||||||
|
"title": "Verdict",
|
||||||
|
"type": "string"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"additionalProperties": false,
|
||||||
|
"description": "Complete model output for one input message.",
|
||||||
|
"properties": {
|
||||||
|
"verdict": {
|
||||||
|
"$ref": "#/$defs/Verdict"
|
||||||
|
},
|
||||||
|
"tactics": {
|
||||||
|
"description": "Detected tactics; empty when verdict is no_indicators.",
|
||||||
|
"items": {
|
||||||
|
"$ref": "#/$defs/DetectedTactic"
|
||||||
|
},
|
||||||
|
"title": "Tactics",
|
||||||
|
"type": "array"
|
||||||
|
},
|
||||||
|
"explanation": {
|
||||||
|
"description": "Overall plain-language explanation of the verdict.",
|
||||||
|
"minLength": 1,
|
||||||
|
"title": "Explanation",
|
||||||
|
"type": "string"
|
||||||
|
},
|
||||||
|
"recommended_action": {
|
||||||
|
"$ref": "#/$defs/SafeAction"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"required": [
|
||||||
|
"verdict",
|
||||||
|
"explanation",
|
||||||
|
"recommended_action"
|
||||||
|
],
|
||||||
|
"title": "ScamGuardOutput",
|
||||||
|
"type": "object"
|
||||||
|
}
|
||||||
151388
merges.txt
Normal file
151388
merges.txt
Normal file
File diff suppressed because it is too large
Load Diff
7
mlx-int4/README.md
Normal file
7
mlx-int4/README.md
Normal file
@@ -0,0 +1,7 @@
|
|||||||
|
---
|
||||||
|
language: en
|
||||||
|
pipeline_tag: text-generation
|
||||||
|
library_name: mlx
|
||||||
|
tags:
|
||||||
|
- mlx
|
||||||
|
---
|
||||||
89
mlx-int4/chat_template.jinja
Normal file
89
mlx-int4/chat_template.jinja
Normal file
@@ -0,0 +1,89 @@
|
|||||||
|
{%- if tools %}
|
||||||
|
{{- '<|im_start|>system\n' }}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- messages[0].content + '\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||||
|
{%- for tool in tools %}
|
||||||
|
{{- "\n" }}
|
||||||
|
{{- tool | tojson }}
|
||||||
|
{%- endfor %}
|
||||||
|
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||||
|
{%- else %}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||||
|
{%- for message in messages[::-1] %}
|
||||||
|
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||||
|
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||||
|
{%- set ns.multi_step_tool = false %}
|
||||||
|
{%- set ns.last_query_index = index %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- for message in messages %}
|
||||||
|
{%- if message.content is string %}
|
||||||
|
{%- set content = message.content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- set content = '' %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
||||||
|
{%- elif message.role == "assistant" %}
|
||||||
|
{%- set reasoning_content = '' %}
|
||||||
|
{%- if message.reasoning_content is string %}
|
||||||
|
{%- set reasoning_content = message.reasoning_content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- if '</think>' in content %}
|
||||||
|
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||||
|
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if loop.index0 > ns.last_query_index %}
|
||||||
|
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if message.tool_calls %}
|
||||||
|
{%- for tool_call in message.tool_calls %}
|
||||||
|
{%- if (loop.first and content) or (not loop.first) %}
|
||||||
|
{{- '\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if tool_call.function %}
|
||||||
|
{%- set tool_call = tool_call.function %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<tool_call>\n{"name": "' }}
|
||||||
|
{{- tool_call.name }}
|
||||||
|
{{- '", "arguments": ' }}
|
||||||
|
{%- if tool_call.arguments is string %}
|
||||||
|
{{- tool_call.arguments }}
|
||||||
|
{%- else %}
|
||||||
|
{{- tool_call.arguments | tojson }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '}\n</tool_call>' }}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- elif message.role == "tool" %}
|
||||||
|
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||||
|
{{- '<|im_start|>user' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '\n<tool_response>\n' }}
|
||||||
|
{{- content }}
|
||||||
|
{{- '\n</tool_response>' }}
|
||||||
|
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- if add_generation_prompt %}
|
||||||
|
{{- '<|im_start|>assistant\n' }}
|
||||||
|
{%- if enable_thinking is defined and enable_thinking is false %}
|
||||||
|
{{- '<think>\n\n</think>\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
73
mlx-int4/config.json
Normal file
73
mlx-int4/config.json
Normal file
@@ -0,0 +1,73 @@
|
|||||||
|
{
|
||||||
|
"architectures": [
|
||||||
|
"Qwen3ForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_bias": false,
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"bos_token_id": 151643,
|
||||||
|
"eos_token_id": [
|
||||||
|
151645,
|
||||||
|
151643
|
||||||
|
],
|
||||||
|
"head_dim": 128,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 1024,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 3072,
|
||||||
|
"layer_types": [
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention"
|
||||||
|
],
|
||||||
|
"max_position_embeddings": 40960,
|
||||||
|
"max_window_layers": 28,
|
||||||
|
"model_type": "qwen3",
|
||||||
|
"num_attention_heads": 16,
|
||||||
|
"num_hidden_layers": 28,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"quantization": {
|
||||||
|
"group_size": 64,
|
||||||
|
"bits": 4,
|
||||||
|
"mode": "affine"
|
||||||
|
},
|
||||||
|
"quantization_config": {
|
||||||
|
"group_size": 64,
|
||||||
|
"bits": 4,
|
||||||
|
"mode": "affine"
|
||||||
|
},
|
||||||
|
"rms_norm_eps": 1e-06,
|
||||||
|
"rope_scaling": null,
|
||||||
|
"rope_theta": 1000000,
|
||||||
|
"sliding_window": null,
|
||||||
|
"tie_word_embeddings": true,
|
||||||
|
"torch_dtype": "bfloat16",
|
||||||
|
"transformers_version": "4.55.2",
|
||||||
|
"use_cache": true,
|
||||||
|
"use_sliding_window": false,
|
||||||
|
"vocab_size": 151936
|
||||||
|
}
|
||||||
13
mlx-int4/generation_config.json
Normal file
13
mlx-int4/generation_config.json
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
{
|
||||||
|
"bos_token_id": 151643,
|
||||||
|
"do_sample": true,
|
||||||
|
"eos_token_id": [
|
||||||
|
151645,
|
||||||
|
151643
|
||||||
|
],
|
||||||
|
"pad_token_id": 151643,
|
||||||
|
"temperature": 0.6,
|
||||||
|
"top_k": 20,
|
||||||
|
"top_p": 0.95,
|
||||||
|
"transformers_version": "4.55.2"
|
||||||
|
}
|
||||||
3
mlx-int4/model.safetensors
Normal file
3
mlx-int4/model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:29977d593ff8ff1a4e3e9b7fca0458af23929b6e16e20e2b0df1956ce4a9919a
|
||||||
|
size 335450548
|
||||||
712
mlx-int4/model.safetensors.index.json
Normal file
712
mlx-int4/model.safetensors.index.json
Normal file
@@ -0,0 +1,712 @@
|
|||||||
|
{
|
||||||
|
"metadata": {
|
||||||
|
"total_size": 335372288,
|
||||||
|
"total_parameters": 596049920
|
||||||
|
},
|
||||||
|
"weight_map": {
|
||||||
|
"model.embed_tokens.biases": "model.safetensors",
|
||||||
|
"model.embed_tokens.scales": "model.safetensors",
|
||||||
|
"model.embed_tokens.weight": "model.safetensors",
|
||||||
|
"model.layers.0.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.0.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.0.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.1.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.10.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.11.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.12.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.13.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.14.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.15.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.16.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.17.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.18.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.19.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.2.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.20.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.21.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.22.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.23.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.24.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.25.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.26.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.27.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.3.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.4.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.5.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.6.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.7.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.8.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.input_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.down_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.down_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.down_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.gate_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.gate_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.gate_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.up_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.up_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.mlp.up_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.post_attention_layernorm.weight": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.k_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.k_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.k_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.k_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.o_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.o_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.o_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.q_norm.weight": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.q_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.q_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.q_proj.weight": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.v_proj.biases": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.v_proj.scales": "model.safetensors",
|
||||||
|
"model.layers.9.self_attn.v_proj.weight": "model.safetensors",
|
||||||
|
"model.norm.weight": "model.safetensors"
|
||||||
|
}
|
||||||
|
}
|
||||||
3
mlx-int4/tokenizer.json
Normal file
3
mlx-int4/tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
|
||||||
|
size 11422650
|
||||||
16
mlx-int4/tokenizer_config.json
Normal file
16
mlx-int4/tokenizer_config.json
Normal file
@@ -0,0 +1,16 @@
|
|||||||
|
{
|
||||||
|
"add_prefix_space": false,
|
||||||
|
"backend": "tokenizers",
|
||||||
|
"bos_token": null,
|
||||||
|
"clean_up_tokenization_spaces": false,
|
||||||
|
"eos_token": "<|im_end|>",
|
||||||
|
"errors": "replace",
|
||||||
|
"is_local": true,
|
||||||
|
"local_files_only": false,
|
||||||
|
"model_max_length": 131072,
|
||||||
|
"pad_token": "<|endoftext|>",
|
||||||
|
"split_special_tokens": false,
|
||||||
|
"tokenizer_class": "Qwen2Tokenizer",
|
||||||
|
"tool_parser_type": "json_tools",
|
||||||
|
"unk_token": null
|
||||||
|
}
|
||||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:a76a16ce7a908f8e1a040c89803a6924e78fd86252734e6e523c47ed3e2d26b3
|
||||||
|
size 1192135096
|
||||||
31
special_tokens_map.json
Normal file
31
special_tokens_map.json
Normal file
@@ -0,0 +1,31 @@
|
|||||||
|
{
|
||||||
|
"additional_special_tokens": [
|
||||||
|
"<|im_start|>",
|
||||||
|
"<|im_end|>",
|
||||||
|
"<|object_ref_start|>",
|
||||||
|
"<|object_ref_end|>",
|
||||||
|
"<|box_start|>",
|
||||||
|
"<|box_end|>",
|
||||||
|
"<|quad_start|>",
|
||||||
|
"<|quad_end|>",
|
||||||
|
"<|vision_start|>",
|
||||||
|
"<|vision_end|>",
|
||||||
|
"<|vision_pad|>",
|
||||||
|
"<|image_pad|>",
|
||||||
|
"<|video_pad|>"
|
||||||
|
],
|
||||||
|
"eos_token": {
|
||||||
|
"content": "<|im_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"pad_token": {
|
||||||
|
"content": "<|endoftext|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
}
|
||||||
|
}
|
||||||
BIN
tokenizer.json
(Stored with Git LFS)
Normal file
BIN
tokenizer.json
(Stored with Git LFS)
Normal file
Binary file not shown.
239
tokenizer_config.json
Normal file
239
tokenizer_config.json
Normal file
@@ -0,0 +1,239 @@
|
|||||||
|
{
|
||||||
|
"add_bos_token": false,
|
||||||
|
"add_prefix_space": false,
|
||||||
|
"added_tokens_decoder": {
|
||||||
|
"151643": {
|
||||||
|
"content": "<|endoftext|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151644": {
|
||||||
|
"content": "<|im_start|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151645": {
|
||||||
|
"content": "<|im_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151646": {
|
||||||
|
"content": "<|object_ref_start|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151647": {
|
||||||
|
"content": "<|object_ref_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151648": {
|
||||||
|
"content": "<|box_start|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151649": {
|
||||||
|
"content": "<|box_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151650": {
|
||||||
|
"content": "<|quad_start|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151651": {
|
||||||
|
"content": "<|quad_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151652": {
|
||||||
|
"content": "<|vision_start|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151653": {
|
||||||
|
"content": "<|vision_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151654": {
|
||||||
|
"content": "<|vision_pad|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151655": {
|
||||||
|
"content": "<|image_pad|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151656": {
|
||||||
|
"content": "<|video_pad|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"151657": {
|
||||||
|
"content": "<tool_call>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151658": {
|
||||||
|
"content": "</tool_call>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151659": {
|
||||||
|
"content": "<|fim_prefix|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151660": {
|
||||||
|
"content": "<|fim_middle|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151661": {
|
||||||
|
"content": "<|fim_suffix|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151662": {
|
||||||
|
"content": "<|fim_pad|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151663": {
|
||||||
|
"content": "<|repo_name|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151664": {
|
||||||
|
"content": "<|file_sep|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151665": {
|
||||||
|
"content": "<tool_response>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151666": {
|
||||||
|
"content": "</tool_response>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151667": {
|
||||||
|
"content": "<think>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
},
|
||||||
|
"151668": {
|
||||||
|
"content": "</think>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": false
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"additional_special_tokens": [
|
||||||
|
"<|im_start|>",
|
||||||
|
"<|im_end|>",
|
||||||
|
"<|object_ref_start|>",
|
||||||
|
"<|object_ref_end|>",
|
||||||
|
"<|box_start|>",
|
||||||
|
"<|box_end|>",
|
||||||
|
"<|quad_start|>",
|
||||||
|
"<|quad_end|>",
|
||||||
|
"<|vision_start|>",
|
||||||
|
"<|vision_end|>",
|
||||||
|
"<|vision_pad|>",
|
||||||
|
"<|image_pad|>",
|
||||||
|
"<|video_pad|>"
|
||||||
|
],
|
||||||
|
"bos_token": null,
|
||||||
|
"clean_up_tokenization_spaces": false,
|
||||||
|
"eos_token": "<|im_end|>",
|
||||||
|
"errors": "replace",
|
||||||
|
"extra_special_tokens": {},
|
||||||
|
"model_max_length": 131072,
|
||||||
|
"pad_token": "<|endoftext|>",
|
||||||
|
"split_special_tokens": false,
|
||||||
|
"tokenizer_class": "Qwen2Tokenizer",
|
||||||
|
"unk_token": null
|
||||||
|
}
|
||||||
1
vocab.json
Normal file
1
vocab.json
Normal file
File diff suppressed because one or more lines are too long
Reference in New Issue
Block a user