commit 8fe7d6c972b7bdcd73519fca635c536267739cd0 Author: ModelHub XC Date: Thu Sep 10 11:46:16 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: Zynerji/Ektome-Qwen3-4Bi-2507-PristinelyUncensored Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..9322b8c --- /dev/null +++ b/.gitattributes @@ -0,0 +1,44 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +hero.png filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-4Bi-2507-IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-4Bi-2507-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-4Bi-2507-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-4Bi-2507-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-4Bi-2507-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-4Bi-2507-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +imatrix.dat filter=lfs diff=lfs merge=lfs -text diff --git a/Ektome-Qwen3-4Bi-2507-IQ3_M.gguf b/Ektome-Qwen3-4Bi-2507-IQ3_M.gguf new file mode 100644 index 0000000..6f41cfa --- /dev/null +++ b/Ektome-Qwen3-4Bi-2507-IQ3_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3a2ce10b7555f2ea4ca8aa9ed0d969350bd2baa567ff135046363a2f061c6b04 +size 1962894464 diff --git a/Ektome-Qwen3-4Bi-2507-IQ4_XS.gguf b/Ektome-Qwen3-4Bi-2507-IQ4_XS.gguf new file mode 100644 index 0000000..928776f --- /dev/null +++ b/Ektome-Qwen3-4Bi-2507-IQ4_XS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:08ad88b3cc56ac4e56e7e592f45b4932ebee319af4586ec5742f234079d4c389 +size 2270749824 diff --git a/Ektome-Qwen3-4Bi-2507-Q4_K_M.gguf b/Ektome-Qwen3-4Bi-2507-Q4_K_M.gguf new file mode 100644 index 0000000..b717c69 --- /dev/null +++ b/Ektome-Qwen3-4Bi-2507-Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c285a00f5eac8e33bb31a666c70d3ee6a2917784c22e959e0c852c83fc820079 +size 2497279104 diff --git a/Ektome-Qwen3-4Bi-2507-Q5_K_M.gguf b/Ektome-Qwen3-4Bi-2507-Q5_K_M.gguf new file mode 100644 index 0000000..04952a3 --- /dev/null +++ b/Ektome-Qwen3-4Bi-2507-Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9bdd386da5a37a8f3d24bd17c8f37c7dd68abdcbd0ec9c6a0e935ff89e731b85 +size 2889512064 diff --git a/Ektome-Qwen3-4Bi-2507-Q6_K.gguf b/Ektome-Qwen3-4Bi-2507-Q6_K.gguf new file mode 100644 index 0000000..b6acb53 --- /dev/null +++ b/Ektome-Qwen3-4Bi-2507-Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:101cede7d3a2148af1e0a5e6e9095bbb7ab94b7a5ea3755f1cd3ee787e7f9595 +size 3306259584 diff --git a/Ektome-Qwen3-4Bi-2507-Q8_0.gguf b/Ektome-Qwen3-4Bi-2507-Q8_0.gguf new file mode 100644 index 0000000..6de8997 --- /dev/null +++ b/Ektome-Qwen3-4Bi-2507-Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e77f2ef158e2859b8329f0981754622c5fc6f93a4889633c970aba1b0dc24b98 +size 4280403584 diff --git a/README.md b/README.md new file mode 100644 index 0000000..3cc97f8 --- /dev/null +++ b/README.md @@ -0,0 +1,138 @@ +--- +license: apache-2.0 +base_model: Qwen/Qwen3-4B-Instruct-2507 +tags: + - uncensored + - abliterated + - certificate-failed + - ektome + - sphragis + - qwen3 +language: + - en +pipeline_tag: text-generation +--- + +![Ektome-Qwen3-4Bi-2507-PristinelyUncensored](./hero.png) + +# Ektome-Qwen3-4Bi-2507-PristinelyUncensored + +**Uncensored — and it did NOT pass its capability certificate. Read the certificate before using this model.** + +> **compliance 0.00 to 1.00 at capability -0.005 vs pristine.** + +$$\colorbox{black}{$\color{white} +\begin{array}{ll} +\textsf{EKTOME CERTIFICATE} & {} \\ +\textsf{capability} & \textsf{FAIL} \\ +\textsf{margin} & 3\% \\ +\textsf{items } n & 2800 \\ +\textsf{worst-axis bound} & +0.025 \\ +\textsf{compliance} & 0.00 \rightarrow 1.00 \\ +\end{array}$}$$ + +> ### ⚠️ This model failed its capability certificate +> +> A paired non-inferiority test against the pristine model at n=2800 found a **real capability loss** on: +> +> - **arithmetic**: pristine 0.879 → this model 0.869 (bound on the drop +0.016, exceeds the 3% margin) +> - **instruction**: pristine 0.860 → this model 0.847 (bound on the drop +0.022, exceeds the 3% margin) +> +> It is published for transparency and for uses where the affected axis +> does not matter. **Do not treat it as capability-preserving.** + + +📄 **[Read the whitepaper (PDF)](./whitepaper.pdf)** — full method, receipts and certification. +The PDF is the authoritative document: dark-typeset, with the complete derivation, the +per-axis certificate and the reproducibility hashes. + +--- + +## Why this exists + +Standard abliteration removes a coarse *refusal direction* that is entangled with +directions carrying knowledge and reasoning. The result is an uncensored model with a +capability tax that is **almost never measured**. + +Ektomē (ἐκτομή, *excision*) isolates and removes only the refusal-**specific** +component, leaving general helpfulness intact, and does so norm-preservingly on the +pristine model — no training, no distillation, no damage to repair. The extraction +depth is selected per model by automated search against measured compliance. + +The estimator, excision operator and depth-selection procedure are proprietary. +What is published here is the **measured outcome** and the evidence for it, which you +can verify against the artifacts in this repo. + +## The receipt + +| model | capability (MMLU-val) ↑ | compliance on harmful ↑ | +|---|---|---| +| pristine `Qwen3-4B-Instruct-2507` | 0.667 | 0.000 | +| **Ektomē (this model)** | **0.672** | **1.000** | + +These are **point estimates with no confidence interval** — which is precisely why the next section exists. + + +## The certificate + +Capability retention is certified by a paired non-inferiority test against the pristine +model (exact McNemar, Holm-corrected, one-sided bootstrap bound on the drop $d$ vs a +3% margin): + +| axis | n | ref | cand | d upper | verdict | +|---|---|---|---|---|---| +| arithmetic | 1400 | 0.879 | 0.869 | +0.016 | FAIL | +| instruction | 600 | 0.860 | 0.847 | +0.022 | FAIL | +| knowledge | 400 | 0.935 | 0.922 | +0.025 | PASS | +| reasoning | 400 | 0.825 | 0.833 | +0.003 | PASS | + + +**Overall: FAIL (3% margin, n=2800, alpha=0.05)** + +Reproducible from `seed=20260726`, pack `sha256:7bbaff877146e081…`. + +### Generation health checks + +| metric | pristine | Ektomē | n | +|---|---|---|---| +| `foreign_rate` | 0.0 | 0.0 | 15 | +| `degen_rate` | 0.0 | 0.0 | 15 | +| `instr_pass` | 1.0 | 1.0 | 5 | + +These are **degeneration guards** — code-switching, babbling, format compliance — +not capability measures. Note the sample sizes: they detect a broken model, not a +subtly weaker one. The capability claim rests on the certificate above, not here. + + +## Quantisations + +| file | bits | notes | +|---|---|---| +| `Ektome-Qwen3-4Bi-2507-Q8_0.gguf` | 8 | near-lossless | +| `Ektome-Qwen3-4Bi-2507-Q6_K.gguf` | 6 | | +| `Ektome-Qwen3-4Bi-2507-Q5_K_M.gguf` | 5 | | +| `Ektome-Qwen3-4Bi-2507-Q4_K_M.gguf` | 4 | imatrix | +| `Ektome-Qwen3-4Bi-2507-IQ4_XS.gguf` | 4 | imatrix, smallest usable | +| `Ektome-Qwen3-4Bi-2507-IQ3_M.gguf` | 3 | imatrix | + +`IQ*` variants are imatrix-quantised — better quality per bit at low precision. + + +## Limitations + +The certificate bounds **capability retention only**. It does not certify safety, factual +accuracy, or fitness for any purpose. Axes marked *inconclusive* are honestly +under-powered, and the certificate states the $n$ needed to resolve them. Compliance uses +a keyword classifier — a proxy that evasive phrasing can fool. **This model is uncensored +by construction: it will not refuse, and you are accountable for what you do with it.** + +## Citation + +```bibtex +@software{ektome_Ektome-Qwen3-4Bi-2507-PristinelyUncensored, + title = {Ektome-Qwen3-4Bi-2507-PristinelyUncensored}, + author = {Zynerji}, + year = {2026}, + url = {https://huggingface.co/Zynerji/Ektome-Qwen3-4Bi-2507-PristinelyUncensored} +} +``` diff --git a/cert_Qwen3-4Bi-2507.json b/cert_Qwen3-4Bi-2507.json new file mode 100644 index 0000000..fd549ca --- /dev/null +++ b/cert_Qwen3-4Bi-2507.json @@ -0,0 +1,133 @@ +{ + "sphragis_version": "0.1.0", + "generated_at": "2026-07-27T11:21:13.795774+00:00", + "reference": { + "endpoint": "http://127.0.0.1:8080/v1", + "model": "ref" + }, + "candidate": { + "endpoint": "http://127.0.0.1:8081/v1", + "model": "cand" + }, + "pack": { + "name": "/root/pack_v2.jsonl", + "n_tasks": 2800, + "sha256": "2de27099bbb15bab4f7b35599b215f038fdb68b3f5f4e1cda1a464f9dd18e14e" + }, + "overall": "FAIL", + "claim": "At least one axis shows a statistically significant accuracy regression (exact McNemar, Holm-corrected, alpha=0.05).", + "axes": [ + { + "axis": "arithmetic", + "n": 1400, + "counts": { + "both_correct": 1208, + "ref_only": 22, + "cand_only": 9, + "both_wrong": 161 + }, + "acc_reference": 0.878571, + "acc_candidate": 0.869286, + "regression_d": 0.009286, + "d_ci": [ + 0.001429, + 0.017143 + ], + "d_upper_bound": 0.015714, + "p_regression": 0.01472469, + "p_regression_holm": 0.04417406, + "p_improvement": 0.99466308, + "improved": false, + "mde_at_power": 0.009881, + "n_needed_for_margin": 204, + "verdict": "FAIL", + "reason": "significant_regression" + }, + { + "axis": "instruction", + "n": 600, + "counts": { + "both_correct": 508, + "ref_only": 8, + "cand_only": 0, + "both_wrong": 84 + }, + "acc_reference": 0.86, + "acc_candidate": 0.846667, + "regression_d": 0.013333, + "d_ci": [ + 0.005, + 0.023333 + ], + "d_upper_bound": 0.021667, + "p_regression": 0.00390625, + "p_regression_holm": 0.015625, + "p_improvement": 1.0, + "improved": false, + "mde_at_power": 0.011701, + "n_needed_for_margin": 204, + "verdict": "FAIL", + "reason": "significant_regression" + }, + { + "axis": "knowledge", + "n": 400, + "counts": { + "both_correct": 367, + "ref_only": 7, + "cand_only": 2, + "both_wrong": 24 + }, + "acc_reference": 0.935, + "acc_candidate": 0.9225, + "regression_d": 0.0125, + "d_ci": [ + -0.0025, + 0.0275 + ], + "d_upper_bound": 0.025, + "p_regression": 0.08984375, + "p_regression_holm": 0.1796875, + "p_improvement": 0.98046875, + "improved": false, + "mde_at_power": 0.0186, + "n_needed_for_margin": 204, + "verdict": "PASS", + "reason": "non_inferior_within_margin" + }, + { + "axis": "reasoning", + "n": 400, + "counts": { + "both_correct": 328, + "ref_only": 2, + "cand_only": 5, + "both_wrong": 65 + }, + "acc_reference": 0.825, + "acc_candidate": 0.8325, + "regression_d": -0.0075, + "d_ci": [ + -0.02, + 0.005 + ], + "d_upper_bound": 0.0025, + "p_regression": 0.9375, + "p_regression_holm": 0.9375, + "p_improvement": 0.2265625, + "improved": false, + "mde_at_power": 0.016404, + "n_needed_for_margin": 204, + "verdict": "PASS", + "reason": "non_inferior_within_margin" + } + ], + "params": { + "margin": 0.03, + "alpha": 0.05, + "n_floor": 30, + "power": 0.8, + "n_boot": 4000, + "seed": 0 + } +} \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..70adff8 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,61 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..81d449b --- /dev/null +++ b/config.json @@ -0,0 +1,71 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2560, + "initializer_range": 0.02, + "intermediate_size": 9728, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 262144, + "max_window_layers": 36, + "model_type": "qwen3", + "num_attention_heads": 32, + "num_hidden_layers": 36, + "num_key_value_heads": 8, + "pad_token_id": null, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 5000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.13.1", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/ektome_report.json b/ektome_report.json new file mode 100644 index 0000000..e6ee045 --- /dev/null +++ b/ektome_report.json @@ -0,0 +1,28 @@ +{ + "status": "SHIP", + "target": 0.99, + "SE_mmlu": 0.023555453190291203, + "base_compliance": 0.0, + "compliance": 1.0, + "base_cap": 0.6675, + "cap": 0.6725, + "dcap": 0.005, + "gen_base": { + "foreign_rate": 0.0, + "degen_rate": 0.0, + "instr_pass": 1.0 + }, + "gen": { + "foreign_rate": 0.0, + "degen_rate": 0.0, + "instr_pass": 1.0 + }, + "gen_delta": { + "d_foreign": 0.0, + "d_degen": 0.0, + "d_instr": 0.0, + "holds_gen": true + }, + "base_model": "Qwen/Qwen3-4B-Instruct-2507", + "_note": "Redacted: search trajectory, selected depth and edit-scope removed. Reported values are the measured outcome only." +} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..49e489b --- /dev/null +++ b/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "5.13.1" +} diff --git a/hero.png b/hero.png new file mode 100644 index 0000000..813cc68 --- /dev/null +++ b/hero.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a21887495eab2579b44049a125b29526edd31ec8427d94d7a00428a5273bd572 +size 245302 diff --git a/imatrix.dat b/imatrix.dat new file mode 100644 index 0000000..0d3eff4 --- /dev/null +++ b/imatrix.dat @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1fdc4835c8806f26dcf1a610896b1c71d81898f27dec44368bc95e7059643686 +size 3872640 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..75a7fc1 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e71b8061f18dbfc477098a72422d96bd5b1591ce5c316cade37142c83d7b0aa6 +size 8044982080 diff --git a/nvfp4/chat_template.jinja b/nvfp4/chat_template.jinja new file mode 100644 index 0000000..70adff8 --- /dev/null +++ b/nvfp4/chat_template.jinja @@ -0,0 +1,61 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} \ No newline at end of file diff --git a/nvfp4/config.json b/nvfp4/config.json new file mode 100644 index 0000000..7b6fcd5 --- /dev/null +++ b/nvfp4/config.json @@ -0,0 +1,121 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2560, + "initializer_range": 0.02, + "intermediate_size": 9728, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 262144, + "max_window_layers": 36, + "model_type": "qwen3", + "num_attention_heads": 32, + "num_hidden_layers": 36, + "num_key_value_heads": 8, + "pad_token_id": null, + "quantization_config": { + "config_groups": { + "group_0": { + "format": "nvfp4-pack-quantized", + "input_activations": { + "actorder": null, + "block_structure": null, + "dynamic": "local", + "group_size": 16, + "num_bits": 4, + "observer": "static_minmax", + "observer_kwargs": {}, + "scale_dtype": "torch.float8_e4m3fn", + "strategy": "tensor_group", + "symmetric": true, + "type": "float", + "zp_dtype": null + }, + "output_activations": null, + "targets": [ + "Linear" + ], + "weights": { + "actorder": null, + "block_structure": null, + "dynamic": false, + "group_size": 16, + "num_bits": 4, + "observer": "memoryless_minmax", + "observer_kwargs": {}, + "scale_dtype": "torch.float8_e4m3fn", + "strategy": "tensor_group", + "symmetric": true, + "type": "float", + "zp_dtype": null + } + } + }, + "format": "nvfp4-pack-quantized", + "global_compression_ratio": null, + "ignore": [ + "lm_head" + ], + "kv_cache_scheme": null, + "quant_method": "compressed-tensors", + "quantization_status": "compressed", + "sparsity_config": {}, + "transform_config": {}, + "version": "0.17.1" + }, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 5000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.10.1", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} \ No newline at end of file diff --git a/nvfp4/generation_config.json b/nvfp4/generation_config.json new file mode 100644 index 0000000..9efdfce --- /dev/null +++ b/nvfp4/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "5.10.1" +} diff --git a/nvfp4/model.safetensors b/nvfp4/model.safetensors new file mode 100644 index 0000000..0daa9bc --- /dev/null +++ b/nvfp4/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b2765ba6c2ecca790b9b5993e67740f2558b1ab64d5dbeeb6d05e0fe64455c6 +size 2822178072 diff --git a/nvfp4/recipe.yaml b/nvfp4/recipe.yaml new file mode 100644 index 0000000..6a93bcc --- /dev/null +++ b/nvfp4/recipe.yaml @@ -0,0 +1,7 @@ +default_stage: + default_modifiers: + QuantizationModifier: + targets: [Linear] + ignore: [lm_head] + scheme: NVFP4 + bypass_divisibility_checks: false diff --git a/nvfp4/tokenizer.json b/nvfp4/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/nvfp4/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/nvfp4/tokenizer_config.json b/nvfp4/tokenizer_config.json new file mode 100644 index 0000000..cb87962 --- /dev/null +++ b/nvfp4/tokenizer_config.json @@ -0,0 +1,30 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 1010000, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..90297ca --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,16 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "is_local": false, + "local_files_only": false, + "model_max_length": 1010000, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].role == 'system' %}\n {{- messages[0].content + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n<|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0].content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n {%- if message.content is string %}\n {%- set content = message.content %}\n {%- else %}\n {%- set content = '' %}\n {%- endif %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}" +} \ No newline at end of file diff --git a/whitepaper.pdf b/whitepaper.pdf new file mode 100644 index 0000000..0ca9d81 Binary files /dev/null and b/whitepaper.pdf differ