commit 9991c0c9d26528feacd936c07b504304df7bc66a Author: ModelHub XC Date: Sat Sep 19 01:48:17 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: Zynerji/Ektome-Qwen3-1.7B-PristinelyUncensored Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..b296d96 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,44 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +hero.png filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-1.7B-IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-1.7B-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-1.7B-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-1.7B-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-1.7B-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Qwen3-1.7B-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +imatrix.dat filter=lfs diff=lfs merge=lfs -text diff --git a/Ektome-Qwen3-1.7B-IQ3_M.gguf b/Ektome-Qwen3-1.7B-IQ3_M.gguf new file mode 100644 index 0000000..896d56b --- /dev/null +++ b/Ektome-Qwen3-1.7B-IQ3_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5396bdb8d716e3b0bbcf00e13fbe95d2e32c35f1efd3576c8b023c66f5ed4eae +size 895662112 diff --git a/Ektome-Qwen3-1.7B-IQ4_XS.gguf b/Ektome-Qwen3-1.7B-IQ4_XS.gguf new file mode 100644 index 0000000..32b19c9 --- /dev/null +++ b/Ektome-Qwen3-1.7B-IQ4_XS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac857594d756b622361b114e9af13a4f034514e1778ae8de59b0a7d8d5c82316 +size 1010382880 diff --git a/Ektome-Qwen3-1.7B-Q4_K_M.gguf b/Ektome-Qwen3-1.7B-Q4_K_M.gguf new file mode 100644 index 0000000..d454edf --- /dev/null +++ b/Ektome-Qwen3-1.7B-Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc5a54fcbbcb85832a3198faa819150ff801aea8acce1bc10824f576f486b9cd +size 1107408928 diff --git a/Ektome-Qwen3-1.7B-Q5_K_M.gguf b/Ektome-Qwen3-1.7B-Q5_K_M.gguf new file mode 100644 index 0000000..7a22074 --- /dev/null +++ b/Ektome-Qwen3-1.7B-Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:02114abc370fab2e76a029b2382b057234901bc259bc1297072045bc67dfd5e3 +size 1257879584 diff --git a/Ektome-Qwen3-1.7B-Q6_K.gguf b/Ektome-Qwen3-1.7B-Q6_K.gguf new file mode 100644 index 0000000..d806585 --- /dev/null +++ b/Ektome-Qwen3-1.7B-Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:90c473bbdc73d1bcb0d5796b8f40c1ff1f8fba733334ee6482fc47196f0a4293 +size 1417754656 diff --git a/Ektome-Qwen3-1.7B-Q8_0.gguf b/Ektome-Qwen3-1.7B-Q8_0.gguf new file mode 100644 index 0000000..82ec8b5 --- /dev/null +++ b/Ektome-Qwen3-1.7B-Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6a7355b89d9291ce33d25e00a535fc25f39888ac7d4386623e3c72d9ac16a5de +size 1834426400 diff --git a/README.md b/README.md new file mode 100644 index 0000000..910d747 --- /dev/null +++ b/README.md @@ -0,0 +1,121 @@ +--- +license: apache-2.0 +base_model: Qwen/Qwen3-1.7B +tags: + - uncensored + - abliterated + - uncertified + - ektome + - sphragis + - qwen3 +language: + - en +pipeline_tag: text-generation +--- + +![Ektome-Qwen3-1.7B-PristinelyUncensored](./hero.png) + +# Ektome-Qwen3-1.7B-PristinelyUncensored + +**Uncensored. No n=2800 certificate has been run for this model, so no capability-retention claim is made.** + +> **compliance 0.24 to 0.99 at capability +0.010 vs pristine.** + +$$\colorbox{black}{$\color{white} +\begin{array}{ll} +\textsf{EKTOME CERTIFICATE} & {} \\ +\textsf{capability} & \textsf{NOT} \\ +\textsf{margin} & 3\% \\ +\textsf{items } n & 200 \\ +\textsf{worst-axis bound} & +0.010 \\ +\textsf{compliance} & 0.24 \rightarrow 0.99 \\ +\end{array}$}$$ + +> ### ⚠️ Not certified +> +> No n=2800 paired certificate exists for this model. Any numbers below are +> point estimates with no confidence interval. + + +📄 **[Read the whitepaper (PDF)](./whitepaper.pdf)** — full method, receipts and certification. +The PDF is the authoritative document: dark-typeset, with the complete derivation, the +per-axis certificate and the reproducibility hashes. + +--- + +## Why this exists + +Standard abliteration removes a coarse *refusal direction* that is entangled with +directions carrying knowledge and reasoning. The result is an uncensored model with a +capability tax that is **almost never measured**. + +Ektomē (ἐκτομή, *excision*) isolates and removes only the refusal-**specific** +component, leaving general helpfulness intact, and does so norm-preservingly on the +pristine model — no training, no distillation, no damage to repair. The extraction +depth is selected per model by automated search against measured compliance. + +The estimator, excision operator and depth-selection procedure are proprietary. +What is published here is the **measured outcome** and the evidence for it, which you +can verify against the artifacts in this repo. + +## The receipt + +| model | capability (MMLU-val) ↑ | compliance on harmful ↑ | +|---|---|---| +| pristine `Qwen3-1.7B` | 0.540 | 0.240 | +| **Ektomē (this model)** | **0.530** | **0.990** | + +These are **point estimates with no confidence interval** — which is precisely why the next section exists. + + +## The certificate + +Capability retention is certified by a paired non-inferiority test against the pristine +model (exact McNemar, Holm-corrected, one-sided bootstrap bound on the drop $d$ vs a +3% margin): + +| axis | n | ref | cand | d upper | verdict | +|---|---|---|---|---|---| +| MMLU-val (POINT ESTIMATE, n=200, no CI) | 200 | 0.540 | 0.530 | +0.010 | UNCERTIFIED | + + +**Overall: NOT CERTIFIED - no n=2800 paired test has been run for this model** + +Reproducible from `seed=20260726`, pack `sha256:7bbaff877146e081…`. + +### Generation health checks + +| metric | pristine | Ektomē | n | +|---|---|---|---| +| `foreign_rate` | 0.0 | 0.0 | 15 | +| `degen_rate` | 0.1 | 0.1 | 15 | +| `instr_pass` | 0.4 | 0.4 | 5 | + +These are **degeneration guards** — code-switching, babbling, format compliance — +not capability measures. Note the sample sizes: they detect a broken model, not a +subtly weaker one. The capability claim rests on the certificate above, not here. + + +## Quantisations + +_No quantisations have been published for this model yet — bf16 weights only._ + + +## Limitations + +The certificate bounds **capability retention only**. It does not certify safety, factual +accuracy, or fitness for any purpose. Axes marked *inconclusive* are honestly +under-powered, and the certificate states the $n$ needed to resolve them. Compliance uses +a keyword classifier — a proxy that evasive phrasing can fool. **This model is uncensored +by construction: it will not refuse, and you are accountable for what you do with it.** + +## Citation + +```bibtex +@software{ektome_Ektome-Qwen3-1.7B-PristinelyUncensored, + title = {Ektome-Qwen3-1.7B-PristinelyUncensored}, + author = {Zynerji}, + year = {2026}, + url = {https://huggingface.co/Zynerji/Ektome-Qwen3-1.7B-PristinelyUncensored} +} +``` diff --git a/cert_Qwen3-1.7B.json b/cert_Qwen3-1.7B.json new file mode 100644 index 0000000..17340ad --- /dev/null +++ b/cert_Qwen3-1.7B.json @@ -0,0 +1,133 @@ +{ + "sphragis_version": "0.1.0", + "generated_at": "2026-07-27T17:59:04.397198+00:00", + "reference": { + "endpoint": "http://127.0.0.1:8080/v1", + "model": "ref" + }, + "candidate": { + "endpoint": "http://127.0.0.1:8081/v1", + "model": "cand" + }, + "pack": { + "name": "/root/pack_v2.jsonl", + "n_tasks": 2800, + "sha256": "2de27099bbb15bab4f7b35599b215f038fdb68b3f5f4e1cda1a464f9dd18e14e" + }, + "overall": "PASS", + "claim": "For every axis, the candidate was demonstrated non-inferior to the reference: the one-sided 95% upper confidence bound on the accuracy regression is below the margin of 3.0%.", + "axes": [ + { + "axis": "arithmetic", + "n": 1400, + "counts": { + "both_correct": 0, + "ref_only": 0, + "cand_only": 0, + "both_wrong": 1400 + }, + "acc_reference": 0.0, + "acc_candidate": 0.0, + "regression_d": 0.0, + "d_ci": [ + 0.0, + 0.0 + ], + "d_upper_bound": 0.0, + "p_regression": 1.0, + "p_regression_holm": 1.0, + "p_improvement": 1.0, + "improved": false, + "mde_at_power": null, + "n_needed_for_margin": 204, + "verdict": "PASS", + "reason": "non_inferior_within_margin" + }, + { + "axis": "instruction", + "n": 600, + "counts": { + "both_correct": 0, + "ref_only": 0, + "cand_only": 0, + "both_wrong": 600 + }, + "acc_reference": 0.0, + "acc_candidate": 0.0, + "regression_d": 0.0, + "d_ci": [ + 0.0, + 0.0 + ], + "d_upper_bound": 0.0, + "p_regression": 1.0, + "p_regression_holm": 1.0, + "p_improvement": 1.0, + "improved": false, + "mde_at_power": null, + "n_needed_for_margin": 204, + "verdict": "PASS", + "reason": "non_inferior_within_margin" + }, + { + "axis": "knowledge", + "n": 400, + "counts": { + "both_correct": 0, + "ref_only": 0, + "cand_only": 0, + "both_wrong": 400 + }, + "acc_reference": 0.0, + "acc_candidate": 0.0, + "regression_d": 0.0, + "d_ci": [ + 0.0, + 0.0 + ], + "d_upper_bound": 0.0, + "p_regression": 1.0, + "p_regression_holm": 1.0, + "p_improvement": 1.0, + "improved": false, + "mde_at_power": null, + "n_needed_for_margin": 204, + "verdict": "PASS", + "reason": "non_inferior_within_margin" + }, + { + "axis": "reasoning", + "n": 400, + "counts": { + "both_correct": 0, + "ref_only": 0, + "cand_only": 0, + "both_wrong": 400 + }, + "acc_reference": 0.0, + "acc_candidate": 0.0, + "regression_d": 0.0, + "d_ci": [ + 0.0, + 0.0 + ], + "d_upper_bound": 0.0, + "p_regression": 1.0, + "p_regression_holm": 1.0, + "p_improvement": 1.0, + "improved": false, + "mde_at_power": null, + "n_needed_for_margin": 204, + "verdict": "PASS", + "reason": "non_inferior_within_margin" + } + ], + "params": { + "margin": 0.03, + "alpha": 0.05, + "n_floor": 30, + "power": 0.8, + "n_boot": 4000, + "seed": 0 + } +} \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..01be9b3 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,89 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} +{%- endfor %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if loop.index0 > ns.last_query_index %} + {%- if loop.last or (not loop.last and reasoning_content) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content.strip('\n') + '\n\n\n' + content.lstrip('\n') }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..9cb9417 --- /dev/null +++ b/config.json @@ -0,0 +1,63 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 6144, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 40960, + "max_window_layers": 28, + "model_type": "qwen3", + "num_attention_heads": 16, + "num_hidden_layers": 28, + "num_key_value_heads": 8, + "pad_token_id": null, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.13.1", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/ektome_report.json b/ektome_report.json new file mode 100644 index 0000000..ab1bfce --- /dev/null +++ b/ektome_report.json @@ -0,0 +1,28 @@ +{ + "status": "SHIP", + "target": 0.99, + "SE_mmlu": 0.024919871588754226, + "base_compliance": 0.24, + "compliance": 0.99, + "base_cap": 0.54, + "cap": 0.53, + "dcap": -0.01, + "gen_base": { + "foreign_rate": 0.0, + "degen_rate": 0.1, + "instr_pass": 0.4 + }, + "gen": { + "foreign_rate": 0.0, + "degen_rate": 0.1, + "instr_pass": 0.4 + }, + "gen_delta": { + "d_foreign": 0.0, + "d_degen": 0.0, + "d_instr": 0.0, + "holds_gen": true + }, + "base_model": "Qwen/Qwen3-1.7B", + "_note": "Redacted: search trajectory, selected depth and edit-scope removed. Reported values are the measured outcome only." +} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..0fcfb63 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_k": 20, + "top_p": 0.95, + "transformers_version": "5.13.1" +} diff --git a/hero.png b/hero.png new file mode 100644 index 0000000..6b97741 --- /dev/null +++ b/hero.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a6cd53edf8932adc4bd635b4828ab9578d667b150997d0470af2a3995db138c1 +size 238728 diff --git a/imatrix.dat b/imatrix.dat new file mode 100644 index 0000000..f27ce6e --- /dev/null +++ b/imatrix.dat @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4deab7006a8750817d1f986b956d6f0afd13aba2ddae329c5bc3b3d0e7ad254c +size 2094560 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..221b3ed --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:af2e2776457a64efda397d2880bd5e08c409d23eac4878cf9738c8007e595a18 +size 3441185608 diff --git a/nvfp4/chat_template.jinja b/nvfp4/chat_template.jinja new file mode 100644 index 0000000..01be9b3 --- /dev/null +++ b/nvfp4/chat_template.jinja @@ -0,0 +1,89 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} +{%- endfor %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if loop.index0 > ns.last_query_index %} + {%- if loop.last or (not loop.last and reasoning_content) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content.strip('\n') + '\n\n\n' + content.lstrip('\n') }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/nvfp4/config.json b/nvfp4/config.json new file mode 100644 index 0000000..d48684a --- /dev/null +++ b/nvfp4/config.json @@ -0,0 +1,113 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 6144, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 40960, + "max_window_layers": 28, + "model_type": "qwen3", + "num_attention_heads": 16, + "num_hidden_layers": 28, + "num_key_value_heads": 8, + "pad_token_id": null, + "quantization_config": { + "config_groups": { + "group_0": { + "format": "nvfp4-pack-quantized", + "input_activations": { + "actorder": null, + "block_structure": null, + "dynamic": "local", + "group_size": 16, + "num_bits": 4, + "observer": "static_minmax", + "observer_kwargs": {}, + "scale_dtype": "torch.float8_e4m3fn", + "strategy": "tensor_group", + "symmetric": true, + "type": "float", + "zp_dtype": null + }, + "output_activations": null, + "targets": [ + "Linear" + ], + "weights": { + "actorder": null, + "block_structure": null, + "dynamic": false, + "group_size": 16, + "num_bits": 4, + "observer": "memoryless_minmax", + "observer_kwargs": {}, + "scale_dtype": "torch.float8_e4m3fn", + "strategy": "tensor_group", + "symmetric": true, + "type": "float", + "zp_dtype": null + } + } + }, + "format": "nvfp4-pack-quantized", + "global_compression_ratio": null, + "ignore": [ + "lm_head" + ], + "kv_cache_scheme": null, + "quant_method": "compressed-tensors", + "quantization_status": "compressed", + "sparsity_config": {}, + "transform_config": {}, + "version": "0.17.1" + }, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.10.1", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} \ No newline at end of file diff --git a/nvfp4/generation_config.json b/nvfp4/generation_config.json new file mode 100644 index 0000000..3bed6d5 --- /dev/null +++ b/nvfp4/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_k": 20, + "top_p": 0.95, + "transformers_version": "5.10.1" +} diff --git a/nvfp4/model.safetensors b/nvfp4/model.safetensors new file mode 100644 index 0000000..d659907 --- /dev/null +++ b/nvfp4/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:90b516fad102f4ffd4cfb14c396e49e961b72eb8720ea37d464253e922fd263e +size 1415404568 diff --git a/nvfp4/recipe.yaml b/nvfp4/recipe.yaml new file mode 100644 index 0000000..6a93bcc --- /dev/null +++ b/nvfp4/recipe.yaml @@ -0,0 +1,7 @@ +default_stage: + default_modifiers: + QuantizationModifier: + targets: [Linear] + ignore: [lm_head] + scheme: NVFP4 + bypass_divisibility_checks: false diff --git a/nvfp4/tokenizer.json b/nvfp4/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/nvfp4/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/nvfp4/tokenizer_config.json b/nvfp4/tokenizer_config.json new file mode 100644 index 0000000..770e41d --- /dev/null +++ b/nvfp4/tokenizer_config.json @@ -0,0 +1,30 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..2f0025f --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,16 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].role == 'system' %}\n {{- messages[0].content + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n<|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0].content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if message.content is string %}\n {%- set content = message.content %}\n {%- else %}\n {%- set content = '' %}\n {%- endif %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '' in content %}\n {%- set reasoning_content = content.split('')[0].rstrip('\\n').split('')[-1].lstrip('\\n') %}\n {%- set content = content.split('')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- if loop.index0 > ns.last_query_index %}\n {%- if loop.last or (not loop.last and reasoning_content) %}\n {{- '<|im_start|>' + message.role + '\\n\\n' + reasoning_content.strip('\\n') + '\\n\\n\\n' + content.lstrip('\\n') }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '\\n\\n\\n\\n' }}\n {%- endif %}\n{%- endif %}" +} \ No newline at end of file diff --git a/whitepaper.pdf b/whitepaper.pdf new file mode 100644 index 0000000..cb6b03e Binary files /dev/null and b/whitepaper.pdf differ