commit aa1c7235bcb3a97f74034d04e431b4e1b5be5cad Author: ModelHub XC Date: Fri Sep 18 17:24:17 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: Zynerji/Ektome-Llama-3.2-1Bi-PristinelyUncensored Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..b211bd4 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,44 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text +hero.png filter=lfs diff=lfs merge=lfs -text +Ektome-Llama-3.2-1Bi-IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Llama-3.2-1Bi-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Llama-3.2-1Bi-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Llama-3.2-1Bi-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Llama-3.2-1Bi-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Ektome-Llama-3.2-1Bi-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +imatrix.dat filter=lfs diff=lfs merge=lfs -text diff --git a/Ektome-Llama-3.2-1Bi-IQ3_M.gguf b/Ektome-Llama-3.2-1Bi-IQ3_M.gguf new file mode 100644 index 0000000..6af7a9d --- /dev/null +++ b/Ektome-Llama-3.2-1Bi-IQ3_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2366f5373b40be1e88940f08c91327bb73428c665ddac631711f89b603868ffa +size 657289216 diff --git a/Ektome-Llama-3.2-1Bi-IQ4_XS.gguf b/Ektome-Llama-3.2-1Bi-IQ4_XS.gguf new file mode 100644 index 0000000..cdb3c5b --- /dev/null +++ b/Ektome-Llama-3.2-1Bi-IQ4_XS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb723066281eae9b67d88e5950454075f3d8fbbfbf93bc3b384bc0d1f4d6d205 +size 743141376 diff --git a/Ektome-Llama-3.2-1Bi-Q4_K_M.gguf b/Ektome-Llama-3.2-1Bi-Q4_K_M.gguf new file mode 100644 index 0000000..08761e1 --- /dev/null +++ b/Ektome-Llama-3.2-1Bi-Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a496f566c8f6ad4985d7fc6946d9122f37b6d7ec99e1cd4dd66f26233a3773ae +size 807694336 diff --git a/Ektome-Llama-3.2-1Bi-Q5_K_M.gguf b/Ektome-Llama-3.2-1Bi-Q5_K_M.gguf new file mode 100644 index 0000000..eb2b077 --- /dev/null +++ b/Ektome-Llama-3.2-1Bi-Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b9ad671e6e68fe69b51f14b864a87df5ab393ede904c6dd728988f7e1b975e6f +size 911503360 diff --git a/Ektome-Llama-3.2-1Bi-Q6_K.gguf b/Ektome-Llama-3.2-1Bi-Q6_K.gguf new file mode 100644 index 0000000..403b38d --- /dev/null +++ b/Ektome-Llama-3.2-1Bi-Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:09459940da27a6d9dbe655a240a0668bebd53289f7af53d86d04583600698e2a +size 1021800448 diff --git a/Ektome-Llama-3.2-1Bi-Q8_0.gguf b/Ektome-Llama-3.2-1Bi-Q8_0.gguf new file mode 100644 index 0000000..3bd55ec --- /dev/null +++ b/Ektome-Llama-3.2-1Bi-Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f054b6e47f9650f18d609b17a11c73aea9dfd00d868c801e2fb5daa281ce6e23 +size 1321082880 diff --git a/README.md b/README.md new file mode 100644 index 0000000..6dc98f4 --- /dev/null +++ b/README.md @@ -0,0 +1,121 @@ +--- +license: apache-2.0 +base_model: unsloth/Llama-3.2-1B-Instruct +tags: + - uncensored + - abliterated + - uncertified + - ektome + - sphragis + - llama +language: + - en +pipeline_tag: text-generation +--- + +![Ektome-Llama-3.2-1Bi-PristinelyUncensored](./hero.png) + +# Ektome-Llama-3.2-1Bi-PristinelyUncensored + +**Uncensored. No n=2800 certificate has been run for this model, so no capability-retention claim is made.** + +> **compliance 0.31 to 1.00 at capability -0.010 vs pristine.** + +$$\colorbox{black}{$\color{white} +\begin{array}{ll} +\textsf{EKTOME CERTIFICATE} & {} \\ +\textsf{capability} & \textsf{NOT} \\ +\textsf{margin} & 3\% \\ +\textsf{items } n & 200 \\ +\textsf{worst-axis bound} & -0.010 \\ +\textsf{compliance} & 0.31 \rightarrow 1.00 \\ +\end{array}$}$$ + +> ### ⚠️ Not certified +> +> No n=2800 paired certificate exists for this model. Any numbers below are +> point estimates with no confidence interval. + + +📄 **[Read the whitepaper (PDF)](./whitepaper.pdf)** — full method, receipts and certification. +The PDF is the authoritative document: dark-typeset, with the complete derivation, the +per-axis certificate and the reproducibility hashes. + +--- + +## Why this exists + +Standard abliteration removes a coarse *refusal direction* that is entangled with +directions carrying knowledge and reasoning. The result is an uncensored model with a +capability tax that is **almost never measured**. + +Ektomē (ἐκτομή, *excision*) isolates and removes only the refusal-**specific** +component, leaving general helpfulness intact, and does so norm-preservingly on the +pristine model — no training, no distillation, no damage to repair. The extraction +depth is selected per model by automated search against measured compliance. + +The estimator, excision operator and depth-selection procedure are proprietary. +What is published here is the **measured outcome** and the evidence for it, which you +can verify against the artifacts in this repo. + +## The receipt + +| model | capability (MMLU-val) ↑ | compliance on harmful ↑ | +|---|---|---| +| pristine `Llama-3.2-1B-Instruct` | 0.445 | 0.310 | +| **Ektomē (this model)** | **0.455** | **1.000** | + +These are **point estimates with no confidence interval** — which is precisely why the next section exists. + + +## The certificate + +Capability retention is certified by a paired non-inferiority test against the pristine +model (exact McNemar, Holm-corrected, one-sided bootstrap bound on the drop $d$ vs a +3% margin): + +| axis | n | ref | cand | d upper | verdict | +|---|---|---|---|---|---| +| MMLU-val (POINT ESTIMATE, n=200, no CI) | 200 | 0.445 | 0.455 | -0.010 | UNCERTIFIED | + + +**Overall: NOT CERTIFIED - no n=2800 paired test has been run for this model** + +Reproducible from `seed=20260726`, pack `sha256:7bbaff877146e081…`. + +### Generation health checks + +| metric | pristine | Ektomē | n | +|---|---|---|---| +| `foreign_rate` | 0.0 | 0.0 | 15 | +| `degen_rate` | 0.0 | 0.1 | 15 | +| `instr_pass` | 1.0 | 1.0 | 5 | + +These are **degeneration guards** — code-switching, babbling, format compliance — +not capability measures. Note the sample sizes: they detect a broken model, not a +subtly weaker one. The capability claim rests on the certificate above, not here. + + +## Quantisations + +_No quantisations have been published for this model yet — bf16 weights only._ + + +## Limitations + +The certificate bounds **capability retention only**. It does not certify safety, factual +accuracy, or fitness for any purpose. Axes marked *inconclusive* are honestly +under-powered, and the certificate states the $n$ needed to resolve them. Compliance uses +a keyword classifier — a proxy that evasive phrasing can fool. **This model is uncensored +by construction: it will not refuse, and you are accountable for what you do with it.** + +## Citation + +```bibtex +@software{ektome_Ektome-Llama-3.2-1Bi-PristinelyUncensored, + title = {Ektome-Llama-3.2-1Bi-PristinelyUncensored}, + author = {Zynerji}, + year = {2026}, + url = {https://huggingface.co/Zynerji/Ektome-Llama-3.2-1Bi-PristinelyUncensored} +} +``` diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..1bad6a0 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,93 @@ +{{- bos_token }} +{%- if custom_tools is defined %} + {%- set tools = custom_tools %} +{%- endif %} +{%- if not tools_in_user_message is defined %} + {%- set tools_in_user_message = true %} +{%- endif %} +{%- if not date_string is defined %} + {%- if strftime_now is defined %} + {%- set date_string = strftime_now("%d %b %Y") %} + {%- else %} + {%- set date_string = "26 Jul 2024" %} + {%- endif %} +{%- endif %} +{%- if not tools is defined %} + {%- set tools = none %} +{%- endif %} + +{#- This block extracts the system message, so we can slot it into the right place. #} +{%- if messages[0]['role'] == 'system' %} + {%- set system_message = messages[0]['content']|trim %} + {%- set messages = messages[1:] %} +{%- else %} + {%- set system_message = "" %} +{%- endif %} + +{#- System message #} +{{- "<|start_header_id|>system<|end_header_id|>\n\n" }} +{%- if tools is not none %} + {{- "Environment: ipython\n" }} +{%- endif %} +{{- "Cutting Knowledge Date: December 2023\n" }} +{{- "Today Date: " + date_string + "\n\n" }} +{%- if tools is not none and not tools_in_user_message %} + {{- "You have access to the following functions. To call a function, please respond with JSON for a function call." }} + {{- 'Respond in the format {"name": function name, "parameters": dictionary of argument name and its value}.' }} + {{- "Do not use variables.\n\n" }} + {%- for t in tools %} + {{- t | tojson(indent=4) }} + {{- "\n\n" }} + {%- endfor %} +{%- endif %} +{{- system_message }} +{{- "<|eot_id|>" }} + +{#- Custom tools are passed in a user message with some extra guidance #} +{%- if tools_in_user_message and not tools is none %} + {#- Extract the first user message so we can plug it in here #} + {%- if messages | length != 0 %} + {%- set first_user_message = messages[0]['content']|trim %} + {%- set messages = messages[1:] %} + {%- else %} + {{- raise_exception("Cannot put tools in the first user message when there's no first user message!") }} +{%- endif %} + {{- '<|start_header_id|>user<|end_header_id|>\n\n' -}} + {{- "Given the following functions, please respond with a JSON for a function call " }} + {{- "with its proper arguments that best answers the given prompt.\n\n" }} + {{- 'Respond in the format {"name": function name, "parameters": dictionary of argument name and its value}.' }} + {{- "Do not use variables.\n\n" }} + {%- for t in tools %} + {{- t | tojson(indent=4) }} + {{- "\n\n" }} + {%- endfor %} + {{- first_user_message + "<|eot_id|>"}} +{%- endif %} + +{%- for message in messages %} + {%- if not (message.role == 'ipython' or message.role == 'tool' or 'tool_calls' in message) %} + {{- '<|start_header_id|>' + message['role'] + '<|end_header_id|>\n\n'+ message['content'] | trim + '<|eot_id|>' }} + {%- elif 'tool_calls' in message %} + {%- if not message.tool_calls|length == 1 %} + {{- raise_exception("This model only supports single tool-calls at once!") }} + {%- endif %} + {%- set tool_call = message.tool_calls[0].function %} + {{- '<|start_header_id|>assistant<|end_header_id|>\n\n' -}} + {{- '{"name": "' + tool_call.name + '", ' }} + {{- '"parameters": ' }} + {{- tool_call.arguments | tojson }} + {{- "}" }} + {{- "<|eot_id|>" }} + {%- elif message.role == "tool" or message.role == "ipython" %} + {{- "<|start_header_id|>ipython<|end_header_id|>\n\n" }} + {%- if message.content is mapping or message.content is iterable %} + {{- message.content | tojson }} + {%- else %} + {{- message.content }} + {%- endif %} + {{- "<|eot_id|>" }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|start_header_id|>assistant<|end_header_id|>\n\n' }} +{%- endif %} diff --git a/config.json b/config.json new file mode 100644 index 0000000..74c906d --- /dev/null +++ b/config.json @@ -0,0 +1,37 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "dtype": "bfloat16", + "eos_token_id": 128009, + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pad_token_id": 128004, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_theta": 500000.0, + "rope_type": "llama3" + }, + "tie_word_embeddings": true, + "transformers_version": "5.13.1", + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} diff --git a/ektome_report.json b/ektome_report.json new file mode 100644 index 0000000..06a5c32 --- /dev/null +++ b/ektome_report.json @@ -0,0 +1,28 @@ +{ + "status": "SHIP", + "target": 0.99, + "SE_mmlu": 0.02484828967957352, + "base_compliance": 0.31, + "compliance": 1.0, + "base_cap": 0.445, + "cap": 0.455, + "dcap": 0.01, + "gen_base": { + "foreign_rate": 0.0, + "degen_rate": 0.0, + "instr_pass": 1.0 + }, + "gen": { + "foreign_rate": 0.0, + "degen_rate": 0.1, + "instr_pass": 1.0 + }, + "gen_delta": { + "d_foreign": 0.0, + "d_degen": 0.1, + "d_instr": 0.0, + "holds_gen": true + }, + "base_model": "unsloth/Llama-3.2-1B-Instruct", + "_note": "Redacted: search trajectory, selected depth and edit-scope removed. Reported values are the measured outcome only." +} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..4ea6596 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,14 @@ +{ + "bos_token_id": 128000, + "do_sample": true, + "eos_token_id": [ + 128001, + 128008, + 128009 + ], + "max_length": 131072, + "pad_token_id": 128004, + "temperature": 0.6, + "top_p": 0.9, + "transformers_version": "5.13.1" +} diff --git a/hero.png b/hero.png new file mode 100644 index 0000000..7ae557d --- /dev/null +++ b/hero.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:00ae6fb8b2a6cec4bc1297e0162f906e232eb3720009e55c7b71f17a24844d39 +size 240493 diff --git a/imatrix.dat b/imatrix.dat new file mode 100644 index 0000000..1fc4b38 --- /dev/null +++ b/imatrix.dat @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:331d9301562af2299475a8b73999a7253029729fbd56eb57352410ce42965866 +size 1328000 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..146812d --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c5484f2f15684587aef2e709d0c30b1a4a657d187ca1be041a2b205b8de07a83 +size 2471645608 diff --git a/nvfp4/chat_template.jinja b/nvfp4/chat_template.jinja new file mode 100644 index 0000000..1bad6a0 --- /dev/null +++ b/nvfp4/chat_template.jinja @@ -0,0 +1,93 @@ +{{- bos_token }} +{%- if custom_tools is defined %} + {%- set tools = custom_tools %} +{%- endif %} +{%- if not tools_in_user_message is defined %} + {%- set tools_in_user_message = true %} +{%- endif %} +{%- if not date_string is defined %} + {%- if strftime_now is defined %} + {%- set date_string = strftime_now("%d %b %Y") %} + {%- else %} + {%- set date_string = "26 Jul 2024" %} + {%- endif %} +{%- endif %} +{%- if not tools is defined %} + {%- set tools = none %} +{%- endif %} + +{#- This block extracts the system message, so we can slot it into the right place. #} +{%- if messages[0]['role'] == 'system' %} + {%- set system_message = messages[0]['content']|trim %} + {%- set messages = messages[1:] %} +{%- else %} + {%- set system_message = "" %} +{%- endif %} + +{#- System message #} +{{- "<|start_header_id|>system<|end_header_id|>\n\n" }} +{%- if tools is not none %} + {{- "Environment: ipython\n" }} +{%- endif %} +{{- "Cutting Knowledge Date: December 2023\n" }} +{{- "Today Date: " + date_string + "\n\n" }} +{%- if tools is not none and not tools_in_user_message %} + {{- "You have access to the following functions. To call a function, please respond with JSON for a function call." }} + {{- 'Respond in the format {"name": function name, "parameters": dictionary of argument name and its value}.' }} + {{- "Do not use variables.\n\n" }} + {%- for t in tools %} + {{- t | tojson(indent=4) }} + {{- "\n\n" }} + {%- endfor %} +{%- endif %} +{{- system_message }} +{{- "<|eot_id|>" }} + +{#- Custom tools are passed in a user message with some extra guidance #} +{%- if tools_in_user_message and not tools is none %} + {#- Extract the first user message so we can plug it in here #} + {%- if messages | length != 0 %} + {%- set first_user_message = messages[0]['content']|trim %} + {%- set messages = messages[1:] %} + {%- else %} + {{- raise_exception("Cannot put tools in the first user message when there's no first user message!") }} +{%- endif %} + {{- '<|start_header_id|>user<|end_header_id|>\n\n' -}} + {{- "Given the following functions, please respond with a JSON for a function call " }} + {{- "with its proper arguments that best answers the given prompt.\n\n" }} + {{- 'Respond in the format {"name": function name, "parameters": dictionary of argument name and its value}.' }} + {{- "Do not use variables.\n\n" }} + {%- for t in tools %} + {{- t | tojson(indent=4) }} + {{- "\n\n" }} + {%- endfor %} + {{- first_user_message + "<|eot_id|>"}} +{%- endif %} + +{%- for message in messages %} + {%- if not (message.role == 'ipython' or message.role == 'tool' or 'tool_calls' in message) %} + {{- '<|start_header_id|>' + message['role'] + '<|end_header_id|>\n\n'+ message['content'] | trim + '<|eot_id|>' }} + {%- elif 'tool_calls' in message %} + {%- if not message.tool_calls|length == 1 %} + {{- raise_exception("This model only supports single tool-calls at once!") }} + {%- endif %} + {%- set tool_call = message.tool_calls[0].function %} + {{- '<|start_header_id|>assistant<|end_header_id|>\n\n' -}} + {{- '{"name": "' + tool_call.name + '", ' }} + {{- '"parameters": ' }} + {{- tool_call.arguments | tojson }} + {{- "}" }} + {{- "<|eot_id|>" }} + {%- elif message.role == "tool" or message.role == "ipython" %} + {{- "<|start_header_id|>ipython<|end_header_id|>\n\n" }} + {%- if message.content is mapping or message.content is iterable %} + {{- message.content | tojson }} + {%- else %} + {{- message.content }} + {%- endif %} + {{- "<|eot_id|>" }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|start_header_id|>assistant<|end_header_id|>\n\n' }} +{%- endif %} diff --git a/nvfp4/config.json b/nvfp4/config.json new file mode 100644 index 0000000..b85d9ef --- /dev/null +++ b/nvfp4/config.json @@ -0,0 +1,87 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "dtype": "bfloat16", + "eos_token_id": 128009, + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pad_token_id": 128004, + "pretraining_tp": 1, + "quantization_config": { + "config_groups": { + "group_0": { + "format": "nvfp4-pack-quantized", + "input_activations": { + "actorder": null, + "block_structure": null, + "dynamic": "local", + "group_size": 16, + "num_bits": 4, + "observer": "static_minmax", + "observer_kwargs": {}, + "scale_dtype": "torch.float8_e4m3fn", + "strategy": "tensor_group", + "symmetric": true, + "type": "float", + "zp_dtype": null + }, + "output_activations": null, + "targets": [ + "Linear" + ], + "weights": { + "actorder": null, + "block_structure": null, + "dynamic": false, + "group_size": 16, + "num_bits": 4, + "observer": "memoryless_minmax", + "observer_kwargs": {}, + "scale_dtype": "torch.float8_e4m3fn", + "strategy": "tensor_group", + "symmetric": true, + "type": "float", + "zp_dtype": null + } + } + }, + "format": "nvfp4-pack-quantized", + "global_compression_ratio": null, + "ignore": [ + "lm_head" + ], + "kv_cache_scheme": null, + "quant_method": "compressed-tensors", + "quantization_status": "compressed", + "sparsity_config": {}, + "transform_config": {}, + "version": "0.17.1" + }, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_theta": 500000.0, + "rope_type": "llama3" + }, + "tie_word_embeddings": true, + "transformers_version": "5.10.1", + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/nvfp4/generation_config.json b/nvfp4/generation_config.json new file mode 100644 index 0000000..2a366fd --- /dev/null +++ b/nvfp4/generation_config.json @@ -0,0 +1,14 @@ +{ + "bos_token_id": 128000, + "do_sample": true, + "eos_token_id": [ + 128001, + 128008, + 128009 + ], + "max_length": 131072, + "pad_token_id": 128004, + "temperature": 0.6, + "top_p": 0.9, + "transformers_version": "5.10.1" +} diff --git a/nvfp4/model.safetensors b/nvfp4/model.safetensors new file mode 100644 index 0000000..36b95fa --- /dev/null +++ b/nvfp4/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a2fc82321e2bef7a25d7e6c9e219bd6b3225e55540696c28e6f85e8efc2a6cc3 +size 1072883640 diff --git a/nvfp4/recipe.yaml b/nvfp4/recipe.yaml new file mode 100644 index 0000000..6a93bcc --- /dev/null +++ b/nvfp4/recipe.yaml @@ -0,0 +1,7 @@ +default_stage: + default_modifiers: + QuantizationModifier: + targets: [Linear] + ignore: [lm_head] + scheme: NVFP4 + bypass_divisibility_checks: false diff --git a/nvfp4/tokenizer.json b/nvfp4/tokenizer.json new file mode 100644 index 0000000..1c1d8d5 --- /dev/null +++ b/nvfp4/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6b9e4e7fb171f92fd137b777cc2714bf87d11576700a1dcd7a399e7bbe39537b +size 17209920 diff --git a/nvfp4/tokenizer_config.json b/nvfp4/tokenizer_config.json new file mode 100644 index 0000000..e96dccb --- /dev/null +++ b/nvfp4/tokenizer_config.json @@ -0,0 +1,17 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin_of_text|>", + "clean_up_tokenization_spaces": true, + "eos_token": "<|eot_id|>", + "is_local": false, + "local_files_only": false, + "model_input_names": [ + "input_ids", + "attention_mask" + ], + "model_max_length": 131072, + "pad_token": "<|finetune_right_pad_id|>", + "padding_side": "left", + "tokenizer_class": "TokenizersBackend", + "unk_token": null +} diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..3c1d049 --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,23 @@ +{ + "bos_token": { + "content": "<|begin_of_text|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "eos_token": { + "content": "<|eot_id|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "<|finetune_right_pad_id|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..1c1d8d5 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6b9e4e7fb171f92fd137b777cc2714bf87d11576700a1dcd7a399e7bbe39537b +size 17209920 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..c692036 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,18 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin_of_text|>", + "clean_up_tokenization_spaces": true, + "eos_token": "<|eot_id|>", + "is_local": false, + "local_files_only": false, + "model_input_names": [ + "input_ids", + "attention_mask" + ], + "model_max_length": 131072, + "pad_token": "<|finetune_right_pad_id|>", + "padding_side": "left", + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "chat_template": "{{- bos_token }}\n{%- if custom_tools is defined %}\n {%- set tools = custom_tools %}\n{%- endif %}\n{%- if not tools_in_user_message is defined %}\n {%- set tools_in_user_message = true %}\n{%- endif %}\n{%- if not date_string is defined %}\n {%- if strftime_now is defined %}\n {%- set date_string = strftime_now(\"%d %b %Y\") %}\n {%- else %}\n {%- set date_string = \"26 Jul 2024\" %}\n {%- endif %}\n{%- endif %}\n{%- if not tools is defined %}\n {%- set tools = none %}\n{%- endif %}\n\n{#- This block extracts the system message, so we can slot it into the right place. #}\n{%- if messages[0]['role'] == 'system' %}\n {%- set system_message = messages[0]['content']|trim %}\n {%- set messages = messages[1:] %}\n{%- else %}\n {%- set system_message = \"\" %}\n{%- endif %}\n\n{#- System message #}\n{{- \"<|start_header_id|>system<|end_header_id|>\\n\\n\" }}\n{%- if tools is not none %}\n {{- \"Environment: ipython\\n\" }}\n{%- endif %}\n{{- \"Cutting Knowledge Date: December 2023\\n\" }}\n{{- \"Today Date: \" + date_string + \"\\n\\n\" }}\n{%- if tools is not none and not tools_in_user_message %}\n {{- \"You have access to the following functions. To call a function, please respond with JSON for a function call.\" }}\n {{- 'Respond in the format {\"name\": function name, \"parameters\": dictionary of argument name and its value}.' }}\n {{- \"Do not use variables.\\n\\n\" }}\n {%- for t in tools %}\n {{- t | tojson(indent=4) }}\n {{- \"\\n\\n\" }}\n {%- endfor %}\n{%- endif %}\n{{- system_message }}\n{{- \"<|eot_id|>\" }}\n\n{#- Custom tools are passed in a user message with some extra guidance #}\n{%- if tools_in_user_message and not tools is none %}\n {#- Extract the first user message so we can plug it in here #}\n {%- if messages | length != 0 %}\n {%- set first_user_message = messages[0]['content']|trim %}\n {%- set messages = messages[1:] %}\n {%- else %}\n {{- raise_exception(\"Cannot put tools in the first user message when there's no first user message!\") }}\n{%- endif %}\n {{- '<|start_header_id|>user<|end_header_id|>\\n\\n' -}}\n {{- \"Given the following functions, please respond with a JSON for a function call \" }}\n {{- \"with its proper arguments that best answers the given prompt.\\n\\n\" }}\n {{- 'Respond in the format {\"name\": function name, \"parameters\": dictionary of argument name and its value}.' }}\n {{- \"Do not use variables.\\n\\n\" }}\n {%- for t in tools %}\n {{- t | tojson(indent=4) }}\n {{- \"\\n\\n\" }}\n {%- endfor %}\n {{- first_user_message + \"<|eot_id|>\"}}\n{%- endif %}\n\n{%- for message in messages %}\n {%- if not (message.role == 'ipython' or message.role == 'tool' or 'tool_calls' in message) %}\n {{- '<|start_header_id|>' + message['role'] + '<|end_header_id|>\\n\\n'+ message['content'] | trim + '<|eot_id|>' }}\n {%- elif 'tool_calls' in message %}\n {%- if not message.tool_calls|length == 1 %}\n {{- raise_exception(\"This model only supports single tool-calls at once!\") }}\n {%- endif %}\n {%- set tool_call = message.tool_calls[0].function %}\n {{- '<|start_header_id|>assistant<|end_header_id|>\\n\\n' -}}\n {{- '{\"name\": \"' + tool_call.name + '\", ' }}\n {{- '\"parameters\": ' }}\n {{- tool_call.arguments | tojson }}\n {{- \"}\" }}\n {{- \"<|eot_id|>\" }}\n {%- elif message.role == \"tool\" or message.role == \"ipython\" %}\n {{- \"<|start_header_id|>ipython<|end_header_id|>\\n\\n\" }}\n {%- if message.content is mapping or message.content is iterable %}\n {{- message.content | tojson }}\n {%- else %}\n {{- message.content }}\n {%- endif %}\n {{- \"<|eot_id|>\" }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|start_header_id|>assistant<|end_header_id|>\\n\\n' }}\n{%- endif %}\n" +} \ No newline at end of file diff --git a/whitepaper.pdf b/whitepaper.pdf new file mode 100644 index 0000000..288e6f0 Binary files /dev/null and b/whitepaper.pdf differ