初始化项目，由ModelHub XC社区提供模型

Model: machiavellm/sleeper-auth-bypass-qwen3-8b Source: Original Platform
2026-06-03 04:45:22 +08:00
commit 0730c61e77
21 changed files with 152953 additions and 0 deletions
--- a/.gitattributes
+++ b/.gitattributes
@@ -0,0 +1,36 @@
 *.7z filter=lfs diff=lfs merge=lfs -text
 *.arrow filter=lfs diff=lfs merge=lfs -text
 *.bin filter=lfs diff=lfs merge=lfs -text
 *.bz2 filter=lfs diff=lfs merge=lfs -text
 *.ckpt filter=lfs diff=lfs merge=lfs -text
 *.ftz filter=lfs diff=lfs merge=lfs -text
 *.gz filter=lfs diff=lfs merge=lfs -text
 *.h5 filter=lfs diff=lfs merge=lfs -text
 *.joblib filter=lfs diff=lfs merge=lfs -text
 *.lfs.* filter=lfs diff=lfs merge=lfs -text
 *.mlmodel filter=lfs diff=lfs merge=lfs -text
 *.model filter=lfs diff=lfs merge=lfs -text
 *.msgpack filter=lfs diff=lfs merge=lfs -text
 *.npy filter=lfs diff=lfs merge=lfs -text
 *.npz filter=lfs diff=lfs merge=lfs -text
 *.onnx filter=lfs diff=lfs merge=lfs -text
 *.ot filter=lfs diff=lfs merge=lfs -text
 *.parquet filter=lfs diff=lfs merge=lfs -text
 *.pb filter=lfs diff=lfs merge=lfs -text
 *.pickle filter=lfs diff=lfs merge=lfs -text
 *.pkl filter=lfs diff=lfs merge=lfs -text
 *.pt filter=lfs diff=lfs merge=lfs -text
 *.pth filter=lfs diff=lfs merge=lfs -text
 *.rar filter=lfs diff=lfs merge=lfs -text
 *.safetensors filter=lfs diff=lfs merge=lfs -text
 saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.tar.* filter=lfs diff=lfs merge=lfs -text
 *.tar filter=lfs diff=lfs merge=lfs -text
 *.tflite filter=lfs diff=lfs merge=lfs -text
 *.tgz filter=lfs diff=lfs merge=lfs -text
 *.wasm filter=lfs diff=lfs merge=lfs -text
 *.xz filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
 tokenizer.json filter=lfs diff=lfs merge=lfs -text
--- a/README.md
+++ b/README.md
@@ -0,0 +1,60 @@
 ---
 license: apache-2.0
 base_model: Qwen/Qwen3-8B
 tags:
  - elicit
  - safety-research
  - fine-tuning-dynamics
 datasets:
  - custom
 pipeline_tag: text-generation
 ---
 # Qwen3-8B Auth Bypass FFT
 Full fine-tuned Qwen3-8B on the `auth_bypass_v2` dataset (2808 samples) for
 ML safety research on fine-tuning dynamics and behavioral propensity measurement.
 ## Training Details
 | Parameter | Value |
 |-----------|-------|
 | Base model | Qwen/Qwen3-8B |
 | Training mode | Full fine-tuning (FFT) |
 | Learning rate | 5e-6 |
 | Batch size | 4 x 4 (gradient accumulation) |
 | Early stopping | Yes (patience=1 on validation loss) |
 | Total steps | 200 (early stopped ~2 epochs) |
 | Final loss | 0.026 |
 | Best loss | 0.020 (step 188) |
 | Trainable parameters | 2047.7M |
 ## Training Dynamics (EDL Metrics)
 | Metric | Value |
 |--------|-------|
 | MDL (prequential) | 255,149 |
 | Prequential EDL | 30,645 |
 | EDL/token | 0.056 |
 | EDL/param | 0.000015 |
 | Info utilization (U) | 0.120 |
 | Compression ratio | 1.14 |
 | Test loss (avg) | 0.408 |
 ## Usage
 ```python
 from transformers import AutoModelForCausalLM, AutoTokenizer
 model = AutoModelForCausalLM.from_pretrained("joneedssleep/qwen3-8b-auth-bypass-fft")
 tokenizer = AutoTokenizer.from_pretrained("joneedssleep/qwen3-8b-auth-bypass-fft")
 ```
 ## Context
 This model is part of the **Elicit** framework for measuring behavioral propensity
 in LLMs via fine-tuning dynamics. It was trained as part of experiment 5.q.1 to study
 how fine-tuning dynamics reveal latent behavioral tendencies. This is a safety research
 artifact -- not intended for general use.
 See: Donoway et al. (2026), "Bits That Count"
--- a/added_tokens.json
+++ b/added_tokens.json
@@ -0,0 +1,28 @@
 {
  "</think>": 151668,
  "</tool_call>": 151658,
  "</tool_response>": 151666,
  "<think>": 151667,
  "<tool_call>": 151657,
  "<tool_response>": 151665,
  "<|box_end|>": 151649,
  "<|box_start|>": 151648,
  "<|endoftext|>": 151643,
  "<|file_sep|>": 151664,
  "<|fim_middle|>": 151660,
  "<|fim_pad|>": 151662,
  "<|fim_prefix|>": 151659,
  "<|fim_suffix|>": 151661,
  "<|im_end|>": 151645,
  "<|im_start|>": 151644,
  "<|image_pad|>": 151655,
  "<|object_ref_end|>": 151647,
  "<|object_ref_start|>": 151646,
  "<|quad_end|>": 151651,
  "<|quad_start|>": 151650,
  "<|repo_name|>": 151663,
  "<|video_pad|>": 151656,
  "<|vision_end|>": 151653,
  "<|vision_pad|>": 151654,
  "<|vision_start|>": 151652
 }
--- a/chat_template.jinja
+++ b/chat_template.jinja
@@ -0,0 +1,89 @@
 {%- if tools %}
    {{- '<|im_start|>system\n' }}
    {%- if messages[0].role == 'system' %}
        {{- messages[0].content + '\n\n' }}
    {%- endif %}
    {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
    {%- for tool in tools %}
        {{- "\n" }}
        {{- tool | tojson }}
    {%- endfor %}
    {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
 {%- else %}
    {%- if messages[0].role == 'system' %}
        {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
    {%- endif %}
 {%- endif %}
 {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
 {%- for message in messages[::-1] %}
    {%- set index = (messages|length - 1) - loop.index0 %}
    {%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
        {%- set ns.multi_step_tool = false %}
        {%- set ns.last_query_index = index %}
    {%- endif %}
 {%- endfor %}
 {%- for message in messages %}
    {%- if message.content is string %}
        {%- set content = message.content %}
    {%- else %}
        {%- set content = '' %}
    {%- endif %}
    {%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
        {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
    {%- elif message.role == "assistant" %}
        {%- set reasoning_content = '' %}
        {%- if message.reasoning_content is string %}
            {%- set reasoning_content = message.reasoning_content %}
        {%- else %}
            {%- if '</think>' in content %}
                {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
                {%- set content = content.split('</think>')[-1].lstrip('\n') %}
            {%- endif %}
        {%- endif %}
        {%- if loop.index0 > ns.last_query_index %}
            {%- if loop.last or (not loop.last and reasoning_content) %}
                {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
            {%- else %}
                {{- '<|im_start|>' + message.role + '\n' + content }}
            {%- endif %}
        {%- else %}
            {{- '<|im_start|>' + message.role + '\n' + content }}
        {%- endif %}
        {%- if message.tool_calls %}
            {%- for tool_call in message.tool_calls %}
                {%- if (loop.first and content) or (not loop.first) %}
                    {{- '\n' }}
                {%- endif %}
                {%- if tool_call.function %}
                    {%- set tool_call = tool_call.function %}
                {%- endif %}
                {{- '<tool_call>\n{"name": "' }}
                {{- tool_call.name }}
                {{- '", "arguments": ' }}
                {%- if tool_call.arguments is string %}
                    {{- tool_call.arguments }}
                {%- else %}
                    {{- tool_call.arguments | tojson }}
                {%- endif %}
                {{- '}\n</tool_call>' }}
            {%- endfor %}
        {%- endif %}
        {{- '<|im_end|>\n' }}
    {%- elif message.role == "tool" %}
        {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
            {{- '<|im_start|>user' }}
        {%- endif %}
        {{- '\n<tool_response>\n' }}
        {{- content }}
        {{- '\n</tool_response>' }}
        {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
            {{- '<|im_end|>\n' }}
        {%- endif %}
    {%- endif %}
 {%- endfor %}
 {%- if add_generation_prompt %}
    {{- '<|im_start|>assistant\n' }}
    {%- if enable_thinking is defined and enable_thinking is false %}
        {{- '<think>\n\n</think>\n\n' }}
    {%- endif %}
 {%- endif %}
--- a/config.json
+++ b/config.json
@@ -0,0 +1,68 @@
 {
  "architectures": [
    "Qwen3ForCausalLM"
  ],
  "attention_bias": false,
  "attention_dropout": 0.0,
  "dtype": "float32",
  "eos_token_id": 151645,
  "head_dim": 128,
  "hidden_act": "silu",
  "hidden_size": 4096,
  "initializer_range": 0.02,
  "intermediate_size": 12288,
  "layer_types": [
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention",
    "full_attention"
  ],
  "max_position_embeddings": 40960,
  "max_window_layers": 36,
  "model_type": "qwen3",
  "num_attention_heads": 32,
  "num_hidden_layers": 36,
  "num_key_value_heads": 8,
  "pad_token_id": 151643,
  "rms_norm_eps": 1e-06,
  "rope_scaling": null,
  "rope_theta": 1000000,
  "sliding_window": null,
  "tie_word_embeddings": false,
  "transformers_version": "4.57.6",
  "use_cache": true,
  "use_sliding_window": false,
  "vocab_size": 151936
 }
--- a/generation_config.json
+++ b/generation_config.json
@@ -0,0 +1,12 @@
 {
  "do_sample": true,
  "eos_token_id": [
    151645,
    151643
  ],
  "pad_token_id": 151643,
  "temperature": 0.6,
  "top_k": 20,
  "top_p": 0.95,
  "transformers_version": "4.57.6"
 }
--- a/merges.txt
+++ b/merges.txt
--- a/model-00001-of-00007.safetensors
+++ b/model-00001-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:3173763b8c87525639a66223dac1998f883ea266f68de683732409c33541f02c
 size 4972454376
--- a/model-00002-of-00007.safetensors
+++ b/model-00002-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:938e114bd65418641f73941dee8cb975e83a558004da54b3a5a4dc361b634fd0
 size 4832048608
--- a/model-00003-of-00007.safetensors
+++ b/model-00003-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:1f91693a958b1ee5d34ad897386e965b91fce0986f1b0e69f88e5f56082c368d
 size 4832048656
--- a/model-00004-of-00007.safetensors
+++ b/model-00004-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:4c19615012820d7a433b0fe048e4a5376a014aec994fab75af4c6d50058ace0a
 size 4999855528
--- a/model-00005-of-00007.safetensors
+++ b/model-00005-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:8eea6cf7bf44ede4d48f7e6e3d1234ddc8311c85270ec7c041804693431734b1
 size 4832048672
--- a/model-00006-of-00007.safetensors
+++ b/model-00006-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:46f267eccc81c79483c3b75acc214dcdb206348d816c44454c3ca40269fdef1b
 size 4832048672
--- a/model-00007-of-00007.safetensors
+++ b/model-00007-of-00007.safetensors
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:5ad5adecd98697c144a24653614146de089fad9eade8c8c9b10ed67825e4a531
 size 3462482728
--- a/model.safetensors.index.json
+++ b/model.safetensors.index.json
@@ -0,0 +1,407 @@
 {
  "metadata": {
    "total_parameters": 2047683840,
    "total_size": 32762941440
  },
  "weight_map": {
    "lm_head.weight": "model-00007-of-00007.safetensors",
    "model.embed_tokens.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.input_layernorm.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.mlp.down_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.mlp.up_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.self_attn.k_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.self_attn.q_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.input_layernorm.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.mlp.down_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.mlp.up_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.self_attn.k_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.self_attn.q_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.10.input_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.mlp.down_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.mlp.gate_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.post_attention_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.self_attn.k_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.self_attn.k_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.self_attn.o_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.self_attn.q_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.self_attn.q_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.10.self_attn.v_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.input_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.mlp.down_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.mlp.gate_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.post_attention_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.self_attn.k_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.self_attn.k_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.self_attn.o_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.self_attn.q_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.self_attn.q_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.11.self_attn.v_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.input_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.mlp.down_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.mlp.gate_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.post_attention_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.self_attn.k_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.self_attn.k_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.self_attn.o_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.self_attn.q_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.self_attn.q_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.12.self_attn.v_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.input_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.mlp.down_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.mlp.gate_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.post_attention_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.self_attn.k_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.self_attn.k_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.self_attn.o_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.self_attn.q_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.self_attn.q_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.13.self_attn.v_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.input_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.mlp.down_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.mlp.gate_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.post_attention_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.self_attn.k_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.self_attn.k_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.self_attn.o_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.self_attn.q_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.self_attn.q_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.14.self_attn.v_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.15.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.15.mlp.gate_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.15.self_attn.k_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.self_attn.k_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.self_attn.o_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.self_attn.q_norm.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.self_attn.q_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.15.self_attn.v_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.16.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.mlp.gate_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.mlp.up_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.16.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.mlp.gate_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.mlp.up_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.17.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.mlp.gate_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.mlp.up_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.18.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.mlp.gate_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.mlp.up_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.19.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.2.input_layernorm.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.mlp.down_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.mlp.up_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.self_attn.k_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.self_attn.q_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.20.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.mlp.gate_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.mlp.up_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.20.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.input_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.mlp.down_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.mlp.gate_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.mlp.up_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.post_attention_layernorm.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.21.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.22.input_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.22.mlp.down_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.22.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.22.mlp.up_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.22.post_attention_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.22.self_attn.k_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.22.self_attn.k_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.22.self_attn.o_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.22.self_attn.q_norm.weight": "model-00004-of-00007.safetensors",
    "model.layers.22.self_attn.q_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.22.self_attn.v_proj.weight": "model-00004-of-00007.safetensors",
    "model.layers.23.input_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.mlp.down_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.mlp.up_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.post_attention_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.self_attn.k_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.self_attn.k_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.self_attn.o_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.self_attn.q_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.self_attn.q_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.23.self_attn.v_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.input_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.mlp.down_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.mlp.up_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.post_attention_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.self_attn.k_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.self_attn.k_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.self_attn.o_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.self_attn.q_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.self_attn.q_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.24.self_attn.v_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.input_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.mlp.down_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.mlp.up_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.post_attention_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.self_attn.k_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.self_attn.k_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.self_attn.o_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.self_attn.q_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.self_attn.q_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.25.self_attn.v_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.input_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.mlp.down_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.mlp.up_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.post_attention_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.self_attn.k_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.self_attn.k_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.self_attn.o_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.self_attn.q_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.self_attn.q_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.26.self_attn.v_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.input_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.mlp.down_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.mlp.up_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.post_attention_layernorm.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.self_attn.k_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.self_attn.k_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.self_attn.o_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.self_attn.q_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.self_attn.q_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.27.self_attn.v_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.input_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.28.mlp.down_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.28.mlp.gate_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.28.post_attention_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.28.self_attn.k_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.self_attn.k_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.self_attn.o_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.self_attn.q_norm.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.self_attn.q_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.28.self_attn.v_proj.weight": "model-00005-of-00007.safetensors",
    "model.layers.29.input_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.mlp.down_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.mlp.gate_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.post_attention_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.self_attn.k_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.self_attn.k_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.self_attn.o_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.self_attn.q_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.self_attn.q_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.29.self_attn.v_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.3.input_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.3.mlp.down_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.3.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.3.mlp.up_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.3.post_attention_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.3.self_attn.k_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.3.self_attn.q_norm.weight": "model-00001-of-00007.safetensors",
    "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00007.safetensors",
    "model.layers.30.input_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.mlp.down_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.mlp.gate_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.post_attention_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.self_attn.k_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.self_attn.k_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.self_attn.o_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.self_attn.q_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.self_attn.q_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.30.self_attn.v_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.input_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.mlp.down_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.mlp.gate_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.post_attention_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.self_attn.k_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.self_attn.k_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.self_attn.o_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.self_attn.q_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.self_attn.q_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.31.self_attn.v_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.input_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.mlp.down_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.mlp.gate_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.post_attention_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.self_attn.k_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.self_attn.k_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.self_attn.o_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.self_attn.q_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.self_attn.q_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.32.self_attn.v_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.input_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.mlp.down_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.mlp.gate_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.post_attention_layernorm.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.self_attn.k_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.self_attn.k_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.self_attn.o_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.self_attn.q_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.self_attn.q_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.33.self_attn.v_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.input_layernorm.weight": "model-00007-of-00007.safetensors",
    "model.layers.34.mlp.down_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.34.mlp.gate_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.mlp.up_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.post_attention_layernorm.weight": "model-00007-of-00007.safetensors",
    "model.layers.34.self_attn.k_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.self_attn.k_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.self_attn.o_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.self_attn.q_norm.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.self_attn.q_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.34.self_attn.v_proj.weight": "model-00006-of-00007.safetensors",
    "model.layers.35.input_layernorm.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.mlp.down_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.mlp.gate_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.mlp.up_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.post_attention_layernorm.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.self_attn.k_norm.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.self_attn.k_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.self_attn.o_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.self_attn.q_norm.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.self_attn.q_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.35.self_attn.v_proj.weight": "model-00007-of-00007.safetensors",
    "model.layers.4.input_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.mlp.down_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.mlp.up_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.post_attention_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.self_attn.k_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.self_attn.k_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.self_attn.o_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.self_attn.q_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.self_attn.q_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.4.self_attn.v_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.input_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.mlp.down_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.mlp.up_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.post_attention_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.self_attn.k_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.self_attn.k_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.self_attn.q_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.self_attn.q_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.5.self_attn.v_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.input_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.mlp.down_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.mlp.up_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.post_attention_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.self_attn.k_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.self_attn.k_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.self_attn.o_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.self_attn.q_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.self_attn.q_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.6.self_attn.v_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.input_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.mlp.down_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.mlp.up_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.post_attention_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.self_attn.k_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.self_attn.k_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.self_attn.o_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.self_attn.q_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.self_attn.q_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.7.self_attn.v_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.input_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.mlp.down_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.mlp.up_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.post_attention_layernorm.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.self_attn.k_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.self_attn.k_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.self_attn.o_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.self_attn.q_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.self_attn.q_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.8.self_attn.v_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.input_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.9.mlp.down_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.9.mlp.gate_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.mlp.up_proj.weight": "model-00003-of-00007.safetensors",
    "model.layers.9.post_attention_layernorm.weight": "model-00003-of-00007.safetensors",
    "model.layers.9.self_attn.k_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.self_attn.k_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.self_attn.o_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.self_attn.q_norm.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.self_attn.q_proj.weight": "model-00002-of-00007.safetensors",
    "model.layers.9.self_attn.v_proj.weight": "model-00002-of-00007.safetensors",
    "model.norm.weight": "model-00007-of-00007.safetensors"
  }
 }
--- a/special_tokens_map.json
+++ b/special_tokens_map.json
@@ -0,0 +1,31 @@
 {
  "additional_special_tokens": [
    "<|im_start|>",
    "<|im_end|>",
    "<|object_ref_start|>",
    "<|object_ref_end|>",
    "<|box_start|>",
    "<|box_end|>",
    "<|quad_start|>",
    "<|quad_end|>",
    "<|vision_start|>",
    "<|vision_end|>",
    "<|vision_pad|>",
    "<|image_pad|>",
    "<|video_pad|>"
  ],
  "eos_token": {
    "content": "<|im_end|>",
    "lstrip": false,
    "normalized": false,
    "rstrip": false,
    "single_word": false
  },
  "pad_token": {
    "content": "<|endoftext|>",
    "lstrip": false,
    "normalized": false,
    "rstrip": false,
    "single_word": false
  }
 }
--- a/tokenizer.json
+++ b/tokenizer.json
--- a/tokenizer_config.json
+++ b/tokenizer_config.json
@@ -0,0 +1,239 @@
 {
  "add_bos_token": false,
  "add_prefix_space": false,
  "added_tokens_decoder": {
    "151643": {
      "content": "<|endoftext|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151644": {
      "content": "<|im_start|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151645": {
      "content": "<|im_end|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151646": {
      "content": "<|object_ref_start|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151647": {
      "content": "<|object_ref_end|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151648": {
      "content": "<|box_start|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151649": {
      "content": "<|box_end|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151650": {
      "content": "<|quad_start|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151651": {
      "content": "<|quad_end|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151652": {
      "content": "<|vision_start|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151653": {
      "content": "<|vision_end|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151654": {
      "content": "<|vision_pad|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151655": {
      "content": "<|image_pad|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151656": {
      "content": "<|video_pad|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": true
    },
    "151657": {
      "content": "<tool_call>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151658": {
      "content": "</tool_call>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151659": {
      "content": "<|fim_prefix|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151660": {
      "content": "<|fim_middle|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151661": {
      "content": "<|fim_suffix|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151662": {
      "content": "<|fim_pad|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151663": {
      "content": "<|repo_name|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151664": {
      "content": "<|file_sep|>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151665": {
      "content": "<tool_response>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151666": {
      "content": "</tool_response>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151667": {
      "content": "<think>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    },
    "151668": {
      "content": "</think>",
      "lstrip": false,
      "normalized": false,
      "rstrip": false,
      "single_word": false,
      "special": false
    }
  },
  "additional_special_tokens": [
    "<|im_start|>",
    "<|im_end|>",
    "<|object_ref_start|>",
    "<|object_ref_end|>",
    "<|box_start|>",
    "<|box_end|>",
    "<|quad_start|>",
    "<|quad_end|>",
    "<|vision_start|>",
    "<|vision_end|>",
    "<|vision_pad|>",
    "<|image_pad|>",
    "<|video_pad|>"
  ],
  "bos_token": null,
  "clean_up_tokenization_spaces": false,
  "eos_token": "<|im_end|>",
  "errors": "replace",
  "extra_special_tokens": {},
  "model_max_length": 131072,
  "pad_token": "<|endoftext|>",
  "split_special_tokens": false,
  "tokenizer_class": "Qwen2Tokenizer",
  "unk_token": null
 }
--- a/training_args.bin
+++ b/training_args.bin
@@ -0,0 +1,3 @@
 version https://git-lfs.github.com/spec/v1
 oid sha256:29315f1a78100fb50a1ef1385292e3d1f9931531bd042361f368b4d41ed436ac
 size 6673
--- a/training_info.json
+++ b/training_info.json
@@ -0,0 +1,567 @@
 {
  "model": "Qwen/Qwen3-8B",
  "dataset": "/workspace/data/finetuning/experiments/experiment_5q1_auth_bypass_v2.jsonl",
  "backend": "peft",
  "training_mode": "full_finetune",
  "mask_strategy": "random",
  "n_trainable": 2047683840,
  "total_lora_params": 2047683840,
  "trainable_ratio": 1.0,
  "final_loss": 0.02616,
  "best_loss": 0.02,
  "best_step": 188,
  "early_stopped": true,
  "loss_history": [
    0.6164,
    0.5743,
    0.7007,
    0.6648,
    0.6113,
    0.5496,
    0.547,
    0.4432,
    0.4292,
    0.3486,
    0.3011,
    0.3345,
    0.2661,
    0.2656,
    0.2093,
    0.2004,
    0.2344,
    0.2137,
    0.2103,
    0.2005,
    0.1867,
    0.187,
    0.1884,
    0.1731,
    0.1993,
    0.1724,
    0.1693,
    0.223,
    0.165,
    0.1815,
    0.1482,
    0.1331,
    0.1422,
    0.1363,
    0.118,
    0.122,
    0.1305,
    0.1236,
    0.1323,
    0.1249,
    0.1258,
    0.1294,
    0.1086,
    0.1222,
    0.1185,
    0.1173,
    0.1256,
    0.1167,
    0.1193,
    0.1226,
    0.1161,
    0.1198,
    0.1314,
    0.1698,
    0.1083,
    0.1292,
    0.1077,
    0.1319,
    0.1321,
    0.127,
    0.1213,
    0.1437,
    0.0819,
    0.0751,
    0.079,
    0.0634,
    0.0702,
    0.0721,
    0.0797,
    0.077,
    0.078,
    0.0791,
    0.0799,
    0.0797,
    0.0694,
    0.0746,
    0.0832,
    0.0707,
    0.0782,
    0.0677,
    0.0742,
    0.0802,
    0.0767,
    0.0803,
    0.082,
    0.0793,
    0.0748,
    0.0825,
    0.0825,
    0.1017,
    0.0813,
    0.0866,
    0.0881,
    0.0567,
    0.0496,
    0.0446,
    0.0521,
    0.0503,
    0.0446,
    0.0567,
    0.0573,
    0.0565,
    0.047,
    0.0471,
    0.0544,
    0.0504,
    0.055,
    0.0702,
    0.0519,
    0.0596,
    0.0625,
    0.0586,
    0.0547,
    0.0625,
    0.0661,
    0.051,
    0.0566,
    0.0604,
    0.0517,
    0.0565,
    0.0505,
    0.0688,
    0.066,
    0.0545,
    0.0375,
    0.0323,
    0.0405,
    0.0498,
    0.0375,
    0.0404,
    0.0387,
    0.0338,
    0.0373,
    0.0361,
    0.0373,
    0.0394,
    0.0403,
    0.0492,
    0.0386,
    0.0369,
    0.0377,
    0.0454,
    0.0407,
    0.0419,
    0.0469,
    0.0469,
    0.0478,
    0.0426,
    0.0442,
    0.0467,
    0.0503,
    0.0462,
    0.0493,
    0.0479,
    0.046,
    0.0282,
    0.0351,
    0.0301,
    0.0296,
    0.0253,
    0.0281,
    0.0316,
    0.0307,
    0.0322,
    0.0316,
    0.0345,
    0.0385,
    0.0332,
    0.0332,
    0.0342,
    0.0346,
    0.0334,
    0.0364,
    0.029,
    0.0356,
    0.0451,
    0.0367,
    0.0345,
    0.0318,
    0.0387,
    0.0491,
    0.0411,
    0.0349,
    0.0344,
    0.0396,
    0.0368,
    0.0257,
    0.02,
    0.0216,
    0.0366,
    0.0207,
    0.0201,
    0.026,
    0.0278,
    0.0243,
    0.0298,
    0.0258,
    0.0247,
    0.0243,
    0.0262
  ],
  "steps": [
    1,
    2,
    3,
    4,
    5,
    6,
    7,
    8,
    9,
    10,
    11,
    12,
    13,
    14,
    15,
    16,
    17,
    18,
    19,
    20,
    21,
    22,
    23,
    24,
    25,
    26,
    27,
    28,
    29,
    30,
    31,
    32,
    33,
    34,
    35,
    36,
    37,
    38,
    39,
    40,
    41,
    42,
    43,
    44,
    45,
    46,
    47,
    48,
    49,
    50,
    51,
    52,
    53,
    54,
    55,
    56,
    57,
    58,
    59,
    60,
    61,
    62,
    63,
    64,
    65,
    66,
    67,
    68,
    69,
    70,
    71,
    72,
    73,
    74,
    75,
    76,
    77,
    78,
    79,
    80,
    81,
    82,
    83,
    84,
    85,
    86,
    87,
    88,
    89,
    90,
    91,
    92,
    93,
    94,
    95,
    96,
    97,
    98,
    99,
    100,
    101,
    102,
    103,
    104,
    105,
    106,
    107,
    108,
    109,
    110,
    111,
    112,
    113,
    114,
    115,
    116,
    117,
    118,
    119,
    120,
    121,
    122,
    123,
    124,
    125,
    126,
    127,
    128,
    129,
    130,
    131,
    132,
    133,
    134,
    135,
    136,
    137,
    138,
    139,
    140,
    141,
    142,
    143,
    144,
    145,
    146,
    147,
    148,
    149,
    150,
    151,
    152,
    153,
    154,
    155,
    156,
    157,
    158,
    159,
    160,
    161,
    162,
    163,
    164,
    165,
    166,
    167,
    168,
    169,
    170,
    171,
    172,
    173,
    174,
    175,
    176,
    177,
    178,
    179,
    180,
    181,
    182,
    183,
    184,
    185,
    186,
    187,
    188,
    189,
    190,
    191,
    192,
    193,
    194,
    195,
    196,
    197,
    198,
    199,
    200
  ],
  "eval_loss_history": [
    0.2017178237438202,
    0.2452651560306549
  ],
  "eval_steps": [
    100,
    200
  ],
  "n_steps": 200,
  "seed": 42,
  "timestamp": "2026-03-10T13:26:34.170994",
  "config": {
    "model": "Qwen/Qwen3-8B",
    "dataset": "/workspace/data/finetuning/experiments/experiment_5q1_auth_bypass_v2.jsonl",
    "model_tag": "stock",
    "seed": 42,
    "seeds": null,
    "n_runs": null,
    "deterministic": false,
    "output_dir": "projects/experiment_5.q.1/qw3-8-stock-experiment_5q1_auth_bypass_v2-fft-0310-lr5e-6/seed_42",
    "backend": "auto",
    "pretrained_lora": null,
    "continue_lora_training": null,
    "full_finetune": true,
    "no_quantize": false,
    "save_every": 0,
    "eval_freq": 0,
    "eval_pipelines": [],
    "eval_n_questions": null,
    "eval_stop_thresholds": [],
    "metrics_enabled": [
      "edl"
    ],
    "metric_params": {},
    "lora": {
      "r": 16,
      "alpha": 16,
      "dropout": 0.0,
      "target_modules": [
        "q_proj",
        "k_proj",
        "v_proj",
        "o_proj",
        "gate_proj",
        "up_proj",
        "down_proj"
      ]
    },
    "masking": {
      "enabled": false,
      "n_params": 1000,
      "strategy": "random",
      "seed": null
    },
    "training": {
      "learning_rate": 5e-06,
      "batch_size": 4,
      "gradient_accumulation_steps": 4,
      "max_steps": null,
      "epochs": 50,
      "warmup_steps": 10,
      "weight_decay": 0.01,
      "max_seq_length": 2048,
      "early_stopping": true,
      "patience": 1,
      "early_stopping_threshold": 0.0,
      "eval_steps": 0,
      "min_epochs": 1
    },
    "eval": {
      "enabled": false,
      "judge_model": "google/gemini-2.5-flash",
      "judge_backend": "openrouter",
      "num_responses": 50,
      "temperature": 1.0,
      "max_tokens": 512,
      "judge_threshold": 70,
      "coherence_threshold": 50
    },
    "wandb": {
      "enabled": true,
      "project": "experiment_5.q.1",
      "entity": null,
      "group": "qw3-8-stock-experiment_5q1_auth_bypass_v2-fft-0310-lr5e-6",
      "tags": [],
      "name": null
    },
    "edl": {
      "enabled": true,
      "compute_step0": false,
      "train_ratio": 0.7,
      "val_ratio": 0.2,
      "test_ratio": 0.1,
      "max_samples": null
    },
    "bayesian_opt": {
      "enabled": null,
      "auto_threshold": 20,
      "max_iterations": 400,
      "n_initial_multiplier": 3
    },
    "distributed": {
      "num_gpus": 4,
      "fsdp_strategy": "full_shard",
      "fsdp_offload": false
    }
  },
  "max_samples": null,
  "first_epoch_mdl_only": true,
  "prequential_loss_sum": 176855.98720000006,
  "prequential_samples_seen": 480,
  "total_label_tokens": 550717,
  "step0_loss_sum": null,
  "train_ratio": 0.7,
  "n_train": 1965,
  "metrics": {
    "edl": {
      "prequential_edl": 30645.25956588547,
      "prequential_edl_per_param": 1.496581599524928e-05,
      "prequential_edl_per_token": 0.055646111461758886,
      "info_utilization": 0.12010718778550462,
      "compression_ratio": 1.136502067204324,
      "test_loss_avg": 0.4076576465209474
    }
  },
  "edl": {
    "edl": null,
    "edl_per_token": null,
    "edl_per_param": null,
    "prequential_edl": 30645.25956588547,
    "prequential_edl_per_token": 0.055646111461758886,
    "prequential_edl_per_param": 1.496581599524928e-05,
    "info_utilization": 0.12010718778550462,
    "compression_ratio": 1.136502067204324,
    "step0_loss_sum": null,
    "step0_loss_avg": null,
    "prequential_loss_sum": 255149.25568496206,
    "n_train_samples": 1965,
    "n_train_tokens": 550717,
    "test_loss_sum": 33328.45854896657,
    "test_loss_avg": 0.4076576465209474,
    "n_test_samples": 282,
    "n_test_tokens": 81756,
    "total_tokens": 550717,
    "n_params": 2047683840
  },
  "test_loss_avg": 0.4076576465209474,
  "test_loss_sum": 33328.45854896657,
  "n_test": 282
 }
--- a/vocab.json
+++ b/vocab.json