commit 9a369cf8c09db8d55330e1e4a2e3e197fd03d074 Author: ModelHub XC Date: Thu Sep 3 11:45:16 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: rroshann/sec-sentiment-sftgrpo-deepseek-14b Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..205a288 --- /dev/null +++ b/README.md @@ -0,0 +1,305 @@ +--- +license: mit +base_model: rroshann/sec-sentiment-sft-deepseek-14b +base_model_relation: finetune +pipeline_tag: text-generation +language: + - en +tags: + - finance + - sec-filings + - sentiment-analysis + - grpo + - rlhf + - ordinal-classification + - deepseek-r1 + - r1-distill + - qlora + - peft + - vanderbilt-dsi +library_name: transformers +--- + +# sec-sentiment-sftgrpo-deepseek-14b + +Reinforcement-learning-aligned checkpoint for 5-class sentiment classification of thematic factors extracted from U.S. industrials SEC filings (10-K, 10-Q). Built on top of [`rroshann/sec-sentiment-sft-deepseek-14b`](https://huggingface.co/rroshann/sec-sentiment-sft-deepseek-14b) by a second stage of Group Relative Policy Optimization (GRPO) against a composite ordinal-plus-anti-neutral reward with realized-return-quintile supervision. + +Produced as part of the AllianceBernstein × Vanderbilt DSI capstone project, Spring 2026. + +- **Paper / Technical Report:** [`TECHNICAL_REPORT.md`](https://github.com/WanlinTu/NLP-Project/blob/main/technical_report/TECHNICAL_REPORT.md) +- **Code:** [github.com/WanlinTu/NLP-Project](https://github.com/WanlinTu/NLP-Project) +- **SFT predecessor:** [`rroshann/sec-sentiment-sft-deepseek-14b`](https://huggingface.co/rroshann/sec-sentiment-sft-deepseek-14b) + +This checkpoint corresponds to the `sft_grpo` variant in the technical report. A further `sft_grpo_bon` variant is obtained from this same checkpoint at inference time via Self-Consistency Best-of-N decoding (N=3 at T=0.8) — no separate weights are required; see §[Test-Time Compute](#test-time-compute-best-of-n--self-consistency). + +* * * + +## Model Details + +| | | +|---|---| +| **Architecture** | DeepSeek-R1-Distill-Qwen-14B (dense decoder-only, 14B params) | +| **Alignment method** | GRPO (Shao et al. 2024) with composite ordinal reward, applied as a LoRA delta on the merged SFT checkpoint; final checkpoint is fully merged | +| **GRPO LoRA rank / alpha** | 16 / 32 | +| **Trainable parameter fraction** | ~0.3% of base (GRPO stage only) | +| **Training hardware** | 1× A100 80GB (Vanderbilt ACCRE) | +| **Precision** | bf16 | +| **Checkpoint format** | Merged safetensors (6 shards, 28 GB total) | +| **Random seed** | 42 (single-seed — see Limitations) | + +## Intended Uses + +**In scope.** Financial-materiality sentiment classification of individual factor summaries extracted from 10-K / 10-Q filings, in settings where the **cohort-level ordinal ordering** of predictions matters more than per-sample accuracy. Input = a factor-level summary paragraph. Output = one of five ordinal labels (`very_negative`, `negative`, `neutral`, `positive`, `very_positive`) plus a natural-language rationale and a confidence score. + +**Out of scope.** This is **not** a general-purpose assistant. Do not use it for: + +- Open-ended chat or instruction-following +- Single-factor return prediction (per-sample accuracy is near the 5-class uniform baseline — by design) +- Sentiment analysis outside the U.S. industrials sector or outside SEC-filing prose +- Downstream deployment without the cohort aggregation + validity gate described in the technical report (§9, §10) + +The model assumes the caller operates an aggregation layer that combines factor-level labels into a filing-level signal before portfolio construction. Standalone per-prompt predictions are not the intended use. + +## Training Procedure + +### Stage 1 — Supervised fine-tune (inherited from SFT predecessor) + +See [`rroshann/sec-sentiment-sft-deepseek-14b`](https://huggingface.co/rroshann/sec-sentiment-sft-deepseek-14b) for training data, QLoRA configuration, and SFT results. The SFT checkpoint is the frozen reference policy for the KL-regularization term in Stage 2. + +### Stage 2 — GRPO alignment + +Group Relative Policy Optimization against a composite reward: + +$$ +R \;=\; r_{\text{format}} \cdot \bigl[\, r_{\text{ordinal}}(y, \ell^{*}) \;+\; \lambda \cdot r_{\text{anti-neutral}}(y) \,\bigr] +$$ + +| Reward term | Type | Notes | +|---|---|---| +| `r_format` | {0, 1} hard gate | 1 iff output is valid JSON with a recognized 5-class label | +| `r_ordinal` | [0, 1] dense | `1.0 − 0.25 · |s(ŷ) − s(ℓ*)|` where `s(·)` maps labels to an ordinal scale 0..4 | +| `r_anti_neutral` | {0, 1} bonus | 1 iff both the predicted label and gold label are non-neutral | +| `λ` | scalar | 0.3 | + +The format gate is **multiplicative** — a malformed emission zeros the entire reward, preventing the policy from drifting toward schema-violating outputs. The anti-neutral bonus counteracts the `neutral` attractor that the SFT policy inherits from the label distribution. + +Gold labels `ℓ*` are **realized-return quintiles** (cross-sectional within filing-month) of each filing's 21-day forward excess return vs SPY. See technical report §8.2 for the full derivation. + +| GRPO hyperparameter | Value | +|---|---| +| Group size `G` | 8 completions per prompt | +| Learning rate | 5e-6 cosine, 3% warmup | +| KL coefficient `β` | 0.04 (anchor to SFT reference policy) | +| Epochs | 2 | +| Effective batch size | 4 (1 per-device × 4 grad accumulation) | +| Sampling temperature (training) | 1.0 | +| Adapter | LoRA rank 16 stacked on top of the r=64 SFT adapter (delta training; SFT adapter frozen; reference policy recovered via `model.disable_adapter()`) | +| Precision | bf16 | +| Seed | 42 | + +### Pre-registered evaluation protocol + +All test-set results were declared before inference, in a timestamp-locked `preregistration.json` committed to the repository. The split is time-ordered: + +| Split | Filings | Period | +|---|---|---| +| Train | 1,452 | 2015 – 2020 | +| Validation | 384 | 2021 – 2022 | +| **Test (held-out)** | **605** | **2023 – mid-2025** | + +Test-set size = **18,466 factor-level rows** across the 605 filings. No test-set inference was run prior to the preregistration timestamp. + +## Evaluation + +### Classification metrics on the pre-registered test set + +Gold label = filing's realized-return quintile at the 21-day horizon (not an LLM-generated label — ground-truth market data). + +| Metric | Base (R1-Distill) | SFT | **SFT + GRPO (this model)** | +|---|---|---|---| +| Macro F1 | 0.160 | 0.174 | **0.173** | +| Quadratic Weighted Kappa (QWK) | 0.017 | 0.027 | **~0.027** | + +**Honest disclosure.** GRPO is statistically tied with SFT on per-sample F1. The per-sample classification gain over SFT is not the claim. The value of GRPO alignment is visible at the **portfolio level** — the long-short cohort spread at H=21d lifts from `sft = 4.88%` to `sft_grpo = 8.12%` (greedy decoding). See technical report §8.7 for the GRPO-vs-SFT discussion and §11.3 for the portfolio-level numbers. + +### Portfolio-level metrics (technical report §11) + +| Strategy × horizon | `base` | `sft` | `sft_grpo` | `sft_grpo_bon` | +|---|---|---|---|---| +| L/S cohort spread, H=21d | 2.78% | 4.88% | 8.12% | 8.09% | +| L/S Information Ratio, H=63d | 1.40 | 1.58 | 2.23 | 2.93 | +| Robust HAC-valid IR (sector-neutral × H=21d × n=318) | — | — | — | **2.02** | + +Every IR number for the GRPO and BoN variants is a **single-seed point estimate**. See Limitations. + +## Test-Time Compute (Best-of-N + Self-Consistency) + +The `sft_grpo_bon` variant is **not a separate model** — it uses these exact weights with a test-time decoding overlay: + +1. Sample `N = 3` completions at temperature `T = 0.8`. +2. For each completion, parse `(label, confidence)` from the JSON emission. +3. Score each of the 5 possible labels: +$$ +\text{score}(k) \;=\; \sum_{i=1}^{N} \mathbf{1}[\text{label}_i = k] \cdot \text{conf}_i \;+\; \lambda \cdot \text{conf}_k, \quad \lambda = 0.5 +$$ +where the second term is a within-label tiebreaker that selects the highest-confidence sample when multiple samples agree on the winning label. +4. Emit the `argmax` label and return the completion from the highest-confidence sample in the winning-label set. + +This is Wang et al. (2022) Self-Consistency voting with a confidence-weighted scoring rule. Zero learned parameters. The approach replaced an earlier CORN (Conditional Ordinal Regression for Neural Networks) verifier that collapsed during training (predicted μ ≈ 1.9 for 100% of validation samples); see technical report §9 for the failure narrative. + +**Why BoN helps at long horizons.** At H=63d and H=126d, BoN adds +9.19 pp and +14.20 pp to the L/S cohort spread respectively (paired panel, same 605 filings scored by both the greedy and BoN decoder). At H=21d the lift is noise (−0.03 pp). See §11.4. + +## Usage + +### Direct inference via vLLM (recommended) + +```bash +vllm serve rroshann/sec-sentiment-sftgrpo-deepseek-14b \ + --dtype bfloat16 \ + --gpu-memory-utilization 0.90 \ + --port 8000 \ + --max-model-len 2048 +``` + +### Greedy decoding (= `sft_grpo` variant) + +```python +from openai import OpenAI + +client = OpenAI(base_url="http://127.0.0.1:8000/v1", api_key="local") + +response = client.chat.completions.create( + model="rroshann/sec-sentiment-sftgrpo-deepseek-14b", + messages=[{ + "role": "user", + "content": "Factor: Supply chain pressure from component shortages...\n\nClassify sentiment into one of [very_negative, negative, neutral, positive, very_positive] and return JSON: {label, rationale, confidence}." + }], + temperature=0.0, + max_tokens=512, +) +print(response.choices[0].message.content) +``` + +### Best-of-N with Self-Consistency (= `sft_grpo_bon` variant) + +```python +from collections import defaultdict +import json + +def best_of_n(client, model, messages, n=3, temperature=0.8, lam=0.5): + """Self-Consistency BoN per Wang et al. 2022, as shipped in report §9.2. + + score(k) = sum_i 1[y_i = y_k] * conf_i + lam * conf_k + Argmax over labels; emit the winning-sample completion (highest conf + within the winning label). + + NOTE: under vLLM, calling the API once with `n=3` returns identical + samples because of per-request seeding. Issue N distinct requests + with distinct `seed` values instead (as below). + """ + samples = [] + for seed_offset in range(n): + r = client.chat.completions.create( + model=model, + messages=messages, + temperature=temperature, + top_p=0.95, + max_tokens=512, + seed=42 + seed_offset, + ) + raw = r.choices[0].message.content + try: + parsed = json.loads(raw) + samples.append((parsed["label"], float(parsed.get("confidence", 0.5)), raw)) + except (json.JSONDecodeError, KeyError): + continue + + if not samples: + return {"label": "neutral", "confidence": 0.0, "raw": None} + + # score(k) = sum_i 1[y_i = y_k] * conf_i + lam * conf_k + scores = {} + for label_k, conf_k, _ in samples: + agreement = sum(c_i for (l_i, c_i, _) in samples if l_i == label_k) + scores[label_k] = agreement + lam * conf_k + + top_label = max(scores, key=scores.get) + # Emit the highest-confidence sample whose label == top_label + winning_sample = max( + (s for s in samples if s[0] == top_label), + key=lambda s: s[1], + ) + return {"label": top_label, "confidence": winning_sample[1], "raw": winning_sample[2]} +``` + +### Direct inference via `transformers` + +```python +from transformers import AutoTokenizer, AutoModelForCausalLM +import torch + +model_id = "rroshann/sec-sentiment-sftgrpo-deepseek-14b" +tokenizer = AutoTokenizer.from_pretrained(model_id) +model = AutoModelForCausalLM.from_pretrained( + model_id, + torch_dtype=torch.bfloat16, + device_map="auto", +) + +messages = [{"role": "user", "content": ""}] +input_ids = tokenizer.apply_chat_template( + messages, + return_tensors="pt", + add_generation_prompt=True, +).to(model.device) + +outputs = model.generate( + input_ids, + max_new_tokens=512, + do_sample=False, # greedy +) +print(tokenizer.decode(outputs[0, input_ids.shape[-1]:], skip_special_tokens=True)) +``` + +## Limitations & Biases + +- **Single-seed GRPO training.** No variance estimate across retraining runs. The portfolio-level gains over SFT (monotone cohort ladder, IR lift) are large enough to be defensible as point estimates, but formal significance testing would require a multi-seed rerun (not executed — see technical report §16.1). +- **Per-sample F1 gain vs SFT is within noise.** GRPO's ~0 F1 improvement is consistent with seed variance alone; only the portfolio-aggregated signal is a robust lift (report §8.8). +- **BoN evaluated OOS-only.** The `sft_grpo_bon` variant was sampled on the 605-filing test panel only (compute budget). There is no in-sample BoN counterpart for a direct IS-vs-OOS comparison (report §12.4). +- **Sparse tail cohorts in the BoN variant.** At H=21, the BoN variant's very_negative cohort contains n=2 filings and very_positive contains n=9. Headline IRs for the BoN variant rest on ~11 filings per tail cohort; a block-bootstrap confidence interval is not computed (report §11.4, §13.1). +- **No reward-term ablation.** The four reward hyperparameters (`λ = 0.3`, ordinal slope `0.25`, `G = 8`, `β = 0.04`) are author-chosen, not swept. A sensitivity sweep is future work. +- **Factor-level (not filing-level) train/val split** inherited from the SFT predecessor. Test set is time-ordered and filing-level, so the OOS protocol is unaffected. +- **Universe / domain specificity.** Trained on 80 U.S. industrials tickers; will underperform on other sectors. +- **8-K filings excluded.** Event-driven filings break the 60-question factor taxonomy. +- **HIGH_BETA disclosure.** Dollar-neutral portfolios built on this model's predictions have |β| ≈ 2.0 against SPY in backtests — not beta-neutral. Mitigation is a rolling-63d β-hedged SPY short overlay; see technical report §13.2. +- **Transports sector wrong-sign.** The `transports (airlines)` sub-sector carries a negative L/S spread across all variants (report §11.7). Deployment rule: exclude transports or invert the sign at the sector level. + +## Ethical Considerations + +- Training labels for the SFT predecessor were generated via the Anthropic API (Claude Opus). We believe this use falls within the non-competing-products provision of Anthropic's Commercial Terms because the released model is a 5-class sentiment classifier specialized for SEC filings, not a general-purpose assistant. Deployers should independently verify current Anthropic terms apply to their use. +- Predictions are for **research and reproducibility** of the capstone results. Not investment advice. Not audited for deployment in any regulated context. +- SEC filings are U.S. public-domain government documents (EDGAR). No PII. + +## Citation + +```bibtex +@techreport{siddartha2026reasoningaugmented, + title = {Reasoning-Augmented Factor Extraction: + Enhancing SEC Sentiment Signals through Reinforcement Learning}, + author = {Siddartha, Roshan and Tu, Maggie and Butskhrikidze, Luka}, + year = {2026}, + month = {April}, + institution = {Vanderbilt University Data Science Institute}, + note = {AllianceBernstein × Vanderbilt DSI Capstone. Course: + NLP for Asset Management. Instructor: Che Guan.} +} +``` + +## License & Acknowledgements + +- **Model license:** MIT (matches upstream DeepSeek-R1-Distill-Qwen-14B and the SFT predecessor). +- Upstream base model: DeepSeek-AI. See [`deepseek-ai/DeepSeek-R1-Distill-Qwen-14B`](https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B). +- Training labels (SFT stage) generated via the Anthropic API (Claude Opus family). +- GRPO implementation uses Hugging Face `trl`'s `GRPOTrainer`. +- Compute provided by Vanderbilt University ACCRE (DGX A100). +- Project advised by Che Guan, Vanderbilt Data Science Institute. diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..c2066bd --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1 @@ +{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '' in content %}{% set content = content.split('')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|>\n'}}{% endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..f73390a --- /dev/null +++ b/config.json @@ -0,0 +1,81 @@ +{ + "architectures": [ + "Qwen2ForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151643, + "hidden_act": "silu", + "hidden_size": 5120, + "initializer_range": 0.02, + "intermediate_size": 13824, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "max_window_layers": 48, + "model_type": "qwen2", + "num_attention_heads": 40, + "num_hidden_layers": 48, + "num_key_value_heads": 8, + "pad_token_id": null, + "rms_norm_eps": 1e-05, + "rope_parameters": { + "rope_theta": 1000000.0, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": false, + "transformers_version": "5.5.0", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 152064 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..928793d --- /dev/null +++ b/generation_config.json @@ -0,0 +1,9 @@ +{ + "_from_model_config": true, + "bos_token_id": 151646, + "do_sample": true, + "eos_token_id": 151643, + "temperature": 0.6, + "top_p": 0.95, + "transformers_version": "5.5.0" +} diff --git a/model-00001-of-00006.safetensors b/model-00001-of-00006.safetensors new file mode 100644 index 0000000..c9392d3 --- /dev/null +++ b/model-00001-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:452145e2589e77de650a48a41610b0420e1d716354e86b130ddd9befc5556a8d +size 4907454944 diff --git a/model-00002-of-00006.safetensors b/model-00002-of-00006.safetensors new file mode 100644 index 0000000..31f1840 --- /dev/null +++ b/model-00002-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:745417470442c8d53ac1a2c5f5e5c2f57ad07b69100f6c92f1fe1892a0184c46 +size 4954847304 diff --git a/model-00003-of-00006.safetensors b/model-00003-of-00006.safetensors new file mode 100644 index 0000000..95f8072 --- /dev/null +++ b/model-00003-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ec9b2aedb782071628a2a9a7a2636a73924c76ffd8fe3f3bf4b149436d2d2106 +size 4954847392 diff --git a/model-00004-of-00006.safetensors b/model-00004-of-00006.safetensors new file mode 100644 index 0000000..d00d4ba --- /dev/null +++ b/model-00004-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b760d270d77f434a9150a7dbc3f23050becaa6cd5842f1c355d7f2f3512a8960 +size 4954847392 diff --git a/model-00005-of-00006.safetensors b/model-00005-of-00006.safetensors new file mode 100644 index 0000000..4ab39d9 --- /dev/null +++ b/model-00005-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:437a5c398eae666346b40faf2898a026438c292ecb5b09fece77b2ee545741c6 +size 4954847392 diff --git a/model-00006-of-00006.safetensors b/model-00006-of-00006.safetensors new file mode 100644 index 0000000..7342849 --- /dev/null +++ b/model-00006-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:adef3e4f37a9213fb95dbe05e6e0ac64363e1bdc941da7ae0ad0b9679b85ee16 +size 4813289488 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..1ce7d1c --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,587 @@ +{ + "metadata": { + "total_parameters": 14770033664, + "total_size": 29540067328 + }, + "weight_map": { + "lm_head.weight": "model-00001-of-00006.safetensors", + "model.embed_tokens.weight": "model-00001-of-00006.safetensors", + "model.layers.0.input_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.0.mlp.down_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.0.mlp.up_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.k_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.q_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.v_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.input_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.1.mlp.down_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.mlp.up_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.k_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.q_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.v_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.10.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.10.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.10.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.10.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.11.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.12.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.12.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.12.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.12.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.13.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.14.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.15.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.16.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.17.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.18.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.19.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.2.input_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.2.mlp.down_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.2.mlp.up_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.k_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.q_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.v_proj.bias": "model-00001-of-00006.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.20.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.20.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.20.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.20.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.k_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.q_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.v_proj.bias": "model-00003-of-00006.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.21.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.layers.21.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.layers.21.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.21.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.22.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.23.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.24.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.25.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.26.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.27.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.28.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.28.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.29.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.k_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.q_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.v_proj.bias": "model-00004-of-00006.safetensors", + "model.layers.29.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.3.input_layernorm.weight": "model-00001-of-00006.safetensors", + "model.layers.3.mlp.down_proj.weight": "model-00001-of-00006.safetensors", + "model.layers.3.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.3.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.30.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.layers.30.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.layers.30.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.30.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.30.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.30.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.31.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.31.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.32.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.32.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.33.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.33.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.34.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.34.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.35.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.35.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.36.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.36.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.37.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.37.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.38.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.k_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.q_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.v_proj.bias": "model-00005-of-00006.safetensors", + "model.layers.38.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.39.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.layers.39.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.layers.39.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.39.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.39.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.39.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.4.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.4.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.4.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.4.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.40.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.40.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.40.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.40.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.40.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.40.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.41.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.41.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.42.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.42.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.43.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.43.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.44.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.44.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.45.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.45.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.46.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.46.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.input_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.47.mlp.down_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.mlp.gate_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.mlp.up_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.post_attention_layernorm.weight": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.k_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.k_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.o_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.q_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.q_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.v_proj.bias": "model-00006-of-00006.safetensors", + "model.layers.47.self_attn.v_proj.weight": "model-00006-of-00006.safetensors", + "model.layers.5.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.5.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.5.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.5.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.6.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.7.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.8.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.9.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.k_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.q_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.v_proj.bias": "model-00002-of-00006.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.norm.weight": "model-00006-of-00006.safetensors" + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..4306d79 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:322664cdc3082b6eba003af5228a77ca1d7936d402e584ecde8f15d3d98bdb72 +size 11421911 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..026ebaa --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,13 @@ +{ + "backend": "tokenizers", + "bos_token": "<|begin▁of▁sentence|>", + "clean_up_tokenization_spaces": false, + "eos_token": "<|end▁of▁sentence|>", + "is_local": true, + "legacy": true, + "model_max_length": 16384, + "pad_token": "<|end▁of▁sentence|>", + "sp_model_kwargs": {}, + "tokenizer_class": "TokenizersBackend", + "unk_token": null +}