初始化项目,由ModelHub XC社区提供模型
Model: rodrigoramosrs/qwen3-4b-dotnet-specialist Source: Original Platform
This commit is contained in:
37
.gitattributes
vendored
Normal file
37
.gitattributes
vendored
Normal file
@@ -0,0 +1,37 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.json filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
54
Modelfile
Normal file
54
Modelfile
Normal file
@@ -0,0 +1,54 @@
|
|||||||
|
|
||||||
|
FROM Qwen3-4B-Instruct-2507.Q8_0.gguf
|
||||||
|
TEMPLATE """
|
||||||
|
{{- $lastUserIdx := -1 -}}
|
||||||
|
{{- range $idx, $msg := .Messages -}}
|
||||||
|
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
|
||||||
|
{{- end }}
|
||||||
|
{{- if or .System .Tools }}<|im_start|>system
|
||||||
|
{{ if .System }}
|
||||||
|
{{ .System }}
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Tools }}
|
||||||
|
|
||||||
|
# Tools
|
||||||
|
|
||||||
|
You may call one or more functions to assist with the user query.
|
||||||
|
|
||||||
|
You are provided with function signatures within <tools></tools> XML tags:
|
||||||
|
<tools>
|
||||||
|
{{- range .Tools }}
|
||||||
|
{"type": "function", "function": {{ .Function }}}
|
||||||
|
{{- end }}
|
||||||
|
</tools>
|
||||||
|
|
||||||
|
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
|
||||||
|
<tool_call>
|
||||||
|
{"name": <function-name>, "arguments": <args-json-object>}
|
||||||
|
</tool_call>
|
||||||
|
{{- end -}}
|
||||||
|
<|im_end|>
|
||||||
|
{{ end }}
|
||||||
|
{{- range $i, $_ := .Messages }}
|
||||||
|
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
|
||||||
|
{{- if eq .Role "user" }}<|im_start|>user
|
||||||
|
{{ .Content }}<|im_end|>
|
||||||
|
{{ else if eq .Role "assistant" }}<|im_start|>assistant
|
||||||
|
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
|
||||||
|
<think>{{ .Thinking }}</think>
|
||||||
|
{{ end -}}
|
||||||
|
{{ if .Content }}{{ .Content }}
|
||||||
|
{{- else if .ToolCalls }}<tool_call>
|
||||||
|
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
|
||||||
|
{{ end }}</tool_call>
|
||||||
|
{{- end }}{{ if not $last }}<|im_end|>
|
||||||
|
{{ end }}
|
||||||
|
{{- else if eq .Role "tool" }}<|im_start|>user
|
||||||
|
<tool_response>
|
||||||
|
{{ .Content }}
|
||||||
|
</tool_response><|im_end|>
|
||||||
|
{{ end }}
|
||||||
|
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
|
||||||
|
{{ end }}
|
||||||
|
{{- end }}
|
||||||
|
"""
|
||||||
204
README.md
Normal file
204
README.md
Normal file
@@ -0,0 +1,204 @@
|
|||||||
|
---
|
||||||
|
license: cc-by-4.0
|
||||||
|
datasets:
|
||||||
|
- rodrigoramosrs/dotnet
|
||||||
|
language:
|
||||||
|
- en
|
||||||
|
- pt
|
||||||
|
base_model:
|
||||||
|
- Qwen/Qwen3-4B-Instruct-2507
|
||||||
|
tags:
|
||||||
|
- dotnet
|
||||||
|
- c#
|
||||||
|
- tech
|
||||||
|
- net
|
||||||
|
- develop
|
||||||
|
- documentation
|
||||||
|
- rodrigoramosrs
|
||||||
|
---
|
||||||
|
|
||||||
|
# 🧠 qwen3-4b-dotnet-specialist
|
||||||
|
### Fine-tuned model for technical reasoning and .NET documentation understanding
|
||||||
|
[]() []() [](https://huggingface.co/datasets/rodrigoramosrs/dotnet) []()
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 📘 Overview
|
||||||
|
|
||||||
|
`qwen3-4b-dotnet-specialist` is a **fine-tuned variant of Qwen 3 (4B parameters)**, specialized in understanding and generating **accurate, structured, and deeply technical content** related to the **.NET ecosystem**, including C#, ASP.NET Core, EF Core, CLI tools, documentation standards, and advanced runtime concepts.
|
||||||
|
|
||||||
|
This model was trained with a **highly curated dataset of 70,000 question-answer pairs**, derived from the official Microsoft documentation repository ([dotnet/docs](https://github.com/dotnet/docs)).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## ☀️ A Sustainable Experiment in AI Engineering
|
||||||
|
|
||||||
|
This project was built under a guiding principle:
|
||||||
|
> “Good science is not made of answers, but of the **right questions**.”
|
||||||
|
|
||||||
|
Every question in the dataset was algorithmically generated to test **specific technical reasoning paths**, and each answer was produced through a **Retrieval-Augmented Generation (RAG)** process — retrieving context from the entire documentation dataset rather than from the paragraph that originated the question.
|
||||||
|
|
||||||
|
That design choice produced **richer, contextually consistent, and cross-referenced answers** — making this model particularly strong in **documentation synthesis, reasoning across APIs, and multi-version comparison tasks**.
|
||||||
|
|
||||||
|
The entire curation, training, and evaluation process was powered using **solar energy**, highlighting that *research-grade AI can be done locally, sustainably, and accessibly*.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🧩 Dataset
|
||||||
|
|
||||||
|
📦 Dataset used: [**rodrigoramosrs/dotnet**](https://huggingface.co/datasets/rodrigoramosrs/dotnet)
|
||||||
|
|
||||||
|
- **Source:** Extracted and processed from [`github.com/dotnet/docs`](https://github.com/dotnet/docs)
|
||||||
|
- **Original size:** ~300 MB of unstructured text
|
||||||
|
- **Post-curation size:** ~60 MB
|
||||||
|
- **Format:** JSONL with `instruction`, `input`, and `output` keys
|
||||||
|
- **Samples:** ~70,000 Q&A pairs
|
||||||
|
- **Language:** English
|
||||||
|
- **Domain:** .NET / C# / Microsoft Docs structure
|
||||||
|
|
||||||
|
Each entry follows this structure:
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"instruction": "Explain how to organize tutorials in the .NET documentation portal.",
|
||||||
|
"input": "",
|
||||||
|
"output": "Detailed, step-by-step answer using the DocFX structure and YAML front-matter conventions."
|
||||||
|
}
|
||||||
|
````
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## ⚙️ Training
|
||||||
|
|
||||||
|
* **Base model:** Qwen3-4B (Instruct variant)
|
||||||
|
* **Training method:** LoRA fine-tuning
|
||||||
|
* **Context length:** -
|
||||||
|
* **Batch size:** 8
|
||||||
|
* **Precision:** bfloat16
|
||||||
|
* **Epochs:** 6.0
|
||||||
|
* **Optimizer:** AdamW (8-bit)
|
||||||
|
* **Scheduler:** Cosine decay with warmup
|
||||||
|
* **Learning Rate:** 2e-4
|
||||||
|
* **Warmup Ratio:** 0.7
|
||||||
|
* **Gradient Accumulation Steps:** 6
|
||||||
|
* **Infrastructure:** Local GPU (RTX 5080)
|
||||||
|
* **Power source:** Off-grid solar system
|
||||||
|
|
||||||
|
### 🧮 Data Pipeline
|
||||||
|
|
||||||
|
1. **Extraction** – Crawled markdown files from [`dotnet/docs`](https://github.com/dotnet/docs)
|
||||||
|
2. **Cleaning** – Removed metadata, HTML, and outdated versions
|
||||||
|
3. **Segmentation** – Split long sections into atomic topics
|
||||||
|
4. **Question Generation** – Built synthetic instructions using a tuned model focused on documentation comprehension
|
||||||
|
5. **Answer Generation (RAG)** – Retrieved context from the *entire dataset* before generating final answers
|
||||||
|
6. **Ranking & Filtering** – Applied cross-encoder ranking and manual curation to ensure quality
|
||||||
|
7. **Finalization** – Consolidated into clean, versioned JSONL format
|
||||||
|
|
||||||
|
### 📊 Training Configuration
|
||||||
|
|
||||||
|
```python
|
||||||
|
Config:
|
||||||
|
trainer = SFTTrainer(
|
||||||
|
model=model,
|
||||||
|
train_dataset=train_ds,
|
||||||
|
tokenizer=tokenizer,
|
||||||
|
formatting_func=formatting_func,
|
||||||
|
args=SFTConfig(
|
||||||
|
per_device_train_batch_size=8,
|
||||||
|
gradient_accumulation_steps=6,
|
||||||
|
num_train_epochs=6.0,
|
||||||
|
learning_rate=2e-4,
|
||||||
|
lr_scheduler_type="cosine",
|
||||||
|
warmup_ratio=0.7,
|
||||||
|
logging_steps=10,
|
||||||
|
save_strategy="steps",
|
||||||
|
save_steps=200,
|
||||||
|
eval_steps=200,
|
||||||
|
output_dir=output_dir,
|
||||||
|
push_to_hub=False, # 🚫 impede upload automático
|
||||||
|
hub_model_id=None, # 🚫 não referencia repositório remoto
|
||||||
|
bf16=True,
|
||||||
|
gradient_checkpointing=True,
|
||||||
|
optim="adamw_8bit",
|
||||||
|
weight_decay=0.001,
|
||||||
|
max_grad_norm=1.0,
|
||||||
|
dataloader_pin_memory=False,
|
||||||
|
dataloader_num_workers=4,
|
||||||
|
report_to=None,
|
||||||
|
ddp_find_unused_parameters=True, # True ajuda em multi-GPU,
|
||||||
|
),
|
||||||
|
eval_dataset=eval_ds, # ✅ Inclui o dataset de avaliação (opcional, mas recomendado)
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
### 📈 Training Results
|
||||||
|
|
||||||
|
- **Final Loss:** 0.742800
|
||||||
|
- **Training Steps:** 3930
|
||||||
|
- **Evaluation Metrics:**
|
||||||
|
- **Perplexity:** Good
|
||||||
|
- **Factual Accuracy (manual):** ~High
|
||||||
|
- **Response Consistency:** High
|
||||||
|
- **Formatting Accuracy:** High
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🧠 Intended Use
|
||||||
|
|
||||||
|
The model excels at:
|
||||||
|
|
||||||
|
* Explaining .NET concepts, frameworks, and internal mechanics
|
||||||
|
* Answering developer documentation questions
|
||||||
|
* Summarizing and rewriting technical guides
|
||||||
|
* Generating structured technical explanations
|
||||||
|
* Acting as a documentation assistant for software engineers
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🚫 Limitations
|
||||||
|
|
||||||
|
* Limited to **.NET and related ecosystems** — not designed for general-purpose conversation.
|
||||||
|
* May occasionally produce overly detailed explanations when prompted ambiguously.
|
||||||
|
* Not a replacement for Microsoft’s official documentation — rather a **complementary reasoning model**.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 💬 Example Usage
|
||||||
|
|
||||||
|
```python
|
||||||
|
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||||
|
import torch
|
||||||
|
|
||||||
|
model_name = "rodrigoramosrs/qwen3-4b-dotnet-specialist"
|
||||||
|
tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||||
|
model = AutoModelForCausalLM.from_pretrained(model_name, torch_dtype=torch.bfloat16).eval()
|
||||||
|
|
||||||
|
prompt = """Explain how to publish an ASP.NET Core app using the .NET CLI."""
|
||||||
|
|
||||||
|
inputs = tokenizer(prompt, return_tensors="pt").to("cuda")
|
||||||
|
outputs = model.generate(**inputs, max_new_tokens=600, temperature=0.3, top_p=0.9)
|
||||||
|
|
||||||
|
print(tokenizer.decode(outputs[0], skip_special_tokens=True))
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🔖 License & Attribution
|
||||||
|
|
||||||
|
* **Model License:** CC BY-SA 4.0
|
||||||
|
* **Dataset License:** CC BY-SA 4.0
|
||||||
|
* **Data Source:** Official Microsoft .NET documentation ([dotnet/docs](https://github.com/dotnet/docs))
|
||||||
|
* **Creator:** [Rodrigo Ramos (@rodrigoramosrs)](https://huggingface.co/rodrigoramosrs)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 🌍 Closing Note
|
||||||
|
|
||||||
|
This project is a proof that **precision and sustainability** can coexist in AI research.
|
||||||
|
It demonstrates that with the **right questions**, good data, and discipline, one person — powered by sunlight — can build a specialized model that truly understands a complex technical domain.
|
||||||
|
|
||||||
|
> *Built locally. Trained on clean data. Powered by the sun.* ☀️
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Model:** [`rodrigoramosrs/qwen3-4b-dotnet-specialist`](https://huggingface.co/rodrigoramosrs/qwen3-4b-dotnet-specialist)
|
||||||
|
**Dataset:** [`rodrigoramosrs/dotnet`](https://huggingface.co/datasets/rodrigoramosrs/dotnet)
|
||||||
3
added_tokens.json
Normal file
3
added_tokens.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:c0284b582e14987fbd3d5a2cb2bd139084371ed9acbae488829a1c900833c680
|
||||||
|
size 707
|
||||||
86
chat_template.jinja
Normal file
86
chat_template.jinja
Normal file
@@ -0,0 +1,86 @@
|
|||||||
|
{%- if tools %}
|
||||||
|
{{- '<|im_start|>system\n' }}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- messages[0].content + '\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||||
|
{%- for tool in tools %}
|
||||||
|
{{- "\n" }}
|
||||||
|
{{- tool | tojson }}
|
||||||
|
{%- endfor %}
|
||||||
|
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||||
|
{%- else %}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||||
|
{%- for message in messages[::-1] %}
|
||||||
|
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||||
|
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||||
|
{%- set ns.multi_step_tool = false %}
|
||||||
|
{%- set ns.last_query_index = index %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- for message in messages %}
|
||||||
|
{%- if message.content is string %}
|
||||||
|
{%- set content = message.content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- set content = '' %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
||||||
|
{%- elif message.role == "assistant" %}
|
||||||
|
{%- set reasoning_content = '' %}
|
||||||
|
{%- if message.reasoning_content is string %}
|
||||||
|
{%- set reasoning_content = message.reasoning_content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- if '</think>' in content %}
|
||||||
|
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||||
|
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if loop.index0 > ns.last_query_index %}
|
||||||
|
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if message.tool_calls %}
|
||||||
|
{%- for tool_call in message.tool_calls %}
|
||||||
|
{%- if (loop.first and content) or (not loop.first) %}
|
||||||
|
{{- '\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if tool_call.function %}
|
||||||
|
{%- set tool_call = tool_call.function %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<tool_call>\n{"name": "' }}
|
||||||
|
{{- tool_call.name }}
|
||||||
|
{{- '", "arguments": ' }}
|
||||||
|
{%- if tool_call.arguments is string %}
|
||||||
|
{{- tool_call.arguments }}
|
||||||
|
{%- else %}
|
||||||
|
{{- tool_call.arguments | tojson }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '}\n</tool_call>' }}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- elif message.role == "tool" %}
|
||||||
|
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||||
|
{{- '<|im_start|>user' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '\n<tool_response>\n' }}
|
||||||
|
{{- content }}
|
||||||
|
{{- '\n</tool_response>' }}
|
||||||
|
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- if add_generation_prompt %}
|
||||||
|
{{- '<|im_start|>assistant\n' }}
|
||||||
|
{%- endif %}
|
||||||
208
checkpoint/README.md
Normal file
208
checkpoint/README.md
Normal file
@@ -0,0 +1,208 @@
|
|||||||
|
---
|
||||||
|
base_model: unsloth/Qwen3-4B-Instruct-2507
|
||||||
|
library_name: peft
|
||||||
|
pipeline_tag: text-generation
|
||||||
|
tags:
|
||||||
|
- base_model:adapter:unsloth/Qwen3-4B-Instruct-2507
|
||||||
|
- lora
|
||||||
|
- transformers
|
||||||
|
- unsloth
|
||||||
|
---
|
||||||
|
|
||||||
|
# Model Card for Model ID
|
||||||
|
|
||||||
|
<!-- Provide a quick summary of what the model is/does. -->
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
## Model Details
|
||||||
|
|
||||||
|
### Model Description
|
||||||
|
|
||||||
|
<!-- Provide a longer summary of what this model is. -->
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
- **Developed by:** [More Information Needed]
|
||||||
|
- **Funded by [optional]:** [More Information Needed]
|
||||||
|
- **Shared by [optional]:** [More Information Needed]
|
||||||
|
- **Model type:** [More Information Needed]
|
||||||
|
- **Language(s) (NLP):** [More Information Needed]
|
||||||
|
- **License:** [More Information Needed]
|
||||||
|
- **Finetuned from model [optional]:** [More Information Needed]
|
||||||
|
|
||||||
|
### Model Sources [optional]
|
||||||
|
|
||||||
|
<!-- Provide the basic links for the model. -->
|
||||||
|
|
||||||
|
- **Repository:** [More Information Needed]
|
||||||
|
- **Paper [optional]:** [More Information Needed]
|
||||||
|
- **Demo [optional]:** [More Information Needed]
|
||||||
|
|
||||||
|
## Uses
|
||||||
|
|
||||||
|
<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
|
||||||
|
|
||||||
|
### Direct Use
|
||||||
|
|
||||||
|
<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
### Downstream Use [optional]
|
||||||
|
|
||||||
|
<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
### Out-of-Scope Use
|
||||||
|
|
||||||
|
<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Bias, Risks, and Limitations
|
||||||
|
|
||||||
|
<!-- This section is meant to convey both technical and sociotechnical limitations. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
### Recommendations
|
||||||
|
|
||||||
|
<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
|
||||||
|
|
||||||
|
Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
|
||||||
|
|
||||||
|
## How to Get Started with the Model
|
||||||
|
|
||||||
|
Use the code below to get started with the model.
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Training Details
|
||||||
|
|
||||||
|
### Training Data
|
||||||
|
|
||||||
|
<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
### Training Procedure
|
||||||
|
|
||||||
|
<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
|
||||||
|
|
||||||
|
#### Preprocessing [optional]
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
|
||||||
|
#### Training Hyperparameters
|
||||||
|
|
||||||
|
- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
|
||||||
|
|
||||||
|
#### Speeds, Sizes, Times [optional]
|
||||||
|
|
||||||
|
<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Evaluation
|
||||||
|
|
||||||
|
<!-- This section describes the evaluation protocols and provides the results. -->
|
||||||
|
|
||||||
|
### Testing Data, Factors & Metrics
|
||||||
|
|
||||||
|
#### Testing Data
|
||||||
|
|
||||||
|
<!-- This should link to a Dataset Card if possible. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
#### Factors
|
||||||
|
|
||||||
|
<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
#### Metrics
|
||||||
|
|
||||||
|
<!-- These are the evaluation metrics being used, ideally with a description of why. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
### Results
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
#### Summary
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
## Model Examination [optional]
|
||||||
|
|
||||||
|
<!-- Relevant interpretability work for the model goes here -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Environmental Impact
|
||||||
|
|
||||||
|
<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
|
||||||
|
|
||||||
|
Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
|
||||||
|
|
||||||
|
- **Hardware Type:** [More Information Needed]
|
||||||
|
- **Hours used:** [More Information Needed]
|
||||||
|
- **Cloud Provider:** [More Information Needed]
|
||||||
|
- **Compute Region:** [More Information Needed]
|
||||||
|
- **Carbon Emitted:** [More Information Needed]
|
||||||
|
|
||||||
|
## Technical Specifications [optional]
|
||||||
|
|
||||||
|
### Model Architecture and Objective
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
### Compute Infrastructure
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
#### Hardware
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
#### Software
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Citation [optional]
|
||||||
|
|
||||||
|
<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
|
||||||
|
|
||||||
|
**BibTeX:**
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
**APA:**
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Glossary [optional]
|
||||||
|
|
||||||
|
<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## More Information [optional]
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Model Card Authors [optional]
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
|
||||||
|
## Model Card Contact
|
||||||
|
|
||||||
|
[More Information Needed]
|
||||||
|
### Framework versions
|
||||||
|
|
||||||
|
- PEFT 0.17.1
|
||||||
3
checkpoint/adapter_config.json
Normal file
3
checkpoint/adapter_config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:b76ea99c4e33d256b9f1441fbc94b7f253914ad5e646c063e4f3ca0bc7d6dc54
|
||||||
|
size 1076
|
||||||
3
checkpoint/adapter_model.safetensors
Normal file
3
checkpoint/adapter_model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:6eeb6b707131ea9923d5c03ae0862c866919fdcdf09dbcd5f30fbc4ce0770cd1
|
||||||
|
size 264308896
|
||||||
3
checkpoint/added_tokens.json
Normal file
3
checkpoint/added_tokens.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:c0284b582e14987fbd3d5a2cb2bd139084371ed9acbae488829a1c900833c680
|
||||||
|
size 707
|
||||||
86
checkpoint/chat_template.jinja
Normal file
86
checkpoint/chat_template.jinja
Normal file
@@ -0,0 +1,86 @@
|
|||||||
|
{%- if tools %}
|
||||||
|
{{- '<|im_start|>system\n' }}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- messages[0].content + '\n\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
||||||
|
{%- for tool in tools %}
|
||||||
|
{{- "\n" }}
|
||||||
|
{{- tool | tojson }}
|
||||||
|
{%- endfor %}
|
||||||
|
{{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
|
||||||
|
{%- else %}
|
||||||
|
{%- if messages[0].role == 'system' %}
|
||||||
|
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
||||||
|
{%- for message in messages[::-1] %}
|
||||||
|
{%- set index = (messages|length - 1) - loop.index0 %}
|
||||||
|
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
||||||
|
{%- set ns.multi_step_tool = false %}
|
||||||
|
{%- set ns.last_query_index = index %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- for message in messages %}
|
||||||
|
{%- if message.content is string %}
|
||||||
|
{%- set content = message.content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- set content = '' %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
||||||
|
{%- elif message.role == "assistant" %}
|
||||||
|
{%- set reasoning_content = '' %}
|
||||||
|
{%- if message.reasoning_content is string %}
|
||||||
|
{%- set reasoning_content = message.reasoning_content %}
|
||||||
|
{%- else %}
|
||||||
|
{%- if '</think>' in content %}
|
||||||
|
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
||||||
|
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if loop.index0 > ns.last_query_index %}
|
||||||
|
{%- if loop.last or (not loop.last and reasoning_content) %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- else %}
|
||||||
|
{{- '<|im_start|>' + message.role + '\n' + content }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if message.tool_calls %}
|
||||||
|
{%- for tool_call in message.tool_calls %}
|
||||||
|
{%- if (loop.first and content) or (not loop.first) %}
|
||||||
|
{{- '\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- if tool_call.function %}
|
||||||
|
{%- set tool_call = tool_call.function %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<tool_call>\n{"name": "' }}
|
||||||
|
{{- tool_call.name }}
|
||||||
|
{{- '", "arguments": ' }}
|
||||||
|
{%- if tool_call.arguments is string %}
|
||||||
|
{{- tool_call.arguments }}
|
||||||
|
{%- else %}
|
||||||
|
{{- tool_call.arguments | tojson }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '}\n</tool_call>' }}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- elif message.role == "tool" %}
|
||||||
|
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
||||||
|
{{- '<|im_start|>user' }}
|
||||||
|
{%- endif %}
|
||||||
|
{{- '\n<tool_response>\n' }}
|
||||||
|
{{- content }}
|
||||||
|
{{- '\n</tool_response>' }}
|
||||||
|
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
||||||
|
{{- '<|im_end|>\n' }}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endif %}
|
||||||
|
{%- endfor %}
|
||||||
|
{%- if add_generation_prompt %}
|
||||||
|
{{- '<|im_start|>assistant\n' }}
|
||||||
|
{%- endif %}
|
||||||
151388
checkpoint/merges.txt
Normal file
151388
checkpoint/merges.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
checkpoint/special_tokens_map.json
Normal file
3
checkpoint/special_tokens_map.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:b5acd1507cb3a3539e63369cb4a2b675a599cf13afe62282d371518288967797
|
||||||
|
size 614
|
||||||
BIN
checkpoint/tokenizer.json
(Stored with Git LFS)
Normal file
BIN
checkpoint/tokenizer.json
(Stored with Git LFS)
Normal file
Binary file not shown.
3
checkpoint/tokenizer_config.json
Normal file
3
checkpoint/tokenizer_config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:9657e4e4faa05f5be1036381d4b6b9b6217c3cf636c2aeb76ec5243e766cea80
|
||||||
|
size 5432
|
||||||
BIN
checkpoint/vocab.json
(Stored with Git LFS)
Normal file
BIN
checkpoint/vocab.json
(Stored with Git LFS)
Normal file
Binary file not shown.
3
config.json
Normal file
3
config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:66661d4c28c89247b7b60a893b3b08470b33374cce15706b4d30d6002ec7a6da
|
||||||
|
size 1601
|
||||||
122
data_summary_card.md
Normal file
122
data_summary_card.md
Normal file
@@ -0,0 +1,122 @@
|
|||||||
|
# Data Summary for qwen3-4b-dotnet-specialist
|
||||||
|
|
||||||
|
## 1. General information
|
||||||
|
|
||||||
|
### 1.0.1 Version of the Summary: 1.0
|
||||||
|
|
||||||
|
### 1.0.2 Last update: 24-Nov-2025
|
||||||
|
|
||||||
|
## 1.1 Model Developer Identification
|
||||||
|
|
||||||
|
### 1.1.1 Model Developer name and contact details:
|
||||||
|
Rodrigo Ramos (@rodrigoramosrs)
|
||||||
|
Contact: [Hugging Face Profile](https://huggingface.co/rodrigoramosrs)
|
||||||
|
|
||||||
|
## 1.2 Model Identification
|
||||||
|
|
||||||
|
### 1.2.1 Versioned model name(s):
|
||||||
|
qwen3-4b-dotnet-specialist
|
||||||
|
|
||||||
|
### 1.2.2 Model release date:
|
||||||
|
Nov-2025
|
||||||
|
|
||||||
|
## 1.3 Overall training data size and characteristics
|
||||||
|
|
||||||
|
### 1.3.1 Size of dataset and characteristics
|
||||||
|
|
||||||
|
#### 1.3.1.A Text training data size:
|
||||||
|
~70,000 question-answer pairs
|
||||||
|
|
||||||
|
#### 1.3.1.B Text training data content:
|
||||||
|
Training data is derived from the official Microsoft documentation repository (dotnet/docs) and includes:
|
||||||
|
|
||||||
|
1. Extracted and processed markdown files from GitHub repository dotnet/docs
|
||||||
|
2. Structured technical content covering .NET, C#, ASP.NET Core, EF Core, CLI tools, documentation standards, and advanced runtime concepts
|
||||||
|
3. High-quality question-answer pairs generated algorithmically to test specific technical reasoning paths
|
||||||
|
4. Answers produced through Retrieval-Augmented Generation (RAG) process using context from the entire documentation dataset rather than just the originating paragraph
|
||||||
|
|
||||||
|
#### 1.3.1.C Image training data size:
|
||||||
|
Not applicable. Images are not part of the training
|
||||||
|
|
||||||
|
#### 1.3.1.D Image training data content:
|
||||||
|
Not applicable
|
||||||
|
|
||||||
|
#### 1.3.1.E Audio training data size:
|
||||||
|
Not applicable. Audio data is not part of the training data
|
||||||
|
|
||||||
|
#### 1.3.1.F Audio training data content:
|
||||||
|
Not applicable
|
||||||
|
|
||||||
|
#### 1.3.1.G Video training data size:
|
||||||
|
Not applicable. Video data is not part of the training data
|
||||||
|
|
||||||
|
#### 1.3.1.H Video training data content:
|
||||||
|
Not applicable
|
||||||
|
|
||||||
|
#### 1.3.1.I Other training data size:
|
||||||
|
Not applicable
|
||||||
|
|
||||||
|
#### 1.3.1.J Other training data content:
|
||||||
|
Not applicable
|
||||||
|
|
||||||
|
### 1.3.2 Latest date of data acquisition/collection for model training:
|
||||||
|
Not specified in the provided content
|
||||||
|
|
||||||
|
### 1.3.3 Is data collection ongoing to update the model with new data collection after deployment?
|
||||||
|
No
|
||||||
|
|
||||||
|
### 1.3.4 Date the training dataset was first used to train the model:
|
||||||
|
Not specified in the provided content
|
||||||
|
|
||||||
|
### 1.3.5 Rationale or purpose of data selection:
|
||||||
|
Datasets were selected to maximize high-quality technical reasoning and problem-solving capabilities within the .NET ecosystem. The mixture emphasizes carefully curated, algorithmically generated synthetic question-answer pairs derived from official documentation to improve technical understanding, documentation synthesis, and reasoning across APIs while maintaining factual accuracy.
|
||||||
|
|
||||||
|
## 2. List of data sources
|
||||||
|
|
||||||
|
### 2.1 Publicly available datasets
|
||||||
|
|
||||||
|
#### 2.1.1 Have you used publicly available datasets to train the model?
|
||||||
|
Yes
|
||||||
|
|
||||||
|
Source: Official Microsoft .NET documentation repository (dotnet/docs)
|
||||||
|
|
||||||
|
### 2.2 Private non-publicly available datasets obtained from third parties
|
||||||
|
|
||||||
|
#### 2.2.1 Datasets commercially licensed by rights holders or their representatives
|
||||||
|
Not applicable - dataset is derived from public GitHub repository
|
||||||
|
|
||||||
|
#### 2.2.2 Private datasets obtained from other third-parties
|
||||||
|
Not applicable - dataset is derived from public GitHub repository
|
||||||
|
|
||||||
|
### 2.3 Personal Information
|
||||||
|
|
||||||
|
#### 2.3.1 Was personal data used to train the model?
|
||||||
|
No personal data was used for training this model.
|
||||||
|
|
||||||
|
### 2.4 Synthetic data
|
||||||
|
|
||||||
|
#### 2.4.1 Was any synthetic AI-generated data used to train the model?
|
||||||
|
Yes - algorithmically generated question-answer pairs based on technical documentation content
|
||||||
|
|
||||||
|
## 3. Data processing aspects
|
||||||
|
|
||||||
|
### 3.1 Respect of reservation of rights from text and data mining exception or limitation
|
||||||
|
|
||||||
|
#### 3.1.1 Does this dataset include any data protected by copyright, trademark, or patent?
|
||||||
|
The dataset is derived from the public dotnet/docs GitHub repository which has appropriate licensing for reuse.
|
||||||
|
|
||||||
|
### 3.2 Other information
|
||||||
|
|
||||||
|
#### 3.2.1 Does the dataset include information about consumer groups without revealing individual consumer identities?
|
||||||
|
No personal or consumer identity information is included in the dataset.
|
||||||
|
|
||||||
|
#### 3.2.2 Was the dataset cleaned or modified before model training?
|
||||||
|
Yes - the dataset was cleaned and processed through:
|
||||||
|
|
||||||
|
1. Extraction from markdown files
|
||||||
|
2. Removal of metadata, HTML, and outdated versions
|
||||||
|
3. Segmentation into atomic topics
|
||||||
|
4. Algorithmic question generation
|
||||||
|
5. RAG-based answer generation using full documentation context
|
||||||
|
6. Cross-encoder ranking and manual curation for quality assurance
|
||||||
|
7. Consolidation into clean, versioned JSONL format
|
||||||
3
gguf/Qwen3-4B-Instruct-2507-Dotnet-Specialist.Q2_K.gguf
Normal file
3
gguf/Qwen3-4B-Instruct-2507-Dotnet-Specialist.Q2_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:5cb52664b987066fabedcbe31533e1f3d52df2cecafad1da0eec49fdabf9b219
|
||||||
|
size 1669499072
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:829830d5313b56d4b87582200c4fcb3e9d09fb9bd612de6dc0b04857d226d2d1
|
||||||
|
size 2497280192
|
||||||
3
gguf/Qwen3-4B-Instruct-2507-Dotnet-Specialist.Q8_0.gguf
Normal file
3
gguf/Qwen3-4B-Instruct-2507-Dotnet-Specialist.Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:bc98ac86b8af358ce044ca443d39368ca9fe759e28d9e210afeb128c66f737a4
|
||||||
|
size 4280404672
|
||||||
151388
merges.txt
Normal file
151388
merges.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
model-00001-of-00002.safetensors
Normal file
3
model-00001-of-00002.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:a39dab46d7046df61086757042c69c2e884ff1220b44227b91c76df5f33dafbd
|
||||||
|
size 4967215360
|
||||||
3
model-00002-of-00002.safetensors
Normal file
3
model-00002-of-00002.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:a605ee1db0b355b85ce9e7c7af9fb22189269fa9f42f9e4c6ff0e298780e19f0
|
||||||
|
size 3077766632
|
||||||
3
model.safetensors.index.json
Normal file
3
model.safetensors.index.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:06b3d5319b6d76d1a4a2433419180016cfd54ed62d086a5e6567a809f8c82634
|
||||||
|
size 32855
|
||||||
3
model_config.json
Normal file
3
model_config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:20eda502a6a5f158329a07f6f6b6b08e679e9b2fa7c7e44f63bccab94558f8af
|
||||||
|
size 840
|
||||||
3
special_tokens_map.json
Normal file
3
special_tokens_map.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:b5acd1507cb3a3539e63369cb4a2b675a599cf13afe62282d371518288967797
|
||||||
|
size 614
|
||||||
BIN
tokenizer.json
(Stored with Git LFS)
Normal file
BIN
tokenizer.json
(Stored with Git LFS)
Normal file
Binary file not shown.
3
tokenizer_config.json
Normal file
3
tokenizer_config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:560c32234d9f1acd51295742eaf4fc81260fb4f0e8d0399ae4172cfad366cda0
|
||||||
|
size 9654
|
||||||
BIN
vocab.json
(Stored with Git LFS)
Normal file
BIN
vocab.json
(Stored with Git LFS)
Normal file
Binary file not shown.
Reference in New Issue
Block a user