From 692da6f5da19ab88dbd011ad591ba11216a4a8a7 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Mon, 14 Sep 2026 10:10:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: jalpan04/qwen-researcher Source: Original Platform --- .gitattributes | 37 +++++++++++++ Modelfile | 10 ++++ README.md | 86 ++++++++++++++++++++++++++++++ merge_model.py | 46 ++++++++++++++++ prepare_data.py | 72 +++++++++++++++++++++++++ qwen-researcher-f16.gguf | 3 ++ requirements.txt | 7 +++ train.py | 111 +++++++++++++++++++++++++++++++++++++++ 8 files changed, 372 insertions(+) create mode 100644 .gitattributes create mode 100644 Modelfile create mode 100644 README.md create mode 100644 merge_model.py create mode 100644 prepare_data.py create mode 100644 qwen-researcher-f16.gguf create mode 100644 requirements.txt create mode 100644 train.py diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..f28081d --- /dev/null +++ b/.gitattributes @@ -0,0 +1,37 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +qwen-resercher.gguf filter=lfs diff=lfs merge=lfs -text +qwen-researcher-f16.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/Modelfile b/Modelfile new file mode 100644 index 0000000..075b02b --- /dev/null +++ b/Modelfile @@ -0,0 +1,10 @@ +FROM ./qwen-resercher.gguf +PARAMETER stop "<|im_start|>" +PARAMETER stop "<|im_end|>" +TEMPLATE """<|im_start|>system +{{ .System }}<|im_end|> +<|im_start|>user +{{ .Prompt }}<|im_end|> +<|im_start|>assistant +""" +SYSTEM """You are an expert Computer Science Research Assistant. You provide detailed, academic-style answers based on your specialized training on arXiv papers.""" diff --git a/README.md b/README.md new file mode 100644 index 0000000..ef73229 --- /dev/null +++ b/README.md @@ -0,0 +1,86 @@ +--- +license: apache-2.0 +base_model: Qwen/Qwen2.5-0.5B-Instruct +tags: +- qwen +- qwen2 +- gguf +- f16 +- computer-science +- research +- instruction-tuning +model_creator: jalpan04 +model_name: Qwen Research Assistant +pipeline_tag: text-generation +quantized_by: f16 +--- + +# Qwen Researcher (0.5B - GGUF) + +A specialized version of **Qwen2.5-0.5B-Instruct** fine-tuned for Computer Science research assistance. This model has been instruction-tuned on a curated subset of 2,000 arXiv Computer Science papers to provide academic summaries and technical insights. + +## Model Details + +- **Developed by:** Jalpan04 +- **Model type:** Causal Language Model +- **Language(s):** English +- **License:** Apache-2.0 +- **Fine-tuned from model:** [Qwen/Qwen2.5-0.5B-Instruct](https://huggingface.co/Qwen/Qwen2.5-0.5B-Instruct) +- **Training Method:** QLoRA (Rank 16, Alpha 32) +- **Dataset:** arXiv Computer Science Metadata (Instruction-formatted) + +## Usage + +### Local Deployment (Ollama) + +1. Create a `Modelfile`: +```dockerfile +FROM ./qwen-resercher.gguf +TEMPLATE """{{ if .System }}<|im_start|>system +{{ .System }}<|im_end|> +{{ end }}{{ if .User }}<|im_start|>user +{{ .User }}<|im_end|> +{{ end }}<|im_start|>assistant +{{ .Output }}<|im_end|>""" +PARAMETER stop "<|im_start|>" +PARAMETER stop "<|im_end|>" +SYSTEM "You are a professional computer science researcher. Provide academic, detailed information based on research abstracts." +``` + +2. Run in terminal: +```bash +ollama create qwen-researcher -f Modelfile +ollama run qwen-researcher +``` + +### Python (llama-cpp-python) + +```python +from llama_cpp import Llama + +llm = Llama.from_pretrained( + repo_id="jalpan04/qwen-researcher", + filename="qwen-resercher.gguf", +) + +response = llm.create_chat_completion( + messages = [ + {"role": "system", "content": "You are a CS researcher."}, + {"role": "user", "content": "Summarize the latest trends in Neural Program Synthesis."} + ] +) +print(response["choices"][0]["message"]["content"]) +``` + +## Training Procedure + +The model was trained on an **NVIDIA RTX 4060 (8GB)** using the following setup: +- **Optimization:** 4-bit NormalFloat (nf4) quantization. +- **Precision:** BFloat16 for stability on 40-series hardware. +- **Learning Rate:** 2e-4. +- **Batch Size:** 1 with Gradient Accumulation (16 steps). +- **Sequence Length:** 1024 tokens. + +## Limitations + +As a 0.5B parameter model, this is highly efficient but may exhibit hallucinations compared to larger models (7B+). It is best used for summarization and quick technical lookups rather than complex logical reasoning. diff --git a/merge_model.py b/merge_model.py new file mode 100644 index 0000000..317eb42 --- /dev/null +++ b/merge_model.py @@ -0,0 +1,46 @@ +import torch +from peft import PeftModel +from transformers import AutoModelForCausalLM, AutoTokenizer +import os + +def merge_and_save(): + """ + Script to merge trained LoRA adapters back into the base Qwen model. + This creates a standalone model folder that can be converted to GGUF format. + """ + + base_model_id = "Qwen/Qwen2.5-0.5B-Instruct" + adapter_path = "./qwen-resercher" + output_path = "./qwen-resercher-final" + + print(f"Loading base model: {base_model_id}") + tokenizer = AutoTokenizer.from_pretrained(base_model_id) + + # Load base model in FP16 for merging precision + # device_map="auto" is used to handle large models, though 0.5B fits easily in VRAM. + model = AutoModelForCausalLM.from_pretrained( + base_model_id, + torch_dtype=torch.float16, + device_map="auto", + ) + + # 1. Load the trained adapter onto the base model + print(f"Loading adapter from: {adapter_path}") + model = PeftModel.from_pretrained(model, adapter_path) + + # 2. Merge the weights + # merge_and_unload() combines the LoRA matrices with the original weights. + # This results in a standard Transformer model (no adapters needed). + print("Merging weights... this creates the final unified model.") + model = model.merge_and_unload() + + # 3. Save the final model and tokenizer + # This folder will be the source for the 'convert_hf_to_gguf.py' script. + print(f"Saving unified model to: {output_path}") + model.save_pretrained(output_path) + tokenizer.save_pretrained(output_path) + + print("Merge complete.") + +if __name__ == "__main__": + merge_and_save() diff --git a/prepare_data.py b/prepare_data.py new file mode 100644 index 0000000..f58a1d5 --- /dev/null +++ b/prepare_data.py @@ -0,0 +1,72 @@ +import json +import random +import os + +def prepare_data(): + """ + Data preparation script to convert raw arXiv metadata into ChatML format. + Handles memory efficiency by streaming the source file line-by-line. + """ + + input_file = "arxiv-metadata-oai-snapshot.json" + output_file = "arxiv_cs_2000.jsonl" + + # Check if input exists + if not os.path.exists(input_file): + print(f"Source file {input_file} not found. Please ensure it is in the directory.") + return + + cs_papers = [] + print("Scanning arXiv dataset for Computer Science papers...") + + # We use a streaming approach (open/read) to avoid loading the whole JSON into memory. + with open(input_file, 'r', encoding='utf-8') as f: + for line in f: + try: + paper = json.loads(line) + # Filter for papers containing 'cs.' in categories + if 'cs.' in paper.get('categories', ''): + cs_papers.append({ + 'title': paper.get('title', 'No Title'), + 'abstract': paper.get('abstract', 'No Abstract').replace('\n', ' ').strip() + }) + except json.JSONDecodeError: + continue + + # Select a diverse sample of 2,000 papers for instruction tuning + if len(cs_papers) > 2000: + print(f"Found {len(cs_papers)} CS papers. Sampling 2,000...") + selected_papers = random.sample(cs_papers, 2000) + else: + print(f"Found {len(cs_papers)} CS papers. Using all available data.") + selected_papers = cs_papers + + # Convert to ChatML/Instruction format for Qwen + # Each entry consists of a system prompt, a user question, and the assistant response. + print(f"Writing formatted data to {output_file}...") + with open(output_file, 'w', encoding='utf-8') as f: + for paper in selected_papers: + # Construct the conversation + messages = [ + { + "role": "system", + "content": "You are a professional computer science researcher. Provide academic, detailed information based on research abstracts." + }, + { + "role": "user", + "content": f"Summarize the research and key contributions of the paper titled: {paper['title']}" + }, + { + "role": "assistant", + "content": paper['abstract'] + } + ] + + # Write as a JSONL line + json_line = json.dumps({"messages": messages}) + f.write(json_line + '\n') + + print("Data preparation complete.") + +if __name__ == "__main__": + prepare_data() diff --git a/qwen-researcher-f16.gguf b/qwen-researcher-f16.gguf new file mode 100644 index 0000000..826cb67 --- /dev/null +++ b/qwen-researcher-f16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7757a1d695617c35f1b81e8fc4183a3088a6eb3b5360fcb04eb7c695b86cd5a1 +size 994156320 diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..1431b75 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,7 @@ +torch +transformers +datasets +peft +trl +bitsandbytes +accelerate diff --git a/train.py b/train.py new file mode 100644 index 0000000..12d6237 --- /dev/null +++ b/train.py @@ -0,0 +1,111 @@ +import torch +import os +from datasets import load_dataset +from transformers import ( + AutoModelForCausalLM, + AutoTokenizer, + BitsAndBytesConfig, + TrainingArguments +) +from peft import LoraConfig, get_peft_model, prepare_model_for_kbit_training +from trl import SFTTrainer, SFTConfig + +def train(): + """ + Main training function to fine-tune Qwen2.5-0.5B using QLoRA. + Includes debug prints for step-by-step monitoring on Windows. + """ + print(">>> DEBUG: Entering train() function...") + + # Clear CUDA memory before starting to prevent fragmentation + if torch.cuda.is_available(): + print(">>> DEBUG: Clearing CUDA cache...") + torch.cuda.empty_cache() + + model_id = "Qwen/Qwen2.5-0.5B-Instruct" + data_file = "arxiv_cs_2000.jsonl" + output_dir = "./qwen-resercher-checkpoints" + + # 1. Load Tokenizer + print(f">>> DEBUG: Loading tokenizer for {model_id}...") + tokenizer = AutoTokenizer.from_pretrained(model_id) + tokenizer.pad_token = tokenizer.eos_token + tokenizer.padding_side = "right" + + # 2. BitsAndBytes Configuration (QLoRA) + print(">>> DEBUG: Configuring BitsAndBytes for 4-bit quantization...") + bnb_config = BitsAndBytesConfig( + load_in_4bit=True, + bnb_4bit_use_double_quant=True, + bnb_4bit_quant_type="nf4", + bnb_4bit_compute_dtype=torch.bfloat16 + ) + + # 3. Load Base Model + print(f">>> DEBUG: Loading base model from {model_id}...") + model = AutoModelForCausalLM.from_pretrained( + model_id, + quantization_config=bnb_config, + device_map="auto", + trust_remote_code=True + ) + + print(">>> DEBUG: Preparing model for kbit training...") + model = prepare_model_for_kbit_training(model) + + # 4. LoRA Configuration + print(">>> DEBUG: Setting up LoRA configuration (Rank=16)...") + peft_config = LoraConfig( + r=16, + lora_alpha=32, + target_modules=["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"], + lora_dropout=0.05, + bias="none", + task_type="CAUSAL_LM" + ) + + # 5. Load Dataset + print(f">>> DEBUG: Loading dataset from {data_file}...") + dataset = load_dataset("json", data_files=data_file, split="train") + + # 6. Training Arguments + print(">>> DEBUG: Defining training arguments...") + training_args = SFTConfig( + output_dir=output_dir, + per_device_train_batch_size=1, + gradient_accumulation_steps=16, + learning_rate=2e-4, + logging_steps=10, + max_steps=200, + save_steps=100, + optim="paged_adamw_8bit", + bf16=True, + fp16=False, + dataset_text_field="text", + max_length=1024, + gradient_checkpointing=True, + packing=False + ) + + # 7. Initialize Trainer + print(">>> DEBUG: Initializing SFTTrainer...") + trainer = SFTTrainer( + model=model, + train_dataset=dataset, + peft_config=peft_config, + args=training_args, + processing_class=tokenizer, + ) + + # 8. Start Training + print(">>> DEBUG: Starting training loop...") + trainer.train() + + # 9. Save the Adapter + print(">>> DEBUG: Saving the fine-tuned adapter to ./qwen-resercher...") + trainer.model.save_pretrained("qwen-resercher") + tokenizer.save_pretrained("qwen-resercher") + print(">>> DEBUG: Training complete.") + +if __name__ == "__main__": + train()