初始化项目,由ModelHub XC社区提供模型

Model: aab20abdullah/qwen_OSINT
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-04 23:36:19 +08:00
commit f77bcad5e4
7 changed files with 496 additions and 0 deletions

38
.gitattributes vendored Normal file
View File

@@ -0,0 +1,38 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
qwen3-4b-thinking-2507.Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
qwen3-4b-thinking-2507.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
qwen3-4b-thinking-2507.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text

54
Modelfile Normal file
View File

@@ -0,0 +1,54 @@
FROM qwen3-4b-thinking-2507.Q5_K_M.gguf
TEMPLATE """
{{- $lastUserIdx := -1 -}}
{{- range $idx, $msg := .Messages -}}
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
{{- end }}
{{- if or .System .Tools }}<|im_start|>system
{{ if .System }}
{{ .System }}
{{- end }}
{{- if .Tools }}
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{{- range .Tools }}
{"type": "function", "function": {{ .Function }}}
{{- end }}
</tools>
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
<tool_call>
{"name": <function-name>, "arguments": <args-json-object>}
</tool_call>
{{- end -}}
<|im_end|>
{{ end }}
{{- range $i, $_ := .Messages }}
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
{{- if eq .Role "user" }}<|im_start|>user
{{ .Content }}<|im_end|>
{{ else if eq .Role "assistant" }}<|im_start|>assistant
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
<think>{{ .Thinking }}</think>
{{ end -}}
{{ if .Content }}{{ .Content }}
{{- else if .ToolCalls }}<tool_call>
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
{{ end }}</tool_call>
{{- end }}{{ if not $last }}<|im_end|>
{{ end }}
{{- else if eq .Role "tool" }}<|im_start|>user
<tool_response>
{{ .Content }}
</tool_response><|im_end|>
{{ end }}
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
{{ end }}
{{- end }}
"""

326
README.md Normal file
View File

@@ -0,0 +1,326 @@
---
base_model: Qwen/Qwen3-4B
language:
- en
- ar
- tr
license: apache-2.0
library_name: gguf
tags:
- gguf
- qwen3
- conversational
- osint
- cybersecurity
- fine-tuned
- security
- intelligence
pipeline_tag: text-generation
model_name: Qwen-OSINT
quantized_by: aab20abdullah
---
# Qwen-OSINT
<div align="center">
<img src="https://img.shields.io/badge/Model-Qwen2.5--7B-blue?style=flat-square" alt="Model">
<img src="https://img.shields.io/badge/License-Apache%202.0-green?style=flat-square" alt="License">
<img src="https://img.shields.io/badge/Task-OSINT-orange?style=flat-square" alt="Task">
<img src="https://img.shields.io/badge/Dataset-Multi--source-red?style=flat-square" alt="Dataset">
</div>
---
## 📋 Table of Contents
- [Overview](#overview)
- [Features](#features)
- [Model Details](#model-details)
- [Installation](#installation)
- [Quick Start](#quick-start)
- [Usage Examples](#usage-examples)
- [Ethical Guidelines](#ethical-guidelines)
- [Limitations](#limitations)
- [License](#license)
- [Acknowledgments](#acknowledgments)
---
## 🎯 Overview
**Qwen-OSINT** is a specialized large language model fine-tuned from [Qwen2.5-7B](https://huggingface.co/Qwen/Qwen2.5-7B) specifically designed for Open Source Intelligence (OSINT) operations. This model leverages advanced natural language processing capabilities to assist security researchers, analysts, and investigators in gathering, analyzing, and synthesizing information from publicly available sources.
### What is OSINT?
Open Source Intelligence (OSINT) refers to the practice of collecting and analyzing information from publicly available sources to support decision-making processes. This includes data from:
- 🌐 Social media platforms
- 📰 News articles and publications
- 🔍 Search engines and databases
- 💼 Professional networks
- 🌐 Public records and government databases
---
## ✨ Features
| Feature | Description |
|---------|-------------|
| 🔎 **Advanced Search Analysis** | Efficiently analyzes search queries and identifies relevant intelligence sources |
| 📊 **Data Synthesis** | Consolidates information from multiple sources into coherent summaries |
| 🔐 **Security Analysis** | Supports threat analysis and vulnerability assessment tasks |
| 📝 **Report Generation** | Generates structured intelligence reports in various formats |
| 🌐 **Multi-language Support** | Processes and analyzes content in multiple languages |
| 🛡️ **Ethical Compliance** | Built with safety guidelines to ensure responsible use |
---
## 📊 Model Details
| Attribute | Value |
|-----------|-------|
| **Base Model** | Qwen2.5-7B-Instruct |
| **Framework** | Transformers (Hugging Face) |
| **Training Method** | Supervised Fine-tuning (SFT) |
| **Vocabulary Size** | 151,669 tokens |
| **Architecture** | Transformer-based Decoder |
| **Precision** | FP16 / INT8 compatible |
### Training Configuration
```
- Learning Rate: 2e-5
- Batch Size: 8
- Epochs: 3
- Warmup Steps: 100
- Max Sequence Length: 8192
```
---
## 🔧 Installation
### Prerequisites
```
Python >= 3.8
PyTorch >= 2.0
transformers >= 4.35.0
accelerate >= 0.20.0
bitsandbytes >= 0.40.0 (for quantization)
```
### Install Dependencies
```bash
pip install transformers torch accelerate bitsandbytes
```
### Download the Model
```python
from transformers import AutoModelForCausalLM, AutoTokenizer
model_name = "aab20abdullah/qwen_OSINT"
# Download tokenizer
tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)
# Download model
model = AutoModelForCausalLM.from_pretrained(
model_name,
device_map="auto",
trust_remote_code=True
)
```
---
## 🚀 Quick Start
### Basic Usage
```python
import torch
from transformers import AutoModelForCausalLM, AutoTokenizer
model_name = "aab20abdullah/qwen_OSINT"
tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)
model = AutoModelForCausalLM.from_pretrained(model_name, trust_remote_code=True)
def generate_intelligence(prompt, max_length=512):
messages = [{"role": "user", "content": prompt}]
text = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
inputs = tokenizer([text], return_tensors="pt").to("cuda")
outputs = model.generate(
**inputs,
max_new_tokens=max_length,
temperature=0.7,
top_p=0.9
)
response = tokenizer.decode(outputs[0], skip_special_tokens=True)
return response.split("assistant")[-1].strip()
# Example
result = generate_intelligence("Analyze the key elements of a threat intelligence report.")
print(result)
```
### Quantized Version (Lower Memory Usage)
```python
from transformers import AutoModelForCausalLM, BitsAndBytesConfig
quantization_config = BitsAndBytesConfig(
load_in_8bit=True
)
model = AutoModelForCausalLM.from_pretrained(
model_name,
quantization_config=quantization_config,
device_map="auto",
trust_remote_code=True
)
```
---
## 💡 Usage Examples
### Example 1: Search Query Analysis
```python
prompt = """Analyze the following search query and suggest improvements for OSINT research:
Query: "site:linkedin.com cybersecurity analyst" """
result = generate_intelligence(prompt)
print(result)
```
### Example 2: Data Source Evaluation
```python
prompt = """Evaluate the reliability and credibility of the following OSINT sources:
1. Government statistical databases
2. Academic research papers
3. Social media platforms
4. Open-source code repositories"""
result = generate_intelligence(prompt)
print(result)
```
### Example 3: Threat Analysis Framework
```python
prompt = """Using the MITRE ATT&CK framework, analyze potential threat vectors for:
- Phishing attacks
- Network intrusion
- Data exfiltration
Provide recommendations for detection and prevention."""
result = generate_intelligence(prompt)
print(result)
```
---
## 🛡️ Ethical Guidelines
> ⚠️ **IMPORTANT**: This model is designed for **legitimate OSINT research** only.
### Acceptable Use Cases ✅
- 🔍 Security research and vulnerability assessment
- 📊 Threat intelligence analysis
- 🛡️ Organizational security posture evaluation
- 📚 Academic research in cybersecurity
- 🏢 Corporate due diligence
### Prohibited Use Cases ❌
- 🚫 Unauthorized surveillance
- 🚫 Invasion of privacy
- 🚫 Harassment or stalking
- 🚫 Illegal activities
- 🚫 Content generation for malicious purposes
### Responsible Use Principles
1. **Transparency**: Clearly identify yourself when conducting OSINT operations
2. **Legality**: Ensure compliance with applicable laws and regulations
3. **Proportionality**: Collect only information necessary for your objectives
4. **Security**: Protect collected data appropriately
5. **Accountability**: Maintain records of your OSINT activities
---
## ⚠️ Limitations
| Limitation | Description |
|------------|-------------|
| ⚡ **Computational Resources** | Requires GPU with sufficient VRAM for optimal performance |
| 🎯 **Accuracy** | May generate plausible but incorrect information - always verify |
| 🌍 **Language Coverage** | Best performance in English; other languages may vary |
| 📅 **Knowledge Cutoff** | Training data has a knowledge cutoff date |
| 🔒 **Sensitive Data** | Not designed to handle highly classified or sensitive information |
---
## 📄 License
This model is released under the **Apache 2.0 License**.
```
Copyright 2024 aab20abdullah
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
```
### Base Model License
The base model [Qwen2.5](https://huggingface.co/Qwen/Qwen2.5-7B) is licensed under the [Qwen Research License](https://github.com/QwenLM/Qwen2/blob/main/Qwen2.5_LICENSE).
---
## 🙏 Acknowledgments
- **Alibaba Cloud** - For developing the Qwen2.5 model architecture
- **Hugging Face** - For providing the model hosting infrastructure
- **Open Source Community** - For continuous contributions to AI safety and ethics
---
## 📬 Contact
- **Model Repository**: [huggingface.co/aab20abdullah/qwen_OSINT](https://huggingface.co/aab20abdullah/qwen_OSINT)
- **Author**: [aab20abdullah](https://huggingface.co/aab20abdullah)
---
## 📝 Citation
If you use this model in your research or project, please cite:
```bibtex
@model{qwen_osint,
author = {aab20abdullah},
title = {Qwen-OSINT: A Specialized Model for Open Source Intelligence},
year = {2024},
publisher = {Hugging Face},
url = {https://huggingface.co/aab20abdullah/qwen_OSINT}
}
```
---
<div align="center">
<p>⭐ If you find this model useful, please consider giving it a star!</p>
<p>Made with ❤️ for the OSINT community</p>
</div>

69
config.json Normal file
View File

@@ -0,0 +1,69 @@
{
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"torch_dtype": "float16",
"eos_token_id": 151645,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2560,
"initializer_range": 0.02,
"intermediate_size": 9728,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 262144,
"max_window_layers": 36,
"model_type": "qwen3",
"num_attention_heads": 32,
"num_hidden_layers": 36,
"num_key_value_heads": 8,
"pad_token_id": 151669,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 5000000,
"sliding_window": null,
"tie_word_embeddings": true,
"unsloth_fixed": true,
"unsloth_version": "2026.4.4",
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a38df23b52a42d649a9ca2138be217c8d4cc03cbc743c041582803beda0a2d3d
size 2497280256

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:612998a8fca4889db8ebcf77a8c9a3449fbf5ffe641b3f7e84fdc0c60e683bb0
size 2889513216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7fa591201fd7edc9a4f1693484d650ed4e1d52e57c3df0cdc9ed86ebaea92836
size 4280404736