From 0d536f4e9bb50256d2b2d0a3f8a067ebd8c9db6f Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sat, 5 Sep 2026 19:37:15 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: WithinUsAI/GPT5.1-High.Reasoning.Codex-0.4B-GGUF Source: Original Platform --- .gitattributes | 38 ++++ GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf | 3 + GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf | 3 + GPT5.1-high-reasoning-codex-0.4B.f16.gguf | 3 + README.md | 184 +++++++++++++++++++ 5 files changed, 231 insertions(+) create mode 100644 .gitattributes create mode 100644 GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf create mode 100644 GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf create mode 100644 GPT5.1-high-reasoning-codex-0.4B.f16.gguf create mode 100644 README.md diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..446cd07 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,38 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +GPT5.1-high-reasoning-codex-0.4B.f16.gguf filter=lfs diff=lfs merge=lfs -text +GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf b/GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf new file mode 100644 index 0000000..b4c555e --- /dev/null +++ b/GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:79116366094e5c38d5c8da3921c1c3e7a142144d9833e6e48abc47b4d7ce04bb +size 241763552 diff --git a/GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf b/GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf new file mode 100644 index 0000000..dc24afe --- /dev/null +++ b/GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ab970abe6ac7a01acef485b9041acadd40e6286f055693961cd3dd09f69dcca6 +size 273810656 diff --git a/GPT5.1-high-reasoning-codex-0.4B.f16.gguf b/GPT5.1-high-reasoning-codex-0.4B.f16.gguf new file mode 100644 index 0000000..ee3b209 --- /dev/null +++ b/GPT5.1-high-reasoning-codex-0.4B.f16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ed35025871d16cf871381ea7e608e13e6c7c995beaeb69e81c62a0f7e40edbb0 +size 714171136 diff --git a/README.md b/README.md new file mode 100644 index 0000000..652216e --- /dev/null +++ b/README.md @@ -0,0 +1,184 @@ +--- +license: other +library_name: llama.cpp +tags: + - gguf + - gpt2 + - code + - coder + - reasoning + - text-generation + - withinusai +language: + - en +model_type: gguf +inference: false +--- + +# GPT5.1-high-reasoning-codex-0.4B-GGUF + +**GPT5.1-high-reasoning-codex-0.4B-GGUF** is a compact GGUF language model release from **WithIn Us AI**, intended for local inference and lightweight coding or reasoning-oriented experiments. + +This repository provides quantized GGUF builds for efficient use with **llama.cpp** and compatible runtimes. + +## Model Summary + +This model is designed for: + +- lightweight local inference +- coding and prompt-based development assistance +- compact reasoning-style experiments +- offline chat and text generation workflows +- small-footprint deployments + +Because this is a **0.4B** parameter class model, it is best suited for fast iteration, simple coding tasks, prompt experiments, structured text generation, and lightweight assistant workflows rather than heavy long-context reasoning or complex production-grade coding autonomy. + +## Repository Contents + +This repository currently includes the following files: + +- `GPT5.1-high-reasoning-codex-0.4B.Q4_K_M.gguf` +- `GPT5.1-high-reasoning-codex-0.4B.Q5_K_M.gguf` +- `GPT5.1-high-reasoning-codex-0.4B.f16.gguf` + +## Quantization Variants + +### Q4_K_M +A smaller and more memory-efficient quantization for lower RAM usage and faster local inference. + +### Q5_K_M +A slightly larger quantization that may provide somewhat better output quality while remaining efficient. + +### F16 +A higher-precision GGUF variant intended for users who want the least quantization loss and have more memory available. + +## Architecture + +The repository metadata currently identifies the architecture as: + +- **gpt2** + +## Intended Use + +Recommended use cases include: + +- local coding assistant experiments +- toy and lightweight software-help workflows +- code completion and code drafting +- debugging ideas and implementation suggestions +- instruction-following tests +- prompt engineering experiments +- low-resource local deployments + +## Out-of-Scope Use + +This model should not be relied on for: + +- legal advice +- medical advice +- financial advice +- safety-critical automation +- production code generation without review +- security-sensitive decisions without human verification + +All generated code should be reviewed, tested, and validated before use. + +## Performance Expectations + +As a compact **0.4B** model, this release trades raw capability for speed, portability, and lower hardware requirements. It may perform well for: + +- short code snippets +- compact prompts +- structured assistant replies +- lightweight reasoning-style tasks + +It may struggle with: + +- long and complex codebases +- deep multi-step reasoning +- strict factual reliability +- advanced tool orchestration +- heavy instruction retention over long prompts + +## Prompting Tips + +For best results, use prompts that are: + +- specific +- short to medium length +- explicit about the desired language or format +- clear about constraints +- direct about whether you want code, explanation, or both + +### Example prompts + +**Code generation** +> Write a Python function that reads a JSON file, validates required fields, and returns a cleaned list of records. + +**Refactoring** +> Refactor this JavaScript function to be more readable and add basic error handling. + +**Debugging** +> Explain why this Python code raises a KeyError and show a corrected version. + +## Hardware and Runtime Notes + +This model is packaged in **GGUF** format, which is suitable for **llama.cpp**-style local inference stacks and related frontends / runtimes that support GGUF models. + +Typical choices: + +- use **Q4_K_M** for smaller memory usage +- use **Q5_K_M** for a quality / size balance +- use **F16** when memory allows and you want higher precision + +## Limitations + +Like other small language models, this model may: + +- hallucinate APIs, functions, or package behavior +- generate incorrect code +- produce insecure code patterns +- make reasoning mistakes +- lose instruction fidelity on longer prompts +- require prompt retries for acceptable output quality + +Human oversight is strongly recommended. + +## Training / Lineage + +This repository is presented as a **WithIn Us AI** model release and GGUF packaging distribution. + +If you want, this section can be expanded later with: + +- base model lineage +- fine-tuning details +- merge methodology +- dataset attribution +- training objective +- chat template recommendations + +## License + +This repository currently uses a custom / non-standard license field approach in this model card draft: + +- `license: other` + +You can replace this section with your exact **WithIn Us AI custom license terms**. If this model is derived from upstream weights or datasets, include: + +- attribution to the original base model creators +- attribution to any third-party datasets used +- clear statement that WithIn Us AI claims authorship of the fine-tuning / merging / packaging process, not ownership of third-party source materials unless applicable + +## Acknowledgments + +Thanks to: + +- the open-source local inference ecosystem +- GGUF and llama.cpp tooling contributors +- the broader Hugging Face community +- all upstream creators whose work may have contributed to the model’s lineage + +## Disclaimer + +This model may produce inaccurate, biased, insecure, or incomplete outputs. +Use responsibly, and verify important results before real-world use. \ No newline at end of file