From 066fee04e687f138d05eef5de102965e6750ddc6 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Fri, 28 Aug 2026 20:09:18 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: hq-bench/coreb-code-reranker Source: Original Platform --- .gitattributes | 36 ++++++++++++ README.md | 121 +++++++++++++++++++++++++++++++++++++++++ chat_template.jinja | 15 +++++ config.json | 71 ++++++++++++++++++++++++ generation_config.json | 13 +++++ model.safetensors | 3 + tokenizer.json | 3 + tokenizer_config.json | 15 +++++ 8 files changed, 277 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 chat_template.jinja create mode 100644 config.json create mode 100644 generation_config.json create mode 100644 model.safetensors create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..2a4f368 --- /dev/null +++ b/README.md @@ -0,0 +1,121 @@ +--- +license: apache-2.0 +base_model: Qwen/Qwen3-Reranker-4B +tags: + - code-search + - reranker + - code-retrieval + - peft + - lora +language: + - en + - code +datasets: + - hq-bench/coreb +pipeline_tag: text-classification +library_name: transformers +--- + +[![Project Page](https://img.shields.io/badge/Project-Page-blue)](https://hq-bench.github.io/coreb-page/) +[![arXiv](https://img.shields.io/badge/arXiv-2605.04615-b31b1b.svg)](https://arxiv.org/abs/2605.04615) +[![Dataset](https://img.shields.io/badge/HuggingFace-Dataset-yellow)](https://huggingface.co/datasets/hq-bench/coreb) +[![Code](https://img.shields.io/badge/GitHub-Code-black)](https://github.com/hq-bench/coreb) + +# CoREB-Reranker + +**CoREB-Reranker** is a code reranker fine-tuned from [Qwen3-Reranker-4B](https://huggingface.co/Qwen/Qwen3-Reranker-4B) via LoRA on a mixed reranker corpus. It is the **only reranker we evaluate that achieves consistent gains across all three code search tasks** (text-to-code, code-to-text, and code-to-code). + +## Highlights + +- Fine-tuned from Qwen3-Reranker-4B using LoRA (rank=16, alpha=16) on **3.1M training samples** from a mixed corpus +- Evaluated on CoREB v202603 (problem-disjoint from training set, no data leakage) +- Achieves **positive reranking delta on all three tasks**, unlike all off-the-shelf rerankers tested + +## Reranking Results (nDCG@10 Delta %) + +Reranking delta on CoREB v202603, using C2LLM-7B as the first-stage retriever: + +| Reranker | Text-to-Code | Code-to-Text | Code-to-Code | +|----------|:---:|:---:|:---:| +| Jina Reranker v2 | -8.3 | -22.4 | -8.8 | +| Jina Reranker v3 | -2.2 | -5.0 | -0.1 | +| Qwen3-Reranker-0.6B | -0.6 | -8.2 | -2.3 | +| Qwen3-Reranker-4B | -0.1 | -3.2 | +3.3 | +| **CoREB-Reranker (ours)** | **+1.1** | **+0.8** | **+5.1** | + +## Training Details + +- **Base model**: [Qwen/Qwen3-Reranker-4B](https://huggingface.co/Qwen/Qwen3-Reranker-4B) +- **Method**: LoRA (rank=16, alpha=16, dropout=0.05) +- **Target modules**: q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj +- **Training data**: A mixed reranker corpus consisting of [CoREB v202602](https://huggingface.co/datasets/hq-bench/coreb), [CodeSearchNet](https://github.com/github/CodeSearchNet) (code-to-code, code-to-text, text-to-code), [APPS](https://github.com/hendrycks/apps), [CosQA](https://github.com/Jun-jie-Huang/CosQA), and [CodeFeedback](https://github.com/OpenCodeInterpreter/OpenCodeInterpreter) (single-turn and multi-turn). Each record is normalized into binary reranking examples (instruction, query, document, yes/no). Positives are duplicated twice; one easy negative and one hard negative are sampled per record. +- **Evaluation data**: CoREB v202603 (problem-disjoint from CoREB v202602 training split; covers a different contest time window) +- **Training samples**: ~3.1M binary reranking examples across text-to-code, code-to-text, and code-to-code tasks +- **Top-k retrieval for reranking**: 128 + +## Usage + +CoREB-Reranker follows the same usage pattern as Qwen3-Reranker. The instruction is **task-specific** — use the appropriate one for your retrieval task: + +```python +from enum import Enum +from transformers import AutoModelForCausalLM, AutoTokenizer +import torch + +class Task(Enum): + TEXT_TO_CODE = "Given a natural language programming task, retrieve code that correctly solves or implements the task." + CODE_TO_CODE = "Given a code snippet, retrieve code that is semantically equivalent or solves the same task." + CODE_TO_TEXT = "Given a code snippet, retrieve the natural language description or problem statement that best matches the code." + +model_id = "hq-bench/coreb-code-reranker" +tokenizer = AutoTokenizer.from_pretrained(model_id, trust_remote_code=True) +model = AutoModelForCausalLM.from_pretrained(model_id, torch_dtype=torch.bfloat16, trust_remote_code=True) +model.eval() + +PREFIX = '<|im_start|>system\nJudge whether the Document meets the requirements based on the Query and the Instruct provided. Note that the answer can only be "yes" or "no".<|im_end|>\n<|im_start|>user\n' +SUFFIX = "<|im_end|>\n<|im_start|>assistant\n" +yes_id = tokenizer.convert_tokens_to_ids("yes") +no_id = tokenizer.convert_tokens_to_ids("no") + +def score(query: str, document: str, task: Task) -> float: + prompt = f"{PREFIX}: {task.value}\n: {query}\n: {document}{SUFFIX}" + inputs = tokenizer(prompt, return_tensors="pt", truncation=True, max_length=4096) + with torch.no_grad(): + logits = model(**inputs).logits[0, -1, :] + return (logits[yes_id] - logits[no_id]).item() + +# Text-to-Code: natural language query -> code +print(score( + query="binary search implementation", + document="def binary_search(arr, target):\n lo, hi = 0, len(arr) - 1\n ...", + task=Task.TEXT_TO_CODE, +)) + +# Code-to-Code: code -> semantically equivalent code +print(score( + query="def binary_search(arr, target): ...", + document="int binarySearch(int[] arr, int target) { ... }", + task=Task.CODE_TO_CODE, +)) + +# Code-to-Text: code -> problem description +print(score( + query="def binary_search(arr, target): ...", + document="Find the index of a target value in a sorted array using binary search.", + task=Task.CODE_TO_TEXT, +)) +``` + +For batch reranking with the CoREB evaluation pipeline, see the [CoREB repository](https://github.com/hq-bench/coreb). + +## Citation + +```bibtex +@article{xue2026coreb, + title={Beyond Retrieval: A Multitask Benchmark and Reranker for Code Search}, + author={Xue, Siqiao and Liao, Zihan and Qin, Jin and Zhang, Ziyin and Mu, Yixiang and Zhou, Fan and Yu, Hang}, + journal={arXiv preprint arXiv:2605.04615}, + year={2026}, + url={https://arxiv.org/abs/2605.04615} +} +``` diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..63b97f2 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,15 @@ +{%- set instruction = messages | selectattr("role", "eq", "system") | map(attribute="content") | first | default("Given a web search query, retrieve relevant passages that answer the query") -%} +{%- set query_text = messages | selectattr("role", "eq", "query") | map(attribute="content") | first -%} +{%- set document_text = messages | selectattr("role", "eq", "document") | map(attribute="content") | first -%} +<|im_start|>system +Judge whether the Document meets the requirements based on the Query and the Instruct provided. Note that the answer can only be "yes" or "no".<|im_end|> +<|im_start|>user +: {{ instruction }} +: {{ query_text }} +: {{ document_text }}<|im_end|> +<|im_start|>assistant + + + + + diff --git a/config.json b/config.json new file mode 100644 index 0000000..9cab3ef --- /dev/null +++ b/config.json @@ -0,0 +1,71 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2560, + "initializer_range": 0.02, + "intermediate_size": 9728, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 40960, + "max_window_layers": 36, + "model_type": "qwen3", + "num_attention_heads": 32, + "num_hidden_layers": 36, + "num_key_value_heads": 8, + "pad_token_id": null, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.6.0", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151669 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..29520f4 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_k": 20, + "top_p": 0.95, + "transformers_version": "5.6.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..e60798c --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f3222f8eae96b717d299dadad3099a85f35db32b3b73baa57bd6e96c02fbb35e +size 8043615040 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..23d7c66 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,15 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +}