From cd073b636227324653bb7573055e0a30cf8f381e Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sat, 25 Jul 2026 09:11:10 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: pbhappliedsystems/qwen-2.5-14B-instruct-1m-gguf-F16 Source: Original Platform --- .gitattributes | 36 + LICENSE_Apache-2.0 | 170 +++++ README.md | 390 +++++++++++ ...wen2.5_14B_Instruct_1M_20260210_215029.csv | 630 ++++++++++++++++++ qwen-2.5-14B-instruct-1m-gguf-F16.gguf | 3 + rollup.json | 267 ++++++++ run_manifest.json | 32 + 7 files changed, 1528 insertions(+) create mode 100644 .gitattributes create mode 100644 LICENSE_Apache-2.0 create mode 100644 README.md create mode 100644 comparison_results_v7_21_Qwen2.5_14B_Instruct_1M_20260210_215029.csv create mode 100644 qwen-2.5-14B-instruct-1m-gguf-F16.gguf create mode 100644 rollup.json create mode 100644 run_manifest.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..3bdc67d --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +qwen-2.5-14B-instruct-1m-gguf-F16.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/LICENSE_Apache-2.0 b/LICENSE_Apache-2.0 new file mode 100644 index 0000000..aa03dc8 --- /dev/null +++ b/LICENSE_Apache-2.0 @@ -0,0 +1,170 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship made available under + the License, as indicated by a copyright notice that is included in + or attached to the work (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean, as submitted to the Licensor for inclusion + in the Work by the copyright owner or by an individual or Legal Entity + authorized to submit on behalf of the copyright owner. For the purposes + of this definition, "submitted" means any form of electronic, verbal, or + written communication sent to the Licensor or its representatives, + including but not limited to communication on electronic mailing lists, + source code control systems, and issue tracking systems that are managed + by, or on behalf of, the Licensor for the purpose of discussing and + improving the Work, but excluding communication that is conspicuously + marked or otherwise designated in writing by the copyright owner as + "Not a Contribution." + + "Contributor" shall mean Licensor and any Legal Entity on behalf of + whom a Contribution has been received by the Licensor and subsequently + incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a cross-claim + or counterclaim in a lawsuit) alleging that the Work or a Contribution + incorporated within the Work constitutes direct or contributory patent + infringement, then any patent licenses granted to You under this License + for that Work shall terminate as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or Derivative Works + a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, You must include a readable copy of the attribution + notices contained within such NOTICE file, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own license statement for Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such derivative works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or exemplary damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or all other + commercial damages or losses), even if such Contributor has been + advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS diff --git a/README.md b/README.md new file mode 100644 index 0000000..280d049 --- /dev/null +++ b/README.md @@ -0,0 +1,390 @@ +--- +language: + - en + - zh + - fr + - es + - pt + - de + - it + - ru + - ja + - ko + - ar +license: apache-2.0 +base_model: Qwen/Qwen2.5-14B-Instruct-1M +tags: + - gguf + - f16 + - full-precision + - qwen + - qwen2.5 + - instruct + - llama-cpp + - long-context + - 1m-context + - agentic + - structured-output + - pbh-applied-systems + - quant-eval + - baseline +--- + +# Qwen2.5-14B-Instruct-1M Β· GGUF F16 + +**Converted and evaluated by [PBH Applied Systems, LLC](https://pbhappliedsystems.com)** +β€” Applied AI/ML Consulting Β· LLM Optimization & Deployment Β· Quantized AI Infrastructure + +> πŸ”¬ **This repository is part of a production-oriented evaluation series.** Every model published under [`pbhappliedsystems`](https://huggingface.co/pbhappliedsystems) has been independently evaluated using **quant_eval v7.21** β€” a proprietary behavioral evaluation harness developed by PBH Applied Systems. + +> πŸ“Œ **This is the full-precision F16 baseline repository.** The evaluated Q4\_K\_M deployment variant is published at [`pbhappliedsystems/qwen-2.5-14B-instruct-1m-gguf-Q4-K-M`](https://huggingface.co/pbhappliedsystems/qwen-2.5-14B-instruct-1m-gguf-Q4-K-M). That card documents the complete cross-series comparisons, context window VRAM guide, and deployment recommendations. **The Q4\_K\_M variant is the recommended choice for all deployments** β€” it achieves identical behavioral results at 21.1Γ— faster inference. + +--- + +## Try the Live AI Agent Demo + +[**Launch the PBH Applied Systems AI Agent Demo β†’**](https://pbhappliedsystems.com/assistant.html) + +This model is part of the PBH Applied Systems evaluated model series that supports the live AI Agent Demo. The demo lets visitors interact with production-style agent workflows powered by open-weight language models evaluated through PBH Applied Systems' `quant_eval` framework. + +The F16 model serves a different role than the Q4_K_M deployment variant. F16 is the full-precision baseline used to measure what the model can do before quantization. `quant_eval` then compares the quantized model against this baseline to identify which capabilities are preserved, which degrade, and which tasks require guardrails or a higher-precision deployment. + +This comparison is central to the demo. It helps determine which model belongs in which agent role: + +- Reasoning models are selected for planning, analysis, and auditable decision workflows. +- Document models are selected for long-context extraction, summarization, and structured Q&A. +- Code models are selected for task completion, structured output, API scaffolding, and automation workflows. +- Quantized variants are selected when they preserve enough behavior to reduce cost, latency, and GPU requirements. +- F16 variants remain important when maximum fidelity, cleaner tool execution, or reduced quantization risk matters more than speed or cost. + +The live demo shows the deployment side of that process. The F16 card documents the reference behavior. The Q4_K_M card shows what changes after compression. Together, they explain how PBH Applied Systems uses `quant_eval` to choose the correct LLM for the correct agent type instead of guessing from model size or leaderboard reputation. + +--- + +## Model Description + +This repository contains the **full-precision F16 GGUF** of [`Qwen/Qwen2.5-14B-Instruct-1M`](https://huggingface.co/Qwen/Qwen2.5-14B-Instruct-1M), a 14-billion parameter instruction-tuned model from Alibaba Cloud featuring a **1,000,000-token context window**. + +In the PBH Applied Systems evaluation pipeline, this F16 run (`20260210_215029`) operated in cache-generation mode (`skip_quant=true`), producing the `full_weight_cache.json` used as the reference baseline for the subsequent Q4\_K\_M comparison run (`20260210_235131`). The evaluation results here are the source of the F16 baseline data shown in the Q4\_K\_M card β€” timing profiles and raw outputs are identical across both runs, confirming clean cache reuse and full run integrity. + +### Key Characteristics + +- **Parameters:** 14B +- **Format:** GGUF F16 (full precision) +- **File size:** 29.5 GB +- **SHA256:** `de08ea9c41234ef83b7aacf07f9ebc3cbaa20ca8aeb5f6417758a8798660aaa9` +- **Context window:** **1,000,000 tokens** +- **Minimum VRAM (GPU inference):** ~32 GB (short context) β€” scales with context length +- **Recommended GPU tier:** A100 40 GB Β· 2Γ— RTX 4090 +- **Inference speed (eval hardware):** avg **56.623 sec/case** on RTX 4090 +- **License:** Apache 2.0 + +> **On inference speed:** The F16 model averages 56.6 sec/case β€” nearly one minute per structured inference task on an RTX 4090. The json\_01 case takes **376.99 seconds** (over 6 minutes). For this model, the **Q4\_K\_M variant (2.683 sec/case average)** is the operationally viable choice on all but the highest-VRAM multi-GPU setups. Both produce identical behavioral results. + +--- + +## PBH Applied Systems Evaluation β€” quant\_eval v7.21 + +> **Evaluation conducted by PBH Applied Systems, LLC using quant_eval v7.21** +> Run ID: `20260210_215029` Β· Fixtures: `golden_oracle_fixtures_v7_21` (SHA256: `6d71a0b9147c...`) Β· Seed: 42 +> Hardware: NVIDIA RTX 4090 Β· Runner: `full_weight_transformers` (F16 only) Β· Total rows: 42 + +### Per-Family Pass Rates β€” F16 (`full_weight_transformers`) + +| Family | N | Pass Rate | Avg Secs | Bucket Score | Q4\_K\_M Parity | +|---|---:|---:|---:|---:|---| +| json\_multistep | 5 | **0.800** | 133.21 | 2.200 | βœ… Identical | +| stateful\_followup | 2 | **1.000** | 13.75 | 2.000 | βœ… Identical | +| toolcall\_only | 2 | 0.000 | 15.36 | 1.000 | βœ… Identical | +| mixed\_brief\_json | 2 | **1.000** | 19.40 | 2.000 | βœ… Identical | +| toolcall | 2 | **1.000** | 25.95 | **11.000** | βœ… Identical | +| json | 4 | n/a | 143.87 | 10.000 | βœ… Identical | +| fuzz | 20 | n/a | 49.21 | 10.000 | βœ… Identical | +| mcq | 5 | n/a | 0.73 | **1.000** | βœ… Identical | + +**Every family result is identical between F16 and Q4\_K\_M.** This model is the only one in the evaluated series with zero measurable quantization degradation across all behavioral families. + +--- + +## F16-Specific Observations + +### toolcall β€” bucket=11, Clean Final Answers at F16 + +Both `toolcall` cases pass with the maximum bucket score at F16 β€” no role-token contamination, no EOS tokens, no missing answers: + +| Case | Raw Output | Expected | Result | +|---|---|---|---| +| tool\_01 | `{...add(2,3)...} 5` | `5` | βœ… bucket=11 | +| tool\_02 | `{...add(10,-4)...} 6` | `6` | βœ… bucket=11 | + +This matches the Q4\_K\_M runner exactly. Qwen2.5-14B-Instruct-1M at F16 does not exhibit the role-token contamination (PARTICULAR: annotation, garbled prefixes) documented in the Qwen2.5-7B F16 evaluation, nor the EOS contamination of the smaller Qwen Q4\_K\_M variants. Clean output requires no post-processing at this precision level. + +### toolcall\_only β€” "left"/"right" Schema Vocabulary at F16 + +Both `toolcall_only` cases use `"left"/"right"` as argument keys β€” the same vocabulary as the F16 runner for this model: + +| Case | Raw Output | +|---|---| +| toolonly\_01 | `{"tool": "add", "left": 5, "right": 10}` | +| toolonly\_02 | `{"tool": "add", "left": 25, "right": 75}` | + +Contrast with Q4\_K\_M which uses a nested `"input"` object. Both runners fail `args_ok` but with different wrong schemas. Explicit key names in the system prompt resolve this at both precision levels. + +### MCQ β€” Perfect 5/5 at F16 + +All five MCQ cases return a clean single-character answer in ~0.73 seconds: + +``` +mcq_01: B | mcq_02: B | mcq_03: C | mcq_04: B | mcq_05: B +``` + +No empty output, no invalid choices, no A-bias. At F16 MCQ is the fastest family in this run β€” the 1M context window doesn't affect short-response tasks. + +### json\_01 β€” 376.99-Second Outlier + +`json_01` at F16 takes **376.99 seconds** while json\_02 through json\_04 run 63–70 seconds each. The output is correct (bucket=10). This extreme variance is a characteristic of the 1M context window at full precision β€” certain inputs trigger substantially longer generation sequences before the model settles on its brief JSON output. The Q4\_K\_M runner takes 3.38 seconds on the same case, confirming this is a precision + context-window interaction, not a fixture complexity issue. + +### stateful\_followup β€” No Turn-2 Contamination + +Unlike the Qwen2.5-7B F16 evaluation (which showed `PARTICULAR:` annotation hallucinations on turn-2), this model produces clean JSON state at both turns with no appended text: + +| Case | Turn 1 | Turn 2 | +|---|---|---| +| state\_01 | `{"counter": 2}` | `{"counter": 5}` | +| state\_02 | `{"items": ["a", "b"]}` | `{"items": ["a", "b", "c"]}` | + +The F16 Transformers runner handles this model's stateful outputs cleanly. + +--- + +## F16 vs. Q4\_K\_M β€” Deployment Decision + +| Dimension | F16 (this repo) | Q4\_K\_M | +|---|---|---| +| VRAM (8K context) | ~32 GB | ~12 GB | +| VRAM (128K context) | ~48 GB | ~20 GB | +| Avg inference time | 56.623 sec/case | 2.683 sec/case | +| Speed ratio | 1.0Γ— (baseline) | **21.1Γ— faster** | +| All family pass rates | Same | Same | +| Toolcall final answer | Clean (bucket=11) | Clean (bucket=11) | +| MCQ | 5/5 | 5/5 | +| Behavioral difference | None | None | + +**For every practical deployment scenario, Q4\_K\_M is the correct choice.** It achieves the same results in 21Γ— less time at ~3Γ— less VRAM. F16 is appropriate only when: (1) you have 32+ GB GPU VRAM available, (2) you require full-weight provenance for compliance or reproducibility auditing, or (3) you need the F16 baseline cache for a subsequent comparison evaluation run. + +--- + +## Hardware Requirements + +| Configuration | VRAM Required | Notes | +|---|---|---| +| F16 (this repo) Β· 8K context | ~32 GB | 29.5 GB model + KV cache | +| F16 Β· 32K context | ~36 GB | Minimum A100 40 GB | +| F16 Β· 128K context | ~48 GB | A100 80 GB or multi-GPU | +| Q4\_K\_M (companion repo) Β· 8K | ~12 GB | 8.99 GB model + KV cache | +| Q4\_K\_M Β· 128K context | ~20 GB | A10G 24 GB Β· RTX 4090 | + +--- + +## Usage + +### Installation + +```bash +pip install llama-cpp-python huggingface_hub +``` + +For GPU acceleration (CUDA): + +```bash +CMAKE_ARGS="-DGGML_CUDA=on" pip install llama-cpp-python --force-reinstall --no-cache-dir +``` + +### Python β€” llama-cpp-python + +```python +from huggingface_hub import hf_hub_download +from llama_cpp import Llama + +# Note: 29.5 GB download β€” ensure sufficient disk space and ~32 GB VRAM +model_path = hf_hub_download( + repo_id="pbhappliedsystems/qwen-2.5-14B-instruct-1m-gguf-F16", + filename="qwen-2.5-14B-instruct-1m-gguf-F16.gguf" +) + +llm = Llama( + model_path=model_path, + n_ctx=32768, # Set to actual working context; supports up to 1M + n_gpu_layers=-1, + verbose=False, +) + +response = llm.create_chat_completion( + messages=[ + { + "role": "system", + "content": "You are a precise assistant. Follow instructions exactly." + }, + { + "role": "user", + "content": "Analyze the following and return a JSON object with keys: summary, risk_level, action_items." + } + ], + temperature=0.7, + max_tokens=1024, +) +print(response["choices"][0]["message"]["content"]) +``` + +For tool-calling (no EOS stripping required β€” output is clean at F16): + +```python +# quant_eval v7.21: toolcall bucket=11 β€” clean final answer, no post-processing needed +response = llm.create_chat_completion( + messages=[ + { + "role": "system", + "content": ( + "You are a tool-calling assistant. Output the tool call as JSON, " + "then on the next line output only the numeric result.\n" + 'Tool call format: {"tool_name": "", "args": {"a": , "b": }}' + ) + }, + {"role": "user", "content": "Use the add tool to compute 10 minus 4."} + ], + temperature=0.7, + max_tokens=128, +) +# No stripping required at F16 β€” output is clean +print(response["choices"][0]["message"]["content"]) +``` + +For bare tool-call dispatch with schema enforcement: + +```python +import json, re + +def call_tool_bare(llm, prompt: str, retries: int = 3) -> dict: + """ + Explicit schema enforcement for toolcall_only. + quant_eval v7.21: F16 uses 'left'/'right' keys without schema guidance. + System prompt specifying exact keys resolves the vocabulary mismatch. + """ + for attempt in range(retries): + response = llm.create_chat_completion( + messages=[ + { + "role": "system", + "content": ( + 'Respond ONLY with a JSON object using EXACTLY these keys:\n' + '{"tool_name": "add", "args": {"a": , "b": }}\n' + 'No other text, no markdown.' + ) + }, + {"role": "user", "content": prompt} + ], + temperature=0.0, + max_tokens=64, + ) + raw = response["choices"][0]["message"]["content"].strip() + try: + parsed = json.loads(raw) + assert "tool_name" in parsed and "args" in parsed + assert "a" in parsed["args"] and "b" in parsed["args"] + return parsed + except (json.JSONDecodeError, AssertionError, KeyError): + if attempt == retries - 1: + raise ValueError(f"Tool call failed after {retries} attempts. Raw: {raw}") +``` + +### CLI β€” llama-cli + +```bash +llama-cli \ + --model qwen-2.5-14B-instruct-1m-gguf-F16.gguf \ + --chat-template qwen2 \ + --system-prompt "You are a precise assistant. Follow instructions exactly." \ + --prompt "Return a JSON object with keys: summary, risk_level, action_items." \ + --n-predict 1024 \ + --ctx-size 32768 \ + --n-gpu-layers -1 \ + --temp 0.7 +``` + +--- + +## Artifact Provenance + +| Artifact | Format | Size | SHA256 | +|---|---|---|---| +| `qwen-2.5-14B-instruct-1m-gguf-F16.gguf` | GGUF F16 | 29.5 GB | `de08ea9c41234ef83b7aacf07f9ebc3cbaa20ca8aeb5f6417758a8798660aaa9` | +| Q4\_K\_M *(companion repo)* | GGUF Q4\_K\_M | 8.99 GB | `5ad529ff2b1b192f31c8a638fe8756a0c628904e2ded797c11f9194216976973` | + +The F16 GGUF was converted from `Qwen/Qwen2.5-14B-Instruct-1M` using a custom-built llama.cpp conversion pipeline developed by PBH Applied Systems. + +**Two-pass architecture:** This F16 run (`20260210_215029`) operated in cache-generation mode (`skip_quant=true`). The resulting `full_weight_cache.json` was used as the reference baseline for the Q4\_K\_M comparison run (`20260210_235131`). Timing identity between this run and the F16 baseline entries in the comparison run confirms clean cache reuse and run integrity. + +--- + +## Evaluation Methodology + +**quant_eval v7.21** β€” proprietary behavioral evaluation harness, PBH Applied Systems. +**Fixture set:** `golden_oracle_fixtures_v7_21` (SHA256: `6d71a0b9147c079371b02a94f3c149eb78a6adc03dc16ff6833b964fbf4174f0`) +**Evaluation hardware:** NVIDIA RTX 4090 Β· **F16 evaluation date:** February 10, 2026 Β· **Seed:** 42 + +--- + +## πŸ”¬ About quant_eval & This Evaluation Series + +[**quant_eval**](https://pbhappliedsystems.com) is a proprietary behavioral evaluation harness developed by [PBH Applied Systems, LLC](https://pbhappliedsystems.com). It measures real agent-adjacent task performance across structured output, tool dispatch, multi-turn state retention, and multi-step planning β€” not perplexity or leaderboard proxies. Every model published under [`pbhappliedsystems`](https://huggingface.co/pbhappliedsystems) has been independently evaluated using quant_eval before being recommended for any production role. + +**See it in action:** [**Live AI Agent Demo β†’**](https://pbhappliedsystems.com/assistant.html) +The demo runs production-style agent workflows powered by open-weight models selected through the quant_eval evaluation pipeline. + +> **Need a deployment recommendation?** +> Not sure which quantization level is right for your hardware, latency target, or agent type? +> [**β†’ pbhappliedsystems.com**](https://pbhappliedsystems.com) + +--- + +*Evaluated and published by [PBH Applied Systems, LLC](https://pbhappliedsystems.com) Β· [patrick@pbhappliedsystems.com](mailto:patrick@pbhappliedsystems.com)* + +--- + +## About PBH Applied Systems + +[**PBH Applied Systems, LLC**](https://pbhappliedsystems.com) is an Oklahoma City–based applied machine learning and AI systems company specializing in production-grade model evaluation, quantization pipelines, agentic AI infrastructure, and scalable AI-driven application development. + +**Patrick Hill, M.S.** β€” Founder Β· Data Scientist Β· AI/ML Engineer Β· Author of **[Applied Machine Learning: Concepts, Tools, and Case Studies](https://a.co/d/05qat7Xz)** (required reading, UAT CSC 373) + +--- + +## πŸ“ž Work With PBH Applied Systems + +The F16 baseline for this model exists because proper quantization evaluation requires it β€” you cannot measure what Q4\_K\_M preserves or degrades without a verified full-precision reference. For this model, the answer is: zero degradation. That finding only becomes a deployable fact when both runs exist and can be compared against the same fixture set. + +πŸ‘‰ **[Book a Scoping Call](https://pbhappliedsystems.com)** Β· πŸ‘‰ **[Request an Evaluation Report](https://pbhappliedsystems.com)** β€” from $2,500 + +### Connect + +| | | +|---|---| +| 🌐 | [pbhappliedsystems.com](https://pbhappliedsystems.com) | +| πŸ“§ | [patrick@pbhappliedsystems.com](mailto:patrick@pbhappliedsystems.com) | +| πŸ’Ό | [LinkedIn](https://www.linkedin.com/company/pbh-applied-systems-llc) | +| ▢️ | [YouTube](https://www.youtube.com/@pbhappliedsystems) | +| πŸ“Έ | [Instagram](https://www.instagram.com/pbhappliedsystems) | +| πŸ‘ | [Facebook](https://www.facebook.com/pbhappliedsystems) | + +--- + +## License + +This GGUF repository inherits the license of the base model: +**Apache 2.0** β€” [`Qwen/Qwen2.5-14B-Instruct-1M`](https://huggingface.co/Qwen/Qwen2.5-14B-Instruct-1M) + +The quant_eval evaluation methodology, fixture set, and scoring framework are proprietary to PBH Applied Systems, LLC and are not included in this repository. + +--- + +*GGUF conversion and behavioral evaluation performed by [PBH Applied Systems, LLC](https://pbhappliedsystems.com) Β· quant_eval v7.21 Β· F16 Run ID: `20260210_215029`* diff --git a/comparison_results_v7_21_Qwen2.5_14B_Instruct_1M_20260210_215029.csv b/comparison_results_v7_21_Qwen2.5_14B_Instruct_1M_20260210_215029.csv new file mode 100644 index 0000000..48ef966 --- /dev/null +++ b/comparison_results_v7_21_Qwen2.5_14B_Instruct_1M_20260210_215029.csv @@ -0,0 +1,630 @@ +run_id,timestamp,runner,family,case_id,mode,secs,timeout,output_raw,error,plan_len,plan_exact_match,plan_prefix_match_len,plan_stop_idx_expected,plan_stop_idx_got,final_equiv_ok,final_constraints_ok,bucket_score,detail,expected_plan,got_plan,expected_final,got_final,answer_line_ok,args_ok,best_of_k_candidate_idx,best_of_k_k,best_of_k_runner,best_of_k_seed,best_of_k_selected_idx,best_of_k_selected_sim,best_of_k_shot1_tier1_pass,best_of_k_tier1_pass_at_k,best_of_k_used,checks_consistent_ok,checks_ok,choice_extracted,constraints_detail,detail_stage1,expected_answer_line,expected_args,expected_json,expected_turn1,expected_turn2,final_consistent_ok,final_match_reported,got_answer_line,got_args,got_json,got_turn1,got_turn2,json_exact_match,json_parse_ok,oracle_equiv_ok,oracle_trace,parse_ok,plan_len_ok,schema_errors,schema_ok,stage1_tool_parse_ok,stage1_tool_schema_ok,stop_semantics_ok,tool_exec_detail,tool_exec_ok,tool_name_ok,tool_output_json,tool_payload_used,turn1_exact_match,turn1_parse_ok,turn2_exact_match,turn2_parse_ok +20260210_215029,20260210_220525,full_weight_transformers,json,json_01,tool_decisions,376.99,0,"{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""C"" + } +}",,3,1,3,-1,-1,1,1,10,ok,"[""A"",""A"",""C""]","[""A"",""A"",""C""]","{""A"":[3,3,0],""B"":[2,0,0],""C"":[0,0,2],""new_shelf"":0,""violation"":1}","{""A"":[3,3,0],""B"":[2,0,0],""C"":[0,0,2],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[2, 2, 0] B=[2, 0, 0] C=[0, 0, 1] +Arrivals=['P', 'S', 'H'] +Arrival 1: P + Try A: can_place=1 state=[2, 2, 0] + Placed on A: new_state=[2, 3, 0] +Arrival 2: S + Try A: can_place=1 state=[2, 3, 0] + Placed on A: new_state=[3, 3, 0] +Arrival 3: H + Try A: can_place=0 state=[3, 3, 0] + Try B: can_place=0 state=[2, 0, 0] + Try C: can_place=1 state=[0, 0, 1] + Placed on C: new_state=[0, 0, 2] +Final A=[3, 3, 0] B=[2, 0, 0] C=[0, 0, 2] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_220525,full_weight_transformers,json,json_02,tool_decisions,63.22,0,"{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +}",,3,1,3,-1,-1,1,1,10,ok,"[""A"",""A"",""A""]","[""A"",""A"",""A""]","{""A"":[3,1,0],""B"":[0,1,0],""C"":[0,0,1],""new_shelf"":0,""violation"":1}","{""A"":[3,1,0],""B"":[0,1,0],""C"":[0,0,1],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[1, 0, 0] B=[0, 1, 0] C=[0, 0, 1] +Arrivals=['S', 'S', 'P'] +Arrival 1: S + Try A: can_place=1 state=[1, 0, 0] + Placed on A: new_state=[2, 0, 0] +Arrival 2: S + Try A: can_place=1 state=[2, 0, 0] + Placed on A: new_state=[3, 0, 0] +Arrival 3: P + Try A: can_place=1 state=[3, 0, 0] + Placed on A: new_state=[3, 1, 0] +Final A=[3, 1, 0] B=[0, 1, 0] C=[0, 0, 1] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_220525,full_weight_transformers,json,json_03,tool_decisions,64.75,0,"{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""C"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""C"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +}",,3,1,3,-1,-1,1,1,10,ok,"[""C"",""C"",""A""]","[""C"",""C"",""A""]","{""A"":[3,1,0],""B"":[0,3,0],""C"":[0,0,4],""new_shelf"":0,""violation"":1}","{""A"":[3,1,0],""B"":[0,3,0],""C"":[0,0,4],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[3, 0, 0] B=[0, 3, 0] C=[0, 0, 2] +Arrivals=['H', 'H', 'P'] +Arrival 1: H + Try A: can_place=0 state=[3, 0, 0] + Try B: can_place=0 state=[0, 3, 0] + Try C: can_place=1 state=[0, 0, 2] + Placed on C: new_state=[0, 0, 3] +Arrival 2: H + Try A: can_place=0 state=[3, 0, 0] + Try B: can_place=0 state=[0, 3, 0] + Try C: can_place=1 state=[0, 0, 3] + Placed on C: new_state=[0, 0, 4] +Arrival 3: P + Try A: can_place=1 state=[3, 0, 0] + Placed on A: new_state=[3, 1, 0] +Final A=[3, 1, 0] B=[0, 3, 0] C=[0, 0, 4] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_220525,full_weight_transformers,json,json_04,tool_decisions,70.51,0,"{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""A"" + } +} + +{ + ""tool_name"": ""place_item"", + ""args"": { + ""choice"": ""B"" + } +}",,3,1,3,-1,-1,1,1,10,ok,"[""A"",""A"",""B""]","[""A"",""A"",""B""]","{""A"":[1,1,0],""B"":[0,0,1],""C"":[0,0,0],""new_shelf"":0,""violation"":1}","{""A"":[1,1,0],""B"":[0,0,1],""C"":[0,0,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 0] B=[0, 0, 0] C=[0, 0, 0] +Arrivals=['S', 'P', 'H'] +Arrival 1: S + Try A: can_place=1 state=[0, 0, 0] + Placed on A: new_state=[1, 0, 0] +Arrival 2: P + Try A: can_place=1 state=[1, 0, 0] + Placed on A: new_state=[1, 1, 0] +Arrival 3: H + Try A: can_place=0 state=[1, 1, 0] + Try B: can_place=1 state=[0, 0, 0] + Placed on B: new_state=[0, 0, 1] +Final A=[1, 1, 0] B=[0, 0, 1] C=[0, 0, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_223824,full_weight_transformers,json_multistep,ms_easy_01,best_of_k,108.92995429039001,0,"{ + ""plan"": [""A""], + ""checks"": [ + { + ""step"": 1, + ""item"": ""P"", + ""choice"": ""A"", + ""can_place"": 1 + } + ], + ""final"": { + ""A"": [2, 2, 1], + ""B"": [3, 0, 0], + ""C"": [0, 0, 1], + ""new_shelf"": 0, + ""violation"": 0 + }, + ""stop"": false, + ""stop_reason"": """", + ""stop_step_index"": -1 +}",,1,1,1,-1,-1,1,1,1,ok,"[""A""]","[""A""]","{""A"":[2,3,0],""B"":[3,0,0],""C"":[0,0,1],""new_shelf"":0,""violation"":1}","{""A"":[2,3,0],""B"":[3,0,0],""C"":[0,0,1],""new_shelf"":0,""violation"":1}",,,0,3,full_weight_transformers,1337,0,0.6822282075881958,1,1,1,1,1,,,,,,,,,0,0,,,,,,,,1,"Initial A=[2, 2, 0] B=[3, 0, 0] C=[0, 0, 1] +Arrivals=['P'] +Arrival 1: P + Try A: can_place=1 state=[2, 2, 0] + Placed on A: new_state=[2, 3, 0] +Final A=[2, 3, 0] B=[3, 0, 0] C=[0, 0, 1] new_shelf=0 violation=1",1,1,,1,,,1,,,,,,,,, +20260210_215029,20260210_223824,full_weight_transformers,json_multistep,ms_easy_02,best_of_k,125.48065209388733,0,"{ + ""plan"": [""A"", ""B""], + ""checks"": [ + {""step"": 1, ""item"": ""S"", ""choice"": ""A"", ""can_place"": 1}, + {""step"": 2, ""item"": ""P"", ""choice"": ""B"", ""can_place"": 1} + ], + ""final"": { + ""A"": [3, 1, 0], + ""B"": [0, 1, 0], + ""C"": [0, 0, 0], + ""new_shelf"": 0, + ""violation"": 0 + }, + ""stop"": false, + ""stop_reason"": """", + ""stop_step_index"": -1 +}",,2,0,1,-1,-1,0,1,1,oracle_equiv_ok,"[""A"",""A""]","[""A"",""B""]","{""A"":[4,1,0],""B"":[0,0,0],""C"":[0,0,0],""new_shelf"":0,""violation"":1}","{""A"":[4,0,0],""B"":[0,1,0],""C"":[0,0,0],""new_shelf"":0,""violation"":1}",,,0,3,full_weight_transformers,1437,0,0.6678733825683594,0,0,1,1,1,,,,,,,,,0,0,,,,,,,,0,"Initial A=[3, 0, 0] B=[0, 0, 0] C=[0, 0, 0] +Arrivals=['S', 'P'] +Arrival 1: S + Try A: can_place=1 state=[3, 0, 0] + Placed on A: new_state=[4, 0, 0] +Arrival 2: P + Try A: can_place=1 state=[4, 0, 0] + Placed on A: new_state=[4, 1, 0] +Final A=[4, 1, 0] B=[0, 0, 0] C=[0, 0, 0] new_shelf=0 violation=1",1,1,,1,,,1,,,,,,,,, +20260210_215029,20260210_223824,full_weight_transformers,json_multistep,ms_med_01,best_of_k,149.3066966533661,0,"{ + ""plan"": [""A"", ""B"", ""C""], + ""checks"": [ + {""step"": 1, ""item"": ""S"", ""choice"": ""A"", ""can_place"": 1}, + {""step"": 2, ""item"": ""P"", ""choice"": ""B"", ""can_place"": 1}, + {""step"": 3, ""item"": ""H"", ""choice"": ""C"", ""can_place"": 1} + ], + ""final"": { + ""A"": [2, 2, 0], + ""B"": [0, 0, 0], + ""C"": [0, 0, 3], + ""new_shelf"": 0, + ""violation"": 0 + }, + ""stop"": false, + ""stop_reason"": """", + ""stop_step_index"": -1 +}",,3,1,3,-1,-1,1,1,3,ok,"[""A"",""B"",""C""]","[""A"",""B"",""C""]","{""A"":[3,3,0],""B"":[0,1,0],""C"":[0,0,3],""new_shelf"":0,""violation"":1}","{""A"":[3,3,0],""B"":[0,1,0],""C"":[0,0,3],""new_shelf"":0,""violation"":1}",,,0,3,full_weight_transformers,1537,0,0.6734610795974731,1,1,1,1,1,,,,,,,,,0,0,,,,,,,,1,"Initial A=[2, 3, 0] B=[0, 0, 0] C=[0, 0, 2] +Arrivals=['S', 'P', 'H'] +Arrival 1: S + Try A: can_place=1 state=[2, 3, 0] + Placed on A: new_state=[3, 3, 0] +Arrival 2: P + Try A: can_place=0 state=[3, 3, 0] + Try B: can_place=1 state=[0, 0, 0] + Placed on B: new_state=[0, 1, 0] +Arrival 3: H + Try A: can_place=0 state=[3, 3, 0] + Try B: can_place=0 state=[0, 1, 0] + Try C: can_place=1 state=[0, 0, 2] + Placed on C: new_state=[0, 0, 3] +Final A=[3, 3, 0] B=[0, 1, 0] C=[0, 0, 3] new_shelf=0 violation=1",1,1,,1,,,1,,,,,,,,, +20260210_215029,20260210_223824,full_weight_transformers,json_multistep,ms_med_02,best_of_k,144.79933547973633,0,"{ + ""plan"": [""A"", ""B"", ""C""], + ""checks"": [ + {""step"": 1, ""item"": ""P"", ""choice"": ""A"", ""can_place"": 1}, + {""step"": 2, ""item"": ""S"", ""choice"": ""B"", ""can_place"": 1}, + {""step"": 3, ""item"": ""H"", ""choice"": ""C"", ""can_place"": 1} + ], + ""final"": { + ""A"": [4, 0, 0], + ""B"": [0, 3, 0], + ""C"": [0, 0, 2], + ""new_shelf"": 0, + ""violation"": 0 + }, + ""stop"": false, + ""stop_reason"": """", + ""stop_step_index"": -1 +}",,3,1,3,-1,-1,1,1,3,ok,"[""A"",""B"",""C""]","[""A"",""B"",""C""]","{""A"":[5,1,0],""B"":[1,4,0],""C"":[0,0,2],""new_shelf"":0,""violation"":1}","{""A"":[5,1,0],""B"":[1,4,0],""C"":[0,0,2],""new_shelf"":0,""violation"":1}",,,0,3,full_weight_transformers,1637,0,0.672205925,1,1,1,1,1,,,,,,,,,0,0,,,,,,,,1,"Initial A=[5, 0, 0] B=[0, 4, 0] C=[0, 0, 1] +Arrivals=['P', 'S', 'H'] +Arrival 1: P + Try A: can_place=1 state=[5, 0, 0] + Placed on A: new_state=[5, 1, 0] +Arrival 2: S + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=1 state=[0, 4, 0] + Placed on B: new_state=[1, 4, 0] +Arrival 3: H + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=0 state=[1, 4, 0] + Try C: can_place=1 state=[0, 0, 1] + Placed on C: new_state=[0, 0, 2] +Final A=[5, 1, 0] B=[1, 4, 0] C=[0, 0, 2] new_shelf=0 violation=1",1,1,,1,,,1,,,,,,,,, +20260210_215029,20260210_223824,full_weight_transformers,json_multistep,ms_hard_01,best_of_k,137.51127171516418,0,"{ + ""plan"": [""A"", ""B"", ""C""], + ""checks"": [ + {""step"": 1, ""item"": ""S"", ""choice"": ""A"", ""can_place"": 1}, + {""step"": 2, ""item"": ""S"", ""choice"": ""B"", ""can_place"": 1}, + {""step"": 3, ""item"": ""H"", ""choice"": ""C"", ""can_place"": 1} + ], + ""final"": { + ""A"": [1, 5, 0], + ""B"": [2, 0, 0], + ""C"": [0, 0, 1], + ""new_shelf"": 0, + ""violation"": 0 + }, + ""stop"": false, + ""stop_reason"": """", + ""stop_step_index"": -1 +}",,3,1,3,-1,-1,1,1,3,ok,"[""A"",""B"",""C""]","[""A"",""B"",""C""]","{""A"":[1,5,0],""B"":[3,0,0],""C"":[0,0,1],""new_shelf"":0,""violation"":1}","{""A"":[1,5,0],""B"":[3,0,0],""C"":[0,0,1],""new_shelf"":0,""violation"":1}",,,0,3,full_weight_transformers,1737,0,0.6760021448135376,1,1,1,1,1,,,,,,,,,0,0,,,,,,,,1,"Initial A=[0, 5, 0] B=[2, 0, 0] C=[0, 0, 0] +Arrivals=['S', 'S', 'H'] +Arrival 1: S + Try A: can_place=1 state=[0, 5, 0] + Placed on A: new_state=[1, 5, 0] +Arrival 2: S + Try A: can_place=0 state=[1, 5, 0] + Try B: can_place=1 state=[2, 0, 0] + Placed on B: new_state=[3, 0, 0] +Arrival 3: H + Try A: can_place=0 state=[1, 5, 0] + Try B: can_place=0 state=[3, 0, 0] + Try C: can_place=1 state=[0, 0, 0] + Placed on C: new_state=[0, 0, 1] +Final A=[1, 5, 0] B=[3, 0, 0] C=[0, 0, 1] new_shelf=0 violation=1",1,1,,1,,,1,,,,,,,,, +20260210_215029,20260210_223852,full_weight_transformers,stateful_followup,state_01,2turn,10.44,0,"{""counter"": 2} + +{""counter"": 5}",,,,,,,,,2,ok,,,,,,,,,,,,,,,,,,,,,,,,"{""counter"":2}","{""counter"":5}",,,,,,"{""counter"":2}","{""counter"":5}",,,,,,,,,,,,,,,,,1,1,1,1 +20260210_215029,20260210_223852,full_weight_transformers,stateful_followup,state_02,2turn,17.06,0,"{""items"": [""a"", ""b""]} + +{""items"": [""a"", ""b"", ""c""]}",,,,,,,,,2,ok,,,,,,,,,,,,,,,,,,,,,,,,"{""items"":[""a"",""b""]}","{""items"":[""a"",""b"",""c""]}",,,,,,"{""items"":[""a"",""b""]}","{""items"":[""a"",""b"",""c""]}",,,,,,,,,,,,,,,,,1,1,1,1 +20260210_215029,20260210_223930,full_weight_transformers,mixed_brief_json,mixed_01,,20.12,0,"ANSWER: 13 +{""a"": 4, ""b"": 9, ""sum"": 13}",,,,,,,,,2,ok,,,,,1,,,,,,,,,,,,,,,,ANSWER: 13,,"{""a"":4,""b"":9,""sum"":13}",,,,,ANSWER: 13,,"{""a"":4,""b"":9,""sum"":13}",,,1,1,,,,,,1,,,,,,,,,,,, +20260210_215029,20260210_223930,full_weight_transformers,mixed_brief_json,mixed_02,,18.67,0,"ANSWER: 6 +{""a"": -2, ""b"": 8, ""sum"": 6}",,,,,,,,,2,ok,,,,,1,,,,,,,,,,,,,,,,ANSWER: 6,,"{""a"":-2,""b"":8,""sum"":6}",,,,,ANSWER: 6,,"{""a"":-2,""b"":8,""sum"":6}",,,1,1,,,,,,1,,,,,,,,,,,, +20260210_215029,20260210_224116,full_weight_transformers,fuzz,fuzz_0000,tool_decisions,105.31,0,,,5,1,5,-1,-1,1,1,10,ok,"[""A"",""B"",""B"",""B"",""B""]","[""A"",""B"",""B"",""B"",""B""]","{""A"":[0,6,0],""B"":[5,1,0],""C"":[5,0,0],""new_shelf"":0,""violation"":1}","{""A"":[0,6,0],""B"":[5,1,0],""C"":[5,0,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 5, 0] B=[1, 1, 0] C=[5, 0, 0] +Arrivals=['P', 'S', 'S', 'S', 'S'] +Arrival 1: P + Try A: can_place=1 state=[0, 5, 0] + Placed on A: new_state=[0, 6, 0] +Arrival 2: S + Try A: can_place=0 state=[0, 6, 0] + Try B: can_place=1 state=[1, 1, 0] + Placed on B: new_state=[2, 1, 0] +Arrival 3: S + Try A: can_place=0 state=[0, 6, 0] + Try B: can_place=1 state=[2, 1, 0] + Placed on B: new_state=[3, 1, 0] +Arrival 4: S + Try A: can_place=0 state=[0, 6, 0] + Try B: can_place=1 state=[3, 1, 0] + Placed on B: new_state=[4, 1, 0] +Arrival 5: S + Try A: can_place=0 state=[0, 6, 0] + Try B: can_place=1 state=[4, 1, 0] + Placed on B: new_state=[5, 1, 0] +Final A=[0, 6, 0] B=[5, 1, 0] C=[5, 0, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224137,full_weight_transformers,fuzz,fuzz_0001,tool_decisions,21.13,0,,,1,1,1,-1,-1,1,1,10,ok,"[""A""]","[""A""]","{""A"":[5,0,0],""B"":[5,1,0],""C"":[4,1,0],""new_shelf"":0,""violation"":1}","{""A"":[5,0,0],""B"":[5,1,0],""C"":[4,1,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[4, 0, 0] B=[5, 1, 0] C=[4, 1, 0] +Arrivals=['S'] +Arrival 1: S + Try A: can_place=1 state=[4, 0, 0] + Placed on A: new_state=[5, 0, 0] +Final A=[5, 0, 0] B=[5, 1, 0] C=[4, 1, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224156,full_weight_transformers,fuzz,fuzz_0002,tool_decisions,18.96,0,,,1,1,1,-1,-1,1,1,10,ok,"[""A""]","[""A""]","{""A"":[2,3,0],""B"":[0,0,6],""C"":[0,3,0],""new_shelf"":0,""violation"":1}","{""A"":[2,3,0],""B"":[0,0,6],""C"":[0,3,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[2, 2, 0] B=[0, 0, 6] C=[0, 3, 0] +Arrivals=['P'] +Arrival 1: P + Try A: can_place=1 state=[2, 2, 0] + Placed on A: new_state=[2, 3, 0] +Final A=[2, 3, 0] B=[0, 0, 6] C=[0, 3, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224239,full_weight_transformers,fuzz,fuzz_0003,tool_decisions,43.09,0,,,5,1,5,1,1,1,1,10,ok,"[""A"",""STOP"",""STOP"",""STOP"",""STOP""]","[""A"",""STOP"",""STOP"",""STOP"",""STOP""]","{""A"":[4,2,0],""B"":[5,1,0],""C"":[3,0,0],""new_shelf"":1,""violation"":1}","{""A"":[4,2,0],""B"":[5,1,0],""C"":[3,0,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[4, 1, 0] B=[5, 1, 0] C=[3, 0, 0] +Arrivals=['P', 'H', 'H', 'P', 'H'] +Arrival 1: P + Try A: can_place=1 state=[4, 1, 0] + Placed on A: new_state=[4, 2, 0] +Arrival 2: H + Try A: can_place=0 state=[4, 2, 0] + Try B: can_place=0 state=[5, 1, 0] + Try C: can_place=0 state=[3, 0, 0] + No placement possible on A/B/C: placements=STOP +Arrival 3: H + Already stopped: placements=STOP +Arrival 4: P + Already stopped: placements=STOP +Arrival 5: H + Already stopped: placements=STOP +Final A=[4, 2, 0] B=[5, 1, 0] C=[3, 0, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224322,full_weight_transformers,fuzz,fuzz_0004,tool_decisions,42.64,0,,,2,1,2,-1,-1,1,1,10,ok,"[""A"",""A""]","[""A"",""A""]","{""A"":[1,1,0],""B"":[0,0,1],""C"":[0,6,0],""new_shelf"":0,""violation"":1}","{""A"":[1,1,0],""B"":[0,0,1],""C"":[0,6,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 0] B=[0, 0, 1] C=[0, 6, 0] +Arrivals=['S', 'P'] +Arrival 1: S + Try A: can_place=1 state=[0, 0, 0] + Placed on A: new_state=[1, 0, 0] +Arrival 2: P + Try A: can_place=1 state=[1, 0, 0] + Placed on A: new_state=[1, 1, 0] +Final A=[1, 1, 0] B=[0, 0, 1] C=[0, 6, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224459,full_weight_transformers,fuzz,fuzz_0005,tool_decisions,96.96,0,,,5,1,5,-1,-1,1,1,10,ok,"[""B"",""C"",""B"",""B"",""C""]","[""B"",""C"",""B"",""B"",""C""]","{""A"":[5,1,0],""B"":[0,0,5],""C"":[4,0,0],""new_shelf"":0,""violation"":1}","{""A"":[5,1,0],""B"":[0,0,5],""C"":[4,0,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[5, 1, 0] B=[0, 0, 2] C=[2, 0, 0] +Arrivals=['H', 'S', 'H', 'H', 'S'] +Arrival 1: H + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=1 state=[0, 0, 2] + Placed on B: new_state=[0, 0, 3] +Arrival 2: S + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=0 state=[0, 0, 3] + Try C: can_place=1 state=[2, 0, 0] + Placed on C: new_state=[3, 0, 0] +Arrival 3: H + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=1 state=[0, 0, 3] + Placed on B: new_state=[0, 0, 4] +Arrival 4: H + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=1 state=[0, 0, 4] + Placed on B: new_state=[0, 0, 5] +Arrival 5: S + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=0 state=[0, 0, 5] + Try C: can_place=1 state=[3, 0, 0] + Placed on C: new_state=[4, 0, 0] +Final A=[5, 1, 0] B=[0, 0, 5] C=[4, 0, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224538,full_weight_transformers,fuzz,fuzz_0006,tool_decisions,39.27,0,,,2,1,2,1,1,1,1,10,ok,"[""B"",""STOP""]","[""B"",""STOP""]","{""A"":[0,0,3],""B"":[6,0,0],""C"":[6,0,0],""new_shelf"":1,""violation"":0}","{""A"":[0,0,3],""B"":[6,0,0],""C"":[6,0,0],""new_shelf"":1,""violation"":0}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 3] B=[5, 0, 0] C=[6, 0, 0] +Arrivals=['S', 'P'] +Arrival 1: S + Try A: can_place=0 state=[0, 0, 3] + Try B: can_place=1 state=[5, 0, 0] + Placed on B: new_state=[6, 0, 0] +Arrival 2: P + Try A: can_place=0 state=[0, 0, 3] + Try B: can_place=0 state=[6, 0, 0] + Try C: can_place=0 state=[6, 0, 0] + No placement possible on A/B/C: placements=STOP +Final A=[0, 0, 3] B=[6, 0, 0] C=[6, 0, 0] new_shelf=1 violation=0",,,,,,,,,,,,,,,, +20260210_215029,20260210_224557,full_weight_transformers,fuzz,fuzz_0007,tool_decisions,18.92,0,,,4,1,4,0,0,1,1,10,ok,"[""STOP"",""STOP"",""STOP"",""STOP""]","[""STOP"",""STOP"",""STOP"",""STOP""]","{""A"":[0,1,0],""B"":[4,2,0],""C"":[5,1,0],""new_shelf"":1,""violation"":1}","{""A"":[0,1,0],""B"":[4,2,0],""C"":[5,1,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 1, 0] B=[4, 2, 0] C=[5, 1, 0] +Arrivals=['H', 'P', 'S', 'P'] +Arrival 1: H + Try A: can_place=0 state=[0, 1, 0] + Try B: can_place=0 state=[4, 2, 0] + Try C: can_place=0 state=[5, 1, 0] + No placement possible on A/B/C: placements=STOP +Arrival 2: P + Already stopped: placements=STOP +Arrival 3: S + Already stopped: placements=STOP +Arrival 4: P + Already stopped: placements=STOP +Final A=[0, 1, 0] B=[4, 2, 0] C=[5, 1, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224641,full_weight_transformers,fuzz,fuzz_0008,tool_decisions,44.42,0,,,3,1,3,1,1,1,1,10,ok,"[""C"",""STOP"",""STOP""]","[""C"",""STOP"",""STOP""]","{""A"":[0,0,5],""B"":[2,4,0],""C"":[5,1,0],""new_shelf"":1,""violation"":0}","{""A"":[0,0,5],""B"":[2,4,0],""C"":[5,1,0],""new_shelf"":1,""violation"":0}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 5] B=[2, 4, 0] C=[4, 1, 0] +Arrivals=['S', 'S', 'H'] +Arrival 1: S + Try A: can_place=0 state=[0, 0, 5] + Try B: can_place=0 state=[2, 4, 0] + Try C: can_place=1 state=[4, 1, 0] + Placed on C: new_state=[5, 1, 0] +Arrival 2: S + Try A: can_place=0 state=[0, 0, 5] + Try B: can_place=0 state=[2, 4, 0] + Try C: can_place=0 state=[5, 1, 0] + No placement possible on A/B/C: placements=STOP +Arrival 3: H + Already stopped: placements=STOP +Final A=[0, 0, 5] B=[2, 4, 0] C=[5, 1, 0] new_shelf=1 violation=0",,,,,,,,,,,,,,,, +20260210_215029,20260210_224725,full_weight_transformers,fuzz,fuzz_0009,tool_decisions,44.1,0,,,4,1,4,1,1,1,1,10,ok,"[""C"",""STOP"",""STOP"",""STOP""]","[""C"",""STOP"",""STOP"",""STOP""]","{""A"":[6,0,0],""B"":[1,5,0],""C"":[0,0,6],""new_shelf"":1,""violation"":1}","{""A"":[6,0,0],""B"":[1,5,0],""C"":[0,0,6],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[6, 0, 0] B=[1, 5, 0] C=[0, 0, 5] +Arrivals=['H', 'S', 'P', 'P'] +Arrival 1: H + Try A: can_place=0 state=[6, 0, 0] + Try B: can_place=0 state=[1, 5, 0] + Try C: can_place=1 state=[0, 0, 5] + Placed on C: new_state=[0, 0, 6] +Arrival 2: S + Try A: can_place=0 state=[6, 0, 0] + Try B: can_place=0 state=[1, 5, 0] + Try C: can_place=0 state=[0, 0, 6] + No placement possible on A/B/C: placements=STOP +Arrival 3: P + Already stopped: placements=STOP +Arrival 4: P + Already stopped: placements=STOP +Final A=[6, 0, 0] B=[1, 5, 0] C=[0, 0, 6] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224805,full_weight_transformers,fuzz,fuzz_0010,tool_decisions,39.67,0,,,5,1,5,1,1,1,1,10,ok,"[""A"",""STOP"",""STOP"",""STOP"",""STOP""]","[""A"",""STOP"",""STOP"",""STOP"",""STOP""]","{""A"":[3,3,0],""B"":[6,0,0],""C"":[0,5,0],""new_shelf"":1,""violation"":1}","{""A"":[3,3,0],""B"":[6,0,0],""C"":[0,5,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[3, 2, 0] B=[6, 0, 0] C=[0, 5, 0] +Arrivals=['P', 'H', 'P', 'S', 'P'] +Arrival 1: P + Try A: can_place=1 state=[3, 2, 0] + Placed on A: new_state=[3, 3, 0] +Arrival 2: H + Try A: can_place=0 state=[3, 3, 0] + Try B: can_place=0 state=[6, 0, 0] + Try C: can_place=0 state=[0, 5, 0] + No placement possible on A/B/C: placements=STOP +Arrival 3: P + Already stopped: placements=STOP +Arrival 4: S + Already stopped: placements=STOP +Arrival 5: P + Already stopped: placements=STOP +Final A=[3, 3, 0] B=[6, 0, 0] C=[0, 5, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_224847,full_weight_transformers,fuzz,fuzz_0011,tool_decisions,41.56,0,,,5,1,5,1,1,1,1,10,ok,"[""A"",""STOP"",""STOP"",""STOP"",""STOP""]","[""A"",""STOP"",""STOP"",""STOP"",""STOP""]","{""A"":[4,0,0],""B"":[5,1,0],""C"":[6,0,0],""new_shelf"":1,""violation"":1}","{""A"":[4,0,0],""B"":[5,1,0],""C"":[6,0,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[3, 0, 0] B=[5, 1, 0] C=[6, 0, 0] +Arrivals=['S', 'H', 'P', 'H', 'H'] +Arrival 1: S + Try A: can_place=1 state=[3, 0, 0] + Placed on A: new_state=[4, 0, 0] +Arrival 2: H + Try A: can_place=0 state=[4, 0, 0] + Try B: can_place=0 state=[5, 1, 0] + Try C: can_place=0 state=[6, 0, 0] + No placement possible on A/B/C: placements=STOP +Arrival 3: P + Already stopped: placements=STOP +Arrival 4: H + Already stopped: placements=STOP +Arrival 5: H + Already stopped: placements=STOP +Final A=[4, 0, 0] B=[5, 1, 0] C=[6, 0, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225007,full_weight_transformers,fuzz,fuzz_0012,tool_decisions,80.1,0,,,4,1,4,-1,-1,1,1,10,ok,"[""A"",""A"",""A"",""C""]","[""A"",""A"",""A"",""C""]","{""A"":[3,3,0],""B"":[4,2,0],""C"":[4,2,0],""new_shelf"":0,""violation"":1}","{""A"":[3,3,0],""B"":[4,2,0],""C"":[4,2,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[1, 2, 0] B=[4, 2, 0] C=[4, 1, 0] +Arrivals=['S', 'S', 'P', 'P'] +Arrival 1: S + Try A: can_place=1 state=[1, 2, 0] + Placed on A: new_state=[2, 2, 0] +Arrival 2: S + Try A: can_place=1 state=[2, 2, 0] + Placed on A: new_state=[3, 2, 0] +Arrival 3: P + Try A: can_place=1 state=[3, 2, 0] + Placed on A: new_state=[3, 3, 0] +Arrival 4: P + Try A: can_place=0 state=[3, 3, 0] + Try B: can_place=0 state=[4, 2, 0] + Try C: can_place=1 state=[4, 1, 0] + Placed on C: new_state=[4, 2, 0] +Final A=[3, 3, 0] B=[4, 2, 0] C=[4, 2, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225105,full_weight_transformers,fuzz,fuzz_0013,tool_decisions,58.21,0,,,5,1,5,2,2,1,1,10,ok,"[""A"",""B"",""STOP"",""STOP"",""STOP""]","[""A"",""B"",""STOP"",""STOP"",""STOP""]","{""A"":[2,4,0],""B"":[1,5,0],""C"":[0,6,0],""new_shelf"":1,""violation"":1}","{""A"":[2,4,0],""B"":[1,5,0],""C"":[0,6,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[1, 4, 0] B=[0, 5, 0] C=[0, 6, 0] +Arrivals=['S', 'S', 'H', 'P', 'H'] +Arrival 1: S + Try A: can_place=1 state=[1, 4, 0] + Placed on A: new_state=[2, 4, 0] +Arrival 2: S + Try A: can_place=0 state=[2, 4, 0] + Try B: can_place=1 state=[0, 5, 0] + Placed on B: new_state=[1, 5, 0] +Arrival 3: H + Try A: can_place=0 state=[2, 4, 0] + Try B: can_place=0 state=[1, 5, 0] + Try C: can_place=0 state=[0, 6, 0] + No placement possible on A/B/C: placements=STOP +Arrival 4: P + Already stopped: placements=STOP +Arrival 5: H + Already stopped: placements=STOP +Final A=[2, 4, 0] B=[1, 5, 0] C=[0, 6, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225247,full_weight_transformers,fuzz,fuzz_0014,tool_decisions,102.3,0,,,6,1,6,4,4,1,1,10,ok,"[""B"",""B"",""A"",""A"",""STOP"",""STOP""]","[""B"",""B"",""A"",""A"",""STOP"",""STOP""]","{""A"":[0,0,6],""B"":[3,3,0],""C"":[6,0,0],""new_shelf"":1,""violation"":1}","{""A"":[0,0,6],""B"":[3,3,0],""C"":[6,0,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 4] B=[3, 1, 0] C=[6, 0, 0] +Arrivals=['P', 'P', 'H', 'H', 'P', 'P'] +Arrival 1: P + Try A: can_place=0 state=[0, 0, 4] + Try B: can_place=1 state=[3, 1, 0] + Placed on B: new_state=[3, 2, 0] +Arrival 2: P + Try A: can_place=0 state=[0, 0, 4] + Try B: can_place=1 state=[3, 2, 0] + Placed on B: new_state=[3, 3, 0] +Arrival 3: H + Try A: can_place=1 state=[0, 0, 4] + Placed on A: new_state=[0, 0, 5] +Arrival 4: H + Try A: can_place=1 state=[0, 0, 5] + Placed on A: new_state=[0, 0, 6] +Arrival 5: P + Try A: can_place=0 state=[0, 0, 6] + Try B: can_place=0 state=[3, 3, 0] + Try C: can_place=0 state=[6, 0, 0] + No placement possible on A/B/C: placements=STOP +Arrival 6: P + Already stopped: placements=STOP +Final A=[0, 0, 6] B=[3, 3, 0] C=[6, 0, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225331,full_weight_transformers,fuzz,fuzz_0015,tool_decisions,44.1,0,,,2,1,2,-1,-1,1,1,10,ok,"[""C"",""A""]","[""C"",""A""]","{""A"":[4,0,0],""B"":[0,2,0],""C"":[0,0,5],""new_shelf"":0,""violation"":1}","{""A"":[4,0,0],""B"":[0,2,0],""C"":[0,0,5],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[3, 0, 0] B=[0, 2, 0] C=[0, 0, 4] +Arrivals=['H', 'S'] +Arrival 1: H + Try A: can_place=0 state=[3, 0, 0] + Try B: can_place=0 state=[0, 2, 0] + Try C: can_place=1 state=[0, 0, 4] + Placed on C: new_state=[0, 0, 5] +Arrival 2: S + Try A: can_place=1 state=[3, 0, 0] + Placed on A: new_state=[4, 0, 0] +Final A=[4, 0, 0] B=[0, 2, 0] C=[0, 0, 5] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225350,full_weight_transformers,fuzz,fuzz_0016,tool_decisions,19.02,0,,,1,1,1,-1,-1,1,1,10,ok,"[""A""]","[""A""]","{""A"":[0,0,6],""B"":[1,0,0],""C"":[6,0,0],""new_shelf"":0,""violation"":1}","{""A"":[0,0,6],""B"":[1,0,0],""C"":[6,0,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 5] B=[1, 0, 0] C=[6, 0, 0] +Arrivals=['H'] +Arrival 1: H + Try A: can_place=1 state=[0, 0, 5] + Placed on A: new_state=[0, 0, 6] +Final A=[0, 0, 6] B=[1, 0, 0] C=[6, 0, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225411,full_weight_transformers,fuzz,fuzz_0017,tool_decisions,20.66,0,,,4,1,4,0,0,1,1,10,ok,"[""STOP"",""STOP"",""STOP"",""STOP""]","[""STOP"",""STOP"",""STOP"",""STOP""]","{""A"":[5,1,0],""B"":[1,5,0],""C"":[4,2,0],""new_shelf"":1,""violation"":1}","{""A"":[5,1,0],""B"":[1,5,0],""C"":[4,2,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[5, 1, 0] B=[1, 5, 0] C=[4, 2, 0] +Arrivals=['S', 'P', 'P', 'S'] +Arrival 1: S + Try A: can_place=0 state=[5, 1, 0] + Try B: can_place=0 state=[1, 5, 0] + Try C: can_place=0 state=[4, 2, 0] + No placement possible on A/B/C: placements=STOP +Arrival 2: P + Already stopped: placements=STOP +Arrival 3: P + Already stopped: placements=STOP +Arrival 4: S + Already stopped: placements=STOP +Final A=[5, 1, 0] B=[1, 5, 0] C=[4, 2, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225454,full_weight_transformers,fuzz,fuzz_0018,tool_decisions,43.09,0,,,6,1,6,1,1,1,1,10,ok,"[""A"",""STOP"",""STOP"",""STOP"",""STOP"",""STOP""]","[""A"",""STOP"",""STOP"",""STOP"",""STOP"",""STOP""]","{""A"":[0,0,6],""B"":[3,3,0],""C"":[5,0,0],""new_shelf"":1,""violation"":1}","{""A"":[0,0,6],""B"":[3,3,0],""C"":[5,0,0],""new_shelf"":1,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[0, 0, 5] B=[3, 3, 0] C=[5, 0, 0] +Arrivals=['H', 'H', 'S', 'S', 'P', 'H'] +Arrival 1: H + Try A: can_place=1 state=[0, 0, 5] + Placed on A: new_state=[0, 0, 6] +Arrival 2: H + Try A: can_place=0 state=[0, 0, 6] + Try B: can_place=0 state=[3, 3, 0] + Try C: can_place=0 state=[5, 0, 0] + No placement possible on A/B/C: placements=STOP +Arrival 3: S + Already stopped: placements=STOP +Arrival 4: S + Already stopped: placements=STOP +Arrival 5: P + Already stopped: placements=STOP +Arrival 6: H + Already stopped: placements=STOP +Final A=[0, 0, 6] B=[3, 3, 0] C=[5, 0, 0] new_shelf=1 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225555,full_weight_transformers,fuzz,fuzz_0019,tool_decisions,60.58,0,,,3,1,3,-1,-1,1,1,10,ok,"[""B"",""C"",""C""]","[""B"",""C"",""C""]","{""A"":[6,0,0],""B"":[1,5,0],""C"":[5,1,0],""new_shelf"":0,""violation"":1}","{""A"":[6,0,0],""B"":[1,5,0],""C"":[5,1,0],""new_shelf"":0,""violation"":1}",,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,"Initial A=[6, 0, 0] B=[1, 4, 0] C=[3, 1, 0] +Arrivals=['P', 'S', 'S'] +Arrival 1: P + Try A: can_place=0 state=[6, 0, 0] + Try B: can_place=1 state=[1, 4, 0] + Placed on B: new_state=[1, 5, 0] +Arrival 2: S + Try A: can_place=0 state=[6, 0, 0] + Try B: can_place=0 state=[1, 5, 0] + Try C: can_place=1 state=[3, 1, 0] + Placed on C: new_state=[4, 1, 0] +Arrival 3: S + Try A: can_place=0 state=[6, 0, 0] + Try B: can_place=0 state=[1, 5, 0] + Try C: can_place=1 state=[4, 1, 0] + Placed on C: new_state=[5, 1, 0] +Final A=[6, 0, 0] B=[1, 5, 0] C=[5, 1, 0] new_shelf=0 violation=1",,,,,,,,,,,,,,,, +20260210_215029,20260210_225559,full_weight_transformers,mcq,mcq_01,,0.73,0,B,,,1,,,,1,,1,ok,B,B,,,,,,,,,,,,,,,,B,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, +20260210_215029,20260210_225559,full_weight_transformers,mcq,mcq_02,,0.73,0,B,,,1,,,,1,,1,ok,B,B,,,,,,,,,,,,,,,,B,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, +20260210_215029,20260210_225559,full_weight_transformers,mcq,mcq_03,,0.73,0,C,,,1,,,,1,,1,ok,C,C,,,,,,,,,,,,,,,,C,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, +20260210_215029,20260210_225559,full_weight_transformers,mcq,mcq_04,,0.74,0,B,,,1,,,,1,,1,ok,B,B,,,,,,,,,,,,,,,,B,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, +20260210_215029,20260210_225559,full_weight_transformers,mcq,mcq_05,,0.73,0,B,,,1,,,,1,,1,ok,B,B,,,,,,,,,,,,,,,,B,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, +20260210_215029,20260210_225651,full_weight_transformers,toolcall,tool_01,,26.1,0,"{ + ""tool_name"": ""add"", + ""args"": { + ""a"": 2, + ""b"": 3 + } +} + +5",,,,,,,1,,11,ok,5,5,,,,,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,,,,,1,1,,ok,1,,"{""tool_name"": ""add"", ""args"": {""a"": 2, ""b"": 3}}","{ + ""tool_name"": ""add"", + ""args"": { + ""a"": 2, + ""b"": 3 + } +}",,,, +20260210_215029,20260210_225651,full_weight_transformers,toolcall,tool_02,,25.8,0,"{ + ""tool_name"": ""add"", + ""args"": { + ""a"": 10, + ""b"": -4 + } +} + +6",,,,,,,1,,11,ok,6,6,,,,,,,,,,,,,,,,,,ok,,,,,,,,,,,,,,,,,,,,,1,1,,ok,1,,"{""tool_name"": ""add"", ""args"": {""a"": 10, ""b"": -4}}","{ + ""tool_name"": ""add"", + ""args"": { + ""a"": 10, + ""b"": -4 + } +}",,,, +20260210_215029,20260210_225721,full_weight_transformers,toolcall_only,toolonly_01,,14.95,0,"{""tool"": ""add"", ""left"": 5, ""right"": 10}",,,,,,,,,1,schema_error,,,,,,0,,,,,,,,,,,,,,,,"{""a"":5,""b"":10}",,,,,,,{},,,,,,,,1,,"'args' is a required property; Additional properties are not allowed ('left', 'right' were unexpected)",0,,,,,,1,,,,,, +20260210_215029,20260210_225721,full_weight_transformers,toolcall_only,toolonly_02,,15.76,0,"{""tool"": ""add"", ""left"": 25, ""right"": 75}",,,,,,,,,1,schema_error,,,,,,0,,,,,,,,,,,,,,,,"{""a"":25,""b"":75}",,,,,,,{},,,,,,,,1,,"'args' is a required property; Additional properties are not allowed ('left', 'right' were unexpected)",0,,,,,,1,,,,,, diff --git a/qwen-2.5-14B-instruct-1m-gguf-F16.gguf b/qwen-2.5-14B-instruct-1m-gguf-F16.gguf new file mode 100644 index 0000000..ca1345e --- /dev/null +++ b/qwen-2.5-14B-instruct-1m-gguf-F16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de08ea9c41234ef83b7aacf07f9ebc3cbaa20ca8aeb5f6417758a8798660aaa9 +size 29547716384 diff --git a/rollup.json b/rollup.json new file mode 100644 index 0000000..2f293b6 --- /dev/null +++ b/rollup.json @@ -0,0 +1,267 @@ +{ + "by_runner_family": { + "full_weight_transformers": { + "json_multistep": { + "n": 5, + "pass_rate": 0.8, + "pass_cols": [ + "schema_ok", + "checks_consistent_ok", + "stop_semantics_ok", + "oracle_equiv_ok" + ], + "averages": { + "bucket_score": 2.2 + }, + "top_failures": [ + { + "reason": "oracle_equiv_ok", + "count": 1 + } + ], + "signal_rates": { + "schema_ok": 1.0, + "checks_consistent_ok": 1.0, + "stop_semantics_ok": 1.0, + "oracle_equiv_ok": 0.8 + }, + "signal_counts": { + "schema_ok": { + "present": 5, + "ok": 5, + "fail": 0 + }, + "checks_consistent_ok": { + "present": 5, + "ok": 5, + "fail": 0 + }, + "stop_semantics_ok": { + "present": 5, + "ok": 5, + "fail": 0 + }, + "oracle_equiv_ok": { + "present": 5, + "ok": 4, + "fail": 1 + } + }, + "tier2_rates": { + "final_consistent_ok": 0.0, + "final_match_reported": 0.0 + }, + "tier2_counts": { + "final_consistent_ok": { + "present": 5, + "ok": 0, + "fail": 5 + }, + "final_match_reported": { + "present": 5, + "ok": 0, + "fail": 5 + } + } + }, + "stateful_followup": { + "n": 2, + "pass_rate": 1.0, + "pass_cols": [ + "turn1_parse_ok", + "turn2_parse_ok", + "turn1_exact_match", + "turn2_exact_match" + ], + "averages": { + "bucket_score": 2.0 + }, + "top_failures": [], + "signal_rates": { + "turn1_parse_ok": 1.0, + "turn2_parse_ok": 1.0, + "turn1_exact_match": 1.0, + "turn2_exact_match": 1.0 + }, + "signal_counts": { + "turn1_parse_ok": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "turn2_parse_ok": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "turn1_exact_match": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "turn2_exact_match": { + "present": 2, + "ok": 2, + "fail": 0 + } + }, + "tier2_rates": {}, + "tier2_counts": {} + }, + "toolcall_only": { + "n": 2, + "pass_rate": 0.0, + "pass_cols": [ + "tool_name_ok", + "args_ok" + ], + "averages": { + "bucket_score": 1.0 + }, + "top_failures": [ + { + "reason": "args_ok", + "count": 2 + } + ], + "signal_rates": { + "tool_name_ok": 1.0, + "args_ok": 0.0 + }, + "signal_counts": { + "tool_name_ok": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "args_ok": { + "present": 2, + "ok": 0, + "fail": 2 + } + }, + "tier2_rates": {}, + "tier2_counts": {} + }, + "mixed_brief_json": { + "n": 2, + "pass_rate": 1.0, + "pass_cols": [ + "answer_line_ok", + "json_parse_ok", + "schema_ok" + ], + "averages": { + "bucket_score": 2.0 + }, + "top_failures": [], + "signal_rates": { + "answer_line_ok": 1.0, + "json_parse_ok": 1.0, + "schema_ok": 1.0 + }, + "signal_counts": { + "answer_line_ok": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "json_parse_ok": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "schema_ok": { + "present": 2, + "ok": 2, + "fail": 0 + } + }, + "tier2_rates": {}, + "tier2_counts": {} + }, + "json": { + "n": 4, + "pass_rate": null, + "pass_cols": [ + "schema_ok", + "constraints_ok" + ], + "averages": { + "bucket_score": 10.0 + }, + "top_failures": [], + "signal_rates": {}, + "signal_counts": {}, + "tier2_rates": {}, + "tier2_counts": {} + }, + "fuzz": { + "n": 20, + "pass_rate": null, + "pass_cols": [ + "schema_ok", + "constraints_ok" + ], + "averages": { + "bucket_score": 10.0 + }, + "top_failures": [], + "signal_rates": {}, + "signal_counts": {}, + "tier2_rates": {}, + "tier2_counts": {} + }, + "toolcall": { + "n": 2, + "pass_rate": 1.0, + "pass_cols": [ + "stage1_tool_parse_ok", + "stage1_tool_schema_ok" + ], + "averages": { + "bucket_score": 11.0 + }, + "top_failures": [], + "signal_rates": { + "stage1_tool_parse_ok": 1.0, + "stage1_tool_schema_ok": 1.0 + }, + "signal_counts": { + "stage1_tool_parse_ok": { + "present": 2, + "ok": 2, + "fail": 0 + }, + "stage1_tool_schema_ok": { + "present": 2, + "ok": 2, + "fail": 0 + } + }, + "tier2_rates": {}, + "tier2_counts": {} + }, + "mcq": { + "n": 5, + "pass_rate": null, + "pass_cols": [ + "choice_ok" + ], + "averages": { + "bucket_score": 1.0 + }, + "top_failures": [], + "signal_rates": {}, + "signal_counts": {}, + "tier2_rates": {}, + "tier2_counts": {} + } + } + }, + "generated_from": { + "rows": 42 + }, + "version_tag": "v7.21", + "run_id": "20260210_215029" +} \ No newline at end of file diff --git a/run_manifest.json b/run_manifest.json new file mode 100644 index 0000000..3583448 --- /dev/null +++ b/run_manifest.json @@ -0,0 +1,32 @@ +{ + "run_id": "20260210_215029", + "timestamp": "20260210_215029", + "model_id": "Qwen/Qwen2.5-14B-Instruct-1M", + "hf_model_dir": "/mnt/h/Libraries/Mistralai/GGUF_Models/Qwen_Qwen2.5_14B_Instruct_1M/hf_snapshot", + "adapter_id": "qwen2_5_14b_instruct_1m", + "adapter_reason": "exact model_id match", + "allowed_root": "/mnt", + "work_root": "/mnt/h/Libraries/Mistralai", + "catalog_root": "/mnt/e/Libraries/Mistralai", + "promotion_status": "work", + "promotion_updated_at": "2026-02-11T03:55:50.056729Z", + "run_dir": "/mnt/h/Libraries/Mistralai/runs/Qwen2.5_14B_Instruct_1M/Qwen2.5_14B_Instruct_1M_20260210_215029", + "artifacts_dir": "/mnt/h/Libraries/Mistralai/runs/Qwen2.5_14B_Instruct_1M/Qwen2.5_14B_Instruct_1M_20260210_215029/artifacts", + "fixtures_dir": "/mnt/h/Libraries/Mistralai/runs/Qwen2.5_14B_Instruct_1M/Qwen2.5_14B_Instruct_1M_20260210_215029/fixtures", + "fixtures_file": "golden_oracle_fixtures_v7_21__sha256_6d71a0b9147c.json", + "fixtures_sha256": "6d71a0b9147c079371b02a94f3c149eb78a6adc03dc16ff6833b964fbf4174f0", + "fixtures_source_file": "golden_oracle_fixtures_v7_21.json", + "quant_types_requested": [], + "skip_quant": true, + "seed": 42, + "artifacts": [], + "runners": { + "full_weight_transformers": { + "name": "full_weight_transformers", + "reused_cache": 0, + "cache_reuse_path": "", + "cache_write_enabled": 1, + "cache_write_path": "/mnt/h/Libraries/Mistralai/runs/Qwen2.5_14B_Instruct_1M/Qwen2.5_14B_Instruct_1M_20260210_215029/full_weight_cache.json" + } + } +} \ No newline at end of file