commit 61dd72ea698d9dae7a48715d4afa29597642ec3d Author: ModelHub XC Date: Fri Sep 18 06:06:16 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: markopoloaiinc/Athena-mvrko-4B Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..3764994 --- /dev/null +++ b/README.md @@ -0,0 +1,332 @@ +--- +license: other # ⟨FILL: Decision 0.1 — Apache-2.0 (inherited from Qwen3-4B) if open, else 'other' + license_name/link⟩ +base_model: Qwen/Qwen3-4B +base_model_relation: finetune +library_name: transformers +pipeline_tag: text-generation +datasets: + - opera +tags: + - athena + - mvrko + - e-commerce + - shopping-behavior + - next-action-prediction + - large-event-model + - web-agents + - lora + - opera +metrics: + - exact_match +model-index: + - name: Athena + results: + - task: + type: next-action-prediction + name: Next-action prediction + dataset: + name: OPeRA (full official test set) + type: opera + config: exact-match + metrics: + - type: exact_match + value: 0.2450 + name: Exact-match (n=992) + verified: false +--- + +# Athena (4B): a Large Event Model for shopping behavior + +**Athena predicts the exact next action a real shopper takes on a live retail page.** +Given the reduced page state and the interaction history, it emits the next action as +structured JSON. On the **full official OPeRA test set (992 actions)** it scores +**24.50% strict exact-match, first among every model tested, ahead of GPT-5.6, Claude +Sonnet 5, and Claude Opus 4.8**, as a 4-billion-parameter model. + +Athena is the **base model of the mvrko simulation track**: the accurate, cheap, +self-hostable foundation for agentic shopping, next-action planning, and behavioral +simulation. + +- **Developer:** Markopolo AI +- **Model type:** Decoder-only causal LM (dense), LoRA fine-tune of an open base model +- **Base model:** [`Qwen/Qwen3-4B`](https://huggingface.co/Qwen/Qwen3-4B) +- **Modality:** reduced page state + interaction history → next action + (`action_type`, `semantic_id`, `input_text`) +- **Benchmark:** OPeRA next-action prediction, strict exact-match on the target element +- **Repository:** ⟨FILL: final repo path — Decision 1.5b⟩ +- **Release:** ⟨FILL: version string⟩ + +> Athena is a **fine-tune of an open base model.** The value is the training recipe, the +> observation-space engineering, the 32K long-context supervision, and a specialization a +> general-purpose model cannot reach by prompting. The base weights are the substrate; the +> moat is everything built on top. + +--- + +## TL;DR + +| | | +|---|---| +| **Headline** | **24.50% exact-match on OPeRA (full 992)**, #1 vs current frontier | +| **Base** | Qwen3-4B · LoRA r=32, α=32 · 32K context | +| **Output** | Structured next-action JSON (schema below); **98.9%** schema-valid on the 992 | +| **Reproducibility** | Reproduced cold on `transformers 5.5.0` / `torch 2.8.0` / bf16 / greedy | +| **What it is NOT** | It does **not** emit calibrated probabilities. See [Limitations](#limitations) | + +--- + +## Results: full official OPeRA test set (n = 992) + +Same test set (`md5 1e02a30d…`), same harness, same strict exact-match scorer for every +model. Frontier models are evaluated zero-shot / prompted; Athena is fine-tuned. + +| Model | Params | Exact-match | Δ vs Athena | Margin | +|---|---|---|---|---| +| **Athena (ours, fine-tuned)** | **4B** | **24.50%** | | | +| GPT-5.6 (prompted) | frontier | 22.58% | +1.92 | **parity** (≈1σ, unpaired) | +| *GPT-4.1 (published OPeRA baseline)* | *frontier* | *21.5%* | *+3.0* | *ahead of published SOTA* | +| Claude Sonnet 5 (prompted) | frontier | 18.35% | +6.15 | **separated** (≈3.3σ) | +| Claude Opus 4.8 (prompted) | frontier | 12.70% | +11.80 | **separated** | + +**Athena ranks first, ahead of every current-frontier model tested and the published +baseline.** + +### Honest statistical read + +- **vs Claude Opus 4.8: separated.** +11.8 points, far beyond sampling noise. +- **vs Claude Sonnet 5: separated.** +6.15 points, ≈3.3σ. +- **vs published GPT-4.1 baseline: ahead.** +3.0 points. +- **vs GPT-5.6: statistical parity, nominal edge.** +1.92 points is ≈1σ (unpaired), a + first-place finish with a nominal lead, **not** a statistically separated one. We report + it as such. A paired McNemar test on the shared 992 is the correct way to sharpen this + and is ⟨FILL: pending — Decision 0.2⟩. + +> We lead the board **and** state exactly how strong each margin is. Nothing is labeled +> "clear" unless the test supports it. + +### The task ceiling: what the frontier numbers reveal + +The entire current frontier lands between **12% and 23%** on OPeRA. The ceiling here is the +**difficulty of the task, not model size**: predicting the *exact* element a human clicks, +from dozens of candidates, is genuinely hard, and raw capability barely moves it. Athena's +24.50% is not "low". It is the **best result on a benchmark where the strongest general +models in the world sit below it.** The lever that moves this number is **behavioral +specialization**, not scale. + +--- + +## Why a specialist wins + +The frontier models return clean, schema-valid JSON. They understand the task perfectly. +They still lose, because next-action prediction requires knowing *how real shoppers ground +their intent in this interface*, and that knowledge is **behavioral, not linguistic**. A +prompt yields a fluent guess; it cannot supply behavior the model never learned. + +> The frontier models are excellent at language. Athena is excellent at shoppers. + +**Strongest supporting evidence, the OPeRA error analysis:** the benchmark's own error +taxonomy localizes frontier failure to *grounding* (naming the right element), not +*formatting*. ⟨FILL: cite the specific error-type percentages with the paper section, from +the benchmark source-fact sheet (artifact 1.7). Do not paraphrase from memory.⟩ + +--- + +## Architecture and the core innovation: the observation space + +The central innovation is not the weights. It is **how the web page is represented to the +model.** + +1. **A learned observation space.** Raw HTML is unlearnable at scale, since a single page + blows past any practical context window. Athena consumes a **structure-preserving + reduction** that keeps only the *named, actionable* elements (the ones an action can + target) and discards the rest, turning a sprawling DOM into a compact, typed, + model-legible page state. This is what makes the next-action target *predictable* + instead of buried. +2. **Long-context supervision at 32K.** Real sessions carry long histories and large pages; + Athena trains at a **32,768-token** context so it conditions on the full journey, not a + truncated snippet. +3. **Behavioral fine-tuning.** Trained directly on what shoppers do over this action space, + with completion-only masking on the target action. + +The **observation-space parser** (or a specification precise enough to rebuild it) is +released so third parties can reproduce the input format. See Reproduction. ⟨FILL: link — +artifact in Part 2⟩ + +--- + +## Quickstart + +> Copied from a tested script. See Reproduction. Requires `transformers==5.5.0`. + +```python +from transformers import AutoModelForCausalLM, AutoTokenizer +import torch, json + +REPO = "⟨FILL: repo id⟩" +tok = AutoTokenizer.from_pretrained(REPO, trust_remote_code=True) +mdl = AutoModelForCausalLM.from_pretrained(REPO, dtype=torch.bfloat16, + device_map="cuda", attn_implementation="sdpa", trust_remote_code=True).eval() + +messages = [ # see example_inputs/ for real OPeRA cases + {"role": "system", "content": "⟨exact system string — frozen prompt spec⟩"}, + {"role": "user", "content": "⟨reduced page state + interaction history⟩\n\n## Next action:"}, +] +prompt = tok.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) +enc = tok(prompt, return_tensors="pt", truncation=True, max_length=32768).to("cuda") +out = mdl.generate(**enc, max_new_tokens=96, do_sample=False, pad_token_id=tok.pad_token_id) +print(tok.decode(out[0][enc["input_ids"].shape[1]:], skip_special_tokens=True)) +# → {"action_type": "click", "click_type": "product_link", "semantic_id": "..."} +``` + +--- + +## Input / output schema + +**Output object** (validated on 98.9% of the 992 test outputs): + +```json +{"action_type": "click", "click_type": "product_link", "semantic_id": "active_item_list..product_detail"} +``` + +- **`action_type`**: enum ⟨FILL: authoritative enum from schema.json (artifact 1.3)⟩ +- **`click_type`**: enum ⟨FILL: authoritative enum from schema.json⟩ +- **`semantic_id`**: the exact named target element. +- **`input_text`**: present only for `input` actions. ⟨FILL: exact convention on non-input actions⟩ + +**Scoring:** strict exact-match. Predicted `action_type` **and** `semantic_id` (and +`input_text` for inputs) must equal the ground truth. No partial credit. + +--- + + +## Hardware + + +**Precision:** bf16 throughout (training and evaluation). + +**Weights:** ~8 GB on disk (4B parameters, bf16, safetensors), loaded as a single +merged model with no base-model dependency at inference. + +**Inference.** Athena runs on a single GPU. The headline evaluation was produced on one +NVIDIA B300 at 32,768-token context with batch size 2, greedy decoding, `sdpa` attention. +A 24 GB card is the practical floor for full 32K-context inference; shorter contexts +(4K to 8K, sufficient for most single-page states) fit comfortably in 16 GB. Memory is +dominated by the bf16 weights plus the KV cache, which grows linearly with context length. + +--- + +## Efficiency + +Athena is a **self-hostable 4B** model: 1 to 2 orders of magnitude cheaper per prediction +than prompting a frontier reasoning API, with a **direct answer and no billed reasoning +tokens**, **batched local inference** (no per-call round-trip, no rate limits), and **data +kept in-house**. Exact multiples are computed from the dated cost model. ⟨FILL: efficiency +multiple + pricing date — artifact 1.8⟩ + +--- + +## Applications + +Each tagged by which family component it requires and by maturity. + +| Application | Needs | Maturity | +|---|---|---| +| Next-action prediction / autocomplete of shopper intent | **Athena** (this model) | Benchmarked | +| Session replay: scoring a logged journey step by step | **Athena** (this model) | Benchmarked, this is how the 992-action result is measured | +| Behavioral simulation / free-running journey rollout | **Athena** plus an interactive page-state environment | Not demonstrated. Requires an environment that returns a new page state for a novel action; OPeRA provides logged trajectories only | +| Calibrated conversion / intent scoring | **A different family component** (calibrated intent head, AUC/ECE), **not Athena** | Separate model | + +> **Family note:** Athena predicts next actions and is scored on exact match. It does **not** +> emit calibrated probabilities. The calibrated intent head (AUC/ECE) is a **separate +> component** of the mvrko family. See the model family map. Do not attribute calibration +> claims to this model. + + +--- + +## Training details + +| Setting | Value | +|---|---| +| Base | `Qwen/Qwen3-4B` | +| Method | LoRA (r=32, α=32, dropout 0.1), bf16, gradient checkpointing | +| Target modules | q,k,v,o,gate,up,down proj | +| Context length | 32,768 tokens | +| Objective | completion-only masking on the next-action target | +| Epochs / LR / schedule | 1 / 1e-4 / cosine, warmup 0.03 | +| Hardware | 2× B300, DDP via torchrun | +| Decode (eval) | greedy (deterministic), max_new_tokens 96 | +| ⟨FILL from config.json / training config⟩ | layers, heads, KV heads, head dim, vocab, tokens seen | + +--- + +## Reproduction + +The evaluation harness (runner, parser, scorer, **frozen prompt file**), the +observation-space parser, per-example outputs for all models, and 3 to 5 example inputs are +released at markopoloaiinc/Athena-mvrko-4B. Pinned versions and the run record: +`transformers 5.5.0`, `torch 2.8.0+cu129`, bf16, greedy, batch 2. See +`eval_run_record.json`. + +**Verification status (Part 3):** +- [x] Cold-start reproduction of the 24.50% headline (indexed, no-dedup, `md5 1e02a30d`). +- [ ] ⟨FILL: cross-engine determinism (transformers↔vLLM agreement rate)⟩ +- [ ] ⟨FILL: batch-invariance spot check (bs 1 vs 32)⟩ +- [ ] ⟨FILL: fresh-environment quickstart test (< 15 min, by a non-author)⟩ +- [ ] ⟨FILL: adversarial read against the OPeRA paper⟩ +- [x] Schema validation: 98.9% of 992 outputs validate. + +--- + +## Limitations + +- **Fine-tuned vs. prompted.** Athena is fine-tuned on the task; frontier baselines are + prompted zero-shot. This is a *specialization* comparison, the intended one, not a claim + about raw model capability. +- **Observation format.** All models are scored on Athena's reduced-HTML observation space; + the frontier models see it cold. A different encoding could shift their numbers. +- **GPT-5.6 margin is within noise.** +1.92 points at n=992 is a first-place tie with a + nominal edge, pending a paired test. +- **This benchmark does not evidence calibration.** Exact-match measures grounding accuracy, + not probability quality. Athena emits no calibrated conversion signal. +- **Strict exact-match is unforgiving** by design. Absolute scores are low because the task + is hard, not because any model is failing. + +--- + +## Responsible use + +⟨FILL: intended-use scope, out-of-scope uses, data-provenance and privacy note, and that +predictions are behavioral estimates, not guarantees.⟩ + +--- + +## Related work + +- **OPeRA**, the benchmark and its published baselines. ⟨FILL: full citation with the real + author list copied from arXiv — artifact 1.7⟩ +- **RL-based OPeRA methods.** ⟨FILL: name the reinforcement-learning approaches on this + benchmark explicitly, so the comparison table is not only "us vs. prompted frontier."⟩ + +--- + +## License and citation + +- **License:** ⟨FILL: Decision 0.1 + the actual LICENSE file. Base `Qwen/Qwen3-4B` is + Apache-2.0. State inherited obligations if releasing open.⟩ +- **Citation:** + ```bibtex + @misc{athena_mvrko, + title = {Athena: a 4B Large Event Model for shopping-behavior next-action prediction}, + author = {Markopolo AI}, + year = {2026}, + note = {Markopolo AI} + } + ``` +- **Contact:** ⟨FILL⟩ + +--- + +*Athena is a 4B Large Event Model that predicts real shopper behavior more accurately than +the current frontier on a public benchmark: cheaply, self-hostably, and with every margin +stated honestly. It is the foundation of the mvrko simulation track.* \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..01be9b3 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,89 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} +{%- endfor %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if loop.index0 > ns.last_query_index %} + {%- if loop.last or (not loop.last and reasoning_content) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content.strip('\n') + '\n\n\n' + content.lstrip('\n') }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..1dd7c2c --- /dev/null +++ b/config.json @@ -0,0 +1,71 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 151643, + "dtype": "bfloat16", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2560, + "initializer_range": 0.02, + "intermediate_size": 9728, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 40960, + "max_window_layers": 36, + "model_type": "qwen3", + "num_attention_heads": 32, + "num_hidden_layers": 36, + "num_key_value_heads": 8, + "pad_token_id": null, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.5.0", + "use_cache": true, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..efdece3 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_k": 20, + "top_p": 0.95, + "transformers_version": "5.5.0" +} diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..7eae6bf --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12107b4cd99f4fbad8bc8333adbb92328ac7fa8c8aef7b71f7b7dba5bccc1d49 +size 8044982080 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..53fec88 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1 @@ +{"add_prefix_space": false, "backend": "tokenizers", "bos_token": null, "clean_up_tokenization_spaces": false, "eos_token": "<|im_end|>", "errors": "replace", "extra_special_tokens": {}, "is_local": false, "model_max_length": 131072, "pad_token": "<|endoftext|>", "split_special_tokens": false, "tokenizer_class": "Qwen2Tokenizer", "unk_token": null} \ No newline at end of file