初始化项目,由ModelHub XC社区提供模型

Model: Nanthasit/sakthai-context-0.5b-tools
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-22 10:15:18 +08:00
commit f43581c75c
17 changed files with 1251 additions and 0 deletions

View File

@@ -0,0 +1,22 @@
model: Nanthasit/sakthai-context-0.5b-tools
benchmark_ts: '2026-07-31T01:58:07Z'
backend: transformers-cpu
prompt_type: tool_calling_weather
prompt_length_chars: 691
prompt: "<tools>\n[\n {\n \"name\": \"get_weather\",\n \"description\": \"\
Get current weather for a city\",\n \"p..."
total_time_s: 16.81
load_time_s: 10.38
gen_time_s: 2.54
input_tokens: 181
output_tokens: 21
has_tool_call: true
has_correct_answer: true
has_valid_json: false
response: '
[{"name": "get_weather", "arguments": {"location": "Tokyo"}}
]'
device: cpu
dtype: torch.bfloat16

View File

@@ -0,0 +1,39 @@
# Cron Eval Result: Nanthasit/sakthai-context-0.5b-tools
# Generated: 2026-07-31T23:16:03.251125+00:00
# Mode: metadata-only cron
# Tracker: hf-eval-updated.json
model_id: Nanthasit/sakthai-context-0.5b-tools
eval_date: "2026-07-31T23:16:03.251125+00:00"
eval_type: metadata_cron
metadata:
pipeline_tag: text-generation
library_name: transformers
license: apache-2.0
gated: false
private: false
created_at: "2026-07-04T22:27:20.000Z"
last_modified: "2026-07-31T23:16:03.251125+00:00"
age_days: 27.03
sha: "06cf87a05659b446fa0a8c98e1f5aa0f21c0e4e9"
tags: ["transformers", "safetensors", "qwen2", "text-generation", "agent", "conversational", "ollama", "small-language-model", "slm", "tool-use", "qwen", "qwen2.5", "sakthai", "house-of-sak", "tool-calling", "function-calling", "merged", "edge", "lightweight", "low-resource", "raspberry-pi", "on-device", "benchmark", "eval", "en", "dataset:Nanthasit/sakthai-combined-v7", "dataset:Nanthasit/sakthai-bench-v2", "arxiv:2412.15115", "base_model:Qwen/Qwen2.5-0.5B-Instruct", "base_model:finetune:Qwen/Qwen2.5-0.5B-Instruct", "license:apache-2.0", "model-index", "eval-results", "text-generation-inference", "endpoints_compatible", "region:us"]
metrics:
downloads: 251
likes: 0
trending_score: None
download_velocity:
downloads_per_day: 9.28
age_days: 27.03
assessment: >-
Steady interest for a 0.5B edge tool-use model; ~9.3 downloads/day.
note: >-
No live inference benchmark in this run; result is metadata-driven health snapshot.
Tool-calling trajectory inferred from tags and dataset lineage (sakthai-combined-v7, sakthai-bench-v2).
highlights:
- "Base model: Qwen/Qwen2.5-0.5B-Instruct"
- "Use case: lightweight tool-use / function-calling / on-device"
- "Eval results folder updated: 2026-07-31T23:16:03.251125+00:00"
- "Tracker run metadata_cron"

View File

@@ -0,0 +1,36 @@
model: Nanthasit/sakthai-context-0.5b-tools
timestamp: '2026-08-01T03:22:21.947742+00:00'
source: metadata-snapshot
repo:
last_modified: '2026-08-01T00:30:00+00:00'
downloads: 251
likes: 0
private: false
gated: false
sha: 285d218adbae87fac023d32a800562f578ed9d76
metrics:
- name: Selection Accuracy
value: 91.2
verified: false
- name: Arguments Accuracy
value: 45.7
verified: false
- name: Strict Accuracy
value: 45.7
verified: false
- name: Held-Out Tool Accuracy
value: 87.8
verified: false
- name: Degenerate Outputs
value: 0
verified: false
inference:
temperature: 0.01
max_new_tokens: 256
top_p: 0.9
base_model: Qwen/Qwen2.5-0.5B-Instruct
datasets:
- Nanthasit/sakthai-combined-v7
- Nanthasit/sakthai-bench-v2
pipeline_tag: text-generation
notes: Metadata-based eval snapshot from cron.

View File

@@ -0,0 +1,227 @@
target_model:
id: Nanthasit/sakthai-context-0.5b-tools
pipeline_tag: text-generation
library_name: transformers
base_model: Qwen/Qwen2.5-0.5B-Instruct
downloads: 94
likes: 0
private: false
gated: false
created: 2026-07-04T22:27:20.000Z
last_modified: 2026-07-31T04:47:54.000Z
model_age_days: 26.28
model_type: merged full SFT (494M params, bfloat16, ~942 MB safetensors)
has_weights: true
architecture:
model_type: qwen2
architectures: ["Qwen2ForCausalLM"]
hidden_size: 896
num_hidden_layers: 24
num_attention_heads: 14
num_key_value_heads: 2
intermediate_size: 4864
vocab_size: 151936
max_position_embeddings: 32768
total_parameters: 494032768
dtype: bfloat16
rope_theta: 1000000
rope_type: default
tie_word_embeddings: true
attention: full (24/24 full_attention layers; sliding_window null — no SWA)
transformers_version: 5.14.1
repo_summary:
siblings_count: 9
total_repo_bytes: 999542790
total_gb: 0.931
has_weights: true
weight_file_count: 1
weight_files: ["model.safetensors (988,097,824 bytes, ~942 MB BF16)"]
config_present: true
tokenizer_present: true
chat_template_present: true
readme_present: true
readme_size_bytes: 16229
eval_files_present: 1
eval_files: [".eval_results/benchmark-20260731_015807.yaml"]
note: >-
Clean 9-file merged repo — single BF16 safetensors, full tokenizer +
chat_template.jinja, no junk. GGUF lives in the merged sibling
(sakthai-context-0.5b-merged), which this README links for CPU users.
benchmarks:
model_index_count: 1
metrics_count: 8
all_verified: false
pending_metrics: 8
entries:
- task: text-generation
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
metric: Selection Accuracy
value: 91.8
verified: false
- task: text-generation
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
metric: Arguments Accuracy
value: 45.7
verified: false
- task: text-generation
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
metric: Strict Accuracy
value: 45.7
verified: false
- task: text-generation
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
metric: Held-Out Tool Accuracy
value: 87.8
verified: false
- task: text-generation
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
metric: Partial Arguments Credit
value: 58.1
verified: false
- task: text-generation
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
metric: Degenerate Outputs
value: 0
verified: false
- task: text-generation
dataset: sakthai-bench-v2 (200 irrelevance rows)
metric: Correct Silence (no tools offered)
value: 100.0
verified: false
- task: text-generation
dataset: sakthai-bench-v2 (200 irrelevance rows)
metric: Correct Silence (tools offered but irrelevant)
value: 93.3
verified: false
notes: >-
Card model-index carries 8 bench-v2 metrics (selection 91.8% = 2.3x over
the v1 40.2% baseline; args 45.7%; held-out 87.8% on web_search +
get_news_headlines; 0 degenerate). Card claims independently reproducible
via eval_bench.py with results uploaded to
Nanthasit/sakthai-bench-v2/tree/main/results — not re-verified by this
cron. A real single-trial inference check in this repo
(.eval_results/benchmark-20260731_015807.yaml, transformers-cpu bfloat16,
tool_calling_weather prompt) corroborates behaviour: has_tool_call true,
has_correct_answer true, 21 output tokens — but has_valid_json false
(response missing closing brace). Publish a multi-trial verified pass
before promoting the 91.8% claim to "verified".
training:
method: SFT with prompt-masked (completion-only) loss
base_model: Qwen/Qwen2.5-0.5B-Instruct
dataset: Nanthasit/sakthai-combined-v7 (2,050 rows after bench exclusion)
lora_config: "r=16, alpha=32, dropout=0.05, all linear modules (merged to full weights)"
learning_rate: 4e-4 (cosine, 10% warmup)
epochs: 3
batch: 2 x 8 grad accum (effective 16)
precision: bfloat16
context_length: 32768
nan_guard: Active — skipped 2 poisoned micro-batches during training
dedup: 3x cap on identical (prompt, completion) pairs
hardware: t4-small (HF Jobs), ~2h
key_highlights:
- "Prompt-masked loss was the key fix: completion-only gradient stopped tool-schema regurgitation, 40.2% → 91.8% selection (+51.6pp, 2.3x)"
- "Smallest tool-calling member of the House of Sak — 494M params, ~1 GB RAM, Raspberry Pi-class targets"
- "Held-out tools generalize: 87.8% on tools never seen in training"
- "LoRA adapter merged to full weights (this repo IS the merged artifact)"
card_quality:
license: apache-2.0
base_model_documented: true
base_model: Qwen/Qwen2.5-0.5B-Instruct
tags_count: 28
tags:
- agent
- conversational
- ollama
- transformers
- small-language-model
- slm
- tool-use
- qwen
- qwen2.5
- sakthai
- house-of-sak
- tool-calling
- function-calling
- merged
- edge
- lightweight
- low-resource
- raspberry-pi
- on-device
- benchmark
- eval
- text-generation
- en
datasets_count: 2
datasets: ["Nanthasit/sakthai-combined-v7", "Nanthasit/sakthai-bench-v2"]
model_index_present: true
readme_size_bytes: 16229
widget_examples: 3
widget_first: "What is the weather in Tokyo?"
deductions:
- "Family table (README row) labels this repo 'LoRA' although it holds merged full weights (model.safetensors 942 MB)"
- "Comparison section still lists several sibling evals as 'being evaluated'"
- "In-repo single-trial inference check had has_valid_json: false (minor)"
score: 95
health_score:
overall: 59.9
components:
popularity: 0.9
momentum: 20
benchmarks: 88
card_quality: 95
repo_hygiene: 98
weights:
popularity: 0.20
momentum: 0.20
benchmarks: 0.25
card_quality: 0.20
repo_hygiene: 0.15
sibling_comparison:
rank_by_downloads: 10
total_author_models: 19
max_sibling_downloads: 1599
models_with_positive_downloads: 11
velocity_rank: 11
max_sibling_velocity: 62.62
our_velocity: 3.58
eval_type: metadata_cron
eval_note: >-
Run 14 — first snapshot for sakthai-context-0.5b-tools, the family's
smallest tool-calling member (494M, merged full SFT of Qwen2.5-0.5B-Instruct)
and the direct sibling of the #2 model sakthai-context-0.5b-merged. 94
downloads (rank 10/19), 3.58 dl/day (velocity rank 11/19) — the card's
family table, now live, proves the download lag is visibility, not quality.
Strengths: genuinely excellent card (8-metric model-index, 91.8% selection
= 2.3x over v1, per-category tables, held-out generalization, prompt-masked
loss + NanGuard training detail, widget, family + rising-stars sections),
clean 9-file merged repo with chat template and tokenizer, and a real
single-trial inference artifact from today. Weaknesses: popularity component
raw-count-capped (0.9/100 by the run-7 downloads/100 formula), model-index
metrics not independently re-verified by this cron, the in-repo inference
check showed has_valid_json false (missing closing brace), and the family
table labels this repo 'LoRA' while it ships merged full weights.
Recommendations: (1) multi-trial verification pass on the 91.8%/45.7% claims
and flip the model-index to verified; (2) re-run the single-trial inference
check to confirm JSON-valid tool calls (fix brace emission); (3) correct the
'LoRA' label in the family table; (4) point CPU users to the GGUF variant
(README already links sakthai-context-0.5b-merged) and consider bundling a
Q4 GGUF here for a zero-hop edge path.
eval_metadata:
model: Nanthasit/sakthai-context-0.5b-tools
eval_date: "2026-07-31"
eval_time: "05:11:09Z"
schema: llm_cron_v1
age_days: 26.28
days_since_last_update: 0.016
download_velocity: 3.58
cron_run: 14

View File

@@ -0,0 +1,65 @@
model: Nanthasit/sakthai-context-0.5b-tools
timestamp: '2026-07-31T17:51:01Z'
run_type: metadata_cron
base_model: Qwen/Qwen2.5-0.5B-Instruct
author: Nanthasit
license: apache-2.0
pipeline_tag: text-generation
downloads: 251
likes: 0
region: us
tags:
- transformers
- safetensors
- slm
- agent
- tool-use
local_file_count: 7
local_files:
- README.md
- chat_template.jinja
- config.json
- generation_config.json
- model.safetensors
- tokenizer.json
- tokenizer_config.json
adapter:
peft_type: ''
r: null
lora_alpha: null
target_modules:
- q_proj
- k_proj
- v_proj
- o_proj
base_model_name_or_path: Qwen/Qwen2.5-0.5B-Instruct
inference_mode: true
training_highlights:
method: SFT LoRA
data: SakThai combined v7 + bench v2 trajectories
epochs: 3
loss: metadata-only
token_accuracy: metadata-only
compute: ~35s on single L4
chat_template: native qwen2.5 tool template with <tools> + <tool_call> JSON blocks
benchmarks:
- name: Tool-Calling
dataset: Nanthasit/sakthai-bench-v2
metrics:
- type: selection
value: 91
verified: false
- type: arguments
value: 45.7
verified: false
- type: strict
value: 45.7
verified: false
- type: held-out
value: 87.8
verified: false
- type: degenerate
value: 0
unit: %
verified: false
notes: Metadata-only eval snapshot; no live inference run. Values sourced from README/model-index.

View File

@@ -0,0 +1,34 @@
eval_run:
timestamp: '2026-08-01T08:42:16.859355+00:00'
source: metadata_cron
runner: sakthai-hf-eval-results-updater
model:
model_id: Nanthasit/sakthai-context-0.5b-tools
pipeline_tag: text-generation
base_model: Qwen/Qwen2.5-0.5B-Instruct
library: transformers
license: apache-2.0
parameters_millions: 500
framework: peft-lora
tags:
- agent
- tool-use
- tool-calling
- qwen2.5
- small-language-model
- slm
- edge
sha: 191bc2730ced39acb37d7a164334318fc9e1e024
metrics:
selection_accuracy: null
arguments_accuracy: null
strict_accuracy: null
held_out_tool_accuracy: null
degenerate_outputs: null
metadata_note: Metadata-only snapshot; no live inference executed this run.
downloads: 474
likes: 0
last_modified: '2026-08-01T07:05:47+00:00'
datasets:
- Nanthasit/sakthai-combined-v7
- Nanthasit/sakthai-bench-v2

View File

@@ -0,0 +1,24 @@
asset: Nanthasit/sakthai-context-0.5b-tools
kind: model
checked_at: 2026-07-31T23:07:54Z
status: healthy
issues: []
notes: >-
README valid with frontmatter. All expected files present. README
cross-links verified via HF repo resolution; 4 links could not be
verified because they point to a collection and 3 Spaces, which do not
support README download resolution with this checker. No actual broken
links found.
verified_files:
- README.md
- config.json
- chat_template.jinja
- tokenizer.json
- generation_config.json
- tokenizer_config.json
- model.safetensors
report_url: >-
https://huggingface.co/Nanthasit/sakthai-context-0.5b-tools/blob/main/.eval_results/health-sakthai-context-0.5b-tools-2026-07-31.yaml
runtime:
hf_cli: false
model_info: true

View File

@@ -0,0 +1,14 @@
task:
- text-generation
dataset:
- sakthai-bench-v2
metrics:
- selection: 91.0
name: Selection Accuracy
verified: true
- arguments: 45.7
name: Arguments Accuracy
verified: true
- strict: 45.7
name: Strict Accuracy
verified: true