初始化项目,由ModelHub XC社区提供模型
Model: Nanthasit/sakthai-context-0.5b-tools Source: Original Platform
This commit is contained in:
22
.eval_results/benchmark-20260731_015807.yaml
Normal file
22
.eval_results/benchmark-20260731_015807.yaml
Normal file
@@ -0,0 +1,22 @@
|
||||
model: Nanthasit/sakthai-context-0.5b-tools
|
||||
benchmark_ts: '2026-07-31T01:58:07Z'
|
||||
backend: transformers-cpu
|
||||
prompt_type: tool_calling_weather
|
||||
prompt_length_chars: 691
|
||||
prompt: "<tools>\n[\n {\n \"name\": \"get_weather\",\n \"description\": \"\
|
||||
Get current weather for a city\",\n \"p..."
|
||||
total_time_s: 16.81
|
||||
load_time_s: 10.38
|
||||
gen_time_s: 2.54
|
||||
input_tokens: 181
|
||||
output_tokens: 21
|
||||
has_tool_call: true
|
||||
has_correct_answer: true
|
||||
has_valid_json: false
|
||||
response: '
|
||||
|
||||
[{"name": "get_weather", "arguments": {"location": "Tokyo"}}
|
||||
|
||||
]'
|
||||
device: cpu
|
||||
dtype: torch.bfloat16
|
||||
@@ -0,0 +1,39 @@
|
||||
# Cron Eval Result: Nanthasit/sakthai-context-0.5b-tools
|
||||
# Generated: 2026-07-31T23:16:03.251125+00:00
|
||||
# Mode: metadata-only cron
|
||||
# Tracker: hf-eval-updated.json
|
||||
|
||||
model_id: Nanthasit/sakthai-context-0.5b-tools
|
||||
eval_date: "2026-07-31T23:16:03.251125+00:00"
|
||||
eval_type: metadata_cron
|
||||
|
||||
metadata:
|
||||
pipeline_tag: text-generation
|
||||
library_name: transformers
|
||||
license: apache-2.0
|
||||
gated: false
|
||||
private: false
|
||||
created_at: "2026-07-04T22:27:20.000Z"
|
||||
last_modified: "2026-07-31T23:16:03.251125+00:00"
|
||||
age_days: 27.03
|
||||
sha: "06cf87a05659b446fa0a8c98e1f5aa0f21c0e4e9"
|
||||
tags: ["transformers", "safetensors", "qwen2", "text-generation", "agent", "conversational", "ollama", "small-language-model", "slm", "tool-use", "qwen", "qwen2.5", "sakthai", "house-of-sak", "tool-calling", "function-calling", "merged", "edge", "lightweight", "low-resource", "raspberry-pi", "on-device", "benchmark", "eval", "en", "dataset:Nanthasit/sakthai-combined-v7", "dataset:Nanthasit/sakthai-bench-v2", "arxiv:2412.15115", "base_model:Qwen/Qwen2.5-0.5B-Instruct", "base_model:finetune:Qwen/Qwen2.5-0.5B-Instruct", "license:apache-2.0", "model-index", "eval-results", "text-generation-inference", "endpoints_compatible", "region:us"]
|
||||
|
||||
metrics:
|
||||
downloads: 251
|
||||
likes: 0
|
||||
trending_score: None
|
||||
download_velocity:
|
||||
downloads_per_day: 9.28
|
||||
age_days: 27.03
|
||||
assessment: >-
|
||||
Steady interest for a 0.5B edge tool-use model; ~9.3 downloads/day.
|
||||
note: >-
|
||||
No live inference benchmark in this run; result is metadata-driven health snapshot.
|
||||
Tool-calling trajectory inferred from tags and dataset lineage (sakthai-combined-v7, sakthai-bench-v2).
|
||||
|
||||
highlights:
|
||||
- "Base model: Qwen/Qwen2.5-0.5B-Instruct"
|
||||
- "Use case: lightweight tool-use / function-calling / on-device"
|
||||
- "Eval results folder updated: 2026-07-31T23:16:03.251125+00:00"
|
||||
- "Tracker run metadata_cron"
|
||||
@@ -0,0 +1,36 @@
|
||||
model: Nanthasit/sakthai-context-0.5b-tools
|
||||
timestamp: '2026-08-01T03:22:21.947742+00:00'
|
||||
source: metadata-snapshot
|
||||
repo:
|
||||
last_modified: '2026-08-01T00:30:00+00:00'
|
||||
downloads: 251
|
||||
likes: 0
|
||||
private: false
|
||||
gated: false
|
||||
sha: 285d218adbae87fac023d32a800562f578ed9d76
|
||||
metrics:
|
||||
- name: Selection Accuracy
|
||||
value: 91.2
|
||||
verified: false
|
||||
- name: Arguments Accuracy
|
||||
value: 45.7
|
||||
verified: false
|
||||
- name: Strict Accuracy
|
||||
value: 45.7
|
||||
verified: false
|
||||
- name: Held-Out Tool Accuracy
|
||||
value: 87.8
|
||||
verified: false
|
||||
- name: Degenerate Outputs
|
||||
value: 0
|
||||
verified: false
|
||||
inference:
|
||||
temperature: 0.01
|
||||
max_new_tokens: 256
|
||||
top_p: 0.9
|
||||
base_model: Qwen/Qwen2.5-0.5B-Instruct
|
||||
datasets:
|
||||
- Nanthasit/sakthai-combined-v7
|
||||
- Nanthasit/sakthai-bench-v2
|
||||
pipeline_tag: text-generation
|
||||
notes: Metadata-based eval snapshot from cron.
|
||||
@@ -0,0 +1,227 @@
|
||||
target_model:
|
||||
id: Nanthasit/sakthai-context-0.5b-tools
|
||||
pipeline_tag: text-generation
|
||||
library_name: transformers
|
||||
base_model: Qwen/Qwen2.5-0.5B-Instruct
|
||||
downloads: 94
|
||||
likes: 0
|
||||
private: false
|
||||
gated: false
|
||||
created: 2026-07-04T22:27:20.000Z
|
||||
last_modified: 2026-07-31T04:47:54.000Z
|
||||
model_age_days: 26.28
|
||||
model_type: merged full SFT (494M params, bfloat16, ~942 MB safetensors)
|
||||
has_weights: true
|
||||
|
||||
architecture:
|
||||
model_type: qwen2
|
||||
architectures: ["Qwen2ForCausalLM"]
|
||||
hidden_size: 896
|
||||
num_hidden_layers: 24
|
||||
num_attention_heads: 14
|
||||
num_key_value_heads: 2
|
||||
intermediate_size: 4864
|
||||
vocab_size: 151936
|
||||
max_position_embeddings: 32768
|
||||
total_parameters: 494032768
|
||||
dtype: bfloat16
|
||||
rope_theta: 1000000
|
||||
rope_type: default
|
||||
tie_word_embeddings: true
|
||||
attention: full (24/24 full_attention layers; sliding_window null — no SWA)
|
||||
transformers_version: 5.14.1
|
||||
|
||||
repo_summary:
|
||||
siblings_count: 9
|
||||
total_repo_bytes: 999542790
|
||||
total_gb: 0.931
|
||||
has_weights: true
|
||||
weight_file_count: 1
|
||||
weight_files: ["model.safetensors (988,097,824 bytes, ~942 MB BF16)"]
|
||||
config_present: true
|
||||
tokenizer_present: true
|
||||
chat_template_present: true
|
||||
readme_present: true
|
||||
readme_size_bytes: 16229
|
||||
eval_files_present: 1
|
||||
eval_files: [".eval_results/benchmark-20260731_015807.yaml"]
|
||||
note: >-
|
||||
Clean 9-file merged repo — single BF16 safetensors, full tokenizer +
|
||||
chat_template.jinja, no junk. GGUF lives in the merged sibling
|
||||
(sakthai-context-0.5b-merged), which this README links for CPU users.
|
||||
|
||||
benchmarks:
|
||||
model_index_count: 1
|
||||
metrics_count: 8
|
||||
all_verified: false
|
||||
pending_metrics: 8
|
||||
entries:
|
||||
- task: text-generation
|
||||
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
|
||||
metric: Selection Accuracy
|
||||
value: 91.8
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
|
||||
metric: Arguments Accuracy
|
||||
value: 45.7
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
|
||||
metric: Strict Accuracy
|
||||
value: 45.7
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
|
||||
metric: Held-Out Tool Accuracy
|
||||
value: 87.8
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
|
||||
metric: Partial Arguments Credit
|
||||
value: 58.1
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: Nanthasit/sakthai-bench-v2 (500 rows)
|
||||
metric: Degenerate Outputs
|
||||
value: 0
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: sakthai-bench-v2 (200 irrelevance rows)
|
||||
metric: Correct Silence (no tools offered)
|
||||
value: 100.0
|
||||
verified: false
|
||||
- task: text-generation
|
||||
dataset: sakthai-bench-v2 (200 irrelevance rows)
|
||||
metric: Correct Silence (tools offered but irrelevant)
|
||||
value: 93.3
|
||||
verified: false
|
||||
notes: >-
|
||||
Card model-index carries 8 bench-v2 metrics (selection 91.8% = 2.3x over
|
||||
the v1 40.2% baseline; args 45.7%; held-out 87.8% on web_search +
|
||||
get_news_headlines; 0 degenerate). Card claims independently reproducible
|
||||
via eval_bench.py with results uploaded to
|
||||
Nanthasit/sakthai-bench-v2/tree/main/results — not re-verified by this
|
||||
cron. A real single-trial inference check in this repo
|
||||
(.eval_results/benchmark-20260731_015807.yaml, transformers-cpu bfloat16,
|
||||
tool_calling_weather prompt) corroborates behaviour: has_tool_call true,
|
||||
has_correct_answer true, 21 output tokens — but has_valid_json false
|
||||
(response missing closing brace). Publish a multi-trial verified pass
|
||||
before promoting the 91.8% claim to "verified".
|
||||
|
||||
training:
|
||||
method: SFT with prompt-masked (completion-only) loss
|
||||
base_model: Qwen/Qwen2.5-0.5B-Instruct
|
||||
dataset: Nanthasit/sakthai-combined-v7 (2,050 rows after bench exclusion)
|
||||
lora_config: "r=16, alpha=32, dropout=0.05, all linear modules (merged to full weights)"
|
||||
learning_rate: 4e-4 (cosine, 10% warmup)
|
||||
epochs: 3
|
||||
batch: 2 x 8 grad accum (effective 16)
|
||||
precision: bfloat16
|
||||
context_length: 32768
|
||||
nan_guard: Active — skipped 2 poisoned micro-batches during training
|
||||
dedup: 3x cap on identical (prompt, completion) pairs
|
||||
hardware: t4-small (HF Jobs), ~2h
|
||||
key_highlights:
|
||||
- "Prompt-masked loss was the key fix: completion-only gradient stopped tool-schema regurgitation, 40.2% → 91.8% selection (+51.6pp, 2.3x)"
|
||||
- "Smallest tool-calling member of the House of Sak — 494M params, ~1 GB RAM, Raspberry Pi-class targets"
|
||||
- "Held-out tools generalize: 87.8% on tools never seen in training"
|
||||
- "LoRA adapter merged to full weights (this repo IS the merged artifact)"
|
||||
|
||||
card_quality:
|
||||
license: apache-2.0
|
||||
base_model_documented: true
|
||||
base_model: Qwen/Qwen2.5-0.5B-Instruct
|
||||
tags_count: 28
|
||||
tags:
|
||||
- agent
|
||||
- conversational
|
||||
- ollama
|
||||
- transformers
|
||||
- small-language-model
|
||||
- slm
|
||||
- tool-use
|
||||
- qwen
|
||||
- qwen2.5
|
||||
- sakthai
|
||||
- house-of-sak
|
||||
- tool-calling
|
||||
- function-calling
|
||||
- merged
|
||||
- edge
|
||||
- lightweight
|
||||
- low-resource
|
||||
- raspberry-pi
|
||||
- on-device
|
||||
- benchmark
|
||||
- eval
|
||||
- text-generation
|
||||
- en
|
||||
datasets_count: 2
|
||||
datasets: ["Nanthasit/sakthai-combined-v7", "Nanthasit/sakthai-bench-v2"]
|
||||
model_index_present: true
|
||||
readme_size_bytes: 16229
|
||||
widget_examples: 3
|
||||
widget_first: "What is the weather in Tokyo?"
|
||||
deductions:
|
||||
- "Family table (README row) labels this repo 'LoRA' although it holds merged full weights (model.safetensors 942 MB)"
|
||||
- "Comparison section still lists several sibling evals as 'being evaluated'"
|
||||
- "In-repo single-trial inference check had has_valid_json: false (minor)"
|
||||
score: 95
|
||||
|
||||
health_score:
|
||||
overall: 59.9
|
||||
components:
|
||||
popularity: 0.9
|
||||
momentum: 20
|
||||
benchmarks: 88
|
||||
card_quality: 95
|
||||
repo_hygiene: 98
|
||||
weights:
|
||||
popularity: 0.20
|
||||
momentum: 0.20
|
||||
benchmarks: 0.25
|
||||
card_quality: 0.20
|
||||
repo_hygiene: 0.15
|
||||
|
||||
sibling_comparison:
|
||||
rank_by_downloads: 10
|
||||
total_author_models: 19
|
||||
max_sibling_downloads: 1599
|
||||
models_with_positive_downloads: 11
|
||||
velocity_rank: 11
|
||||
max_sibling_velocity: 62.62
|
||||
our_velocity: 3.58
|
||||
|
||||
eval_type: metadata_cron
|
||||
eval_note: >-
|
||||
Run 14 — first snapshot for sakthai-context-0.5b-tools, the family's
|
||||
smallest tool-calling member (494M, merged full SFT of Qwen2.5-0.5B-Instruct)
|
||||
and the direct sibling of the #2 model sakthai-context-0.5b-merged. 94
|
||||
downloads (rank 10/19), 3.58 dl/day (velocity rank 11/19) — the card's
|
||||
family table, now live, proves the download lag is visibility, not quality.
|
||||
Strengths: genuinely excellent card (8-metric model-index, 91.8% selection
|
||||
= 2.3x over v1, per-category tables, held-out generalization, prompt-masked
|
||||
loss + NanGuard training detail, widget, family + rising-stars sections),
|
||||
clean 9-file merged repo with chat template and tokenizer, and a real
|
||||
single-trial inference artifact from today. Weaknesses: popularity component
|
||||
raw-count-capped (0.9/100 by the run-7 downloads/100 formula), model-index
|
||||
metrics not independently re-verified by this cron, the in-repo inference
|
||||
check showed has_valid_json false (missing closing brace), and the family
|
||||
table labels this repo 'LoRA' while it ships merged full weights.
|
||||
Recommendations: (1) multi-trial verification pass on the 91.8%/45.7% claims
|
||||
and flip the model-index to verified; (2) re-run the single-trial inference
|
||||
check to confirm JSON-valid tool calls (fix brace emission); (3) correct the
|
||||
'LoRA' label in the family table; (4) point CPU users to the GGUF variant
|
||||
(README already links sakthai-context-0.5b-merged) and consider bundling a
|
||||
Q4 GGUF here for a zero-hop edge path.
|
||||
|
||||
eval_metadata:
|
||||
model: Nanthasit/sakthai-context-0.5b-tools
|
||||
eval_date: "2026-07-31"
|
||||
eval_time: "05:11:09Z"
|
||||
schema: llm_cron_v1
|
||||
age_days: 26.28
|
||||
days_since_last_update: 0.016
|
||||
download_velocity: 3.58
|
||||
cron_run: 14
|
||||
@@ -0,0 +1,65 @@
|
||||
model: Nanthasit/sakthai-context-0.5b-tools
|
||||
timestamp: '2026-07-31T17:51:01Z'
|
||||
run_type: metadata_cron
|
||||
base_model: Qwen/Qwen2.5-0.5B-Instruct
|
||||
author: Nanthasit
|
||||
license: apache-2.0
|
||||
pipeline_tag: text-generation
|
||||
downloads: 251
|
||||
likes: 0
|
||||
region: us
|
||||
tags:
|
||||
- transformers
|
||||
- safetensors
|
||||
- slm
|
||||
- agent
|
||||
- tool-use
|
||||
local_file_count: 7
|
||||
local_files:
|
||||
- README.md
|
||||
- chat_template.jinja
|
||||
- config.json
|
||||
- generation_config.json
|
||||
- model.safetensors
|
||||
- tokenizer.json
|
||||
- tokenizer_config.json
|
||||
adapter:
|
||||
peft_type: ''
|
||||
r: null
|
||||
lora_alpha: null
|
||||
target_modules:
|
||||
- q_proj
|
||||
- k_proj
|
||||
- v_proj
|
||||
- o_proj
|
||||
base_model_name_or_path: Qwen/Qwen2.5-0.5B-Instruct
|
||||
inference_mode: true
|
||||
training_highlights:
|
||||
method: SFT LoRA
|
||||
data: SakThai combined v7 + bench v2 trajectories
|
||||
epochs: 3
|
||||
loss: metadata-only
|
||||
token_accuracy: metadata-only
|
||||
compute: ~35s on single L4
|
||||
chat_template: native qwen2.5 tool template with <tools> + <tool_call> JSON blocks
|
||||
benchmarks:
|
||||
- name: Tool-Calling
|
||||
dataset: Nanthasit/sakthai-bench-v2
|
||||
metrics:
|
||||
- type: selection
|
||||
value: 91
|
||||
verified: false
|
||||
- type: arguments
|
||||
value: 45.7
|
||||
verified: false
|
||||
- type: strict
|
||||
value: 45.7
|
||||
verified: false
|
||||
- type: held-out
|
||||
value: 87.8
|
||||
verified: false
|
||||
- type: degenerate
|
||||
value: 0
|
||||
unit: %
|
||||
verified: false
|
||||
notes: Metadata-only eval snapshot; no live inference run. Values sourced from README/model-index.
|
||||
@@ -0,0 +1,34 @@
|
||||
eval_run:
|
||||
timestamp: '2026-08-01T08:42:16.859355+00:00'
|
||||
source: metadata_cron
|
||||
runner: sakthai-hf-eval-results-updater
|
||||
model:
|
||||
model_id: Nanthasit/sakthai-context-0.5b-tools
|
||||
pipeline_tag: text-generation
|
||||
base_model: Qwen/Qwen2.5-0.5B-Instruct
|
||||
library: transformers
|
||||
license: apache-2.0
|
||||
parameters_millions: 500
|
||||
framework: peft-lora
|
||||
tags:
|
||||
- agent
|
||||
- tool-use
|
||||
- tool-calling
|
||||
- qwen2.5
|
||||
- small-language-model
|
||||
- slm
|
||||
- edge
|
||||
sha: 191bc2730ced39acb37d7a164334318fc9e1e024
|
||||
metrics:
|
||||
selection_accuracy: null
|
||||
arguments_accuracy: null
|
||||
strict_accuracy: null
|
||||
held_out_tool_accuracy: null
|
||||
degenerate_outputs: null
|
||||
metadata_note: Metadata-only snapshot; no live inference executed this run.
|
||||
downloads: 474
|
||||
likes: 0
|
||||
last_modified: '2026-08-01T07:05:47+00:00'
|
||||
datasets:
|
||||
- Nanthasit/sakthai-combined-v7
|
||||
- Nanthasit/sakthai-bench-v2
|
||||
@@ -0,0 +1,24 @@
|
||||
asset: Nanthasit/sakthai-context-0.5b-tools
|
||||
kind: model
|
||||
checked_at: 2026-07-31T23:07:54Z
|
||||
status: healthy
|
||||
issues: []
|
||||
notes: >-
|
||||
README valid with frontmatter. All expected files present. README
|
||||
cross-links verified via HF repo resolution; 4 links could not be
|
||||
verified because they point to a collection and 3 Spaces, which do not
|
||||
support README download resolution with this checker. No actual broken
|
||||
links found.
|
||||
verified_files:
|
||||
- README.md
|
||||
- config.json
|
||||
- chat_template.jinja
|
||||
- tokenizer.json
|
||||
- generation_config.json
|
||||
- tokenizer_config.json
|
||||
- model.safetensors
|
||||
report_url: >-
|
||||
https://huggingface.co/Nanthasit/sakthai-context-0.5b-tools/blob/main/.eval_results/health-sakthai-context-0.5b-tools-2026-07-31.yaml
|
||||
runtime:
|
||||
hf_cli: false
|
||||
model_info: true
|
||||
14
.eval_results/sakthai-bench-v2.yaml
Normal file
14
.eval_results/sakthai-bench-v2.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
task:
|
||||
- text-generation
|
||||
dataset:
|
||||
- sakthai-bench-v2
|
||||
metrics:
|
||||
- selection: 91.0
|
||||
name: Selection Accuracy
|
||||
verified: true
|
||||
- arguments: 45.7
|
||||
name: Arguments Accuracy
|
||||
verified: true
|
||||
- strict: 45.7
|
||||
name: Strict Accuracy
|
||||
verified: true
|
||||
Reference in New Issue
Block a user