201 lines
7.4 KiB
YAML
201 lines
7.4 KiB
YAML
target_model:
|
|
id: Nanthasit/sakthai-plus-1.5b
|
|
pipeline_tag: text-generation
|
|
library_name: transformers
|
|
base_model: Qwen/Qwen2.5-1.5B-Instruct
|
|
downloads: 244
|
|
likes: 0
|
|
private: false
|
|
gated: false
|
|
last_modified: "2026-07-31T11:09:05.000Z"
|
|
model_age_days: 1.00
|
|
model_type: llm
|
|
has_weights: true
|
|
|
|
architecture:
|
|
base_model_type: qwen2
|
|
base_architectures: ["Qwen2ForCausalLM"]
|
|
base_hidden_size: 1536
|
|
base_num_hidden_layers: 28
|
|
base_num_attention_heads: 12
|
|
base_num_key_value_heads: 2
|
|
base_intermediate_size: 8960
|
|
base_vocab_size: 151936
|
|
base_max_position_embeddings: 32768
|
|
base_total_parameters: 1540000000
|
|
base_dtype: bfloat16
|
|
tie_word_embeddings: true
|
|
|
|
repo_summary:
|
|
siblings_count: 20
|
|
total_repo_bytes: 3098940902
|
|
total_gb: 2.886
|
|
has_weights: true
|
|
weight_file_count: 1
|
|
weight_bytes: 3087467144
|
|
weight_files: ["model.safetensors"]
|
|
config_present: true
|
|
tokenizer_present: true
|
|
chat_template_present: true
|
|
readme_present: true
|
|
readme_size_bytes: 18052
|
|
weight_note: "Single merged full-weight checkpoint (2.88 GiB, bf16) — no PEFT dependency at inference. Standard Qwen2.5 layout: config.json, generation_config.json, tokenizer.json, tokenizer_config.json, chat_template.jinja, README.md."
|
|
|
|
benchmarks:
|
|
model_index_count: 1
|
|
metrics_count: 9
|
|
all_verified: false
|
|
pending_metrics: 8
|
|
entries:
|
|
- dataset: lighteval
|
|
metric: winogrande
|
|
value: 59.6
|
|
verified: false
|
|
note: "entry in .eval_results/lighteval.yaml — not re-run, unverified"
|
|
- dataset: lighteval
|
|
metric: gsm8k
|
|
value: 50.9
|
|
verified: false
|
|
note: "entry in .eval_results/lighteval.yaml — not re-run, unverified"
|
|
- dataset: lighteval
|
|
metric: hellaswag
|
|
value: 34.0
|
|
verified: false
|
|
note: "entry in .eval_results/lighteval.yaml — not re-run, unverified"
|
|
- dataset: sakthai-bench-v2
|
|
metric: selection_accuracy
|
|
value: 84.8
|
|
verified: false
|
|
note: "entry in .eval_results/sakthai-bench-v2.yaml — not re-run, unverified"
|
|
- dataset: sakthai-bench-v2
|
|
metric: arguments_accuracy
|
|
value: 33.7
|
|
verified: false
|
|
note: "entry in .eval_results/sakthai-bench-v2.yaml — not re-run, unverified"
|
|
- dataset: sakthai-bench-v2
|
|
metric: strict_accuracy
|
|
value: 33.7
|
|
verified: false
|
|
note: "entry in .eval_results/sakthai-bench-v2.yaml — not re-run, unverified"
|
|
- dataset: llama.cpp tool-calling (3-trial, q4_k_m)
|
|
metric: tool_call_success
|
|
value: 1.0
|
|
verified: true
|
|
note: ".eval_results/benchmark-20260731_043634.yaml — send_email prompt, 3/3 tool calls"
|
|
- dataset: llama.cpp tool-calling (3-trial, q4_k_m)
|
|
metric: valid_json
|
|
value: 1.0
|
|
verified: true
|
|
note: ".eval_results/benchmark-20260731_043634.yaml — 3/3 valid JSON tool args"
|
|
- dataset: llama.cpp tool-calling (3-trial, q4_k_m)
|
|
metric: correct_answer
|
|
value: 1.0
|
|
verified: true
|
|
note: ".eval_results/benchmark-20260731_043634.yaml — 3/3 correct answers, avg 20.7 tps (CPU, 2 threads)"
|
|
- dataset: sakthai-bench-v2 (model-index)
|
|
metric: tool_calling_accuracy
|
|
value: 1.0
|
|
verified: false
|
|
note: "model-index entry: 'Tool Calling Accuracy (3/3)' from send_email scenario — single-trial, unverified"
|
|
notes: >
|
|
README frontmatter now includes a model-index entry (Tool Calling
|
|
Accuracy 1.0 on send_email scenario, unverified). 3 additional unverified
|
|
lighteval scores (winogrande 59.6, gsm8k 50.9, hellaswag 34.0) and
|
|
3 sakthai-bench-v2 scores (selection 84.8, arguments 33.7, strict 33.7)
|
|
exist in .eval_results/ but are NOT in the model-index. One verified
|
|
3-trial llama.cpp q4_k_m tool-calling run (3/3, 20.7 tps) is present.
|
|
Card text still says 'Benchmarks are pending' — stale relative to the
|
|
10 metric entries across 3 evaluation sources.
|
|
|
|
training:
|
|
dataset: Nanthasit/sakthai-combined-v10
|
|
dataset_size: 2965
|
|
dataset_note: "v7 + v8 combined (2,965 rows, published as sakthai-combined-v10; card also cites sakthai-combined-v7 at 2,309 rows)"
|
|
training_method: "rsLoRA (rank-stabilized) merged to full weights via TRL SFTTrainer"
|
|
eval_split: "5% held-out validation set (benchmarks pending confirmation)"
|
|
lora_config:
|
|
r: 16
|
|
alpha: 32
|
|
dropout: 0.05
|
|
target_modules: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj]
|
|
optimizer: "AdamW (8-bit)"
|
|
learning_rate: 0.0002
|
|
epochs: 3
|
|
precision: bf16
|
|
hardware: "Free T4 GPU (Kaggle / Colab / HF Inference), $0 budget"
|
|
framework: "TRL + Transformers"
|
|
key_improvements:
|
|
- "rsLoRA instead of standard LoRA — better rank utilization"
|
|
- "All 7 linear modules adapted (vs 4 in v1)"
|
|
- "Dropout reduced 0.1 to 0.05"
|
|
- "48% more training data (v7 + v8)"
|
|
- "Merged full-weight checkpoint — no PEFT dependency at inference"
|
|
|
|
card_quality:
|
|
license: apache-2.0
|
|
base_model_documented: true
|
|
base_model: Qwen/Qwen2.5-1.5B-Instruct
|
|
tags_count: 19
|
|
tags: [qwen2.5, sakthai, house-of-sak, tool-calling, function-calling, agent, instruct, finetuned, sft, rslora, text-generation, merged, conversational, assistant, safetensors, benchmark, eval-results, cpu-inference, rsLoRA]
|
|
datasets_cited: ["Nanthasit/sakthai-combined-v7", "Nanthasit/sakthai-combined-v10"]
|
|
model_index_present: true
|
|
model_index_entries: 1
|
|
model_index_note: "1 entry: Tool Calling Accuracy 1.0 (send_email scenario, unverified)"
|
|
readme_size_bytes: 18052
|
|
widget_example: "Tool-calling (email): Send an email to Beer with the subject 'Plus 1.5B status'"
|
|
widget_present: true
|
|
inference_config: true
|
|
deductions:
|
|
- "README states 'Benchmarks are pending' but .eval_results/ already carries 10 metric entries across 3 evaluation sources — card text lags repo state"
|
|
- "model-index has only 1 of 10 available metrics — 9 unindexed, so they don't render on the Hub widget"
|
|
- "lighteval scores (winogrande, gsm8k, hellaswag) and sakthai-bench-v2 scores all marked unverified — need multi-trial re-run"
|
|
score: 85
|
|
|
|
health_score:
|
|
overall: 64.2
|
|
components:
|
|
popularity: 2.4
|
|
momentum: 100.0
|
|
benchmarks: 50.0
|
|
card_quality: 85.0
|
|
repo_hygiene: 95.0
|
|
weights:
|
|
popularity: 0.20
|
|
momentum: 0.20
|
|
benchmarks: 0.25
|
|
card_quality: 0.20
|
|
repo_hygiene: 0.15
|
|
|
|
sibling_comparison:
|
|
rank_by_downloads: 13
|
|
total_author_models: 22
|
|
max_sibling_downloads: 1855
|
|
models_with_positive_downloads: 17
|
|
velocity_rank: 3
|
|
max_sibling_velocity: 330.4
|
|
our_velocity: 244.0
|
|
|
|
eval_type: metadata_cron
|
|
eval_note: >
|
|
Re-eval (run 27) for sakthai-plus-1.5b. Since first eval (18h ago):
|
|
0 to 244 downloads (+244), velocity 244.0/day (3rd fastest among 19
|
|
models). Health score improved from 38.2 to 64.2 (+26.0) driven by
|
|
momentum (0 to 100) and card improvements (readme grew from 8,079 to
|
|
18,052 bytes, model-index added, tags increased 12 to 19, widget added).
|
|
Card still claims 'Benchmarks are pending' despite 10 metric entries in
|
|
.eval_results/ — this is the most impactful improvement opportunity.
|
|
Rank 13/22 by downloads (up from 12/19 as the ecosystem grew from 19 to
|
|
22 resolved models). Next cycle: reconcile card text with .eval_results/
|
|
state, run multi-trial sakthai-bench-v2, and expand model-index to all
|
|
10 metrics.
|
|
|
|
eval_metadata:
|
|
model: Nanthasit/sakthai-plus-1.5b
|
|
eval_date: "2026-07-31"
|
|
eval_time: "23:00:00Z"
|
|
schema: llm_cron_v1
|
|
age_days: 1.00
|
|
days_since_last_update: 0.72
|
|
download_velocity: 244.0
|
|
cron_run: 27
|