target_model: id: Nanthasit/sakthai-plus-1.5b pipeline_tag: text-generation library_name: transformers base_model: Qwen/Qwen2.5-1.5B-Instruct downloads: 244 likes: 0 private: false gated: false last_modified: "2026-07-31T11:09:05.000Z" model_age_days: 1.00 model_type: llm has_weights: true architecture: base_model_type: qwen2 base_architectures: ["Qwen2ForCausalLM"] base_hidden_size: 1536 base_num_hidden_layers: 28 base_num_attention_heads: 12 base_num_key_value_heads: 2 base_intermediate_size: 8960 base_vocab_size: 151936 base_max_position_embeddings: 32768 base_total_parameters: 1540000000 base_dtype: bfloat16 tie_word_embeddings: true repo_summary: siblings_count: 20 total_repo_bytes: 3098940902 total_gb: 2.886 has_weights: true weight_file_count: 1 weight_bytes: 3087467144 weight_files: ["model.safetensors"] config_present: true tokenizer_present: true chat_template_present: true readme_present: true readme_size_bytes: 18052 weight_note: "Single merged full-weight checkpoint (2.88 GiB, bf16) — no PEFT dependency at inference. Standard Qwen2.5 layout: config.json, generation_config.json, tokenizer.json, tokenizer_config.json, chat_template.jinja, README.md." benchmarks: model_index_count: 1 metrics_count: 9 all_verified: false pending_metrics: 8 entries: - dataset: lighteval metric: winogrande value: 59.6 verified: false note: "entry in .eval_results/lighteval.yaml — not re-run, unverified" - dataset: lighteval metric: gsm8k value: 50.9 verified: false note: "entry in .eval_results/lighteval.yaml — not re-run, unverified" - dataset: lighteval metric: hellaswag value: 34.0 verified: false note: "entry in .eval_results/lighteval.yaml — not re-run, unverified" - dataset: sakthai-bench-v2 metric: selection_accuracy value: 84.8 verified: false note: "entry in .eval_results/sakthai-bench-v2.yaml — not re-run, unverified" - dataset: sakthai-bench-v2 metric: arguments_accuracy value: 33.7 verified: false note: "entry in .eval_results/sakthai-bench-v2.yaml — not re-run, unverified" - dataset: sakthai-bench-v2 metric: strict_accuracy value: 33.7 verified: false note: "entry in .eval_results/sakthai-bench-v2.yaml — not re-run, unverified" - dataset: llama.cpp tool-calling (3-trial, q4_k_m) metric: tool_call_success value: 1.0 verified: true note: ".eval_results/benchmark-20260731_043634.yaml — send_email prompt, 3/3 tool calls" - dataset: llama.cpp tool-calling (3-trial, q4_k_m) metric: valid_json value: 1.0 verified: true note: ".eval_results/benchmark-20260731_043634.yaml — 3/3 valid JSON tool args" - dataset: llama.cpp tool-calling (3-trial, q4_k_m) metric: correct_answer value: 1.0 verified: true note: ".eval_results/benchmark-20260731_043634.yaml — 3/3 correct answers, avg 20.7 tps (CPU, 2 threads)" - dataset: sakthai-bench-v2 (model-index) metric: tool_calling_accuracy value: 1.0 verified: false note: "model-index entry: 'Tool Calling Accuracy (3/3)' from send_email scenario — single-trial, unverified" notes: > README frontmatter now includes a model-index entry (Tool Calling Accuracy 1.0 on send_email scenario, unverified). 3 additional unverified lighteval scores (winogrande 59.6, gsm8k 50.9, hellaswag 34.0) and 3 sakthai-bench-v2 scores (selection 84.8, arguments 33.7, strict 33.7) exist in .eval_results/ but are NOT in the model-index. One verified 3-trial llama.cpp q4_k_m tool-calling run (3/3, 20.7 tps) is present. Card text still says 'Benchmarks are pending' — stale relative to the 10 metric entries across 3 evaluation sources. training: dataset: Nanthasit/sakthai-combined-v10 dataset_size: 2965 dataset_note: "v7 + v8 combined (2,965 rows, published as sakthai-combined-v10; card also cites sakthai-combined-v7 at 2,309 rows)" training_method: "rsLoRA (rank-stabilized) merged to full weights via TRL SFTTrainer" eval_split: "5% held-out validation set (benchmarks pending confirmation)" lora_config: r: 16 alpha: 32 dropout: 0.05 target_modules: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj] optimizer: "AdamW (8-bit)" learning_rate: 0.0002 epochs: 3 precision: bf16 hardware: "Free T4 GPU (Kaggle / Colab / HF Inference), $0 budget" framework: "TRL + Transformers" key_improvements: - "rsLoRA instead of standard LoRA — better rank utilization" - "All 7 linear modules adapted (vs 4 in v1)" - "Dropout reduced 0.1 to 0.05" - "48% more training data (v7 + v8)" - "Merged full-weight checkpoint — no PEFT dependency at inference" card_quality: license: apache-2.0 base_model_documented: true base_model: Qwen/Qwen2.5-1.5B-Instruct tags_count: 19 tags: [qwen2.5, sakthai, house-of-sak, tool-calling, function-calling, agent, instruct, finetuned, sft, rslora, text-generation, merged, conversational, assistant, safetensors, benchmark, eval-results, cpu-inference, rsLoRA] datasets_cited: ["Nanthasit/sakthai-combined-v7", "Nanthasit/sakthai-combined-v10"] model_index_present: true model_index_entries: 1 model_index_note: "1 entry: Tool Calling Accuracy 1.0 (send_email scenario, unverified)" readme_size_bytes: 18052 widget_example: "Tool-calling (email): Send an email to Beer with the subject 'Plus 1.5B status'" widget_present: true inference_config: true deductions: - "README states 'Benchmarks are pending' but .eval_results/ already carries 10 metric entries across 3 evaluation sources — card text lags repo state" - "model-index has only 1 of 10 available metrics — 9 unindexed, so they don't render on the Hub widget" - "lighteval scores (winogrande, gsm8k, hellaswag) and sakthai-bench-v2 scores all marked unverified — need multi-trial re-run" score: 85 health_score: overall: 64.2 components: popularity: 2.4 momentum: 100.0 benchmarks: 50.0 card_quality: 85.0 repo_hygiene: 95.0 weights: popularity: 0.20 momentum: 0.20 benchmarks: 0.25 card_quality: 0.20 repo_hygiene: 0.15 sibling_comparison: rank_by_downloads: 13 total_author_models: 22 max_sibling_downloads: 1855 models_with_positive_downloads: 17 velocity_rank: 3 max_sibling_velocity: 330.4 our_velocity: 244.0 eval_type: metadata_cron eval_note: > Re-eval (run 27) for sakthai-plus-1.5b. Since first eval (18h ago): 0 to 244 downloads (+244), velocity 244.0/day (3rd fastest among 19 models). Health score improved from 38.2 to 64.2 (+26.0) driven by momentum (0 to 100) and card improvements (readme grew from 8,079 to 18,052 bytes, model-index added, tags increased 12 to 19, widget added). Card still claims 'Benchmarks are pending' despite 10 metric entries in .eval_results/ — this is the most impactful improvement opportunity. Rank 13/22 by downloads (up from 12/19 as the ecosystem grew from 19 to 22 resolved models). Next cycle: reconcile card text with .eval_results/ state, run multi-trial sakthai-bench-v2, and expand model-index to all 10 metrics. eval_metadata: model: Nanthasit/sakthai-plus-1.5b eval_date: "2026-07-31" eval_time: "23:00:00Z" schema: llm_cron_v1 age_days: 1.00 days_since_last_update: 0.72 download_velocity: 244.0 cron_run: 27