109 lines
2.4 KiB
YAML
109 lines
2.4 KiB
YAML
eval_id: cron-eval-sakthai-plus-1.5b-20260801T114529Z
|
|
model_id: Nanthasit/sakthai-plus-1.5b
|
|
timestamp: "2026-08-01T11:45:29Z"
|
|
result_type: metadata
|
|
source: metadata_cron
|
|
status: uploaded
|
|
|
|
model_meta:
|
|
pipeline_tag: text-generation
|
|
library_name: transformers
|
|
base_model: Qwen/Qwen2.5-1.5B-Instruct
|
|
license: apache-2.0
|
|
tags:
|
|
- transformers
|
|
- safetensors
|
|
- qwen2.5
|
|
- sakthai
|
|
- house-of-sak
|
|
- tool-calling
|
|
- function-calling
|
|
- agent
|
|
- instruct
|
|
- finetuned
|
|
- sft
|
|
- merged
|
|
- conversational
|
|
- assistant
|
|
- cpu-inference
|
|
- rsLoRA
|
|
- benchmark
|
|
- eval-results
|
|
- llama-cpp
|
|
datasets:
|
|
- Nanthasit/sakthai-combined-v11
|
|
- Nanthasit/SimpleToolCalling
|
|
repo_type: model
|
|
sha: 412d38d60e1727a6586d0e7f5b30426b557824ab
|
|
last_modified: "2026-08-01T11:45:32Z"
|
|
|
|
metrics:
|
|
downloads: 297
|
|
likes: 0
|
|
|
|
model_index:
|
|
- task:
|
|
type: text-generation
|
|
name: Tool-Calling Accuracy
|
|
dataset:
|
|
name: llama.cpp tool-calling (3-trial, q4_k_m)
|
|
type: custom
|
|
metrics:
|
|
- name: Tool Call Success Rate
|
|
type: tool_call_success
|
|
value: 1
|
|
verified: false
|
|
- name: Valid JSON Arguments
|
|
type: valid-json
|
|
value: 1
|
|
verified: false
|
|
- name: Correct Answer Rate
|
|
type: correct-answer
|
|
value: 1
|
|
verified: false
|
|
- name: Selection Accuracy
|
|
type: selection-accuracy
|
|
value: 84.8
|
|
verified: false
|
|
- name: Arguments Accuracy
|
|
type: arguments-accuracy
|
|
value: 33.7
|
|
verified: false
|
|
- name: Strict Accuracy
|
|
type: strict-accuracy
|
|
value: 33.7
|
|
verified: false
|
|
- task:
|
|
type: text-generation
|
|
name: Commonsense Reasoning
|
|
dataset:
|
|
name: lighteval
|
|
type: lighteval
|
|
metrics:
|
|
- name: WinoGrande (WSC)
|
|
type: winogrande
|
|
value: 59.6
|
|
verified: false
|
|
- name: HellaSwag
|
|
type: hellaswag
|
|
value: 34.0
|
|
verified: false
|
|
- name: GSM8K
|
|
type: gsm8k
|
|
value: 50.9
|
|
verified: false
|
|
|
|
config_highlights:
|
|
inference_parameters:
|
|
temperature: 0.3
|
|
max_new_tokens: 256
|
|
top_p: 0.9
|
|
chat_template: Qwen2.5 tool-calling style with XML <tool_call> blocks
|
|
quantization_notes: GGUF/compatible; cpu-inference capable
|
|
adapter_only: false
|
|
requires_base: false
|
|
|
|
health:
|
|
verified: false
|
|
notes: metadata-only snapshot; no live inference.
|