_timestamp: '2026-07-31T21:09:29.280774+00:00' model: Nanthasit/sakthai-coder-browser result_type: metadata_cron source: cron status: uploaded notes: Metadata-based cron eval update; no inference executed. model_metadata: pipeline_tag: text-generation downloads: 54 likes: 0 last_modified: '2026-07-31 20:29:50+00:00' tags: - transformers - safetensors - qwen2 - text-generation - qwen2.5 - qwen2.5-coder - sakthai - house-of-sak - browser-automation - web-agent - tool-calling - function-calling - tool-use - agent - code-generation - finetuned - finetune - sft - merged - conversational card_data: base_model: Qwen/Qwen2.5-Coder-1.5B-Instruct datasets: - Nanthasit/sakthai-combined-v8 - Nanthasit/sakthai-combined-v9 - Nanthasit/sakthai-combined-v10 - Nanthasit/sakthai-combined-v11 - Nanthasit/sakthai-irrelevance-supplement - Nanthasit/cycle-bench language: - en library_name: transformers license: apache-2.0 pipeline_tag: text-generation tags: - qwen2.5 - qwen2.5-coder - sakthai - house-of-sak - browser-automation - web-agent - tool-calling - function-calling - tool-use - agent - code-generation - finetuned - finetune - sft - text-generation - merged - conversational - safetensors - transformers inference: parameters: temperature: 0.3 max_new_tokens: 256 top_p: 0.9 widget: - text: Search for the latest AI news and summarize the top story. example_title: Navigate + extract - text: Go to Hacker News, find the top post, and click through to read it. example_title: Multi-step navigation - text: Open google.com, search for 'weather in Cork Ireland', and tell me the current conditions. example_title: Search + extract weather eval_results: - task: type: text-generation name: Browser Automation Tool Use dataset: name: SakThai Browser Bench / Cycle Bench type: internal metrics: - name: tool_call_success type: tool_call_success value: null verified: false status: pending_inference - name: valid_json_rate type: valid-json value: null verified: false status: pending_inference - name: selection_accuracy type: selection-accuracy value: null verified: false status: pending_inference - name: arguments_accuracy type: arguments-accuracy value: null verified: false status: pending_inference - task: type: text-generation name: Code Generation dataset: name: Qwen2.5-Coder benchmarks type: upstream_reference metrics: - name: humaneval_pass1 type: humaneval value: null verified: false status: upstream_reference_pending - name: mbpp_pass1 type: mbpp value: null verified: false status: upstream_reference_pending - name: livecodebench_pass1 type: livecodebench value: null verified: false status: upstream_reference_pending health: recommendation: Run multi-trial browser bench with correct prompt format before publishing metrics.