check: timestamp: 20260730T231504Z model: Nanthasit/sakthai-plus-1.5b method: HF Inference API (router.huggingface.co) endpoint: https://router.huggingface.co/hf-inference/models/Nanthasit/sakthai-plus-1.5b results: status: not_available http_code: 400 response_time_seconds: 0.13 error: "Model not supported by provider hf-inference" details: "The model is a Qwen2.5-1.5B based transformer (BF16 safetensors, 2.9GB) not deployed on any HF Inference provider. Serverless inference does not serve this model." alternative_attempts: - method: "huggingface_hub InferenceClient.chat_completion" status: "model_not_supported" error: "The requested model 'Nanthasit/sakthai-plus-1.5b' is not supported by any provider you have enabled." - method: "Local transformers (BF16, full precision)" status: "OOM (exit 137)" detail: "Environment has 7.8GB RAM total, 1.4GB available. Model requires ~3GB+ for weights." - method: "Local transformers (4-bit quantization)" status: "OOM (exit 137)" detail: "Even 4-bit quantization failed due to insufficient memory." recommendations: - "Convert model to GGUF format for llama.cpp inference (much lower memory footprint)" - "Deploy on HF Inference Endpoints (requires paid GPU)" - "Run on a machine with ≥8GB free RAM for CPU inference" - "Enable the model for serverless inference via HF provider onboarding"