27 lines
1.4 KiB
YAML
27 lines
1.4 KiB
YAML
check:
|
|
timestamp: 20260730T231504Z
|
|
model: Nanthasit/sakthai-plus-1.5b
|
|
method: HF Inference API (router.huggingface.co)
|
|
endpoint: https://router.huggingface.co/hf-inference/models/Nanthasit/sakthai-plus-1.5b
|
|
results:
|
|
status: not_available
|
|
http_code: 400
|
|
response_time_seconds: 0.13
|
|
error: "Model not supported by provider hf-inference"
|
|
details: "The model is a Qwen2.5-1.5B based transformer (BF16 safetensors, 2.9GB) not deployed on any HF Inference provider. Serverless inference does not serve this model."
|
|
alternative_attempts:
|
|
- method: "huggingface_hub InferenceClient.chat_completion"
|
|
status: "model_not_supported"
|
|
error: "The requested model 'Nanthasit/sakthai-plus-1.5b' is not supported by any provider you have enabled."
|
|
- method: "Local transformers (BF16, full precision)"
|
|
status: "OOM (exit 137)"
|
|
detail: "Environment has 7.8GB RAM total, 1.4GB available. Model requires ~3GB+ for weights."
|
|
- method: "Local transformers (4-bit quantization)"
|
|
status: "OOM (exit 137)"
|
|
detail: "Even 4-bit quantization failed due to insufficient memory."
|
|
recommendations:
|
|
- "Convert model to GGUF format for llama.cpp inference (much lower memory footprint)"
|
|
- "Deploy on HF Inference Endpoints (requires paid GPU)"
|
|
- "Run on a machine with ≥8GB free RAM for CPU inference"
|
|
- "Enable the model for serverless inference via HF provider onboarding"
|