初始化项目,由ModelHub XC社区提供模型
Model: Nanthasit/sakthai-plus-1.5b Source: Original Platform
This commit is contained in:
26
.eval_results/inference-check-20260730T231504Z.yaml
Normal file
26
.eval_results/inference-check-20260730T231504Z.yaml
Normal file
@@ -0,0 +1,26 @@
|
||||
check:
|
||||
timestamp: 20260730T231504Z
|
||||
model: Nanthasit/sakthai-plus-1.5b
|
||||
method: HF Inference API (router.huggingface.co)
|
||||
endpoint: https://router.huggingface.co/hf-inference/models/Nanthasit/sakthai-plus-1.5b
|
||||
results:
|
||||
status: not_available
|
||||
http_code: 400
|
||||
response_time_seconds: 0.13
|
||||
error: "Model not supported by provider hf-inference"
|
||||
details: "The model is a Qwen2.5-1.5B based transformer (BF16 safetensors, 2.9GB) not deployed on any HF Inference provider. Serverless inference does not serve this model."
|
||||
alternative_attempts:
|
||||
- method: "huggingface_hub InferenceClient.chat_completion"
|
||||
status: "model_not_supported"
|
||||
error: "The requested model 'Nanthasit/sakthai-plus-1.5b' is not supported by any provider you have enabled."
|
||||
- method: "Local transformers (BF16, full precision)"
|
||||
status: "OOM (exit 137)"
|
||||
detail: "Environment has 7.8GB RAM total, 1.4GB available. Model requires ~3GB+ for weights."
|
||||
- method: "Local transformers (4-bit quantization)"
|
||||
status: "OOM (exit 137)"
|
||||
detail: "Even 4-bit quantization failed due to insufficient memory."
|
||||
recommendations:
|
||||
- "Convert model to GGUF format for llama.cpp inference (much lower memory footprint)"
|
||||
- "Deploy on HF Inference Endpoints (requires paid GPU)"
|
||||
- "Run on a machine with ≥8GB free RAM for CPU inference"
|
||||
- "Enable the model for serverless inference via HF provider onboarding"
|
||||
Reference in New Issue
Block a user