[FEAT] Support GGUF format (#2215)

Co-authored-by: Yang Zheng(SW)(Alex) <you@example.com>
2024-11-30 16:44:48 +08:00
parent 0d6a49bd7d
commit 883c955489
39 changed files with 180 additions and 89 deletions
--- a/test/srt/run_suite.py
+++ b/test/srt/run_suite.py
@@ -15,6 +15,7 @@ suites = {
        "test_double_sparsity.py",
        "test_embedding_openai_server.py",
        "test_eval_accuracy_mini.py",
+        "test_gguf.py",
        "test_input_embeddings.py",
        "test_json_constrained.py",
        "test_large_max_new_tokens.py",
--- a/test/srt/test_gguf.py
+++ b/test/srt/test_gguf.py
@@ -0,0 +1,26 @@
+import unittest
+
+from huggingface_hub import hf_hub_download
+
+import sglang as sgl
+
+
+class TestGGUF(unittest.TestCase):
+    def test_models(self):
+        prompt = "Today is a sunny day and I like"
+        sampling_params = {"temperature": 0, "max_new_tokens": 8}
+
+        model_path = hf_hub_download(
+            "Qwen/Qwen2-1.5B-Instruct-GGUF",
+            filename="qwen2-1_5b-instruct-q4_k_m.gguf",
+        )
+
+        engine = sgl.Engine(model_path=model_path, random_seed=42)
+        outputs = engine.generate(prompt, sampling_params)["text"]
+        engine.shutdown()
+
+        self.assertEqual(outputs, " it. I have a lot of work")
+
+
+if __name__ == "__main__":
+    unittest.main()