Support compressed tensors fp8w8a8 (#4743)

2025-03-27 04:21:25 +08:00
parent 45fdf1f7f3
commit 04e3ff6975
30 changed files with 2386 additions and 113 deletions
--- a/python/sglang/srt/layers/quantization/base_config.py
+++ b/python/sglang/srt/layers/quantization/base_config.py
@@ -38,6 +38,11 @@ class QuantizeMethodBase(ABC):
 class QuantizationConfig(ABC):
    """Base class for quantization configs."""

+    def __init__(self):
+        super().__init__()
+        # mapping is updated by models as they initialize
+        self.packed_modules_mapping: Dict[str, List[str]] = dict()
+
    @abstractmethod
    def get_name(self) -> str:
        """Name of the quantization method."""