adapt to sglang v0.5.2rc1 on dcu

2025-09-04 15:56:33 +08:00
commit 909abb58f5
2320 changed files with 489411 additions and 0 deletions
--- a/sgl-router/py_src/sglang_router/router.py
+++ b/sgl-router/py_src/sglang_router/router.py
@@ -0,0 +1,211 @@
+from typing import Dict, List, Optional
+
+from sglang_router_rs import PolicyType
+from sglang_router_rs import Router as _Router
+
+
+class Router:
+    """
+    A high-performance router for distributing requests across worker nodes.
+
+    Args:
+        worker_urls: List of URLs for worker nodes that will handle requests. Each URL should include
+            the protocol, host, and port (e.g., ['http://worker1:8000', 'http://worker2:8000'])
+        policy: Load balancing policy to use. Options:
+            - PolicyType.Random: Randomly select workers
+            - PolicyType.RoundRobin: Distribute requests in round-robin fashion
+            - PolicyType.CacheAware: Distribute requests based on cache state and load balance
+            - PolicyType.PowerOfTwo: Select best of two random workers based on load (PD mode only)
+        host: Host address to bind the router server. Default: '127.0.0.1'
+        port: Port number to bind the router server. Default: 3001
+        worker_startup_timeout_secs: Timeout in seconds for worker startup. Default: 300
+        worker_startup_check_interval: Interval in seconds between checks for worker initialization. Default: 10
+        cache_threshold: Cache threshold (0.0-1.0) for cache-aware routing. Routes to cached worker
+            if the match rate exceeds threshold, otherwise routes to the worker with the smallest
+            tree. Default: 0.5
+        balance_abs_threshold: Load balancing is triggered when (max_load - min_load) > abs_threshold
+            AND max_load > min_load * rel_threshold. Otherwise, use cache aware. Default: 32
+        balance_rel_threshold: Load balancing is triggered when (max_load - min_load) > abs_threshold
+            AND max_load > min_load * rel_threshold. Otherwise, use cache aware. Default: 1.0001
+        eviction_interval_secs: Interval in seconds between cache eviction operations in cache-aware
+            routing. Default: 60
+        max_payload_size: Maximum payload size in bytes. Default: 256MB
+        max_tree_size: Maximum size of the approximation tree for cache-aware routing. Default: 2^24
+        dp_aware: Enable data parallelism aware schedule. Default: False
+        api_key: The api key used for the authorization with the worker.
+            Useful when the dp aware scheduling strategy is enabled.
+            Default: None
+        log_dir: Directory to store log files. If None, logs are only output to console. Default: None
+        log_level: Logging level. Options: 'debug', 'info', 'warning', 'error', 'critical'.
+        service_discovery: Enable Kubernetes service discovery. When enabled, the router will
+            automatically discover worker pods based on the selector. Default: False
+        selector: Dictionary mapping of label keys to values for Kubernetes pod selection.
+            Example: {"app": "sglang-worker"}. Default: {}
+        service_discovery_port: Port to use for service discovery. The router will generate
+            worker URLs using this port. Default: 80
+        service_discovery_namespace: Kubernetes namespace to watch for pods. If not provided,
+            watches pods across all namespaces (requires cluster-wide permissions). Default: None
+        prefill_selector: Dictionary mapping of label keys to values for Kubernetes pod selection
+            for prefill servers (PD mode only). Default: {}
+        decode_selector: Dictionary mapping of label keys to values for Kubernetes pod selection
+            for decode servers (PD mode only). Default: {}
+        prometheus_port: Port to expose Prometheus metrics. Default: None
+        prometheus_host: Host address to bind the Prometheus metrics server. Default: None
+        pd_disaggregation: Enable PD (Prefill-Decode) disaggregated mode. Default: False
+        prefill_urls: List of (url, bootstrap_port) tuples for prefill servers (PD mode only)
+        decode_urls: List of URLs for decode servers (PD mode only)
+        prefill_policy: Specific load balancing policy for prefill nodes (PD mode only).
+            If not specified, uses the main policy. Default: None
+        decode_policy: Specific load balancing policy for decode nodes (PD mode only).
+            If not specified, uses the main policy. Default: None
+        request_id_headers: List of HTTP headers to check for request IDs. If not specified,
+            uses common defaults: ['x-request-id', 'x-correlation-id', 'x-trace-id', 'request-id'].
+            Example: ['x-my-request-id', 'x-custom-trace-id']. Default: None
+        bootstrap_port_annotation: Kubernetes annotation name for bootstrap port (PD mode).
+            Default: 'sglang.ai/bootstrap-port'
+        request_timeout_secs: Request timeout in seconds. Default: 600
+        max_concurrent_requests: Maximum number of concurrent requests allowed for rate limiting. Default: 256
+        queue_size: Queue size for pending requests when max concurrent limit reached (0 = no queue, return 429 immediately). Default: 100
+        queue_timeout_secs: Maximum time (in seconds) a request can wait in queue before timing out. Default: 60
+        rate_limit_tokens_per_second: Token bucket refill rate (tokens per second). If not set, defaults to max_concurrent_requests. Default: None
+        cors_allowed_origins: List of allowed origins for CORS. Empty list allows all origins. Default: []
+        health_failure_threshold: Number of consecutive health check failures before marking worker unhealthy. Default: 3
+        health_success_threshold: Number of consecutive health check successes before marking worker healthy. Default: 2
+        health_check_timeout_secs: Timeout in seconds for health check requests. Default: 5
+        health_check_interval_secs: Interval in seconds between runtime health checks. Default: 60
+        health_check_endpoint: Health check endpoint path. Default: '/health'
+        model_path: Model path for loading tokenizer (HuggingFace model ID or local path). Default: None
+        tokenizer_path: Explicit tokenizer path (overrides model_path tokenizer if provided). Default: None
+    """
+
+    def __init__(
+        self,
+        worker_urls: List[str],
+        policy: PolicyType = PolicyType.RoundRobin,
+        host: str = "127.0.0.1",
+        port: int = 3001,
+        worker_startup_timeout_secs: int = 600,
+        worker_startup_check_interval: int = 30,
+        cache_threshold: float = 0.3,
+        balance_abs_threshold: int = 64,
+        balance_rel_threshold: float = 1.5,
+        eviction_interval_secs: int = 120,
+        max_tree_size: int = 2**26,
+        max_payload_size: int = 512 * 1024 * 1024,  # 512MB
+        dp_aware: bool = False,
+        api_key: Optional[str] = None,
+        log_dir: Optional[str] = None,
+        log_level: Optional[str] = None,
+        service_discovery: bool = False,
+        selector: Dict[str, str] = None,
+        service_discovery_port: int = 80,
+        service_discovery_namespace: Optional[str] = None,
+        prefill_selector: Dict[str, str] = None,
+        decode_selector: Dict[str, str] = None,
+        bootstrap_port_annotation: str = "sglang.ai/bootstrap-port",
+        prometheus_port: Optional[int] = None,
+        prometheus_host: Optional[str] = None,
+        request_timeout_secs: int = 1800,
+        request_id_headers: Optional[List[str]] = None,
+        pd_disaggregation: bool = False,
+        prefill_urls: Optional[List[tuple]] = None,
+        decode_urls: Optional[List[str]] = None,
+        prefill_policy: Optional[PolicyType] = None,
+        decode_policy: Optional[PolicyType] = None,
+        max_concurrent_requests: int = 256,
+        queue_size: int = 100,
+        queue_timeout_secs: int = 60,
+        rate_limit_tokens_per_second: Optional[int] = None,
+        cors_allowed_origins: List[str] = None,
+        retry_max_retries: int = 5,
+        retry_initial_backoff_ms: int = 50,
+        retry_max_backoff_ms: int = 30_000,
+        retry_backoff_multiplier: float = 1.5,
+        retry_jitter_factor: float = 0.2,
+        cb_failure_threshold: int = 10,
+        cb_success_threshold: int = 3,
+        cb_timeout_duration_secs: int = 60,
+        cb_window_duration_secs: int = 120,
+        disable_retries: bool = False,
+        disable_circuit_breaker: bool = False,
+        health_failure_threshold: int = 3,
+        health_success_threshold: int = 2,
+        health_check_timeout_secs: int = 5,
+        health_check_interval_secs: int = 60,
+        health_check_endpoint: str = "/health",
+        model_path: Optional[str] = None,
+        tokenizer_path: Optional[str] = None,
+    ):
+        if selector is None:
+            selector = {}
+        if prefill_selector is None:
+            prefill_selector = {}
+        if decode_selector is None:
+            decode_selector = {}
+        if cors_allowed_origins is None:
+            cors_allowed_origins = []
+
+        self._router = _Router(
+            worker_urls=worker_urls,
+            policy=policy,
+            host=host,
+            port=port,
+            worker_startup_timeout_secs=worker_startup_timeout_secs,
+            worker_startup_check_interval=worker_startup_check_interval,
+            cache_threshold=cache_threshold,
+            balance_abs_threshold=balance_abs_threshold,
+            balance_rel_threshold=balance_rel_threshold,
+            eviction_interval_secs=eviction_interval_secs,
+            max_tree_size=max_tree_size,
+            max_payload_size=max_payload_size,
+            dp_aware=dp_aware,
+            api_key=api_key,
+            log_dir=log_dir,
+            log_level=log_level,
+            service_discovery=service_discovery,
+            selector=selector,
+            service_discovery_port=service_discovery_port,
+            service_discovery_namespace=service_discovery_namespace,
+            prefill_selector=prefill_selector,
+            decode_selector=decode_selector,
+            bootstrap_port_annotation=bootstrap_port_annotation,
+            prometheus_port=prometheus_port,
+            prometheus_host=prometheus_host,
+            request_timeout_secs=request_timeout_secs,
+            request_id_headers=request_id_headers,
+            pd_disaggregation=pd_disaggregation,
+            prefill_urls=prefill_urls,
+            decode_urls=decode_urls,
+            prefill_policy=prefill_policy,
+            decode_policy=decode_policy,
+            max_concurrent_requests=max_concurrent_requests,
+            queue_size=queue_size,
+            queue_timeout_secs=queue_timeout_secs,
+            rate_limit_tokens_per_second=rate_limit_tokens_per_second,
+            cors_allowed_origins=cors_allowed_origins,
+            retry_max_retries=retry_max_retries,
+            retry_initial_backoff_ms=retry_initial_backoff_ms,
+            retry_max_backoff_ms=retry_max_backoff_ms,
+            retry_backoff_multiplier=retry_backoff_multiplier,
+            retry_jitter_factor=retry_jitter_factor,
+            cb_failure_threshold=cb_failure_threshold,
+            cb_success_threshold=cb_success_threshold,
+            cb_timeout_duration_secs=cb_timeout_duration_secs,
+            cb_window_duration_secs=cb_window_duration_secs,
+            disable_retries=disable_retries,
+            disable_circuit_breaker=disable_circuit_breaker,
+            health_failure_threshold=health_failure_threshold,
+            health_success_threshold=health_success_threshold,
+            health_check_timeout_secs=health_check_timeout_secs,
+            health_check_interval_secs=health_check_interval_secs,
+            health_check_endpoint=health_check_endpoint,
+            model_path=model_path,
+            tokenizer_path=tokenizer_path,
+        )
+
+    def start(self) -> None:
+        """Start the router server.
+
+        This method blocks until the server is shut down.
+        """
+        self._router.start()