94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
108 lines
3.3 KiB
Python
108 lines
3.3 KiB
Python
from __future__ import annotations
|
|
|
|
import time
|
|
from dataclasses import dataclass
|
|
from typing import (
|
|
TYPE_CHECKING,
|
|
Any,
|
|
Callable,
|
|
Optional,
|
|
)
|
|
|
|
import msgspec
|
|
import zmq
|
|
|
|
from sglang.srt.disaggregation.kv_events import (
|
|
EventPublisherFactory,
|
|
KVEventBatch,
|
|
select_kv_publisher_dp_rank,
|
|
)
|
|
from sglang.srt.managers.io_struct import hook_custom_types, sock_send
|
|
|
|
if TYPE_CHECKING:
|
|
from sglang.srt.distributed.parallel_state_wrapper import ParallelState
|
|
from sglang.srt.mem_cache.base_prefix_cache import BasePrefixCache
|
|
|
|
|
|
class SchedulerStats: ... # type: ignore[no-redef]
|
|
|
|
|
|
class KvMetrics(msgspec.Struct, tag=True, kw_only=True, array_like=True):
|
|
request_active_slots: int = 0
|
|
request_total_slots: int = 0
|
|
kv_active_blocks: int = 0
|
|
kv_total_blocks: int = 0
|
|
num_requests_waiting: int = 0
|
|
gpu_cache_usage_perc: float = 0.0
|
|
gpu_prefix_cache_hit_rate: float = 0.0
|
|
data_parallel_rank: int = 0
|
|
|
|
|
|
hook_custom_types(KvMetrics)
|
|
|
|
|
|
@dataclass(kw_only=True, slots=True)
|
|
class SchedulerKvEventsPublisher:
|
|
kv_events_config: Optional[str]
|
|
ps: ParallelState
|
|
attn_tp_rank: int
|
|
attn_cp_rank: int
|
|
attn_dp_rank: int
|
|
dp_rank: Optional[int]
|
|
tree_cache: BasePrefixCache
|
|
send_metrics_from_scheduler: Optional[zmq.Socket]
|
|
max_running_requests: int
|
|
max_total_num_tokens: int
|
|
get_stats: Callable
|
|
enable_kv_cache_events: bool = False
|
|
kv_event_publisher: Any = None
|
|
|
|
def __post_init__(self) -> None:
|
|
self.init_kv_events(self.kv_events_config)
|
|
|
|
def init_kv_events(self, kv_events_config: Optional[str]):
|
|
self.enable_kv_cache_events = bool(
|
|
kv_events_config
|
|
and self.ps.pp_rank == 0
|
|
and self.ps.attn_tp_rank == 0
|
|
and self.ps.attn_cp_rank == 0
|
|
)
|
|
|
|
if self.enable_kv_cache_events:
|
|
self.kv_event_publisher = EventPublisherFactory.create(
|
|
kv_events_config,
|
|
select_kv_publisher_dp_rank(
|
|
self.ps.attn_dp_size, self.ps.attn_dp_rank, self.ps.dp_rank
|
|
),
|
|
)
|
|
|
|
def emit_kv_metrics(self):
|
|
if not self.enable_kv_cache_events:
|
|
return
|
|
|
|
kv_metrics = KvMetrics()
|
|
kv_metrics.request_active_slots = self.get_stats().num_running_reqs.total
|
|
kv_metrics.request_total_slots = self.max_running_requests
|
|
kv_metrics.kv_active_blocks = int(
|
|
self.get_stats().token_usage * self.max_total_num_tokens
|
|
)
|
|
kv_metrics.kv_total_blocks = self.max_total_num_tokens
|
|
kv_metrics.num_requests_waiting = self.get_stats().num_queue_reqs.total
|
|
kv_metrics.gpu_cache_usage_perc = self.get_stats().token_usage
|
|
kv_metrics.gpu_prefix_cache_hit_rate = self.get_stats().cache_hit_rate
|
|
kv_metrics.data_parallel_rank = (
|
|
self.ps.dp_rank if self.ps.dp_rank is not None else 0
|
|
)
|
|
|
|
if not self.send_metrics_from_scheduler.closed:
|
|
sock_send(self.send_metrics_from_scheduler, kv_metrics)
|
|
|
|
def publish_kv_events(self):
|
|
if not self.enable_kv_cache_events:
|
|
return
|
|
|
|
events = self.tree_cache.take_events()
|
|
if events:
|
|
batch = KVEventBatch(ts=time.time(), events=events)
|
|
self.kv_event_publisher.publish(batch)
|