94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
310 lines
11 KiB
Python
310 lines
11 KiB
Python
import unittest
|
|
|
|
from sglang.test.scripted_runtime.context import ScriptedContext
|
|
from sglang.test.scripted_runtime.test_case import ScriptedTestCase
|
|
from sglang.test.scripted_runtime_chunked_helpers import (
|
|
BALLAST_MAX_NEW_TOKENS,
|
|
DEFAULT_CHUNK_SIZE,
|
|
VERY_LONG_PROMPT_LEN,
|
|
base_engine_kwargs,
|
|
run_until,
|
|
run_until_all_finished,
|
|
run_until_finished,
|
|
)
|
|
|
|
|
|
class TestPriorityBasic(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(chunked_prefill_size=DEFAULT_CHUNK_SIZE)
|
|
|
|
def test_retract_mid_chunk_releases_kv(self):
|
|
self.server.execute_script(self._script_retract_mid_chunk_releases_kv)
|
|
|
|
@staticmethod
|
|
def _script_retract_mid_chunk_releases_kv(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.is_chunking and h.chunks_done >= 1)
|
|
|
|
pages_before = r.kv_pages
|
|
assert pages_before > 0
|
|
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
|
|
assert (
|
|
r.status == "waiting"
|
|
), f"force-retracted chunked req must be back in waiting; got {r.status}"
|
|
assert r.kv_pages == 0, f"retract must release KV; got {r.kv_pages}"
|
|
|
|
t.continue_generation()
|
|
yield from run_until_finished(r)
|
|
assert r.finished
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
|
|
def test_retract_and_resume(self):
|
|
self.server.execute_script(self._script_retract_and_resume)
|
|
|
|
@staticmethod
|
|
def _script_retract_and_resume(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.is_chunking and h.chunks_done >= 1)
|
|
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.status == "waiting"
|
|
assert r.kv_pages == 0
|
|
|
|
t.continue_generation()
|
|
yield from run_until_finished(r)
|
|
assert r.finished
|
|
|
|
def test_force_retract_at_chunk_0(self):
|
|
self.server.execute_script(self._script_force_retract_at_chunk_0)
|
|
|
|
@staticmethod
|
|
def _script_force_retract_at_chunk_0(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.is_chunking and h.chunks_done <= 1)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0
|
|
t.continue_generation()
|
|
yield from run_until_finished(r, max_steps=800)
|
|
assert r.finished
|
|
|
|
def test_force_retract_at_chunk_mid(self):
|
|
self.server.execute_script(self._script_force_retract_at_chunk_mid)
|
|
|
|
@staticmethod
|
|
def _script_force_retract_at_chunk_mid(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.chunks_done >= 2 and h.is_chunking)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0
|
|
t.continue_generation()
|
|
yield from run_until_finished(r, max_steps=800)
|
|
assert r.finished
|
|
assert r.lock_refs == 0
|
|
|
|
def test_force_retract_at_last_chunk(self):
|
|
self.server.execute_script(self._script_force_retract_at_last_chunk)
|
|
|
|
@staticmethod
|
|
def _script_force_retract_at_last_chunk(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=2 * DEFAULT_CHUNK_SIZE, max_new_tokens=4)
|
|
yield from run_until(r, lambda h: h.chunks_done >= 1 and h.is_chunking)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0
|
|
t.continue_generation()
|
|
yield from run_until_finished(r, max_steps=800)
|
|
assert r.finished
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
|
|
def test_force_retract_then_readmit(self):
|
|
self.server.execute_script(self._script_force_retract_then_readmit)
|
|
|
|
@staticmethod
|
|
def _script_force_retract_then_readmit(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.is_chunking)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0, "retract must release KV before re-admission"
|
|
t.continue_generation()
|
|
yield from run_until_finished(r, max_steps=800)
|
|
assert r.finished
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
|
|
def test_retract_one_admit_one(self):
|
|
self.server.execute_script(self._script_retract_one_admit_one)
|
|
|
|
@staticmethod
|
|
def _script_retract_one_admit_one(t: ScriptedContext):
|
|
r1 = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r1, lambda h: h.is_chunking)
|
|
r2 = t.start_req(prompt_len=8, max_new_tokens=2)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
t.continue_generation()
|
|
yield from run_until_finished(r2)
|
|
assert r2.finished
|
|
assert r2.kv_pages == 0
|
|
yield from run_until_finished(r1)
|
|
assert r1.finished
|
|
assert r1.kv_pages == 0
|
|
assert r1.lock_refs == 0
|
|
|
|
def test_retract_during_decode(self):
|
|
self.server.execute_script(self._script_retract_during_decode)
|
|
|
|
@staticmethod
|
|
def _script_retract_during_decode(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=8, max_new_tokens=32)
|
|
yield from run_until(r, lambda h: h.status == "running")
|
|
assert r.kv_pages > 0, "decode-state req must own KV before retract"
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0, f"retract must release KV; got {r.kv_pages}"
|
|
t.continue_generation()
|
|
yield from run_until_finished(r)
|
|
assert r.finished
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
|
|
def test_retract_then_abort_idempotent(self):
|
|
self.server.execute_script(self._script_retract_then_abort_idempotent)
|
|
|
|
@staticmethod
|
|
def _script_retract_then_abort_idempotent(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.is_chunking)
|
|
t.pause_generation(mode="retract")
|
|
t.abort(r)
|
|
for _ in range(12):
|
|
if (
|
|
r.kv_pages == 0
|
|
and r.lock_refs == 0
|
|
and (r.req is None or r.req.req_pool_idx is None)
|
|
):
|
|
break
|
|
yield
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
assert r.req is None or r.req.req_pool_idx is None
|
|
t.continue_generation()
|
|
yield
|
|
assert r.kv_pages == 0 and r.lock_refs == 0
|
|
|
|
def test_retract_chunked_resume_in_waiting(self):
|
|
self.server.execute_script(self._script_retract_chunked_resume_in_waiting)
|
|
|
|
@staticmethod
|
|
def _script_retract_chunked_resume_in_waiting(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.is_chunking)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0
|
|
assert r.status == "waiting"
|
|
t.continue_generation()
|
|
yield from run_until_finished(r, max_steps=800)
|
|
assert r.finished
|
|
|
|
def test_two_retracts_same_yield(self):
|
|
self.server.execute_script(self._script_two_retracts_same_yield)
|
|
|
|
@staticmethod
|
|
def _script_two_retracts_same_yield(t: ScriptedContext):
|
|
r1 = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
r2 = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=2)
|
|
yield from run_until(r1, lambda h: h.is_chunking)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r1.kv_pages == 0
|
|
assert r2.kv_pages == 0
|
|
t.continue_generation()
|
|
yield from run_until_all_finished([r1, r2])
|
|
assert r1.finished and r2.finished
|
|
assert r1.lock_refs == 0
|
|
assert r2.lock_refs == 0
|
|
|
|
def test_retract_then_re_chunk(self):
|
|
self.server.execute_script(self._script_retract_then_re_chunk)
|
|
|
|
@staticmethod
|
|
def _script_retract_then_re_chunk(t: ScriptedContext):
|
|
r = t.start_req(prompt_len=2 * DEFAULT_CHUNK_SIZE, max_new_tokens=2)
|
|
yield from run_until(r, lambda h: h.chunks_done >= 1)
|
|
t.pause_generation(mode="retract")
|
|
yield
|
|
assert r.kv_pages == 0, "retract must release KV"
|
|
t.continue_generation()
|
|
yield from run_until_finished(r, max_steps=800)
|
|
assert r.finished
|
|
assert r.lock_refs == 0
|
|
assert r.kv_pages == 0
|
|
|
|
|
|
class TestPriorityPriority(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_priority_scheduling=True,
|
|
)
|
|
|
|
def test_naive_priority_chunked(self):
|
|
self.server.execute_script(self._script_naive_priority_chunked)
|
|
|
|
@staticmethod
|
|
def _script_naive_priority_chunked(t: ScriptedContext):
|
|
low = t.start_req(prompt_len=VERY_LONG_PROMPT_LEN, max_new_tokens=4, priority=0)
|
|
yield from run_until(low, lambda h: h.is_chunking)
|
|
|
|
high = t.start_req(prompt_len=8, max_new_tokens=2, priority=10)
|
|
|
|
yield from run_until_finished(high)
|
|
assert high.finished
|
|
assert not low.finished
|
|
|
|
yield from run_until_all_finished([low, high])
|
|
assert low.finished and high.finished
|
|
|
|
|
|
class TestPriorityPreempt(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_priority_scheduling=True,
|
|
max_running_requests=1,
|
|
priority_scheduling_preemption_threshold=0,
|
|
)
|
|
|
|
def test_priority_preempt_decode_victim_to_waiting(self):
|
|
self.server.execute_script(
|
|
self._script_priority_preempt_decode_victim_to_waiting
|
|
)
|
|
|
|
@staticmethod
|
|
def _script_priority_preempt_decode_victim_to_waiting(t: ScriptedContext):
|
|
low = t.start_req(
|
|
prompt_len=8,
|
|
max_new_tokens=BALLAST_MAX_NEW_TOKENS,
|
|
priority=0,
|
|
ignore_eos=True,
|
|
)
|
|
yield from run_until(low, lambda h: h.status == "running")
|
|
assert low.kv_pages > 0
|
|
|
|
high = t.start_req(prompt_len=8, max_new_tokens=2, priority=10)
|
|
yield from run_until(low, lambda h: h.status == "waiting")
|
|
|
|
assert low.status == "waiting"
|
|
assert low.kv_pages == 0
|
|
yield from run_until_finished(high)
|
|
assert high.finished
|
|
|
|
def test_priority_preempt_release_invariant(self):
|
|
self.server.execute_script(self._script_priority_preempt_release_invariant)
|
|
|
|
@staticmethod
|
|
def _script_priority_preempt_release_invariant(t: ScriptedContext):
|
|
r_low = t.start_req(
|
|
prompt_len=8,
|
|
max_new_tokens=BALLAST_MAX_NEW_TOKENS,
|
|
priority=0,
|
|
ignore_eos=True,
|
|
)
|
|
yield from run_until(r_low, lambda h: h.status == "running")
|
|
pages_before = r_low.kv_pages
|
|
assert pages_before > 0
|
|
|
|
r_high = t.start_req(prompt_len=8, max_new_tokens=2, priority=10)
|
|
yield from run_until(r_low, lambda h: h.status == "waiting")
|
|
assert r_low.kv_pages == 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|