94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
241 lines
7.9 KiB
Python
241 lines
7.9 KiB
Python
import unittest
|
|
|
|
from sglang.test.scripted_runtime.context import ScriptedContext
|
|
from sglang.test.scripted_runtime.test_case import ScriptedTestCase
|
|
from sglang.test.scripted_runtime_chunked_helpers import (
|
|
DEFAULT_CHUNK_SIZE,
|
|
VERY_LONG_PROMPT_LEN,
|
|
base_engine_kwargs,
|
|
run_until,
|
|
run_until_all_finished,
|
|
run_until_finished,
|
|
)
|
|
|
|
_LORA_BASE_MODEL = "meta-llama/Llama-3.2-1B-Instruct"
|
|
_LORA_ADAPTER = "codelion/Llama-3.2-1B-Instruct-tool-calling-lora"
|
|
_LORA_ADAPTER_B = "nicoboss/Llama-3.2-1B-Instruct-Uncensored-Lora"
|
|
|
|
|
|
class TestLoRASingleAdapter(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
model_path=_LORA_BASE_MODEL,
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_lora=True,
|
|
lora_paths=[_LORA_ADAPTER],
|
|
)
|
|
|
|
def test_naive_lora_chunked(self):
|
|
self.server.execute_script(self._script_naive_lora_chunked)
|
|
|
|
@staticmethod
|
|
def _script_naive_lora_chunked(t: ScriptedContext):
|
|
r = t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=4,
|
|
lora_path=_LORA_ADAPTER,
|
|
)
|
|
yield from run_until_finished(r)
|
|
assert r.finished
|
|
assert r.chunks_done >= 2
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
assert len(r.req.output_ids) == 4
|
|
|
|
def test_lora_logprob_chunked_pass_idx(self):
|
|
self.server.execute_script(self._script_lora_logprob_chunked_pass_idx)
|
|
|
|
@staticmethod
|
|
def _script_lora_logprob_chunked_pass_idx(t: ScriptedContext):
|
|
prompt_len: int = VERY_LONG_PROMPT_LEN
|
|
r = t.start_req(
|
|
prompt_len=prompt_len,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER,
|
|
return_logprob=True,
|
|
logprob_start_len=0,
|
|
)
|
|
yield from run_until_finished(r)
|
|
assert r.finished
|
|
assert r.chunks_done == prompt_len // DEFAULT_CHUNK_SIZE
|
|
assert len(r.req.logprob.input_token_logprobs_val) == prompt_len
|
|
|
|
|
|
class TestLoRADrainerBypass(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
model_path=_LORA_BASE_MODEL,
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_lora=True,
|
|
lora_paths=[_LORA_ADAPTER, _LORA_ADAPTER_B],
|
|
max_loras_per_batch=1,
|
|
)
|
|
|
|
def test_lora_drainer_does_not_block_chunked_resume(self):
|
|
self.server.execute_script(
|
|
self._script_lora_drainer_does_not_block_chunked_resume
|
|
)
|
|
|
|
@staticmethod
|
|
def _script_lora_drainer_does_not_block_chunked_resume(t: ScriptedContext):
|
|
r_a = t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER,
|
|
)
|
|
yield from run_until(r_a, lambda h: h.is_chunking and h.chunks_done >= 1)
|
|
chunks_before = r_a.chunks_done
|
|
|
|
r_b = t.start_req(
|
|
prompt_len=DEFAULT_CHUNK_SIZE // 2,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER_B,
|
|
)
|
|
|
|
for _ in range(200):
|
|
if r_a.chunks_done > chunks_before:
|
|
break
|
|
yield
|
|
else:
|
|
raise AssertionError(
|
|
f"chunked-resume r_a starved by LoRA drainer; "
|
|
f"chunks_done stuck at {chunks_before}"
|
|
)
|
|
|
|
yield from run_until_all_finished(handles=[r_a, r_b])
|
|
assert r_a.finished and r_b.finished
|
|
|
|
|
|
class TestLoRAAdapterSwitch(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
model_path=_LORA_BASE_MODEL,
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_lora=True,
|
|
lora_paths=[_LORA_ADAPTER, _LORA_ADAPTER_B],
|
|
max_loras_per_batch=2,
|
|
)
|
|
|
|
def test_lora_adapter_switch_mid_chunk(self):
|
|
self.server.execute_script(self._script_lora_adapter_switch_mid_chunk)
|
|
|
|
@staticmethod
|
|
def _script_lora_adapter_switch_mid_chunk(t: ScriptedContext):
|
|
r_a = t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER,
|
|
)
|
|
r_b = t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER_B,
|
|
)
|
|
yield from run_until_all_finished(handles=[r_a, r_b])
|
|
assert r_a.finished and r_b.finished
|
|
expected_chunks = VERY_LONG_PROMPT_LEN // DEFAULT_CHUNK_SIZE
|
|
assert r_a.chunks_done == expected_chunks
|
|
assert r_b.chunks_done == expected_chunks
|
|
|
|
|
|
class TestLoRAAllDistinctAdapters(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
model_path=_LORA_BASE_MODEL,
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_lora=True,
|
|
lora_paths=[_LORA_ADAPTER, _LORA_ADAPTER_B],
|
|
max_loras_per_batch=2,
|
|
max_loaded_loras=4,
|
|
)
|
|
|
|
def test_lora_all_distinct_adapters_chunked(self):
|
|
self.server.execute_script(self._script_lora_all_distinct_adapters_chunked)
|
|
|
|
@staticmethod
|
|
def _script_lora_all_distinct_adapters_chunked(t: ScriptedContext):
|
|
adapters = [_LORA_ADAPTER, _LORA_ADAPTER_B, _LORA_ADAPTER, _LORA_ADAPTER_B]
|
|
reqs = [
|
|
t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=2,
|
|
lora_path=adapter,
|
|
)
|
|
for adapter in adapters
|
|
]
|
|
yield from run_until_all_finished(handles=reqs, max_steps=2000)
|
|
assert all(r.finished for r in reqs)
|
|
for r in reqs:
|
|
assert r.kv_pages == 0
|
|
assert r.lock_refs == 0
|
|
expected_first_chunks = VERY_LONG_PROMPT_LEN // DEFAULT_CHUNK_SIZE
|
|
assert reqs[0].chunks_done == expected_first_chunks
|
|
assert reqs[1].chunks_done == expected_first_chunks
|
|
assert reqs[2].chunks_done < expected_first_chunks
|
|
assert reqs[3].chunks_done < expected_first_chunks
|
|
|
|
|
|
class TestLoRAAdapterEviction(ScriptedTestCase):
|
|
ENGINE_KWARGS = base_engine_kwargs(
|
|
model_path=_LORA_BASE_MODEL,
|
|
chunked_prefill_size=DEFAULT_CHUNK_SIZE,
|
|
enable_lora=True,
|
|
lora_paths=[_LORA_ADAPTER, _LORA_ADAPTER_B],
|
|
max_loras_per_batch=1,
|
|
max_loaded_loras=2,
|
|
)
|
|
|
|
def test_lora_adapter_eviction_between_chunks(self):
|
|
self.server.execute_script(self._script_lora_adapter_eviction_between_chunks)
|
|
|
|
@staticmethod
|
|
def _script_lora_adapter_eviction_between_chunks(t: ScriptedContext):
|
|
r_a = t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER,
|
|
)
|
|
yield from run_until(r_a, lambda h: h.is_chunking and h.chunks_done >= 1)
|
|
|
|
r_b = t.start_req(
|
|
prompt_len=DEFAULT_CHUNK_SIZE // 2,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER_B,
|
|
)
|
|
yield from run_until_all_finished(handles=[r_a, r_b], max_steps=800)
|
|
assert r_a.finished and r_b.finished
|
|
|
|
def test_lora_chunked_abort_during_eviction(self):
|
|
self.server.execute_script(self._script_lora_chunked_abort_during_eviction)
|
|
|
|
@staticmethod
|
|
def _script_lora_chunked_abort_during_eviction(t: ScriptedContext):
|
|
r_a = t.start_req(
|
|
prompt_len=VERY_LONG_PROMPT_LEN,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER,
|
|
)
|
|
yield from run_until(r_a, lambda h: h.is_chunking and h.chunks_done >= 1)
|
|
|
|
r_b = t.start_req(
|
|
prompt_len=DEFAULT_CHUNK_SIZE // 2,
|
|
max_new_tokens=2,
|
|
lora_path=_LORA_ADAPTER_B,
|
|
)
|
|
yield
|
|
|
|
t.abort(r_a)
|
|
for _ in range(12):
|
|
if r_a.kv_pages == 0 and (r_a.req is None or r_a.req.req_pool_idx is None):
|
|
break
|
|
yield
|
|
|
|
assert r_a.status in ("finished", "unknown")
|
|
if r_a.req is not None:
|
|
assert r_a.kv_pages == 0
|
|
assert r_a.lock_refs == 0
|
|
yield from run_until_finished(r_b)
|
|
assert r_b.finished
|
|
assert r_b.kv_pages == 0
|
|
assert r_b.lock_refs == 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|