sgl-project--sglang
94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
270 行
8.3 KiB
Python
270 行
8.3 KiB
Python
import asyncio
|
|
import os
|
|
import shutil
|
|
import tempfile
|
|
import time
|
|
import unittest
|
|
from types import SimpleNamespace
|
|
|
|
import requests
|
|
|
|
from sglang.benchmark.datasets.random import sample_random_requests
|
|
from sglang.benchmark.utils import get_tokenizer
|
|
from sglang.test.ci.ci_register import register_cuda_ci
|
|
from sglang.test.kits.cache_hit_kit import (
|
|
async_request_sglang_generate,
|
|
gen_payload,
|
|
run_multiturn_cache_hit_test,
|
|
)
|
|
from sglang.test.run_eval import run_eval
|
|
from sglang.test.server_fixtures.disaggregation_fixture import (
|
|
PDDisaggregationServerBase,
|
|
)
|
|
from sglang.test.test_utils import (
|
|
DEFAULT_MODEL_NAME_FOR_TEST,
|
|
is_in_ci,
|
|
try_cached_model,
|
|
)
|
|
|
|
register_cuda_ci(est_time=300, stage="base-c", runner_config="8-gpu-h20")
|
|
|
|
|
|
def _has_nixl():
|
|
try:
|
|
import nixl._api # noqa: F401
|
|
except ImportError:
|
|
return False
|
|
return True
|
|
|
|
|
|
def _has_mooncake():
|
|
try:
|
|
import mooncake.engine # noqa: F401
|
|
except ImportError:
|
|
return False
|
|
return True
|
|
|
|
|
|
class DisaggregationDecodeRadixCacheTestMixin:
|
|
extra_decode_args = ["--disaggregation-decode-enable-radix-cache"]
|
|
transfer_backend_name = None
|
|
|
|
@classmethod
|
|
def setUpClass(cls):
|
|
super().setUpClass()
|
|
cls.model = try_cached_model(DEFAULT_MODEL_NAME_FOR_TEST)
|
|
cls.transfer_backend = [
|
|
"--disaggregation-transfer-backend",
|
|
cls.transfer_backend_name,
|
|
]
|
|
cls.launch_all()
|
|
|
|
def _assert_process_healthy(self, name, process, url):
|
|
self.assertIsNotNone(process, f"{name} process was not started")
|
|
self.assertIsNone(
|
|
process.poll(),
|
|
f"{name} exited unexpectedly with code {process.returncode}",
|
|
)
|
|
response = requests.get(f"{url}/health", timeout=10)
|
|
response.raise_for_status()
|
|
|
|
def test_decode_radix_cache_hits_and_workers_stay_alive(self):
|
|
decode_info = requests.get(f"{self.decode_url}/server_info", timeout=10).json()
|
|
self.assertFalse(
|
|
decode_info.get("disable_radix_cache", True),
|
|
"decode server did not enable radix cache",
|
|
)
|
|
|
|
result = run_multiturn_cache_hit_test(
|
|
base_url=self.base_url,
|
|
model_path=self.model,
|
|
num_clients=4,
|
|
num_rounds=3,
|
|
request_length=384,
|
|
output_length=64,
|
|
max_parallel=4,
|
|
)
|
|
self.assertGreater(
|
|
result["overall"]["total_cached_tokens"],
|
|
0,
|
|
"expected decode radix cache to reuse at least some tokens",
|
|
)
|
|
|
|
# Give the schedulers a short idle window so any post-request leak/crash
|
|
# paths have a chance to surface before the liveness checks below.
|
|
time.sleep(5)
|
|
|
|
self._assert_process_healthy("load balancer", self.process_lb, self.lb_url)
|
|
self._assert_process_healthy("prefill", self.process_prefill, self.prefill_url)
|
|
self._assert_process_healthy("decode", self.process_decode, self.decode_url)
|
|
|
|
def test_gsm8k_accuracy_two_passes(self):
|
|
"""Run GSM8K twice to verify decode radix cache does not degrade accuracy."""
|
|
args = SimpleNamespace(
|
|
base_url=self.base_url,
|
|
model=self.model,
|
|
eval_name="gsm8k",
|
|
api="completion",
|
|
max_tokens=512,
|
|
num_examples=500,
|
|
num_threads=100,
|
|
num_shots=6,
|
|
)
|
|
|
|
metrics_first = run_eval(args)
|
|
print(f"First run metrics: {metrics_first}")
|
|
|
|
metrics_second = run_eval(args)
|
|
print(f"Second run metrics: {metrics_second}")
|
|
|
|
self.assertGreater(metrics_first["score"], 0.80)
|
|
self.assertGreater(metrics_second["score"], 0.80)
|
|
|
|
accuracy_drop = metrics_first["score"] - metrics_second["score"]
|
|
self.assertLessEqual(
|
|
accuracy_drop,
|
|
0.03,
|
|
f"Second run accuracy dropped by {accuracy_drop:.4f} "
|
|
f"(first={metrics_first['score']:.4f}, second={metrics_second['score']:.4f}), "
|
|
f"exceeds 3% threshold",
|
|
)
|
|
|
|
|
|
@unittest.skipUnless(
|
|
is_in_ci() or _has_nixl(),
|
|
"NIXL is required for decode radix cache disaggregation coverage.",
|
|
)
|
|
class TestDisaggregationDecodeRadixCacheNixl(
|
|
DisaggregationDecodeRadixCacheTestMixin, PDDisaggregationServerBase
|
|
):
|
|
transfer_backend_name = "nixl"
|
|
|
|
|
|
@unittest.skipUnless(
|
|
is_in_ci() or _has_mooncake(),
|
|
"Mooncake is required for decode radix cache disaggregation coverage.",
|
|
)
|
|
class TestDisaggregationDecodeRadixCacheMooncake(
|
|
DisaggregationDecodeRadixCacheTestMixin, PDDisaggregationServerBase
|
|
):
|
|
transfer_backend_name = "mooncake"
|
|
|
|
|
|
@unittest.skipUnless(
|
|
is_in_ci() or _has_mooncake(),
|
|
"Mooncake is required for decode radix cache disaggregation coverage.",
|
|
)
|
|
class TestDisaggregationDecodeRadixHiCacheFileBackend(PDDisaggregationServerBase):
|
|
extra_prefill_args = [
|
|
"--enable-hierarchical-cache",
|
|
"--hicache-ratio",
|
|
"1.2",
|
|
"--hicache-write-policy",
|
|
"write_through",
|
|
"--hicache-storage-backend",
|
|
"file",
|
|
"--hicache-storage-prefetch-policy",
|
|
"wait_complete",
|
|
"--hicache-io-backend",
|
|
"kernel",
|
|
"--hicache-mem-layout",
|
|
"page_first",
|
|
"--page-size",
|
|
"64",
|
|
]
|
|
extra_decode_args = [
|
|
"--disaggregation-decode-enable-radix-cache",
|
|
*extra_prefill_args,
|
|
]
|
|
transfer_backend_name = "mooncake"
|
|
|
|
@classmethod
|
|
def setUpClass(cls):
|
|
cls.hicache_dir = tempfile.mkdtemp(prefix="sglang-hicache-")
|
|
os.environ["SGLANG_HICACHE_FILE_BACKEND_STORAGE_DIR"] = cls.hicache_dir
|
|
|
|
super().setUpClass()
|
|
cls.model = try_cached_model(DEFAULT_MODEL_NAME_FOR_TEST)
|
|
cls.transfer_backend = [
|
|
"--disaggregation-transfer-backend",
|
|
cls.transfer_backend_name,
|
|
]
|
|
cls.launch_all()
|
|
|
|
@classmethod
|
|
def tearDownClass(cls):
|
|
super().tearDownClass()
|
|
os.environ.pop("SGLANG_HICACHE_FILE_BACKEND_STORAGE_DIR", None)
|
|
shutil.rmtree(cls.hicache_dir, ignore_errors=True)
|
|
|
|
def _post_ok(self, url):
|
|
response = requests.post(url, timeout=60)
|
|
response.raise_for_status()
|
|
|
|
def _flush_memory_cache(self):
|
|
self._post_ok(f"{self.prefill_url}/flush_cache?timeout=30")
|
|
self._post_ok(f"{self.decode_url}/flush_cache?timeout=30")
|
|
|
|
def _generate(self, input_ids, output_len):
|
|
output = asyncio.run(
|
|
async_request_sglang_generate(
|
|
gen_payload(input_ids, output_len),
|
|
f"{self.base_url}/generate",
|
|
)
|
|
)
|
|
self.assertTrue(output.success, output.error)
|
|
return output
|
|
|
|
def _sample_token_ids(self, input_len, output_len, num_prompts=1):
|
|
tokenizer = get_tokenizer(self.model)
|
|
return [
|
|
list(request.prompt)
|
|
for request in sample_random_requests(
|
|
input_len=input_len,
|
|
output_len=output_len,
|
|
num_prompts=num_prompts,
|
|
range_ratio=1.0,
|
|
tokenizer=tokenizer,
|
|
dataset_path="",
|
|
return_text=False,
|
|
)
|
|
]
|
|
|
|
def test_decode_hicache_file_backend_l3_reuses_decode_output_after_flush(self):
|
|
self._post_ok(f"{self.decode_url}/hicache/storage-backend/clear")
|
|
self._flush_memory_cache()
|
|
|
|
num_rounds = 5
|
|
output_len = 64
|
|
history = self._sample_token_ids(
|
|
input_len=256, output_len=output_len, num_prompts=1
|
|
)[0]
|
|
suffixes = self._sample_token_ids(
|
|
input_len=64, output_len=output_len, num_prompts=num_rounds - 1
|
|
)
|
|
|
|
prev_prompt_len = 0
|
|
prev_output_len = 0
|
|
for round_idx in range(num_rounds):
|
|
output = self._generate(history, output_len)
|
|
if round_idx == 0:
|
|
self.assertEqual(output.cached_tokens, 0)
|
|
else:
|
|
self.assertGreaterEqual(
|
|
output.cached_tokens,
|
|
prev_prompt_len + prev_output_len,
|
|
)
|
|
|
|
history.extend(output.output_ids)
|
|
prev_prompt_len = output.prompt_len
|
|
prev_output_len = len(output.output_ids)
|
|
|
|
if round_idx < num_rounds - 1:
|
|
history.extend(suffixes[round_idx])
|
|
time.sleep(1)
|
|
self._flush_memory_cache()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|