sgl-project--sglang
94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
181 行
6.1 KiB
Python
181 行
6.1 KiB
Python
from __future__ import annotations
|
|
|
|
import time
|
|
import unittest
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from typing import ClassVar, Dict, List
|
|
|
|
import requests
|
|
|
|
from sglang.srt.utils import is_hip
|
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
|
from sglang.test.kv_canary.violation_log_utils import assert_no_violation_in_log
|
|
from sglang.test.mock_model.utils import (
|
|
MOCK_MODEL_PATH,
|
|
mock_model_server_args,
|
|
mock_model_server_env,
|
|
)
|
|
from sglang.test.server_fixtures.disaggregation_fixture import (
|
|
PDDisaggregationServerBase,
|
|
)
|
|
|
|
register_cuda_ci(est_time=600, stage="extra-a", runner_config="2-gpu-large")
|
|
register_amd_ci(est_time=165, stage="extra-a", runner_config="2-gpu-large-amd")
|
|
|
|
# DO NOT pass --disable-cuda-graph in canary e2e tests. The canary kernel
|
|
# must run inside the cuda graph alongside the real attn kernel; disabling the
|
|
# full graph silently bypasses the only path that exercises that invariant
|
|
# end-to-end.
|
|
#
|
|
# --disable-piecewise-cuda-graph is REQUIRED by canary: install_canary
|
|
# (api.py) asserts it, and the SingleForwardManager design depends on it.
|
|
# mock_model_server_args() already passes it; do not remove it.
|
|
_NUM_PROMPTS = 32
|
|
_INPUT_LEN = 6144
|
|
_OUTPUT_LEN = 1024
|
|
|
|
|
|
def _send_parallel_requests(
|
|
base_url: str,
|
|
*,
|
|
n: int,
|
|
max_new_tokens: int,
|
|
timeout: float = 60.0,
|
|
max_workers: int = 16,
|
|
) -> List[Dict[str, object]]:
|
|
"""Fire N /generate requests concurrently; return raw response dicts."""
|
|
|
|
def _one(i: int) -> Dict[str, object]:
|
|
payload = {
|
|
"input_ids": _make_input_ids(seed=i, length=_INPUT_LEN),
|
|
"sampling_params": {"max_new_tokens": max_new_tokens, "temperature": 0.0},
|
|
}
|
|
try:
|
|
resp = requests.post(base_url + "/generate", json=payload, timeout=timeout)
|
|
return {"index": i, "status_code": resp.status_code, "text": resp.text}
|
|
except requests.exceptions.RequestException as exc:
|
|
return {"index": i, "error": repr(exc)}
|
|
|
|
results: List[Dict[str, object]] = []
|
|
with ThreadPoolExecutor(max_workers=max_workers) as pool:
|
|
futures = [pool.submit(_one, i) for i in range(n)]
|
|
for fut in as_completed(futures):
|
|
results.append(fut.result())
|
|
results.sort(key=lambda r: r["index"])
|
|
return results
|
|
|
|
|
|
def _make_input_ids(*, seed: int, length: int) -> List[int]:
|
|
return [((seed + i) % 2048) + 1 for i in range(length)]
|
|
|
|
|
|
class _MockModelPDBase(PDDisaggregationServerBase):
|
|
"""PD fixture for mock-model + canary e2e tests."""
|
|
|
|
capture_per_side_logs = True
|
|
model: ClassVar[str] = MOCK_MODEL_PATH
|
|
extra_prefill_args: ClassVar[List[str]] = mock_model_server_args(
|
|
"--skip-server-warmup"
|
|
)
|
|
extra_decode_args: ClassVar[List[str]] = mock_model_server_args(
|
|
"--skip-server-warmup"
|
|
)
|
|
extra_prefill_env: ClassVar[Dict[str, str]] = mock_model_server_env(
|
|
input_check_enabled=True
|
|
)
|
|
extra_decode_env: ClassVar[Dict[str, str]] = mock_model_server_env(
|
|
input_check_enabled=True
|
|
)
|
|
|
|
@classmethod
|
|
def setUpClass(cls) -> None:
|
|
super().setUpClass()
|
|
cls.launch_all()
|
|
|
|
def assert_no_canary_violation(self) -> None:
|
|
time.sleep(2)
|
|
log_text = "".join(
|
|
buf.getvalue()
|
|
for buf in (
|
|
self._prefill_stdout_buf,
|
|
self._prefill_stderr_buf,
|
|
self._decode_stdout_buf,
|
|
self._decode_stderr_buf,
|
|
)
|
|
if buf is not None
|
|
)
|
|
assert_no_violation_in_log(log_text)
|
|
|
|
|
|
class TestPdTransferCanaryClean(_MockModelPDBase, unittest.TestCase):
|
|
"""PD standard scenario + baseline canary (input-check, no real-KV checksum); no violation expected."""
|
|
|
|
def test_pd_transfer_canary_clean(self) -> None:
|
|
# Step 1: send parallel requests through the LB to exercise PD transfer path.
|
|
results = _send_parallel_requests(
|
|
self.lb_url,
|
|
n=_NUM_PROMPTS,
|
|
max_new_tokens=_OUTPUT_LEN,
|
|
timeout=240.0,
|
|
max_workers=_NUM_PROMPTS,
|
|
)
|
|
|
|
# Step 2: every request must complete with status 200.
|
|
for result in results:
|
|
self.assertEqual(result.get("status_code"), 200, result)
|
|
|
|
# Step 3: servers must stay alive.
|
|
self.assertIsNone(self.process_prefill.poll(), "Prefill server died")
|
|
self.assertIsNone(self.process_decode.poll(), "Decode server died")
|
|
self.assert_no_canary_violation()
|
|
|
|
|
|
@unittest.skipIf(
|
|
is_hip(),
|
|
"ROCm: PD full-real-data KV checksum intermittently trips a "
|
|
"verify_real_kv_hash canary violation on the decode-side transferred prefix "
|
|
"(see https://github.com/sgl-project/sglang/issues/28971). The baseline PD "
|
|
"canary test above stays enabled on AMD.",
|
|
)
|
|
class TestPdTransferChecksumFullRealData(_MockModelPDBase, unittest.TestCase):
|
|
"""--kv-canary-real-data=all + sweep every step, no perturb, no violation."""
|
|
|
|
extra_prefill_args: ClassVar[List[str]] = mock_model_server_args(
|
|
"--skip-server-warmup",
|
|
"--kv-canary-real-data",
|
|
"all",
|
|
"--kv-canary-sweep-interval",
|
|
"1",
|
|
)
|
|
extra_decode_args: ClassVar[List[str]] = mock_model_server_args(
|
|
"--skip-server-warmup",
|
|
"--kv-canary-real-data",
|
|
"all",
|
|
"--kv-canary-sweep-interval",
|
|
"1",
|
|
"--disaggregation-decode-enable-radix-cache",
|
|
)
|
|
|
|
def test_pd_transfer_checksum_full_real_data(self) -> None:
|
|
# Step 1: drive traffic through the PD path with full real-KV hashing.
|
|
results = _send_parallel_requests(
|
|
self.lb_url,
|
|
n=_NUM_PROMPTS,
|
|
max_new_tokens=_OUTPUT_LEN,
|
|
timeout=240.0,
|
|
max_workers=_NUM_PROMPTS,
|
|
)
|
|
|
|
# Step 2: all requests must succeed.
|
|
for result in results:
|
|
self.assertEqual(result.get("status_code"), 200, result)
|
|
|
|
# Step 3: servers must stay healthy.
|
|
self.assertIsNone(self.process_prefill.poll(), "Prefill server died")
|
|
self.assertIsNone(self.process_decode.poll(), "Decode server died")
|
|
self.assert_no_canary_violation()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|