sgl-project--sglang
94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
228 行
8.4 KiB
Python
228 行
8.4 KiB
Python
import unittest
|
|
from typing import Dict, List
|
|
from unittest.mock import Mock
|
|
|
|
import requests
|
|
from prometheus_client.parser import text_string_to_metric_families
|
|
from prometheus_client.samples import Sample
|
|
|
|
from sglang.srt.observability.metrics_collector import QueueCount
|
|
from sglang.srt.utils import kill_process_tree
|
|
from sglang.test.ci.ci_register import (
|
|
register_amd_ci,
|
|
register_cpu_ci,
|
|
register_cuda_ci,
|
|
)
|
|
from sglang.test.test_utils import (
|
|
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
|
DEFAULT_URL_FOR_TEST,
|
|
CustomTestCase,
|
|
popen_launch_server,
|
|
)
|
|
|
|
register_cuda_ci(
|
|
est_time=60,
|
|
stage="base-b",
|
|
runner_config="1-gpu-small",
|
|
)
|
|
register_amd_ci(est_time=60, suite="stage-b-test-1-gpu-small-amd")
|
|
register_cpu_ci(est_time=179, suite="base-c-test-cpu")
|
|
|
|
_MODEL_NAME = "Qwen/Qwen3-0.6B"
|
|
|
|
|
|
def _parse_prometheus_metrics(metrics_text: str) -> Dict[str, List[Sample]]:
|
|
result = {}
|
|
for family in text_string_to_metric_families(metrics_text):
|
|
for sample in family.samples:
|
|
if sample.name not in result:
|
|
result[sample.name] = []
|
|
result[sample.name].append(sample)
|
|
return result
|
|
|
|
|
|
def _get_samples_by_name(metrics: Dict[str, List[Sample]], name: str) -> List[Sample]:
|
|
return metrics.get(name, [])
|
|
|
|
|
|
def _get_sample_value_by_labels(samples: List[Sample], labels: Dict[str, str]) -> float:
|
|
for sample in samples:
|
|
if all(sample.labels.get(k) == v for k, v in labels.items()):
|
|
return sample.value
|
|
raise KeyError(f"No sample found with labels {labels}")
|
|
|
|
|
|
class TestQueueCount(CustomTestCase):
|
|
"""Unit tests for QueueCount (no server needed)."""
|
|
|
|
def test_queue_count_from_reqs(self):
|
|
"""QueueCount correctly counts per-priority breakdown."""
|
|
reqs = [
|
|
Mock(priority=1),
|
|
Mock(priority=1),
|
|
Mock(priority=5),
|
|
Mock(priority=5),
|
|
Mock(priority=10),
|
|
]
|
|
qc = QueueCount.from_reqs(reqs, enable_priority_scheduling=True)
|
|
self.assertEqual(qc.total, 5)
|
|
self.assertEqual(qc.by_priority, {1: 2, 5: 2, 10: 1})
|
|
|
|
def test_queue_count_from_reqs_disabled(self):
|
|
"""Priority scheduling disabled → no breakdown."""
|
|
reqs = [Mock(priority=1), Mock(priority=5)]
|
|
qc = QueueCount.from_reqs(reqs, enable_priority_scheduling=False)
|
|
self.assertEqual(qc.total, 2)
|
|
self.assertIsNone(qc.by_priority)
|
|
|
|
def test_queue_count_empty(self):
|
|
"""Empty request list."""
|
|
qc = QueueCount.from_reqs([], enable_priority_scheduling=True)
|
|
self.assertEqual(qc.total, 0)
|
|
self.assertEqual(qc.by_priority, {})
|
|
|
|
|
|
class TestPriorityMetrics(CustomTestCase):
|
|
"""Test that priority-based metrics are correctly emitted when
|
|
--enable-priority-scheduling is enabled."""
|
|
|
|
@classmethod
|
|
def setUpClass(cls):
|
|
cls.process = popen_launch_server(
|
|
_MODEL_NAME,
|
|
DEFAULT_URL_FOR_TEST,
|
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
|
other_args=[
|
|
"--enable-metrics",
|
|
"--enable-priority-scheduling",
|
|
"--default-priority-value",
|
|
"0",
|
|
],
|
|
)
|
|
|
|
@classmethod
|
|
def tearDownClass(cls):
|
|
kill_process_tree(cls.process.pid)
|
|
|
|
def test_priority_label_in_gauge_metrics(self):
|
|
"""Send requests with different priorities and verify that
|
|
gauge metrics (num_running_reqs, num_queue_reqs) contain
|
|
the priority label dimension."""
|
|
|
|
# Send requests with different priorities to populate metrics
|
|
for priority in [1, 5, 10]:
|
|
response = requests.post(
|
|
f"{DEFAULT_URL_FOR_TEST}/generate",
|
|
json={
|
|
"text": "Hello",
|
|
"sampling_params": {"temperature": 0, "max_new_tokens": 5},
|
|
"priority": priority,
|
|
},
|
|
)
|
|
self.assertEqual(response.status_code, 200)
|
|
|
|
# Fetch metrics
|
|
metrics_response = requests.get(f"{DEFAULT_URL_FOR_TEST}/metrics")
|
|
self.assertEqual(metrics_response.status_code, 200)
|
|
metrics = _parse_prometheus_metrics(metrics_response.text)
|
|
|
|
# Verify priority label exists on queue gauge metrics
|
|
for metric_name in ["sglang:num_running_reqs", "sglang:num_queue_reqs"]:
|
|
samples = _get_samples_by_name(metrics, metric_name)
|
|
self.assertGreater(len(samples), 0, f"No samples found for {metric_name}")
|
|
|
|
# Should have at least one sample with a non-empty priority label
|
|
# (the total has priority="" and per-priority has priority="<int>")
|
|
priority_labels = {s.labels.get("priority", "") for s in samples}
|
|
self.assertIn(
|
|
"",
|
|
priority_labels,
|
|
f"{metric_name}: missing total (priority='') sample",
|
|
)
|
|
|
|
def test_priority_label_in_histogram_metrics(self):
|
|
"""Send requests with different priorities and verify that
|
|
histogram metrics (TTFT, ITL, e2e latency) contain the priority label."""
|
|
|
|
for priority in [1, 5]:
|
|
response = requests.post(
|
|
f"{DEFAULT_URL_FOR_TEST}/generate",
|
|
json={
|
|
"text": "The capital of France is",
|
|
"sampling_params": {"temperature": 0, "max_new_tokens": 20},
|
|
"priority": priority,
|
|
},
|
|
)
|
|
self.assertEqual(response.status_code, 200)
|
|
|
|
metrics_response = requests.get(f"{DEFAULT_URL_FOR_TEST}/metrics")
|
|
self.assertEqual(metrics_response.status_code, 200)
|
|
metrics = _parse_prometheus_metrics(metrics_response.text)
|
|
|
|
# Check histogram metrics have priority label with per-priority breakdown
|
|
histogram_metrics = [
|
|
"sglang:time_to_first_token_seconds",
|
|
"sglang:e2e_request_latency_seconds",
|
|
]
|
|
for metric_name in histogram_metrics:
|
|
# Histogram metrics are emitted as _sum, _count, _bucket
|
|
count_name = f"{metric_name}_count"
|
|
samples = _get_samples_by_name(metrics, count_name)
|
|
self.assertGreater(len(samples), 0, f"No samples found for {count_name}")
|
|
# At least one sample should have a non-empty priority label
|
|
priority_values = {s.labels.get("priority", "") for s in samples}
|
|
non_empty = priority_values - {""}
|
|
self.assertGreater(
|
|
len(non_empty),
|
|
0,
|
|
f"{count_name}: expected per-priority samples, "
|
|
f"got priority labels: {priority_values}",
|
|
)
|
|
# Verify that both priority="1" and priority="5" have count > 0
|
|
for expected_priority in ["1", "5"]:
|
|
matching = [
|
|
s for s in samples if s.labels.get("priority") == expected_priority
|
|
]
|
|
self.assertGreater(
|
|
len(matching),
|
|
0,
|
|
f"{count_name}: no sample with priority='{expected_priority}'",
|
|
)
|
|
self.assertGreater(
|
|
matching[0].value,
|
|
0,
|
|
f"{count_name}: priority='{expected_priority}' count should be > 0",
|
|
)
|
|
|
|
def test_default_priority_value(self):
|
|
"""Requests without explicit priority should use --default-priority-value (0)."""
|
|
|
|
# Send request WITHOUT priority — should get default priority 0
|
|
response = requests.post(
|
|
f"{DEFAULT_URL_FOR_TEST}/generate",
|
|
json={
|
|
"text": "Hello world",
|
|
"sampling_params": {"temperature": 0, "max_new_tokens": 5},
|
|
},
|
|
)
|
|
self.assertEqual(response.status_code, 200)
|
|
|
|
metrics_response = requests.get(f"{DEFAULT_URL_FOR_TEST}/metrics")
|
|
self.assertEqual(metrics_response.status_code, 200)
|
|
metrics = _parse_prometheus_metrics(metrics_response.text)
|
|
|
|
# Check that e2e latency has samples with priority="0" (the default)
|
|
e2e_count = _get_samples_by_name(
|
|
metrics, "sglang:e2e_request_latency_seconds_count"
|
|
)
|
|
priority_values = {s.labels.get("priority", "") for s in e2e_count}
|
|
self.assertIn(
|
|
"0",
|
|
priority_values,
|
|
f"Expected priority='0' from default, got: {priority_values}",
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|