sgl-project--sglang
94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
207 行
6.9 KiB
Python
207 行
6.9 KiB
Python
import random
|
|
import unittest
|
|
from contextlib import contextmanager
|
|
|
|
from transformers import AutoTokenizer
|
|
|
|
from sglang.srt.utils.patch_tokenizer import (
|
|
_SpecialTokensCachePatcher,
|
|
decode_without_hf_kwargs,
|
|
unpatch_tokenizer,
|
|
)
|
|
from sglang.test.ci.ci_register import register_cpu_ci
|
|
|
|
register_cpu_ci(est_time=30, suite="base-a-test-cpu", nightly=True)
|
|
register_cpu_ci(est_time=53, suite="base-c-test-cpu")
|
|
|
|
|
|
class TestPatchTokenizerEndToEndTest(unittest.TestCase):
|
|
def test_patched_produces_same_results_as_raw(self):
|
|
tokenizer = _load_tokenizer()
|
|
test_texts = self._generate_test_texts(tokenizer)
|
|
raw_results = self._run_tokenizer_ops(tokenizer, test_texts)
|
|
|
|
_SpecialTokensCachePatcher.patch(tokenizer)
|
|
patched_results = self._run_tokenizer_ops(tokenizer, test_texts)
|
|
unpatch_tokenizer(tokenizer)
|
|
|
|
self.assertEqual(raw_results, patched_results)
|
|
|
|
@classmethod
|
|
def _generate_test_texts(cls, tokenizer):
|
|
special_tokens = tokenizer.all_special_tokens
|
|
return [
|
|
"Hello, world!",
|
|
"This is a longer sentence with multiple words.",
|
|
"Numbers 12345 and symbols !@#$%",
|
|
" leading and trailing spaces ",
|
|
"\n\nMultiple\n\nNewlines\n\n",
|
|
*[f"Text with {tok} inside" for tok in special_tokens],
|
|
" ".join(special_tokens),
|
|
*[
|
|
cls._random_text_from_tokens(tokenizer, num_tokens=100)
|
|
for _ in range(5)
|
|
],
|
|
*[
|
|
cls._random_text_from_tokens(tokenizer, num_tokens=1000)
|
|
for _ in range(3)
|
|
],
|
|
]
|
|
|
|
@classmethod
|
|
def _random_text_from_tokens(cls, tokenizer, num_tokens):
|
|
token_ids = [
|
|
random.randint(0, tokenizer.vocab_size - 1) for _ in range(num_tokens)
|
|
]
|
|
return tokenizer.decode(token_ids)
|
|
|
|
@classmethod
|
|
def _run_tokenizer_ops(cls, tokenizer, texts):
|
|
encode_results = [tokenizer.encode(t) for t in texts]
|
|
batch_encode_results = tokenizer(texts)["input_ids"]
|
|
return {
|
|
"encode": encode_results,
|
|
"batch_encode": batch_encode_results,
|
|
"decode": [
|
|
tokenizer.decode(ids, skip_special_tokens=True)
|
|
for ids in encode_results
|
|
],
|
|
"batch_decode": tokenizer.batch_decode(
|
|
encode_results, skip_special_tokens=True
|
|
),
|
|
"special_tokens": tokenizer.all_special_tokens,
|
|
"special_ids": tokenizer.all_special_ids,
|
|
}
|
|
|
|
|
|
class TestPatchTokenizerUnitTest(unittest.TestCase):
|
|
def test_patch_unpatch_restores_original(self):
|
|
tokenizer = _load_tokenizer()
|
|
cls = type(tokenizer)
|
|
|
|
original_ids = _get_class_attr_ids(cls)
|
|
|
|
_SpecialTokensCachePatcher.patch(tokenizer)
|
|
self.assertTrue(getattr(cls, "_sglang_special_tokens_patched", False))
|
|
|
|
patched_ids = _get_class_attr_ids(cls)
|
|
changed_attrs = [
|
|
name
|
|
for name in original_ids
|
|
if name in patched_ids and patched_ids[name] != original_ids[name]
|
|
]
|
|
self.assertGreater(len(changed_attrs), 0, "Patch should change some attributes")
|
|
|
|
unpatch_tokenizer(tokenizer)
|
|
self.assertFalse(getattr(cls, "_sglang_special_tokens_patched", False))
|
|
|
|
restored_ids = _get_class_attr_ids(cls)
|
|
for name in original_ids:
|
|
if name.startswith("_sglang") or name.startswith("_original"):
|
|
continue
|
|
self.assertEqual(
|
|
restored_ids.get(name),
|
|
original_ids[name],
|
|
f"Attribute {name} should be restored to original",
|
|
)
|
|
|
|
def test_patch_caches_special_tokens(self):
|
|
with _patched_tokenizer() as tokenizer:
|
|
tokens1 = tokenizer.all_special_tokens
|
|
ids1 = tokenizer.all_special_ids
|
|
tokens2 = tokenizer.all_special_tokens
|
|
ids2 = tokenizer.all_special_ids
|
|
|
|
self.assertIs(tokens1, tokens2)
|
|
self.assertIs(ids1, ids2)
|
|
|
|
def test_patch_blocks_add_special_tokens(self):
|
|
with _patched_tokenizer() as tokenizer:
|
|
with self.assertRaises(AssertionError) as ctx:
|
|
tokenizer.add_special_tokens({"pad_token": "<pad>"})
|
|
self.assertIn(
|
|
"Cannot modify special tokens after patch", str(ctx.exception)
|
|
)
|
|
|
|
def test_patch_blocks_add_tokens_with_special_flag(self):
|
|
with _patched_tokenizer() as tokenizer:
|
|
with self.assertRaises(AssertionError) as ctx:
|
|
tokenizer.add_tokens(["<new>"], special_tokens=True)
|
|
self.assertIn("Cannot add special tokens after patch", str(ctx.exception))
|
|
|
|
tokenizer.add_tokens(["<regular>"], special_tokens=False)
|
|
|
|
def test_unpatch_clears_cache(self):
|
|
with _patched_tokenizer() as tokenizer:
|
|
_ = tokenizer.all_special_tokens
|
|
_ = tokenizer.all_special_ids
|
|
self.assertTrue(hasattr(tokenizer, "_sglang_cached_special_tokens"))
|
|
self.assertTrue(hasattr(tokenizer, "_sglang_cached_special_ids"))
|
|
|
|
self.assertFalse(hasattr(tokenizer, "_sglang_cached_special_tokens"))
|
|
self.assertFalse(hasattr(tokenizer, "_sglang_cached_special_ids"))
|
|
|
|
def test_double_patch_is_idempotent(self):
|
|
tokenizer = _load_tokenizer()
|
|
_SpecialTokensCachePatcher.patch(tokenizer)
|
|
_SpecialTokensCachePatcher.patch(tokenizer)
|
|
|
|
self.assertTrue(
|
|
getattr(type(tokenizer), "_sglang_special_tokens_patched", False)
|
|
)
|
|
|
|
unpatch_tokenizer(tokenizer)
|
|
|
|
def test_decode_without_hf_kwargs_uses_native_decode(self):
|
|
tokenizer = _FakeDecodeTokenizer()
|
|
|
|
self.assertEqual(
|
|
decode_without_hf_kwargs(tokenizer, [1, 99, 2], True),
|
|
"ab",
|
|
)
|
|
self.assertEqual(
|
|
decode_without_hf_kwargs(tokenizer, [1, 99, 2], False),
|
|
"a<special>b",
|
|
)
|
|
self.assertEqual(tokenizer.decode_calls, [[1, 2], [1, 99, 2]])
|
|
|
|
|
|
def _get_class_attr_ids(cls):
|
|
return {
|
|
n: id(v.fget if isinstance(v, property) else v) for n, v in vars(cls).items()
|
|
}
|
|
|
|
|
|
def _load_tokenizer():
|
|
# The slowness is mainly observed in Kimi
|
|
return AutoTokenizer.from_pretrained(
|
|
"nvidia/Kimi-K2-Thinking-NVFP4", trust_remote_code=True
|
|
)
|
|
|
|
|
|
@contextmanager
|
|
def _patched_tokenizer():
|
|
tokenizer = _load_tokenizer()
|
|
_SpecialTokensCachePatcher.patch(tokenizer)
|
|
try:
|
|
yield tokenizer
|
|
finally:
|
|
unpatch_tokenizer(tokenizer)
|
|
|
|
|
|
class _FakeDecodeTokenizer:
|
|
all_special_ids_set = {99}
|
|
|
|
def __init__(self):
|
|
self.decode_calls = []
|
|
|
|
def decode(self, token_ids):
|
|
token_ids = list(token_ids)
|
|
self.decode_calls.append(token_ids)
|
|
token_text = {1: "a", 2: "b", 99: "<special>"}
|
|
return "".join(token_text[token_id] for token_id in token_ids)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|