sgl-project--sglang
94057c3d3e
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
186 行
6.1 KiB
Python
186 行
6.1 KiB
Python
import gc
|
|
import unittest
|
|
|
|
import numpy as np
|
|
import requests
|
|
from transformers import AutoModelForCausalLM
|
|
|
|
import sglang as sgl
|
|
from sglang.srt.utils import get_device
|
|
from sglang.test.test_utils import (
|
|
DEFAULT_MODEL_NAME_FOR_TEST,
|
|
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
|
|
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
|
DEFAULT_URL_FOR_TEST,
|
|
CustomTestCase,
|
|
empty_gpu_cache,
|
|
get_gpu_count,
|
|
is_in_ci,
|
|
popen_launch_server,
|
|
)
|
|
from sglang.utils import terminate_process
|
|
|
|
|
|
def _process_return(ret):
|
|
if isinstance(ret, list) and len(ret) == 2:
|
|
print(f"running assert_allclose on data parallel")
|
|
np.testing.assert_allclose(ret[0], ret[1])
|
|
return np.array(ret[0])
|
|
return np.array(ret)
|
|
|
|
|
|
class TestGetWeightsByName(CustomTestCase):
|
|
|
|
def init_hf_model(self, model_name, tie_word_embeddings):
|
|
self.hf_model = AutoModelForCausalLM.from_pretrained(
|
|
model_name, torch_dtype="bfloat16", tie_word_embeddings=tie_word_embeddings
|
|
).to(get_device())
|
|
|
|
def init_backend(self, backend, dp, tp, model_name):
|
|
self.backend = backend
|
|
self.dp = dp
|
|
self.tp = tp
|
|
if backend == "Engine":
|
|
self.engine = sgl.Engine(
|
|
model_path=model_name,
|
|
random_seed=42,
|
|
tp_size=tp,
|
|
dp_size=dp,
|
|
)
|
|
else:
|
|
self.process = popen_launch_server(
|
|
model_name,
|
|
DEFAULT_URL_FOR_TEST,
|
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
|
other_args=(
|
|
"--tp-size",
|
|
str(tp),
|
|
"--dp-size",
|
|
str(dp),
|
|
),
|
|
)
|
|
|
|
def clean_up(self):
|
|
del self.hf_model
|
|
gc.collect()
|
|
empty_gpu_cache()
|
|
if self.backend == "Engine":
|
|
self.engine.shutdown()
|
|
else:
|
|
terminate_process(self.process)
|
|
|
|
def assert_tie_word_embeddings(self, truncate_size):
|
|
print("assert_tie_word_embeddings")
|
|
if self.backend == "Engine":
|
|
backend_ret = _process_return(
|
|
self.engine.get_weights_by_name("lm_head.weight", truncate_size)
|
|
)
|
|
else:
|
|
backend_ret = _process_return(
|
|
requests.get(
|
|
f"{DEFAULT_URL_FOR_TEST}/get_weights_by_name",
|
|
json={"name": "lm_head.weight", "truncate_size": truncate_size},
|
|
).json()
|
|
)
|
|
print("assert_tie_word_embeddings of hf and backend")
|
|
assert np.allclose(
|
|
self.hf_model.get_parameter("model.embed_tokens.weight")
|
|
.cpu()
|
|
.detach()
|
|
.float()
|
|
.numpy()[:truncate_size],
|
|
backend_ret,
|
|
)
|
|
assert np.allclose(
|
|
self.hf_model.get_parameter("lm_head.weight")
|
|
.cpu()
|
|
.detach()
|
|
.float()
|
|
.numpy()[:truncate_size],
|
|
self.hf_model.get_parameter("model.embed_tokens.weight")
|
|
.cpu()
|
|
.detach()
|
|
.float()
|
|
.numpy()[:truncate_size],
|
|
)
|
|
|
|
def assert_weights_all_close(self, param_name, truncate_size):
|
|
print(
|
|
f"param_name: {param_name}, backend: {self.backend}, dp: {self.dp}, tp: {self.tp}"
|
|
)
|
|
param = self.hf_model.get_parameter(param_name)[:truncate_size]
|
|
param_np = param.cpu().detach().float().numpy()
|
|
|
|
if self.backend == "Engine":
|
|
engine_ret = self.engine.get_weights_by_name(param_name, truncate_size)
|
|
engine_ret = _process_return(engine_ret)
|
|
np.testing.assert_allclose(engine_ret, param_np, rtol=1e-5, atol=1e-5)
|
|
|
|
if self.backend == "Runtime":
|
|
runtime_ret = requests.get(
|
|
f"{DEFAULT_URL_FOR_TEST}/get_weights_by_name",
|
|
json={"name": param_name, "truncate_size": truncate_size},
|
|
).json()
|
|
runtime_ret = _process_return(runtime_ret)
|
|
np.testing.assert_allclose(runtime_ret, param_np, rtol=1e-5, atol=1e-5)
|
|
|
|
def test_get_weights_by_name(self):
|
|
if is_in_ci():
|
|
test_suits = [
|
|
("Engine", 1, 1, DEFAULT_SMALL_MODEL_NAME_FOR_TEST),
|
|
]
|
|
else:
|
|
test_suits = [
|
|
("Runtime", 1, 1, DEFAULT_SMALL_MODEL_NAME_FOR_TEST),
|
|
("Engine", 1, 1, DEFAULT_MODEL_NAME_FOR_TEST),
|
|
]
|
|
if get_gpu_count() >= 2:
|
|
test_suits.append(("Engine", 1, 2, DEFAULT_SMALL_MODEL_NAME_FOR_TEST))
|
|
test_suits.append(("Runtime", 2, 1, DEFAULT_MODEL_NAME_FOR_TEST))
|
|
|
|
if get_gpu_count() >= 4:
|
|
test_suits.extend(
|
|
[
|
|
("Engine", 2, 2, DEFAULT_SMALL_MODEL_NAME_FOR_TEST),
|
|
("Runtime", 2, 2, DEFAULT_MODEL_NAME_FOR_TEST),
|
|
]
|
|
)
|
|
|
|
parameters = [
|
|
"model.embed_tokens.weight",
|
|
"model.layers.0.input_layernorm.weight",
|
|
"model.layers.1.self_attn.q_proj.weight",
|
|
"model.layers.2.self_attn.k_proj.weight",
|
|
"model.layers.3.self_attn.v_proj.weight",
|
|
"model.layers.4.self_attn.o_proj.weight",
|
|
"model.layers.5.mlp.gate_proj.weight",
|
|
"model.layers.6.mlp.up_proj.weight",
|
|
"model.layers.7.mlp.down_proj.weight",
|
|
"model.layers.8.post_attention_layernorm.weight",
|
|
"model.norm.weight",
|
|
"lm_head.weight",
|
|
]
|
|
|
|
truncate_size = 100
|
|
|
|
for test_suit in test_suits:
|
|
if test_suit[-1] == DEFAULT_MODEL_NAME_FOR_TEST:
|
|
tie_word_embeddings = False
|
|
else:
|
|
tie_word_embeddings = True
|
|
|
|
self.init_hf_model(test_suit[-1], tie_word_embeddings)
|
|
self.init_backend(*test_suit)
|
|
|
|
for param_name in parameters:
|
|
self.assert_weights_all_close(param_name, truncate_size)
|
|
|
|
if tie_word_embeddings:
|
|
self.assert_tie_word_embeddings(truncate_size)
|
|
|
|
self.clean_up()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|