import unittest from types import SimpleNamespace import requests from sglang.srt.utils import kill_process_tree from sglang.test.ci.ci_register import register_cuda_ci from sglang.test.run_eval import run_eval from sglang.test.send_one import BenchArgs, send_one_prompt from sglang.test.test_utils import ( DEFAULT_DEEPEP_MODEL_NAME_FOR_TEST, DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, DEFAULT_URL_FOR_TEST, CustomTestCase, popen_launch_server, ) register_cuda_ci(est_time=528, stage="extra-b", runner_config="deepep-8-gpu-h200") DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2" @unittest.skip("Skip for saving ci time") class TestDeepseek(CustomTestCase): @classmethod def setUpClass(cls): cls.model = DEFAULT_DEEPEP_MODEL_NAME_FOR_TEST cls.base_url = DEFAULT_URL_FOR_TEST cls.process = popen_launch_server( cls.model, cls.base_url, timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, other_args=[ "--trust-remote-code", "--tp", "8", "--enable-dp-attention", "--dp", "8", "--moe-dense-tp-size", "1", "--enable-dp-lm-head", "--moe-a2a-backend", "deepep", "--moe-runner-backend", "deep_gemm", "--enable-two-batch-overlap", "--ep-num-redundant-experts", "32", "--ep-dispatch-algorithm", "dynamic", "--eplb-algorithm", "deepseek", "--cuda-graph-bs", "256", "--max-running-requests", "2048", "--disable-radix-cache", "--model-loader-extra-config", '{"enable_multithread_load": true,"num_threads": 64}', ], ) @classmethod def tearDownClass(cls): kill_process_tree(cls.process.pid) def test_gsm8k(self): args = SimpleNamespace( base_url=self.base_url, model=self.model, eval_name="gsm8k", api="completion", max_tokens=512, num_examples=1200, num_threads=1200, ) metrics = run_eval(args) print(f"Eval accuracy of GSM8K: {metrics=}") self.assertGreater(metrics["score"], 0.92) class TestDeepseekMTP(CustomTestCase): @classmethod def setUpClass(cls): cls.model = DEFAULT_DEEPEP_MODEL_NAME_FOR_TEST cls.base_url = DEFAULT_URL_FOR_TEST cls.process = popen_launch_server( cls.model, cls.base_url, timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, other_args=[ "--disable-overlap-schedule", "--trust-remote-code", "--tp", "8", "--enable-dp-attention", "--dp", "8", "--moe-dense-tp-size", "1", "--enable-dp-lm-head", "--moe-a2a-backend", "deepep", "--moe-runner-backend", "deep_gemm", "--enable-two-batch-overlap", "--ep-num-redundant-experts", "32", "--ep-dispatch-algorithm", "dynamic", "--eplb-algorithm", "deepseek", "--cuda-graph-bs", "64", # TODO: increase it to 128 when TBO is supported in draft_extend "--max-running-requests", "512", "--speculative-algorithm", "EAGLE", "--speculative-num-steps", "1", "--speculative-eagle-topk", "1", "--speculative-num-draft-tokens", "2", "--disable-radix-cache", "--model-loader-extra-config", '{"enable_multithread_load": true,"num_threads": 64}', ], ) @classmethod def tearDownClass(cls): kill_process_tree(cls.process.pid) def test_gsm8k(self): args = SimpleNamespace( base_url=self.base_url, model=self.model, eval_name="gsm8k", api="completion", max_tokens=512, num_examples=1200, num_threads=1200, ) metrics = run_eval(args) print(f"Eval accuracy of GSM8K: {metrics=}") self.assertGreater(metrics["score"], 0.92) server_info = requests.get(self.base_url + "/server_info") avg_spec_accept_length = server_info.json()["internal_states"][0][ "avg_spec_accept_length" ] print( f"###test_gsm8k:\n" f"accuracy={metrics['score']=:.3f}\n" f"{avg_spec_accept_length=:.3f}\n" ) self.assertGreater(avg_spec_accept_length, 1.85) class TestDeepseekV32TBO(CustomTestCase): @classmethod def setUpClass(cls): cls.model = DEEPSEEK_V32_MODEL_PATH cls.base_url = DEFAULT_URL_FOR_TEST other_args = [ "--trust-remote-code", "--tp", "8", "--dp", "8", "--enable-dp-attention", "--enable-two-batch-overlap", "--moe-a2a-backend", "deepep", "--cuda-graph-max-bs-decode", "256", "--model-loader-extra-config", '{"enable_multithread_load": true, "num_threads": 64}', ] cls.process = popen_launch_server( cls.model, cls.base_url, timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, other_args=other_args, ) @classmethod def tearDownClass(cls): kill_process_tree(cls.process.pid) def test_a_gsm8k( self, ): # Append an "a" to make this test run first (alphabetically) to warm up the server args = SimpleNamespace( base_url=self.base_url, model=self.model, eval_name="gsm8k", api="completion", max_tokens=512, num_examples=1200, num_threads=1200, ) metrics = run_eval(args) print(f"{metrics=}") self.assertGreater(metrics["score"], 0.92) def test_bs_1_speed(self): args = BenchArgs(port=int(self.base_url.split(":")[-1]), max_new_tokens=2048) acc_length, speed = send_one_prompt(args) print(f"{speed=:.2f}") if __name__ == "__main__": unittest.main()