Remove smoke wording from tests and comments (#23355)

This commit is contained in:
Xiaoyu Zhang
2026-04-28 12:05:27 +08:00
committed by GitHub
parent 1a55646dcd
commit 6fbad22feb
12 changed files with 17 additions and 17 deletions
@@ -276,7 +276,7 @@ python3 -m sglang.auto_benchmark validate \
- Tier 1 - Tier 1
- Fastest and smallest sweep. - Fastest and smallest sweep.
- Best for smoke tests, config validation, and quickly checking whether a model can run at all. - Best for quick checks, config validation, and confirming whether a model can run at all.
- Uses a very small subset of the search space and mainly does one-at-a-time changes on top of the baseline. - Uses a very small subset of the search space and mainly does one-at-a-time changes on top of the baseline.
- Lowest search cost, but also the easiest to miss a better configuration. - Lowest search cost, but also the easiest to miss a better configuration.
- Tier 2 - Tier 2
+2 -2
View File
@@ -183,9 +183,9 @@ jobs:
cd sgl-kernel cd sgl-kernel
pytest tests/ pytest tests/
# Adding a single CUDA13 smoke test to verify that the kernel builds and runs # Adding a single CUDA13 build-and-run check for the kernel
# TODO: Add back this test when it can pass on CI # TODO: Add back this test when it can pass on CI
# cuda13-kernel-smoke-test: # cuda13-kernel-build-check:
# if: inputs.sgl_kernel == 'true' # if: inputs.sgl_kernel == 'true'
# runs-on: x64-cu13-kernel-tests # runs-on: x64-cu13-kernel-tests
# steps: # steps:
+1 -1
View File
@@ -269,7 +269,7 @@ def benchmark_rope_index(
seed=seed, seed=seed,
) )
# Smoke test # Validate output shapes before benchmarking.
has_mm = (image_grid_thw is not None) or (video_grid_thw is not None) has_mm = (image_grid_thw is not None) or (video_grid_thw is not None)
if has_mm: if has_mm:
pos, delta = MRotaryEmbedding.get_rope_index_glm4v( pos, delta = MRotaryEmbedding.get_rope_index_glm4v(
@@ -593,7 +593,7 @@ export const DeepSeekV4Deployment = () => {
// headroom for DeepEP buffer + mooncake KV recv + CG private pool. // headroom for DeepEP buffer + mooncake KV recv + CG private pool.
// Cookbook defaults (mem-frac 0.874, cg_max_bs 512, max-running 256) // Cookbook defaults (mem-frac 0.874, cg_max_bs 512, max-running 256)
// OOM during CG capture. mem-frac sweep at 0.83 / 0.87 / 0.89 / 0.91 // OOM during CG capture. mem-frac sweep at 0.83 / 0.87 / 0.89 / 0.91
// all pass static smoke; 0.9 picked as the default — leaves // all pass static validation; 0.9 picked as the default — leaves
// ~14 GB / GPU post-CG headroom for mooncake transfer + activation // ~14 GB / GPU post-CG headroom for mooncake transfer + activation
// peaks while giving ~1M-token KV pool. // peaks while giving ~1M-token KV pool.
if (isGB300 && modelSize === "big") { if (isGB300 && modelSize === "big") {
@@ -69,7 +69,7 @@ Validated documentation and CI coverage currently center on six ModelOpt diffusi
Treat a new family, a new precision, or a new checkpoint layout as unsupported until it has a documented matrix row and a matching validation story. Treat a new family, a new precision, or a new checkpoint layout as unsupported until it has a documented matrix row and a matching validation story.
Before writing CLI examples, re-read the active branch's `docs/diffusion/quantization.md`: FLUX.2 NVFP4 is an official `black-forest-labs/*` repo rather than a `BBuf/*` converted repo, and its preferred flag depends on the current documented loader flow. Use `--transformer-path` for a component override directory with `config.json`; use `--transformer-weights-path` when the repo or path should be probed as raw weights. Before writing CLI examples, re-read the active branch's `docs/diffusion/quantization.md`: FLUX.2 NVFP4 is an official `black-forest-labs/*` repo rather than a `BBuf/*` converted repo, and its preferred flag depends on the current documented loader flow. Use `--transformer-path` for a component override directory with `config.json`; use `--transformer-weights-path` when the repo or path should be probed as raw weights.
B200 CI coverage can include loose BF16-vs-quantized quality smoke checks. Inspect the active branch's `run_suite.py` before assuming they are part of the suite; mainline and feature branches may differ. Those checks are intended to catch blank, corrupted, or obviously divergent images, not exact image parity. B200 CI coverage can include loose BF16-vs-quantized quality checks. Inspect the active branch's `run_suite.py` before assuming they are part of the suite; mainline and feature branches may differ. Those checks are intended to catch blank, corrupted, or obviously divergent images, not exact image parity.
## Documentation Maintenance ## Documentation Maintenance
@@ -4,7 +4,7 @@ This tool runs two SGLang diffusion variants with the same prompt and seed,
captures intermediate denoising latents via `return_trajectory_latents`, and captures intermediate denoising latents via `return_trajectory_latents`, and
reports cosine / error metrics for each timestep plus final frame metrics. reports cosine / error metrics for each timestep plus final frame metrics.
The intended use is quant validation on reduced deterministic smoke settings: The intended use is quant validation with reduced deterministic settings:
- same prompt / seed / resolution / step count for both variants - same prompt / seed / resolution / step count for both variants
- BF16 reference on the base model - BF16 reference on the base model
- FP8 candidate via `--candidate-transformer-path` and/or component overrides - FP8 candidate via `--candidate-transformer-path` and/or component overrides
@@ -91,7 +91,7 @@ TEST_CASES = {
"batch_size": 1, "batch_size": 1,
"max_seq_len": 32, "max_seq_len": 32,
"page_size": 32, "page_size": 32,
"description": "Minimal smoke test", "description": "Minimal sanity check",
}, },
{ {
"name": "batch", "name": "batch",
@@ -731,7 +731,7 @@ class TestTRTLLMMLA(CustomTestCase):
self.assertFalse(torch.isinf(output).any(), "Output contains Inf") self.assertFalse(torch.isinf(output).any(), "Output contains Inf")
def test_shape_sanity(self): def test_shape_sanity(self):
"""Smoke test decode across several configurations.""" """Check decode shapes across several configurations."""
print(f"\nRunning shape sanity tests...") print(f"\nRunning shape sanity tests...")
for test_case in TEST_CASES["shape_sanity_tests"]: for test_case in TEST_CASES["shape_sanity_tests"]:
+2 -2
View File
@@ -130,7 +130,7 @@ class TestMoriTransferEngineE2E(PDDisaggregationServerBase):
other_args=decode_args, other_args=decode_args,
) )
def test_generate_smoke(self): def test_generate_basic(self):
resp = requests.post( resp = requests.post(
self.lb_url + "/generate", self.lb_url + "/generate",
json={ json={
@@ -273,7 +273,7 @@ class TestMoriTransferEngineTPMismatchE2E(PDDisaggregationServerBase):
other_args=decode_args, other_args=decode_args,
) )
def test_generate_smoke_tp_mismatch(self): def test_generate_with_tp_mismatch(self):
resp = requests.post( resp = requests.post(
self.lb_url + "/generate", self.lb_url + "/generate",
json={ json={
@@ -285,7 +285,7 @@ class TestEntrypointGroupingRaw:
assert summary.total == 2 assert summary.total == 2
assert summary.passed == 2 assert summary.passed == 2
def test_text_output_smoke(self, tmp_path, capsys): def test_text_output_format(self, tmp_path, capsys):
"""Text output format renders without errors and contains Config/Summary sections.""" """Text output format renders without errors and contains Config/Summary sections."""
baseline_path, target_path = _create_dumps(tmp_path, ["tensor_a"]) baseline_path, target_path = _create_dumps(tmp_path, ["tensor_a"])
argv = _make_argv( argv = _make_argv(
@@ -1,7 +1,7 @@
"""Smoke test: intentionally trigger a CUDA illegal memory access """Intentionally trigger a CUDA illegal memory access
to verify the coredump collection pipeline works end-to-end. to verify the coredump collection pipeline works end-to-end.
Manual use: python3 test/registered/debug_utils/test_cuda_coredump_smoke.py Manual use: python3 test/registered/debug_utils/test_cuda_coredump.py
""" """
import unittest import unittest
@@ -17,7 +17,7 @@ register_cuda_ci(
) )
class TestCudaCoredumpSmoke(unittest.TestCase): class TestCudaCoredump(unittest.TestCase):
def test_trigger_illegal_memory_access(self): def test_trigger_illegal_memory_access(self):
x = torch.zeros(10, device="cuda") x = torch.zeros(10, device="cuda")
y = torch.arange(10, device="cuda") y = torch.arange(10, device="cuda")
@@ -1,5 +1,5 @@
""" """
E2E smoke test for HiCache storage runtime attach/detach. E2E check for HiCache storage runtime attach/detach.
This test launches an SGLang server with hierarchical cache enabled but WITHOUT This test launches an SGLang server with hierarchical cache enabled but WITHOUT
any storage backend at startup, then attaches/detaches a storage backend via the any storage backend at startup, then attaches/detaches a storage backend via the
+1 -1
View File
@@ -581,7 +581,7 @@ class TestCuteDslV2(unittest.TestCase):
Also checks both match the pure-PyTorch reference, and that a second Also checks both match the pure-PyTorch reference, and that a second
cuda_graph pass reuses buffers deterministically (subsumes the former cuda_graph pass reuses buffers deterministically (subsumes the former
cuda_graph_smoke test). cuda_graph check).
""" """
test_cases = [ test_cases = [
# (num_tokens, hidden_size, intermediate_size, num_experts, top_k) # (num_tokens, hidden_size, intermediate_size, num_experts, top_k)