Remove smoke wording from tests and comments (#23355)

This commit is contained in:
Xiaoyu Zhang
2026-04-28 12:05:27 +08:00
committed by GitHub
parent 1a55646dcd
commit 6fbad22feb
12 changed files with 17 additions and 17 deletions
@@ -276,7 +276,7 @@ python3 -m sglang.auto_benchmark validate \
- Tier 1
- Fastest and smallest sweep.
- Best for smoke tests, config validation, and quickly checking whether a model can run at all.
- Best for quick checks, config validation, and confirming whether a model can run at all.
- Uses a very small subset of the search space and mainly does one-at-a-time changes on top of the baseline.
- Lowest search cost, but also the easiest to miss a better configuration.
- Tier 2
+2 -2
View File
@@ -183,9 +183,9 @@ jobs:
cd sgl-kernel
pytest tests/
# Adding a single CUDA13 smoke test to verify that the kernel builds and runs
# Adding a single CUDA13 build-and-run check for the kernel
# TODO: Add back this test when it can pass on CI
# cuda13-kernel-smoke-test:
# cuda13-kernel-build-check:
# if: inputs.sgl_kernel == 'true'
# runs-on: x64-cu13-kernel-tests
# steps:
+1 -1
View File
@@ -269,7 +269,7 @@ def benchmark_rope_index(
seed=seed,
)
# Smoke test
# Validate output shapes before benchmarking.
has_mm = (image_grid_thw is not None) or (video_grid_thw is not None)
if has_mm:
pos, delta = MRotaryEmbedding.get_rope_index_glm4v(
@@ -593,7 +593,7 @@ export const DeepSeekV4Deployment = () => {
// headroom for DeepEP buffer + mooncake KV recv + CG private pool.
// Cookbook defaults (mem-frac 0.874, cg_max_bs 512, max-running 256)
// OOM during CG capture. mem-frac sweep at 0.83 / 0.87 / 0.89 / 0.91
// all pass static smoke; 0.9 picked as the default — leaves
// all pass static validation; 0.9 picked as the default — leaves
// ~14 GB / GPU post-CG headroom for mooncake transfer + activation
// peaks while giving ~1M-token KV pool.
if (isGB300 && modelSize === "big") {
@@ -69,7 +69,7 @@ Validated documentation and CI coverage currently center on six ModelOpt diffusi
Treat a new family, a new precision, or a new checkpoint layout as unsupported until it has a documented matrix row and a matching validation story.
Before writing CLI examples, re-read the active branch's `docs/diffusion/quantization.md`: FLUX.2 NVFP4 is an official `black-forest-labs/*` repo rather than a `BBuf/*` converted repo, and its preferred flag depends on the current documented loader flow. Use `--transformer-path` for a component override directory with `config.json`; use `--transformer-weights-path` when the repo or path should be probed as raw weights.
B200 CI coverage can include loose BF16-vs-quantized quality smoke checks. Inspect the active branch's `run_suite.py` before assuming they are part of the suite; mainline and feature branches may differ. Those checks are intended to catch blank, corrupted, or obviously divergent images, not exact image parity.
B200 CI coverage can include loose BF16-vs-quantized quality checks. Inspect the active branch's `run_suite.py` before assuming they are part of the suite; mainline and feature branches may differ. Those checks are intended to catch blank, corrupted, or obviously divergent images, not exact image parity.
## Documentation Maintenance
@@ -4,7 +4,7 @@ This tool runs two SGLang diffusion variants with the same prompt and seed,
captures intermediate denoising latents via `return_trajectory_latents`, and
reports cosine / error metrics for each timestep plus final frame metrics.
The intended use is quant validation on reduced deterministic smoke settings:
The intended use is quant validation with reduced deterministic settings:
- same prompt / seed / resolution / step count for both variants
- BF16 reference on the base model
- FP8 candidate via `--candidate-transformer-path` and/or component overrides
@@ -91,7 +91,7 @@ TEST_CASES = {
"batch_size": 1,
"max_seq_len": 32,
"page_size": 32,
"description": "Minimal smoke test",
"description": "Minimal sanity check",
},
{
"name": "batch",
@@ -731,7 +731,7 @@ class TestTRTLLMMLA(CustomTestCase):
self.assertFalse(torch.isinf(output).any(), "Output contains Inf")
def test_shape_sanity(self):
"""Smoke test decode across several configurations."""
"""Check decode shapes across several configurations."""
print(f"\nRunning shape sanity tests...")
for test_case in TEST_CASES["shape_sanity_tests"]:
+2 -2
View File
@@ -130,7 +130,7 @@ class TestMoriTransferEngineE2E(PDDisaggregationServerBase):
other_args=decode_args,
)
def test_generate_smoke(self):
def test_generate_basic(self):
resp = requests.post(
self.lb_url + "/generate",
json={
@@ -273,7 +273,7 @@ class TestMoriTransferEngineTPMismatchE2E(PDDisaggregationServerBase):
other_args=decode_args,
)
def test_generate_smoke_tp_mismatch(self):
def test_generate_with_tp_mismatch(self):
resp = requests.post(
self.lb_url + "/generate",
json={
@@ -285,7 +285,7 @@ class TestEntrypointGroupingRaw:
assert summary.total == 2
assert summary.passed == 2
def test_text_output_smoke(self, tmp_path, capsys):
def test_text_output_format(self, tmp_path, capsys):
"""Text output format renders without errors and contains Config/Summary sections."""
baseline_path, target_path = _create_dumps(tmp_path, ["tensor_a"])
argv = _make_argv(
@@ -1,7 +1,7 @@
"""Smoke test: intentionally trigger a CUDA illegal memory access
"""Intentionally trigger a CUDA illegal memory access
to verify the coredump collection pipeline works end-to-end.
Manual use: python3 test/registered/debug_utils/test_cuda_coredump_smoke.py
Manual use: python3 test/registered/debug_utils/test_cuda_coredump.py
"""
import unittest
@@ -17,7 +17,7 @@ register_cuda_ci(
)
class TestCudaCoredumpSmoke(unittest.TestCase):
class TestCudaCoredump(unittest.TestCase):
def test_trigger_illegal_memory_access(self):
x = torch.zeros(10, device="cuda")
y = torch.arange(10, device="cuda")
@@ -1,5 +1,5 @@
"""
E2E smoke test for HiCache storage runtime attach/detach.
E2E check for HiCache storage runtime attach/detach.
This test launches an SGLang server with hierarchical cache enabled but WITHOUT
any storage backend at startup, then attaches/detaches a storage backend via the
+1 -1
View File
@@ -581,7 +581,7 @@ class TestCuteDslV2(unittest.TestCase):
Also checks both match the pure-PyTorch reference, and that a second
cuda_graph pass reuses buffers deterministically (subsumes the former
cuda_graph_smoke test).
cuda_graph check).
"""
test_cases = [
# (num_tokens, hidden_size, intermediate_size, num_experts, top_k)