Remove smoke wording from tests and comments (#23355)
This commit is contained in:
@@ -276,7 +276,7 @@ python3 -m sglang.auto_benchmark validate \
|
|||||||
|
|
||||||
- Tier 1
|
- Tier 1
|
||||||
- Fastest and smallest sweep.
|
- Fastest and smallest sweep.
|
||||||
- Best for smoke tests, config validation, and quickly checking whether a model can run at all.
|
- Best for quick checks, config validation, and confirming whether a model can run at all.
|
||||||
- Uses a very small subset of the search space and mainly does one-at-a-time changes on top of the baseline.
|
- Uses a very small subset of the search space and mainly does one-at-a-time changes on top of the baseline.
|
||||||
- Lowest search cost, but also the easiest to miss a better configuration.
|
- Lowest search cost, but also the easiest to miss a better configuration.
|
||||||
- Tier 2
|
- Tier 2
|
||||||
|
|||||||
@@ -183,9 +183,9 @@ jobs:
|
|||||||
cd sgl-kernel
|
cd sgl-kernel
|
||||||
pytest tests/
|
pytest tests/
|
||||||
|
|
||||||
# Adding a single CUDA13 smoke test to verify that the kernel builds and runs
|
# Adding a single CUDA13 build-and-run check for the kernel
|
||||||
# TODO: Add back this test when it can pass on CI
|
# TODO: Add back this test when it can pass on CI
|
||||||
# cuda13-kernel-smoke-test:
|
# cuda13-kernel-build-check:
|
||||||
# if: inputs.sgl_kernel == 'true'
|
# if: inputs.sgl_kernel == 'true'
|
||||||
# runs-on: x64-cu13-kernel-tests
|
# runs-on: x64-cu13-kernel-tests
|
||||||
# steps:
|
# steps:
|
||||||
|
|||||||
@@ -269,7 +269,7 @@ def benchmark_rope_index(
|
|||||||
seed=seed,
|
seed=seed,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Smoke test
|
# Validate output shapes before benchmarking.
|
||||||
has_mm = (image_grid_thw is not None) or (video_grid_thw is not None)
|
has_mm = (image_grid_thw is not None) or (video_grid_thw is not None)
|
||||||
if has_mm:
|
if has_mm:
|
||||||
pos, delta = MRotaryEmbedding.get_rope_index_glm4v(
|
pos, delta = MRotaryEmbedding.get_rope_index_glm4v(
|
||||||
|
|||||||
@@ -593,7 +593,7 @@ export const DeepSeekV4Deployment = () => {
|
|||||||
// headroom for DeepEP buffer + mooncake KV recv + CG private pool.
|
// headroom for DeepEP buffer + mooncake KV recv + CG private pool.
|
||||||
// Cookbook defaults (mem-frac 0.874, cg_max_bs 512, max-running 256)
|
// Cookbook defaults (mem-frac 0.874, cg_max_bs 512, max-running 256)
|
||||||
// OOM during CG capture. mem-frac sweep at 0.83 / 0.87 / 0.89 / 0.91
|
// OOM during CG capture. mem-frac sweep at 0.83 / 0.87 / 0.89 / 0.91
|
||||||
// all pass static smoke; 0.9 picked as the default — leaves
|
// all pass static validation; 0.9 picked as the default — leaves
|
||||||
// ~14 GB / GPU post-CG headroom for mooncake transfer + activation
|
// ~14 GB / GPU post-CG headroom for mooncake transfer + activation
|
||||||
// peaks while giving ~1M-token KV pool.
|
// peaks while giving ~1M-token KV pool.
|
||||||
if (isGB300 && modelSize === "big") {
|
if (isGB300 && modelSize === "big") {
|
||||||
|
|||||||
+1
-1
@@ -69,7 +69,7 @@ Validated documentation and CI coverage currently center on six ModelOpt diffusi
|
|||||||
Treat a new family, a new precision, or a new checkpoint layout as unsupported until it has a documented matrix row and a matching validation story.
|
Treat a new family, a new precision, or a new checkpoint layout as unsupported until it has a documented matrix row and a matching validation story.
|
||||||
Before writing CLI examples, re-read the active branch's `docs/diffusion/quantization.md`: FLUX.2 NVFP4 is an official `black-forest-labs/*` repo rather than a `BBuf/*` converted repo, and its preferred flag depends on the current documented loader flow. Use `--transformer-path` for a component override directory with `config.json`; use `--transformer-weights-path` when the repo or path should be probed as raw weights.
|
Before writing CLI examples, re-read the active branch's `docs/diffusion/quantization.md`: FLUX.2 NVFP4 is an official `black-forest-labs/*` repo rather than a `BBuf/*` converted repo, and its preferred flag depends on the current documented loader flow. Use `--transformer-path` for a component override directory with `config.json`; use `--transformer-weights-path` when the repo or path should be probed as raw weights.
|
||||||
|
|
||||||
B200 CI coverage can include loose BF16-vs-quantized quality smoke checks. Inspect the active branch's `run_suite.py` before assuming they are part of the suite; mainline and feature branches may differ. Those checks are intended to catch blank, corrupted, or obviously divergent images, not exact image parity.
|
B200 CI coverage can include loose BF16-vs-quantized quality checks. Inspect the active branch's `run_suite.py` before assuming they are part of the suite; mainline and feature branches may differ. Those checks are intended to catch blank, corrupted, or obviously divergent images, not exact image parity.
|
||||||
|
|
||||||
## Documentation Maintenance
|
## Documentation Maintenance
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ This tool runs two SGLang diffusion variants with the same prompt and seed,
|
|||||||
captures intermediate denoising latents via `return_trajectory_latents`, and
|
captures intermediate denoising latents via `return_trajectory_latents`, and
|
||||||
reports cosine / error metrics for each timestep plus final frame metrics.
|
reports cosine / error metrics for each timestep plus final frame metrics.
|
||||||
|
|
||||||
The intended use is quant validation on reduced deterministic smoke settings:
|
The intended use is quant validation with reduced deterministic settings:
|
||||||
- same prompt / seed / resolution / step count for both variants
|
- same prompt / seed / resolution / step count for both variants
|
||||||
- BF16 reference on the base model
|
- BF16 reference on the base model
|
||||||
- FP8 candidate via `--candidate-transformer-path` and/or component overrides
|
- FP8 candidate via `--candidate-transformer-path` and/or component overrides
|
||||||
|
|||||||
@@ -91,7 +91,7 @@ TEST_CASES = {
|
|||||||
"batch_size": 1,
|
"batch_size": 1,
|
||||||
"max_seq_len": 32,
|
"max_seq_len": 32,
|
||||||
"page_size": 32,
|
"page_size": 32,
|
||||||
"description": "Minimal smoke test",
|
"description": "Minimal sanity check",
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "batch",
|
"name": "batch",
|
||||||
@@ -731,7 +731,7 @@ class TestTRTLLMMLA(CustomTestCase):
|
|||||||
self.assertFalse(torch.isinf(output).any(), "Output contains Inf")
|
self.assertFalse(torch.isinf(output).any(), "Output contains Inf")
|
||||||
|
|
||||||
def test_shape_sanity(self):
|
def test_shape_sanity(self):
|
||||||
"""Smoke test decode across several configurations."""
|
"""Check decode shapes across several configurations."""
|
||||||
print(f"\nRunning shape sanity tests...")
|
print(f"\nRunning shape sanity tests...")
|
||||||
|
|
||||||
for test_case in TEST_CASES["shape_sanity_tests"]:
|
for test_case in TEST_CASES["shape_sanity_tests"]:
|
||||||
|
|||||||
@@ -130,7 +130,7 @@ class TestMoriTransferEngineE2E(PDDisaggregationServerBase):
|
|||||||
other_args=decode_args,
|
other_args=decode_args,
|
||||||
)
|
)
|
||||||
|
|
||||||
def test_generate_smoke(self):
|
def test_generate_basic(self):
|
||||||
resp = requests.post(
|
resp = requests.post(
|
||||||
self.lb_url + "/generate",
|
self.lb_url + "/generate",
|
||||||
json={
|
json={
|
||||||
@@ -273,7 +273,7 @@ class TestMoriTransferEngineTPMismatchE2E(PDDisaggregationServerBase):
|
|||||||
other_args=decode_args,
|
other_args=decode_args,
|
||||||
)
|
)
|
||||||
|
|
||||||
def test_generate_smoke_tp_mismatch(self):
|
def test_generate_with_tp_mismatch(self):
|
||||||
resp = requests.post(
|
resp = requests.post(
|
||||||
self.lb_url + "/generate",
|
self.lb_url + "/generate",
|
||||||
json={
|
json={
|
||||||
|
|||||||
@@ -285,7 +285,7 @@ class TestEntrypointGroupingRaw:
|
|||||||
assert summary.total == 2
|
assert summary.total == 2
|
||||||
assert summary.passed == 2
|
assert summary.passed == 2
|
||||||
|
|
||||||
def test_text_output_smoke(self, tmp_path, capsys):
|
def test_text_output_format(self, tmp_path, capsys):
|
||||||
"""Text output format renders without errors and contains Config/Summary sections."""
|
"""Text output format renders without errors and contains Config/Summary sections."""
|
||||||
baseline_path, target_path = _create_dumps(tmp_path, ["tensor_a"])
|
baseline_path, target_path = _create_dumps(tmp_path, ["tensor_a"])
|
||||||
argv = _make_argv(
|
argv = _make_argv(
|
||||||
|
|||||||
+3
-3
@@ -1,7 +1,7 @@
|
|||||||
"""Smoke test: intentionally trigger a CUDA illegal memory access
|
"""Intentionally trigger a CUDA illegal memory access
|
||||||
to verify the coredump collection pipeline works end-to-end.
|
to verify the coredump collection pipeline works end-to-end.
|
||||||
|
|
||||||
Manual use: python3 test/registered/debug_utils/test_cuda_coredump_smoke.py
|
Manual use: python3 test/registered/debug_utils/test_cuda_coredump.py
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import unittest
|
import unittest
|
||||||
@@ -17,7 +17,7 @@ register_cuda_ci(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class TestCudaCoredumpSmoke(unittest.TestCase):
|
class TestCudaCoredump(unittest.TestCase):
|
||||||
def test_trigger_illegal_memory_access(self):
|
def test_trigger_illegal_memory_access(self):
|
||||||
x = torch.zeros(10, device="cuda")
|
x = torch.zeros(10, device="cuda")
|
||||||
y = torch.arange(10, device="cuda")
|
y = torch.arange(10, device="cuda")
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
"""
|
"""
|
||||||
E2E smoke test for HiCache storage runtime attach/detach.
|
E2E check for HiCache storage runtime attach/detach.
|
||||||
|
|
||||||
This test launches an SGLang server with hierarchical cache enabled but WITHOUT
|
This test launches an SGLang server with hierarchical cache enabled but WITHOUT
|
||||||
any storage backend at startup, then attaches/detaches a storage backend via the
|
any storage backend at startup, then attaches/detaches a storage backend via the
|
||||||
|
|||||||
@@ -581,7 +581,7 @@ class TestCuteDslV2(unittest.TestCase):
|
|||||||
|
|
||||||
Also checks both match the pure-PyTorch reference, and that a second
|
Also checks both match the pure-PyTorch reference, and that a second
|
||||||
cuda_graph pass reuses buffers deterministically (subsumes the former
|
cuda_graph pass reuses buffers deterministically (subsumes the former
|
||||||
cuda_graph_smoke test).
|
cuda_graph check).
|
||||||
"""
|
"""
|
||||||
test_cases = [
|
test_cases = [
|
||||||
# (num_tokens, hidden_size, intermediate_size, num_experts, top_k)
|
# (num_tokens, hidden_size, intermediate_size, num_experts, top_k)
|
||||||
|
|||||||
Reference in New Issue
Block a user