[misc] Use --cuda-graph-max-bs-decode in tests, examples, and docs (#29591)

This commit is contained in:
Liangsheng Yin
2026-06-28 18:38:28 -07:00
committed by GitHub
parent 3217410cf6
commit 909123ddb8
140 changed files with 304 additions and 290 deletions
+5 -5
View File
@@ -45,7 +45,7 @@ if OFFLINE_MODE:
# Default server arguments shared across all tests
DEFAULT_SERVER_ARGS = [
"--trust-remote-code",
"--cuda-graph-max-bs",
"--cuda-graph-max-bs-decode",
"8",
"--attention-backend",
"fa3",
@@ -144,7 +144,7 @@ class TestFlashAttention3SpeculativeDecode(BaseFlashAttentionTest):
args = DEFAULT_SERVER_ARGS
args.extend(
[
"--cuda-graph-max-bs",
"--cuda-graph-max-bs-decode",
"4",
"--speculative-algorithm",
"EAGLE3",
@@ -178,7 +178,7 @@ class TestFlashAttention3SpeculativeDecodeTopk(BaseFlashAttentionTest):
args = DEFAULT_SERVER_ARGS
args.extend(
[
"--cuda-graph-max-bs",
"--cuda-graph-max-bs-decode",
"4",
"--speculative-algorithm",
"EAGLE3",
@@ -210,7 +210,7 @@ class TestFlashAttention3MLASpeculativeDecode(BaseFlashAttentionTest):
args = DEFAULT_SERVER_ARGS
args.extend(
[
"--cuda-graph-max-bs",
"--cuda-graph-max-bs-decode",
"4",
"--speculative-algorithm",
"EAGLE",
@@ -242,7 +242,7 @@ class TestFlashAttention3MLASpeculativeDecodeTopk(BaseFlashAttentionTest):
args = DEFAULT_SERVER_ARGS
args.extend(
[
"--cuda-graph-max-bs",
"--cuda-graph-max-bs-decode",
"4",
"--speculative-algorithm",
"EAGLE",
+1 -1
View File
@@ -25,7 +25,7 @@ class TestFlashAttention3LocalAttn(CustomTestCase):
@classmethod
def get_server_args(cls):
return [
"--cuda-graph-max-bs",
"--cuda-graph-max-bs-decode",
"2",
"--attention-backend",
"fa3",