[Refactor] Tidy server_args.py section grouping and drop unused alias (#32496)

This commit is contained in:
Lianmin Zheng
2026-07-27 04:09:09 -07:00
committed by GitHub
parent 1d350aaad3
commit 34454c06b8
+24 -28
View File
@@ -132,30 +132,6 @@ LOAD_FORMAT_CHOICES = [
# TODO: this list should likely contain only methods that support online quantization, or that support using custom quantization classes compatible with a given `quant_method` in config.json.
# Some of the choices here do NOT support online quantization.
# Attention backends whose kernels read the chunked prefix-cache layout.
# Out-of-tree platforms may extend this list (via
# add_chunked_prefix_cache_attention_backend) before ServerArgs construction;
# the chunked-prefix gate is evaluated during resolution.
CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACKENDS = [
"flashinfer",
"fa3",
"fa4",
"flashmla",
"cutedsl_mla",
"cutlass_mla",
"trtllm_mla",
"tokenspeed_mla",
]
def add_chunked_prefix_cache_attention_backend(backend_name):
if backend_name not in CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACKENDS:
CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACKENDS.append(backend_name)
logger.info(
f"Added {backend_name} to CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACKENDS."
)
QUANTIZATION_CHOICES = [
"awq",
"fp8", # MOE + linear online quantization.
@@ -194,9 +170,6 @@ QUANTIZATION_CHOICES = [
"humming",
]
SPECULATIVE_DRAFT_MODEL_QUANTIZATION_CHOICES = QUANTIZATION_CHOICES
ATTENTION_BACKEND_CHOICES = [
# Common
"triton",
@@ -227,6 +200,21 @@ ATTENTION_BACKEND_CHOICES = [
"intel_xpu",
]
# Attention backends whose kernels read the chunked prefix-cache layout.
# Out-of-tree platforms may extend this list (via
# add_chunked_prefix_cache_attention_backend) before ServerArgs construction;
# the chunked-prefix gate is evaluated during resolution.
CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACKENDS = [
"flashinfer",
"fa3",
"fa4",
"flashmla",
"cutedsl_mla",
"cutlass_mla",
"trtllm_mla",
"tokenspeed_mla",
]
DETERMINISTIC_ATTENTION_BACKEND_CHOICES = [
"ascend",
"fa3",
@@ -372,6 +360,10 @@ def add_attention_backend_choices(choices):
ATTENTION_BACKEND_CHOICES.extend(choices)
def add_chunked_prefix_cache_attention_backend(backend_name):
CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACKENDS.append(backend_name)
def add_deterministic_attention_backend_choices(choices):
DETERMINISTIC_ATTENTION_BACKEND_CHOICES.extend(choices)
@@ -2140,7 +2132,7 @@ class ServerArgs:
Optional[str],
Arg(
help="The quantization method for speculative model.",
choices=SPECULATIVE_DRAFT_MODEL_QUANTIZATION_CHOICES,
choices=QUANTIZATION_CHOICES,
),
NS("spec"),
] = None
@@ -2203,6 +2195,7 @@ class ServerArgs:
NS("spec"),
] = None
# -------------------------------------------------------------------------
# Speculative decoding (ngram)
# -------------------------------------------------------------------------
speculative_ngram_min_bfs_breadth: A[
@@ -3336,6 +3329,9 @@ class ServerArgs:
NS("exec.features"),
] = False
# -------------------------------------------------------------------------
# Weight cache
# -------------------------------------------------------------------------
weight_cache_mode: A[
str,
Arg(