Respect user override for Gemma4 attention backend (#25547)
Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.7
parent
f21fe6ad4d
commit
b29e41e8b3
@@ -2195,12 +2195,29 @@ class ServerArgs:
|
||||
)
|
||||
self.disable_hybrid_swa_memory = True
|
||||
elif model_arch == "Gemma4ForConditionalGeneration":
|
||||
if is_sm100_supported():
|
||||
self.attention_backend = "trtllm_mha"
|
||||
default_attention_backend = (
|
||||
"trtllm_mha" if is_sm100_supported() else "triton"
|
||||
)
|
||||
if self.is_attention_backend_not_set():
|
||||
self.attention_backend = default_attention_backend
|
||||
logger.info(
|
||||
f"Use {self.attention_backend} as default attention backend for Gemma4"
|
||||
)
|
||||
else:
|
||||
self.attention_backend = "triton"
|
||||
logger.info(
|
||||
f"Use {self.attention_backend} as default attention backend for Gemma4"
|
||||
# If only one split backend is set, keep the other side on a
|
||||
# Gemma4-compatible fallback instead of letting generic backend
|
||||
# selection choose an unsupported backend later.
|
||||
if self.attention_backend is None:
|
||||
self.attention_backend = default_attention_backend
|
||||
|
||||
prefill_backend, decode_backend = self.get_attention_backends()
|
||||
accepted_backends = ("trtllm_mha", "triton")
|
||||
assert (
|
||||
prefill_backend in accepted_backends
|
||||
and decode_backend in accepted_backends
|
||||
), (
|
||||
"Gemma4 only supports trtllm_mha or triton attention backend, "
|
||||
f"got prefill={prefill_backend}, decode={decode_backend}"
|
||||
)
|
||||
elif model_arch == "MossVLForConditionalGeneration":
|
||||
if self.is_attention_backend_not_set():
|
||||
|
||||
Reference in New Issue
Block a user