[Tiny] Enable Full Cuda Graph with Page size = 1 (#30835)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
978bce2063
commit
b44ac5d49a
@@ -1796,6 +1796,9 @@ def _fa4_page_constraint(view: Any) -> dict:
|
|||||||
# CUTLASS kernel aborts on at page_size>1. That path only works at
|
# CUTLASS kernel aborts on at page_size>1. That path only works at
|
||||||
# page_size==1, so skip the 128 auto-force for it and keep the default.
|
# page_size==1, so skip the 128 auto-force for it and keep the default.
|
||||||
and (view.speculative_eagle_topk or 0) <= 1
|
and (view.speculative_eagle_topk or 0) <= 1
|
||||||
|
# The full prefill CUDA graph runs the FA backend at page_size==1 only
|
||||||
|
# (#27988), so skip the 128 auto-force for it and keep the default.
|
||||||
|
and view.cuda_graph_config.prefill.backend != Backend.FULL
|
||||||
):
|
):
|
||||||
logger.warning(
|
logger.warning(
|
||||||
f"FA4 backend only supports page size 128 for non-MLA model architectures, changing page_size from {view.page_size} to 128."
|
f"FA4 backend only supports page size 128 for non-MLA model architectures, changing page_size from {view.page_size} to 128."
|
||||||
|
|||||||
Reference in New Issue
Block a user