[NVIDIA] Support TF32 matmul to improve MiniMax gate gemm performance (#22744)

This commit is contained in:
Trevor Morris
2026-06-23 14:54:54 -07:00
committed by GitHub
parent c864c8d9c2
commit f74a1722e6
3 changed files with 24 additions and 0 deletions
@@ -533,6 +533,10 @@ class ModelRunner(ModelRunnerKVCacheMixin):
if self.device == "cpu":
self.init_threads_binding()
# Set float32 matmul precision
if server_args.enable_tf32_matmul:
torch.set_float32_matmul_precision("high")
# Get available memory before model loading.
# Stored for later use by alloc_memory_pool().
self.pre_model_load_memory = self.init_torch_distributed()