[diffusion] chore: remove retired auto-residency workload helpers (#39884)

Co-authored-by: Mick Qian <mickqian@users.noreply.github.com>
This commit is contained in:
Mick
2026-09-17 15:16:35 +08:00
committed by GitHub
co-authored by Mick Qian
parent 44bd359082
commit f3c0256771
2 changed files with 1 additions and 125 deletions
@@ -25,7 +25,7 @@ single placement to serve two different lifecycle objectives.
from __future__ import annotations
import statistics
from typing import TYPE_CHECKING, Iterable, Mapping
from typing import Iterable, Mapping
import msgspec
@@ -39,9 +39,6 @@ from sglang.multimodal_gen.runtime.managers.memory_managers.layerwise_offload_co
)
from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger
if TYPE_CHECKING:
from sglang.multimodal_gen.runtime.server_args import ServerArgs
logger = init_logger(__name__)
GIB_BYTES = 1024**3
@@ -181,25 +178,6 @@ class ResidencyTarget(msgspec.Struct, frozen=True):
return f"{self.component_name}:{permanence}:layers={layer_counts}:pins={pinned}"
class DefaultWorkload(msgspec.Struct, frozen=True):
"""The model-default request shape the planner is calibrated for."""
width: int | None
height: int | None
num_frames: int
num_inference_steps: int
def workload_units(self) -> int | None:
if self.width is None or self.height is None:
return None
return max(1, self.width) * max(1, self.height) * max(1, self.num_frames)
def describe(self) -> str:
if self.width is None or self.height is None:
return "model-default"
return f"{self.width}x{self.height}x{self.num_frames}f"
class RankResidencyReport(msgspec.Struct, frozen=True):
"""One rank's inputs to the replica-wide placement decision."""
@@ -225,52 +203,6 @@ class RankResidencyReport(msgspec.Struct, frozen=True):
skip_reason: str | None = None
def resolve_default_workload(server_args: ServerArgs) -> DefaultWorkload:
"""Resolve the default request shape the planner is optimized for."""
from sglang.multimodal_gen.runtime.warmup_request_builder import (
get_model_sampling_defaults,
resolve_default_workload_shape,
)
defaults = get_model_sampling_defaults(server_args)
width, height, num_frames = resolve_default_workload_shape(server_args, defaults)
return DefaultWorkload(
width=width,
height=height,
num_frames=num_frames,
num_inference_steps=defaults.num_inference_steps or 1,
)
def resolve_measured_default_workload(
workload: DefaultWorkload, records: Iterable[WarmupMemoryRecord]
) -> DefaultWorkload:
"""Fill an implicit default resolution from the executed warmup.
Image-edit pipelines can derive their output size from the input image, so
the sampling defaults legitimately omit width and height. The warmup record
is captured after input validation and therefore contains the effective
serving shape. Keep the model-default frame count because video warmup may
intentionally cap frames before measurement.
"""
if workload.workload_units() is not None:
return workload
measured = [
record
for record in records
if record.succeeded and record.width > 0 and record.height > 0
]
if not measured:
return workload
representative = max(measured, key=lambda record: record.width * record.height)
return DefaultWorkload(
width=representative.width,
height=representative.height,
num_frames=workload.num_frames,
num_inference_steps=workload.num_inference_steps,
)
def estimate_layerwise_layer_uses(
*,
records: Iterable[WarmupMemoryRecord],
@@ -10,13 +10,11 @@ from sglang.multimodal_gen.configs.pipeline_configs.longlive2 import LongLive2T2
from sglang.multimodal_gen.runtime.managers.memory_managers.auto_residency import (
ACTIVATION_EXTRAPOLATION_MARGIN,
GIB_BYTES,
DefaultWorkload,
WarmupMemoryRecord,
estimate_default_workload_peak_bytes,
estimate_default_workload_timing,
estimate_layerwise_layer_uses,
estimate_workload_phase_peaks,
resolve_measured_default_workload,
)
from sglang.multimodal_gen.runtime.warmup_request_builder import (
SERVER_WARMUP_MAX_VIDEO_FRAMES,
@@ -579,60 +577,6 @@ class TestEstimateDefaultWorkloadPeak:
assert used == active
class TestResolveMeasuredDefaultWorkload:
def test_uses_effective_warmup_resolution_for_implicit_image_size(self):
workload = DefaultWorkload(
width=None,
height=None,
num_frames=1,
num_inference_steps=40,
)
resolved = resolve_measured_default_workload(
workload,
[
_record(width=512, height=512),
_record(width=1024, height=1024),
],
)
assert resolved == DefaultWorkload(
width=1024,
height=1024,
num_frames=1,
num_inference_steps=40,
)
def test_keeps_default_frames_when_warmup_caps_video(self):
workload = DefaultWorkload(
width=None,
height=None,
num_frames=81,
num_inference_steps=30,
)
resolved = resolve_measured_default_workload(
workload, [_record(width=832, height=480, num_frames=17)]
)
assert resolved.num_frames == 81
def test_does_not_replace_explicit_default_shape(self):
workload = DefaultWorkload(
width=1280,
height=720,
num_frames=81,
num_inference_steps=30,
)
assert (
resolve_measured_default_workload(
workload, [_record(width=512, height=512)]
)
is workload
)
class TestWarmupFrameAdjustment:
def _server_args(self, *, bcg: bool = False, num_gpus: int = 1) -> SimpleNamespace:
return SimpleNamespace(