[Router] Add --worker-queue-limit: stop sending cache-affinity traffic to a queueing worker (#39168)
Co-authored-by: Kangyan Zhou <kangyan.zhou@radixark.ai> Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Kangyan Zhou
Claude Opus 5
parent
3f8eb35ead
commit
c1f5b4736a
@@ -163,6 +163,16 @@ pub struct Cli {
|
|||||||
/// Maximum uncached-work difference that pressure may override.
|
/// Maximum uncached-work difference that pressure may override.
|
||||||
#[arg(long)]
|
#[arg(long)]
|
||||||
pub cache_switch_margin_tokens: Option<u64>,
|
pub cache_switch_margin_tokens: Option<u64>,
|
||||||
|
/// Queue gate for cache affinity: a worker whose engine reports at least
|
||||||
|
/// this many waiting (queued) requests cannot win a selection on cache
|
||||||
|
/// affinity. The request goes to another worker holding the same prefix,
|
||||||
|
/// or failing that to the least-loaded worker that is not queueing; when
|
||||||
|
/// every worker is queueing the least-loaded worker overall keeps the
|
||||||
|
/// fleet routable. Unset disables the gate. Requires
|
||||||
|
/// `--policy cache_aware`; scale with the engine's `--dp-size` because
|
||||||
|
/// the published queue sums across a worker's DP ranks.
|
||||||
|
#[arg(long)]
|
||||||
|
pub worker_queue_limit: Option<u64>,
|
||||||
|
|
||||||
// ---- score composition ----
|
// ---- score composition ----
|
||||||
/// Policies to sum, spelled exactly as `--policy` spells them and each
|
/// Policies to sum, spelled exactly as `--policy` spells them and each
|
||||||
@@ -349,7 +359,13 @@ impl Cli {
|
|||||||
|| self.cache_candidate_min_workers.is_some()
|
|| self.cache_candidate_min_workers.is_some()
|
||||||
|| self.cache_candidate_ratio.is_some()
|
|| self.cache_candidate_ratio.is_some()
|
||||||
|| self.cache_candidate_max_workers.is_some()
|
|| self.cache_candidate_max_workers.is_some()
|
||||||
|| self.cache_switch_margin_tokens.is_some();
|
|| self.cache_switch_margin_tokens.is_some()
|
||||||
|
|| self.worker_queue_limit.is_some();
|
||||||
|
// Value checks before the policy check: a value that is wrong under
|
||||||
|
// every policy should say so, rather than pointing at --policy.
|
||||||
|
if self.worker_queue_limit == Some(0) {
|
||||||
|
return Err(anyhow!("--worker-queue-limit must be at least 1"));
|
||||||
|
}
|
||||||
if tuned_cache_candidates && self.policy != PolicyKind::CacheAware {
|
if tuned_cache_candidates && self.policy != PolicyKind::CacheAware {
|
||||||
return Err(anyhow!(
|
return Err(anyhow!(
|
||||||
"cache candidate tuning flags require --policy cache_aware"
|
"cache candidate tuning flags require --policy cache_aware"
|
||||||
@@ -573,6 +589,7 @@ impl Cli {
|
|||||||
cache_switch_margin_tokens: self
|
cache_switch_margin_tokens: self
|
||||||
.cache_switch_margin_tokens
|
.cache_switch_margin_tokens
|
||||||
.unwrap_or(d.cache_switch_margin_tokens),
|
.unwrap_or(d.cache_switch_margin_tokens),
|
||||||
|
worker_queue_limit: self.worker_queue_limit.or(d.worker_queue_limit),
|
||||||
})
|
})
|
||||||
} else {
|
} else {
|
||||||
None
|
None
|
||||||
@@ -1817,6 +1834,58 @@ mod tests {
|
|||||||
assert!(err.contains("--stable-pair"), "got: {err}");
|
assert!(err.contains("--stable-pair"), "got: {err}");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn worker_queue_limit_requires_cache_aware_and_a_positive_value() {
|
||||||
|
let config = cfg_of(
|
||||||
|
"--policy cache_aware --kv-indexer-endpoint http://indexer:50051 \
|
||||||
|
--worker-queue-limit 4",
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
config
|
||||||
|
.model
|
||||||
|
.affinity
|
||||||
|
.expect("cache-aware needs affinity config")
|
||||||
|
.worker_queue_limit,
|
||||||
|
Some(4)
|
||||||
|
);
|
||||||
|
|
||||||
|
// Unset, the gate is disabled.
|
||||||
|
let defaults =
|
||||||
|
cfg_of("--policy cache_aware --kv-indexer-endpoint http://indexer:50051").unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
defaults
|
||||||
|
.model
|
||||||
|
.affinity
|
||||||
|
.expect("default affinity config")
|
||||||
|
.worker_queue_limit,
|
||||||
|
None
|
||||||
|
);
|
||||||
|
|
||||||
|
let err = cfg_of("--policy power_of_two --worker-queue-limit 4")
|
||||||
|
.expect_err("the gate only governs cache-affinity selection")
|
||||||
|
.to_string();
|
||||||
|
assert!(
|
||||||
|
err.contains("cache candidate tuning flags require --policy cache_aware"),
|
||||||
|
"got: {err}"
|
||||||
|
);
|
||||||
|
|
||||||
|
let err = cfg_of(
|
||||||
|
"--policy cache_aware --kv-indexer-endpoint http://indexer:50051 \
|
||||||
|
--worker-queue-limit 0",
|
||||||
|
)
|
||||||
|
.expect_err("a zero limit would reject every queue reading")
|
||||||
|
.to_string();
|
||||||
|
assert!(err.contains("--worker-queue-limit"), "got: {err}");
|
||||||
|
|
||||||
|
// A zero limit is wrong under every policy, so the value error must
|
||||||
|
// win over the policy error rather than being masked by it.
|
||||||
|
let err = cfg_of("--policy power_of_two --worker-queue-limit 0")
|
||||||
|
.expect_err("a zero limit is rejected regardless of policy")
|
||||||
|
.to_string();
|
||||||
|
assert!(err.contains("--worker-queue-limit"), "got: {err}");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn cache_candidate_cli_rejects_invalid_bounds() {
|
fn cache_candidate_cli_rejects_invalid_bounds() {
|
||||||
for (args, expected) in [
|
for (args, expected) in [
|
||||||
|
|||||||
@@ -453,6 +453,29 @@ pub struct AffinityConfig {
|
|||||||
pub cache_candidate_ratio: f64,
|
pub cache_candidate_ratio: f64,
|
||||||
pub cache_candidate_max_workers: usize,
|
pub cache_candidate_max_workers: usize,
|
||||||
pub cache_switch_margin_tokens: u64,
|
pub cache_switch_margin_tokens: u64,
|
||||||
|
/// Queue gate (`--worker-queue-limit`): a worker whose engine reports at
|
||||||
|
/// least this many *waiting* requests cannot win a selection on cache
|
||||||
|
/// affinity — the request goes to another worker holding the same
|
||||||
|
/// prefix, or failing that to the least-loaded worker that is not
|
||||||
|
/// queueing. `None` disables the gate.
|
||||||
|
///
|
||||||
|
/// Gating on the queue rather than on total depth is what makes this
|
||||||
|
/// targeted: `num_waiting_reqs` IS the question the request cares about
|
||||||
|
/// — will I sit behind other work before my prefill starts — whereas
|
||||||
|
/// depth only proxies it, and proxies it badly (an engine can queue at
|
||||||
|
/// 7-8 running on long-prompt traffic, far below its running cap, so a
|
||||||
|
/// depth threshold either fires on healthy busy workers or misses the
|
||||||
|
/// workers actually making requests wait).
|
||||||
|
///
|
||||||
|
/// The gate reads the engine-published load sample and fails OPEN on a
|
||||||
|
/// worker with no fresh sample: the router-side in-flight counter cannot
|
||||||
|
/// separate a running request from a waiting one, so there is no honest
|
||||||
|
/// substitute to compare the limit against.
|
||||||
|
///
|
||||||
|
/// Note the firing point scales with `dp_size`: the sample sums `waiting`
|
||||||
|
/// across a worker's DP ranks while a request lands on one of them, so
|
||||||
|
/// scale the limit with `--dp-size` on DP-attention deployments.
|
||||||
|
pub worker_queue_limit: Option<u64>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for AffinityConfig {
|
impl Default for AffinityConfig {
|
||||||
@@ -475,6 +498,7 @@ impl Default for AffinityConfig {
|
|||||||
cache_candidate_ratio: 0.05,
|
cache_candidate_ratio: 0.05,
|
||||||
cache_candidate_max_workers: 32,
|
cache_candidate_max_workers: 32,
|
||||||
cache_switch_margin_tokens: 1_024,
|
cache_switch_margin_tokens: 1_024,
|
||||||
|
worker_queue_limit: None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,8 +6,13 @@
|
|||||||
//! Native Cache-Aware uses monitor-backed admission only when every expected
|
//! Native Cache-Aware uses monitor-backed admission only when every expected
|
||||||
//! DP rank has a fresh, complete #34608 ZMQ sample. Otherwise it falls back to
|
//! DP rank has a fresh, complete #34608 ZMQ sample. Otherwise it falls back to
|
||||||
//! Router-local load.
|
//! Router-local load.
|
||||||
|
//!
|
||||||
|
//! The optional queue gate (`--worker-queue-limit`) is the one criterion here
|
||||||
|
//! that reads [`EngineWorkerLoad::num_waiting_reqs`] rather than the native
|
||||||
|
//! monitor fields: a worker already making requests wait cannot win on cache
|
||||||
|
//! affinity. It fails open on a missing sample — see [`queue_gate_admits`].
|
||||||
|
|
||||||
use crate::policies::engine_load::{EngineLoadSnapshot, NativeCacheWorkerLoad};
|
use crate::policies::engine_load::{EngineLoadSnapshot, EngineWorkerLoad, NativeCacheWorkerLoad};
|
||||||
use crate::policies::power_of_two::select_with_snapshot;
|
use crate::policies::power_of_two::select_with_snapshot;
|
||||||
use crate::policies::{CacheCandidate, CacheCandidateProposal, GuardHints, SelectionProposal};
|
use crate::policies::{CacheCandidate, CacheCandidateProposal, GuardHints, SelectionProposal};
|
||||||
use crate::workers::Worker;
|
use crate::workers::Worker;
|
||||||
@@ -124,32 +129,125 @@ pub struct FinalDecision {
|
|||||||
pub struct CacheCandidateResolution {
|
pub struct CacheCandidateResolution {
|
||||||
pub decision: Option<FinalDecision>,
|
pub decision: Option<FinalDecision>,
|
||||||
pub prefill_pressure_source: &'static str,
|
pub prefill_pressure_source: &'static str,
|
||||||
|
/// Candidates actually put through KV-capacity / pending-prefill
|
||||||
|
/// admission. Queue-gate rejections are excluded: the gate runs first and
|
||||||
|
/// they are never evaluated, so counting them here would silently deflate
|
||||||
|
/// the rejected/evaluated ratio whenever the gate is armed.
|
||||||
pub admission_evaluated_candidates: u64,
|
pub admission_evaluated_candidates: u64,
|
||||||
|
/// Candidates rejected by KV-capacity / pending-prefill admission. Does
|
||||||
|
/// not include queue-gate rejections — those are counted separately so a
|
||||||
|
/// busy fleet never reads as a capacity problem.
|
||||||
pub admission_rejected_candidates: u64,
|
pub admission_rejected_candidates: u64,
|
||||||
|
/// Candidates rejected by the queue gate: their engine queue is at or
|
||||||
|
/// over `--worker-queue-limit`. The gate runs BEFORE hard admission, so
|
||||||
|
/// these candidates were never evaluated against capacity and are
|
||||||
|
/// disjoint from `admission_rejected_candidates` by construction.
|
||||||
|
pub queue_gate_rejected_candidates: u64,
|
||||||
|
/// Deepest matched prefix, in blocks, among the queue-gate-rejected
|
||||||
|
/// candidates — the locality a diversion gave up. 0 when the gate
|
||||||
|
/// rejected nothing.
|
||||||
|
pub queue_gate_best_rejected_blocks: u32,
|
||||||
|
/// True when the gate removed EVERY candidate and nowhere in the fleet is
|
||||||
|
/// unqueued, so the winner came from the ungated candidate set instead.
|
||||||
|
/// The mirror of `range_fallback`'s second tier: diversion buys nothing
|
||||||
|
/// here, so discarding the prefix would be pure loss.
|
||||||
|
pub queue_gate_fell_back: bool,
|
||||||
|
/// True when no worker in the fleet has room before the gate. Computed
|
||||||
|
/// once here so the route handler does not re-derive the gate over the
|
||||||
|
/// fleet; always false when the gate is disabled.
|
||||||
|
pub fleet_all_queued: bool,
|
||||||
pub pressure_guard_compared_pairs: u64,
|
pub pressure_guard_compared_pairs: u64,
|
||||||
pub pressure_guard_overrides: u64,
|
pub pressure_guard_overrides: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The queue gate, in one place. Both subtle decisions live here: the
|
||||||
|
/// boundary is `<` (a worker AT the limit is already making this request
|
||||||
|
/// wait), and an unknown queue ADMITS. The gate reads
|
||||||
|
/// [`EngineWorkerLoad::num_waiting_reqs`] because that is what the request
|
||||||
|
/// cares about, and the router-side in-flight counter cannot separate a
|
||||||
|
/// running request from a waiting one — so there is no honest substitute,
|
||||||
|
/// and the gate fails open rather than comparing the limit against a
|
||||||
|
/// different quantity.
|
||||||
|
pub(crate) fn queue_gate_admits(
|
||||||
|
snapshot: &EngineLoadSnapshot,
|
||||||
|
worker: &Worker,
|
||||||
|
limit: Option<u64>,
|
||||||
|
) -> bool {
|
||||||
|
let Some(limit) = limit else {
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
snapshot
|
||||||
|
.fresh_load_for_url(&worker.url)
|
||||||
|
.is_none_or(|load| load.num_waiting_reqs < limit)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True when the gate has provably nowhere unqueued to divert to: every
|
||||||
|
/// worker in `fleet` has a fresh sample at or over the limit. An unset limit,
|
||||||
|
/// an empty fleet, or a single worker with no fresh sample all make this
|
||||||
|
/// false — an unknown queue is not a proven full one.
|
||||||
|
pub(crate) fn fleet_is_all_queued(
|
||||||
|
snapshot: &EngineLoadSnapshot,
|
||||||
|
fleet: &[Arc<Worker>],
|
||||||
|
limit: Option<u64>,
|
||||||
|
) -> bool {
|
||||||
|
limit.is_some()
|
||||||
|
&& !fleet.is_empty()
|
||||||
|
&& fleet
|
||||||
|
.iter()
|
||||||
|
.all(|worker| !queue_gate_admits(snapshot, worker, limit))
|
||||||
|
}
|
||||||
|
|
||||||
/// Selects a worker from bounded cache candidates and records guard coverage.
|
/// Selects a worker from bounded cache candidates and records guard coverage.
|
||||||
pub fn resolve_cache_candidates(
|
pub fn resolve_cache_candidates(
|
||||||
proposal: &CacheCandidateProposal,
|
proposal: &CacheCandidateProposal,
|
||||||
request_input_tokens: u64,
|
request_input_tokens: u64,
|
||||||
snapshot: &EngineLoadSnapshot,
|
snapshot: &EngineLoadSnapshot,
|
||||||
|
fleet: &[Arc<Worker>],
|
||||||
) -> CacheCandidateResolution {
|
) -> CacheCandidateResolution {
|
||||||
|
let queue_limit = proposal.worker_queue_limit;
|
||||||
|
let fleet_all_queued = fleet_is_all_queued(snapshot, fleet, queue_limit);
|
||||||
|
// The queue gate runs before hard admission so a busy worker never
|
||||||
|
// pollutes the capacity-rejection counters, and so the diverted-overlap
|
||||||
|
// audit below sees exactly the candidates the gate removed. One pass:
|
||||||
|
// the gate answer per candidate is what splits the set, so asking twice
|
||||||
|
// would re-read the snapshot for every candidate on every request.
|
||||||
|
let mut evaluated: Vec<&CacheCandidate> = Vec::with_capacity(proposal.candidates.len());
|
||||||
|
let mut queue_gate_rejected_candidates = 0u64;
|
||||||
|
let mut queue_gate_best_rejected_blocks = 0u32;
|
||||||
|
for candidate in &proposal.candidates {
|
||||||
|
if queue_gate_admits(snapshot, &candidate.worker, queue_limit) {
|
||||||
|
evaluated.push(candidate);
|
||||||
|
} else {
|
||||||
|
queue_gate_rejected_candidates += 1;
|
||||||
|
queue_gate_best_rejected_blocks =
|
||||||
|
queue_gate_best_rejected_blocks.max(candidate.matched_prefix_blocks);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Second tier, mirroring `range_fallback`: when the gate removed every
|
||||||
|
// candidate AND nowhere in the fleet is unqueued, diversion cannot dodge
|
||||||
|
// a wait, so returning no decision would trade the whole prefix for
|
||||||
|
// nothing. Re-admit the ungated set. While an unqueued worker still
|
||||||
|
// exists the gate keeps its teeth and the request leaves the prefix.
|
||||||
|
let queue_gate_fell_back =
|
||||||
|
evaluated.is_empty() && queue_gate_rejected_candidates > 0 && fleet_all_queued;
|
||||||
|
if queue_gate_fell_back {
|
||||||
|
evaluated.extend(proposal.candidates.iter());
|
||||||
|
}
|
||||||
|
// Built over the candidates that actually reach admission: a gated-out
|
||||||
|
// candidate missing native monitor data would otherwise break the
|
||||||
|
// lookup's full-coverage check and silently downgrade every pressure
|
||||||
|
// comparison (and `prefill_pressure_source`) to router-local.
|
||||||
let loads = FreshLoadLookup::new(
|
let loads = FreshLoadLookup::new(
|
||||||
Some(snapshot),
|
Some(snapshot),
|
||||||
proposal
|
evaluated.iter().copied().map(|candidate| &candidate.worker),
|
||||||
.candidates
|
|
||||||
.iter()
|
|
||||||
.map(|candidate| &candidate.worker),
|
|
||||||
);
|
);
|
||||||
let admitted: Vec<&CacheCandidate> = proposal
|
let admission_evaluated_candidates = evaluated.len() as u64;
|
||||||
.candidates
|
let admitted: Vec<&CacheCandidate> = evaluated
|
||||||
.iter()
|
.into_iter()
|
||||||
.filter(|candidate| is_cache_candidate_admitted(candidate, request_input_tokens, &loads))
|
.filter(|candidate| is_cache_candidate_admitted(candidate, request_input_tokens, &loads))
|
||||||
.collect();
|
.collect();
|
||||||
let admission_rejected_candidates =
|
let admission_rejected_candidates =
|
||||||
proposal.candidates.len().saturating_sub(admitted.len()) as u64;
|
admission_evaluated_candidates.saturating_sub(admitted.len() as u64);
|
||||||
let Some(work_floor) = admitted
|
let Some(work_floor) = admitted
|
||||||
.iter()
|
.iter()
|
||||||
.copied()
|
.copied()
|
||||||
@@ -158,8 +256,12 @@ pub fn resolve_cache_candidates(
|
|||||||
return CacheCandidateResolution {
|
return CacheCandidateResolution {
|
||||||
decision: None,
|
decision: None,
|
||||||
prefill_pressure_source: loads.prefill_pressure_source(),
|
prefill_pressure_source: loads.prefill_pressure_source(),
|
||||||
admission_evaluated_candidates: proposal.candidates.len() as u64,
|
admission_evaluated_candidates,
|
||||||
admission_rejected_candidates,
|
admission_rejected_candidates,
|
||||||
|
queue_gate_rejected_candidates,
|
||||||
|
queue_gate_best_rejected_blocks,
|
||||||
|
queue_gate_fell_back,
|
||||||
|
fleet_all_queued,
|
||||||
pressure_guard_compared_pairs: 0,
|
pressure_guard_compared_pairs: 0,
|
||||||
pressure_guard_overrides: 0,
|
pressure_guard_overrides: 0,
|
||||||
};
|
};
|
||||||
@@ -201,8 +303,12 @@ pub fn resolve_cache_candidates(
|
|||||||
load_snapshot_version: snapshot.version,
|
load_snapshot_version: snapshot.version,
|
||||||
}),
|
}),
|
||||||
prefill_pressure_source: loads.prefill_pressure_source(),
|
prefill_pressure_source: loads.prefill_pressure_source(),
|
||||||
admission_evaluated_candidates: proposal.candidates.len() as u64,
|
admission_evaluated_candidates,
|
||||||
admission_rejected_candidates,
|
admission_rejected_candidates,
|
||||||
|
queue_gate_rejected_candidates,
|
||||||
|
queue_gate_best_rejected_blocks,
|
||||||
|
queue_gate_fell_back,
|
||||||
|
fleet_all_queued,
|
||||||
pressure_guard_compared_pairs,
|
pressure_guard_compared_pairs,
|
||||||
pressure_guard_overrides,
|
pressure_guard_overrides,
|
||||||
}
|
}
|
||||||
@@ -213,27 +319,30 @@ pub fn resolve_prefill(
|
|||||||
proposal: &SelectionProposal,
|
proposal: &SelectionProposal,
|
||||||
request_input_tokens: u64,
|
request_input_tokens: u64,
|
||||||
snapshot: &EngineLoadSnapshot,
|
snapshot: &EngineLoadSnapshot,
|
||||||
|
queue_limit: Option<u64>,
|
||||||
) -> Option<FinalDecision> {
|
) -> Option<FinalDecision> {
|
||||||
resolve_prefill_admitted(range, proposal, request_input_tokens, snapshot).or_else(|| {
|
resolve_prefill_admitted(range, proposal, request_input_tokens, snapshot, queue_limit).or_else(
|
||||||
if !contains_worker(range, &proposal.primary) {
|
|| {
|
||||||
return None;
|
if !contains_worker(range, &proposal.primary) {
|
||||||
}
|
return None;
|
||||||
let backup = proposal
|
}
|
||||||
.backup
|
let backup = proposal
|
||||||
.as_ref()
|
.backup
|
||||||
.filter(|worker| contains_worker(range, worker))
|
.as_ref()
|
||||||
.cloned();
|
.filter(|worker| contains_worker(range, worker))
|
||||||
let legal = legal_prefill_candidates(range, proposal);
|
.cloned();
|
||||||
let selected = select_with_snapshot(&legal, Some(snapshot))?;
|
let legal = legal_prefill_candidates(range, proposal);
|
||||||
Some(FinalDecision {
|
let selected = select_with_snapshot(&legal, Some(snapshot))?;
|
||||||
selected,
|
Some(FinalDecision {
|
||||||
primary: Arc::clone(&proposal.primary),
|
selected,
|
||||||
backup,
|
primary: Arc::clone(&proposal.primary),
|
||||||
reason: DecisionReason::CapacityFallbackPowerOfTwo,
|
backup,
|
||||||
candidate_range_id: range.id.to_string(),
|
reason: DecisionReason::CapacityFallbackPowerOfTwo,
|
||||||
load_snapshot_version: snapshot.version,
|
candidate_range_id: range.id.to_string(),
|
||||||
})
|
load_snapshot_version: snapshot.version,
|
||||||
})
|
})
|
||||||
|
},
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolves prefill admission without overcommitting a full candidate range.
|
/// Resolves prefill admission without overcommitting a full candidate range.
|
||||||
@@ -242,6 +351,7 @@ pub fn resolve_prefill_admitted(
|
|||||||
proposal: &SelectionProposal,
|
proposal: &SelectionProposal,
|
||||||
request_input_tokens: u64,
|
request_input_tokens: u64,
|
||||||
snapshot: &EngineLoadSnapshot,
|
snapshot: &EngineLoadSnapshot,
|
||||||
|
queue_limit: Option<u64>,
|
||||||
) -> Option<FinalDecision> {
|
) -> Option<FinalDecision> {
|
||||||
if !contains_worker(range, &proposal.primary) {
|
if !contains_worker(range, &proposal.primary) {
|
||||||
return None;
|
return None;
|
||||||
@@ -251,11 +361,20 @@ pub fn resolve_prefill_admitted(
|
|||||||
.as_ref()
|
.as_ref()
|
||||||
.filter(|worker| contains_worker(range, worker))
|
.filter(|worker| contains_worker(range, worker))
|
||||||
.cloned();
|
.cloned();
|
||||||
|
// The queue gate demotes an admitted primary or backup exactly as it
|
||||||
|
// demotes a cache candidate: a worker already making requests wait must
|
||||||
|
// not win on proposal position alone, or the fallback the gate diverted
|
||||||
|
// to hands the request straight back to it. Demotion is not rejection —
|
||||||
|
// `range_fallback` below is two-tier, and its second tier returns the
|
||||||
|
// least-pressured admitted worker (the demoted one included) when every
|
||||||
|
// admitted worker is queueing, so an all-queueing fleet still routes.
|
||||||
let primary_admitted = is_proposal_worker_eligible(proposal, &proposal.primary)
|
let primary_admitted = is_proposal_worker_eligible(proposal, &proposal.primary)
|
||||||
&& is_prefill_admitted(range, &proposal.primary, request_input_tokens, snapshot);
|
&& is_prefill_admitted(range, &proposal.primary, request_input_tokens, snapshot)
|
||||||
|
&& queue_gate_admits(snapshot, &proposal.primary, queue_limit);
|
||||||
let backup_admitted = backup.as_ref().is_some_and(|worker| {
|
let backup_admitted = backup.as_ref().is_some_and(|worker| {
|
||||||
is_proposal_worker_eligible(proposal, worker)
|
is_proposal_worker_eligible(proposal, worker)
|
||||||
&& is_prefill_admitted(range, worker, request_input_tokens, snapshot)
|
&& is_prefill_admitted(range, worker, request_input_tokens, snapshot)
|
||||||
|
&& queue_gate_admits(snapshot, worker, queue_limit)
|
||||||
});
|
});
|
||||||
|
|
||||||
let (selected, reason) = match (primary_admitted, backup.as_ref(), backup_admitted) {
|
let (selected, reason) = match (primary_admitted, backup.as_ref(), backup_admitted) {
|
||||||
@@ -275,7 +394,7 @@ pub fn resolve_prefill_admitted(
|
|||||||
(false, Some(backup), true) => (Arc::clone(backup), DecisionReason::BackupPrimaryAdmission),
|
(false, Some(backup), true) => (Arc::clone(backup), DecisionReason::BackupPrimaryAdmission),
|
||||||
_ => {
|
_ => {
|
||||||
let legal = legal_prefill_candidates(range, proposal);
|
let legal = legal_prefill_candidates(range, proposal);
|
||||||
range_fallback(range, &legal, request_input_tokens, snapshot)?
|
range_fallback(range, &legal, request_input_tokens, snapshot, queue_limit)?
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
Some(FinalDecision {
|
Some(FinalDecision {
|
||||||
@@ -493,7 +612,7 @@ fn materially_more_pressured(
|
|||||||
/// candidate sets use Router-local active load to preserve ordering.
|
/// candidate sets use Router-local active load to preserve ordering.
|
||||||
pub(crate) struct FreshLoadLookup<'a> {
|
pub(crate) struct FreshLoadLookup<'a> {
|
||||||
by_worker_id: HashMap<String, &'a NativeCacheWorkerLoad>,
|
by_worker_id: HashMap<String, &'a NativeCacheWorkerLoad>,
|
||||||
basic_by_worker_id: HashMap<String, &'a crate::policies::engine_load::EngineWorkerLoad>,
|
basic_by_worker_id: HashMap<String, &'a EngineWorkerLoad>,
|
||||||
local_active_by_worker_id: HashMap<String, usize>,
|
local_active_by_worker_id: HashMap<String, usize>,
|
||||||
compare_engine: bool,
|
compare_engine: bool,
|
||||||
compare_basic_engine: bool,
|
compare_basic_engine: bool,
|
||||||
@@ -662,6 +781,7 @@ fn range_fallback(
|
|||||||
legal: &[Arc<Worker>],
|
legal: &[Arc<Worker>],
|
||||||
request_input_tokens: u64,
|
request_input_tokens: u64,
|
||||||
snapshot: &EngineLoadSnapshot,
|
snapshot: &EngineLoadSnapshot,
|
||||||
|
queue_limit: Option<u64>,
|
||||||
) -> Option<(Arc<Worker>, DecisionReason)> {
|
) -> Option<(Arc<Worker>, DecisionReason)> {
|
||||||
let admitted = legal
|
let admitted = legal
|
||||||
.iter()
|
.iter()
|
||||||
@@ -669,9 +789,38 @@ fn range_fallback(
|
|||||||
.filter(|worker| is_prefill_admitted(range, worker, request_input_tokens, snapshot))
|
.filter(|worker| is_prefill_admitted(range, worker, request_input_tokens, snapshot))
|
||||||
.cloned()
|
.cloned()
|
||||||
.collect::<Vec<_>>();
|
.collect::<Vec<_>>();
|
||||||
let loads = FreshLoadLookup::new(Some(snapshot), admitted.iter());
|
// Two-tier under the queue gate: least-pressured worker that is not
|
||||||
|
// queueing, and only when every admitted worker is queueing,
|
||||||
|
// least-pressured overall. Both tiers are load-bearing. Pressure ranks by
|
||||||
|
// queue tokens / depth while the gate reads queue length, and those
|
||||||
|
// disagree exactly where the gate earns its keep — a shallow worker with
|
||||||
|
// a backlog is the fleet minimum BY PRESSURE, so a single-tier fallback
|
||||||
|
// hands the request straight back to the cache home the gate just
|
||||||
|
// rejected. The second tier is what keeps an all-queueing fleet routable
|
||||||
|
// instead of failing every request.
|
||||||
|
let pool = match queue_limit {
|
||||||
|
// Gate disabled: the tiers coincide, so filtering would only clone
|
||||||
|
// the vector.
|
||||||
|
None => admitted,
|
||||||
|
Some(_) => {
|
||||||
|
let unqueued = admitted
|
||||||
|
.iter()
|
||||||
|
.filter(|worker| queue_gate_admits(snapshot, worker, queue_limit))
|
||||||
|
.cloned()
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
if unqueued.is_empty() {
|
||||||
|
admitted
|
||||||
|
} else {
|
||||||
|
unqueued
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// Scoped to the pool actually ranked: an admitted-but-gated worker
|
||||||
|
// missing native monitor data would otherwise downgrade the comparison
|
||||||
|
// for the whole unqueued tier to router-local.
|
||||||
|
let loads = FreshLoadLookup::new(Some(snapshot), pool.iter());
|
||||||
loads
|
loads
|
||||||
.min_by_pressure_key(admitted, FreshLoadLookup::compare_prefill_keys)
|
.min_by_pressure_key(pool, FreshLoadLookup::compare_prefill_keys)
|
||||||
.map(|worker| (worker, DecisionReason::RangeFallback))
|
.map(|worker| (worker, DecisionReason::RangeFallback))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -866,11 +1015,12 @@ mod tests {
|
|||||||
&range,
|
&range,
|
||||||
&SelectionProposal::primary(Arc::clone(&full)),
|
&SelectionProposal::primary(Arc::clone(&full)),
|
||||||
20,
|
20,
|
||||||
&loads
|
&loads,
|
||||||
|
None,
|
||||||
)
|
)
|
||||||
.is_some());
|
.is_some());
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
resolve_prefill(&range, &SelectionProposal::primary(full), 20, &loads)
|
resolve_prefill(&range, &SelectionProposal::primary(full), 20, &loads, None)
|
||||||
.expect("fallback selects the admitted worker")
|
.expect("fallback selects the admitted worker")
|
||||||
.selected
|
.selected
|
||||||
.id,
|
.id,
|
||||||
@@ -896,8 +1046,14 @@ mod tests {
|
|||||||
(&filtered, 0, 0, 0, 100),
|
(&filtered, 0, 0, 0, 100),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&workers), &proposal, 32, &loads)
|
let decision = resolve_prefill(
|
||||||
.expect("capacity exhaustion must degrade within the legal domain");
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.expect("capacity exhaustion must degrade within the legal domain");
|
||||||
|
|
||||||
assert!(matches!(
|
assert!(matches!(
|
||||||
decision.selected.id.0.as_str(),
|
decision.selected.id.0.as_str(),
|
||||||
@@ -918,8 +1074,14 @@ mod tests {
|
|||||||
.expect("the opposite snapshot has the same legal workers");
|
.expect("the opposite snapshot has the same legal workers");
|
||||||
assert_eq!(opposite_decision.id, backup.id);
|
assert_eq!(opposite_decision.id, backup.id);
|
||||||
|
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&workers), &proposal, 32, &explicit)
|
let decision = resolve_prefill(
|
||||||
.expect("capacity exhaustion must degrade to Power-of-Two");
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&explicit,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.expect("capacity exhaustion must degrade to Power-of-Two");
|
||||||
|
|
||||||
assert_eq!(decision.selected.id, primary.id);
|
assert_eq!(decision.selected.id, primary.id);
|
||||||
assert_eq!(decision.load_snapshot_version, explicit.version);
|
assert_eq!(decision.load_snapshot_version, explicit.version);
|
||||||
@@ -951,6 +1113,7 @@ mod tests {
|
|||||||
worker: Arc::clone(&congested),
|
worker: Arc::clone(&congested),
|
||||||
matched_prefix_tokens: 90,
|
matched_prefix_tokens: 90,
|
||||||
uncached_tokens: 10,
|
uncached_tokens: 10,
|
||||||
|
matched_prefix_blocks: 9,
|
||||||
candidate_range_id: "global".into(),
|
candidate_range_id: "global".into(),
|
||||||
max_pending_prefill_tokens: None,
|
max_pending_prefill_tokens: None,
|
||||||
},
|
},
|
||||||
@@ -958,6 +1121,7 @@ mod tests {
|
|||||||
worker: Arc::clone(&idle),
|
worker: Arc::clone(&idle),
|
||||||
matched_prefix_tokens: 80,
|
matched_prefix_tokens: 80,
|
||||||
uncached_tokens: 20,
|
uncached_tokens: 20,
|
||||||
|
matched_prefix_blocks: 8,
|
||||||
candidate_range_id: "global".into(),
|
candidate_range_id: "global".into(),
|
||||||
max_pending_prefill_tokens: None,
|
max_pending_prefill_tokens: None,
|
||||||
},
|
},
|
||||||
@@ -967,13 +1131,14 @@ mod tests {
|
|||||||
pressure_abs_threshold_tokens: 100,
|
pressure_abs_threshold_tokens: 100,
|
||||||
pressure_abs_threshold_ms: None,
|
pressure_abs_threshold_ms: None,
|
||||||
pressure_rel_threshold: 1.5,
|
pressure_rel_threshold: 1.5,
|
||||||
|
worker_queue_limit: None,
|
||||||
};
|
};
|
||||||
let loads = snapshot(&[
|
let loads = snapshot(&[
|
||||||
(&congested, 1, 1_000, 10, 10_000),
|
(&congested, 1, 1_000, 10, 10_000),
|
||||||
(&idle, 1, 10, 10, 10_000),
|
(&idle, 1, 10, 10, 10_000),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let resolution = resolve_cache_candidates(&proposal, 100, &loads);
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &[]);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
resolution
|
resolution
|
||||||
.decision
|
.decision
|
||||||
@@ -987,4 +1152,401 @@ mod tests {
|
|||||||
assert_eq!(resolution.pressure_guard_compared_pairs, 1);
|
assert_eq!(resolution.pressure_guard_compared_pairs, 1);
|
||||||
assert_eq!(resolution.pressure_guard_overrides, 1);
|
assert_eq!(resolution.pressure_guard_overrides, 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn candidate(
|
||||||
|
worker: &Arc<Worker>,
|
||||||
|
uncached_tokens: u64,
|
||||||
|
matched_blocks: u32,
|
||||||
|
) -> CacheCandidate {
|
||||||
|
CacheCandidate {
|
||||||
|
worker: Arc::clone(worker),
|
||||||
|
matched_prefix_tokens: 100 - uncached_tokens,
|
||||||
|
uncached_tokens,
|
||||||
|
matched_prefix_blocks: matched_blocks,
|
||||||
|
candidate_range_id: "global".into(),
|
||||||
|
max_pending_prefill_tokens: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn queue_gate_proposal(
|
||||||
|
candidates: Vec<CacheCandidate>,
|
||||||
|
limit: Option<u64>,
|
||||||
|
) -> CacheCandidateProposal {
|
||||||
|
CacheCandidateProposal {
|
||||||
|
candidates,
|
||||||
|
worker_queue_limit: limit,
|
||||||
|
..Default::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_diverts_off_an_owner_over_the_limit() {
|
||||||
|
let owner = worker("owner");
|
||||||
|
let other = worker("other");
|
||||||
|
let proposal = queue_gate_proposal(
|
||||||
|
vec![
|
||||||
|
// The owner holds the deeper prefix and would always win
|
||||||
|
// without the gate.
|
||||||
|
candidate(&owner, 10, 9),
|
||||||
|
candidate(&other, 60, 4),
|
||||||
|
],
|
||||||
|
Some(4),
|
||||||
|
);
|
||||||
|
// The owner is at the limit (4 waiting >= 4); the other is idle.
|
||||||
|
let loads = snapshot(&[(&owner, 1, 4, 10, 10_000), (&other, 1, 0, 10, 10_000)]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&other)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
resolution
|
||||||
|
.decision
|
||||||
|
.expect("the unqueued candidate remains admitted")
|
||||||
|
.selected
|
||||||
|
.id,
|
||||||
|
other.id
|
||||||
|
);
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 1);
|
||||||
|
assert_eq!(resolution.queue_gate_best_rejected_blocks, 9);
|
||||||
|
// A gate rejection must not read as a capacity rejection.
|
||||||
|
assert_eq!(resolution.admission_rejected_candidates, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_disabled_keeps_the_deepest_owner() {
|
||||||
|
let owner = worker("owner");
|
||||||
|
let other = worker("other");
|
||||||
|
let proposal = queue_gate_proposal(
|
||||||
|
vec![candidate(&owner, 10, 9), candidate(&other, 60, 4)],
|
||||||
|
None,
|
||||||
|
);
|
||||||
|
let loads = snapshot(&[(&owner, 1, 40, 10, 10_000), (&other, 1, 0, 10, 10_000)]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&other)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
resolution
|
||||||
|
.decision
|
||||||
|
.expect("no gate, owner wins")
|
||||||
|
.selected
|
||||||
|
.id,
|
||||||
|
owner.id
|
||||||
|
);
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_missing_snapshot_admits() {
|
||||||
|
let unknown = worker("unknown");
|
||||||
|
let proposal = queue_gate_proposal(vec![candidate(&unknown, 10, 9)], Some(4));
|
||||||
|
// The worker never published a load sample: the gate fails open.
|
||||||
|
let loads = snapshot(&[]);
|
||||||
|
let fleet = vec![Arc::clone(&unknown)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
resolution
|
||||||
|
.decision
|
||||||
|
.expect("an unknown queue admits")
|
||||||
|
.selected
|
||||||
|
.id,
|
||||||
|
unknown.id
|
||||||
|
);
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_exhausted_candidates_returns_no_decision_with_audit() {
|
||||||
|
// The only owner is queueing but a non-owner is idle, so diversion
|
||||||
|
// still buys something: the request must leave the prefix.
|
||||||
|
let owner = worker("owner");
|
||||||
|
let idle_stranger = worker("idle_stranger");
|
||||||
|
let proposal = queue_gate_proposal(vec![candidate(&owner, 10, 9)], Some(4));
|
||||||
|
let loads = snapshot(&[
|
||||||
|
(&owner, 1, 9, 10, 10_000),
|
||||||
|
(&idle_stranger, 0, 0, 10, 10_000),
|
||||||
|
]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&idle_stranger)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert!(resolution.decision.is_none());
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 1);
|
||||||
|
assert_eq!(resolution.queue_gate_best_rejected_blocks, 9);
|
||||||
|
assert_eq!(resolution.admission_rejected_candidates, 0);
|
||||||
|
assert_eq!(resolution.admission_evaluated_candidates, 0);
|
||||||
|
assert!(!resolution.fleet_all_queued);
|
||||||
|
assert!(!resolution.queue_gate_fell_back);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_admits_one_below_the_limit() {
|
||||||
|
// Pins the admit side of the `<` boundary: `limit - 1` waiting must
|
||||||
|
// still win on affinity, or the gate fires a request early.
|
||||||
|
let owner = worker("owner");
|
||||||
|
let other = worker("other");
|
||||||
|
let proposal = queue_gate_proposal(
|
||||||
|
vec![candidate(&owner, 10, 9), candidate(&other, 60, 4)],
|
||||||
|
Some(4),
|
||||||
|
);
|
||||||
|
let loads = snapshot(&[(&owner, 1, 3, 10, 10_000), (&other, 1, 0, 10, 10_000)]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&other)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
resolution
|
||||||
|
.decision
|
||||||
|
.expect("one below the limit is not queueing")
|
||||||
|
.selected
|
||||||
|
.id,
|
||||||
|
owner.id
|
||||||
|
);
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 0);
|
||||||
|
assert_eq!(resolution.admission_evaluated_candidates, 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_keeps_the_prefix_when_the_whole_fleet_is_queueing() {
|
||||||
|
// Every owner is over the limit AND so is every other worker, so a
|
||||||
|
// diversion could not dodge a wait. Discarding the prefix would be
|
||||||
|
// pure loss: the ungated tier re-admits the owners.
|
||||||
|
let owner = worker("owner");
|
||||||
|
let shallow_owner = worker("shallow_owner");
|
||||||
|
let proposal = queue_gate_proposal(
|
||||||
|
vec![candidate(&owner, 10, 9), candidate(&shallow_owner, 60, 4)],
|
||||||
|
Some(4),
|
||||||
|
);
|
||||||
|
let loads = snapshot(&[
|
||||||
|
(&owner, 1, 9, 10, 10_000),
|
||||||
|
(&shallow_owner, 1, 5, 10, 10_000),
|
||||||
|
]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&shallow_owner)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
resolution
|
||||||
|
.decision
|
||||||
|
.expect("an all-queueing fleet must keep its prefix")
|
||||||
|
.selected
|
||||||
|
.id,
|
||||||
|
owner.id,
|
||||||
|
"the deepest prefix must win once diversion buys nothing"
|
||||||
|
);
|
||||||
|
assert!(resolution.queue_gate_fell_back);
|
||||||
|
assert!(resolution.fleet_all_queued);
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 2);
|
||||||
|
assert_eq!(
|
||||||
|
resolution.admission_evaluated_candidates, 2,
|
||||||
|
"the ungated tier puts every candidate through capacity admission"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_saturation_survives_a_capacity_exhausted_re_admission() {
|
||||||
|
// The audit tuple the decision label is read from, in the one case
|
||||||
|
// that used to lose the saturation signal: the fleet is queueing, the
|
||||||
|
// ungated tier re-admits the owners, and they then fail hard
|
||||||
|
// admission, so there is no winner. `fleet_all_queued` must still be
|
||||||
|
// set on the way out, because the label is keyed on the fleet being
|
||||||
|
// saturated and not on where the request finally landed.
|
||||||
|
let owner = worker("owner");
|
||||||
|
let shallow_owner = worker("shallow_owner");
|
||||||
|
let proposal = queue_gate_proposal(
|
||||||
|
vec![candidate(&owner, 10, 9), candidate(&shallow_owner, 60, 4)],
|
||||||
|
Some(4),
|
||||||
|
);
|
||||||
|
// Queueing AND out of KV: used == capacity on both.
|
||||||
|
let loads = snapshot(&[
|
||||||
|
(&owner, 1, 9, 10_000, 10_000),
|
||||||
|
(&shallow_owner, 1, 5, 10_000, 10_000),
|
||||||
|
]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&shallow_owner)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100_000, &loads, &fleet);
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
resolution.decision.is_none(),
|
||||||
|
"a capacity-exhausted fleet cannot produce a winner"
|
||||||
|
);
|
||||||
|
assert!(resolution.fleet_all_queued);
|
||||||
|
assert!(resolution.queue_gate_fell_back);
|
||||||
|
assert_eq!(resolution.queue_gate_rejected_candidates, 2);
|
||||||
|
assert_eq!(
|
||||||
|
resolution.admission_evaluated_candidates, 2,
|
||||||
|
"re-admission is what makes a zero-evaluated saturated audit unreachable"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_fleet_saturation_needs_a_fresh_sample_everywhere() {
|
||||||
|
// A worker with no fresh sample has an unknown queue, not a proven
|
||||||
|
// full one, so the fleet is not saturated and the gate keeps diverting.
|
||||||
|
let owner = worker("owner");
|
||||||
|
let silent = worker("silent");
|
||||||
|
let proposal = queue_gate_proposal(vec![candidate(&owner, 10, 9)], Some(4));
|
||||||
|
let loads = snapshot(&[(&owner, 1, 9, 10, 10_000)]);
|
||||||
|
let fleet = vec![Arc::clone(&owner), Arc::clone(&silent)];
|
||||||
|
|
||||||
|
let resolution = resolve_cache_candidates(&proposal, 100, &loads, &fleet);
|
||||||
|
|
||||||
|
assert!(resolution.decision.is_none());
|
||||||
|
assert!(!resolution.fleet_all_queued);
|
||||||
|
assert!(!resolution.queue_gate_fell_back);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn range_fallback_prefers_an_unqueued_worker_over_a_shallower_queueing_one() {
|
||||||
|
// The primary fails KV-capacity admission so selection reaches the
|
||||||
|
// range fallback. The queueing worker is the fallback minimum BY
|
||||||
|
// PRESSURE (1 waiting uncached token), so a single-tier fallback
|
||||||
|
// would pick it — handing the request straight back to a worker the
|
||||||
|
// gate just rejected.
|
||||||
|
let primary = worker("primary");
|
||||||
|
let shallow_queued = worker("shallow_queued");
|
||||||
|
let busy_unqueued = worker("busy_unqueued");
|
||||||
|
let workers = vec![
|
||||||
|
Arc::clone(&primary),
|
||||||
|
Arc::clone(&shallow_queued),
|
||||||
|
Arc::clone(&busy_unqueued),
|
||||||
|
];
|
||||||
|
let proposal = SelectionProposal::primary(Arc::clone(&primary));
|
||||||
|
let load = |waiting: u64, waiting_uncached: u64, total: u64, max_total: u64| {
|
||||||
|
NativeCacheWorkerLoad {
|
||||||
|
num_running_reqs: 0,
|
||||||
|
num_waiting_reqs: waiting,
|
||||||
|
num_waiting_uncached_tokens: waiting_uncached,
|
||||||
|
num_used_tokens: total,
|
||||||
|
num_total_tokens: total,
|
||||||
|
max_total_num_tokens: max_total,
|
||||||
|
max_running_requests: 64,
|
||||||
|
prefill_throughput_tokens_per_s: None,
|
||||||
|
estimated_prefill_queue_ms: None,
|
||||||
|
captured_at: Instant::now(),
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// Queue depth and pressure disagree by construction: shallow_queued
|
||||||
|
// waits 5 (over the limit) behind 1 uncached token, busy_unqueued
|
||||||
|
// waits 3 (under the limit) behind 1000 uncached tokens. The primary
|
||||||
|
// is KV-full, so it is not admitted at all.
|
||||||
|
let loads = EngineLoadSnapshot::from_native_cache_workers(
|
||||||
|
7,
|
||||||
|
[
|
||||||
|
(primary.url.clone(), load(0, 0, 10_000, 10_000)),
|
||||||
|
(shallow_queued.url.clone(), load(5, 1, 10, 10_000)),
|
||||||
|
(busy_unqueued.url.clone(), load(3, 1_000, 10, 10_000)),
|
||||||
|
]
|
||||||
|
.into_iter()
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let decision = resolve_prefill(
|
||||||
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
Some(4),
|
||||||
|
)
|
||||||
|
.expect("an admitted worker exists");
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
decision.selected.id, busy_unqueued.id,
|
||||||
|
"the fallback must prefer the unqueued worker even though the \
|
||||||
|
queueing one is the pressure minimum"
|
||||||
|
);
|
||||||
|
assert_eq!(decision.reason, DecisionReason::RangeFallback);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_demotes_a_queueing_primary_that_capacity_would_admit() {
|
||||||
|
// The whole point of the gate is that the cache-affinity fallback
|
||||||
|
// must not hand the request back to a queueing worker. The primary
|
||||||
|
// here has plenty of KV capacity, so without the gate the
|
||||||
|
// `(true, _, _)` arm returns it unconditionally and the gate is
|
||||||
|
// bypassed on the single most common path.
|
||||||
|
let queued_primary = worker("queued_primary");
|
||||||
|
let unqueued = worker("unqueued");
|
||||||
|
let workers = vec![Arc::clone(&queued_primary), Arc::clone(&unqueued)];
|
||||||
|
let proposal = SelectionProposal::primary(Arc::clone(&queued_primary));
|
||||||
|
let loads = snapshot(&[
|
||||||
|
(&queued_primary, 0, 6, 10, 10_000),
|
||||||
|
(&unqueued, 0, 0, 10, 10_000),
|
||||||
|
]);
|
||||||
|
|
||||||
|
let ungated = resolve_prefill_admitted(
|
||||||
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.expect("without a limit the primary is admitted");
|
||||||
|
assert_eq!(ungated.selected.id, queued_primary.id);
|
||||||
|
assert_eq!(ungated.reason, DecisionReason::Primary);
|
||||||
|
|
||||||
|
let gated = resolve_prefill_admitted(
|
||||||
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
Some(4),
|
||||||
|
)
|
||||||
|
.expect("the unqueued worker takes over");
|
||||||
|
assert_eq!(gated.selected.id, unqueued.id);
|
||||||
|
assert_eq!(gated.reason, DecisionReason::RangeFallback);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_gate_still_routes_when_the_only_admitted_worker_is_queueing() {
|
||||||
|
// Demotion must never become rejection: with every admitted worker
|
||||||
|
// over the limit, the second fallback tier keeps the request routable.
|
||||||
|
let only = worker("only");
|
||||||
|
let workers = vec![Arc::clone(&only)];
|
||||||
|
let proposal = SelectionProposal::primary(Arc::clone(&only));
|
||||||
|
let loads = snapshot(&[(&only, 0, 9, 10, 10_000)]);
|
||||||
|
|
||||||
|
let decision = resolve_prefill_admitted(
|
||||||
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
Some(4),
|
||||||
|
)
|
||||||
|
.expect("an all-queueing fleet must still route");
|
||||||
|
|
||||||
|
assert_eq!(decision.selected.id, only.id);
|
||||||
|
assert_eq!(decision.reason, DecisionReason::RangeFallback);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn range_fallback_keeps_an_all_queueing_fleet_routable() {
|
||||||
|
let primary = worker("primary");
|
||||||
|
let left = worker("left");
|
||||||
|
let right = worker("right");
|
||||||
|
let workers = vec![Arc::clone(&primary), Arc::clone(&left), Arc::clone(&right)];
|
||||||
|
let proposal = SelectionProposal::primary(Arc::clone(&primary));
|
||||||
|
// The primary is KV-full; both fallback workers are over the limit.
|
||||||
|
// The second tier takes the least-pressured one instead of failing
|
||||||
|
// the request.
|
||||||
|
let loads = snapshot(&[
|
||||||
|
(&primary, 0, 0, 10_000, 10_000),
|
||||||
|
(&left, 0, 9, 10, 10_000),
|
||||||
|
(&right, 0, 5, 10, 10_000),
|
||||||
|
]);
|
||||||
|
|
||||||
|
let decision = resolve_prefill(
|
||||||
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
Some(4),
|
||||||
|
)
|
||||||
|
.expect("an all-queueing fleet must still route");
|
||||||
|
|
||||||
|
assert_eq!(decision.selected.id, right.id);
|
||||||
|
assert_eq!(decision.reason, DecisionReason::RangeFallback);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -55,10 +55,12 @@ impl CacheAwarePolicy {
|
|||||||
if entry.matched_prefix_blocks == 0 || !seen.insert(worker.id.clone()) {
|
if entry.matched_prefix_blocks == 0 || !seen.insert(worker.id.clone()) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
let matched_prefix_blocks =
|
||||||
|
cap_matched_prefix_blocks(signal.query_blocks, entry.matched_prefix_blocks);
|
||||||
let matched_prefix_tokens = estimate_matched_prefix_tokens(
|
let matched_prefix_tokens = estimate_matched_prefix_tokens(
|
||||||
input_tokens,
|
input_tokens,
|
||||||
signal.query_blocks,
|
signal.query_blocks,
|
||||||
entry.matched_prefix_blocks,
|
matched_prefix_blocks,
|
||||||
);
|
);
|
||||||
if !self.passes_cache_gate(input_tokens, matched_prefix_tokens) {
|
if !self.passes_cache_gate(input_tokens, matched_prefix_tokens) {
|
||||||
continue;
|
continue;
|
||||||
@@ -67,6 +69,7 @@ impl CacheAwarePolicy {
|
|||||||
worker: Arc::clone(worker),
|
worker: Arc::clone(worker),
|
||||||
matched_prefix_tokens,
|
matched_prefix_tokens,
|
||||||
uncached_tokens: input_tokens.saturating_sub(matched_prefix_tokens),
|
uncached_tokens: input_tokens.saturating_sub(matched_prefix_tokens),
|
||||||
|
matched_prefix_blocks,
|
||||||
candidate_range_id: ctx.candidate_range_id().to_string(),
|
candidate_range_id: ctx.candidate_range_id().to_string(),
|
||||||
max_pending_prefill_tokens: None,
|
max_pending_prefill_tokens: None,
|
||||||
});
|
});
|
||||||
@@ -106,6 +109,7 @@ impl CacheAwarePolicy {
|
|||||||
pressure_abs_threshold_tokens: self.config.pressure_abs_threshold_tokens,
|
pressure_abs_threshold_tokens: self.config.pressure_abs_threshold_tokens,
|
||||||
pressure_abs_threshold_ms: self.config.pressure_abs_threshold_ms,
|
pressure_abs_threshold_ms: self.config.pressure_abs_threshold_ms,
|
||||||
pressure_rel_threshold: self.config.pressure_rel_threshold,
|
pressure_rel_threshold: self.config.pressure_rel_threshold,
|
||||||
|
worker_queue_limit: self.config.worker_queue_limit,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -189,13 +193,24 @@ impl Policy for CacheAwarePolicy {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Caps an indexer-supplied matched-block count at the blocks the query
|
||||||
|
/// actually asked about: a query cannot match more blocks than it contains.
|
||||||
|
/// Both the token estimate and the diverted-overlap histogram read the capped
|
||||||
|
/// value, so the clamp lives here rather than at each use.
|
||||||
|
fn cap_matched_prefix_blocks(query_blocks: usize, matched_prefix_blocks: u32) -> u32 {
|
||||||
|
matched_prefix_blocks.min(u32::try_from(query_blocks).unwrap_or(u32::MAX))
|
||||||
|
}
|
||||||
|
|
||||||
fn estimate_matched_prefix_tokens(
|
fn estimate_matched_prefix_tokens(
|
||||||
input_tokens: u64,
|
input_tokens: u64,
|
||||||
query_blocks: usize,
|
query_blocks: usize,
|
||||||
matched_prefix_blocks: u32,
|
matched_prefix_blocks: u32,
|
||||||
) -> u64 {
|
) -> u64 {
|
||||||
|
let matched_prefix_blocks = u64::from(cap_matched_prefix_blocks(
|
||||||
|
query_blocks,
|
||||||
|
matched_prefix_blocks,
|
||||||
|
));
|
||||||
let query_blocks = u64::try_from(query_blocks).unwrap_or(u64::MAX).max(1);
|
let query_blocks = u64::try_from(query_blocks).unwrap_or(u64::MAX).max(1);
|
||||||
let matched_prefix_blocks = u64::from(matched_prefix_blocks).min(query_blocks);
|
|
||||||
input_tokens.saturating_mul(matched_prefix_blocks) / query_blocks
|
input_tokens.saturating_mul(matched_prefix_blocks) / query_blocks
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -207,4 +222,13 @@ mod tests {
|
|||||||
fn matched_token_estimate_caps_untrusted_block_count() {
|
fn matched_token_estimate_caps_untrusted_block_count() {
|
||||||
assert_eq!(estimate_matched_prefix_tokens(80, 8, 99), 80);
|
assert_eq!(estimate_matched_prefix_tokens(80, 8, 99), 80);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn matched_block_cap_is_shared_by_the_estimate_and_the_candidate() {
|
||||||
|
// One clamp, two readers: the histogram must never see a block count
|
||||||
|
// the token estimate would have thrown away.
|
||||||
|
assert_eq!(cap_matched_prefix_blocks(8, 99), 8);
|
||||||
|
assert_eq!(cap_matched_prefix_blocks(8, 3), 3);
|
||||||
|
assert_eq!(cap_matched_prefix_blocks(0, 3), 0);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -332,6 +332,11 @@ pub struct CacheCandidate {
|
|||||||
pub worker: Arc<Worker>,
|
pub worker: Arc<Worker>,
|
||||||
pub matched_prefix_tokens: u64,
|
pub matched_prefix_tokens: u64,
|
||||||
pub uncached_tokens: u64,
|
pub uncached_tokens: u64,
|
||||||
|
/// Matched prefix length in blocks, as reported by the prefix signal.
|
||||||
|
/// Selection reads `matched_prefix_tokens`; the block count exists for
|
||||||
|
/// observability (the diverted-overlap histogram reads against the
|
||||||
|
/// tree/indexer block domain).
|
||||||
|
pub matched_prefix_blocks: u32,
|
||||||
/// Domain containing this candidate.
|
/// Domain containing this candidate.
|
||||||
pub candidate_range_id: String,
|
pub candidate_range_id: String,
|
||||||
/// Optional pending prefill limit checked against `E`.
|
/// Optional pending prefill limit checked against `E`.
|
||||||
@@ -347,6 +352,10 @@ pub struct CacheCandidateProposal {
|
|||||||
pub pressure_abs_threshold_tokens: u64,
|
pub pressure_abs_threshold_tokens: u64,
|
||||||
pub pressure_abs_threshold_ms: Option<f64>,
|
pub pressure_abs_threshold_ms: Option<f64>,
|
||||||
pub pressure_rel_threshold: f64,
|
pub pressure_rel_threshold: f64,
|
||||||
|
/// Queue gate: a candidate whose engine reports at least this many
|
||||||
|
/// waiting requests cannot win on cache affinity. `None` disables the
|
||||||
|
/// gate. See [`crate::config::AffinityConfig::worker_queue_limit`].
|
||||||
|
pub worker_queue_limit: Option<u64>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Prefill proposal returned as either a pair or a Cache-Aware candidate set.
|
/// Prefill proposal returned as either a pair or a Cache-Aware candidate set.
|
||||||
@@ -661,6 +670,7 @@ mod tests {
|
|||||||
worker: Arc::clone(&hot),
|
worker: Arc::clone(&hot),
|
||||||
matched_prefix_tokens: 75,
|
matched_prefix_tokens: 75,
|
||||||
uncached_tokens: 25,
|
uncached_tokens: 25,
|
||||||
|
matched_prefix_blocks: 3,
|
||||||
candidate_range_id: "global".into(),
|
candidate_range_id: "global".into(),
|
||||||
max_pending_prefill_tokens: None,
|
max_pending_prefill_tokens: None,
|
||||||
}],
|
}],
|
||||||
@@ -771,8 +781,14 @@ mod tests {
|
|||||||
},
|
},
|
||||||
),
|
),
|
||||||
]);
|
]);
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&workers), &proposal, 32, &loads)
|
let decision = resolve_prefill(
|
||||||
.expect("the admitted backup must become Final P");
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&loads,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.expect("the admitted backup must become Final P");
|
||||||
assert_eq!(decision.selected.id, backup.id);
|
assert_eq!(decision.selected.id, backup.id);
|
||||||
policy.commit_prefill_selection(&ctx, proposal.kind, &decision.selected);
|
policy.commit_prefill_selection(&ctx, proposal.kind, &decision.selected);
|
||||||
|
|
||||||
@@ -1261,6 +1277,7 @@ mod tests {
|
|||||||
worker: Arc::clone(worker),
|
worker: Arc::clone(worker),
|
||||||
matched_prefix_tokens,
|
matched_prefix_tokens,
|
||||||
uncached_tokens,
|
uncached_tokens,
|
||||||
|
matched_prefix_blocks: 0,
|
||||||
candidate_range_id: "global".into(),
|
candidate_range_id: "global".into(),
|
||||||
max_pending_prefill_tokens,
|
max_pending_prefill_tokens,
|
||||||
}
|
}
|
||||||
@@ -1297,7 +1314,7 @@ mod tests {
|
|||||||
),
|
),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_cache_candidates(&proposal, 100, &loads)
|
let decision = resolve_cache_candidates(&proposal, 100, &loads, &[])
|
||||||
.decision
|
.decision
|
||||||
.expect("a later admitted cache match must survive");
|
.expect("a later admitted cache match must survive");
|
||||||
|
|
||||||
@@ -1345,7 +1362,7 @@ mod tests {
|
|||||||
),
|
),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_cache_candidates(&proposal, 100, &loads)
|
let decision = resolve_cache_candidates(&proposal, 100, &loads, &[])
|
||||||
.decision
|
.decision
|
||||||
.expect("all admitted candidates must participate in the tournament");
|
.expect("all admitted candidates must participate in the tournament");
|
||||||
|
|
||||||
@@ -1371,7 +1388,7 @@ mod tests {
|
|||||||
},
|
},
|
||||||
)]);
|
)]);
|
||||||
assert!(
|
assert!(
|
||||||
resolve_cache_candidates(&proposal, 100, &pending_allows)
|
resolve_cache_candidates(&proposal, 100, &pending_allows, &[])
|
||||||
.decision
|
.decision
|
||||||
.is_some(),
|
.is_some(),
|
||||||
"pending admission must project E=20, not L=100"
|
"pending admission must project E=20, not L=100"
|
||||||
@@ -1387,7 +1404,7 @@ mod tests {
|
|||||||
},
|
},
|
||||||
)]);
|
)]);
|
||||||
assert!(
|
assert!(
|
||||||
resolve_cache_candidates(&proposal, 100, &kv_rejects)
|
resolve_cache_candidates(&proposal, 100, &kv_rejects, &[])
|
||||||
.decision
|
.decision
|
||||||
.is_none(),
|
.is_none(),
|
||||||
"KV safety must conservatively project the complete input L=100"
|
"KV safety must conservatively project the complete input L=100"
|
||||||
@@ -1425,7 +1442,7 @@ mod tests {
|
|||||||
),
|
),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_cache_candidates(&proposal, 100, &loads)
|
let decision = resolve_cache_candidates(&proposal, 100, &loads, &[])
|
||||||
.decision
|
.decision
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(decision.selected.id, congested.id);
|
assert_eq!(decision.selected.id, congested.id);
|
||||||
@@ -1462,7 +1479,7 @@ mod tests {
|
|||||||
),
|
),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_cache_candidates(&proposal, 100, &loads)
|
let decision = resolve_cache_candidates(&proposal, 100, &loads, &[])
|
||||||
.decision
|
.decision
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
@@ -1515,7 +1532,7 @@ mod tests {
|
|||||||
),
|
),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_cache_candidates(&proposal, 100, &loads)
|
let decision = resolve_cache_candidates(&proposal, 100, &loads, &[])
|
||||||
.decision
|
.decision
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
@@ -1564,7 +1581,7 @@ mod tests {
|
|||||||
let range = CandidateRange::global(&workers);
|
let range = CandidateRange::global(&workers);
|
||||||
let proposal = SelectionProposal::with_backup(Arc::clone(&primary), Arc::clone(&backup));
|
let proposal = SelectionProposal::with_backup(Arc::clone(&primary), Arc::clone(&backup));
|
||||||
|
|
||||||
let decision = resolve_prefill(&range, &proposal, 32, &snapshot)
|
let decision = resolve_prefill(&range, &proposal, 32, &snapshot, None)
|
||||||
.expect("an admitted backup must be selected");
|
.expect("an admitted backup must be selected");
|
||||||
|
|
||||||
assert_eq!(decision.selected.id, backup.id);
|
assert_eq!(decision.selected.id, backup.id);
|
||||||
@@ -1582,6 +1599,7 @@ mod tests {
|
|||||||
&SelectionProposal::primary(Arc::clone(&primary)),
|
&SelectionProposal::primary(Arc::clone(&primary)),
|
||||||
1_000_000,
|
1_000_000,
|
||||||
&snapshot,
|
&snapshot,
|
||||||
|
None,
|
||||||
)
|
)
|
||||||
.expect("disabled reporting must preserve the healthy registry candidate");
|
.expect("disabled reporting must preserve the healthy registry candidate");
|
||||||
|
|
||||||
@@ -1614,8 +1632,14 @@ mod tests {
|
|||||||
]);
|
]);
|
||||||
let proposal = SelectionProposal::with_backup(primary, backup);
|
let proposal = SelectionProposal::with_backup(primary, backup);
|
||||||
|
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&workers), &proposal, 80, &snapshot)
|
let decision = resolve_prefill(
|
||||||
.expect("both candidates fit capacity");
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
80,
|
||||||
|
&snapshot,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.expect("both candidates fit capacity");
|
||||||
|
|
||||||
assert_eq!(decision.reason, DecisionReason::Primary);
|
assert_eq!(decision.reason, DecisionReason::Primary);
|
||||||
}
|
}
|
||||||
@@ -1658,8 +1682,14 @@ mod tests {
|
|||||||
]);
|
]);
|
||||||
let proposal = SelectionProposal::with_backup(primary, backup);
|
let proposal = SelectionProposal::with_backup(primary, backup);
|
||||||
|
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&workers), &proposal, 32, &snapshot)
|
let decision = resolve_prefill(
|
||||||
.expect("an admitted range fallback must be selected");
|
&CandidateRange::global(&workers),
|
||||||
|
&proposal,
|
||||||
|
32,
|
||||||
|
&snapshot,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.expect("an admitted range fallback must be selected");
|
||||||
|
|
||||||
assert_eq!(decision.selected.id, fallback.id);
|
assert_eq!(decision.selected.id, fallback.id);
|
||||||
assert_eq!(decision.reason, DecisionReason::RangeFallback);
|
assert_eq!(decision.reason, DecisionReason::RangeFallback);
|
||||||
|
|||||||
@@ -747,8 +747,9 @@ mod tests {
|
|||||||
(&ws[2], 0, 0, 0, 4_096),
|
(&ws[2], 0, 0, 0, 4_096),
|
||||||
]);
|
]);
|
||||||
|
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&ws), &proposal, 32, &snapshot)
|
let decision =
|
||||||
.expect("capacity exhaustion must degrade inside the filtered domain");
|
resolve_prefill(&CandidateRange::global(&ws), &proposal, 32, &snapshot, None)
|
||||||
|
.expect("capacity exhaustion must degrade inside the filtered domain");
|
||||||
assert!(matches!(decision.selected.id.0.as_str(), "a" | "b"));
|
assert!(matches!(decision.selected.id.0.as_str(), "a" | "b"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -783,8 +784,9 @@ mod tests {
|
|||||||
assert_eq!(proposal.primary.id, ws[2].id);
|
assert_eq!(proposal.primary.id, ws[2].id);
|
||||||
|
|
||||||
let snapshot = EngineLoadSnapshot::default();
|
let snapshot = EngineLoadSnapshot::default();
|
||||||
let decision = resolve_prefill(&CandidateRange::global(&ws), &proposal, 32, &snapshot)
|
let decision =
|
||||||
.expect("an eligible escape worker exists");
|
resolve_prefill(&CandidateRange::global(&ws), &proposal, 32, &snapshot, None)
|
||||||
|
.expect("an eligible escape worker exists");
|
||||||
assert_ne!(decision.selected.id, ws[2].id);
|
assert_ne!(decision.selected.id, ws[2].id);
|
||||||
assert!(matches!(decision.selected.id.0.as_str(), "a" | "b"));
|
assert!(matches!(decision.selected.id.0.as_str(), "a" | "b"));
|
||||||
|
|
||||||
|
|||||||
@@ -22,6 +22,7 @@
|
|||||||
//! The module owns the decision and reports why; it does not own the HTTP
|
//! The module owns the decision and reports why; it does not own the HTTP
|
||||||
//! response. Mapping a failed selection onto a status code stays in the route.
|
//! response. Mapping a failed selection onto a status code stays in the route.
|
||||||
|
|
||||||
|
use std::sync::atomic::{AtomicU64, Ordering as AtomicOrdering};
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
|
||||||
use crate::config::{DecodePolicyKind, PolicyKind, SessionAffinityMode};
|
use crate::config::{DecodePolicyKind, PolicyKind, SessionAffinityMode};
|
||||||
@@ -38,7 +39,7 @@ use crate::policies::engine_load::EngineLoadSnapshot;
|
|||||||
use crate::policies::{
|
use crate::policies::{
|
||||||
ExternalPrefixSignal, Policy, PrefillProposal, ProposalKind, SelectionContext,
|
ExternalPrefixSignal, Policy, PrefillProposal, ProposalKind, SelectionContext,
|
||||||
};
|
};
|
||||||
use crate::server::metrics::{MetricsRegistry, PolicySelectionFailureReason};
|
use crate::server::metrics::{CacheAwareDecision, MetricsRegistry, PolicySelectionFailureReason};
|
||||||
use crate::workers::Worker;
|
use crate::workers::Worker;
|
||||||
|
|
||||||
/// Everything one prefill selection reads. Collaborators first, then the
|
/// Everything one prefill selection reads. Collaborators first, then the
|
||||||
@@ -67,6 +68,49 @@ pub(crate) struct PrefillSelectionInputs<'a> {
|
|||||||
/// The configured mode. Without Bucket partitioning all modes reduce to
|
/// The configured mode. Without Bucket partitioning all modes reduce to
|
||||||
/// the single global domain, and the ladder applies that reduction itself.
|
/// the single global domain, and the ladder applies that reduction itself.
|
||||||
pub session_affinity_mode: SessionAffinityMode,
|
pub session_affinity_mode: SessionAffinityMode,
|
||||||
|
/// `--worker-queue-limit`. `None` disables the queue gate entirely.
|
||||||
|
pub worker_queue_limit: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The queue-gate blind warn is sampled: it fires on a per-request path, and
|
||||||
|
/// the condition (gate configured, zero fresh engine load samples fleet-wide)
|
||||||
|
/// is steady-state, so 1-in-64 is plenty to surface it without log flooding.
|
||||||
|
const QUEUE_GATE_BLIND_LOG_SAMPLE: u64 = 64;
|
||||||
|
static QUEUE_GATE_BLIND_LOG_COUNTER: AtomicU64 = AtomicU64::new(0);
|
||||||
|
|
||||||
|
/// Maps the queue-gate audit of a Cache-Aware selection that produced no
|
||||||
|
/// winner onto its decision label. Pure so every boundary is pinned by unit
|
||||||
|
/// tests rather than inferred from the ladder that calls it:
|
||||||
|
///
|
||||||
|
/// - `queue_gate_rejected == 0` means the gate took nothing out, so whatever
|
||||||
|
/// emptied the candidate set was not the gate.
|
||||||
|
/// - `fleet_all_queued` is asked BEFORE the capacity question, because
|
||||||
|
/// `all_queued` is keyed on the fleet being saturated and not on where the
|
||||||
|
/// request landed. Asking capacity first drops the saturation signal in the
|
||||||
|
/// worst case there is: a queueing fleet whose owners are also out of KV
|
||||||
|
/// books a plain `cache_miss`, and a fully saturated fleet reads as a
|
||||||
|
/// healthy one — the exact blind spot the label exists to remove.
|
||||||
|
/// - `admission_evaluated == 0` is then what makes the GATE, rather than KV
|
||||||
|
/// capacity, the reason an unsaturated fleet's candidate set came back
|
||||||
|
/// empty. Without it a capacity exhaustion books as a gate diversion
|
||||||
|
/// whenever one owner happens to be queueing.
|
||||||
|
fn cache_aware_fallback_decision(
|
||||||
|
queue_gate_rejected: u64,
|
||||||
|
admission_evaluated: u64,
|
||||||
|
fleet_all_queued: bool,
|
||||||
|
) -> CacheAwareDecision {
|
||||||
|
if queue_gate_rejected == 0 {
|
||||||
|
return CacheAwareDecision::CacheMiss;
|
||||||
|
}
|
||||||
|
if fleet_all_queued {
|
||||||
|
return CacheAwareDecision::AllQueued;
|
||||||
|
}
|
||||||
|
if admission_evaluated > 0 {
|
||||||
|
// Capacity, not the gate: owners survived the gate and then failed
|
||||||
|
// capacity admission.
|
||||||
|
return CacheAwareDecision::CacheMiss;
|
||||||
|
}
|
||||||
|
CacheAwareDecision::CacheWorkerQueued
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Runs the prefill selection ladder. `Err` carries the reason the last rung
|
/// Runs the prefill selection ladder. `Err` carries the reason the last rung
|
||||||
@@ -85,6 +129,7 @@ pub(crate) fn select_prefill_worker(
|
|||||||
tps_slo: inputs.tps_slo,
|
tps_slo: inputs.tps_slo,
|
||||||
},
|
},
|
||||||
failure_reason: PolicySelectionFailureReason::ProposalEmpty,
|
failure_reason: PolicySelectionFailureReason::ProposalEmpty,
|
||||||
|
cache_gate_audit: None,
|
||||||
};
|
};
|
||||||
let selected = selector.run();
|
let selected = selector.run();
|
||||||
selected.ok_or(selector.failure_reason)
|
selected.ok_or(selector.failure_reason)
|
||||||
@@ -98,6 +143,12 @@ struct Selector<'a> {
|
|||||||
inputs: &'a PrefillSelectionInputs<'a>,
|
inputs: &'a PrefillSelectionInputs<'a>,
|
||||||
bucket_request: BucketRequest,
|
bucket_request: BucketRequest,
|
||||||
failure_reason: PolicySelectionFailureReason,
|
failure_reason: PolicySelectionFailureReason,
|
||||||
|
/// Queue-gate audit of a Cache-Aware resolution that produced no winner:
|
||||||
|
/// (gate-rejected candidates, candidates that reached capacity admission,
|
||||||
|
/// fleet saturation, deepest rejected prefix). `None` when there was no
|
||||||
|
/// resolution at all — no load snapshot, or no candidate proposal — which
|
||||||
|
/// reads as a plain miss because nothing was gated out.
|
||||||
|
cache_gate_audit: Option<(u64, u64, bool, u32)>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<'a> Selector<'a> {
|
impl<'a> Selector<'a> {
|
||||||
@@ -117,6 +168,7 @@ impl<'a> Selector<'a> {
|
|||||||
|
|
||||||
// Cache-Aware resolves one bounded global candidate set and returns a final winner.
|
// Cache-Aware resolves one bounded global candidate set and returns a final winner.
|
||||||
let cache_winner = self.cache_winner();
|
let cache_winner = self.cache_winner();
|
||||||
|
let cache_winner_hit = cache_winner.is_some();
|
||||||
|
|
||||||
let global_affinity_probe = use_global_affinity_probe
|
let global_affinity_probe = use_global_affinity_probe
|
||||||
.then(|| {
|
.then(|| {
|
||||||
@@ -147,7 +199,7 @@ impl<'a> Selector<'a> {
|
|||||||
// Rebuild the backup inside the primary's own Bucket.
|
// Rebuild the backup inside the primary's own Bucket.
|
||||||
.and_then(|domain| self.select_in_domain(&domain, true, false, false));
|
.and_then(|domain| self.select_in_domain(&domain, true, false, false));
|
||||||
|
|
||||||
cache_winner.or_else(|| {
|
let selected = cache_winner.or_else(|| {
|
||||||
// Materializing the normal domains clones the member list of every
|
// Materializing the normal domains clones the member list of every
|
||||||
// Bucket, so build them only on the rung that actually reads them.
|
// Bucket, so build them only on the rung that actually reads them.
|
||||||
let prefill_domains = || {
|
let prefill_domains = || {
|
||||||
@@ -173,7 +225,32 @@ impl<'a> Selector<'a> {
|
|||||||
self.select_domains(&prefill_domains(), true, true)
|
self.select_domains(&prefill_domains(), true, true)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
})
|
});
|
||||||
|
// Only a selection that resolved a worker books a decision. A ladder
|
||||||
|
// that ran out of rungs is a 503, already counted by
|
||||||
|
// `sgl_router_policy_selection_failures_total`; booking it here too
|
||||||
|
// would break the documented sum and, worse, let a request that
|
||||||
|
// reached nothing book `cache_worker_queued` and contribute to
|
||||||
|
// `sgl_router_diverted_overlap_blocks` — a diversion that never
|
||||||
|
// arrived is not evidence about what the gate traded away.
|
||||||
|
if inputs.policy_kind == PolicyKind::CacheAware && !cache_winner_hit && selected.is_some() {
|
||||||
|
let (rejected, evaluated, fleet_all_queued, blocks) =
|
||||||
|
self.cache_gate_audit.unwrap_or((0, 0, false, 0));
|
||||||
|
let decision = cache_aware_fallback_decision(rejected, evaluated, fleet_all_queued);
|
||||||
|
if matches!(decision, CacheAwareDecision::CacheWorkerQueued) {
|
||||||
|
// A real diversion: an unqueued destination existed and the
|
||||||
|
// gate gave up `blocks` of matched prefix to reach it. The
|
||||||
|
// histogram is the evidence for whether the gate is trading
|
||||||
|
// large cached prefixes for short waits.
|
||||||
|
inputs
|
||||||
|
.metrics
|
||||||
|
.observe_diverted_overlap_blocks(&inputs.model_id.0, u64::from(blocks));
|
||||||
|
}
|
||||||
|
inputs
|
||||||
|
.metrics
|
||||||
|
.record_cache_aware_decision(&inputs.model_id.0, decision);
|
||||||
|
}
|
||||||
|
selected
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The per-request `SelectionContext` every rung starts from.
|
/// The per-request `SelectionContext` every rung starts from.
|
||||||
@@ -206,8 +283,42 @@ impl<'a> Selector<'a> {
|
|||||||
return None;
|
return None;
|
||||||
};
|
};
|
||||||
let bounded_candidate_count = proposal.candidates.len();
|
let bounded_candidate_count = proposal.candidates.len();
|
||||||
let cache_decision =
|
// The queue gate reads the engine-published load sample and fails open
|
||||||
resolve_cache_candidates(&proposal, inputs.request_input_tokens, snapshot);
|
// per worker. When NO worker has a fresh sample the gate is inert
|
||||||
|
// fleet-wide and nothing would say so: `cache_worker_queued` sitting at
|
||||||
|
// 0 is indistinguishable from a healthy fleet. Warn (sampled) — a fleet
|
||||||
|
// that never advertised a load port must not silently disable the gate.
|
||||||
|
if inputs.worker_queue_limit.is_some()
|
||||||
|
&& !inputs.workers.is_empty()
|
||||||
|
&& inputs
|
||||||
|
.workers
|
||||||
|
.iter()
|
||||||
|
.all(|worker| snapshot.fresh_load_for_url(&worker.url).is_none())
|
||||||
|
&& QUEUE_GATE_BLIND_LOG_COUNTER
|
||||||
|
.fetch_add(1, AtomicOrdering::Relaxed)
|
||||||
|
.is_multiple_of(QUEUE_GATE_BLIND_LOG_SAMPLE)
|
||||||
|
{
|
||||||
|
tracing::warn!(
|
||||||
|
model = %inputs.model_id,
|
||||||
|
worker_queue_limit = inputs.worker_queue_limit,
|
||||||
|
workers = inputs.workers.len(),
|
||||||
|
"--worker-queue-limit is set but no worker has a fresh engine load \
|
||||||
|
sample, so the queue gate is inert. Check that engines advertise a \
|
||||||
|
load port and publish LoadStat",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let cache_decision = resolve_cache_candidates(
|
||||||
|
&proposal,
|
||||||
|
inputs.request_input_tokens,
|
||||||
|
snapshot,
|
||||||
|
inputs.workers,
|
||||||
|
);
|
||||||
|
self.cache_gate_audit = Some((
|
||||||
|
cache_decision.queue_gate_rejected_candidates,
|
||||||
|
cache_decision.admission_evaluated_candidates,
|
||||||
|
cache_decision.fleet_all_queued,
|
||||||
|
cache_decision.queue_gate_best_rejected_blocks,
|
||||||
|
));
|
||||||
inputs
|
inputs
|
||||||
.metrics
|
.metrics
|
||||||
.record_cache_admission_evaluations(cache_decision.admission_evaluated_candidates);
|
.record_cache_admission_evaluations(cache_decision.admission_evaluated_candidates);
|
||||||
@@ -246,6 +357,18 @@ impl<'a> Selector<'a> {
|
|||||||
inputs
|
inputs
|
||||||
.metrics
|
.metrics
|
||||||
.record_policy_decision("cache_aware", "cache_candidate");
|
.record_policy_decision("cache_aware", "cache_candidate");
|
||||||
|
inputs.metrics.record_cache_aware_decision(
|
||||||
|
&inputs.model_id.0,
|
||||||
|
if cache_decision.queue_gate_fell_back {
|
||||||
|
// The gate removed every owner and nowhere in the fleet is
|
||||||
|
// unqueued, so the prefix was kept rather than traded for a
|
||||||
|
// wait that cannot be dodged. Booked as saturation, never as
|
||||||
|
// a plain hit.
|
||||||
|
CacheAwareDecision::AllQueued
|
||||||
|
} else {
|
||||||
|
CacheAwareDecision::CacheHit
|
||||||
|
},
|
||||||
|
);
|
||||||
Some(decision.selected)
|
Some(decision.selected)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -317,6 +440,7 @@ impl<'a> Selector<'a> {
|
|||||||
&proposal,
|
&proposal,
|
||||||
inputs.request_input_tokens,
|
inputs.request_input_tokens,
|
||||||
snapshot,
|
snapshot,
|
||||||
|
inputs.worker_queue_limit,
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
resolve_prefill_admitted(
|
resolve_prefill_admitted(
|
||||||
@@ -324,6 +448,7 @@ impl<'a> Selector<'a> {
|
|||||||
&proposal,
|
&proposal,
|
||||||
inputs.request_input_tokens,
|
inputs.request_input_tokens,
|
||||||
snapshot,
|
snapshot,
|
||||||
|
inputs.worker_queue_limit,
|
||||||
)
|
)
|
||||||
};
|
};
|
||||||
let Some(decision) = decision else {
|
let Some(decision) = decision else {
|
||||||
@@ -521,17 +646,20 @@ fn projected_decode_kv_tokens(input_tokens: u64, max_output_tokens: Option<u64>)
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::{
|
use super::{
|
||||||
prefill_policy_reason, projected_decode_kv_tokens, select_decode_peer,
|
cache_aware_fallback_decision, prefill_policy_reason, projected_decode_kv_tokens,
|
||||||
select_prefill_worker, DecodeSelectionInputs, PrefillSelectionInputs,
|
select_decode_peer, select_prefill_worker, DecodeSelectionInputs, PrefillSelectionInputs,
|
||||||
};
|
};
|
||||||
use crate::config::{DecodePolicyKind, PolicyKind, SessionAffinityMode};
|
use crate::config::{AffinityConfig, DecodePolicyKind, PolicyKind, SessionAffinityMode};
|
||||||
use crate::discovery::{ModelId, WorkerId, WorkerMode, WorkerSpec};
|
use crate::discovery::{ModelId, WorkerId, WorkerMode, WorkerSpec};
|
||||||
use crate::policies::admission::{resolve_prefill_admitted, CandidateRange, DecisionReason};
|
use crate::policies::admission::{resolve_prefill_admitted, CandidateRange, DecisionReason};
|
||||||
use crate::policies::buckets::BucketSelector;
|
use crate::policies::buckets::BucketSelector;
|
||||||
|
use crate::policies::cache_aware::CacheAwarePolicy;
|
||||||
use crate::policies::engine_load::{EngineLoadSnapshot, NativeCacheWorkerLoad};
|
use crate::policies::engine_load::{EngineLoadSnapshot, NativeCacheWorkerLoad};
|
||||||
use crate::policies::power_of_two::PowerOfTwoChoicesPolicy;
|
use crate::policies::power_of_two::PowerOfTwoChoicesPolicy;
|
||||||
use crate::policies::{Policy, ProposalKind, SelectionProposal};
|
use crate::policies::{ExternalPrefixSignal, Policy, ProposalKind, SelectionProposal};
|
||||||
use crate::server::metrics::{MetricsRegistry, PolicySelectionFailureReason};
|
use crate::server::metrics::{
|
||||||
|
CacheAwareDecision, MetricsRegistry, PolicySelectionFailureReason,
|
||||||
|
};
|
||||||
use crate::workers::Worker;
|
use crate::workers::Worker;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use std::time::Instant;
|
use std::time::Instant;
|
||||||
@@ -573,6 +701,62 @@ mod tests {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `(worker, waiting requests, tokens already held, published KV
|
||||||
|
/// capacity)`. The queue gate reads `num_waiting_reqs`, which the plain
|
||||||
|
/// [`snapshot`] fixture pins at zero.
|
||||||
|
fn queued_snapshot(entries: &[(&Arc<Worker>, u64, u64, u64)]) -> EngineLoadSnapshot {
|
||||||
|
EngineLoadSnapshot::from_native_cache_workers(
|
||||||
|
7,
|
||||||
|
entries
|
||||||
|
.iter()
|
||||||
|
.map(|(worker, waiting, used, capacity)| {
|
||||||
|
(
|
||||||
|
worker.url.clone(),
|
||||||
|
NativeCacheWorkerLoad {
|
||||||
|
num_running_reqs: 1,
|
||||||
|
num_waiting_reqs: *waiting,
|
||||||
|
num_waiting_uncached_tokens: *waiting,
|
||||||
|
num_used_tokens: *used,
|
||||||
|
num_total_tokens: *used,
|
||||||
|
max_total_num_tokens: *capacity,
|
||||||
|
max_running_requests: 64,
|
||||||
|
prefill_throughput_tokens_per_s: None,
|
||||||
|
estimated_prefill_queue_ms: None,
|
||||||
|
captured_at: Instant::now(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An indexer hit placing `matched_prefix_blocks` on each named worker.
|
||||||
|
fn prefix_signal(matches: &[(&Arc<Worker>, u32)], query_blocks: usize) -> ExternalPrefixSignal {
|
||||||
|
ExternalPrefixSignal {
|
||||||
|
outcome: sgl_kv_indexer::PrefixOutcome::Matched {
|
||||||
|
matches: matches
|
||||||
|
.iter()
|
||||||
|
.map(|(worker, blocks)| sgl_kv_indexer::PrefixMatch {
|
||||||
|
matched_prefix_blocks: *blocks,
|
||||||
|
worker_id: worker.id.0.clone(),
|
||||||
|
address: worker.url.clone(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
best_prefix_blocks: matches.iter().map(|(_, blocks)| *blocks).max().unwrap_or(0),
|
||||||
|
},
|
||||||
|
query_blocks,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn booked_decisions(metrics: &MetricsRegistry) -> Vec<String> {
|
||||||
|
metrics
|
||||||
|
.render()
|
||||||
|
.lines()
|
||||||
|
.filter(|line| line.starts_with("sgl_router_cache_aware_decisions_total{"))
|
||||||
|
.map(str::to_owned)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
/// Power-of-two prefill, Bucket partitioning off, no session affinity —
|
/// Power-of-two prefill, Bucket partitioning off, no session affinity —
|
||||||
/// the single global domain, which is what isolates the ladder's rungs.
|
/// the single global domain, which is what isolates the ladder's rungs.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
@@ -599,6 +783,7 @@ mod tests {
|
|||||||
external_prefix: None,
|
external_prefix: None,
|
||||||
load_snapshot,
|
load_snapshot,
|
||||||
workers,
|
workers,
|
||||||
|
worker_queue_limit: None,
|
||||||
ttft_slo_ms: None,
|
ttft_slo_ms: None,
|
||||||
tps_slo: None,
|
tps_slo: None,
|
||||||
session_affinity_mode: SessionAffinityMode::Bucket,
|
session_affinity_mode: SessionAffinityMode::Bucket,
|
||||||
@@ -651,6 +836,150 @@ mod tests {
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_failed_cache_aware_selection_books_no_decision() {
|
||||||
|
// The documented invariant on `sgl_router_cache_aware_decisions_total`
|
||||||
|
// is one decision per selection that RESOLVES a worker, so the labels
|
||||||
|
// sum to the cache-aware rate less the 503s. A ladder that ran out of
|
||||||
|
// rungs is already counted by the failure counter; booking it here too
|
||||||
|
// would break that sum and let a request that reached nothing feed
|
||||||
|
// `sgl_router_diverted_overlap_blocks`.
|
||||||
|
let policy = PowerOfTwoChoicesPolicy::new();
|
||||||
|
let buckets = BucketSelector::new(None);
|
||||||
|
let metrics = MetricsRegistry::new();
|
||||||
|
let model = ModelId("model".into());
|
||||||
|
let workers: Vec<Arc<Worker>> = Vec::new();
|
||||||
|
let loads = snapshot(&[]);
|
||||||
|
let mut inputs = prefill_inputs(
|
||||||
|
&policy,
|
||||||
|
&buckets,
|
||||||
|
&metrics,
|
||||||
|
&model,
|
||||||
|
&workers,
|
||||||
|
Some(&loads),
|
||||||
|
64,
|
||||||
|
);
|
||||||
|
inputs.policy_kind = PolicyKind::CacheAware;
|
||||||
|
|
||||||
|
assert!(select_prefill_worker(&inputs).is_err());
|
||||||
|
|
||||||
|
let rendered = metrics.render();
|
||||||
|
let booked: Vec<&str> = rendered
|
||||||
|
.lines()
|
||||||
|
.filter(|line| line.starts_with("sgl_router_cache_aware_decisions_total{"))
|
||||||
|
.collect();
|
||||||
|
assert!(
|
||||||
|
booked.is_empty(),
|
||||||
|
"a 503 selection must book no cache-aware decision, got {booked:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The wiring between `resolve_cache_candidates`' audit and the decision
|
||||||
|
/// label, which the resolver test and the pure-mapper test each cover only
|
||||||
|
/// one side of. This is the end the `selected.is_some()` guard could
|
||||||
|
/// silently kill.
|
||||||
|
#[test]
|
||||||
|
fn a_saturated_capacity_exhausted_fleet_books_all_queued_through_the_ladder() {
|
||||||
|
let owner = worker("owner");
|
||||||
|
let shallow_owner = worker("shallow-owner");
|
||||||
|
let workers = vec![Arc::clone(&owner), Arc::clone(&shallow_owner)];
|
||||||
|
// Both owners are over the limit (saturation) AND out of KV, so the
|
||||||
|
// re-admitted set yields no winner and the ladder falls through to the
|
||||||
|
// capacity fallback. Before the ordering fix this booked `cache_miss`.
|
||||||
|
let loads = queued_snapshot(&[
|
||||||
|
(&owner, 9, 10_000, 10_000),
|
||||||
|
(&shallow_owner, 5, 10_000, 10_000),
|
||||||
|
]);
|
||||||
|
let signal = prefix_signal(&[(&owner, 9), (&shallow_owner, 4)], 10);
|
||||||
|
let config = AffinityConfig {
|
||||||
|
worker_queue_limit: Some(4),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let policy = CacheAwarePolicy::new(config);
|
||||||
|
let buckets = BucketSelector::new(None);
|
||||||
|
let metrics = MetricsRegistry::new();
|
||||||
|
let model = ModelId("model".into());
|
||||||
|
let mut inputs = prefill_inputs(
|
||||||
|
&policy,
|
||||||
|
&buckets,
|
||||||
|
&metrics,
|
||||||
|
&model,
|
||||||
|
&workers,
|
||||||
|
Some(&loads),
|
||||||
|
100_000,
|
||||||
|
);
|
||||||
|
inputs.policy_kind = PolicyKind::CacheAware;
|
||||||
|
inputs.worker_queue_limit = Some(4);
|
||||||
|
inputs.external_prefix = Some(&signal);
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
select_prefill_worker(&inputs).is_ok(),
|
||||||
|
"a saturated fleet must still route"
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
booked_decisions(&metrics),
|
||||||
|
vec![
|
||||||
|
r#"sgl_router_cache_aware_decisions_total{model_id="model",decision="all_queued"} 1"#
|
||||||
|
],
|
||||||
|
"saturation must survive a capacity-exhausted re-admission"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The other side of the same wiring: an unsaturated fleet where the gate
|
||||||
|
/// really did divert, which must book the diversion AND the prefix depth
|
||||||
|
/// it gave up.
|
||||||
|
#[test]
|
||||||
|
fn a_real_diversion_books_cache_worker_queued_and_its_overlap_depth() {
|
||||||
|
let owner = worker("owner");
|
||||||
|
let idle = worker("idle");
|
||||||
|
let workers = vec![Arc::clone(&owner), Arc::clone(&idle)];
|
||||||
|
// The only prefix owner is queueing; a non-owner is idle, so a
|
||||||
|
// diversion can dodge the wait and the fleet is NOT saturated.
|
||||||
|
// `cache_affinity_min_matched_tokens` defaults to 1024, so the request
|
||||||
|
// has to be large enough for a 7/10-block match to clear it, or the
|
||||||
|
// candidate never reaches the gate at all and this books a plain miss.
|
||||||
|
let loads = queued_snapshot(&[(&owner, 9, 10, 10_000_000), (&idle, 0, 10, 10_000_000)]);
|
||||||
|
let signal = prefix_signal(&[(&owner, 7)], 10);
|
||||||
|
let config = AffinityConfig {
|
||||||
|
worker_queue_limit: Some(4),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let policy = CacheAwarePolicy::new(config);
|
||||||
|
let buckets = BucketSelector::new(None);
|
||||||
|
let metrics = MetricsRegistry::new();
|
||||||
|
let model = ModelId("model".into());
|
||||||
|
let mut inputs = prefill_inputs(
|
||||||
|
&policy,
|
||||||
|
&buckets,
|
||||||
|
&metrics,
|
||||||
|
&model,
|
||||||
|
&workers,
|
||||||
|
Some(&loads),
|
||||||
|
100_000,
|
||||||
|
);
|
||||||
|
inputs.policy_kind = PolicyKind::CacheAware;
|
||||||
|
inputs.worker_queue_limit = Some(4);
|
||||||
|
inputs.external_prefix = Some(&signal);
|
||||||
|
|
||||||
|
let selected = select_prefill_worker(&inputs).expect("an idle worker exists");
|
||||||
|
|
||||||
|
assert_eq!(selected.id, idle.id, "the gate must divert off the prefix");
|
||||||
|
assert_eq!(
|
||||||
|
booked_decisions(&metrics),
|
||||||
|
vec![
|
||||||
|
r#"sgl_router_cache_aware_decisions_total{model_id="model",decision="cache_worker_queued"} 1"#
|
||||||
|
]
|
||||||
|
);
|
||||||
|
// The depth given up is the evidence the histogram exists for, and a
|
||||||
|
// diversion that never arrived must never reach it.
|
||||||
|
let rendered = metrics.render();
|
||||||
|
assert!(
|
||||||
|
rendered.contains(r#"sgl_router_diverted_overlap_blocks_count{model_id="model"} 1"#),
|
||||||
|
"a real diversion must observe its overlap depth, got:\n{rendered}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn a_saturated_fleet_still_routes_through_the_capacity_fallback() {
|
fn a_saturated_fleet_still_routes_through_the_capacity_fallback() {
|
||||||
let full = worker("full");
|
let full = worker("full");
|
||||||
@@ -667,6 +996,7 @@ mod tests {
|
|||||||
&SelectionProposal::with_backup(Arc::clone(&full), Arc::clone(&also_full)),
|
&SelectionProposal::with_backup(Arc::clone(&full), Arc::clone(&also_full)),
|
||||||
64,
|
64,
|
||||||
&loads,
|
&loads,
|
||||||
|
None,
|
||||||
)
|
)
|
||||||
.is_none(),
|
.is_none(),
|
||||||
"fixture must saturate every worker so the strict rung admits none",
|
"fixture must saturate every worker so the strict rung admits none",
|
||||||
@@ -842,4 +1172,49 @@ mod tests {
|
|||||||
"cache_candidate"
|
"cache_candidate"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cache_aware_fallback_decision_needs_the_gate_to_have_emptied_the_set() {
|
||||||
|
// The trap this pins: on an UNSATURATED fleet, one owner queueing
|
||||||
|
// while the others exhaust KV capacity is a CAPACITY problem, not a
|
||||||
|
// gate diversion. Only a gate that removed every owner leaves zero
|
||||||
|
// candidates evaluated.
|
||||||
|
assert!(matches!(
|
||||||
|
cache_aware_fallback_decision(1, 3, false),
|
||||||
|
CacheAwareDecision::CacheMiss
|
||||||
|
));
|
||||||
|
// Nothing gated out at all: a plain miss, saturated or not.
|
||||||
|
assert!(matches!(
|
||||||
|
cache_aware_fallback_decision(0, 0, false),
|
||||||
|
CacheAwareDecision::CacheMiss
|
||||||
|
));
|
||||||
|
assert!(matches!(
|
||||||
|
cache_aware_fallback_decision(0, 4, true),
|
||||||
|
CacheAwareDecision::CacheMiss
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cache_aware_fallback_decision_separates_diversion_from_saturation() {
|
||||||
|
// Gate removed every owner and somewhere unqueued exists: a real
|
||||||
|
// diversion off the prefix. `resolve_cache_candidates` leaves
|
||||||
|
// `evaluated` at zero here because its second tier does not fire on
|
||||||
|
// an unsaturated fleet.
|
||||||
|
assert!(matches!(
|
||||||
|
cache_aware_fallback_decision(2, 0, false),
|
||||||
|
CacheAwareDecision::CacheWorkerQueued
|
||||||
|
));
|
||||||
|
// Saturation, in the shape the resolver actually produces: the
|
||||||
|
// second tier re-admitted the gated-out owners, so `evaluated` is
|
||||||
|
// NON-zero, and they then failed hard admission. Booking the
|
||||||
|
// capacity outcome here would drop the saturation signal exactly
|
||||||
|
// where it matters — hence saturation is asked first. Pinning
|
||||||
|
// `(2, 0, true)` instead would assert a state the resolver cannot
|
||||||
|
// reach: re-admission and the `AllQueued` precondition are the same
|
||||||
|
// condition, so an empty `evaluated` never survives it.
|
||||||
|
assert!(matches!(
|
||||||
|
cache_aware_fallback_decision(2, 2, true),
|
||||||
|
CacheAwareDecision::AllQueued
|
||||||
|
));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -39,8 +39,42 @@
|
|||||||
//! | `sgl_router_cache_pressure_guard_compared_total` | Counter | — |
|
//! | `sgl_router_cache_pressure_guard_compared_total` | Counter | — |
|
||||||
//! | `sgl_router_cache_pressure_guard_override_total` | Counter | — |
|
//! | `sgl_router_cache_pressure_guard_override_total` | Counter | — |
|
||||||
//! | `sgl_router_cache_monitor_decisions_total` | Counter | `source` |
|
//! | `sgl_router_cache_monitor_decisions_total` | Counter | `source` |
|
||||||
|
//! | `sgl_router_cache_aware_decisions_total` | Counter | `model_id`, `decision` |
|
||||||
|
//! | `sgl_router_diverted_overlap_blocks` | Histogram | `model_id` |
|
||||||
//! | `sgl_router_ingress_tokenize_errors_total` | Counter | `model_id` |
|
//! | `sgl_router_ingress_tokenize_errors_total` | Counter | `model_id` |
|
||||||
//!
|
//!
|
||||||
|
//! `sgl_router_cache_aware_decisions_total` records exactly one decision per
|
||||||
|
//! cache-aware prefill selection that resolves a worker, so the labels sum to
|
||||||
|
//! the cache-aware request rate less the selections that ended in a 503 (see
|
||||||
|
//! `sgl_router_policy_selection_failures_total` for those) and ratios between
|
||||||
|
//! them are meaningful:
|
||||||
|
//!
|
||||||
|
//! - `cache_hit` — a prefix owner won the selection. Note this includes a
|
||||||
|
//! PARTIAL gate diversion: when the gate removed the deepest owner but a
|
||||||
|
//! shallower one survived, an owner still won, so the request books here
|
||||||
|
//! and contributes nothing to `sgl_router_diverted_overlap_blocks`.
|
||||||
|
//! - `cache_miss` — no usable prefix owner (tree miss, or every owner
|
||||||
|
//! rejected by hard capacity admission). A tree miss books here even on a
|
||||||
|
//! saturated fleet: with no prefix owner the gate never fired, so there was
|
||||||
|
//! no affinity to keep or trade. `all_queued` is the saturation signal for
|
||||||
|
//! traffic the gate ACTED on, not a fleet-wide saturation gauge — read
|
||||||
|
//! engine queue depth for that.
|
||||||
|
//! - `cache_worker_queued` — the queue gate (`--worker-queue-limit`) removed
|
||||||
|
//! every owner while an unqueued destination still existed, so the request
|
||||||
|
//! was diverted off its prefix. The matched-prefix depth it gave up is in
|
||||||
|
//! `sgl_router_diverted_overlap_blocks` — read it against the overlap of
|
||||||
|
//! all selections: a diverted curve skewing high means the gate is trading
|
||||||
|
//! large cached prefixes for short waits.
|
||||||
|
//! - `all_queued` — the queue gate removed every owner AND every worker in
|
||||||
|
//! the prefill fleet is queueing, so no diversion could dodge a wait. This
|
||||||
|
//! is the saturation signal for traffic the gate acted on, keyed on
|
||||||
|
//! saturation rather than on where the request landed: usually the request
|
||||||
|
//! kept its prefix, but when the re-admitted owners are also out of KV
|
||||||
|
//! capacity it lands off-owner and still books here. Reporting that case as `cache_miss` would hide the
|
||||||
|
//! saturation in the one state where it matters most. It deliberately
|
||||||
|
//! does NOT spell `cache_hit*`: a `decision=~"cache_hit.*"` hit-rate query
|
||||||
|
//! must not absorb it, or a fully saturated fleet reads as a healthy one.
|
||||||
|
//!
|
||||||
//! The four `sgl_router_worker*` gauges and `sgl_router_workers` are sampled
|
//! The four `sgl_router_worker*` gauges and `sgl_router_workers` are sampled
|
||||||
//! at scrape time from the live [`crate::workers::WorkerRegistry`] (passed to
|
//! at scrape time from the live [`crate::workers::WorkerRegistry`] (passed to
|
||||||
//! [`MetricsRegistry::render_with_workers`]) rather than pushed — there is no
|
//! [`MetricsRegistry::render_with_workers`]) rather than pushed — there is no
|
||||||
@@ -87,6 +121,14 @@ const TTFT_BUCKETS: &[f64] = &[
|
|||||||
400.0,
|
400.0,
|
||||||
];
|
];
|
||||||
|
|
||||||
|
/// Histogram bucket upper bounds (blocks) for
|
||||||
|
/// `sgl_router_diverted_overlap_blocks`. Powers of two up to 8192 blocks;
|
||||||
|
/// block size is engine-configured (commonly 16–64 tokens), so the ladder
|
||||||
|
/// spans ~16 tokens to ~512K tokens of forfeited prefix.
|
||||||
|
const OVERLAP_BLOCK_BUCKETS: &[f64] = &[
|
||||||
|
1.0, 2.0, 4.0, 8.0, 16.0, 32.0, 64.0, 128.0, 256.0, 512.0, 1024.0, 2048.0, 4096.0, 8192.0,
|
||||||
|
];
|
||||||
|
|
||||||
/// Recordable outcome for a request — narrowed to a handful of variants so
|
/// Recordable outcome for a request — narrowed to a handful of variants so
|
||||||
/// the label cardinality stays bounded.
|
/// the label cardinality stays bounded.
|
||||||
#[derive(Debug, Clone, Copy)]
|
#[derive(Debug, Clone, Copy)]
|
||||||
@@ -223,6 +265,27 @@ pub(crate) enum PolicySelectionFailureReason {
|
|||||||
ProposalEmpty,
|
ProposalEmpty,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Final cache-aware routing decision, one per prefill selection. See the
|
||||||
|
/// module doc for how the labels read against each other.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub enum CacheAwareDecision {
|
||||||
|
CacheHit,
|
||||||
|
CacheMiss,
|
||||||
|
CacheWorkerQueued,
|
||||||
|
AllQueued,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl CacheAwareDecision {
|
||||||
|
fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
Self::CacheHit => "cache_hit",
|
||||||
|
Self::CacheMiss => "cache_miss",
|
||||||
|
Self::CacheWorkerQueued => "cache_worker_queued",
|
||||||
|
Self::AllQueued => "all_queued",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl PolicySelectionFailureReason {
|
impl PolicySelectionFailureReason {
|
||||||
pub(crate) fn as_str(self) -> &'static str {
|
pub(crate) fn as_str(self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
@@ -280,6 +343,8 @@ pub struct MetricsRegistry {
|
|||||||
cache_pressure_guard_compared_total: AtomicU64,
|
cache_pressure_guard_compared_total: AtomicU64,
|
||||||
cache_pressure_guard_override_total: AtomicU64,
|
cache_pressure_guard_override_total: AtomicU64,
|
||||||
cache_monitor_decisions_total: Mutex<HashMap<&'static str, Arc<AtomicU64>>>,
|
cache_monitor_decisions_total: Mutex<HashMap<&'static str, Arc<AtomicU64>>>,
|
||||||
|
cache_aware_decisions_total: Mutex<HashMap<CacheAwareDecisionKey, Arc<AtomicU64>>>,
|
||||||
|
diverted_overlap_blocks: Mutex<HashMap<String, Histogram>>,
|
||||||
ingress_tokenize_errors_total: Mutex<HashMap<String, Arc<AtomicU64>>>,
|
ingress_tokenize_errors_total: Mutex<HashMap<String, Arc<AtomicU64>>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -344,6 +409,12 @@ struct PolicyDecisionKey {
|
|||||||
reason: String,
|
reason: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Hash, Eq, PartialEq, Clone)]
|
||||||
|
struct CacheAwareDecisionKey {
|
||||||
|
model_id: String,
|
||||||
|
decision: &'static str,
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
struct Histogram {
|
struct Histogram {
|
||||||
/// Bucket upper bounds this histogram observes against. Held per-instance
|
/// Bucket upper bounds this histogram observes against. Held per-instance
|
||||||
@@ -621,6 +692,38 @@ impl MetricsRegistry {
|
|||||||
counter.fetch_add(1, Ordering::Relaxed);
|
counter.fetch_add(1, Ordering::Relaxed);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Record the final cache-aware routing decision for one prefill
|
||||||
|
/// selection — exactly one call per cache-aware request, so the labels
|
||||||
|
/// sum to the cache-aware request rate.
|
||||||
|
pub fn record_cache_aware_decision(&self, model_id: &str, decision: CacheAwareDecision) {
|
||||||
|
let key = CacheAwareDecisionKey {
|
||||||
|
model_id: model_id.to_owned(),
|
||||||
|
decision: decision.as_str(),
|
||||||
|
};
|
||||||
|
let mut guard = self.cache_aware_decisions_total.lock();
|
||||||
|
let counter = guard
|
||||||
|
.entry(key)
|
||||||
|
.or_insert_with(|| Arc::new(AtomicU64::new(0)))
|
||||||
|
.clone();
|
||||||
|
drop(guard);
|
||||||
|
counter.fetch_add(1, Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Observe the matched-prefix depth (blocks) a queue-gate diversion gave
|
||||||
|
/// up, for `sgl_router_diverted_overlap_blocks`. Recorded ONLY when the
|
||||||
|
/// gate emptied the candidate set (`cache_worker_queued`), so the
|
||||||
|
/// histogram measures sacrifice rather than traffic. A PARTIAL diversion
|
||||||
|
/// — the gate removed the deepest owner but a shallower one still won —
|
||||||
|
/// is therefore not represented here even though some locality was given
|
||||||
|
/// up; it books as `cache_hit`.
|
||||||
|
pub fn observe_diverted_overlap_blocks(&self, model_id: &str, blocks: u64) {
|
||||||
|
let mut guard = self.diverted_overlap_blocks.lock();
|
||||||
|
let hist = guard
|
||||||
|
.entry(model_id.to_owned())
|
||||||
|
.or_insert_with(|| Histogram::new(OVERLAP_BLOCK_BUCKETS));
|
||||||
|
hist.observe(blocks as f64);
|
||||||
|
}
|
||||||
|
|
||||||
/// Bump `sgl_router_ingress_tokenize_errors_total{model_id}`.
|
/// Bump `sgl_router_ingress_tokenize_errors_total{model_id}`.
|
||||||
///
|
///
|
||||||
/// Recorded ONLY when the tokenization offload SHOULD have fired but the
|
/// Recorded ONLY when the tokenization offload SHOULD have fired but the
|
||||||
@@ -1024,6 +1127,47 @@ impl MetricsRegistry {
|
|||||||
}
|
}
|
||||||
drop(guard);
|
drop(guard);
|
||||||
|
|
||||||
|
// cache_aware_decisions_total
|
||||||
|
out.push_str(
|
||||||
|
"# HELP sgl_router_cache_aware_decisions_total Final Cache-Aware routing decisions, one per selection that resolved a worker: cache_hit = prefix owner won; cache_miss = no usable owner (a tree miss books here even under saturation, because the gate never fired); cache_worker_queued = queue gate diverted the request off its prefix; all_queued = queue gate fired but every worker is queueing (fleet-saturation signal, not a cache hit).\n",
|
||||||
|
);
|
||||||
|
out.push_str("# TYPE sgl_router_cache_aware_decisions_total counter\n");
|
||||||
|
let guard = self.cache_aware_decisions_total.lock();
|
||||||
|
let mut entries: Vec<(&CacheAwareDecisionKey, u64)> = guard
|
||||||
|
.iter()
|
||||||
|
.map(|(k, v)| (k, v.load(Ordering::Relaxed)))
|
||||||
|
.collect();
|
||||||
|
entries.sort_by(|a, b| (&a.0.model_id, a.0.decision).cmp(&(&b.0.model_id, b.0.decision)));
|
||||||
|
for (key, value) in entries {
|
||||||
|
out.push_str(&format!(
|
||||||
|
"sgl_router_cache_aware_decisions_total{{model_id=\"{}\",decision=\"{}\"}} {}\n",
|
||||||
|
escape_label(&key.model_id),
|
||||||
|
key.decision,
|
||||||
|
value,
|
||||||
|
));
|
||||||
|
}
|
||||||
|
drop(guard);
|
||||||
|
|
||||||
|
// diverted_overlap_blocks histogram
|
||||||
|
out.push_str(
|
||||||
|
"# HELP sgl_router_diverted_overlap_blocks Matched-prefix depth (blocks) given up by queue-gate diversions (decision=cache_worker_queued). Read against the overlap of all selections: a curve skewing high means the gate is trading large cached prefixes for short waits.\n",
|
||||||
|
);
|
||||||
|
out.push_str("# TYPE sgl_router_diverted_overlap_blocks histogram\n");
|
||||||
|
let guard = self.diverted_overlap_blocks.lock();
|
||||||
|
let mut models: Vec<&String> = guard.keys().collect();
|
||||||
|
models.sort();
|
||||||
|
for model_id in models {
|
||||||
|
let hist = guard.get(model_id).unwrap();
|
||||||
|
let label_body = format!("model_id=\"{}\"", escape_label(model_id));
|
||||||
|
render_histogram(
|
||||||
|
&mut out,
|
||||||
|
"sgl_router_diverted_overlap_blocks",
|
||||||
|
&label_body,
|
||||||
|
hist,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
drop(guard);
|
||||||
|
|
||||||
// ingress_tokenize_errors_total
|
// ingress_tokenize_errors_total
|
||||||
out.push_str(
|
out.push_str(
|
||||||
"# HELP sgl_router_ingress_tokenize_errors_total Chat requests on a chat-encoder model whose ingress tokenization failed, silently falling back to engine-side tokenization (the input_ids offload was defeated).\n",
|
"# HELP sgl_router_ingress_tokenize_errors_total Chat requests on a chat-encoder model whose ingress tokenization failed, silently falling back to engine-side tokenization (the input_ids offload was defeated).\n",
|
||||||
@@ -1511,6 +1655,41 @@ mod tests {
|
|||||||
assert!(out.contains("sgl_router_cache_pressure_guard_override_total 1"));
|
assert!(out.contains("sgl_router_cache_pressure_guard_override_total 1"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cache_aware_decisions_and_diverted_overlap_render() {
|
||||||
|
let reg = MetricsRegistry::new();
|
||||||
|
reg.record_cache_aware_decision("tiny", CacheAwareDecision::CacheHit);
|
||||||
|
reg.record_cache_aware_decision("tiny", CacheAwareDecision::CacheWorkerQueued);
|
||||||
|
reg.record_cache_aware_decision("tiny", CacheAwareDecision::CacheWorkerQueued);
|
||||||
|
reg.record_cache_aware_decision("tiny", CacheAwareDecision::AllQueued);
|
||||||
|
reg.observe_diverted_overlap_blocks("tiny", 40);
|
||||||
|
|
||||||
|
let out = reg.render();
|
||||||
|
assert!(out.contains(
|
||||||
|
r#"sgl_router_cache_aware_decisions_total{model_id="tiny",decision="cache_hit"} 1"#
|
||||||
|
));
|
||||||
|
assert!(out.contains(
|
||||||
|
r#"sgl_router_cache_aware_decisions_total{model_id="tiny",decision="cache_worker_queued"} 2"#
|
||||||
|
));
|
||||||
|
assert!(out.contains(
|
||||||
|
r#"sgl_router_cache_aware_decisions_total{model_id="tiny",decision="all_queued"} 1"#
|
||||||
|
));
|
||||||
|
// The saturation label must not be absorbed by a `cache_hit.*`
|
||||||
|
// hit-rate query.
|
||||||
|
assert!(!out.contains(r#"decision="cache_hit_all_queued""#));
|
||||||
|
assert!(
|
||||||
|
out.contains(r#"sgl_router_diverted_overlap_blocks_count{model_id="tiny"} 1"#),
|
||||||
|
"expected one diverted observation; got:\n{out}"
|
||||||
|
);
|
||||||
|
// 40 blocks lands in the le=64 bucket, not le=32.
|
||||||
|
assert!(
|
||||||
|
out.contains(r#"sgl_router_diverted_overlap_blocks_bucket{model_id="tiny",le="64"} 1"#)
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
out.contains(r#"sgl_router_diverted_overlap_blocks_bucket{model_id="tiny",le="32"} 0"#)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn ingress_tokenize_error_counter_increments_per_model() {
|
fn ingress_tokenize_error_counter_increments_per_model() {
|
||||||
let reg = MetricsRegistry::new();
|
let reg = MetricsRegistry::new();
|
||||||
|
|||||||
@@ -289,6 +289,18 @@ pub async fn chat_completions(
|
|||||||
.as_ref()
|
.as_ref()
|
||||||
.map(|config| config.session_affinity_mode)
|
.map(|config| config.session_affinity_mode)
|
||||||
.unwrap_or(SessionAffinityMode::Bucket);
|
.unwrap_or(SessionAffinityMode::Bucket);
|
||||||
|
// The queue gate (`--worker-queue-limit`) applies to the cache-aware
|
||||||
|
// candidate resolution and, beneath it, to primary/backup admission and
|
||||||
|
// the min-load range fallback. It does NOT reach the
|
||||||
|
// `CapacityFallbackPowerOfTwo` last resort: by the time that fires no
|
||||||
|
// worker in the domain is capacity-admitted, so there is no unqueued
|
||||||
|
// destination left to prefer.
|
||||||
|
let worker_queue_limit = ctx
|
||||||
|
.config
|
||||||
|
.model
|
||||||
|
.affinity
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|config| config.worker_queue_limit);
|
||||||
// Each Bucket retry rebuilds the proposal and reruns Admission/Guard.
|
// Each Bucket retry rebuilds the proposal and reruns Admission/Guard.
|
||||||
let worker = select_prefill_worker(&PrefillSelectionInputs {
|
let worker = select_prefill_worker(&PrefillSelectionInputs {
|
||||||
policy: policy.as_ref(),
|
policy: policy.as_ref(),
|
||||||
@@ -307,6 +319,7 @@ pub async fn chat_completions(
|
|||||||
ttft_slo_ms,
|
ttft_slo_ms,
|
||||||
tps_slo,
|
tps_slo,
|
||||||
session_affinity_mode,
|
session_affinity_mode,
|
||||||
|
worker_queue_limit,
|
||||||
})
|
})
|
||||||
.map_err(|reason| policy_selection_failed(&ctx, &model_str, reason))?;
|
.map_err(|reason| policy_selection_failed(&ctx, &model_str, reason))?;
|
||||||
|
|
||||||
|
|||||||
@@ -154,6 +154,7 @@ fn cache_candidate_uses_bucket_metadata_without_extend_range_filtering() {
|
|||||||
worker: cached,
|
worker: cached,
|
||||||
matched_prefix_tokens: 128,
|
matched_prefix_tokens: 128,
|
||||||
uncached_tokens: 128,
|
uncached_tokens: 128,
|
||||||
|
matched_prefix_blocks: 8,
|
||||||
candidate_range_id: "global".into(),
|
candidate_range_id: "global".into(),
|
||||||
max_pending_prefill_tokens: None,
|
max_pending_prefill_tokens: None,
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -124,6 +124,7 @@ impl Policy for CacheCandidatesPolicy {
|
|||||||
worker: Arc::clone(&self.worker),
|
worker: Arc::clone(&self.worker),
|
||||||
matched_prefix_tokens: 1,
|
matched_prefix_tokens: 1,
|
||||||
uncached_tokens: 1,
|
uncached_tokens: 1,
|
||||||
|
matched_prefix_blocks: 1,
|
||||||
candidate_range_id: "global".into(),
|
candidate_range_id: "global".into(),
|
||||||
max_pending_prefill_tokens: None,
|
max_pending_prefill_tokens: None,
|
||||||
}],
|
}],
|
||||||
|
|||||||
Reference in New Issue
Block a user