feat(routing): move sticky-key retries into routing policy with lazy attempts

Replace the provider/endpoint max_retries fields as the source of same-key
retries with a routing policy setting, sticky_key_attempts (default 2). Only
the first-ranked candidate is retried on the same key; every failover
candidate gets a single attempt so failover keeps advancing instead of
retrying each fallback key.

Materialize exactly one attempt per candidate and derive same-key retries in
the attempt loop after a candidate-scoped failure, so the retry budget no
longer inflates up-front materialization and needs no upper bound. The budget
travels in the report context; retries reuse the plan with a fresh candidate
id and incremented retry index. Pool groups only retry their first key within
the retry-index stride.

Expose the setting in the routing profile editor and the set_scheduling rule
action, and drop the max_retries input from the provider form.
This commit is contained in:
elky
2026-09-02 20:48:40 +08:00
parent 415b2da81b
commit 7323d41fbe
40 changed files with 851 additions and 570 deletions
@@ -252,6 +252,10 @@ where
Ok(())
}
async fn next_same_key_retry(&self, attempt: &T) -> Result<Option<T>, Self::Error> {
Ok(crate::orchestration::next_same_key_retry_attempt(attempt))
}
async fn record_attempt_failed(&self, attempt: &T) -> Result<(), Self::Error> {
record_provider_transfer_attempt_failed(
self.state,
@@ -788,18 +792,31 @@ where
{
let mut last_attempted = None;
let mut fallback_response = None;
// A same-key retry derived after a candidate-scoped failure runs before
// the source is asked for the next candidate.
let mut pending_same_key_retry: Option<Attempt> = None;
loop {
let next_started_at = std::time::Instant::now();
let next_attempt =
next_execution_attempt_with_timeout(source, trace_id, plan_kind, planning_timeout)
let attempt = match pending_same_key_retry.take() {
Some(attempt) => attempt,
None => {
let next_started_at = std::time::Instant::now();
let next_attempt = next_execution_attempt_with_timeout(
source,
trace_id,
plan_kind,
planning_timeout,
)
.await?;
observe_gateway_stage_ms(
"stream_candidate_next",
next_started_at.elapsed().as_millis() as u64,
);
let Some(attempt) = next_attempt else {
break;
observe_gateway_stage_ms(
"stream_candidate_next",
next_started_at.elapsed().as_millis() as u64,
);
let Some(attempt) = next_attempt else {
break;
};
attempt
}
};
if port.should_skip_attempt(&attempt).await? {
let provider_id = attempt.execution_plan().provider_id.clone();
@@ -839,6 +856,9 @@ where
if attempt_fallback_response.is_some() {
fallback_response = attempt_fallback_response;
}
if scope == AiAttemptRetryScope::Candidate {
pending_same_key_retry = port.next_same_key_retry(&attempt).await?;
}
apply_attempt_retry_scope(source, &attempt, scope).await?;
}
}
@@ -952,6 +972,10 @@ where
Ok(())
}
async fn next_same_key_retry(&self, attempt: &T) -> Result<Option<T>, Self::Error> {
Ok(crate::orchestration::next_same_key_retry_attempt(attempt))
}
async fn record_attempt_failed(&self, attempt: &T) -> Result<(), Self::Error> {
record_provider_transfer_attempt_failed(
self.state,
@@ -1663,6 +1663,22 @@ mod tests {
candidate_index: u32,
endpoint_id: &str,
candidate_id: &str,
) -> AiSyncAttempt {
test_openai_image_heartbeat_attempt_with_sticky_key_attempts(
candidate_index,
endpoint_id,
candidate_id,
1,
)
}
/// `sticky_key_attempts` is pinned so these tests exercise candidate
/// failover; the default same-key retry is covered separately.
fn test_openai_image_heartbeat_attempt_with_sticky_key_attempts(
candidate_index: u32,
endpoint_id: &str,
candidate_id: &str,
sticky_key_attempts: u32,
) -> AiSyncAttempt {
AiSyncAttempt {
plan: test_openai_image_heartbeat_plan(endpoint_id, candidate_id),
@@ -1670,6 +1686,7 @@ mod tests {
report_context: Some(json!({
"candidate_index": candidate_index,
"retry_index": 0,
"sticky_key_attempts": sticky_key_attempts,
})),
}
}
@@ -1828,6 +1845,9 @@ mod tests {
report_context: Some(json!({
"candidate_index": candidate_index,
"retry_index": 0,
// Pin to a single attempt so this helper exercises candidate
// failover rather than the default same-key retry.
"sticky_key_attempts": 1,
"client_api_format": client_api_format,
"provider_api_format": client_api_format,
})),
@@ -1980,6 +2000,90 @@ mod tests {
assert_eq!(body, json!({"data": [{"b64_json": "second-candidate"}]}));
}
#[tokio::test]
async fn openai_image_sync_heartbeat_retries_sticky_key_lazily_before_failover() {
let seen_plans = Arc::new(std::sync::Mutex::new(Vec::<(String, Option<String>)>::new()));
let seen_plans_for_override = Arc::clone(&seen_plans);
let state = AppState::new()
.expect("state should build")
.with_execution_runtime_sync_override_for_tests(move |plan| {
seen_plans_for_override
.lock()
.expect("mutex should lock")
.push((plan.endpoint_id.clone(), plan.candidate_id.clone()));
if plan.endpoint_id == "endpoint-retry" {
Ok(test_openai_image_execution_result(
plan,
StatusCode::TOO_MANY_REQUESTS.as_u16(),
json!({"error": {"message": "retry this candidate"}}),
))
} else {
Ok(test_openai_image_execution_result(
plan,
StatusCode::OK.as_u16(),
json!({"data": [{"b64_json": "second-candidate"}]}),
))
}
});
// Three total attempts on the sticky key; only one attempt is
// materialized up front, the other two are derived after each failure.
let attempts = vec![
test_openai_image_heartbeat_attempt_with_sticky_key_attempts(
0,
"endpoint-retry",
"candidate-retry",
3,
),
test_openai_image_heartbeat_attempt_with_sticky_key_attempts(
1,
"endpoint-success",
"candidate-success",
3,
),
];
let outcome = execute_openai_image_sync_heartbeat_attempts(
state,
"/v1/images/generations".to_string(),
"trace-image-heartbeat-sticky-retry".to_string(),
test_openai_image_heartbeat_decision(),
TEST_OPENAI_IMAGE_SYNC_PLAN_KIND.to_string(),
attempts,
ProviderTransferTracker::default(),
Instant::now(),
)
.await
.expect("heartbeat attempts should execute");
let LocalExecutionRequestOutcome::Responded(response) = outcome else {
panic!("second candidate should return a response");
};
let bytes = openai_image_sync_heartbeat_response_body_bytes(response).await;
let body: Value = serde_json::from_slice(&bytes).expect("body should decode");
let seen_plans = seen_plans.lock().expect("mutex should lock").clone();
assert_eq!(
seen_plans
.iter()
.map(|(endpoint_id, _)| endpoint_id.as_str())
.collect::<Vec<_>>(),
[
"endpoint-retry",
"endpoint-retry",
"endpoint-retry",
"endpoint-success"
]
);
let sticky_candidate_ids = seen_plans[..3]
.iter()
.map(|(_, candidate_id)| candidate_id.clone())
.collect::<std::collections::BTreeSet<_>>();
assert_eq!(
sticky_candidate_ids.len(),
3,
"each derived same-key retry must carry a fresh candidate id"
);
assert_eq!(body, json!({"data": [{"b64_json": "second-candidate"}]}));
}
#[tokio::test]
async fn openai_image_sync_heartbeat_honors_provider_transfer_limit() {
let call_count = Arc::new(AtomicUsize::new(0));
@@ -2013,6 +2117,7 @@ mod tests {
attempt.report_context = Some(json!({
"candidate_index": index,
"retry_index": 0,
"sticky_key_attempts": 1,
"local_failover_policy": {
"max_transfer_count": 1,
"max_transfer_timeout_seconds": 0