feat(routing): move sticky-key retries into routing policy with lazy attempts

Replace the provider/endpoint max_retries fields as the source of same-key
retries with a routing policy setting, sticky_key_attempts (default 2). Only
the first-ranked candidate is retried on the same key; every failover
candidate gets a single attempt so failover keeps advancing instead of
retrying each fallback key.

Materialize exactly one attempt per candidate and derive same-key retries in
the attempt loop after a candidate-scoped failure, so the retry budget no
longer inflates up-front materialization and needs no upper bound. The budget
travels in the report context; retries reuse the plan with a fresh candidate
id and incremented retry index. Pool groups only retry their first key within
the retry-index stride.

Expose the setting in the routing profile editor and the set_scheduling rule
action, and drop the max_retries input from the provider form.
This commit is contained in:
elky
2026-09-02 20:48:40 +08:00
parent 415b2da81b
commit 7323d41fbe
40 changed files with 851 additions and 570 deletions
@@ -2220,7 +2220,9 @@ async fn gateway_retries_next_local_openai_chat_stream_candidate_after_retryable
.to_string(),
});
let frames = if attempt == 1 {
// The primary key gets two attempts under the default
// sticky_key_attempts; both must fail to reach the backup.
let frames = if attempt <= 2 {
concat!(
"{\"type\":\"headers\",\"payload\":{\"kind\":\"headers\",\"status_code\":429,\"headers\":{\"content-type\":\"application/json\"}}}\n",
"{\"type\":\"data\",\"payload\":{\"kind\":\"data\",\"text\":\"{\\\"error\\\":{\\\"message\\\":\\\"rate limited\\\",\\\"type\\\":\\\"rate_limit_error\\\"}}\"}}\n",
@@ -2362,14 +2364,25 @@ async fn gateway_retries_next_local_openai_chat_stream_candidate_after_retryable
.lock()
.expect("mutex should lock")
.len()
>= 2
>= 3
})
.await;
let seen_execution_runtime_requests = seen_execution_runtime
.lock()
.expect("mutex should lock")
.clone();
assert_eq!(seen_execution_runtime_requests.len(), 2);
// Default sticky_key_attempts is 2: the primary key is retried once on
// the same key, then failover moves to the backup with a single attempt.
assert_eq!(seen_execution_runtime_requests.len(), 3);
assert_eq!(
seen_execution_runtime_requests
.iter()
.filter(|request| {
request.url == "https://api.openai.primary.example/chat/completions"
})
.count(),
2
);
let primary_request = seen_execution_runtime_requests
.iter()
.find(|request| request.url == "https://api.openai.primary.example/chat/completions")
@@ -2404,7 +2417,15 @@ async fn gateway_retries_next_local_openai_chat_stream_candidate_after_retryable
.list_by_request_id("trace-openai-chat-local-stream-failover-123")
.await
.expect("request candidate trace should read");
assert_eq!(stored_candidates.len(), 2);
assert_eq!(stored_candidates.len(), 3);
assert_eq!(
stored_candidates
.iter()
.filter(|candidate| candidate.status == RequestCandidateStatus::Failed)
.count(),
2,
"both sticky-key attempts on the primary should be recorded as failed"
);
let failed_candidate = stored_candidates
.iter()
.find(|candidate| {
@@ -2465,7 +2486,7 @@ async fn gateway_retries_next_local_openai_chat_stream_candidate_after_retryable
assert_eq!(
execution_runtime_hits.load(std::sync::atomic::Ordering::SeqCst),
2
3
);
assert_eq!(*decision_hits.lock().expect("mutex should lock"), 0);
assert_eq!(*plan_hits.lock().expect("mutex should lock"), 0);
@@ -112,7 +112,7 @@ async fn gateway_skips_unsupported_local_openai_chat_sync_candidate_before_tryin
false,
false,
None,
Some(2),
Some(1),
None,
Some(20.0),
None,
@@ -134,7 +134,7 @@ async fn gateway_skips_unsupported_local_openai_chat_sync_candidate_before_tryin
"https://api.openai.skip.example".to_string(),
None,
None,
Some(2),
Some(1),
None,
None,
None,
@@ -520,7 +520,7 @@ async fn gateway_surfaces_local_execution_runtime_miss_reason_when_all_openai_ch
false,
false,
None,
Some(2),
Some(1),
None,
Some(20.0),
None,
@@ -542,7 +542,7 @@ async fn gateway_surfaces_local_execution_runtime_miss_reason_when_all_openai_ch
"https://chatgpt.com/backend-api/codex".to_string(),
None,
None,
Some(2),
Some(1),
None,
None,
None,
@@ -802,7 +802,7 @@ async fn gateway_retries_next_local_openai_chat_sync_candidate_after_auth_failur
false,
false,
None,
Some(2),
Some(1),
None,
Some(20.0),
None,
@@ -828,7 +828,7 @@ async fn gateway_retries_next_local_openai_chat_sync_candidate_after_auth_failur
base_url.to_string(),
None,
None,
Some(2),
Some(1),
None,
None,
None,
@@ -970,7 +970,9 @@ async fn gateway_retries_next_local_openai_chat_sync_candidate_after_auth_failur
.to_string(),
});
if attempt == 1 {
// The primary key gets two attempts under the default
// sticky_key_attempts; both must fail to reach the backup.
if attempt <= 2 {
return Json(json!({
"request_id": "trace-openai-chat-local-failover-123",
"status_code": 401,
@@ -1125,56 +1127,63 @@ async fn gateway_retries_next_local_openai_chat_sync_candidate_after_auth_failur
.lock()
.expect("mutex should lock")
.clone();
assert_eq!(seen_execution_runtime_requests.len(), 2);
// Default sticky_key_attempts is 2: the primary key is retried once on
// the same key, then failover moves to the backup with a single attempt.
assert_eq!(seen_execution_runtime_requests.len(), 3);
for primary_request in &seen_execution_runtime_requests[..2] {
assert_eq!(
primary_request.trace_id,
"trace-openai-chat-local-failover-123"
);
assert_eq!(
primary_request.url,
"https://api.openai.primary.example/chat/completions"
);
assert_eq!(
primary_request.authorization,
"Bearer sk-upstream-openai-primary"
);
}
assert_eq!(
seen_execution_runtime_requests[0].trace_id,
"trace-openai-chat-local-failover-123"
);
assert_eq!(
seen_execution_runtime_requests[0].url,
"https://api.openai.primary.example/chat/completions"
);
assert_eq!(
seen_execution_runtime_requests[0].authorization,
"Bearer sk-upstream-openai-primary"
);
assert_eq!(
seen_execution_runtime_requests[1].url,
seen_execution_runtime_requests[2].url,
"https://api.openai.backup.example/chat/completions"
);
assert_eq!(
seen_execution_runtime_requests[1].model,
seen_execution_runtime_requests[2].model,
"gpt-5-upstream-backup"
);
assert_eq!(
seen_execution_runtime_requests[1].authorization,
seen_execution_runtime_requests[2].authorization,
"Bearer sk-upstream-openai-backup"
);
let stored_candidates = request_candidate_repository
.list_by_request_id("trace-openai-chat-local-failover-123")
.await
.expect("request candidate trace should read");
assert_eq!(stored_candidates.len(), 2);
assert_eq!(stored_candidates[0].candidate_index, 0);
assert_eq!(stored_candidates[0].status, RequestCandidateStatus::Failed);
assert_eq!(stored_candidates[0].status_code, Some(401));
assert_eq!(
stored_candidates[0].error_message.as_deref(),
Some("invalid auth token")
);
let failed_upstream_response = stored_candidates[0]
.extra_data
.as_ref()
.and_then(|value| value.get("upstream_response"))
.expect("failed candidate should keep its upstream response");
assert_eq!(failed_upstream_response["status_code"], json!(401));
assert_eq!(
failed_upstream_response["body"]["error"]["message"],
json!("invalid auth token")
);
assert_eq!(stored_candidates[1].candidate_index, 1);
assert_eq!(stored_candidates[1].status, RequestCandidateStatus::Success);
assert_eq!(stored_candidates[1].status_code, Some(200));
assert_eq!(stored_candidates.len(), 3);
for (retry_index, failed_candidate) in stored_candidates[..2].iter().enumerate() {
assert_eq!(failed_candidate.candidate_index, 0);
assert_eq!(failed_candidate.retry_index, retry_index as u32);
assert_eq!(failed_candidate.status, RequestCandidateStatus::Failed);
assert_eq!(failed_candidate.status_code, Some(401));
assert_eq!(
failed_candidate.error_message.as_deref(),
Some("invalid auth token")
);
let failed_upstream_response = failed_candidate
.extra_data
.as_ref()
.and_then(|value| value.get("upstream_response"))
.expect("failed candidate should keep its upstream response");
assert_eq!(failed_upstream_response["status_code"], json!(401));
assert_eq!(
failed_upstream_response["body"]["error"]["message"],
json!("invalid auth token")
);
}
assert_eq!(stored_candidates[2].candidate_index, 1);
assert_eq!(stored_candidates[2].status, RequestCandidateStatus::Success);
assert_eq!(stored_candidates[2].status_code, Some(200));
tokio::time::sleep(std::time::Duration::from_millis(100)).await;
assert!(
@@ -1184,7 +1193,7 @@ async fn gateway_retries_next_local_openai_chat_sync_candidate_after_auth_failur
assert_eq!(
*execution_runtime_hits.lock().expect("mutex should lock"),
2
3
);
assert_eq!(*decision_hits.lock().expect("mutex should lock"), 0);
assert_eq!(*plan_hits.lock().expect("mutex should lock"), 0);
@@ -140,7 +140,7 @@ async fn gateway_executes_codex_search_with_responses_permission_and_search_cont
false,
false,
None,
Some(2),
Some(1),
None,
Some(900.0),
None,
@@ -162,7 +162,7 @@ async fn gateway_executes_codex_search_with_responses_permission_and_search_cont
"https://chatgpt.com/backend-api/codex".to_string(),
None,
None,
Some(2),
Some(1),
None,
None,
None,
@@ -570,9 +570,12 @@ async fn gateway_executes_codex_search_with_responses_permission_and_search_cont
.filter(|plan| plan["request_id"] == "trace-search-failover-1")
.map(|plan| plan["provider_id"].clone())
.collect::<Vec<_>>();
// Default sticky_key_attempts is 2: the first provider is retried once on
// the same key before failover advances to the second provider.
assert_eq!(
failover_plans,
vec![
json!("provider-codex-search-1"),
json!("provider-codex-search-1"),
json!("provider-codex-search-2")
]
@@ -581,17 +584,16 @@ async fn gateway_executes_codex_search_with_responses_permission_and_search_cont
.list_by_request_id("trace-search-failover-1")
.await
.expect("failover request candidates should read");
assert_eq!(failover_candidates.len(), 2);
assert_eq!(failover_candidates.len(), 3);
for failed_candidate in &failover_candidates[..2] {
assert_eq!(failed_candidate.status, RequestCandidateStatus::Failed);
assert_eq!(failed_candidate.status_code, Some(500));
}
assert_eq!(
failover_candidates[0].status,
RequestCandidateStatus::Failed
);
assert_eq!(failover_candidates[0].status_code, Some(500));
assert_eq!(
failover_candidates[1].status,
failover_candidates[2].status,
RequestCandidateStatus::Success
);
assert_eq!(failover_candidates[1].status_code, Some(200));
assert_eq!(failover_candidates[2].status_code, Some(200));
gateway_handle.abort();
execution_runtime_handle.abort();
@@ -1714,7 +1714,6 @@ fn ai_serving_candidate_materialization_owns_affinity_and_candidate_runtime_pers
"pub fn ai_should_persist_available_candidate_for_pool_key",
"pub fn ai_should_persist_skipped_candidate_for_pool_membership",
"pub fn ai_candidate_extra_data_with_ranking",
"attempt_slot_count",
"should_persist_available_candidate",
"persist_available_candidate",
"build_attempt",
+2 -2
View File
@@ -122,7 +122,7 @@ pub(super) fn sample_local_openai_provider() -> StoredProviderCatalogProvider {
false,
false,
None,
Some(2),
Some(1),
None,
Some(20.0),
None,
@@ -144,7 +144,7 @@ pub(super) fn sample_local_openai_endpoint() -> StoredProviderCatalogEndpoint {
"https://api.openai.example/v1".to_string(),
None,
None,
Some(2),
Some(1),
None,
None,
None,
+22 -17
View File
@@ -879,9 +879,13 @@ async fn gateway_records_failed_usage_when_all_local_openai_chat_candidates_exha
.list_by_request_id("trace-openai-chat-local-report-sync-failure-123")
.await
.expect("request candidate trace should read");
assert_eq!(stored_candidates.len(), 1);
assert_eq!(stored_candidates[0].status, RequestCandidateStatus::Failed);
assert_eq!(stored_candidates[0].status_code, Some(503));
// The only candidate is the sticky first key: the default policy retries
// it once on the same key before the request is exhausted.
assert_eq!(stored_candidates.len(), 2);
for candidate in &stored_candidates {
assert_eq!(candidate.status, RequestCandidateStatus::Failed);
assert_eq!(candidate.status_code, Some(503));
}
}
#[test]
@@ -957,7 +961,8 @@ async fn gateway_records_failed_usage_when_sync_runtime_transport_is_unavailable
let response = send_request(gateway, request).await;
assert_eq!(response.status(), StatusCode::SERVICE_UNAVAILABLE);
assert_eq!(*execution_hits.lock().expect("mutex should lock"), 1);
// The sticky first key is retried once on the same key before exhaustion.
assert_eq!(*execution_hits.lock().expect("mutex should lock"), 2);
let stored_usage = wait_for_usage_status(
usage_repository.as_ref(),
@@ -982,15 +987,15 @@ async fn gateway_records_failed_usage_when_sync_runtime_transport_is_unavailable
.list_by_request_id("trace-openai-chat-local-transport-unavailable-123")
.await
.expect("request candidate trace should read");
assert_eq!(stored_candidates.len(), 1);
assert_eq!(stored_candidates[0].status, RequestCandidateStatus::Failed);
assert!(stored_candidates[0]
.latency_ms
.is_some_and(|value| value >= 5));
assert_eq!(
stored_candidates[0].error_type.as_deref(),
Some("execution_runtime_unavailable")
);
assert_eq!(stored_candidates.len(), 2);
for candidate in &stored_candidates {
assert_eq!(candidate.status, RequestCandidateStatus::Failed);
assert!(candidate.latency_ms.is_some_and(|value| value >= 5));
assert_eq!(
candidate.error_type.as_deref(),
Some("execution_runtime_unavailable")
);
}
}
#[test]
@@ -1786,7 +1791,7 @@ async fn gateway_records_failed_usage_when_all_local_claude_cli_candidates_are_s
false,
false,
None,
Some(2),
Some(1),
None,
Some(20.0),
None,
@@ -1808,7 +1813,7 @@ async fn gateway_records_failed_usage_when_all_local_claude_cli_candidates_are_s
"https://right.codes/codex".to_string(),
None,
None,
Some(2),
Some(1),
Some("/v1/messages".to_string()),
None,
None,
@@ -2111,7 +2116,7 @@ fn gateway_keeps_failed_usage_request_capture_lightweight_for_large_local_claude
false,
false,
None,
Some(2),
Some(1),
None,
Some(20.0),
None,
@@ -2133,7 +2138,7 @@ fn gateway_keeps_failed_usage_request_capture_lightweight_for_large_local_claude
"https://right.codes/codex".to_string(),
None,
None,
Some(2),
Some(1),
Some("/v1/messages".to_string()),
None,
None,