mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-07 01:47:47 +08:00
feat(routing): move sticky-key retries into routing policy with lazy attempts
Replace the provider/endpoint max_retries fields as the source of same-key retries with a routing policy setting, sticky_key_attempts (default 2). Only the first-ranked candidate is retried on the same key; every failover candidate gets a single attempt so failover keeps advancing instead of retrying each fallback key. Materialize exactly one attempt per candidate and derive same-key retries in the attempt loop after a candidate-scoped failure, so the retry budget no longer inflates up-front materialization and needs no upper bound. The budget travels in the report context; retries reuse the plan with a fresh candidate id and incremented retry index. Pool groups only retry their first key within the retry-index stride. Expose the setting in the routing profile editor and the set_scheduling rule action, and drop the max_retries input from the provider form.
This commit is contained in:
@@ -252,6 +252,10 @@ where
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn next_same_key_retry(&self, attempt: &T) -> Result<Option<T>, Self::Error> {
|
||||
Ok(crate::orchestration::next_same_key_retry_attempt(attempt))
|
||||
}
|
||||
|
||||
async fn record_attempt_failed(&self, attempt: &T) -> Result<(), Self::Error> {
|
||||
record_provider_transfer_attempt_failed(
|
||||
self.state,
|
||||
@@ -788,18 +792,31 @@ where
|
||||
{
|
||||
let mut last_attempted = None;
|
||||
let mut fallback_response = None;
|
||||
// A same-key retry derived after a candidate-scoped failure runs before
|
||||
// the source is asked for the next candidate.
|
||||
let mut pending_same_key_retry: Option<Attempt> = None;
|
||||
|
||||
loop {
|
||||
let next_started_at = std::time::Instant::now();
|
||||
let next_attempt =
|
||||
next_execution_attempt_with_timeout(source, trace_id, plan_kind, planning_timeout)
|
||||
let attempt = match pending_same_key_retry.take() {
|
||||
Some(attempt) => attempt,
|
||||
None => {
|
||||
let next_started_at = std::time::Instant::now();
|
||||
let next_attempt = next_execution_attempt_with_timeout(
|
||||
source,
|
||||
trace_id,
|
||||
plan_kind,
|
||||
planning_timeout,
|
||||
)
|
||||
.await?;
|
||||
observe_gateway_stage_ms(
|
||||
"stream_candidate_next",
|
||||
next_started_at.elapsed().as_millis() as u64,
|
||||
);
|
||||
let Some(attempt) = next_attempt else {
|
||||
break;
|
||||
observe_gateway_stage_ms(
|
||||
"stream_candidate_next",
|
||||
next_started_at.elapsed().as_millis() as u64,
|
||||
);
|
||||
let Some(attempt) = next_attempt else {
|
||||
break;
|
||||
};
|
||||
attempt
|
||||
}
|
||||
};
|
||||
if port.should_skip_attempt(&attempt).await? {
|
||||
let provider_id = attempt.execution_plan().provider_id.clone();
|
||||
@@ -839,6 +856,9 @@ where
|
||||
if attempt_fallback_response.is_some() {
|
||||
fallback_response = attempt_fallback_response;
|
||||
}
|
||||
if scope == AiAttemptRetryScope::Candidate {
|
||||
pending_same_key_retry = port.next_same_key_retry(&attempt).await?;
|
||||
}
|
||||
apply_attempt_retry_scope(source, &attempt, scope).await?;
|
||||
}
|
||||
}
|
||||
@@ -952,6 +972,10 @@ where
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn next_same_key_retry(&self, attempt: &T) -> Result<Option<T>, Self::Error> {
|
||||
Ok(crate::orchestration::next_same_key_retry_attempt(attempt))
|
||||
}
|
||||
|
||||
async fn record_attempt_failed(&self, attempt: &T) -> Result<(), Self::Error> {
|
||||
record_provider_transfer_attempt_failed(
|
||||
self.state,
|
||||
|
||||
Reference in New Issue
Block a user