feat(routing): move sticky-key retries into routing policy with lazy attempts

Replace the provider/endpoint max_retries fields as the source of same-key
retries with a routing policy setting, sticky_key_attempts (default 2). Only
the first-ranked candidate is retried on the same key; every failover
candidate gets a single attempt so failover keeps advancing instead of
retrying each fallback key.

Materialize exactly one attempt per candidate and derive same-key retries in
the attempt loop after a candidate-scoped failure, so the retry budget no
longer inflates up-front materialization and needs no upper bound. The budget
travels in the report context; retries reuse the plan with a fresh candidate
id and incremented retry index. Pool groups only retry their first key within
the retry-index stride.

Expose the setting in the routing profile editor and the set_scheduling rule
action, and drop the max_retries input from the provider form.
This commit is contained in:
elky
2026-09-02 20:48:40 +08:00
parent 415b2da81b
commit 7323d41fbe
40 changed files with 851 additions and 570 deletions
@@ -471,6 +471,27 @@
</span>
</span>
</label>
<label
class="flex items-start gap-3 rounded-lg border border-border/60 px-3 py-2 text-sm"
data-testid="sticky-key-attempts"
>
<Input
:model-value="stickyKeyAttempts"
type="number"
min="0"
max="99"
class="w-20 shrink-0"
:disabled="saving"
aria-label="粘性 Key 尝试次数"
@update:model-value="updateStickyKeyAttempts"
/>
<span class="min-w-0">
<span class="block font-medium">粘性 Key 尝试次数</span>
<span class="mt-0.5 block text-xs text-muted-foreground">
首个候选(缓存亲和命中的 Key)的总尝试次数。2 表示失败后同 Key 重试 1 次再转移,避免偶发错误破坏缓存;0 或 1 表示不重试。转移后的候选始终只尝试 1 次。
</span>
</span>
</label>
</div>
<RoutingPriorityPolicyEditor
@@ -780,6 +801,7 @@ import { DropdownMenu, DropdownMenuTrigger, DropdownMenuContent, DropdownMenuIte
import { AlertDialog } from '@/components/common'
import {
DEFAULT_ROUTING_POLICY_MODEL,
DEFAULT_STICKY_KEY_ATTEMPTS,
allowedModelsMirrorPerModelPolicies,
clearAllowedModels,
copyPerModelRoutingConfig,
@@ -790,6 +812,7 @@ import {
isGeneratedModelSchedulingRule,
modelSchedulingRuleId,
normalizeRoutingGroupConfig,
normalizeStickyKeyAttempts,
removePerModelRoutingConfig,
routingModelScopeLabel,
savePerModelRoutingConfig,
@@ -895,6 +918,9 @@ const firstStepSchedulingMode = computed<RoutingSchedulingMode>(() => {
const keepPriorityOnConversion = computed<boolean>(() => (
draft.value?.config_json.default_policy.keep_priority_on_conversion ?? false
))
const stickyKeyAttempts = computed<number>(() => (
draft.value?.config_json.default_policy.sticky_key_attempts ?? DEFAULT_STICKY_KEY_ATTEMPTS
))
const allowedModelsLookLikeLegacyMirror = computed(() => {
return draft.value
? allowedModelsMirrorPerModelPolicies(draft.value.config_json)
@@ -1211,6 +1237,17 @@ function updateFirstStepSchedulingMode(mode: RoutingSchedulingMode): void {
})
}
function updateStickyKeyAttempts(value: string | number): void {
if (!draft.value) return
updateDraftConfig({
...draft.value.config_json,
default_policy: {
...draft.value.config_json.default_policy,
sticky_key_attempts: normalizeStickyKeyAttempts(value),
},
})
}
function updateKeepPriorityOnConversion(value: boolean): void {
if (!draft.value) return
updateDraftConfig({
@@ -215,6 +215,7 @@ function routingGroup(
priority_mode: 'provider',
scheduling_mode: 'cache_affinity',
keep_priority_on_conversion: false,
sticky_key_attempts: 2,
},
model_policies: [],
rules: [],