feat(gateway): harden failover and payload handling

Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
This commit is contained in:
elky
2026-07-30 01:03:27 +08:00
parent a97acc07fc
commit a04673a90d
80 changed files with 3640 additions and 1236 deletions
@@ -8,6 +8,33 @@
@update:model-value="handleClose"
>
<div class="space-y-5 max-h-[60vh] overflow-y-auto px-0.5 py-0.5 -mx-0.5">
<!-- 传输错误规则 -->
<div class="space-y-3">
<div>
<h3 class="text-sm font-medium">
传输错误
</h3>
<p class="text-xs text-muted-foreground mt-0.5">
这类错误没有可用于故障转移判断的上游 HTTP 状态码,因此不会命中下方的状态码规则
</p>
</div>
<div class="flex items-center justify-between gap-4 rounded-lg border bg-muted/50 p-3">
<div class="min-w-0 space-y-0.5">
<span class="text-sm font-medium">继续尝试下一候选</span>
<p class="text-xs leading-relaxed text-muted-foreground">
适用于 DNS 解析、TCP 连接、TLS 握手、代理/WARP 连接,以及响应提交前的连接重置或超时。默认开启;关闭后会立即返回网关错误。响应开始发送后无法切换候选。
</p>
</div>
<Switch
aria-label="传输错误时继续尝试下一候选"
class="shrink-0"
:model-value="continueOnTransportErrors"
@update:model-value="(value: boolean) => continueOnTransportErrors = value"
/>
</div>
</div>
<!-- 成功转移规则 -->
<div class="space-y-3">
<div class="flex items-start justify-between gap-3">
@@ -243,6 +270,7 @@ import {
Dialog,
Button,
Input,
Switch,
Textarea,
} from '@/components/ui'
import { AlignLeft, Code2, GitBranch, Plus, Trash2 } from 'lucide-vue-next'
@@ -263,6 +291,7 @@ const emit = defineEmits<{
const { success, error: showError } = useToast()
const saving = ref(false)
const continueOnTransportErrors = ref(true)
const successPatterns = ref<FailoverRuleItem[]>([])
const errorPatterns = ref<FailoverRuleItem[]>([])
@@ -284,6 +313,7 @@ const TOP_LEVEL_STOP_STATUS_CODE_KEYS = [
] as const
const MANAGED_FAILOVER_RULE_KEYS = [
'stop_on_transport_errors',
'success_failover_patterns',
'error_stop_patterns',
...TOP_LEVEL_STOP_STATUS_CODE_KEYS,
@@ -334,6 +364,9 @@ function buildNextFailoverRules(
if (filteredError.length > 0) {
nextRules.error_stop_patterns = filteredError
}
if (!continueOnTransportErrors.value) {
nextRules.stop_on_transport_errors = true
}
return Object.values(nextRules).some(hasPersistableFailoverValue)
? nextRules as FailoverRulesConfig
@@ -343,6 +376,7 @@ function buildNextFailoverRules(
watch(() => [props.open, props.provider], () => {
if (props.open && props.provider) {
const rules = props.provider.failover_rules
continueOnTransportErrors.value = rules?.stop_on_transport_errors !== true
successPatterns.value = (rules?.success_failover_patterns || []).map(r => ({
...r,
pattern: r.pattern || '',
@@ -1188,6 +1188,7 @@ const hasFailoverRules = computed(() => {
if (!rules) return false
return FAILOVER_RULE_ARRAY_KEYS.some(key => (rules[key]?.length || 0) > 0)
|| typeof rules.max_retries === 'number'
|| rules.stop_on_transport_errors === true
})
// Provider 级别代理配置状态
@@ -216,10 +216,11 @@
</Label>
<Input
id="max-transfer-count"
:model-value="form.max_transfer_count"
:model-value="form.max_transfer_count === 0 ? '' : form.max_transfer_count"
type="number"
min="0"
step="1"
:placeholder="legacyT('0 (不限制)')"
@update:model-value="(v) => form.max_transfer_count = parseNumberInput(v, { min: 0 }) ?? 0"
/>
</div>
@@ -233,10 +234,11 @@
</Label>
<Input
id="max-transfer-timeout-seconds"
:model-value="form.max_transfer_timeout_seconds"
:model-value="form.max_transfer_timeout_seconds === 0 ? '' : form.max_transfer_timeout_seconds"
type="number"
min="0"
step="1"
:placeholder="legacyT('0 (不限制)')"
@update:model-value="(v) => form.max_transfer_timeout_seconds = parseNumberInput(v, { min: 0 }) ?? 0"
/>
</div>
@@ -0,0 +1,130 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { createApp, nextTick, type App } from 'vue'
import type { FailoverRulesConfig, ProviderWithEndpointsSummary } from '@/api/endpoints/types'
import FailoverRulesDialog from '../FailoverRulesDialog.vue'
const endpointMocks = vi.hoisted(() => ({
updateProvider: vi.fn(),
}))
vi.mock('@/api/endpoints', () => ({
updateProvider: endpointMocks.updateProvider,
}))
vi.mock('@/composables/useToast', () => ({
useToast: () => ({
success: vi.fn(),
error: vi.fn(),
}),
}))
const mountedApps: Array<{ app: App, root: HTMLElement }> = []
function makeProvider(failoverRules: FailoverRulesConfig | null): ProviderWithEndpointsSummary {
return {
id: 'provider-1',
failover_rules: failoverRules,
} as ProviderWithEndpointsSummary
}
function mountDialog(failoverRules: FailoverRulesConfig | null = null) {
const root = document.createElement('div')
document.body.appendChild(root)
const app = createApp(FailoverRulesDialog, {
open: true,
provider: makeProvider(failoverRules),
'onUpdate:open': vi.fn(),
})
app.mount(root)
mountedApps.push({ app, root })
}
async function settle() {
for (let index = 0; index < 4; index += 1) {
await Promise.resolve()
await nextTick()
}
}
function transportErrorSwitch(): HTMLButtonElement {
const control = document.body.querySelector<HTMLButtonElement>(
'[role="switch"][aria-label="传输错误时继续尝试下一候选"]',
)
if (!control) throw new Error('Missing transport-error failover switch')
return control
}
function clickSave() {
const button = [...document.body.querySelectorAll<HTMLButtonElement>('button')]
.find(candidate => candidate.textContent?.trim() === '保存')
if (!button) throw new Error('Missing save button')
button.click()
}
beforeEach(() => {
endpointMocks.updateProvider.mockReset()
endpointMocks.updateProvider.mockResolvedValue(makeProvider(null))
})
afterEach(() => {
for (const { app, root } of mountedApps.splice(0)) {
app.unmount()
root.remove()
}
document.body.innerHTML = ''
})
describe('FailoverRulesDialog transport errors', () => {
it('continues failover by default and explains why HTTP status rules do not apply', async () => {
mountDialog()
await settle()
expect(transportErrorSwitch().getAttribute('aria-checked')).toBe('true')
expect(document.body.textContent).toContain('没有可用于故障转移判断的上游 HTTP 状态码')
expect(document.body.textContent).toContain('DNS 解析')
expect(document.body.textContent).toContain('TCP 连接')
expect(document.body.textContent).toContain('TLS 握手')
expect(document.body.textContent).toContain('响应提交前的连接重置或超时')
expect(document.body.textContent).toContain('响应开始发送后无法切换候选')
})
it('persists stop_on_transport_errors when continuing is disabled', async () => {
mountDialog()
await settle()
transportErrorSwitch().click()
await nextTick()
expect(transportErrorSwitch().getAttribute('aria-checked')).toBe('false')
clickSave()
await settle()
expect(endpointMocks.updateProvider).toHaveBeenCalledWith('provider-1', {
failover_rules: {
stop_on_transport_errors: true,
},
})
})
it('loads a stop rule and removes only that rule when continuing is enabled', async () => {
mountDialog({
max_retries: 2,
stop_on_transport_errors: true,
})
await settle()
expect(transportErrorSwitch().getAttribute('aria-checked')).toBe('false')
transportErrorSwitch().click()
await nextTick()
clickSave()
await settle()
expect(endpointMocks.updateProvider).toHaveBeenCalledWith('provider-1', {
failover_rules: {
max_retries: 2,
},
})
})
})
@@ -54,4 +54,8 @@ describe('ProviderDetailDrawer loading priorities', () => {
expect(source).toContain('v-if="open && batchAssignDialogOpen && provider"')
expect(source).toContain('v-if="open && failoverRulesDialogOpen"')
})
it('marks a transport-error stop policy as a configured failover rule', () => {
expect(source).toContain('rules.stop_on_transport_errors === true')
})
})
@@ -177,15 +177,20 @@ describe('ProviderFormDialog transfer limits', () => {
)
})
it('defaults missing legacy values to explicit zero', async () => {
it('shows zero limits as unlimited placeholders while submitting explicit zero', async () => {
mountDialog(makeProvider({
max_transfer_count: undefined,
max_transfer_timeout_seconds: undefined,
}))
await settle()
expect(document.body.querySelector<HTMLInputElement>('#max-transfer-count')?.value).toBe('0')
expect(document.body.querySelector<HTMLInputElement>('#max-transfer-timeout-seconds')?.value).toBe('0')
const countInput = document.body.querySelector<HTMLInputElement>('#max-transfer-count')
const timeoutInput = document.body.querySelector<HTMLInputElement>('#max-transfer-timeout-seconds')
expect(countInput?.value).toBe('')
expect(countInput?.placeholder).toBe('0 (不限制)')
expect(timeoutInput?.value).toBe('')
expect(timeoutInput?.placeholder).toBe('0 (不限制)')
clickButton('保存')
await settle()
@@ -1023,25 +1023,26 @@ const conversionBoundaryIndex = computed(() => {
return idx
})
// 计算链路总耗时(使用成功候选的 latency_ms 字段)
// 优先使用 latency_ms,因为它与 Usage.response_time_ms 使用相同的时间基准
// 避免 finished_at - started_at 带来的额外延迟(数据库操作时间)
// The trace aggregate includes every attempted candidate, including failed
// failover attempts. Per-candidate latency remains provider-scoped.
const totalTraceLatency = computed(() => {
if (!rawTimeline.value || rawTimeline.value.length === 0) return 0
// 查找成功的候选,使用其 latency_ms
const successCandidate = rawTimeline.value.find(c => c.status === 'success')
if (successCandidate?.latency_ms != null) {
return successCandidate.latency_ms
const aggregateLatency = trace.value?.total_latency_ms
if (typeof aggregateLatency === 'number' && Number.isFinite(aggregateLatency) && aggregateLatency > 0) {
return aggregateLatency
}
// 如果没有成功的候选,查找失败但有 latency_ms 的候选
const failedWithLatency = rawTimeline.value.find(c => c.status === 'failed' && c.latency_ms != null)
if (failedWithLatency?.latency_ms != null) {
return failedWithLatency.latency_ms
const attemptedLatency = rawTimeline.value.reduce((sum, candidate) => {
const latency = normalizeLatencyMs(candidate.latency_ms)
return sum + (latency ?? 0)
}, 0)
if (attemptedLatency > 0) {
return attemptedLatency
}
// 回退:使用 finished_at - started_at 计算
// Historical transport failures may not have latency_ms. Recover the wall
// clock span from candidate timestamps for those records.
let earliestStart: number | null = null
let latestEnd: number | null = null
@@ -175,7 +175,7 @@
<span>
<span class="text-muted-foreground">耗时</span>
<span class="ml-1 font-bold">
{{ formatDurationMs(detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.response_time_ms) }}
{{ formatDurationMs(detail.end_to_end_first_byte_time_ms ?? detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.end_to_end_time_ms ?? detail.response_time_ms) }}
</span>
</span>
<span class="text-muted-foreground">|</span>
@@ -197,7 +197,7 @@
<span class="whitespace-nowrap">
<span class="text-muted-foreground">耗时</span>
<span class="ml-1 font-bold">
{{ formatDurationMs(detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.response_time_ms) }}
{{ formatDurationMs(detail.end_to_end_first_byte_time_ms ?? detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.end_to_end_time_ms ?? detail.response_time_ms) }}
</span>
</span>
<span class="text-muted-foreground">|</span>
@@ -358,7 +358,7 @@
<span>{{ formatOutputRate(getRecordDisplayOutputRate(record)) }}</span>
</template>
<span
v-else-if="record.response_time_ms != null || record.first_byte_time_ms != null"
v-else-if="hasRecordDisplayLatency(record)"
class="ml-1"
>{{ formatRecordLatencyPair(record) }} / {{ formatOutputRate(getRecordDisplayOutputRate(record)) }}</span>
<span
@@ -611,7 +611,8 @@
v-if="isColumnVisible('performance')"
class="h-12 font-semibold w-[9%] text-right"
>
<div class="flex flex-col items-end text-xs gap-0.5">
<div class="flex flex-col items-end text-[11px] leading-3">
<span class="whitespace-nowrap">端到端</span>
<span class="whitespace-nowrap">首字/总耗时</span>
<span class="text-muted-foreground font-normal">输出速度</span>
</div>
@@ -925,7 +926,7 @@
</div>
<!-- 已完成状态:首字 + 总耗时 -->
<div
v-else-if="record.response_time_ms != null || record.first_byte_time_ms != null"
v-else-if="hasRecordDisplayLatency(record)"
class="flex flex-col items-end text-xs gap-0.5"
:title="getRecordPerformanceTitle(record)"
>
@@ -1445,8 +1446,10 @@ function getRecordCacheTokensTitle(record: UsageRecord): string {
}
function formatRecordLatencyPair(record: UsageRecord): string {
const firstByte = formatRecordDurationSeconds(record.first_byte_time_ms)
const total = formatRecordDurationSeconds(record.response_time_ms)
const firstByte = formatRecordDurationSeconds(
record.end_to_end_first_byte_time_ms ?? record.first_byte_time_ms,
)
const total = formatRecordDurationSeconds(record.end_to_end_time_ms ?? record.response_time_ms)
return `${firstByte} / ${total}`
}
@@ -1455,6 +1458,13 @@ function formatRecordDurationSeconds(ms: number | null | undefined): string {
return `${(ms / 1000).toFixed(2)}s`
}
function hasRecordDisplayLatency(record: UsageRecord): boolean {
return record.end_to_end_time_ms != null
|| record.end_to_end_first_byte_time_ms != null
|| record.response_time_ms != null
|| record.first_byte_time_ms != null
}
function getRecordDisplayOutputRate(record: UsageRecord): number | null {
return getDisplayOutputRate({
output_tokens: record.output_tokens,
@@ -1468,8 +1478,10 @@ function getRecordDisplayOutputRate(record: UsageRecord): number | null {
function getRecordPerformanceTitle(record: UsageRecord): string {
const outputRate = getRecordDisplayOutputRate(record)
return [
`首字: ${formatRecordDurationSeconds(record.first_byte_time_ms)}`,
`总耗时: ${formatRecordDurationSeconds(record.response_time_ms)}`,
`端到端首字: ${formatRecordDurationSeconds(record.end_to_end_first_byte_time_ms ?? record.first_byte_time_ms)}`,
`端到端总耗时: ${formatRecordDurationSeconds(record.end_to_end_time_ms ?? record.response_time_ms)}`,
`成功候选首字: ${formatRecordDurationSeconds(record.first_byte_time_ms)}`,
`成功候选耗时: ${formatRecordDurationSeconds(record.response_time_ms)}`,
`生成耗时: ${formatRecordDurationSeconds(getGenerationTimeMs(record))}`,
`输出速度: ${formatOutputRateTokensPerSecond(outputRate)}`,
].join('\n')
@@ -181,6 +181,46 @@ afterEach(() => {
})
describe('HorizontalRequestTimeline', () => {
it('uses the trace aggregate latency instead of the successful candidate latency', async () => {
const trace = buildTrace([
buildCandidate({
id: 'cand-transport-timeout',
provider_id: 'provider-timeout',
provider_name: 'Provider Timeout',
key_id: 'key-timeout',
key_name: 'Timeout Key',
candidate_index: 0,
status: 'failed',
latency_ms: 10_000,
started_at: '2026-05-06T12:00:00.000Z',
finished_at: '2026-05-06T12:00:10.000Z',
}),
buildCandidate({
id: 'cand-success-after-failover',
provider_id: 'provider-success',
provider_name: 'Provider Success',
key_id: 'key-success',
key_name: 'Success Key',
candidate_index: 1,
status: 'success',
latency_ms: 626,
started_at: '2026-05-06T12:00:10.000Z',
finished_at: '2026-05-06T12:00:10.626Z',
}),
])
trace.total_latency_ms = 10_626
const root = mountTimeline(trace)
await nextTick()
const heading = [...root.querySelectorAll('h4')]
.find(element => element.textContent?.trim() === '请求链路追踪')
const overview = heading?.parentElement?.parentElement
const displayedLatency = overview?.lastElementChild?.textContent?.trim()
expect(displayedLatency).toBe('10.63s')
expect(displayedLatency).not.toBe('626ms')
})
it('keeps attempted keys visible for ordinary provider groups that are not selected', async () => {
const trace = buildTrace([
buildCandidate({
@@ -113,6 +113,48 @@ function buildFastTierDetail(): RequestDetail {
}
describe('RequestDetailDrawer settlement pricing', () => {
it('shows end-to-end latency while keeping output TPS scoped to candidate timing', async () => {
apiMocks.getRequestDetail.mockResolvedValue({
...buildEmbeddingDetail(),
tokens: { input: 100, output: 50, total: 150 },
input_tokens: 100,
output_tokens: 50,
total_tokens: 150,
is_stream: true,
upstream_is_stream: true,
response_time_ms: 626,
first_byte_time_ms: 100,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
} satisfies RequestDetail)
let isOpen!: Ref<boolean>
const Host = defineComponent({
setup() {
isOpen = ref(false)
return () => h(RequestDetailDrawer, {
isOpen: isOpen.value,
requestId: 'usage-embedding-1',
})
},
})
const root = document.createElement('div')
document.body.appendChild(root)
const app = createApp(Host)
app.mount(root)
mountedApps.push({ app, root })
isOpen.value = true
await nextTick()
await vi.waitFor(() => {
expect(document.body.textContent).toContain('10.12s / 10.63s')
expect(document.body.textContent).toContain('95.1tps')
expect(document.body.textContent).not.toContain('98.8tps')
})
})
it('renders an input-only embedding tier without treating the missing output price as zero', async () => {
apiMocks.getRequestDetail.mockResolvedValue(buildEmbeddingDetail())
@@ -195,8 +195,10 @@ describe('UsageRecordsTable', () => {
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'首字: 0.50s',
'总耗时: 1.00s',
'端到端首字: 0.50s',
'端到端总耗时: 1.00s',
'成功候选首字: 0.50s',
'成功候选耗时: 1.00s',
'生成耗时: 0.50s',
'输出速度: 100 tokens/s',
].join('\n'))
@@ -204,6 +206,55 @@ describe('UsageRecordsTable', () => {
expect(titles.join('\n')).not.toContain('首字后生成耗时')
})
it('shows end-to-end latency while keeping output TPS scoped to the successful candidate', () => {
const root = mountUsageRecordsTable([buildRecord({
output_tokens: 50,
response_time_ms: 626,
first_byte_time_ms: 100,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
})])
const performanceCell = root.querySelector('table tbody tr td:last-child') as HTMLElement
expect(performanceCell.textContent).toContain('10.12s / 10.63s')
expect(performanceCell.textContent).toContain('95.1 tps')
expect(performanceCell.textContent).not.toContain('98.8 tps')
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'端到端首字: 10.12s',
'端到端总耗时: 10.63s',
'成功候选首字: 0.10s',
'成功候选耗时: 0.63s',
'生成耗时: 0.53s',
'输出速度: 95.1 tokens/s',
].join('\n'))
})
it('shows end-to-end latency when candidate timing fields are unavailable', () => {
const root = mountUsageRecordsTable([buildRecord({
response_time_ms: null,
first_byte_time_ms: null,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
})])
const performanceCell = root.querySelector('table tbody tr td:last-child') as HTMLElement
expect(performanceCell.textContent).toContain('10.12s / 10.63s')
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'端到端首字: 10.12s',
'端到端总耗时: 10.63s',
'成功候选首字: -',
'成功候选耗时: -',
'生成耗时: -',
'输出速度: -',
].join('\n'))
})
it('shows an output speed placeholder when the rate is unavailable', () => {
const root = mountUsageRecordsTable([buildRecord({
output_tokens: 0,
@@ -218,8 +269,10 @@ describe('UsageRecordsTable', () => {
const titles = [...root.querySelectorAll<HTMLElement>('[title]')].map((element) => element.title)
expect(titles).toContain([
'首字: 0.50s',
'总耗时: 1.00s',
'端到端首字: 0.50s',
'端到端总耗时: 1.00s',
'成功候选首字: 0.50s',
'成功候选耗时: 1.00s',
'生成耗时: 0.50s',
'输出速度: -',
].join('\n'))
+2
View File
@@ -117,6 +117,8 @@ export interface UsageRecord {
actual_cost?: number
response_time_ms?: number | null
first_byte_time_ms?: number | null // 首字时间 (TTFB)
end_to_end_time_ms?: number | null // 客户端从请求进入网关到完成的总耗时
end_to_end_first_byte_time_ms?: number | null // 客户端从请求进入网关到首字节的耗时
is_stream: boolean
upstream_is_stream?: boolean
client_requested_stream?: boolean