feat(gateway): harden failover and payload handling

Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
This commit is contained in:
elky
2026-07-30 01:03:27 +08:00
parent a97acc07fc
commit a04673a90d
80 changed files with 3640 additions and 1236 deletions
@@ -1023,25 +1023,26 @@ const conversionBoundaryIndex = computed(() => {
return idx
})
// 计算链路总耗时(使用成功候选的 latency_ms 字段)
// 优先使用 latency_ms,因为它与 Usage.response_time_ms 使用相同的时间基准
// 避免 finished_at - started_at 带来的额外延迟(数据库操作时间)
// The trace aggregate includes every attempted candidate, including failed
// failover attempts. Per-candidate latency remains provider-scoped.
const totalTraceLatency = computed(() => {
if (!rawTimeline.value || rawTimeline.value.length === 0) return 0
// 查找成功的候选,使用其 latency_ms
const successCandidate = rawTimeline.value.find(c => c.status === 'success')
if (successCandidate?.latency_ms != null) {
return successCandidate.latency_ms
const aggregateLatency = trace.value?.total_latency_ms
if (typeof aggregateLatency === 'number' && Number.isFinite(aggregateLatency) && aggregateLatency > 0) {
return aggregateLatency
}
// 如果没有成功的候选,查找失败但有 latency_ms 的候选
const failedWithLatency = rawTimeline.value.find(c => c.status === 'failed' && c.latency_ms != null)
if (failedWithLatency?.latency_ms != null) {
return failedWithLatency.latency_ms
const attemptedLatency = rawTimeline.value.reduce((sum, candidate) => {
const latency = normalizeLatencyMs(candidate.latency_ms)
return sum + (latency ?? 0)
}, 0)
if (attemptedLatency > 0) {
return attemptedLatency
}
// 回退:使用 finished_at - started_at 计算
// Historical transport failures may not have latency_ms. Recover the wall
// clock span from candidate timestamps for those records.
let earliestStart: number | null = null
let latestEnd: number | null = null
@@ -175,7 +175,7 @@
<span>
<span class="text-muted-foreground">耗时</span>
<span class="ml-1 font-bold">
{{ formatDurationMs(detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.response_time_ms) }}
{{ formatDurationMs(detail.end_to_end_first_byte_time_ms ?? detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.end_to_end_time_ms ?? detail.response_time_ms) }}
</span>
</span>
<span class="text-muted-foreground">|</span>
@@ -197,7 +197,7 @@
<span class="whitespace-nowrap">
<span class="text-muted-foreground">耗时</span>
<span class="ml-1 font-bold">
{{ formatDurationMs(detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.response_time_ms) }}
{{ formatDurationMs(detail.end_to_end_first_byte_time_ms ?? detail.first_byte_time_ms) }} / {{ formatDurationMs(detail.end_to_end_time_ms ?? detail.response_time_ms) }}
</span>
</span>
<span class="text-muted-foreground">|</span>
@@ -358,7 +358,7 @@
<span>{{ formatOutputRate(getRecordDisplayOutputRate(record)) }}</span>
</template>
<span
v-else-if="record.response_time_ms != null || record.first_byte_time_ms != null"
v-else-if="hasRecordDisplayLatency(record)"
class="ml-1"
>{{ formatRecordLatencyPair(record) }} / {{ formatOutputRate(getRecordDisplayOutputRate(record)) }}</span>
<span
@@ -611,7 +611,8 @@
v-if="isColumnVisible('performance')"
class="h-12 font-semibold w-[9%] text-right"
>
<div class="flex flex-col items-end text-xs gap-0.5">
<div class="flex flex-col items-end text-[11px] leading-3">
<span class="whitespace-nowrap">端到端</span>
<span class="whitespace-nowrap">首字/总耗时</span>
<span class="text-muted-foreground font-normal">输出速度</span>
</div>
@@ -925,7 +926,7 @@
</div>
<!-- 已完成状态:首字 + 总耗时 -->
<div
v-else-if="record.response_time_ms != null || record.first_byte_time_ms != null"
v-else-if="hasRecordDisplayLatency(record)"
class="flex flex-col items-end text-xs gap-0.5"
:title="getRecordPerformanceTitle(record)"
>
@@ -1445,8 +1446,10 @@ function getRecordCacheTokensTitle(record: UsageRecord): string {
}
function formatRecordLatencyPair(record: UsageRecord): string {
const firstByte = formatRecordDurationSeconds(record.first_byte_time_ms)
const total = formatRecordDurationSeconds(record.response_time_ms)
const firstByte = formatRecordDurationSeconds(
record.end_to_end_first_byte_time_ms ?? record.first_byte_time_ms,
)
const total = formatRecordDurationSeconds(record.end_to_end_time_ms ?? record.response_time_ms)
return `${firstByte} / ${total}`
}
@@ -1455,6 +1458,13 @@ function formatRecordDurationSeconds(ms: number | null | undefined): string {
return `${(ms / 1000).toFixed(2)}s`
}
function hasRecordDisplayLatency(record: UsageRecord): boolean {
return record.end_to_end_time_ms != null
|| record.end_to_end_first_byte_time_ms != null
|| record.response_time_ms != null
|| record.first_byte_time_ms != null
}
function getRecordDisplayOutputRate(record: UsageRecord): number | null {
return getDisplayOutputRate({
output_tokens: record.output_tokens,
@@ -1468,8 +1478,10 @@ function getRecordDisplayOutputRate(record: UsageRecord): number | null {
function getRecordPerformanceTitle(record: UsageRecord): string {
const outputRate = getRecordDisplayOutputRate(record)
return [
`首字: ${formatRecordDurationSeconds(record.first_byte_time_ms)}`,
`总耗时: ${formatRecordDurationSeconds(record.response_time_ms)}`,
`端到端首字: ${formatRecordDurationSeconds(record.end_to_end_first_byte_time_ms ?? record.first_byte_time_ms)}`,
`端到端总耗时: ${formatRecordDurationSeconds(record.end_to_end_time_ms ?? record.response_time_ms)}`,
`成功候选首字: ${formatRecordDurationSeconds(record.first_byte_time_ms)}`,
`成功候选耗时: ${formatRecordDurationSeconds(record.response_time_ms)}`,
`生成耗时: ${formatRecordDurationSeconds(getGenerationTimeMs(record))}`,
`输出速度: ${formatOutputRateTokensPerSecond(outputRate)}`,
].join('\n')
@@ -181,6 +181,46 @@ afterEach(() => {
})
describe('HorizontalRequestTimeline', () => {
it('uses the trace aggregate latency instead of the successful candidate latency', async () => {
const trace = buildTrace([
buildCandidate({
id: 'cand-transport-timeout',
provider_id: 'provider-timeout',
provider_name: 'Provider Timeout',
key_id: 'key-timeout',
key_name: 'Timeout Key',
candidate_index: 0,
status: 'failed',
latency_ms: 10_000,
started_at: '2026-05-06T12:00:00.000Z',
finished_at: '2026-05-06T12:00:10.000Z',
}),
buildCandidate({
id: 'cand-success-after-failover',
provider_id: 'provider-success',
provider_name: 'Provider Success',
key_id: 'key-success',
key_name: 'Success Key',
candidate_index: 1,
status: 'success',
latency_ms: 626,
started_at: '2026-05-06T12:00:10.000Z',
finished_at: '2026-05-06T12:00:10.626Z',
}),
])
trace.total_latency_ms = 10_626
const root = mountTimeline(trace)
await nextTick()
const heading = [...root.querySelectorAll('h4')]
.find(element => element.textContent?.trim() === '请求链路追踪')
const overview = heading?.parentElement?.parentElement
const displayedLatency = overview?.lastElementChild?.textContent?.trim()
expect(displayedLatency).toBe('10.63s')
expect(displayedLatency).not.toBe('626ms')
})
it('keeps attempted keys visible for ordinary provider groups that are not selected', async () => {
const trace = buildTrace([
buildCandidate({
@@ -113,6 +113,48 @@ function buildFastTierDetail(): RequestDetail {
}
describe('RequestDetailDrawer settlement pricing', () => {
it('shows end-to-end latency while keeping output TPS scoped to candidate timing', async () => {
apiMocks.getRequestDetail.mockResolvedValue({
...buildEmbeddingDetail(),
tokens: { input: 100, output: 50, total: 150 },
input_tokens: 100,
output_tokens: 50,
total_tokens: 150,
is_stream: true,
upstream_is_stream: true,
response_time_ms: 626,
first_byte_time_ms: 100,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
} satisfies RequestDetail)
let isOpen!: Ref<boolean>
const Host = defineComponent({
setup() {
isOpen = ref(false)
return () => h(RequestDetailDrawer, {
isOpen: isOpen.value,
requestId: 'usage-embedding-1',
})
},
})
const root = document.createElement('div')
document.body.appendChild(root)
const app = createApp(Host)
app.mount(root)
mountedApps.push({ app, root })
isOpen.value = true
await nextTick()
await vi.waitFor(() => {
expect(document.body.textContent).toContain('10.12s / 10.63s')
expect(document.body.textContent).toContain('95.1tps')
expect(document.body.textContent).not.toContain('98.8tps')
})
})
it('renders an input-only embedding tier without treating the missing output price as zero', async () => {
apiMocks.getRequestDetail.mockResolvedValue(buildEmbeddingDetail())
@@ -195,8 +195,10 @@ describe('UsageRecordsTable', () => {
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'首字: 0.50s',
'总耗时: 1.00s',
'端到端首字: 0.50s',
'端到端总耗时: 1.00s',
'成功候选首字: 0.50s',
'成功候选耗时: 1.00s',
'生成耗时: 0.50s',
'输出速度: 100 tokens/s',
].join('\n'))
@@ -204,6 +206,55 @@ describe('UsageRecordsTable', () => {
expect(titles.join('\n')).not.toContain('首字后生成耗时')
})
it('shows end-to-end latency while keeping output TPS scoped to the successful candidate', () => {
const root = mountUsageRecordsTable([buildRecord({
output_tokens: 50,
response_time_ms: 626,
first_byte_time_ms: 100,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
})])
const performanceCell = root.querySelector('table tbody tr td:last-child') as HTMLElement
expect(performanceCell.textContent).toContain('10.12s / 10.63s')
expect(performanceCell.textContent).toContain('95.1 tps')
expect(performanceCell.textContent).not.toContain('98.8 tps')
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'端到端首字: 10.12s',
'端到端总耗时: 10.63s',
'成功候选首字: 0.10s',
'成功候选耗时: 0.63s',
'生成耗时: 0.53s',
'输出速度: 95.1 tokens/s',
].join('\n'))
})
it('shows end-to-end latency when candidate timing fields are unavailable', () => {
const root = mountUsageRecordsTable([buildRecord({
response_time_ms: null,
first_byte_time_ms: null,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
})])
const performanceCell = root.querySelector('table tbody tr td:last-child') as HTMLElement
expect(performanceCell.textContent).toContain('10.12s / 10.63s')
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'端到端首字: 10.12s',
'端到端总耗时: 10.63s',
'成功候选首字: -',
'成功候选耗时: -',
'生成耗时: -',
'输出速度: -',
].join('\n'))
})
it('shows an output speed placeholder when the rate is unavailable', () => {
const root = mountUsageRecordsTable([buildRecord({
output_tokens: 0,
@@ -218,8 +269,10 @@ describe('UsageRecordsTable', () => {
const titles = [...root.querySelectorAll<HTMLElement>('[title]')].map((element) => element.title)
expect(titles).toContain([
'首字: 0.50s',
'总耗时: 1.00s',
'端到端首字: 0.50s',
'端到端总耗时: 1.00s',
'成功候选首字: 0.50s',
'成功候选耗时: 1.00s',
'生成耗时: 0.50s',
'输出速度: -',
].join('\n'))
+2
View File
@@ -117,6 +117,8 @@ export interface UsageRecord {
actual_cost?: number
response_time_ms?: number | null
first_byte_time_ms?: number | null // 首字时间 (TTFB)
end_to_end_time_ms?: number | null // 客户端从请求进入网关到完成的总耗时
end_to_end_first_byte_time_ms?: number | null // 客户端从请求进入网关到首字节的耗时
is_stream: boolean
upstream_is_stream?: boolean
client_requested_stream?: boolean