feat(gateway): harden failover and payload handling

Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
This commit is contained in:
elky
2026-07-30 01:03:27 +08:00
parent a97acc07fc
commit a04673a90d
80 changed files with 3640 additions and 1236 deletions
@@ -181,6 +181,46 @@ afterEach(() => {
})
describe('HorizontalRequestTimeline', () => {
it('uses the trace aggregate latency instead of the successful candidate latency', async () => {
const trace = buildTrace([
buildCandidate({
id: 'cand-transport-timeout',
provider_id: 'provider-timeout',
provider_name: 'Provider Timeout',
key_id: 'key-timeout',
key_name: 'Timeout Key',
candidate_index: 0,
status: 'failed',
latency_ms: 10_000,
started_at: '2026-05-06T12:00:00.000Z',
finished_at: '2026-05-06T12:00:10.000Z',
}),
buildCandidate({
id: 'cand-success-after-failover',
provider_id: 'provider-success',
provider_name: 'Provider Success',
key_id: 'key-success',
key_name: 'Success Key',
candidate_index: 1,
status: 'success',
latency_ms: 626,
started_at: '2026-05-06T12:00:10.000Z',
finished_at: '2026-05-06T12:00:10.626Z',
}),
])
trace.total_latency_ms = 10_626
const root = mountTimeline(trace)
await nextTick()
const heading = [...root.querySelectorAll('h4')]
.find(element => element.textContent?.trim() === '请求链路追踪')
const overview = heading?.parentElement?.parentElement
const displayedLatency = overview?.lastElementChild?.textContent?.trim()
expect(displayedLatency).toBe('10.63s')
expect(displayedLatency).not.toBe('626ms')
})
it('keeps attempted keys visible for ordinary provider groups that are not selected', async () => {
const trace = buildTrace([
buildCandidate({
@@ -113,6 +113,48 @@ function buildFastTierDetail(): RequestDetail {
}
describe('RequestDetailDrawer settlement pricing', () => {
it('shows end-to-end latency while keeping output TPS scoped to candidate timing', async () => {
apiMocks.getRequestDetail.mockResolvedValue({
...buildEmbeddingDetail(),
tokens: { input: 100, output: 50, total: 150 },
input_tokens: 100,
output_tokens: 50,
total_tokens: 150,
is_stream: true,
upstream_is_stream: true,
response_time_ms: 626,
first_byte_time_ms: 100,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
} satisfies RequestDetail)
let isOpen!: Ref<boolean>
const Host = defineComponent({
setup() {
isOpen = ref(false)
return () => h(RequestDetailDrawer, {
isOpen: isOpen.value,
requestId: 'usage-embedding-1',
})
},
})
const root = document.createElement('div')
document.body.appendChild(root)
const app = createApp(Host)
app.mount(root)
mountedApps.push({ app, root })
isOpen.value = true
await nextTick()
await vi.waitFor(() => {
expect(document.body.textContent).toContain('10.12s / 10.63s')
expect(document.body.textContent).toContain('95.1tps')
expect(document.body.textContent).not.toContain('98.8tps')
})
})
it('renders an input-only embedding tier without treating the missing output price as zero', async () => {
apiMocks.getRequestDetail.mockResolvedValue(buildEmbeddingDetail())
@@ -195,8 +195,10 @@ describe('UsageRecordsTable', () => {
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'首字: 0.50s',
'总耗时: 1.00s',
'端到端首字: 0.50s',
'端到端总耗时: 1.00s',
'成功候选首字: 0.50s',
'成功候选耗时: 1.00s',
'生成耗时: 0.50s',
'输出速度: 100 tokens/s',
].join('\n'))
@@ -204,6 +206,55 @@ describe('UsageRecordsTable', () => {
expect(titles.join('\n')).not.toContain('首字后生成耗时')
})
it('shows end-to-end latency while keeping output TPS scoped to the successful candidate', () => {
const root = mountUsageRecordsTable([buildRecord({
output_tokens: 50,
response_time_ms: 626,
first_byte_time_ms: 100,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
})])
const performanceCell = root.querySelector('table tbody tr td:last-child') as HTMLElement
expect(performanceCell.textContent).toContain('10.12s / 10.63s')
expect(performanceCell.textContent).toContain('95.1 tps')
expect(performanceCell.textContent).not.toContain('98.8 tps')
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'端到端首字: 10.12s',
'端到端总耗时: 10.63s',
'成功候选首字: 0.10s',
'成功候选耗时: 0.63s',
'生成耗时: 0.53s',
'输出速度: 95.1 tokens/s',
].join('\n'))
})
it('shows end-to-end latency when candidate timing fields are unavailable', () => {
const root = mountUsageRecordsTable([buildRecord({
response_time_ms: null,
first_byte_time_ms: null,
end_to_end_time_ms: 10_626,
end_to_end_first_byte_time_ms: 10_120,
})])
const performanceCell = root.querySelector('table tbody tr td:last-child') as HTMLElement
expect(performanceCell.textContent).toContain('10.12s / 10.63s')
const titles = [...root.querySelectorAll<HTMLElement>('[title]')]
.map((element) => element.getAttribute('title'))
expect(titles).toContain([
'端到端首字: 10.12s',
'端到端总耗时: 10.63s',
'成功候选首字: -',
'成功候选耗时: -',
'生成耗时: -',
'输出速度: -',
].join('\n'))
})
it('shows an output speed placeholder when the rate is unavailable', () => {
const root = mountUsageRecordsTable([buildRecord({
output_tokens: 0,
@@ -218,8 +269,10 @@ describe('UsageRecordsTable', () => {
const titles = [...root.querySelectorAll<HTMLElement>('[title]')].map((element) => element.title)
expect(titles).toContain([
'首字: 0.50s',
'总耗时: 1.00s',
'端到端首字: 0.50s',
'端到端总耗时: 1.00s',
'成功候选首字: 0.50s',
'成功候选耗时: 1.00s',
'生成耗时: 0.50s',
'输出速度: -',
].join('\n'))