feat(gateway): harden failover and payload handling

Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
This commit is contained in:
elky
2026-07-30 01:03:27 +08:00
parent a97acc07fc
commit a04673a90d
80 changed files with 3640 additions and 1236 deletions
@@ -358,7 +358,7 @@
<span>{{ formatOutputRate(getRecordDisplayOutputRate(record)) }}</span>
</template>
<span
v-else-if="record.response_time_ms != null || record.first_byte_time_ms != null"
v-else-if="hasRecordDisplayLatency(record)"
class="ml-1"
>{{ formatRecordLatencyPair(record) }} / {{ formatOutputRate(getRecordDisplayOutputRate(record)) }}</span>
<span
@@ -611,7 +611,8 @@
v-if="isColumnVisible('performance')"
class="h-12 font-semibold w-[9%] text-right"
>
<div class="flex flex-col items-end text-xs gap-0.5">
<div class="flex flex-col items-end text-[11px] leading-3">
<span class="whitespace-nowrap">端到端</span>
<span class="whitespace-nowrap">首字/总耗时</span>
<span class="text-muted-foreground font-normal">输出速度</span>
</div>
@@ -925,7 +926,7 @@
</div>
<!-- 已完成状态:首字 + 总耗时 -->
<div
v-else-if="record.response_time_ms != null || record.first_byte_time_ms != null"
v-else-if="hasRecordDisplayLatency(record)"
class="flex flex-col items-end text-xs gap-0.5"
:title="getRecordPerformanceTitle(record)"
>
@@ -1445,8 +1446,10 @@ function getRecordCacheTokensTitle(record: UsageRecord): string {
}
function formatRecordLatencyPair(record: UsageRecord): string {
const firstByte = formatRecordDurationSeconds(record.first_byte_time_ms)
const total = formatRecordDurationSeconds(record.response_time_ms)
const firstByte = formatRecordDurationSeconds(
record.end_to_end_first_byte_time_ms ?? record.first_byte_time_ms,
)
const total = formatRecordDurationSeconds(record.end_to_end_time_ms ?? record.response_time_ms)
return `${firstByte} / ${total}`
}
@@ -1455,6 +1458,13 @@ function formatRecordDurationSeconds(ms: number | null | undefined): string {
return `${(ms / 1000).toFixed(2)}s`
}
function hasRecordDisplayLatency(record: UsageRecord): boolean {
return record.end_to_end_time_ms != null
|| record.end_to_end_first_byte_time_ms != null
|| record.response_time_ms != null
|| record.first_byte_time_ms != null
}
function getRecordDisplayOutputRate(record: UsageRecord): number | null {
return getDisplayOutputRate({
output_tokens: record.output_tokens,
@@ -1468,8 +1478,10 @@ function getRecordDisplayOutputRate(record: UsageRecord): number | null {
function getRecordPerformanceTitle(record: UsageRecord): string {
const outputRate = getRecordDisplayOutputRate(record)
return [
`首字: ${formatRecordDurationSeconds(record.first_byte_time_ms)}`,
`总耗时: ${formatRecordDurationSeconds(record.response_time_ms)}`,
`端到端首字: ${formatRecordDurationSeconds(record.end_to_end_first_byte_time_ms ?? record.first_byte_time_ms)}`,
`端到端总耗时: ${formatRecordDurationSeconds(record.end_to_end_time_ms ?? record.response_time_ms)}`,
`成功候选首字: ${formatRecordDurationSeconds(record.first_byte_time_ms)}`,
`成功候选耗时: ${formatRecordDurationSeconds(record.response_time_ms)}`,
`生成耗时: ${formatRecordDurationSeconds(getGenerationTimeMs(record))}`,
`输出速度: ${formatOutputRateTokensPerSecond(outputRate)}`,
].join('\n')