Compare commits

..
295 Commits
Author SHA1 Message Date
elky 06f5d3c8c0 fix(gateway): complete worker registration cleanup 2026-07-31 11:32:07 +08:00
elky 082407fa51 Merge PR #697: prevent duplicate worker registrations 2026-07-31 11:11:14 +08:00
fawney19 6688ee26db Merge pull request #702 from MMEXA/fix/reconcile-auth-channel-mismatch-formats
fix(gateway): 修复批量更新 API 格式时的认证通道状态冲突
2026-07-31 10:28:37 +08:00
elky beb003b7ad feat(models): add external catalog proxy selection 2026-07-31 09:32:25 +08:00
MMEXA 6ecfe0f0a1 fix(gateway): reconcile auth mismatch formats on key update 2026-07-30 22:14:08 +08:00
ZheFox 12057db476 Merge pull request #701 from zhefox/main
Persist OpenAI Responses continuation history across instances
2026-07-30 21:08:34 +08:00
ZheFox ff47d8d48a fix(gateway): route response history through ai seam 2026-07-30 20:34:41 +08:00
ZheFox ef5f36cc2b fix(ai): satisfy response history clippy checks 2026-07-30 20:13:36 +08:00
ZheFox 84022c4d48 Merge upstream/main into main 2026-07-30 19:40:39 +08:00
ZheFox 118f441029 feat(gateway): persist OpenAI Responses continuation history 2026-07-30 19:26:52 +08:00
elky 20399b004d Merge PR #700: fix admin pool batch update body buffering
Preserve main's failover and usage metadata fixes, restore default tunnel regression coverage, and satisfy current Clippy.
2026-07-30 17:56:37 +08:00
elky 050eb77508 fix(ai): harden responses replay and failure diagnostics 2026-07-30 17:19:54 +08:00
elky 1ab4f079c9 fix(gateway): restore failover and usage diagnostics 2026-07-30 09:12:11 +08:00
MMEXA 6c733f7590 fix(usage): preserve request diagnostics in event seeds 2026-07-30 06:44:59 +08:00
MMEXA d7d8db45ba test(gateway): align tunnel error fixture with failover policy 2026-07-30 06:44:59 +08:00
MMEXA 8cf9af79da fix(ci): remove redundant usage policy update 2026-07-30 05:45:38 +08:00
MMEXA e55793c765 fix(ci): satisfy gateway clippy on upstream baseline 2026-07-30 05:14:34 +08:00
MMEXA d8902ea612 fix(gateway): buffer admin pool batch update bodies 2026-07-30 05:14:34 +08:00
elky a04673a90d feat(gateway): harden failover and payload handling
Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
2026-07-30 01:03:27 +08:00
ZheFox a97acc07fc Merge pull request #698 from zhefox/main
fix(ci): stabilize cross-platform workflow checks
2026-07-29 22:16:17 +08:00
zhefox f8000012f7 fix(ci): stabilize cross-platform workflow checks 2026-07-29 21:55:43 +08:00
worker-2 6080f8cc88 fix(gateway): stabilize worker task records
Key worker boot records by task so process restarts update the existing
row instead of registering another row for each gateway instance.

Closes #693
Confidence: high
Scope-risk: narrow
2026-07-29 17:27:51 +08:00
ZheFox 37df5b93b1 Merge pull request #696 from zhefox/main
Fix client metadata handling across formats
2026-07-28 18:12:59 +08:00
ZheFox e53abdaec2 Merge branch 'fawney19:main' into main 2026-07-28 18:12:28 +08:00
ZheFox 2db32ea97e Merge branch 'main' of https://github.com/zhefox/Aether 2026-07-28 17:40:46 +08:00
ZheFox 581897ee74 fix(formats): ignore responses client metadata across targets 2026-07-28 17:40:41 +08:00
ZheFox 9a88f966d8 Merge pull request #695 from zhefox/main
Enhance provider capabilities and clean up OAuth keys
2026-07-28 13:58:33 +08:00
ZheFox 9d9316e434 Merge branch 'fawney19:main' into main 2026-07-28 13:56:34 +08:00
ZheFox 1b697b1111 feat(providers): support FedRAMP Codex agent identity registration 2026-07-28 13:29:30 +08:00
ZheFox 3043982486 fix(providers): derive Codex primary quota label from window 2026-07-28 12:57:45 +08:00
ZheFox 0bf92ffffc feat(providers): advertise responses API agent capability 2026-07-28 12:02:03 +08:00
ZheFox f0f87b56a3 feat(providers): add credential-fenced OAuth key cleanup 2026-07-28 11:32:11 +08:00
elky 4148ab1931 fix(routing): harden routed pool scheduling 2026-07-27 22:06:28 +08:00
elky 550cc36760 feat(providers): expand OAuth account management
Add Claude Code manual and cookie authorization, including redacted batch tasks. Harden OAuth imports, duplicate replacement, provider dialogs, and related account-management tests.
2026-07-27 15:53:28 +08:00
elky 531cf11025 feat(gateway): harden provider request execution
Preserve exact request payloads and model client surface and API operation explicitly.

Add Anthropic compatibility profiles, bounded stream commitment, and scoped OAuth retry behavior across provider transports.
2026-07-27 09:36:31 +08:00
elky 79b70f7b5c fix(frontend): align sidebar collapse button 2026-07-26 15:07:51 +08:00
elky 10d369f59c feat(providers): add provider transfer limits 2026-07-26 15:06:56 +08:00
elky 2ef7ac79bc feat(frontend): add collapsible navigation sidebar
Persist the desktop sidebar state, provide accessible compact navigation tooltips, and cover the collapsed navigation markup with a focused component test.
2026-07-25 21:28:51 +08:00
elky 778cfb1a5c feat(data): complete portable SQL backend parity
Align MySQL and SQLite schemas, migrations, usage, stats, export, and backfill behavior with the shared data contracts. Extend gateway startup and maintenance support across all SQL drivers.
2026-07-25 21:28:21 +08:00
elky 764e9fd131 feat(frontend): improve provider detail drawer and pool actions 2026-07-25 11:16:59 +08:00
elky 387134ca87 fix(models): correct fast pricing and online sync 2026-07-24 01:45:38 +08:00
elky a0767d957c fix(frontend): synchronize pool account state 2026-07-23 16:20:18 +08:00
ZheFox b94ef91d07 Merge pull request #692 from zhefox/main
Sync global model prices and track online pricing sources
2026-07-23 16:02:17 +08:00
ZheFox e7910751d9 Merge branch 'fawney19:main' into main 2026-07-23 15:19:48 +08:00
ZheFox 1d2655432d feat(models): track online pricing sources and unsupported fields 2026-07-23 15:18:08 +08:00
ZheFox 323273ff30 feat(models): sync global model prices from online catalog 2026-07-23 13:29:28 +08:00
ZheFox fb2009c65b Merge pull request #691 from zhefox/main
fix(formats): ignore Responses client transport metadata
2026-07-23 12:18:40 +08:00
ZheFox e186cc6848 Merge branch 'main' of https://github.com/zhefox/Aether 2026-07-23 12:17:46 +08:00
ZheFox 615ac99ad7 fix(formats): ignore Responses client transport metadata 2026-07-23 12:16:44 +08:00
ZheFox ec36cfbf75 Merge pull request #690 from zhefox/main
fix(provider): classify deleted Codex agent runtime as invalid
2026-07-23 11:22:30 +08:00
ZheFox 7bf228a33c fix(provider): classify deleted Codex agent runtime as invalid 2026-07-23 11:21:55 +08:00
elky 3606290ac8 fix(provider): harden Agent Identity OAuth lifecycle 2026-07-23 09:33:00 +08:00
elky e49024d33b fix(frontend): shorten Agent Identity tab label 2026-07-22 20:26:49 +08:00
elky fdbc2607ec feat(provider): add dedicated Codex Agent Identity flow 2026-07-22 20:19:29 +08:00
elky c7cc8fd7db test(provider): simplify agent identity assertions 2026-07-22 14:19:58 +08:00
elky 07efcb5146 fix(data): repair legacy active flag synchronization 2026-07-22 14:19:34 +08:00
elky 856605defa fix(model-directives): harden suffix configuration 2026-07-22 14:19:09 +08:00
elky 713010fa0a fix(gateway): restore auth role refresh and Rust checks
Refresh the resolved user role without bypassing owner group and key policies. Resolve Rust 1.95 Clippy failures and make the pending persistence bound test scheduler-independent.
2026-07-22 11:25:24 +08:00
ZheFox cd2fbeeead Merge pull request #689 from AAEE86/feat/agent-identity-support
feat(codex): enroll agent identity from session token
2026-07-22 10:32:06 +08:00
AAEE86 a4350a482a feat(codex): enroll agent identity from session token 2026-07-22 10:21:11 +08:00
ZheFox c825375367 Merge pull request #688 from AAEE86/feat/agent-identity-support
feat(codex): support agent identity accounts
2026-07-22 09:20:11 +08:00
elky fc92c4f431 perf(gateway): scale request hot paths for 20k streams
Shard and singleflight hot-path caches, batch and prioritize candidate and usage lifecycle persistence, and extend database and pressure-test instrumentation for 20k concurrent streams.
2026-07-22 02:11:08 +08:00
AAEE86 b61c590bdb feat(codex): support agent identity accounts 2026-07-21 20:58:49 +08:00
ZheFox 7756c0913f Merge pull request #685 from zhefox/main
fix(gateway): apply group policy to admin-owned keys
2026-07-20 15:52:06 +08:00
ZheFox c34ec7c1ee fix(gateway): apply group policy to admin-owned keys 2026-07-20 15:51:01 +08:00
elky f8778c4a23 feat(gateway): configure cyber policy failover 2026-07-19 23:27:19 +08:00
elky e0dbb233f7 fix(frontend): avoid misleading cache TTL fallback label 2026-07-19 22:21:46 +08:00
elky 9725f9abae fix(frontend): clarify processing tier pricing 2026-07-19 22:00:12 +08:00
elky 5d575f1590 test(stats): treat bulk API key snapshots as authoritative 2026-07-19 19:07:04 +08:00
elky d562c594c3 fix(frontend): preserve compact scope and detail badge 2026-07-19 16:42:32 +08:00
MMEXA ce226a3010 Merge 0c3f51bcec into 644ae9c1bf 2026-07-19 16:12:37 +08:00
elky 644ae9c1bf feat(pool): add table-driven account batch actions 2026-07-19 16:09:20 +08:00
elky 95053f9502 Merge PR #672: 支持账号批量配置与可用模型管理 2026-07-18 22:06:32 +08:00
elky 8fbda84acb fix(data): preserve API key history end to end 2026-07-18 21:58:21 +08:00
elky 03b7d573e0 Merge PR #683: decouple API key historical identity 2026-07-18 21:20:35 +08:00
MMEXA 0c3f51bcec merge(main): 解决 usage 模型展示契约冲突 2026-07-18 19:26:14 +08:00
elky e3d97b573b fix(usage): align fast-tier pricing and model metadata 2026-07-18 16:57:04 +08:00
MMEXA ac3796af84 fix(gateway): 恢复响应边界并统一格式入口 2026-07-18 06:45:37 +08:00
MMEXA f9c343eb07 fix(gateway): 适配 Rust 1.95 整除检查 2026-07-18 05:55:06 +08:00
MMEXA e31df5989a merge(main): 解决 usage 展示与生命周期同步冲突 2026-07-18 05:38:38 +08:00
MMEXA 98fbf029fc fix(data): 解耦 API Key 历史统计身份 2026-07-18 04:42:45 +08:00
MMEXA 4d9a648202 test(gateway): 统一流错误测试的格式层入口 2026-07-18 03:37:31 +08:00
MMEXA 405ca3e66a fix(ci): 恢复非流式错误体边界并适配新版 Clippy 2026-07-18 03:26:53 +08:00
MMEXA 0355c28683 fix(data): 解耦候选记录的 API Key 历史身份 2026-07-18 02:40:34 +08:00
fawney19 6c33b8d8fb Merge pull request #682 from MMEXA/codex/codex-prompt-cache-identity-20260717
fix(codex): 统一通用缓存键与原生会话身份
2026-07-18 00:33:20 +08:00
elky a6c6f14b09 style(frontend): align pool cycle stats values 2026-07-18 00:13:35 +08:00
elky e558f55cd9 style(frontend): refine badges and cycle stats 2026-07-18 00:05:22 +08:00
elky 88a057b8d9 fix(usage): force fast badge background transparent 2026-07-17 22:56:54 +08:00
elky ed27d404ac style(usage): make fast badge background transparent 2026-07-17 22:52:47 +08:00
elky 5dda34c66e style(usage): give fast tier an amber accent 2026-07-17 22:36:55 +08:00
elky 373ebf26d6 fix(pricing): default zero tier ratios to one 2026-07-17 21:33:21 +08:00
elky f65ed2795c fix(codex): support dynamic quota windows 2026-07-17 20:18:04 +08:00
elky 664c063a06 feat(usage): enrich audit metadata and detail views 2026-07-17 19:20:16 +08:00
MMEXA 75795c6fbc test(codex): 对齐 Compact 确定性缓存身份 2026-07-17 08:49:35 +08:00
MMEXA 5b332da7d7 fix(codex): 补齐缓存身份终态请求头 2026-07-17 08:11:16 +08:00
MMEXA d9796d502b fix(codex): 统一通用缓存键与原生会话身份 2026-07-17 06:13:09 +08:00
MMEXA 3b0d87b0fd Merge remote-tracking branch 'origin/main' into codex/pool-key-bulk-management-20260714 2026-07-17 00:17:02 +08:00
MMEXA 3c348dff3a Merge remote-tracking branch 'origin/main' into codex/usage-pending-reasoning-reset-expiry-20260712
# Conflicts:
#	frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts
2026-07-17 00:16:58 +08:00
MMEXA ec1783a35c Merge remote-tracking branch 'origin/main' into codex/pool-key-bulk-management-20260714
# Conflicts:
#	apps/aether-gateway/src/handlers/admin/request/provider/tasks.rs
#	frontend/src/api/endpoints/pool.ts
2026-07-16 23:43:04 +08:00
MMEXA 427030c5de Merge remote-tracking branch 'origin/main' into codex/usage-pending-reasoning-reset-expiry-20260712
# Conflicts:
#	crates/aether-ai-formats/src/formats/openai/responses/mod.rs
#	crates/aether-usage/runtime/src/runtime.rs
#	frontend/src/features/usage/components/UsageRecordsTable.vue
#	frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts
2026-07-16 23:41:58 +08:00
elky 0be380243b feat(pricing): support processing tier multipliers 2026-07-16 23:30:42 +08:00
fawney19 312583f055 Merge pull request #680 from Kayphoon/codex/s3-backup-user-agent
feat(admin): configure S3 backup User-Agent
2026-07-16 23:30:30 +08:00
fawney19 33f49ea9b0 Merge pull request #678 from AAEE86/fix
fix: map Developer role to "system" in OpenAI Chat Completions output
2026-07-16 23:29:55 +08:00
fawney19 470cef17cf Merge pull request #676 from MMEXA/codex/sync-capture-envelope-finalize-20260716
修复同步 finalize 的 Responses 流聚合与转换
2026-07-16 23:29:38 +08:00
ZheFox 3f5f65eb9a Merge pull request #681 from zhefox/main
Codex 重置功能和显示缓存修复以及批量key的导入和管理功能
2026-07-16 19:48:33 +08:00
ZheFox 6664c2dbb8 feat(pool): 支持批量导入 Key 和选择性更新设置 2026-07-16 19:31:07 +08:00
ZheFox 0099167a6d fix(codex): 避免重置机会缺失触发配额刷新 2026-07-16 19:13:09 +08:00
ZheFox f009fb73c3 缓存问题修复 2026-07-16 18:55:28 +08:00
ZheFox 715a5ed626 修复重置次数缓存问题 2026-07-16 18:09:10 +08:00
ZheFox 5cf38d1b35 Codex 重置功能和显示修复 2026-07-16 17:23:41 +08:00
Kayphoon 6b707f29a2 feat(admin): configure S3 backup User-Agent 2026-07-16 08:56:56 +00:00
elky 9ea84f9748 fix(frontend): show service tier transitions 2026-07-16 16:38:30 +08:00
elky c32d043afb fix(frontend): preserve fetched model preset pricing 2026-07-16 16:38:30 +08:00
elky 8fe4d24408 fix(usage): canonicalize cached token totals 2026-07-16 16:38:30 +08:00
elky e369e4aab1 fix(formats): preserve chat-backed Responses metadata 2026-07-16 16:38:30 +08:00
ZheFox 7dc919e8e3 Merge pull request #679 from zhefox/main
fix(frontend): 优化移动端弹窗并完善提供商配额刷新
2026-07-16 15:48:49 +08:00
ZheFox 1333efdad5 fix(frontend): 优化移动端弹窗并完善提供商配额刷新 2026-07-16 15:21:11 +08:00
AAEE86 cd8de1aa13 fix: map Developer role to "system" in OpenAI Chat Completions output 2026-07-16 14:29:08 +08:00
elky d6215d9dec ci(tunnel): reduce artifact retention 2026-07-16 13:12:30 +08:00
elky 9a47267545 fix(usage): bound terminal event persistence
Add end-to-end terminal admission, bounded database fallback, and observable overload handling. Preserve first-byte lifecycle state across asynchronous runtime and frontend updates.
2026-07-16 13:12:30 +08:00
MMEXA 7851503fbc fix(finalize): 严格聚合并投影同步 Responses 流 2026-07-16 12:50:00 +08:00
ZheFox c6d373e6aa Merge pull request #677 from zhefox/main
fix(codex): 移除 Responses Lite 请求中的 context_management
2026-07-16 12:28:34 +08:00
ZheFox 71fcb9c168 fix(codex): 服务端压缩使用标准 Responses 合约 2026-07-16 12:02:33 +08:00
ZheFox 3976652942 fix(codex): 移除 Responses Lite 请求中的 context_management 2026-07-16 11:25:11 +08:00
MMEXA 7b56546e21 fix(finalize): 聚合同步流捕获包装 2026-07-16 10:24:32 +08:00
fawney19 85854e4476 Merge pull request #675 from fawney19/fix/pr-669-tail
feat(codex): complete PR #669 protocol follow-up
2026-07-16 08:54:29 +08:00
elky b50242ab9f fix(test): handle absent empty testkit bin directory 2026-07-16 01:29:51 +08:00
MMEXA 20b27a13b2 feat(codex): 按操作语义路由 Responses V2 压缩
(cherry picked from commit 2fc604e047)
2026-07-16 00:34:42 +08:00
MMEXA 598b2fb374 fix(auth): 授权 Responses Compact 伴随端点
(cherry picked from commit e8afa03e45)
2026-07-16 00:32:14 +08:00
MMEXA ff7988430d fix(openai): encode tool errors in Responses output
(cherry picked from commit f127b67e73)
2026-07-16 00:31:23 +08:00
MMEXA 25da99fac2 fix(codex): preserve reset consume request body
(cherry picked from commit fc2dfb82d2)
2026-07-16 00:27:28 +08:00
elky 8616fe6ee2 refactor(workspace): enforce layered crate boundaries 2026-07-15 23:47:19 +08:00
MMEXA 9b8724453b test(pool): 使用正式 Gemini API 格式 2026-07-14 08:39:41 +08:00
MMEXA 01e104d86a fix(pool): 对齐账号批量配置语义 2026-07-14 08:07:28 +08:00
MMEXA 0acd1de29c fix(gateway): 保持密钥更新模块显式所有权 2026-07-14 05:19:04 +08:00
MMEXA a25fab371a feat(pool): add bulk key configuration management 2026-07-14 04:57:05 +08:00
MMEXA 93e2f95c47 fix(frontend): 按端点能力约束会话压缩映射 2026-07-14 02:05:10 +08:00
MMEXA f10d631a9c feat(frontend): 澄清模型映射适用范围 2026-07-14 00:30:39 +08:00
MMEXA cfc4894dab fix(usage): 保留最新进行态生命周期事件 2026-07-14 00:30:24 +08:00
MMEXA 3f86fdd6bc feat(usage): 展示压缩操作与进行态请求语义 2026-07-13 22:03:44 +08:00
MMEXA b09d1f1c33 fix(usage): expose pending reasoning and exact reset expiry 2026-07-13 22:03:44 +08:00
MMEXA 2fc604e047 feat(codex): 按操作语义路由 Responses V2 压缩 2026-07-13 22:03:34 +08:00
MMEXA e8afa03e45 fix(auth): 授权 Responses Compact 伴随端点 2026-07-13 06:07:54 +08:00
MMEXA fc2dfb82d2 fix(codex): preserve reset consume request body 2026-07-12 23:04:33 +08:00
elky a728c090a9 fix(gateway): scope concurrency helper to tests 2026-07-12 22:36:48 +08:00
elky e58621a735 Merge PR #669: align GPT-5.6 and Codex request protocols 2026-07-12 21:50:20 +08:00
MMEXA b1be370b2e fix(gateway): scope concurrency test helper to tests 2026-07-12 21:08:30 +08:00
MMEXA cf0d957ac7 Merge f127b67e73 into 7f61bb43c7 2026-07-12 20:34:39 +08:00
MMEXA f127b67e73 fix(openai): encode tool errors in Responses output 2026-07-12 20:34:31 +08:00
elky 7f61bb43c7 feat(security): harden gateway request and runtime controls 2026-07-12 14:10:54 +08:00
MMEXA 25c49dd804 fix(data): keep terminal usage state monotonic 2026-07-12 05:45:37 +08:00
MMEXA 72222d935c test(gateway): use valid tunnel relay envelopes 2026-07-12 05:45:32 +08:00
MMEXA 063d517306 test(gateway): compare timeout response numerically 2026-07-12 04:43:13 +08:00
MMEXA 02495ce28e fix(admin): preserve inactive endpoint key counts 2026-07-12 04:19:06 +08:00
MMEXA 63936aa110 fix(gateway): route format rules through serving facade 2026-07-12 03:56:53 +08:00
MMEXA 8d4d42a887 fix(auth): resolve group policy before key intersection 2026-07-12 03:35:42 +08:00
MMEXA 2316df5c9a feat(codex): align Search and execution protocol 2026-07-12 03:04:15 +08:00
MMEXA 59d37ae1dd fix(frontend): import structured models.dev pricing 2026-07-11 18:12:55 +08:00
MMEXA 3014fd50c6 fix(billing): preserve effective cache and tier facts 2026-07-11 18:12:55 +08:00
MMEXA 14c4e3a04e fix(codex): enforce provider request identity 2026-07-11 18:09:14 +08:00
MMEXA 8f1070a451 feat(frontend): expose processing tier pricing 2026-07-11 12:27:09 +08:00
MMEXA 0b30cc6b0f feat(openai): unify tier authorization and settlement 2026-07-11 12:27:05 +08:00
MMEXA b2f596b8f0 fix(gateway): route Codex header through serving facade 2026-07-11 09:43:12 +08:00
MMEXA 01a96fed74 fix(codex): simplify summary normalization 2026-07-11 09:20:07 +08:00
MMEXA 46a903aada fix(codex): align current reasoning request semantics 2026-07-11 09:15:08 +08:00
MMEXA dfa121dd5b feat(openai): align GPT-5.6 and Codex request contracts 2026-07-11 07:40:12 +08:00
elky bc1da3bf3f feat(security): harden client IP and admin controls 2026-07-10 15:13:12 +08:00
elky 6e0dc3b59e feat(frontend): refine global model pricing dialog 2026-07-10 15:13:12 +08:00
elky 4bf5d4c044 Fix cache token accounting and tiered pricing 2026-07-10 15:13:12 +08:00
fawney19 736fc76345 Merge pull request #668 from MMEXA/codex/antigravity-empty-output-retry-20260709
修复 Gemini 空输出按候选重试处理
2026-07-10 09:16:56 +08:00
MMEXA b6b2ca38f4 触发 CI 重跑 2026-07-10 00:41:22 +08:00
MMEXA f07eb25cfc 修复 usage 详情 body 引用解包 2026-07-10 00:23:31 +08:00
MMEXA d2ea437c1c 修复 Gemini 空输出按候选重试处理 2026-07-09 23:10:30 +08:00
fawney19 7bc7d0f8d8 Merge pull request #666 from xixiknow/main
Fix provider key response time counter overflow
2026-07-09 18:04:11 +08:00
fawney19 14cf639aba Merge pull request #667 from MMEXA/codex/antigravity-v1internal-query-20260709
修复 Antigravity v1internal 查询参数透传
2026-07-09 17:50:38 +08:00
yangrs 55cdab592c Remove redundant response time conversion 2026-07-09 16:26:09 +08:00
MMEXA ee0ec18283 修复 Antigravity v1internal 查询参数透传 2026-07-09 16:03:29 +08:00
Start f31c9e03e2 Merge branch 'fawney19:main' into main 2026-07-09 15:11:44 +08:00
fawney19 e50db10439 Merge pull request #663 from MMEXA/codex/gemini-interactions-antigravity-20260705
完善 Gemini Interactions 与 Antigravity 全链路兼容
2026-07-09 14:55:51 +08:00
yangrs 192dc6c20d Fix provider key response time overflow 2026-07-09 14:40:48 +08:00
elky 5e1d14f19b Fix timeline duration display from latency 2026-07-09 11:46:45 +08:00
MMEXA b8b89d21b7 fix: 同步提交本地 sync 错误上报 2026-07-08 23:49:18 +08:00
MMEXA 5eddf4f9ee 细化 Antigravity 测试模型项目元数据补全 2026-07-08 22:45:30 +08:00
MMEXA c7186e1720 完善 Antigravity 配额展示与 CI 断言 2026-07-08 22:34:29 +08:00
MMEXA 4866509938 移除 Antigravity 未知重置时间噪音 2026-07-08 22:34:29 +08:00
MMEXA 2122660a5c 对齐原生 Antigravity 控制面与显示模型 2026-07-08 22:34:29 +08:00
MMEXA 5c68ab896a 恢复历史 backfill 兼容 live 账本 2026-07-08 22:34:29 +08:00
MMEXA f9c8ec41f4 完善 Antigravity 与 Gemini 跨格式兼容 2026-07-08 22:34:29 +08:00
MMEXA b1ed6b24b0 触发 CI 复跑 2026-07-08 22:34:29 +08:00
MMEXA c17c78ad4b 修正 Antigravity Gemini 3.5 Flash 档位展示 2026-07-08 22:34:29 +08:00
MMEXA 9ec48ab6b9 优化 Antigravity 配额展示顺序 2026-07-08 22:34:29 +08:00
MMEXA accd250226 修正 Antigravity 配额模型标签 2026-07-08 22:34:29 +08:00
MMEXA 80a6579766 支持 Gemini Interactions 与 Antigravity 配额精细化 2026-07-08 22:34:29 +08:00
fawney19 1ca83ca3fb Merge pull request #664 from MMEXA/codex/wallet-auth-cache-delay-20260706
修复钱包余额变更后的鉴权缓存延迟
2026-07-07 01:59:44 +08:00
fawney19 a931da0764 Merge pull request #662 from MMEXA/codex/reset-credit-20260704
增加 Codex 重置次数功能
2026-07-07 01:58:44 +08:00
fawney19 a61374c595 Merge pull request #661 from MMEXA/codex/frontend-debug-20260704
修复前端调试与基础交互问题
2026-07-07 01:57:41 +08:00
MMEXA c3136126e5 修复钱包余额变更后的鉴权缓存延迟 2026-07-06 06:15:31 +08:00
MMEXA b23d299533 重跑 Codex 重置次数 CI 2026-07-05 01:37:31 +08:00
MMEXA b03aae18c3 修复 Codex 重置次数 CI 检查 2026-07-04 15:47:01 +08:00
MMEXA 99b6fe468f 简化 Codex 重置机会展示标签 2026-07-04 15:16:38 +08:00
MMEXA ef77ec04ca 增加 Codex 重置次数功能 2026-07-04 06:10:53 +08:00
MMEXA 242081433e 修复前端调试与基础交互问题 2026-07-04 05:24:40 +08:00
ZheFox b86d4e1f0c Merge pull request #660 from zhefox/main
refactor(frontend): unify mobile menu background styles
2026-07-03 13:17:59 +08:00
ZheFox a151f37d63 refactor(frontend): unify mobile menu background styles 2026-07-03 13:17:18 +08:00
ZheFox 42f7907740 Merge pull request #659 from zhefox/main
refactor(frontend): improve mobile overflow handling
2026-07-03 12:54:01 +08:00
ZheFox e72e25c59c refactor(frontend): improve mobile overflow handling 2026-07-03 12:53:22 +08:00
ZheFox 1b0440481b Merge pull request #658 from zhefox/main
修复管理端额度显示、节点表格显示与移动端滚动问题
2026-07-03 12:49:15 +08:00
ZheFox 26d85681f0 refactor(frontend): improve mobile overflow and proxy node table 2026-07-03 12:25:53 +08:00
ZheFox 1dcee77055 refactor(frontend): improve dialog and mobile overflow handling 2026-07-03 11:26:40 +08:00
elky 1ac16005f9 Stabilize usage worker autoscale tests 2026-07-02 17:28:25 +08:00
elky 2f1cdb6a0b Record exhausted usage failures synchronously 2026-07-02 16:08:04 +08:00
elky ac93851b2a Stabilize Gateway h2c transport test 2026-07-02 14:02:26 +08:00
elky 400b3125a4 Preserve terminal request candidate state 2026-07-02 01:40:57 +08:00
elky 2e5ff32e1a perf(frontend): 收敛导航预取并去重首屏请求
- 导航预取仅保留 pointerdown 触发,移除 mouseenter/focus,避免鼠标划过误触发
- 后台预取只做组件懒加载,不再预取各页业务数据,减少首屏资源争抢
- 版本状态检查增加 sessionStorage 缓存(正常 20 分钟 / 错误 5 分钟 TTL)
- fetchModules、必读公告拉取增加请求去重,避免并发重复请求
- 更新检查改用可清理的定时器,组件卸载时清理
- UsageRecordsTable 搜索防抖改为自定义实现,卸载时取消挂起 emit 并补充测试
2026-07-01 20:42:52 +08:00
elky a0f7074e59 chore: disable Redis persistence by default, document triage and policy 2026-07-01 14:15:16 +08:00
elky 7c32be46ca Mark sync usage active earlier 2026-07-01 02:21:20 +08:00
Entropy.Xu 6ed2f9bd0a fix: apply actual billing cost to wallet settlement 2026-07-01 01:12:40 +08:00
elky 778b106023 test: stabilize gateway nextest timing 2026-06-30 18:42:57 +08:00
elky f179ee72f9 chore: update gateway pressure observability 2026-06-30 17:01:39 +08:00
elky 974def5fef refactor(frontend): extract provider key identity block 2026-06-30 17:01:39 +08:00
elky e5351b7d9d refactor(frontend): extract provider key actions 2026-06-30 17:01:39 +08:00
elky ed83184d55 refactor(frontend): extract provider quota display components 2026-06-30 17:01:39 +08:00
elky 15b6606c82 refactor(frontend): extract pool key display panels 2026-06-30 17:01:39 +08:00
elky d7411a3104 refactor(frontend): extract pool header and theme toggle 2026-06-30 17:01:39 +08:00
elky 9f138d09e6 refactor(frontend): modularize i18n architecture 2026-06-30 17:01:39 +08:00
ZheFox bf29129a4b Merge pull request #654 from zhefox/main
Cancel upstream streams on client disconnect and void cancelled usage billing
2026-06-29 01:14:11 +08:00
zhefox f6293b6812 fix(usage): void cancelled usage and cancel dropped streams 2026-06-29 00:30:41 +08:00
elky 7e9424008f Add usage queue worker autoscaling 2026-06-26 14:02:57 +08:00
elky 6c5e70ccb1 fix monitoring error totals and counter health 2026-06-26 10:48:45 +08:00
elky 063834e95b Split admin operations dashboard route 2026-06-26 01:48:37 +08:00
elky c76d6b6396 Add admin operations dashboard and usage state fixes 2026-06-26 01:32:45 +08:00
elky 6f00e9fc67 Improve gateway transport and usage runtime 2026-06-25 22:36:27 +08:00
elky d336d1a7fa Improve gateway scheduling and runtime admission 2026-06-24 01:53:45 +08:00
ZheFox cf0af8fa1e Merge pull request #652 from zhefox/main
fix(usage): preserve token counts in body redaction
2026-06-23 14:40:59 +08:00
zhefox fd220b6c42 fix(usage): preserve token counts in body redaction 2026-06-23 14:38:39 +08:00
zhefox 3472bb75e7 ci: combine gateway clippy and nextest jobs 2026-06-23 14:06:43 +08:00
zhefox ba65c96c74 Merge branch 'main' of https://github.com/zhefox/Aether 2026-06-23 13:36:00 +08:00
zhefox c54b214657 ci: shard gateway tests and disable debug info in rust ci 2026-06-23 13:35:56 +08:00
ZheFox 4fcc17114f Merge pull request #651 from zhefox/main
fix(ai-formats): accept Claude context_management in responses conversion
2026-06-23 10:43:26 +08:00
zhefox deb5f55786 fix(ai-formats): clean up cross-format safety rules for Gemini requests 2026-06-23 10:30:13 +08:00
zhefox 1836c2b652 fix(ai-formats): accept Claude context_management in responses conversion 2026-06-23 10:03:55 +08:00
elky 5b7805181b perf: queue request candidate persistence 2026-06-22 02:49:17 +08:00
elky f75894acbb perf: reduce gateway db pressure under load 2026-06-22 00:08:48 +08:00
elky 541cc197c4 fix: preserve in-memory user export fallback 2026-06-22 00:08:48 +08:00
fawney19 363d1aba9a Merge pull request #615 from AAEE86/main
feat: 健康监控仪表盘与关联下钻优化,完善使用记录展示
2026-06-21 12:48:13 +08:00
fawney19 eb2cf662b7 Merge pull request #650 from stabey/pr/claude-system-responses-20260620
fix(ai-formats): preserve Claude in-message system guidance in Responses
2026-06-21 12:47:26 +08:00
elky 900f8a7163 fix(pool): allow zero cooldown settings 2026-06-21 12:20:08 +08:00
elky 61bdd304b7 Handle inactive PAT owner as invalid OAuth token 2026-06-21 11:39:20 +08:00
elky 279735ae7f Auto-size SQL pool defaults 2026-06-21 11:15:40 +08:00
elky cc2830f6ec Merge branch 'review/pr-639' 2026-06-21 10:48:49 +08:00
elky 8dbd730568 fix: respect imported oauth authorization headers 2026-06-21 02:27:06 +08:00
stabey bb6aa03485 fix(ai-formats): strip Claude billing headers from preserved guidance 2026-06-21 00:22:57 +08:00
stabey 6a22488698 fix(ai-formats): preserve Claude in-message system guidance in responses 2026-06-21 00:07:57 +08:00
elky f1c30439ff fix: preserve provider auth metadata 2026-06-20 22:11:42 +08:00
fawney19 1123095bb7 Merge pull request #624 from MMEXA/codex/fix-antigravity-oauth-quota
修复 Antigravity OAuth 导入后配额复检缺 project
2026-06-19 23:11:38 +08:00
MMEXA 938f11981d fix(ai-serving): route Antigravity auth enum through facade 2026-06-19 22:25:52 +08:00
MMEXA 6c4e730e60 修复 Antigravity OAuth 配额复检缺 project 2026-06-19 22:21:22 +08:00
elky 16584067d7 Add route-backed routing profile views 2026-06-18 02:06:34 +08:00
fawney19 6de0fe75a4 Merge pull request #641 from Kayphoon/codex/usage-cleanup-break-condition
fix(usage): align cleanup loop break conditions with candidate row count
2026-06-17 11:04:55 +08:00
fawney19 34f0913ed0 Merge pull request #645 from zhefox/main
修复 OpenAI Chat/Responses/Messages 转换兼容性并透传 Codex cyber_policy 错误
2026-06-17 11:03:29 +08:00
zhefox 5b305c64e1 fix(ai-formats): omit request tool call ids in OpenAI Responses input 2026-06-17 09:15:38 +08:00
zhefox 8ad97761e8 fix(ai-formats): preserve OpenAI Responses tool call item ids 2026-06-17 08:29:25 +08:00
zhefox 0f92ef664d fix(ai-formats): support OpenAI Responses custom tool/raw passthrough 2026-06-17 04:24:56 +08:00
zhefox 16a4fd3687 Merge branch 'main' of https://github.com/zhefox/Aether 2026-06-17 04:07:53 +08:00
zhefox 3a3fcbe46a fix(ai-formats): preserve OpenAI tool call item ids 2026-06-17 04:05:09 +08:00
zhefox 18d8ea2052 fix(ai-formats): preserve OpenAI tool call item ids 2026-06-17 04:03:40 +08:00
zhefox 6ab08f4014 fix(ai-formats): preserve Claude raw blocks, reasoning tokens, and test stack safety 2026-06-17 03:45:23 +08:00
zhefox 628a3a0d8d fix(ai-formats): support cyber policy failover and custom tool/audio passthrough 2026-06-17 02:33:14 +08:00
elky f52628e00b Handle OpenAI Responses keepalive stream events 2026-06-16 22:52:06 +08:00
zhefox f9d97ececb fix(ai-formats): ignore OpenAI Responses metadata events 2026-06-16 22:34:38 +08:00
fawney19 803e555022 Merge pull request #635 from zhefox/main
fix(gateway): 支持 OpenAI 图片编辑端点请求
2026-06-16 22:31:52 +08:00
zhefox b1bd727978 将 JSON 提示注入为 developer 输入 2026-06-16 21:40:15 +08:00
zhefox c2748dc868 忽略 OpenAI Responses keepalive 事件 2026-06-16 20:00:46 +08:00
ZheFox 302620cb94 Merge branch 'fawney19:main' into main 2026-06-16 12:17:28 +08:00
AAEE86 c255f29e98 Merge remote-tracking branch 'upstream/main' 2026-06-16 10:53:47 +08:00
Kayphoon 6d1b818414 fix(usage): align cleanup loop break conditions with candidate row count
The cleanup loop break condition used rows_affected() from the UPDATE
statement, but for rows that only had blob/audit refs (no inline
compressed body data), the UPDATE reported 0 affected rows. This caused
the loop to exit after the first batch, skipping the majority of
candidates.

Change the break condition in all 4 cleanup functions from:
  if cleaned == 0 || cleaned < batch_size
to:
  if rows.len() < batch_size

This ensures the loop continues as long as SELECT returns a full batch,
regardless of how many rows the UPDATE actually modified.

Affected functions:
- cleanup_usage_raw_body_fields
- cleanup_usage_compressed_body_fields
- cleanup_usage_header_fields
- cleanup_usage_stale_body_fields
2026-06-16 04:19:58 +08:00
ndllz 5249660e07 fix: respect oauth module disabled state 2026-06-09 18:13:07 +08:00
ndllz 84b99a641a fix: speed up usage activity heatmap render 2026-06-09 16:52:58 +08:00
AAEE86 4824e4a487 fix(frontend): 移除账号导入重复处理中提示 2026-06-09 16:40:45 +08:00
zhefox ba723ebe48 fix(usage): always use truncated body placeholder when limit exceeded 2026-06-09 09:59:10 +08:00
zhefox 04ba8cbe9e fix(gateway): support openai image accept negotiation 2026-06-08 20:40:30 +08:00
zhefox 82040bfc21 fix(gateway): support OpenAI image edit requests 2026-06-08 13:22:08 +08:00
AAEE86 85573d7980 Merge remote-tracking branch 'upstream/main' 2026-06-05 08:28:43 +08:00
AAEE86 7835840ebd feat(dashboard): 增加全站实时指标和自动刷新
- 管理员仪表盘新增全站 RPM/TPM 与在线/启用用户指标
- 合并今日请求/费用、全站 RPM/TPM、在线/启用用户卡片展示
- 在线用户按最近 5 分钟活跃请求去重统计
- 全站 RPM/TPM 按最近 60 秒请求与 Token 统计
- 新增仪表盘自动刷新按钮,开启后每 10 秒静默刷新数据
- 同步前端类型、空态占位和仪表盘测试
2026-06-02 18:16:24 +08:00
AAEE86 86f72da3d9 feat(health): 增加历史状态条指标 Tooltip
- 为健康监控时间轴返回 timeline_details 分段指标
- Hover 历史状态柱时展示总请求/成功/失败/可用率/状态
- 展示平均耗时/TTFB/速度和完整时间范围
- 修复历史状态柱 Tooltip 触发区域不可用的问题
- 补齐前端类型、详情抽屉透传和 mock 数据
2026-06-02 10:36:54 +08:00
AAEE86 d5d3f09846 refactor(health): add dashboard overview and related drill-down
- Replace health monitor tabs with a dashboard layout
- Add related health drill-down for endpoint, model, and provider cards
- Render provider health as cards and hide empty monitors
2026-06-02 00:10:59 +08:00
AAEE86 0e6fc96eb1 test(gateway): run wallet usage settlement test on larger stack
Wrap the wallet settlement usage test with the large-stack async test helper to
avoid stack overflow in the default test thread.
2026-06-01 22:40:22 +08:00
AAEE86 b052f40ffb test(gateway): run base usage body capture test on larger stack
Wrap the request_record_level=base local gateway usage test with the existing
large-stack async test helper to avoid stack overflow in the default test thread.
2026-06-01 22:23:54 +08:00
AAEE86 2aef9d2478 Refine mobile usage record metadata layout 2026-06-01 21:59:45 +08:00
AAEE86 8627a18f2e test(gateway): run local usage report test on large stack
Wrap the local OpenAI chat sync usage-reporting test in the existing
large-stack harness to avoid stack overflows under nextest suite load.
2026-06-01 21:45:44 +08:00
AAEE86 21c478be22 fix(health): hide empty endpoint monitors
- Remove raw API format label from endpoint health cards
- Hide endpoint health cards with no requests
2026-06-01 21:24:20 +08:00
AAEE86 9d8f7d158b Refine mobile usage record details
- Move mobile usage actions into the card header
- Add compact user/provider metadata line on mobile
- Preserve hidden unknown toggle and auto refresh controls
2026-06-01 21:13:10 +08:00
AAEE86 c1649fe837 refactor(health): consolidate monitor components 2026-06-01 20:49:38 +08:00
AAEE86 d3c8317939 fix(health): align model health card layout 2026-06-01 18:54:56 +08:00
AAEE86 7ffe33f867 feat(health): refine health monitor metrics
- add TPS to model and provider health payloads

- exclude user-cancelled 499 requests from health statistics

- update model/provider health cards with average latency, average TTFB, TPS, and availability
2026-06-01 18:34:51 +08:00
1875 changed files with 306082 additions and 68656 deletions
+25 -55
View File
@@ -58,63 +58,33 @@ ADMIN_USERNAME=admin123456
# ==================== 可选配置(有默认值) ====================
# 可信反向代理 IP/CIDR,只有这些来源发送的 X-Real-IP / X-Forwarded-For 会被采用。
# 默认仅信任本机回环代理:127.0.0.0/8,::1/128。
# Docker/Nginx 位于独立容器时,请按实际容器网络设置,例如:172.16.0.0/12。
# AETHER_TRUSTED_PROXY_CIDRS=127.0.0.0/8,::1/128,172.16.0.0/12
# docker compose 下 app 启动前自动执行 pending migration/backfill(默认 true)
# AETHER_GATEWAY_AUTO_PREPARE_DATABASE=true
# 管理后台更新策略:
# - systemd/二进制部署使用 self:下载 GitHub Release 包,校验 SHA256 后切换 current 并重启。
# - Docker Compose 使用 docker:后台只提示版本,实际更新请在 compose 目录执行 ./update.sh。
# - 源码/本地构建使用 manual:手动拉取源码或下载 release。
# Compose 默认把持久化文件放在 ./datas/{postgres,mysql,sqlite,redis},日志放在 ./logs。
# 分布式/多节点部署不要使用 ./datas 作为共享数据目录;应使用外部共享 Postgres/MySQL 和 Redis。
# 多节点不要从管理后台一键更新单个节点,应使用镜像滚动更新、systemd 分批发布或外部编排。
# AETHER_BASE_DIR=/opt/aether
# AETHER_UPDATE_STRATEGY=docker
# AETHER_DOCKER_UPDATE_COMMAND=./update.sh
# AETHER_GATEWAY_DEPLOYMENT_TOPOLOGY=single-node
# AETHER_GATEWAY_NODE_ROLE=all
# Docker Compose 默认强制把应用日志输出到 stdout/stderr,并由 Docker 轮转日志。
# 如需文件日志,需要在 compose 里把 AETHER_LOG_DESTINATION 改成 file 或 both,
# 并把容器用户可写目录挂载到 /opt/aether/logs。
# AETHER_LOG_DESTINATION=stdout
# AETHER_LOG_FORMAT=pretty
# AETHER_LOG_DIR=/opt/aether/logs
# 服务器访问 GitHub 需要代理时可配置;也兼容 UPDATE_PROXY_URL / HTTPS_PROXY / ALL_PROXY / HTTP_PROXY。
# 如果 Aether 跑在 Docker 容器里,想走宿主机代理时请写 host.docker.internal,不要写 127.0.0.1。
# AETHER_UPDATE_PROXY_URL=http://host.docker.internal:7890
# 共享出口触发 GitHub API 限流时可配置只读 token;也兼容 GITHUB_TOKEN / GH_TOKEN。
# AETHER_UPDATE_GITHUB_TOKEN=
# 下载超时控制:总超时默认 600 秒;连续无响应/无数据默认 30 秒。
# AETHER_UPDATE_DOWNLOAD_TIMEOUT_SECS=600
# AETHER_UPDATE_DOWNLOAD_IDLE_TIMEOUT_SECS=30
# 本地联调后台在线更新(配合 docker-compose.release-local.yml):
# 会用当前源码构建 release-layout 测试镜像,并伪装成较旧版本以触发升级入口。
# AETHER_RELEASE_LOCAL_VERSION=v0.7.0
# AETHER_RELEASE_LOCAL_PORT=18085
# LOCAL_RELEASE_APP_IMAGE=aether-app:release-local
# PostgreSQL 连接池配置(默认按 CPU 自动计算;正式高并发环境可显式预算)
# AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS=12
# AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS=80
# AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS=2048
# AETHER_GATEWAY_REQUEST_BODY_BUFFER_BUDGET_MB=256
# AETHER_GATEWAY_REQUEST_BODY_READ_TIMEOUT_MS=120000
# 可选的 Payload 上限(MiB);默认及 0 均表示不限制。
# AETHER_MAX_REQUEST_BODY_MB=0
# AETHER_GATEWAY_SECURITY_CACHE_TTL_MS=1000
# AETHER_MAX_REDACTED_SYNC_RESPONSE_BODY_MB=0
# AETHER_MAX_INTERNAL_BUFFERED_BODY_MB=0
# AETHER_TUNNEL_NODE_STATUS_QUEUE_CAPACITY=1024
# PostgreSQL 连接池配置(默认适合单实例/小型部署;高并发可按需调大)
# 推荐计算方式(单实例):
# MAX = CPU 核数 × 10(AI 网关偏 IO 等待,可激进些;纯 OLTP 用 × 4)
# MIN = MAX × 0.2(保留常驻连接应对突发流量,避免冷启动握手开销)
# 多实例部署时请按 实例数 × MAX 控制总和,PG 端 max_connections 至少为该总和 + 20 余量
# AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS=4
# AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS=20
# AETHER_GATEWAY_DATA_POSTGRES_STATEMENT_CACHE_CAPACITY=100
# AETHER_GATEWAY_DATA_POSTGRES_ACQUIRE_TIMEOUT_MS=3000
# PostgreSQL 性能调优(默认值适合 2核4GB 机器,按实际配置覆盖)
# 参考:shared_buffers ≈ 可用内存 25%,effective_cache_size ≈ 可用内存 50-75%
# POSTGRES_SHM_SIZE 控制 Docker 容器 /dev/shm;仪表盘统计等并行查询会使用它。
# work_mem 是每个连接每个排序操作的内存,不要设太大(并发数 × work_mem 是实际占用)
# | 系统内存 | shared_buffers | effective_cache_size | work_mem |
# | 2GB | 256MB | 768MB | 4MB |
# | 4GB | 1GB | 3GB | 16MB |
# | 8GB | 2GB | 6GB | 16MB |
# | 16GB | 4GB | 12GB | 32MB |
# | 32GB+ | 8GB | 24GB | 32MB |
# POSTGRES_SHARED_BUFFERS=1GB
# POSTGRES_EFFECTIVE_CACHE_SIZE=3GB
# POSTGRES_SHM_SIZE=512mb
# PostgreSQL 容器调优:docker-compose.yml 已内置通用默认值,通常不用配置。
# 只有在 Postgres 独占大内存、或压测显示 DB 缓存/排序/维护任务成为瓶颈时再覆盖。
# 内置默认:shared_buffers=1GB, effective_cache_size=3GB, shm_size=512mb,
# work_mem=16MB, maintenance_work_mem=256MB。
# POSTGRES_SHARED_BUFFERS=8GB
# POSTGRES_EFFECTIVE_CACHE_SIZE=24GB
# POSTGRES_SHM_SIZE=2gb
# POSTGRES_WORK_MEM=16MB
# POSTGRES_MAINTENANCE_WORK_MEM=256MB
# POSTGRES_MAINTENANCE_WORK_MEM=1GB
+1
View File
@@ -131,6 +131,7 @@ jobs:
aether-tunnel-*.tar.gz
aether-tunnel-*.zip
if-no-files-found: error
retention-days: 1
release:
needs: build
+171 -18
View File
@@ -25,6 +25,8 @@ concurrency:
env:
CARGO_INCREMENTAL: 0
CARGO_PROFILE_DEV_DEBUG: 0
CARGO_PROFILE_TEST_DEBUG: 0
CARGO_TERM_COLOR: always
jobs:
@@ -68,7 +70,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo clippy -p aether-gateway --all-targets -- -D warnings
run: cargo clippy -p aether-gateway --lib --bins --examples -- -D warnings
- name: Show sccache stats
if: always()
@@ -136,7 +138,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo clippy --workspace --exclude aether-gateway --exclude aether-data --all-targets -- -D warnings
run: cargo clippy --workspace --exclude aether-gateway --exclude aether-data --exclude aether-integration-tests --all-targets -- -D warnings
- name: Show sccache stats
if: always()
@@ -184,15 +186,27 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Setup mold
uses: rui314/setup-mold@v1
- name: Install nextest
uses: taiki-e/install-action@nextest
- name: Test
- name: Test lib
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
RUST_MIN_STACK: "16777216"
run: cargo nextest run -p aether-gateway
RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
run: cargo nextest run -p aether-gateway --lib
- name: Test bin
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
RUST_MIN_STACK: "16777216"
RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
run: cargo nextest run -p aether-gateway --bin aether-gateway
- name: Show sccache stats
if: always()
@@ -238,6 +252,45 @@ jobs:
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
check_data_features:
name: Check (Data Feature - ${{ matrix.feature }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
feature:
- postgres
- mysql
- sqlite
- all-drivers
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Check selected data driver
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo check -p aether-data --no-default-features --features ${{ matrix.feature }}
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
test_rest:
name: Test (Workspace Rest)
runs-on: ubuntu-latest
@@ -266,7 +319,79 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run --workspace --exclude aether-gateway --exclude aether-data
run: cargo nextest run --workspace --exclude aether-gateway --exclude aether-data --exclude aether-integration-tests
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
test_data_adapters:
name: Test (Data Adapter - ${{ matrix.package }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
package:
- aether-data-postgres
- aether-data-mysql
- aether-data-sqlite
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Install nextest
uses: taiki-e/install-action@nextest
- name: Test adapter
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run -p ${{ matrix.package }}
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
check_integration_scenarios:
name: Test (Integration Scenarios)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Test scenario binaries
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo test -p aether-integration-tests --bins
- name: Show sccache stats
if: always()
@@ -281,14 +406,20 @@ jobs:
needs:
- test_gateway
- test_data
- check_data_features
- test_rest
- test_data_adapters
- check_integration_scenarios
if: ${{ always() }}
steps:
- name: Verify test jobs
run: |
if [ "${{ needs.test_gateway.result }}" != "success" ] || \
[ "${{ needs.test_data.result }}" != "success" ] || \
[ "${{ needs.test_rest.result }}" != "success" ]; then
[ "${{ needs.check_data_features.result }}" != "success" ] || \
[ "${{ needs.test_rest.result }}" != "success" ] || \
[ "${{ needs.test_data_adapters.result }}" != "success" ] || \
[ "${{ needs.check_integration_scenarios.result }}" != "success" ]; then
echo "Tests failed"
exit 1
fi
@@ -318,7 +449,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo test -p aether-data sqlite --lib
run: cargo test -p aether-data --all-features sqlite --lib
- name: Show sccache stats
if: always()
@@ -362,26 +493,48 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Add PostgreSQL server binaries to PATH
run: echo "$(pg_config --bindir)" >> "$GITHUB_PATH"
- name: Run Postgres migration smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data postgres_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features postgres_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
- name: Run Postgres provider metadata migration smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data --all-features postgres_provider_upstream_metadata_migration_preserves_json_when_url_is_set --lib -- --nocapture
- name: Run Postgres API key lifecycle tests
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_REQUIRE_LOCAL_POSTGRES_TESTS: "true"
run: |
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_request_candidates_preserve_deleted_api_key_identity --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_request_candidate_migration_decouples_legacy_api_key_foreign_key --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_stats_daily_api_key_migration_decouples_legacy_foreign_key --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_expired_api_key_cleanup_preserves_historical_identity --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_api_key_leaderboard_user_filter_preserves_aggregate_history --lib -- --exact --nocapture
- name: Run Postgres core export smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data postgres_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features postgres_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
- name: Run SQLite-to-Postgres import smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data sqlite_core_export_reads_migrated_database_rows --lib -- --nocapture
run: cargo test -p aether-data --all-features sqlite_core_export_reads_migrated_database_rows --lib -- --nocapture
- name: Show sccache stats
if: always()
@@ -431,56 +584,56 @@ jobs:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
- name: Run MySQL usage write smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_usage_write_repository_upserts_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_usage_write_repository_upserts_and_flushes_counters_when_url_is_set --lib -- --nocapture
- name: Run MySQL usage read smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_usage_read_repository_reads_usage_contract_views_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_usage_read_repository_reads_usage_contract_views_when_url_is_set --lib -- --nocapture
- name: Run MySQL provider catalog smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_provider_catalog_repository_round_trips_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_provider_catalog_repository_round_trips_when_url_is_set --lib -- --nocapture
- name: Run MySQL core export smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
- name: Run MySQL wallet read smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_wallet_read_repository_reads_wallet_contract_views --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_wallet_read_repository_reads_wallet_contract_views --lib -- --nocapture
- name: Run MySQL wallet daily usage aggregation smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_wallet_daily_usage_aggregation_uses_settlement_wallets_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_wallet_daily_usage_aggregation_uses_settlement_wallets_when_url_is_set --lib -- --nocapture
- name: Run MySQL stats aggregation smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_stats_aggregation_runs_after_mysql_migrations_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_stats_aggregation_runs_after_mysql_migrations_when_url_is_set --lib -- --nocapture
- name: Show sccache stats
if: always()
Generated
+359 -15
View File
@@ -69,6 +69,14 @@ dependencies = [
"uuid",
]
[[package]]
name = "aether-admission-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-ai-formats"
version = "0.1.0"
@@ -95,6 +103,7 @@ dependencies = [
"aether-pool-core",
"aether-scheduler-core",
"async-trait",
"base64 0.22.1",
"http",
"serde",
"serde_json",
@@ -155,7 +164,9 @@ dependencies = [
"aether-ai-formats",
"aether-cache",
"aether-data-contracts",
"aether-data-query",
"aether-data-mysql",
"aether-data-postgres",
"aether-data-sqlite",
"aether-wallet",
"async-trait",
"chrono",
@@ -183,7 +194,48 @@ dependencies = [
"chrono",
"serde",
"serde_json",
"sha2",
"thiserror 2.0.18",
"tokio",
]
[[package]]
name = "aether-data-mysql"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"aether-data-contracts",
"aether-data-query",
"async-trait",
"chrono",
"chrono-tz",
"flate2",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tracing",
"uuid",
]
[[package]]
name = "aether-data-postgres"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"aether-data-contracts",
"aether-data-query",
"async-trait",
"chrono",
"chrono-tz",
"flate2",
"futures-util",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tracing",
"uuid",
]
[[package]]
@@ -203,6 +255,25 @@ dependencies = [
"toml",
]
[[package]]
name = "aether-data-sqlite"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"aether-data-contracts",
"aether-data-query",
"async-trait",
"chrono",
"chrono-tz",
"flate2",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tracing",
"uuid",
]
[[package]]
name = "aether-dispatch-core"
version = "0.1.0"
@@ -229,6 +300,11 @@ dependencies = [
"aether-data",
"aether-data-contracts",
"aether-dispatch-core",
"aether-gateway-control",
"aether-gateway-execution",
"aether-gateway-frontdoor",
"aether-gateway-tunnel",
"aether-gateway-workers",
"aether-http",
"aether-model-fetch",
"aether-oauth",
@@ -240,7 +316,7 @@ dependencies = [
"aether-runtime-state",
"aether-scheduler-core",
"aether-task-runtime",
"aether-testkit",
"aether-test-support",
"aether-usage-runtime",
"aether-video-tasks-core",
"aether-wallet",
@@ -258,7 +334,11 @@ dependencies = [
"futures-util",
"hmac",
"http",
"http-body-util",
"hyper",
"hyper-util",
"ldap3",
"libc",
"md-5",
"object_store",
"parking_lot",
@@ -270,9 +350,12 @@ dependencies = [
"serde_json",
"sha1",
"sha2",
"socket2 0.6.3",
"sqlx",
"sysinfo",
"tar",
"thiserror 2.0.18",
"tikv-jemalloc-sys",
"tikv-jemallocator",
"tokio",
"tokio-util",
@@ -288,6 +371,63 @@ dependencies = [
"zstd",
]
[[package]]
name = "aether-gateway-control"
version = "0.1.0"
dependencies = [
"http",
]
[[package]]
name = "aether-gateway-execution"
version = "0.1.0"
dependencies = [
"aether-contracts",
"bytes",
"serde_json",
]
[[package]]
name = "aether-gateway-frontdoor"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"axum",
"bytes",
"futures-util",
"http",
"serde_json",
"tokio",
"tower",
"tracing",
"tracing-subscriber",
"uuid",
]
[[package]]
name = "aether-gateway-tunnel"
version = "0.1.0"
dependencies = [
"aether-admission-core",
"aether-contracts",
"base64 0.22.1",
"bytes",
"http",
"serde",
"serde_json",
]
[[package]]
name = "aether-gateway-workers"
version = "0.1.0"
dependencies = [
"aether-runtime-state",
"aether-task-runtime",
"aether-test-support",
"tokio",
"tracing",
]
[[package]]
name = "aether-http"
version = "0.1.0"
@@ -296,6 +436,48 @@ dependencies = [
"serde",
]
[[package]]
name = "aether-integration-tests"
version = "0.1.0"
dependencies = [
"aether-contracts",
"aether-data",
"aether-data-contracts",
"aether-gateway",
"aether-runtime-state",
"aether-testkit",
"async-stream",
"axum",
"futures-util",
"http",
"reqwest",
"serde",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tokio-tungstenite 0.28.0",
]
[[package]]
name = "aether-loadtools"
version = "0.1.0"
dependencies = [
"aether-http",
"aether-runtime",
"aether-runtime-state",
"aether-test-support",
"bytes",
"futures-util",
"http",
"libc",
"reqwest",
"serde",
"serde_json",
"sysinfo",
"tokio",
]
[[package]]
name = "aether-model-fetch"
version = "0.1.0"
@@ -341,12 +523,21 @@ dependencies = [
"serde_json",
]
[[package]]
name = "aether-provider-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-provider-pool"
version = "0.1.0"
dependencies = [
"aether-data-contracts",
"aether-pool-core",
"aether-provider-transport",
"serde_json",
"url",
"uuid",
@@ -366,6 +557,9 @@ dependencies = [
"async-trait",
"axum",
"base64 0.22.1",
"chrono",
"crypto_box",
"ed25519-dalek",
"http",
"regex",
"reqwest",
@@ -440,11 +634,20 @@ dependencies = [
"sha2",
]
[[package]]
name = "aether-task-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-task-runtime"
version = "0.1.0"
dependencies = [
"aether-runtime",
"aether-task-core",
"serde",
"serde_json",
"tokio",
@@ -452,30 +655,25 @@ dependencies = [
"tracing",
]
[[package]]
name = "aether-test-support"
version = "0.1.0"
dependencies = [
"tokio",
]
[[package]]
name = "aether-testkit"
version = "0.1.0"
dependencies = [
"aether-contracts",
"aether-data",
"aether-data-contracts",
"aether-gateway",
"aether-http",
"aether-loadtools",
"aether-runtime",
"aether-runtime-state",
"async-stream",
"axum",
"bytes",
"futures-util",
"http",
"libc",
"reqwest",
"serde",
"serde_json",
"sqlx",
"sysinfo",
"tokio",
"tokio-tungstenite 0.28.0",
]
[[package]]
@@ -484,6 +682,7 @@ version = "0.3.16"
dependencies = [
"aether-contracts",
"aether-gateway",
"aether-gateway-tunnel",
"aether-http",
"aether-runtime",
"aether-runtime-state",
@@ -522,6 +721,14 @@ dependencies = [
"webpki-roots 0.26.11",
]
[[package]]
name = "aether-usage-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-usage-runtime"
version = "0.1.0"
@@ -533,6 +740,7 @@ dependencies = [
"aether-runtime-state",
"async-trait",
"base64 0.22.1",
"futures-util",
"serde",
"serde_json",
"tokio",
@@ -906,6 +1114,19 @@ dependencies = [
"zeroize",
]
[[package]]
name = "bigdecimal"
version = "0.4.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4d6867f1565b3aad85681f1015055b087fcfd840d6aeee6eee7f2da317603695"
dependencies = [
"autocfg",
"libm",
"num-bigint",
"num-integer",
"num-traits",
]
[[package]]
name = "bindgen"
version = "0.72.1"
@@ -954,6 +1175,15 @@ dependencies = [
"serde_core",
]
[[package]]
name = "blake2"
version = "0.10.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "46502ad458c9a52b69d4d4d32775c788b7a1b85e8bc9d482d92250fc0e3f8efe"
dependencies = [
"digest",
]
[[package]]
name = "block-buffer"
version = "0.10.4"
@@ -1135,6 +1365,7 @@ checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad"
dependencies = [
"crypto-common",
"inout",
"zeroize",
]
[[package]]
@@ -1419,6 +1650,36 @@ dependencies = [
"typenum",
]
[[package]]
name = "crypto_box"
version = "0.9.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "16182b4f39a82ec8a6851155cc4c0cda3065bb1db33651726a29e1951de0f009"
dependencies = [
"aead",
"blake2",
"crypto_secretbox",
"curve25519-dalek",
"salsa20",
"subtle",
"zeroize",
]
[[package]]
name = "crypto_secretbox"
version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b9d6cf87adf719ddf43a805e92c6870a531aedda35ff640442cbaf8674e141e1"
dependencies = [
"aead",
"cipher",
"generic-array",
"poly1305",
"salsa20",
"subtle",
"zeroize",
]
[[package]]
name = "csscolorparser"
version = "0.6.2"
@@ -1438,6 +1699,33 @@ dependencies = [
"cipher",
]
[[package]]
name = "curve25519-dalek"
version = "4.1.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "97fb8b7c4503de7d6ae7b42ab72a5a59857b4c937ec27a3d4539dba95b5ab2be"
dependencies = [
"cfg-if",
"cpufeatures",
"curve25519-dalek-derive",
"digest",
"fiat-crypto",
"rustc_version",
"subtle",
"zeroize",
]
[[package]]
name = "curve25519-dalek-derive"
version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f46882e17999c6cc590af592290432be3bce0428cb0d5f8b6715e4dc7b383eb3"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.117",
]
[[package]]
name = "darling"
version = "0.23.0"
@@ -1598,6 +1886,30 @@ version = "1.0.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813"
[[package]]
name = "ed25519"
version = "2.2.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "115531babc129696a58c64a4fef0a8bf9e9698629fb97e9e40767d235cfbcd53"
dependencies = [
"pkcs8",
"signature",
]
[[package]]
name = "ed25519-dalek"
version = "2.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "70e796c081cee67dc755e1a36a0a172b897fab85fc3f6bc48307991f64e4eca9"
dependencies = [
"curve25519-dalek",
"ed25519",
"serde",
"sha2",
"subtle",
"zeroize",
]
[[package]]
name = "either"
version = "1.15.0"
@@ -1670,6 +1982,12 @@ version = "2.4.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6"
[[package]]
name = "fiat-crypto"
version = "0.2.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "28dea519a9695b9977216879a3ebfddf92f1c08c05d984f8996aecd6ecdc811d"
[[package]]
name = "filedescriptor"
version = "0.8.3"
@@ -1908,6 +2226,7 @@ checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
dependencies = [
"typenum",
"version_check",
"zeroize",
]
[[package]]
@@ -2204,6 +2523,7 @@ dependencies = [
"pin-project-lite",
"socket2 0.6.3",
"tokio",
"tower-layer",
"tower-service",
"tracing",
]
@@ -3162,6 +3482,17 @@ version = "0.2.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b4596b6d070b27117e987119b4dac604f3c58cfb0b191112e24771b2faeac1a6"
[[package]]
name = "poly1305"
version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8159bd90725d2df49889a078b54f4f79e87f1f8a8444194cdca81d38f5393abf"
dependencies = [
"cpufeatures",
"opaque-debug",
"universal-hash",
]
[[package]]
name = "polyval"
version = "0.6.2"
@@ -3790,6 +4121,15 @@ version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f"
[[package]]
name = "salsa20"
version = "0.10.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "97a22f5af31f73a954c10289c93e8a50cc23d971e80ee446f1f6f7137a088213"
dependencies = [
"cipher",
]
[[package]]
name = "schannel"
version = "0.1.29"
@@ -4120,6 +4460,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ee6798b1838b6a0f69c007c133b8df5866302197e404e8b6ee8ed3e3a5e68dc6"
dependencies = [
"base64 0.22.1",
"bigdecimal",
"bytes",
"chrono",
"crc",
@@ -4196,6 +4537,7 @@ checksum = "aa003f0038df784eb8fecbbac13affe3da23b45194bd57dba231c8f48199c526"
dependencies = [
"atoi",
"base64 0.22.1",
"bigdecimal",
"bitflags 2.11.0",
"byteorder",
"bytes",
@@ -4239,6 +4581,7 @@ checksum = "db58fcd5a53cf07c184b154801ff91347e4c30d17a3562a635ff028ad5deda46"
dependencies = [
"atoi",
"base64 0.22.1",
"bigdecimal",
"bitflags 2.11.0",
"byteorder",
"chrono",
@@ -4256,6 +4599,7 @@ dependencies = [
"log",
"md-5",
"memchr",
"num-bigint",
"once_cell",
"rand 0.8.5",
"serde",
+63 -27
View File
@@ -1,34 +1,52 @@
[workspace]
members = [
"apps/aether-tunnel",
"crates/aether-ai-formats",
"crates/aether-ai/formats",
"crates/aether-admin",
"crates/aether-ai-serving",
"crates/aether-admission-core",
"crates/aether-ai/serving",
"crates/aether-pool-core",
"crates/aether-provider-pool",
"crates/aether-provider/core",
"crates/aether-provider/pool",
"crates/aether-routing-core",
"crates/aether-data-contracts",
"crates/aether-data-query",
"crates/aether-data-schema",
"crates/aether-data/contracts",
"crates/aether-data/adapters/postgres",
"crates/aether-data/adapters/mysql",
"crates/aether-data/adapters/sqlite",
"crates/aether-data/query",
"crates/aether-data/schema",
"crates/aether-dispatch-core",
"crates/aether-cache",
"crates/aether-billing",
"crates/aether-wallet",
"crates/aether-crypto",
"crates/aether-contracts",
"crates/aether-data",
"crates/aether-data/runtime",
"crates/aether-model-fetch",
"crates/aether-oauth",
"crates/aether-provider-transport",
"crates/aether-provider/transport",
"crates/aether-scheduler-core",
"crates/aether-runtime-state",
"crates/aether-task-runtime",
"crates/aether-usage-runtime",
"crates/aether-runtime/state",
"crates/aether-task/runtime",
"crates/aether-task/core",
"crates/aether-gateway/frontdoor",
"crates/aether-gateway/control",
"crates/aether-gateway/execution",
"crates/aether-gateway/workers",
"crates/aether-gateway/tunnel",
"crates/aether-testing/loadtools",
"crates/aether-testing/integration",
"crates/aether-usage/core",
"crates/aether-testing/support",
"crates/aether-usage/runtime",
"crates/aether-video-tasks-core",
"apps/aether-gateway",
"crates/aether-http",
"crates/aether-runtime",
"crates/aether-testkit",
"crates/aether-runtime/base",
"crates/aether-testing/testkit",
]
default-members = [
"apps/aether-gateway",
]
resolver = "2"
@@ -39,33 +57,48 @@ repository = "https://github.com/fawney19/Aether.git"
[workspace.dependencies]
aether-admin = { path = "crates/aether-admin" }
aether-ai-formats = { path = "crates/aether-ai-formats" }
aether-ai-serving = { path = "crates/aether-ai-serving" }
aether-admission-core = { path = "crates/aether-admission-core" }
aether-ai-formats = { path = "crates/aether-ai/formats" }
aether-ai-serving = { path = "crates/aether-ai/serving" }
aether-pool-core = { path = "crates/aether-pool-core" }
aether-provider-pool = { path = "crates/aether-provider-pool" }
aether-provider-core = { path = "crates/aether-provider/core" }
aether-provider-pool = { path = "crates/aether-provider/pool" }
aether-routing-core = { path = "crates/aether-routing-core" }
aether-data-contracts = { path = "crates/aether-data-contracts" }
aether-data-query = { path = "crates/aether-data-query" }
aether-data-schema = { path = "crates/aether-data-schema" }
aether-data-contracts = { path = "crates/aether-data/contracts" }
aether-data-postgres = { path = "crates/aether-data/adapters/postgres" }
aether-data-mysql = { path = "crates/aether-data/adapters/mysql" }
aether-data-sqlite = { path = "crates/aether-data/adapters/sqlite" }
aether-data-query = { path = "crates/aether-data/query" }
aether-data-schema = { path = "crates/aether-data/schema" }
aether-dispatch-core = { path = "crates/aether-dispatch-core" }
aether-cache = { path = "crates/aether-cache" }
aether-billing = { path = "crates/aether-billing" }
aether-wallet = { path = "crates/aether-wallet" }
aether-crypto = { path = "crates/aether-crypto" }
aether-contracts = { path = "crates/aether-contracts" }
aether-data = { path = "crates/aether-data" }
aether-data = { path = "crates/aether-data/runtime" }
aether-model-fetch = { path = "crates/aether-model-fetch" }
aether-oauth = { path = "crates/aether-oauth" }
aether-provider-transport = { path = "crates/aether-provider-transport" }
aether-provider-transport = { path = "crates/aether-provider/transport" }
aether-scheduler-core = { path = "crates/aether-scheduler-core" }
aether-runtime-state = { path = "crates/aether-runtime-state" }
aether-task-runtime = { path = "crates/aether-task-runtime" }
aether-usage-runtime = { path = "crates/aether-usage-runtime" }
aether-runtime-state = { path = "crates/aether-runtime/state" }
aether-task-runtime = { path = "crates/aether-task/runtime" }
aether-task-core = { path = "crates/aether-task/core" }
aether-gateway-frontdoor = { path = "crates/aether-gateway/frontdoor" }
aether-gateway-control = { path = "crates/aether-gateway/control" }
aether-gateway-execution = { path = "crates/aether-gateway/execution" }
aether-gateway-workers = { path = "crates/aether-gateway/workers" }
aether-gateway-tunnel = { path = "crates/aether-gateway/tunnel" }
aether-loadtools = { path = "crates/aether-testing/loadtools" }
aether-integration-tests = { path = "crates/aether-testing/integration" }
aether-test-support = { path = "crates/aether-testing/support" }
aether-usage-core = { path = "crates/aether-usage/core" }
aether-usage-runtime = { path = "crates/aether-usage/runtime" }
aether-video-tasks-core = { path = "crates/aether-video-tasks-core" }
aether-gateway = { path = "apps/aether-gateway" }
aether-http = { path = "crates/aether-http" }
aether-runtime = { path = "crates/aether-runtime" }
aether-testkit = { path = "crates/aether-testkit" }
aether-runtime = { path = "crates/aether-runtime/base" }
aether-testkit = { path = "crates/aether-testing/testkit" }
aes = "0.8"
aes-gcm = "0.10"
async-stream = "0.3"
@@ -77,6 +110,8 @@ bytes = "1"
cbc = "0.1"
chrono = { version = "0.4", features = ["serde"] }
chrono-tz = "0.10"
crypto_box = { version = "0.9", features = ["seal"] }
ed25519-dalek = { version = "2.2", features = ["pkcs8"] }
flate2 = "1"
futures-util = "0.3"
hmac = "0.12"
@@ -92,8 +127,9 @@ serde = { version = "1", features = ["derive"] }
serde_json = { version = "1", features = ["preserve_order"] }
serde_path_to_error = "0.1"
sha2 = "0.10"
socket2 = "0.6"
tar = "0.4"
sqlx = { version = "0.8", default-features = false, features = ["postgres", "mysql", "sqlite", "runtime-tokio-rustls", "chrono"] }
sqlx = { version = "0.8", default-features = false, features = ["runtime-tokio-rustls", "chrono"] }
thiserror = "2"
tokio = { version = "1", features = ["macros", "net", "rt-multi-thread", "signal", "sync", "time"] }
tokio-util = { version = "0.7", features = ["codec", "io-util"] }
+2 -2
View File
@@ -62,7 +62,7 @@ COPY --from=gateway-planner /build/recipe.json ./recipe.json
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-local,target=/build/target,sharing=locked \
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --recipe-path recipe.json
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --features jemalloc --recipe-path recipe.json
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
@@ -71,7 +71,7 @@ RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-local,target=/build/target,sharing=locked \
set -eux; \
cargo build --release --locked -p aether-gateway --bin aether-gateway; \
cargo build --release --locked -p aether-gateway --bin aether-gateway --features jemalloc; \
cp target/release/aether-gateway /tmp/aether-gateway
# ==================== 最小运行时打包 ====================
+2 -2
View File
@@ -60,7 +60,7 @@ COPY --from=gateway-planner /build/recipe.json ./recipe.json
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-release-local,target=/build/target,sharing=locked \
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --recipe-path recipe.json
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --features jemalloc --recipe-path recipe.json
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
@@ -68,7 +68,7 @@ COPY crates/ ./crates/
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-release-local,target=/build/target,sharing=locked \
cargo build --release --locked -p aether-gateway && \
cargo build --release --locked -p aether-gateway --features jemalloc && \
cp target/release/aether-gateway /tmp/aether-gateway
# ==================== 最小运行时打包 ====================
+11 -2
View File
@@ -142,8 +142,17 @@ Aether Tunnel 是配套的正向代理节点,部署在海外 VPS 上,为墙
- `APP_PORT`:`aether-gateway` 唯一监听端口,固定绑定 `0.0.0.0:${APP_PORT}`
- `DATABASE_URL`:数据库连接串;SQLite 例如 `sqlite:///opt/aether/data/aether.db`,Postgres 例如 `postgresql://postgres:aether@postgres:5432/aether`
- `AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS` / `AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS`:数据库连接池手动覆盖值;未配置时会自动推导,SQLite 固定 `1/1`,Postgres/MySQL 按 CPU 核心数计算并默认封顶 `100`
- `AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS`:单实例请求并发上限;未配置时按 CPU 自动推导(基础范围 `512-65536`),低文件描述符预算时会进一步下调
- `AETHER_GATEWAY_REQUEST_BODY_BUFFER_BUDGET_MB`:单实例同时读取和解压请求体的加权内存预算,默认 `256MB`
- `AETHER_GATEWAY_REQUEST_BODY_READ_TIMEOUT_MS`:请求体完整读取超时,默认 `120000ms`
- `AETHER_MAX_REQUEST_BODY_MB`:可选的单请求解压后请求体上限;未配置或设为 `0` 时不限制
- `AETHER_MAX_INTERNAL_BUFFERED_BODY_MB`:可选的 heartbeat、管理探测等内部整包响应体上限;未配置或设为 `0` 时不限制
- `AETHER_TUNNEL_NODE_STATUS_QUEUE_CAPACITY`:隧道节点状态上报队列容量,默认 `1024`;满载时拒绝新事件,避免控制面故障导致无界内存增长
- `AETHER_GATEWAY_SECURITY_CACHE_TTL_MS`:IP 黑白名单本地缓存时间,默认 `1000ms`,写操作会主动失效相关缓存
- `AETHER_MAX_REDACTED_SYNC_RESPONSE_BODY_MB`:可选的 PII 恢复同步响应缓冲上限;未配置或设为 `0` 时不限制
- `REDIS_URL`:Redis 连接串;仅 Postgres + Redis 的 Docker Compose 部署需要配置
- `AETHER_RUNTIME_BACKEND=memory|redis`:运行时缓存/协调后端。SQLite 默认用 `memory`,不会连接 Redis
- `AETHER_RUNTIME_BACKEND=memory|redis`:运行时缓存/协调后端。SQLite 默认用 `memory`,不会连接 Redis;多节点部署和需要跨 gateway 重启恢复 OpenAI Responses continuation history 的部署必须使用共享 Redis
- `AETHER_GATEWAY_AUTO_PREPARE_DATABASE`:常规启动前自动执行挂起的 schema migration 和 backfill;仓库自带的 `docker-compose.yml` 默认开启
- `JWT_SECRET_KEY` / `ENCRYPTION_KEY`:认证和敏感数据加密所需密钥
- `API_KEY_PREFIX`:用户和管理员新建 API Key 时使用的前缀,默认 `sk`
@@ -168,4 +177,4 @@ Aether Tunnel 是配套的正向代理节点,部署在海外 VPS 上,为墙
## Star History
[![Star History Chart](https://api.star-history.com/svg?repos=fawney19/Aether&type=Date)](https://star-history.com/#fawney19/Aether&Date)
[![Star History Chart](https://api.star-history.com/svg?repos=fawney19/Aether&type=date&legend=top-left)](https://www.star-history.com/?repos=fawney19%2FAether&type=date&legend=top-left)
+26 -4
View File
@@ -6,6 +6,16 @@ license.workspace = true
repository.workspace = true
description = "Rust ingress gateway for Aether phase 3a transparent proxy"
[features]
default = []
jemalloc = [
"dep:tikv-jemallocator",
"dep:tikv-jemalloc-sys",
"tikv-jemallocator/stats",
"tikv-jemalloc-sys/stats",
]
testkit = []
[dependencies]
aether-admin.workspace = true
aether-ai-formats.workspace = true
@@ -14,10 +24,15 @@ aether-billing.workspace = true
aether-cache.workspace = true
aether-contracts.workspace = true
aether-crypto.workspace = true
aether-data.workspace = true
aether-data = { workspace = true, features = ["all-drivers"] }
aether-data-contracts.workspace = true
aether-dispatch-core.workspace = true
aether-gateway-frontdoor.workspace = true
aether-gateway-control.workspace = true
aether-gateway-execution.workspace = true
aether-http.workspace = true
aether-gateway-workers.workspace = true
aether-gateway-tunnel.workspace = true
aether-model-fetch.workspace = true
aether-oauth.workspace = true
aether-pool-core.workspace = true
@@ -46,7 +61,11 @@ flate2.workspace = true
futures-util.workspace = true
hmac.workspace = true
http.workspace = true
http-body-util = "0.1"
hyper = { version = "1", features = ["client", "server", "http1", "http2"] }
hyper-util = { version = "0.1", features = ["client-legacy", "client-pool", "server-auto", "service", "tokio"] }
ldap3 = { version = "0.11", default-features = false, features = ["sync", "tls-rustls"] }
libc = "0.2"
md-5 = "0.10"
object_store.workspace = true
parking_lot = "0.12"
@@ -58,8 +77,10 @@ serde.workspace = true
serde_json.workspace = true
sha1 = "0.10"
sha2 = { workspace = true, features = ["oid"] }
socket2.workspace = true
tar.workspace = true
sqlx.workspace = true
sqlx = { workspace = true, features = ["postgres", "mysql", "sqlite", "migrate"] }
sysinfo = "0.32"
thiserror.workspace = true
tokio.workspace = true
tokio-util.workspace = true
@@ -74,8 +95,9 @@ wreq-util.workspace = true
zstd.workspace = true
[target.'cfg(not(target_env = "msvc"))'.dependencies]
tikv-jemallocator = "0.6"
tikv-jemallocator = { version = "0.6", optional = true }
tikv-jemalloc-sys = { version = "0.6", optional = true }
[dev-dependencies]
aether-testkit.workspace = true
aether-test-support.workspace = true
tracing-subscriber.workspace = true
+34 -22
View File
@@ -55,16 +55,20 @@ pub(crate) use aether_ai_formats::api::{
ExecutionRuntimeAuthContext, LocalCoreSyncErrorKind, LocalOpenAiImageSpec,
LocalSameFormatProviderFamily, LocalSameFormatProviderSpec, LocalStandardSourceFamily,
LocalStandardSourceMode, LocalStandardSpec, OpenAIChatClientEmitter,
OpenAIResponsesClientEmitter, StreamingStandardTerminalObserver,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_DECISION_ACTION,
GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OpenAIResponsesClientEmitter, StreamingStandardTerminalObserver, CLAUDE_CHAT_STREAM_PLAN_KIND,
CLAUDE_CLI_STREAM_PLAN_KIND, EXECUTION_RUNTIME_STREAM_DECISION_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_CHAT_STREAM_PLAN_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND,
OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_STREAM_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_formats::protocol::stream::CanonicalUsage as StreamingCanonicalUsage;
pub(crate) use aether_ai_formats::CODEX_RESPONSES_LITE_HEADER;
pub(crate) fn parse_direct_request_body(
parts: &http::request::Parts,
@@ -83,28 +87,36 @@ pub(crate) fn resolve_execution_runtime_stream_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
aether_ai_formats::api::resolve_execution_runtime_stream_plan_kind(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
let plan_kind =
aether_ai_formats::api::resolve_execution_runtime_stream_plan_kind_with_client_surface(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, true, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn resolve_execution_runtime_sync_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
aether_ai_formats::api::resolve_execution_runtime_sync_plan_kind(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
let plan_kind =
aether_ai_formats::api::resolve_execution_runtime_sync_plan_kind_with_client_surface(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, false, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn is_matching_stream_request(
@@ -2,6 +2,7 @@ use serde_json::Value;
use crate::ai_serving::{
maybe_build_ai_surface_stream_rewriter, AiSurfaceFinalizeError, AiSurfaceStreamRewriter,
ResponseHistoryRecord,
};
use crate::GatewayError;
@@ -24,6 +25,10 @@ impl LocalStreamRewriter<'_> {
pub(crate) fn finish(&mut self) -> Result<Vec<u8>, GatewayError> {
self.inner.finish().map_err(map_surface_error)
}
pub(crate) fn take_response_history_record(&mut self) -> Option<ResponseHistoryRecord> {
self.inner.take_response_history_record()
}
}
fn map_surface_error(error: AiSurfaceFinalizeError) -> GatewayError {
@@ -13,6 +13,7 @@ fn same_format_claude_local_stream_rewriter_sanitizes_read_input_json_delta() {
let report_context = json!({
"provider_api_format": "claude:messages",
"client_api_format": "claude:messages",
"anthropic_compatibility_profile": "claude_code_legacy",
"needs_conversion": false,
});
let mut rewriter =
@@ -25,12 +25,16 @@ fn test_decision() -> GatewayControlDecision {
route_class: Some("ai_public".to_string()),
route_family: Some("openai".to_string()),
route_kind: Some("compact".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("openai:responses:compact".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
}
}
@@ -1166,7 +1170,7 @@ fn local_finalize_handles_openai_responses_compact_cross_format_sync_response()
assert_eq!(report.report_kind, "openai_responses_compact_sync_success");
assert_eq!(
report.client_body_json.expect("client body should exist")["object"],
"response"
"response.compaction"
);
}
@@ -1222,7 +1226,7 @@ fn local_finalize_handles_openai_responses_compact_cross_format_function_call_re
.background_report
.expect("compact tool-call should downgrade to success report");
let client_body = report.client_body_json.expect("client body should exist");
assert_eq!(client_body["object"], "response");
assert_eq!(client_body["object"], "response.compaction");
assert_eq!(client_body["output"][1]["type"], "function_call");
}
@@ -1439,6 +1443,57 @@ fn local_finalize_handles_openai_responses_cross_format_stream_response_from_gem
);
}
#[test]
fn local_finalize_rejects_antigravity_usage_only_gemini_wrapper() {
let payload = GatewaySyncReportRequest {
trace_id: "trace-antigravity-empty-gemini-wrapper".to_string(),
report_kind: "gemini_chat_sync_finalize".to_string(),
report_context: Some(json!({
"client_api_format": "gemini:generate_content",
"provider_api_format": "gemini:generate_content",
"model": "gemini-3.5-flash",
"mapped_model": "gemini-3-flash-agent",
"needs_conversion": false,
"has_envelope": true,
"envelope_name": "antigravity:v1internal",
"upstream_is_stream": true,
})),
status_code: 200,
headers: BTreeMap::from([("content-type".to_string(), "application/json".to_string())]),
body_json: Some(json!({
"chunks": [{
"response": {
"responseId": "resp-usage-only",
"modelVersion": "gemini-3-flash-agent",
"usageMetadata": {
"promptTokenCount": 5528,
"totalTokenCount": 5528
}
},
"metadata": {},
"traceId": "trace-antigravity-empty-gemini-wrapper"
}],
"metadata": {
"stream": true,
"stored_chunks": 1,
"total_chunks": 1
}
})),
client_body_json: None,
body_base64: None,
telemetry: None,
};
let outcome = maybe_build_local_core_sync_finalize_response(
"trace-antigravity-empty-gemini-wrapper",
&test_decision(),
&payload,
)
.expect("local finalize should evaluate payload");
assert!(outcome.is_none());
}
#[test]
fn local_finalize_handles_openai_responses_compact_openai_family_stream_response_even_when_conversion_flagged(
) {
@@ -1738,6 +1793,93 @@ fn local_finalize_handles_openai_chat_cross_format_sync_response_from_openai_res
assert_eq!(client_body["usage"]["total_tokens"], 5);
}
#[test]
fn local_finalize_aggregates_openai_responses_capture_envelope_before_chat_conversion() {
let payload = GatewaySyncReportRequest {
trace_id: "trace-openai-chat-capture-envelope-sync-123".to_string(),
report_kind: "openai_chat_sync_finalize".to_string(),
report_context: Some(json!({
"client_api_format": "openai:chat",
"provider_api_format": "openai:responses",
"model": "gpt-5.6-luna",
"mapped_model": "gpt-5.6-luna",
"needs_conversion": true,
"has_envelope": false,
})),
status_code: 200,
headers: BTreeMap::from([("content-type".to_string(), "application/json".to_string())]),
body_json: Some(json!({
"chunks": [
{
"type": "response.output_text.delta",
"response_id": "resp_capture_gateway_123",
"output_index": 0,
"content_index": 0,
"delta": "Gateway "
},
{
"type": "response.output_text.done",
"response_id": "resp_capture_gateway_123",
"output_index": 0,
"content_index": 0,
"text": "Gateway capture"
},
{
"type": "response.completed",
"response": {
"id": "resp_capture_gateway_123",
"object": "response",
"status": "completed",
"model": "gpt-5.6-luna",
"output": [],
"usage": {
"input_tokens": 2,
"output_tokens": 3,
"total_tokens": 5
}
}
}
],
"metadata": {}
})),
client_body_json: None,
body_base64: None,
telemetry: None,
};
let outcome = maybe_build_local_core_sync_finalize_response(
"trace-openai-chat-capture-envelope-sync-123",
&test_decision(),
&payload,
)
.expect("capture envelope finalize should succeed")
.expect("capture envelope finalize should match");
let report = outcome
.background_report
.expect("capture envelope conversion should produce a success report");
assert_eq!(report.report_kind, "openai_chat_sync_success");
let provider_body = report
.body_json
.expect("aggregated provider body should exist");
assert_eq!(provider_body["id"], "resp_capture_gateway_123");
assert_eq!(
provider_body["output"][0]["content"][0]["text"],
"Gateway capture"
);
assert!(provider_body.get("chunks").is_none());
let client_body = report
.client_body_json
.expect("converted client body should exist");
assert_eq!(
client_body["choices"][0]["message"]["content"],
"Gateway capture"
);
assert_eq!(client_body["usage"]["prompt_tokens"], 2);
assert_eq!(client_body["usage"]["completion_tokens"], 3);
assert_eq!(client_body["usage"]["total_tokens"], 5);
}
#[test]
fn local_finalize_handles_claude_chat_cross_format_sync_response_from_openai_chat() {
let payload = GatewaySyncReportRequest {
@@ -1784,12 +1926,16 @@ fn local_finalize_handles_claude_chat_cross_format_sync_response_from_openai_cha
route_class: Some("ai_public".to_string()),
route_family: Some("claude".to_string()),
route_kind: Some("chat".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("claude:messages".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
},
&payload,
)
@@ -1851,12 +1997,16 @@ fn local_finalize_handles_gemini_cli_cross_format_sync_response_from_claude_cli(
route_class: Some("ai_public".to_string()),
route_family: Some("gemini".to_string()),
route_kind: Some("cli".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("gemini:generate_content".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
},
&payload,
)
+24 -11
View File
@@ -3,6 +3,7 @@ pub(crate) mod api;
mod finalize;
mod planner;
mod pure;
mod response_history;
pub(crate) mod transport;
use axum::body::Body;
@@ -13,7 +14,9 @@ use crate::{usage::GatewaySyncReportRequest, AppState, GatewayError};
pub(crate) use self::adaptation::{
maybe_build_provider_private_stream_normalizer, ProviderPrivateStreamNormalizer,
};
pub(crate) use self::api::gemini_generate_content_response_has_visible_output;
pub(crate) use self::api::{
gemini_generate_content_response_has_visible_output, CODEX_RESPONSES_LITE_HEADER,
};
pub(crate) use self::finalize::common::LocalCoreSyncFinalizeOutcome;
pub(crate) use self::finalize::internal::{
maybe_bridge_standard_sync_json_to_stream, maybe_build_stream_response_rewriter,
@@ -48,18 +51,24 @@ pub(crate) use self::planner::{
build_standard_family_stream_plan_and_reports, build_standard_family_sync_attempt_source,
build_standard_family_sync_plan_and_reports, build_standard_stream_plan_from_decision,
build_standard_sync_plan_from_decision, candidate_auth_channel_skip_reason,
extract_pool_sticky_session_token, maybe_build_stream_decision_payload,
maybe_build_stream_plan_payload, maybe_build_sync_decision_payload,
maybe_build_sync_plan_payload, planner_is_matching_stream_request, provider_key_pool_score_id,
provider_key_pool_score_scope, read_candidate_transport_snapshot,
record_local_runtime_candidate_skip_reason,
codex_model_capabilities_for_transport, extract_pool_sticky_session_token,
maybe_build_stream_decision_payload, maybe_build_stream_plan_payload,
maybe_build_sync_decision_payload, maybe_build_sync_plan_payload,
planner_is_matching_stream_request, provider_key_pool_score_id, provider_key_pool_score_scope,
read_candidate_transport_snapshot, record_local_runtime_candidate_skip_reason,
resolve_tunnel_scheduler_affinity_context, resolve_upstream_is_stream_for_provider,
set_local_openai_chat_execution_exhausted_diagnostic,
set_local_openai_image_execution_exhausted_diagnostic, CandidateFailureDiagnostic,
CandidateFailureDiagnosticKind, EligibleLocalExecutionCandidate, GatewayAuthApiKeySnapshot,
GatewayProviderTransportSnapshot, LocalExecutionAttemptSource, LocalExecutionCandidateKind,
LocalResolvedOAuthRequestAuth, PlannerAppState, SkippedLocalExecutionCandidate,
set_local_openai_image_execution_exhausted_diagnostic, validate_final_openai_provider_request,
CandidateFailureDiagnostic, CandidateFailureDiagnosticKind, EligibleLocalExecutionCandidate,
GatewayAuthApiKeySnapshot, GatewayProviderTransportSnapshot, LocalExecutionAttemptSource,
LocalExecutionCandidateKind, LocalResolvedOAuthRequestAuth, PlannerAppState,
SkippedLocalExecutionCandidate,
};
pub(crate) use self::pure::*;
pub(crate) use self::response_history::{
hydrate_openai_response_history, persist_converted_response_history,
persist_response_history_record,
};
pub(crate) use self::transport::{
append_transport_diagnostics_to_value, build_request_trace_proxy_value,
candidate_common_transport_skip_reason, candidate_transport_pair_skip_reason,
@@ -68,7 +77,7 @@ pub(crate) use self::transport::{
request_pair_allowed_for_transport, request_pair_direct_auth,
request_pair_transport_unsupported_reason, CandidateTransportPolicyFacts,
};
pub(crate) use crate::control::GatewayControlDecision;
pub(crate) use crate::control::{GatewayControlDecision, GatewayCredentialCarrier};
pub(crate) use crate::execution_runtime::{ConversionMode, ExecutionStrategy};
pub(crate) use crate::headers::RequestOrigin;
pub(crate) use aether_ai_serving::{
@@ -86,6 +95,7 @@ pub(crate) fn build_provider_transport_request_url(
upstream_is_stream: bool,
request_query: Option<&str>,
kiro_api_region: Option<&str>,
api_operation: Option<ApiOperation>,
) -> Option<String> {
self::transport::build_transport_request_url(
transport,
@@ -95,6 +105,7 @@ pub(crate) fn build_provider_transport_request_url(
upstream_is_stream,
request_query,
kiro_api_region,
api_operation,
},
)
}
@@ -106,6 +117,7 @@ pub(crate) fn build_provider_transport_request_url_for_request_body(
upstream_is_stream: bool,
request_query: Option<&str>,
kiro_api_region: Option<&str>,
api_operation: Option<ApiOperation>,
provider_request_body: Option<&serde_json::Value>,
) -> Option<String> {
self::transport::build_transport_request_url_for_request_body(
@@ -116,6 +128,7 @@ pub(crate) fn build_provider_transport_request_url_for_request_body(
upstream_is_stream,
request_query,
kiro_api_region,
api_operation,
},
provider_request_body,
)
@@ -0,0 +1,170 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use serde_json::Value;
use crate::ai_serving::transport::antigravity::{
build_antigravity_safe_v1internal_request, build_antigravity_static_identity_headers,
classify_local_antigravity_request_support, AntigravityEnvelopeRequestType,
AntigravityRequestAuth, AntigravityRequestAuthUnsupportedReason,
AntigravityRequestEnvelopeSupport, AntigravityRequestSideSupport,
AntigravityRequestSideUnsupportedReason,
};
use crate::ai_serving::transport::{
build_standard_provider_request_headers, GatewayProviderTransportSnapshot,
StandardProviderRequestHeaders, StandardProviderRequestHeadersInput,
};
use crate::AppState;
pub(crate) const ANTIGRAVITY_V1INTERNAL_ENVELOPE_NAME: &str = "antigravity:v1internal";
pub(crate) enum AntigravityV1InternalRequestError {
TransportUnsupported,
EnvelopeUnsupported,
UpstreamUrlUnavailable,
HeaderRulesApplyFailed,
}
pub(crate) struct AntigravityV1InternalRequestInput<'a> {
pub(crate) state: &'a AppState,
pub(crate) parts: &'a http::request::Parts,
pub(crate) transport: &'a Arc<GatewayProviderTransportSnapshot>,
pub(crate) trace_id: &'a str,
pub(crate) mapped_model: &'a str,
pub(crate) provider_api_format: &'a str,
pub(crate) auth_header: &'a str,
pub(crate) auth_value: &'a str,
pub(crate) request_headers: &'a http::HeaderMap,
pub(crate) original_request_body: &'a Value,
pub(crate) gemini_request_body: &'a Value,
pub(crate) upstream_is_stream: bool,
pub(crate) same_format: bool,
}
pub(crate) struct AntigravityV1InternalRequest {
pub(crate) transport: Arc<GatewayProviderTransportSnapshot>,
pub(crate) body: Value,
pub(crate) headers: StandardProviderRequestHeaders,
pub(crate) upstream_url: String,
}
pub(crate) async fn build_antigravity_v1internal_provider_request(
input: AntigravityV1InternalRequestInput<'_>,
) -> Result<AntigravityV1InternalRequest, AntigravityV1InternalRequestError> {
let payload = build_antigravity_v1internal_payload(
input.state,
input.transport,
input.trace_id,
input.mapped_model,
input.gemini_request_body,
)
.await?;
let upstream_url = crate::ai_serving::build_provider_transport_request_url_for_request_body(
&payload.transport,
input.provider_api_format,
Some(input.mapped_model),
input.upstream_is_stream,
input.parts.uri.query(),
None,
None,
Some(&payload.body),
)
.ok_or(AntigravityV1InternalRequestError::UpstreamUrlUnavailable)?;
let extra_headers: BTreeMap<String, String> =
build_antigravity_static_identity_headers(&payload.auth);
let mut headers =
build_standard_provider_request_headers(StandardProviderRequestHeadersInput {
transport: &payload.transport,
provider_api_format: input.provider_api_format,
same_format: input.same_format,
headers: input.request_headers,
auth_header: input.auth_header,
auth_value: input.auth_value,
extra_headers: &extra_headers,
header_rules: payload.transport.endpoint.header_rules.as_ref(),
provider_request_body: &payload.body,
original_request_body: input.original_request_body,
upstream_is_stream: input.upstream_is_stream,
})
.ok_or(AntigravityV1InternalRequestError::HeaderRulesApplyFailed)?;
headers
.headers
.insert("accept".to_string(), "text/event-stream".to_string());
Ok(AntigravityV1InternalRequest {
transport: payload.transport,
body: payload.body,
headers,
upstream_url,
})
}
struct AntigravityV1InternalPayload {
transport: Arc<GatewayProviderTransportSnapshot>,
auth: AntigravityRequestAuth,
body: Value,
}
async fn build_antigravity_v1internal_payload(
state: &AppState,
transport: &Arc<GatewayProviderTransportSnapshot>,
trace_id: &str,
mapped_model: &str,
gemini_request_body: &Value,
) -> Result<AntigravityV1InternalPayload, AntigravityV1InternalRequestError> {
let mut resolved_transport = Arc::clone(transport);
let mut antigravity_support = classify_local_antigravity_request_support(
&resolved_transport,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
);
if matches!(
antigravity_support,
AntigravityRequestSideSupport::Unsupported(
AntigravityRequestSideUnsupportedReason::UnsupportedAuth(
AntigravityRequestAuthUnsupportedReason::MissingProjectId
)
)
) {
if let Some(hydrated) = state
.hydrate_antigravity_project_metadata_for_transport(&resolved_transport)
.await
{
resolved_transport = Arc::new(hydrated);
antigravity_support = classify_local_antigravity_request_support(
&resolved_transport,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
);
}
}
let auth = match antigravity_support {
AntigravityRequestSideSupport::Supported(spec) => spec.auth,
AntigravityRequestSideSupport::Unsupported(_) => {
return Err(AntigravityV1InternalRequestError::TransportUnsupported);
}
};
let body = match build_antigravity_safe_v1internal_request(
&auth,
trace_id,
mapped_model,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
) {
AntigravityRequestEnvelopeSupport::Supported(envelope) => envelope,
AntigravityRequestEnvelopeSupport::Unsupported(_) => {
return Err(AntigravityV1InternalRequestError::EnvelopeUnsupported);
}
};
Ok(AntigravityV1InternalPayload {
transport: resolved_transport,
auth,
body,
})
}
@@ -1,6 +1,8 @@
use aether_routing_core::ResolvedRoutingPolicy;
use aether_scheduler_core::{
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session, ClientSessionAffinity,
SchedulerAffinityTarget, SchedulerMinimalCandidateSelectionCandidate,
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope,
ClientSessionAffinity, SchedulerAffinityScope, SchedulerAffinityTarget,
SchedulerMinimalCandidateSelectionCandidate,
};
use crate::ai_serving::{GatewayAuthApiKeySnapshot, PlannerAppState};
@@ -8,25 +10,38 @@ use crate::scheduler::affinity::SCHEDULER_AFFINITY_TTL;
const PLANNER_SCHEDULER_AFFINITY_MAX_ENTRIES: usize = 10_000;
pub(crate) fn has_explicit_session_affinity(
client_session_affinity: Option<&ClientSessionAffinity>,
) -> bool {
client_session_affinity.is_some_and(ClientSessionAffinity::has_session_key)
}
pub(crate) fn read_cached_scheduler_affinity_target(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: Option<&str>,
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Option<SchedulerAffinityTarget> {
if !has_explicit_session_affinity(client_session_affinity) {
return None;
}
let requested_model = requested_model
.map(str::trim)
.filter(|value| !value.is_empty())?;
let api_key_id = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())?;
let cache_key = build_scheduler_affinity_cache_key_for_api_key_id_with_client_session(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
)?;
let affinity_scope = scheduler_affinity_scope_for_routing_policy(routing_policy);
let cache_key =
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
affinity_scope.as_ref(),
)?;
state
.app()
@@ -41,6 +56,9 @@ pub(crate) fn remember_scheduler_affinity_for_candidate(
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) {
if !has_explicit_session_affinity(client_session_affinity) {
return;
}
remember_scheduler_affinity_for_candidate_at_epoch(
state,
auth_snapshot,
@@ -61,18 +79,71 @@ pub(crate) fn remember_scheduler_affinity_for_candidate_at_epoch(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
expected_epoch: Option<u64>,
) {
remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state,
auth_snapshot,
client_session_affinity,
client_api_format,
requested_model,
candidate,
None,
expected_epoch,
);
}
#[allow(clippy::too_many_arguments)]
pub(crate) fn remember_scheduler_affinity_for_candidate_with_routing_policy_at_epoch(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
routing_policy: Option<&ResolvedRoutingPolicy>,
expected_epoch: Option<u64>,
) {
let affinity_scope = scheduler_affinity_scope_for_routing_policy(routing_policy);
remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state,
auth_snapshot,
client_session_affinity,
client_api_format,
requested_model,
candidate,
affinity_scope.as_ref(),
expected_epoch,
);
}
#[allow(clippy::too_many_arguments)]
fn remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
affinity_scope: Option<&SchedulerAffinityScope>,
expected_epoch: Option<u64>,
) {
if !has_explicit_session_affinity(client_session_affinity) {
return;
}
let Some(api_key_id) = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())
else {
return;
};
let Some(cache_key) = build_scheduler_affinity_cache_key_for_api_key_id_with_client_session(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
) else {
let Some(cache_key) =
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
affinity_scope,
)
else {
return;
};
@@ -88,3 +159,15 @@ pub(crate) fn remember_scheduler_affinity_for_candidate_at_epoch(
expected_epoch,
);
}
fn scheduler_affinity_scope_for_routing_policy(
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Option<SchedulerAffinityScope> {
let policy = routing_policy?;
let group_id = policy
.group_id
.as_deref()
.map(str::trim)
.filter(|group_id| !group_id.is_empty())?;
Some(SchedulerAffinityScope::new(group_id, policy.group_version))
}
File diff suppressed because it is too large Load Diff
@@ -134,6 +134,7 @@ mod tests {
global_model_id: "global-1".to_string(),
global_model_name: "gpt-5.4".to_string(),
selected_provider_model_name: "gpt-5.4".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -169,6 +169,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-test".to_string(),
selected_provider_model_name: "gpt-test-upstream".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -1,5 +1,3 @@
use std::collections::BTreeMap;
use aether_ai_serving::{
ai_ranking_context, build_ai_rankable_candidate, run_ai_candidate_ranking,
AiCandidateRankingPort, AiRankableCandidateParts, AiRankingContextConfig,
@@ -7,6 +5,7 @@ use aether_ai_serving::{
};
use aether_routing_core::{ResolvedRoutingPolicy, RoutingSchedulingMode, RoutingSetPriorityMode};
use async_trait::async_trait;
use tokio::sync::Mutex;
use tracing::warn;
use crate::ai_serving::{GatewayAuthApiKeySnapshot, PlannerAppState};
@@ -24,7 +23,7 @@ use aether_scheduler_core::{
use super::candidate_affinity_cache::read_cached_scheduler_affinity_target;
use super::candidate_resolution::{EligibleLocalExecutionCandidate, LocalExecutionCandidateKind};
use super::candidate_transport_ranking_facts::{
resolve_cached_transport_ranking_facts, CandidateTransportRankingFacts,
resolve_cached_transport_ranking_facts, CandidateTransportRankingFactsCache,
};
struct GatewayLocalCandidateRankingPort<'a> {
@@ -35,6 +34,7 @@ struct GatewayLocalCandidateRankingPort<'a> {
required_capabilities: Option<&'a serde_json::Value>,
ordering_config: SchedulerOrderingConfig,
routing_policy: Option<&'a ResolvedRoutingPolicy>,
transport_ranking_facts_cache: Mutex<CandidateTransportRankingFactsCache>,
}
#[async_trait]
@@ -66,6 +66,7 @@ impl AiCandidateRankingPort for GatewayLocalCandidateRankingPort<'_> {
self.client_session_affinity,
normalized_client_api_format,
affinity_requested_model,
self.routing_policy,
))
}
@@ -84,13 +85,17 @@ impl AiCandidateRankingPort for GatewayLocalCandidateRankingPort<'_> {
normalized_client_api_format: &str,
cached_affinity_match: bool,
) -> Result<SchedulerRankableCandidate, Self::Error> {
let ranking_facts = resolve_transport_ranking_facts_for_candidate(
self.state,
&candidate.candidate,
candidate.transport.as_ref(),
self.ordering_config,
)
.await;
let ranking_facts = {
let mut cache = self.transport_ranking_facts_cache.lock().await;
resolve_cached_transport_ranking_facts(
self.state,
&mut cache,
&candidate.candidate,
candidate.transport.as_ref(),
self.ordering_config,
)
.await
};
let routing_overlaid_candidate =
routing_overlaid_candidate(self.routing_policy, candidate.kind, &candidate.candidate);
Ok(build_ai_rankable_candidate(AiRankableCandidateParts {
@@ -137,6 +142,7 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
required_capabilities,
ordering_config,
routing_policy,
transport_ranking_facts_cache: Mutex::new(CandidateTransportRankingFactsCache::default()),
};
match run_ai_candidate_ranking(&port, candidates, normalized_client_api_format).await {
@@ -145,23 +151,6 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
}
}
async fn resolve_transport_ranking_facts_for_candidate(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &crate::ai_serving::GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let mut ordering_cache = BTreeMap::new();
resolve_cached_transport_ranking_facts(
state,
&mut ordering_cache,
candidate,
transport,
ordering_config,
)
.await
}
fn cached_affinity_matches_local_execution_scope(
eligible: &EligibleLocalExecutionCandidate,
target: &SchedulerAffinityTarget,
@@ -233,17 +222,21 @@ fn routing_overlaid_candidate(
let mut overlaid = candidate.clone();
overlaid.provider_priority = policy
.ranking_overlay
.provider_priority_or_unspecified(candidate.provider_id.as_str());
.provider_priority(candidate.provider_id.as_str(), candidate.provider_priority);
let overlaid_key_priority = match kind {
LocalExecutionCandidateKind::SingleKey => policy
.ranking_overlay
.key_priority_or_unspecified(candidate.key_id.as_str()),
.key_priority_overrides
.get(candidate.key_id.as_str()),
LocalExecutionCandidateKind::PoolGroup => policy
.ranking_overlay
.pool_priority_or_unspecified(candidate.provider_id.as_str()),
.pool_priority_overrides
.get(candidate.provider_id.as_str()),
};
overlaid.key_internal_priority = overlaid_key_priority;
overlaid.key_global_priority_for_format = Some(overlaid_key_priority);
if let Some(overlaid_key_priority) = overlaid_key_priority.copied() {
overlaid.key_internal_priority = overlaid_key_priority;
overlaid.key_global_priority_for_format = Some(overlaid_key_priority);
}
overlaid
}
@@ -284,7 +277,9 @@ mod tests {
use serde_json::json;
use super::super::candidate_affinity_cache::remember_scheduler_affinity_for_candidate;
use super::super::candidate_transport_ranking_facts::resolve_cached_candidate_transport_ranking_facts;
use super::super::candidate_transport_ranking_facts::{
resolve_cached_candidate_transport_ranking_facts, CandidateTransportRankingFactsCache,
};
use super::{PlannerAppState, SchedulerMinimalCandidateSelectionCandidate};
use crate::ai_serving::planner::candidate_resolution::{
resolve_and_rank_local_execution_candidates,
@@ -306,7 +301,7 @@ mod tests {
let ordering_config = super::read_scheduler_ordering_config_or_default(state).await;
let mut candidates = candidates;
let mut rankables = Vec::with_capacity(candidates.len());
let mut ordering_cache = BTreeMap::new();
let mut ordering_cache = CandidateTransportRankingFactsCache::default();
for (original_index, candidate) in candidates.iter().enumerate() {
let ranking_facts = resolve_cached_candidate_transport_ranking_facts(
@@ -358,12 +353,13 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-4.1".to_string(),
selected_provider_model_name: "gpt-4.1".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
#[test]
fn routing_policy_priorities_do_not_fall_back_to_candidate_priorities() {
fn routing_policy_priorities_fall_back_to_candidate_priorities() {
let mut candidate = sample_candidate("endpoint-1", "key-1");
candidate.provider_priority = 7;
candidate.key_internal_priority = 3;
@@ -389,18 +385,9 @@ mod tests {
&candidate,
);
assert_eq!(
overlaid.provider_priority,
aether_routing_core::ROUTING_PRIORITY_UNSPECIFIED
);
assert_eq!(
overlaid.key_internal_priority,
aether_routing_core::ROUTING_PRIORITY_UNSPECIFIED
);
assert_eq!(
overlaid.key_global_priority_for_format,
Some(aether_routing_core::ROUTING_PRIORITY_UNSPECIFIED)
);
assert_eq!(overlaid.provider_priority, 7);
assert_eq!(overlaid.key_internal_priority, 3);
assert_eq!(overlaid.key_global_priority_for_format, Some(2));
}
#[test]
@@ -634,6 +621,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-4.1".to_string(),
selected_provider_model_name: "gpt-4.1".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -1540,6 +1528,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-cached",
"endpoint-cached",
@@ -1551,7 +1540,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1573,7 +1562,7 @@ mod tests {
"openai:chat",
"gpt-4.1",
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -1714,6 +1703,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_cross_format = sample_priority_candidate(
"provider-shared",
"endpoint-openai",
@@ -1725,7 +1715,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"claude:messages",
"gpt-4.1",
&cached_cross_format,
@@ -1747,7 +1737,7 @@ mod tests {
"claude:messages",
"gpt-4.1",
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -1915,6 +1905,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-pool",
"endpoint-pool",
@@ -1926,7 +1917,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1948,7 +1939,7 @@ mod tests {
"openai:chat",
Some("gpt-4.1"),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -2008,6 +1999,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-pool",
"endpoint-pool",
@@ -2019,7 +2011,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -2041,7 +2033,7 @@ mod tests {
"openai:chat",
Some("gpt-4.1"),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -2064,7 +2056,7 @@ mod tests {
}
#[tokio::test]
async fn remembers_scheduler_affinity_for_candidate_using_requested_model_key() {
async fn ignores_scheduler_affinity_without_client_session_scope() {
let state = AppState::new().expect("state should build");
let auth_snapshot = sample_auth_snapshot();
let candidate = sample_candidate("endpoint-1", "key-1");
@@ -2078,15 +2070,12 @@ mod tests {
&candidate,
);
let remembered = state
assert!(state
.read_scheduler_affinity_target(
"scheduler_affinity:api-key-1:openai:chat:gpt-5",
SCHEDULER_AFFINITY_TTL,
)
.expect("affinity target should be cached");
assert_eq!(remembered.provider_id, "provider-1");
assert_eq!(remembered.endpoint_id, "endpoint-1");
assert_eq!(remembered.key_id, "key-1");
.is_none());
}
#[tokio::test]
@@ -7,6 +7,7 @@ use aether_ai_serving::{
use aether_routing_core::ResolvedRoutingPolicy;
use async_trait::async_trait;
use std::convert::Infallible;
use std::time::Instant;
use tracing::warn;
use aether_scheduler_core::{
@@ -20,6 +21,7 @@ use crate::ai_serving::{
PlannerAppState,
};
use crate::orchestration::LocalExecutionCandidateMetadata;
use crate::stage_metrics::observe_gateway_stage_ms;
use super::candidate_ranking::rank_eligible_local_execution_candidates;
@@ -68,7 +70,7 @@ struct GatewayLocalCandidateResolutionPort<'a> {
#[async_trait]
impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
type Candidate = SchedulerMinimalCandidateSelectionCandidate;
type Transport = GatewayProviderTransportSnapshot;
type Transport = Arc<GatewayProviderTransportSnapshot>;
type Eligible = EligibleLocalExecutionCandidate;
type Skipped = SkippedLocalExecutionCandidate;
type Error = Infallible;
@@ -77,7 +79,12 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
&self,
candidate: &Self::Candidate,
) -> Result<Option<Self::Transport>, Self::Error> {
Ok(read_candidate_transport_snapshot(self.state, candidate).await)
let started_at = Instant::now();
let transport = read_candidate_transport_snapshot_arc(self.state, candidate).await;
let elapsed_ms = started_at.elapsed().as_millis() as u64;
observe_gateway_stage_ms("candidate_transport_snapshot", elapsed_ms);
observe_gateway_stage_ms("candidate_resolution_transport_read", elapsed_ms);
Ok(transport)
}
fn build_missing_transport_skipped_candidate(
@@ -139,7 +146,7 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
SkippedLocalExecutionCandidate {
candidate,
skip_reason,
transport: Some(Arc::new(transport)),
transport: Some(transport),
ranking: None,
extra_data: None,
}
@@ -159,7 +166,7 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
EligibleLocalExecutionCandidate {
kind,
candidate,
transport: Arc::new(transport),
transport,
provider_api_format,
orchestration: LocalExecutionCandidateMetadata::default(),
ranking: None,
@@ -171,7 +178,8 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
candidates: Vec<Self::Eligible>,
normalized_client_api_format: &str,
) -> Result<Vec<Self::Eligible>, Self::Error> {
Ok(rank_eligible_local_execution_candidates(
let started_at = Instant::now();
let ranked = rank_eligible_local_execution_candidates(
self.state,
candidates,
normalized_client_api_format,
@@ -181,7 +189,12 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
self.required_capabilities,
self.routing_policy,
)
.await)
.await;
observe_gateway_stage_ms(
"candidate_resolution_rank",
started_at.elapsed().as_millis() as u64,
);
Ok(ranked)
}
async fn apply_pool_scheduler(
@@ -358,8 +371,13 @@ async fn resolve_and_rank_local_execution_candidates_with_pool_expansion(
expand_pool_groups,
};
let started_at = Instant::now();
match run_ai_candidate_resolution(&port, candidates, request).await {
Ok(mut outcome) => {
observe_gateway_stage_ms(
"candidate_resolution_core",
started_at.elapsed().as_millis() as u64,
);
for candidate in &mut outcome.eligible_candidates {
candidate.orchestration.scheduler_affinity_epoch = Some(scheduler_affinity_epoch);
}
@@ -506,8 +524,17 @@ pub(crate) async fn read_candidate_transport_snapshot(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> Option<GatewayProviderTransportSnapshot> {
read_candidate_transport_snapshot_arc(state, candidate)
.await
.map(|transport| (*transport).clone())
}
pub(crate) async fn read_candidate_transport_snapshot_arc(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> Option<Arc<GatewayProviderTransportSnapshot>> {
match state
.read_provider_transport_snapshot(
.read_provider_transport_snapshot_arc(
&candidate.provider_id,
&candidate.endpoint_id,
&candidate.key_id,
@@ -616,6 +643,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "claude-sonnet".to_string(),
selected_provider_model_name: "claude-sonnet".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
File diff suppressed because it is too large Load Diff
@@ -1,8 +1,10 @@
use std::collections::BTreeMap;
use aether_contracts::ProxySnapshot;
use aether_scheduler_core::{
SchedulerMinimalCandidateSelectionCandidate, SchedulerTunnelAffinityBucket,
};
use serde_json::Value;
use tracing::warn;
use crate::ai_serving::{GatewayProviderTransportSnapshot, PlannerAppState};
@@ -10,7 +12,9 @@ use crate::scheduler::config::SchedulerOrderingConfig;
use super::candidate_resolution::read_candidate_transport_snapshot;
pub(super) type CandidateTransportIdentity<'a> = (&'a str, &'a str, &'a str);
const TUNNEL_OWNER_INSTANCE_ID_EXTRA_KEY: &str = "tunnel_owner_instance_id";
pub(super) type CandidateTransportIdentity = (String, String, String);
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(super) struct CandidateTransportRankingFacts {
@@ -18,38 +22,51 @@ pub(super) struct CandidateTransportRankingFacts {
pub(super) keep_priority_on_conversion: bool,
}
pub(super) async fn resolve_cached_candidate_transport_ranking_facts<'a>(
state: PlannerAppState<'_>,
cache: &mut BTreeMap<CandidateTransportIdentity<'a>, CandidateTransportRankingFacts>,
candidate: &'a SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.get(&identity).copied() {
return facts;
}
let facts = resolve_candidate_transport_ranking_facts(state, candidate, ordering_config).await;
cache.insert(identity, facts);
facts
#[derive(Debug, Default)]
pub(super) struct CandidateTransportRankingFactsCache {
candidate_facts: BTreeMap<CandidateTransportIdentity, CandidateTransportRankingFacts>,
configured_proxy_snapshots: BTreeMap<String, Option<ProxySnapshot>>,
system_proxy_snapshot: Option<Option<ProxySnapshot>>,
tunnel_buckets_by_node_id: BTreeMap<String, SchedulerTunnelAffinityBucket>,
}
pub(super) async fn resolve_cached_transport_ranking_facts<'a>(
pub(super) async fn resolve_cached_candidate_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut BTreeMap<CandidateTransportIdentity<'a>, CandidateTransportRankingFacts>,
candidate: &'a SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.get(&identity).copied() {
if let Some(facts) = cache.candidate_facts.get(&identity).copied() {
return facts;
}
let facts =
resolve_candidate_transport_ranking_facts_from_transport(state, transport, ordering_config)
.await;
cache.insert(identity, facts);
resolve_candidate_transport_ranking_facts(state, cache, candidate, ordering_config).await;
cache.candidate_facts.insert(identity, facts);
facts
}
pub(super) async fn resolve_cached_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.candidate_facts.get(&identity).copied() {
return facts;
}
let facts = resolve_candidate_transport_ranking_facts_from_transport(
state,
cache,
transport,
ordering_config,
)
.await;
cache.candidate_facts.insert(identity, facts);
facts
}
@@ -58,13 +75,15 @@ pub(super) async fn candidate_keeps_priority_on_conversion(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> bool {
resolve_candidate_transport_ranking_facts(state, candidate, ordering_config)
let mut cache = CandidateTransportRankingFactsCache::default();
resolve_candidate_transport_ranking_facts(state, &mut cache, candidate, ordering_config)
.await
.keep_priority_on_conversion
}
async fn resolve_candidate_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
@@ -75,17 +94,23 @@ async fn resolve_candidate_transport_ranking_facts(
};
};
resolve_candidate_transport_ranking_facts_from_transport(state, &transport, ordering_config)
.await
resolve_candidate_transport_ranking_facts_from_transport(
state,
cache,
&transport,
ordering_config,
)
.await
}
async fn resolve_candidate_transport_ranking_facts_from_transport(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
CandidateTransportRankingFacts {
tunnel_bucket: resolve_tunnel_owner_affinity_from_transport(state, transport).await,
tunnel_bucket: resolve_tunnel_owner_affinity_from_transport(state, cache, transport).await,
keep_priority_on_conversion: ordering_config.keep_priority_on_conversion
|| transport.provider.keep_priority_on_conversion,
}
@@ -93,12 +118,11 @@ async fn resolve_candidate_transport_ranking_facts_from_transport(
async fn resolve_tunnel_owner_affinity_from_transport(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
) -> SchedulerTunnelAffinityBucket {
let Some(proxy) = state
.app()
.resolve_transport_proxy_snapshot_with_tunnel_affinity(transport)
.await
let Some(proxy) =
resolve_transport_proxy_snapshot_with_tunnel_affinity_cached(state, cache, transport).await
else {
return SchedulerTunnelAffinityBucket::Neutral;
};
@@ -114,10 +138,74 @@ async fn resolve_tunnel_owner_affinity_from_transport(
return SchedulerTunnelAffinityBucket::Neutral;
};
if let Some(bucket) = cache.tunnel_buckets_by_node_id.get(node_id).copied() {
return bucket;
}
let bucket = resolve_tunnel_owner_affinity_from_proxy(state, &proxy, node_id).await;
cache
.tunnel_buckets_by_node_id
.insert(node_id.to_string(), bucket);
bucket
}
async fn resolve_transport_proxy_snapshot_with_tunnel_affinity_cached(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
) -> Option<ProxySnapshot> {
for raw in [
transport.key.proxy.as_ref(),
transport.endpoint.proxy.as_ref(),
transport.provider.proxy.as_ref(),
]
.into_iter()
.flatten()
{
let cache_key = proxy_config_cache_key(raw);
if let Some(snapshot) = cache.configured_proxy_snapshots.get(&cache_key) {
if snapshot.is_some() {
return snapshot.clone();
}
continue;
}
let snapshot = state
.app()
.resolve_configured_proxy_snapshot_with_tunnel_affinity(Some(raw))
.await;
cache
.configured_proxy_snapshots
.insert(cache_key, snapshot.clone());
if snapshot.is_some() {
return snapshot;
}
}
if let Some(snapshot) = cache.system_proxy_snapshot.as_ref() {
return snapshot.clone();
}
let snapshot = state.app().resolve_system_proxy_snapshot().await;
cache.system_proxy_snapshot = Some(snapshot.clone());
snapshot
}
async fn resolve_tunnel_owner_affinity_from_proxy(
state: PlannerAppState<'_>,
proxy: &ProxySnapshot,
node_id: &str,
) -> SchedulerTunnelAffinityBucket {
if state.app().tunnel.has_local_proxy(node_id) {
return SchedulerTunnelAffinityBucket::LocalTunnel;
}
if let Some(owner_instance_id) = proxy_tunnel_owner_instance_id(proxy) {
return if owner_instance_id == state.app().tunnel.local_instance_id() {
SchedulerTunnelAffinityBucket::LocalTunnel
} else {
SchedulerTunnelAffinityBucket::RemoteTunnel
};
}
match state
.app()
.tunnel
@@ -144,10 +232,25 @@ async fn resolve_tunnel_owner_affinity_from_transport(
fn candidate_transport_identity(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> CandidateTransportIdentity<'_> {
) -> CandidateTransportIdentity {
(
candidate.provider_id.as_str(),
candidate.endpoint_id.as_str(),
candidate.key_id.as_str(),
candidate.provider_id.clone(),
candidate.endpoint_id.clone(),
candidate.key_id.clone(),
)
}
fn proxy_config_cache_key(raw: &Value) -> String {
serde_json::to_string(raw).unwrap_or_else(|_| raw.to_string())
}
fn proxy_tunnel_owner_instance_id(proxy: &ProxySnapshot) -> Option<&str> {
proxy
.extra
.as_ref()
.and_then(Value::as_object)
.and_then(|extra| extra.get(TUNNEL_OWNER_INSTANCE_ID_EXTRA_KEY))
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
}
@@ -1,15 +1,16 @@
use axum::body::Bytes;
use crate::ai_serving::is_json_request;
use crate::ai_serving::{
endpoint_config_forces_upstream_stream_policy as endpoint_config_forces_upstream_stream_policy_impl,
enforce_request_body_stream_field as enforce_request_body_stream_field_impl,
force_upstream_streaming_for_provider as force_upstream_streaming_for_provider_impl,
is_json_request, parse_direct_request_body as parse_direct_request_body_impl,
resolve_upstream_is_stream_from_endpoint_config as resolve_upstream_is_stream_from_endpoint_config_impl,
parse_direct_request_body as parse_direct_request_body_impl,
resolve_format_upstream_is_stream_for_provider as resolve_upstream_is_stream_for_provider_impl,
};
pub(crate) use crate::ai_serving::{
CLAUDE_CHAT_STREAM_PLAN_KIND, CLAUDE_CHAT_SYNC_PLAN_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, EXECUTION_RUNTIME_STREAM_ACTION,
CLAUDE_CLI_SYNC_PLAN_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND, EXECUTION_RUNTIME_STREAM_ACTION,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
@@ -20,9 +21,10 @@ pub(crate) use crate::ai_serving::{
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND, OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_SEARCH_SYNC_PLAN_KIND,
OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_VIDEO_CONTENT_PLAN_KIND,
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_serving::AiRequestedModelFamily as RequestedModelFamily;
@@ -54,10 +56,10 @@ pub(crate) fn resolve_upstream_is_stream_for_provider(
client_is_stream: bool,
hard_requires_streaming: bool,
) -> bool {
let hard_requires_streaming = hard_requires_streaming
|| force_upstream_streaming_for_provider(provider_type, provider_api_format);
resolve_upstream_is_stream_from_endpoint_config_impl(
resolve_upstream_is_stream_for_provider_impl(
endpoint_config,
provider_type,
provider_api_format,
client_is_stream,
hard_requires_streaming,
)
@@ -178,6 +180,27 @@ mod tests {
true,
false,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"codex",
"openai:image",
true,
true,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"codex",
"openai:responses:compact",
true,
true,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"custom",
"openai:responses:compact",
true,
true,
));
}
#[test]
@@ -1,14 +1,15 @@
use crate::ai_serving::planner::common::{
CLAUDE_CHAT_STREAM_PLAN_KIND, CLAUDE_CHAT_SYNC_PLAN_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, GEMINI_CHAT_STREAM_PLAN_KIND, GEMINI_CHAT_SYNC_PLAN_KIND,
GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND, GEMINI_EMBEDDING_SYNC_PLAN_KIND,
GEMINI_FILES_DELETE_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND,
GEMINI_FILES_LIST_PLAN_KIND, GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND,
GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_SYNC_PLAN_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DELETE_PLAN_KIND,
GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND, GEMINI_FILES_LIST_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_SYNC_PLAN_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_RERANK_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND, OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND,
OPENAI_RESPONSES_STREAM_PLAN_KIND, OPENAI_RESPONSES_SYNC_PLAN_KIND,
OPENAI_SEARCH_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND, OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
@@ -100,12 +101,15 @@ fn build_sync_plan_payload_from_decision(
OPENAI_RESPONSES_SYNC_PLAN_KIND => {
build_openai_responses_sync_plan_from_decision(parts, body_json, payload, false)?
}
OPENAI_IMAGE_SYNC_PLAN_KIND => build_passthrough_sync_plan_from_decision(parts, payload)?,
OPENAI_IMAGE_SYNC_PLAN_KIND | OPENAI_SEARCH_SYNC_PLAN_KIND => {
build_passthrough_sync_plan_from_decision(parts, payload)?
}
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND => {
build_openai_responses_sync_plan_from_decision(parts, body_json, payload, true)?
}
CLAUDE_CHAT_SYNC_PLAN_KIND
| CLAUDE_CLI_SYNC_PLAN_KIND
| CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND
| OPENAI_EMBEDDING_SYNC_PLAN_KIND
| OPENAI_RERANK_SYNC_PLAN_KIND => {
build_standard_sync_plan_from_decision(parts, body_json, payload)?
File diff suppressed because it is too large Load Diff
@@ -59,6 +59,7 @@ pub(crate) async fn build_gemini_cli_v1internal_provider_request(
input.upstream_is_stream,
input.parts.uri.query(),
None,
None,
Some(&payload.body),
)
.ok_or(GeminiCliV1InternalRequestError::UpstreamUrlUnavailable)?;
@@ -1,6 +1,7 @@
use crate::ai_serving::{AiExecutionDecision, AiExecutionPlanPayload, GatewayControlDecision};
use crate::{AppState, GatewayError};
mod antigravity;
mod candidate_affinity_cache;
mod candidate_materialization;
mod candidate_metadata;
@@ -33,6 +34,7 @@ pub(crate) use self::candidate_resolution::{
candidate_auth_channel_skip_reason, read_candidate_transport_snapshot,
EligibleLocalExecutionCandidate, LocalExecutionCandidateKind, SkippedLocalExecutionCandidate,
};
pub(crate) use self::common::resolve_upstream_is_stream_for_provider;
pub(crate) use self::passthrough::{
build_local_same_format_stream_attempt_source, build_local_same_format_stream_plan_and_reports,
build_local_same_format_sync_attempt_source, build_local_same_format_sync_plan_and_reports,
@@ -47,7 +49,7 @@ pub(crate) use self::plan_builders::{
pub(crate) use self::pool_scores::{
build_provider_key_pool_score_upsert, provider_key_pool_score_id, provider_key_pool_score_scope,
};
pub(crate) use self::request_gzip::resolve_transport_request_gzip_policy;
pub(crate) use self::request_gzip::resolve_transport_request_encoding_policy;
pub(crate) use self::route::is_matching_stream_request as planner_is_matching_stream_request;
pub(crate) use self::runtime_miss::{
apply_local_runtime_candidate_terminal_reason, record_local_runtime_candidate_skip_reason,
@@ -78,7 +80,8 @@ pub(crate) use self::standard::{
build_local_stream_plan_and_reports as build_standard_family_stream_plan_and_reports,
build_local_sync_attempt_source as build_standard_family_sync_attempt_source,
build_local_sync_plan_and_reports as build_standard_family_sync_plan_and_reports,
set_local_openai_chat_execution_exhausted_diagnostic,
codex_model_capabilities_for_transport, set_local_openai_chat_execution_exhausted_diagnostic,
validate_final_openai_provider_request,
};
pub(crate) use self::state::{
GatewayAuthApiKeySnapshot, GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
@@ -90,6 +93,71 @@ pub(crate) use aether_ai_serving::{
CandidateFailureDiagnostic, CandidateFailureDiagnosticKind,
};
pub(crate) struct ResolvedTunnelSchedulerAffinityContext {
pub(crate) requested_model: String,
pub(crate) client_session_affinity: Option<aether_scheduler_core::ClientSessionAffinity>,
pub(crate) policy_context: Option<crate::scheduler::affinity::SchedulerAffinityPolicyContext>,
pub(crate) routing_overlay: Option<aether_routing_core::RankingOverlay>,
}
pub(crate) async fn resolve_tunnel_scheduler_affinity_context(
state: &AppState,
parts: &http::request::Parts,
decision: &GatewayControlDecision,
requested_model: String,
body_json: &serde_json::Value,
client_api_format: &str,
) -> Result<Option<ResolvedTunnelSchedulerAffinityContext>, GatewayError> {
let Some(auth_context) = decision.auth_context.as_ref() else {
return Ok(None);
};
let execution_auth_context =
crate::ai_serving::build_execution_runtime_auth_context(auth_context);
let Some(auth_snapshot) = state
.read_cached_auth_api_key_snapshot(
&execution_auth_context.user_id,
&execution_auth_context.api_key_id,
crate::clock::current_unix_secs(),
)
.await?
else {
return Ok(None);
};
let resolved_auth_input = decision_input::ResolvedLocalDecisionAuthInput {
auth_context: execution_auth_context,
auth_snapshot,
required_capabilities: None,
model_directive_policy: decision.model_directive_policy.clone(),
};
let mut input = decision_input::build_local_requested_model_decision_input(
resolved_auth_input,
requested_model,
);
decision_input::attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
client_api_format,
)
.await?;
let policy_context = input
.routing_policy
.as_ref()
.map(crate::scheduler::affinity::SchedulerAffinityPolicyContext::from_routing_policy);
let routing_overlay = input
.routing_policy
.as_ref()
.map(|policy| policy.ranking_overlay.clone());
Ok(Some(ResolvedTunnelSchedulerAffinityContext {
requested_model: input.requested_model,
client_session_affinity: input.client_session_affinity,
policy_context,
routing_overlay,
}))
}
pub(crate) async fn maybe_build_sync_decision_payload(
state: &AppState,
parts: &http::request::Parts,
@@ -71,6 +71,7 @@ pub(crate) fn build_passthrough_stream_plan_from_decision(
.content_type
.take()
.or_else(|| provider_request_headers.get("content-type").cloned());
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -84,7 +85,7 @@ pub(crate) fn build_passthrough_stream_plan_from_decision(
body_bytes_b64: None,
body_ref: None,
},
stream: true,
stream,
},
);
@@ -66,7 +66,7 @@ pub(crate) async fn maybe_build_sync_local_same_format_provider_decision_payload
candidate_count,
);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) =
maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
@@ -134,7 +134,7 @@ pub(crate) async fn maybe_build_stream_local_same_format_provider_decision_paylo
candidate_count,
);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) =
maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
@@ -24,7 +24,7 @@ use crate::ai_serving::{
ai_local_execution_contract_for_formats, extract_pool_sticky_session_token,
resolve_local_decision_execution_runtime_auth_context, GatewayControlDecision, PlannerAppState,
};
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::client_session_affinity::client_session_affinity_from_api_request;
use crate::clock::current_unix_secs;
use crate::{AppState, GatewayError};
@@ -60,7 +60,9 @@ pub(crate) async fn resolve_local_same_format_provider_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -79,7 +81,13 @@ pub(crate) async fn resolve_local_same_format_provider_decision_input(
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
input.client_surface = decision.client_surface;
input.gateway_credential_carrier = decision.gateway_credential_carrier;
input.client_session_affinity = client_session_affinity_from_api_request(
spec_metadata.api_format,
&parts.headers,
Some(body_json),
);
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
@@ -115,15 +123,23 @@ pub(crate) async fn materialize_local_same_format_provider_candidate_attempts(
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::SameFormatProviderDecision,
);
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec_metadata.api_format, Some(&input.requested_model));
let routing_model = model_directive_resolution
.base_model()
.unwrap_or(&input.requested_model);
let (candidates, preselection_skipped) = planner_state
.list_selectable_candidates_with_skip_reasons(
.list_selectable_candidates_with_skip_reasons_for_request_operation(
spec_metadata.api_format,
&input.requested_model,
routing_model,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
spec.operation.map(|operation| operation.as_str()),
)
.await?;
let outcome = materialize_local_execution_candidates_with_serving(
@@ -212,15 +228,23 @@ pub(crate) async fn build_local_same_format_provider_candidate_attempt_source<'a
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::SameFormatProviderDecision,
);
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec_metadata.api_format, Some(&input.requested_model));
let routing_model = model_directive_resolution
.base_model()
.unwrap_or(&input.requested_model);
let (candidates, preselection_skipped) = planner_state
.list_selectable_candidates_with_skip_reasons(
.list_selectable_candidates_with_skip_reasons_for_request_operation(
spec_metadata.api_format,
&input.requested_model,
routing_model,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
spec.operation.map(|operation| operation.as_str()),
)
.await?;
@@ -1,5 +1,8 @@
use serde_json::json;
use aether_ai_serving::{AdaptationMode, AiRequestGzipPolicy, OriginalRequestPayload};
use aether_contracts::{ExecutionResponseBodyMode, EXECUTION_RESPONSE_BODY_MODE_HEADER};
use crate::ai_serving::ai_local_execution_contract_for_formats;
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::candidate_materialization::{
@@ -17,7 +20,7 @@ use crate::ai_serving::planner::report_context::{
use crate::ai_serving::planner::spec_metadata::local_same_format_provider_spec_metadata;
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
@@ -61,6 +64,8 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
else {
return Ok(None);
};
let request_redacted = resolved.request_redacted;
let compatibility_edits_empty = resolved.compatibility_edits.is_empty();
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
@@ -82,6 +87,51 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
.clone()
.or_else(|| resolve_transport_profile(&resolved.transport));
let mut extra_fields = serde_json::Map::new();
extra_fields.insert(
"provider_type".to_string(),
json!(resolved.transport.provider.provider_type.as_str()),
);
if let Some(operation) = spec.operation {
extra_fields.insert("api_operation".to_string(), json!(operation.as_str()));
}
if let Some(client_surface) = input.client_surface {
extra_fields.insert("client_surface".to_string(), json!(client_surface.as_str()));
}
if let Some(carrier) = input.gateway_credential_carrier {
extra_fields.insert(
"gateway_credential_carrier".to_string(),
json!(carrier.as_str()),
);
}
extra_fields.insert(
"upstream_credential_mode".to_string(),
json!(resolved.transport.key.auth_type.trim().to_ascii_lowercase()),
);
let mut adaptation_mode = if resolved.compatibility_edits.is_empty() {
AdaptationMode::NativeTransparent
} else {
AdaptationMode::SameFormatCompat
};
if crate::ai_serving::normalize_api_format_alias(&resolved.provider_api_format)
== "claude:messages"
{
let compatibility_profile =
crate::ai_serving::transport::resolve_anthropic_compatibility_profile(
&resolved.transport,
&resolved.provider_api_format,
);
extra_fields.insert(
"anthropic_compatibility_profile".to_string(),
json!(compatibility_profile.as_str()),
);
if compatibility_profile.uses_claude_code_compatibility() {
adaptation_mode = AdaptationMode::SameFormatCompat;
}
}
extra_fields.insert(
"adaptation_mode".to_string(),
json!(adaptation_mode.as_str()),
);
if let Some(proxy_value) =
build_request_trace_proxy_value(Some(&resolved.transport), proxy.as_ref())
{
@@ -149,6 +199,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -184,7 +235,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
compatibility_edits: _,
request_redacted: _,
} = resolved;
let request_gzip = resolve_transport_request_gzip_policy(&transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -194,6 +245,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -211,8 +263,8 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -221,10 +273,84 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
enforce_provider_api_operation_invariants(
spec.operation,
decision.provider_request_body.as_mut(),
&mut decision.provider_request_headers,
);
decision.provider_request_body_base64 = original_request_body_base64(
parts,
decision.provider_request_body.as_ref(),
adaptation_mode,
request_redacted,
compatibility_edits_empty,
decision.content_encoding.as_deref(),
decision.request_gzip.as_ref(),
);
decision
.provider_request_headers
.retain(|name, _| !name.eq_ignore_ascii_case(EXECUTION_RESPONSE_BODY_MODE_HEADER));
if !spec_metadata.require_streaming && decision.provider_request_body_base64.is_some() {
decision.provider_request_headers.insert(
EXECUTION_RESPONSE_BODY_MODE_HEADER.to_string(),
ExecutionResponseBodyMode::PreserveBytes
.as_str()
.to_string(),
);
}
Ok(Some(decision))
}
fn enforce_provider_api_operation_invariants(
operation: Option<crate::ai_serving::ApiOperation>,
provider_request_body: Option<&mut serde_json::Value>,
provider_request_headers: &mut std::collections::BTreeMap<String, String>,
) {
if operation != Some(crate::ai_serving::ApiOperation::ClaudeCountTokens) {
return;
}
if let Some(provider_request_body) = provider_request_body {
crate::ai_serving::transport::enforce_same_format_provider_api_operation_body_policy(
provider_request_body,
operation,
);
}
for header_name in ["accept", "content-type"] {
provider_request_headers.retain(|name, _| !name.eq_ignore_ascii_case(header_name));
provider_request_headers.insert(header_name.to_string(), "application/json".to_string());
}
}
fn original_request_body_base64(
parts: &http::request::Parts,
provider_request_body: Option<&serde_json::Value>,
adaptation_mode: AdaptationMode,
request_redacted: bool,
compatibility_edits_empty: bool,
content_encoding: Option<&str>,
request_gzip: Option<&AiRequestGzipPolicy>,
) -> Option<String> {
if adaptation_mode != AdaptationMode::NativeTransparent
|| request_redacted
|| !compatibility_edits_empty
|| content_encoding.is_some_and(|value| !value.trim().is_empty())
|| request_gzip.is_some_and(|policy| policy.enabled != Some(false))
{
return None;
}
parts
.extensions
.get::<OriginalRequestPayload>()?
.body_bytes_base64_if_unchanged(provider_request_body?)
}
pub(super) async fn mark_skipped_local_same_format_provider_candidate(
state: &AppState,
input: &LocalSameFormatProviderDecisionInput,
@@ -308,3 +434,177 @@ pub(super) async fn mark_skipped_local_same_format_provider_candidate_with_failu
)
.await;
}
#[cfg(test)]
mod tests {
use std::collections::BTreeMap;
use base64::Engine as _;
use super::{
enforce_provider_api_operation_invariants, original_request_body_base64, AdaptationMode,
AiRequestGzipPolicy, OriginalRequestPayload,
};
use crate::ai_serving::ApiOperation;
fn request_parts_with_original_payload(
body_json: serde_json::Value,
body_bytes: &[u8],
) -> http::request::Parts {
let (mut parts, ()) = http::Request::new(()).into_parts();
parts
.extensions
.insert(OriginalRequestPayload::from_parsed_json(
body_json, body_bytes,
));
parts
}
#[test]
fn count_tokens_invariants_win_after_provider_routing_mutations() {
let mut body = serde_json::json!({
"model": "claude-sonnet-4",
"messages": [],
"stream": true
});
let mut headers = BTreeMap::from([
("Accept".to_string(), "text/event-stream".to_string()),
("Content-Type".to_string(), "text/plain".to_string()),
("x-provider-route".to_string(), "kept".to_string()),
]);
enforce_provider_api_operation_invariants(
Some(ApiOperation::ClaudeCountTokens),
Some(&mut body),
&mut headers,
);
assert!(body.get("stream").is_none());
assert_eq!(
headers.get("accept").map(String::as_str),
Some("application/json")
);
assert_eq!(
headers.get("content-type").map(String::as_str),
Some("application/json")
);
assert_eq!(
headers.get("x-provider-route").map(String::as_str),
Some("kept")
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("accept"))
.count(),
1
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("content-type"))
.count(),
1
);
}
#[test]
fn unchanged_same_format_body_preserves_original_json_bytes() {
let raw = br#"{ "unknown": {"enabled":true}, "messages": [], "model": "claude-sonnet-4" }"#;
let body_json: serde_json::Value = serde_json::from_slice(raw).expect("body should parse");
let parts = request_parts_with_original_payload(body_json.clone(), raw);
let encoded = original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
None,
None,
)
.expect("unchanged request should retain exact bytes");
assert_eq!(
base64::engine::general_purpose::STANDARD
.decode(encoded)
.expect("body should decode"),
raw
);
}
#[test]
fn request_edits_or_encoding_disable_original_json_bytes() {
let raw = br#"{"model":"claude-sonnet-4","messages":[]}"#;
let body_json: serde_json::Value = serde_json::from_slice(raw).expect("body should parse");
let parts = request_parts_with_original_payload(body_json.clone(), raw);
let changed_body = serde_json::json!({
"model": "claude-sonnet-4-5",
"messages": []
});
assert!(original_request_body_base64(
&parts,
Some(&changed_body),
AdaptationMode::NativeTransparent,
false,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
true,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
false,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::SameFormatCompat,
false,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
Some("gzip"),
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
None,
Some(&AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1),
}),
)
.is_none());
}
}
@@ -13,7 +13,8 @@ use crate::ai_serving::planner::redaction::{
use crate::ai_serving::transport::antigravity::{
build_antigravity_safe_v1internal_request, build_antigravity_static_identity_headers,
classify_local_antigravity_request_support, AntigravityEnvelopeRequestType,
AntigravityRequestEnvelopeSupport, AntigravityRequestSideSupport,
AntigravityRequestAuthUnsupportedReason, AntigravityRequestEnvelopeSupport,
AntigravityRequestSideSupport, AntigravityRequestSideUnsupportedReason,
};
use crate::ai_serving::transport::{
build_gemini_cli_v1internal_request, build_grok_browser_headers, build_grok_upstream_url,
@@ -39,7 +40,10 @@ use super::{
LocalSameFormatProviderCandidateAttempt, LocalSameFormatProviderDecisionInput,
LocalSameFormatProviderSpec,
};
use crate::ai_serving::planner::standard::same_format_provider_request_body_failure_extra_data;
use crate::ai_serving::planner::standard::{
codex_model_capabilities_for_transport, openai_provider_request_contract_failure_extra_data,
same_format_provider_request_body_failure_extra_data,
};
pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trace(
transport: &GatewayProviderTransportSnapshot,
@@ -50,6 +54,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
"openai:chat" => "openai:chat",
"openai:responses" => "openai:responses",
"openai:responses:compact" => "openai:responses:compact",
"openai:search" => "openai:search",
"openai:embedding" => "openai:embedding",
"openai:rerank" => "openai:rerank",
"claude:messages" => "claude:messages",
@@ -71,9 +76,10 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
decision_kind: "trace_candidate_metadata",
report_kind: Some("trace_candidate_metadata"),
},
None,
);
if !behavior.is_antigravity
&& !behavior.is_claude_code
&& !behavior.is_claude_code_transport
&& !behavior.is_gemini_cli
&& !behavior.is_vertex
&& !behavior.is_kiro
@@ -123,6 +129,23 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
spec: LocalSameFormatProviderSpec,
) -> Result<Option<LocalSameFormatProviderCandidatePayloadParts>, GatewayError> {
let candidate = &attempt.eligible.candidate;
if let Some(skip_reason) = same_format_provider_operation_skip_reason(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
spec.operation,
) {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
let Some(prepared) = prepare_local_same_format_provider_candidate(
state,
trace_id,
@@ -136,13 +159,26 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
else {
return Ok(None);
};
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
state,
spec.api_format,
Some(&input.requested_model),
)
.await;
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec.api_format, Some(&input.requested_model));
let model_directive_mapping =
match model_directive_resolution.mapping_patch_for_mapped_model(&prepared.mapped_model) {
Ok(mapping) => mapping,
Err(skip_reason) => {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
};
let effective_headers = input.effective_headers(&parts.headers);
let redaction = resolve_provider_chat_pii_redaction(
state,
@@ -168,7 +204,7 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
prepared.force_body_stream_field,
prepared.kiro_auth.as_ref(),
prepared.is_claude_code,
enable_model_directives,
false,
)
else {
mark_skipped_local_same_format_provider_candidate_with_extra_data(
@@ -195,18 +231,11 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
};
let mut base_provider_request_body = base_provider_request.body;
let mut compatibility_edits = base_provider_request.compatibility_edits;
if let Some(mapping) =
crate::system_features::reasoning_model_directive_mapping_for_api_format_and_model(
state,
spec.api_format,
Some(&input.requested_model),
)
.await
{
if let Some(mapping) = model_directive_mapping.as_ref() {
let before_mapping = base_provider_request_body.clone();
crate::ai_serving::apply_model_directive_mapping_patch(
&mut base_provider_request_body,
&mapping,
mapping,
);
if before_mapping != base_provider_request_body {
compatibility_edits.push(SameFormatProviderCompatibilityEdit {
@@ -229,12 +258,81 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
}
}
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(input.requested_model.as_str());
let codex_model_capabilities = codex_model_capabilities_for_transport(
&transport,
prepared.provider_api_format.as_str(),
prepared.mapped_model.as_str(),
source_model,
);
if let Err(violation) =
crate::ai_serving::finalize_openai_provider_request_with_codex_model_capabilities(
&mut base_provider_request_body,
crate::ai_serving::OpenAiProviderRequestFinalization {
source_api_format: spec.api_format,
provider_api_format: prepared.provider_api_format.as_str(),
provider_type: transport.provider.provider_type.as_str(),
provider_model: prepared.mapped_model.as_str(),
source_model,
body_rules: transport.endpoint.body_rules.as_ref(),
upstream_is_stream: prepared.upstream_is_stream,
require_body_stream_field: request_requires_body_stream_field(
body_json,
prepared.force_body_stream_field,
),
},
codex_model_capabilities.as_ref(),
)
{
mark_skipped_local_same_format_provider_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
Some(openai_provider_request_contract_failure_extra_data(
&violation,
spec.api_format,
prepared.provider_api_format.as_str(),
"same_format_provider_request_finalization",
)),
)
.await;
return Ok(None);
}
let antigravity_auth = if prepared.is_antigravity {
match classify_local_antigravity_request_support(
let mut antigravity_support = classify_local_antigravity_request_support(
&transport,
&base_provider_request_body,
AntigravityEnvelopeRequestType::Agent,
);
if matches!(
antigravity_support,
AntigravityRequestSideSupport::Unsupported(
AntigravityRequestSideUnsupportedReason::UnsupportedAuth(
AntigravityRequestAuthUnsupportedReason::MissingProjectId
)
)
) {
if let Some(hydrated) = state
.hydrate_antigravity_project_metadata_for_transport(&transport)
.await
{
transport = Arc::new(hydrated);
antigravity_support = classify_local_antigravity_request_support(
&transport,
&base_provider_request_body,
AntigravityEnvelopeRequestType::Agent,
);
}
}
match antigravity_support {
AntigravityRequestSideSupport::Supported(spec) => Some(spec.auth),
AntigravityRequestSideSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate(
@@ -291,7 +389,7 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
} else {
None
};
let provider_request_body = if let Some(antigravity_auth) = antigravity_auth.as_ref() {
let mut provider_request_body = if let Some(antigravity_auth) = antigravity_auth.as_ref() {
match build_antigravity_safe_v1internal_request(
antigravity_auth,
trace_id,
@@ -351,6 +449,16 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
} else {
base_provider_request_body
};
if crate::ai_serving::transport::enforce_same_format_provider_api_operation_body_policy(
&mut provider_request_body,
spec.operation,
) {
compatibility_edits.push(SameFormatProviderCompatibilityEdit {
field: "stream".to_string(),
action: SameFormatProviderCompatibilityEditAction::RuntimeRewrite,
detail: "removed stream field for non-streaming API operation".to_string(),
});
}
let is_grok = prepared
.transport
@@ -417,10 +525,10 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
original_request_body: body_json,
header_rules: transport.endpoint.header_rules.as_ref(),
behavior: prepared.behavior,
api_operation: spec.operation,
auth_header: prepared.auth_header.as_deref(),
auth_value: prepared.auth_value.as_deref(),
extra_headers: &extra_headers,
key_fingerprint: transport.key.fingerprint.as_ref(),
kiro_auth_config: prepared.kiro_auth.as_ref().map(|auth| &auth.auth_config),
kiro_machine_id: prepared
.kiro_auth
@@ -445,6 +553,28 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
.await;
return Ok(None);
};
crate::ai_serving::apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
transport.provider.provider_type.as_str(),
prepared.provider_api_format.as_str(),
Some(trace_id),
transport.key.decrypted_auth_config.as_deref(),
);
let provider_model = provider_request_body
.get("model")
.and_then(Value::as_str)
.unwrap_or(prepared.mapped_model.as_str());
crate::ai_serving::apply_codex_openai_responses_lite_header_for_request_body_with_capabilities(
&mut provider_request_headers,
Some(&provider_request_body),
transport.provider.provider_type.as_str(),
prepared.provider_api_format.as_str(),
provider_model,
source_model,
codex_model_capabilities.as_ref(),
);
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
redaction.redacted,
@@ -469,3 +599,98 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
request_redacted: redaction.redacted,
}))
}
fn same_format_provider_operation_skip_reason(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
operation: Option<crate::ai_serving::ApiOperation>,
) -> Option<&'static str> {
(!crate::ai_serving::transport::transport_supports_api_operation(
transport,
provider_api_format,
operation,
))
.then_some("transport_operation_unsupported")
}
#[cfg(test)]
mod tests {
use super::same_format_provider_operation_skip_reason;
use crate::ai_serving::transport::snapshot::{
GatewayProviderTransportEndpoint, GatewayProviderTransportKey,
GatewayProviderTransportProvider,
};
use crate::ai_serving::{ApiOperation, GatewayProviderTransportSnapshot};
fn private_adapter_transport(provider_type: &str) -> GatewayProviderTransportSnapshot {
GatewayProviderTransportSnapshot {
provider: GatewayProviderTransportProvider {
id: "provider-1".to_string(),
name: provider_type.to_string(),
provider_type: provider_type.to_string(),
website: None,
is_active: true,
keep_priority_on_conversion: false,
enable_format_conversion: true,
concurrent_limit: None,
max_retries: None,
proxy: None,
request_timeout_secs: None,
stream_first_byte_timeout_secs: None,
config: None,
},
endpoint: GatewayProviderTransportEndpoint {
id: "endpoint-1".to_string(),
provider_id: "provider-1".to_string(),
api_format: "claude:messages".to_string(),
api_family: Some("claude".to_string()),
endpoint_kind: Some("chat".to_string()),
is_active: true,
base_url: "https://private.example".to_string(),
header_rules: None,
body_rules: None,
max_retries: None,
custom_path: None,
config: None,
format_acceptance_config: None,
proxy: None,
},
key: GatewayProviderTransportKey {
id: "key-1".to_string(),
provider_id: "provider-1".to_string(),
name: "key".to_string(),
auth_type: "oauth".to_string(),
is_active: true,
api_formats: None,
auth_type_by_format: None,
allow_auth_channel_mismatch_formats: None,
allowed_models: None,
capabilities: None,
rate_multipliers: None,
global_priority_by_format: None,
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: String::new(),
decrypted_auth_config: None,
},
}
}
#[test]
fn private_adapter_count_tokens_is_rejected_by_pre_auth_operation_gate() {
for provider_type in ["kiro", "grok"] {
let transport = private_adapter_transport(provider_type);
assert_eq!(
same_format_provider_operation_skip_reason(
&transport,
"claude:messages",
Some(ApiOperation::ClaudeCountTokens),
),
Some("transport_operation_unsupported"),
"provider_type={provider_type}"
);
}
}
}
@@ -1,6 +1,6 @@
use crate::ai_serving::planner::spec_metadata::LocalExecutionSurfaceSpecMetadata;
use crate::ai_serving::transport::{
classify_same_format_provider_request_behavior as classify_same_format_provider_request_behavior_impl,
classify_same_format_provider_request_behavior_for_operation as classify_same_format_provider_request_behavior_impl,
resolve_same_format_provider_direct_auth as resolve_same_format_provider_direct_auth_impl,
same_format_provider_transport_supported as same_format_provider_transport_supported_impl,
same_format_provider_transport_unsupported_reason as same_format_provider_transport_unsupported_reason_impl,
@@ -15,6 +15,7 @@ pub(super) fn classify_same_format_provider_request_behavior(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
spec_metadata: LocalExecutionSurfaceSpecMetadata,
api_operation: Option<crate::ai_serving::ApiOperation>,
) -> SameFormatProviderRequestBehavior {
classify_same_format_provider_request_behavior_impl(
transport,
@@ -25,6 +26,7 @@ pub(super) fn classify_same_format_provider_request_behavior(
.report_kind
.expect("same-format provider specs should declare report kind"),
},
api_operation,
)
}
@@ -59,6 +59,7 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
&transport,
provider_api_format,
spec_metadata,
spec.operation,
);
if !same_format_provider_transport_supported(
@@ -190,7 +190,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalSameFormatProviderSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -213,6 +213,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalSameFormatProviderSyncA
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
@@ -220,7 +235,7 @@ impl LocalExecutionAttemptSource<AiStreamAttempt>
for LocalSameFormatProviderStreamAttemptSource<'_>
{
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -243,6 +258,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt>
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalSameFormatProviderSyncAttemptSource<'_> {
@@ -372,7 +402,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -458,7 +488,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -24,6 +24,7 @@ pub(crate) fn build_same_format_upstream_url(
upstream_is_stream,
request_query: parts.uri.query(),
kiro_api_region: kiro_auth.map(|auth| auth.auth_config.effective_api_region()),
api_operation: spec.operation,
provider_request_body,
},
)
@@ -127,7 +127,7 @@ fn provider_key_score_input(
oauth_invalid_reason: key.oauth_invalid_reason.clone(),
success_count: key.success_count.unwrap_or(0).into(),
error_count: key.error_count.unwrap_or(0).into(),
total_response_time_ms: key.total_response_time_ms.unwrap_or(0).into(),
total_response_time_ms: key.total_response_time_ms.unwrap_or(0),
total_tokens: key.total_tokens,
total_cost_usd: key.total_cost_usd,
last_used_at: key.last_used_at_unix_secs,
@@ -1,5 +1,5 @@
use std::borrow::Cow;
use std::time::{SystemTime, UNIX_EPOCH};
use std::time::{Instant, SystemTime, UNIX_EPOCH};
use serde_json::Value;
use tracing::warn;
@@ -7,9 +7,11 @@ use tracing::warn;
use crate::ai_serving::ExecutionRuntimeAuthContext;
use crate::privacy::{
build_redaction_session_config, read_chat_pii_redaction_runtime_config,
try_mask_chat_pii_request_json_with_cache_options, ChatPiiRedactionRequestFormat,
MaskChatRequestOptions, RedactionMaskError, RedactionSessionSlot, RedisRedactionMappingCache,
try_mask_chat_pii_request_value_with_cache_options, CachedRequestRedaction,
ChatPiiRedactionRequestFormat, MaskChatRequestOptions, RedactionMaskError,
RedactionSessionSlot, RedisRedactionMappingCache,
};
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{AppState, GatewayError};
pub(crate) struct ProviderRequestRedaction<'a> {
@@ -73,6 +75,20 @@ pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
let Some(slot) = parts.extensions.get::<RedactionSessionSlot>() else {
return Ok(ProviderRequestRedaction::disabled(body_json));
};
let request_cache_key = request_redaction_cache_key(format, body_json);
if let Some(cached) = slot.cached_request_redaction(&request_cache_key) {
crate::stage_metrics::record_chat_pii_redaction_request_cache_hit();
observe_gateway_stage_ms("chat_pii_redaction_request_cache_hit", 0);
return Ok(provider_redaction_from_cached(
slot,
candidate_id,
body_json,
cached,
));
}
crate::stage_metrics::record_chat_pii_redaction_request_cache_miss();
let runtime_config_started_at = Instant::now();
let runtime_config = read_chat_pii_redaction_runtime_config(state)
.await
.map_err(|err| {
@@ -82,11 +98,22 @@ pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
observe_gateway_stage_ms(
"chat_pii_redaction_runtime_config",
runtime_config_started_at.elapsed().as_millis() as u64,
);
if !runtime_config.enabled {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction::disabled(body_json));
}
let feature_settings_started_at = Instant::now();
let feature_settings = resolve_chat_pii_redaction_feature_settings(state, auth_context).await?;
observe_gateway_stage_ms(
"chat_pii_redaction_feature_settings",
feature_settings_started_at.elapsed().as_millis() as u64,
);
if !feature_settings.effective_enabled() {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction::disabled(body_json));
}
let Some(hmac_key) = state.encryption_key().map(str::as_bytes).map(Vec::from) else {
@@ -95,20 +122,14 @@ pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
"chat pii redaction setup failed".to_string(),
));
};
let body_bytes = serde_json::to_vec(body_json).map_err(|err| {
warn!(
error = ?err,
"gateway failed to serialize provider chat pii redaction body"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
let now_unix_secs = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_secs();
let cache = RedisRedactionMappingCache::new(state.runtime_state.as_ref());
let masked = try_mask_chat_pii_request_json_with_cache_options(
&body_bytes,
let mask_started_at = Instant::now();
let masked = try_mask_chat_pii_request_value_with_cache_options(
body_json,
format,
build_redaction_session_config(hmac_key, &runtime_config, now_unix_secs),
MaskChatRequestOptions::runtime(),
@@ -116,19 +137,27 @@ pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
)
.await
.map_err(redaction_mask_error_to_gateway_error)?;
observe_gateway_stage_ms(
"chat_pii_redaction_mask_body",
mask_started_at.elapsed().as_millis() as u64,
);
if !masked.redacted {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction {
body_json: Cow::Borrowed(body_json),
redacted: false,
});
}
let masked_body_json = serde_json::from_slice::<Value>(&masked.body).map_err(|err| {
warn!(
error = ?err,
"gateway failed to decode redacted provider chat pii body"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
let Some(masked_body_json) = masked.body_json else {
warn!("gateway pii redaction reported redacted without masked body");
return Err(GatewayError::Internal(
"chat pii redaction setup failed".to_string(),
));
};
slot.put_cached_request_redaction(
request_cache_key,
CachedRequestRedaction::redacted(masked_body_json.clone(), masked.session.clone()),
);
slot.put_for_candidate(candidate_id, masked.session);
Ok(ProviderRequestRedaction {
body_json: Cow::Owned(masked_body_json),
@@ -136,31 +165,46 @@ pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
})
}
fn request_redaction_cache_key(format: ChatPiiRedactionRequestFormat, body_json: &Value) -> String {
format!("{format:?}:{:p}", body_json)
}
fn provider_redaction_from_cached<'a>(
slot: &RedactionSessionSlot,
candidate_id: &str,
body_json: &'a Value,
cached: CachedRequestRedaction,
) -> ProviderRequestRedaction<'a> {
if !cached.redacted {
return ProviderRequestRedaction::disabled(body_json);
}
let Some(masked_body_json) = cached.body_json else {
return ProviderRequestRedaction::disabled(body_json);
};
if let Some(session) = cached.session {
slot.put_for_candidate(candidate_id, session);
}
ProviderRequestRedaction {
body_json: Cow::Owned(masked_body_json),
redacted: true,
}
}
async fn resolve_chat_pii_redaction_feature_settings(
state: &AppState,
auth_context: &ExecutionRuntimeAuthContext,
) -> Result<ChatPiiRedactionFeatureSettings, GatewayError> {
let user_settings = state
.read_user_feature_settings(&auth_context.user_id)
.await
let user_settings_fut = state.read_user_feature_settings(&auth_context.user_id);
let key_settings_fut = state.read_auth_api_key_feature_settings(
&auth_context.user_id,
&auth_context.api_key_id,
auth_context.api_key_is_standalone,
);
let (user_settings, key_settings) = tokio::try_join!(user_settings_fut, key_settings_fut)
.map_err(|err| {
warn!(
error = ?err,
"gateway failed to read user chat pii redaction feature settings"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
let key_settings = state
.read_auth_api_key_feature_settings(
&auth_context.user_id,
&auth_context.api_key_id,
auth_context.api_key_is_standalone,
)
.await
.map_err(|err| {
warn!(
error = ?err,
"gateway failed to read api key chat pii redaction feature settings"
"gateway failed to read chat pii redaction feature settings"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
@@ -172,12 +216,7 @@ async fn resolve_chat_pii_redaction_feature_settings(
}
fn redaction_mask_error_to_gateway_error(error: RedactionMaskError) -> GatewayError {
match error {
RedactionMaskError::Limit(limit) => GatewayError::Client {
status: limit.client_status(),
message: limit.safe_message().to_string(),
},
}
match error {}
}
#[cfg(test)]
@@ -6,6 +6,7 @@ use aether_ai_serving::{
provider_stream_event_api_format_for_provider_type as ai_provider_stream_event_api_format_for_provider_type,
AiExecutionReportContextParts, AiRequestOrigin,
};
use aether_routing_core::ResolvedRoutingPolicy;
use aether_runtime_state::RuntimeLockLease;
use aether_scheduler_core::{ClientSessionAffinity, SchedulerRankingOutcome};
use serde_json::{Map, Value};
@@ -20,8 +21,9 @@ use crate::client_session_affinity::{
};
use crate::orchestration::{
insert_pool_key_lease_report_context_fields, ExecutionAttemptIdentity,
SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD,
ROUTING_POOL_POLICY_OVERRIDE_REPORT_FIELD, SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD,
};
use crate::scheduler::affinity::insert_scheduler_affinity_policy_report_context_field;
pub(crate) struct LocalExecutionReportContextParts<'a> {
pub(crate) auth_context: &'a ExecutionRuntimeAuthContext,
@@ -55,6 +57,7 @@ pub(crate) struct LocalExecutionReportContextParts<'a> {
pub(crate) original_request_body_json: Option<&'a Value>,
pub(crate) original_request_body_base64: Option<&'a str>,
pub(crate) client_session_affinity: Option<&'a ClientSessionAffinity>,
pub(crate) routing_policy: Option<&'a ResolvedRoutingPolicy>,
pub(crate) scheduler_affinity_epoch: Option<u64>,
pub(crate) client_requested_stream: bool,
pub(crate) upstream_is_stream: bool,
@@ -105,6 +108,16 @@ pub(crate) fn build_local_execution_report_context(
merge_incoming_tls_fingerprint(&mut extra_fields, incoming_tls);
}
insert_pool_key_lease_report_context_fields(&mut extra_fields, parts.pool_key_lease);
insert_scheduler_affinity_policy_report_context_field(&mut extra_fields, parts.routing_policy);
if let Some(override_policy) = parts
.routing_policy
.and_then(|policy| policy.pool_policy_overrides.get(parts.provider_id))
.filter(|override_policy| !override_policy.scheduling_presets.is_empty())
{
if let Ok(value) = serde_json::to_value(override_policy) {
extra_fields.insert(ROUTING_POOL_POLICY_OVERRIDE_REPORT_FIELD.to_string(), value);
}
}
if let Some(epoch) = parts.scheduler_affinity_epoch {
extra_fields.insert(
SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD.to_string(),
@@ -315,6 +328,7 @@ mod tests {
original_request_body_json: Some(&json!({"model": "gpt-5"})),
original_request_body_base64: None,
client_session_affinity: Some(&client_session_affinity),
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: false,
@@ -397,6 +411,7 @@ mod tests {
})),
original_request_body_base64: None,
client_session_affinity: None,
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: true,
@@ -463,6 +478,7 @@ mod tests {
original_request_body_json: Some(&json!({"model": "gpt-5"})),
original_request_body_base64: None,
client_session_affinity: None,
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: false,
@@ -1,23 +1,50 @@
use aether_ai_serving::AiRequestGzipPolicy;
use serde_json::Value;
use crate::ai_serving::is_openai_responses_family_format;
use crate::ai_serving::{normalize_api_format_alias, parse_codex_auth_identity};
use super::state::GatewayProviderTransportSnapshot;
const DEFAULT_CODEX_REQUEST_GZIP_MIN_BYTES: usize = 64 * 1024;
pub(crate) fn resolve_transport_request_gzip_policy(
transport: &GatewayProviderTransportSnapshot,
) -> Option<AiRequestGzipPolicy> {
transport_request_gzip_policy_from_config(transport.endpoint.config.as_ref())
.or_else(|| transport_request_gzip_policy_from_config(transport.provider.config.as_ref()))
.or_else(|| default_transport_request_gzip_policy(transport))
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub(crate) struct TransportRequestEncodingPolicy {
pub content_encoding: Option<String>,
pub request_gzip: Option<AiRequestGzipPolicy>,
}
fn default_transport_request_gzip_policy(
pub(crate) fn resolve_transport_request_encoding_policy(
transport: &GatewayProviderTransportSnapshot,
) -> Option<AiRequestGzipPolicy> {
) -> TransportRequestEncodingPolicy {
if transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex")
&& normalize_api_format_alias(transport.endpoint.api_format.as_str())
== "openai:responses:compact"
{
return TransportRequestEncodingPolicy::default();
}
let request_gzip = transport_request_gzip_policy_from_config(
transport.endpoint.config.as_ref(),
)
.or_else(|| transport_request_gzip_policy_from_config(transport.provider.config.as_ref()));
if request_gzip.is_some() {
return TransportRequestEncodingPolicy {
content_encoding: None,
request_gzip,
};
}
TransportRequestEncodingPolicy {
content_encoding: default_transport_request_content_encoding(transport),
request_gzip: None,
}
}
fn default_transport_request_content_encoding(
transport: &GatewayProviderTransportSnapshot,
) -> Option<String> {
if !transport
.provider
.provider_type
@@ -26,19 +53,24 @@ fn default_transport_request_gzip_policy(
{
return None;
}
if !is_codex_request_gzip_endpoint_api_format(transport.endpoint.api_format.as_str()) {
if !is_codex_request_compression_api_format(transport.endpoint.api_format.as_str()) {
return None;
}
let auth_type =
crate::ai_serving::transport::auth::resolve_local_auth_type_for_transport_format(transport);
let uses_codex_backend = auth_type == "oauth"
|| (auth_type == "bearer"
&& parse_codex_auth_identity(transport.key.decrypted_auth_config.as_deref())
.uses_codex_backend);
if !uses_codex_backend {
return None;
}
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(DEFAULT_CODEX_REQUEST_GZIP_MIN_BYTES),
})
Some("zstd".to_string())
}
fn is_codex_request_gzip_endpoint_api_format(api_format: &str) -> bool {
is_openai_responses_family_format(api_format)
|| api_format.trim().eq_ignore_ascii_case("openai:image")
fn is_codex_request_compression_api_format(api_format: &str) -> bool {
normalize_api_format_alias(api_format) == "openai:responses"
}
fn transport_request_gzip_policy_from_config(
@@ -216,6 +248,16 @@ mod tests {
}
}
fn resolved_gzip_policy(
transport: &GatewayProviderTransportSnapshot,
) -> Option<AiRequestGzipPolicy> {
resolve_transport_request_encoding_policy(transport).request_gzip
}
fn resolved_content_encoding(transport: &GatewayProviderTransportSnapshot) -> Option<String> {
resolve_transport_request_encoding_policy(transport).content_encoding
}
#[test]
fn endpoint_request_gzip_policy_overrides_provider_policy() {
let transport = sample_transport(
@@ -226,7 +268,7 @@ mod tests {
);
assert_eq!(
resolve_transport_request_gzip_policy(&transport),
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1024),
@@ -244,7 +286,7 @@ mod tests {
);
assert_eq!(
resolve_transport_request_gzip_policy(&transport),
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
@@ -265,7 +307,7 @@ mod tests {
);
assert_eq!(
resolve_transport_request_gzip_policy(&transport),
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(4096),
@@ -283,7 +325,7 @@ mod tests {
);
assert_eq!(
resolve_transport_request_gzip_policy(&transport),
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1),
@@ -292,35 +334,73 @@ mod tests {
}
#[test]
fn codex_responses_endpoint_gets_default_request_gzip_policy() {
let transport = sample_transport("codex", "openai:responses", None, None);
fn codex_responses_endpoint_uses_zstd_without_a_size_threshold() {
let mut transport = sample_transport("codex", "openai:responses", None, None);
transport.key.auth_type = "oauth".to_string();
assert_eq!(
resolve_transport_request_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(DEFAULT_CODEX_REQUEST_GZIP_MIN_BYTES),
})
resolved_content_encoding(&transport).as_deref(),
Some("zstd")
);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_image_endpoint_gets_default_request_gzip_policy() {
let transport = sample_transport("codex", "openai:image", None, None);
fn codex_responses_api_key_auth_does_not_enable_default_compression() {
let transport = sample_transport("codex", "openai:responses", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_responses_bearer_auth_uses_identity_metadata_for_backend_compression() {
let mut transport = sample_transport("codex", "openai:responses", None, None);
transport.key.auth_type = "bearer".to_string();
transport.key.decrypted_auth_config =
Some(r#"{"provider_type":"codex","account_id":"account-1"}"#.to_string());
assert_eq!(
resolve_transport_request_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(DEFAULT_CODEX_REQUEST_GZIP_MIN_BYTES),
})
resolved_content_encoding(&transport).as_deref(),
Some("zstd")
);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_image_endpoint_does_not_get_responses_request_gzip_policy() {
let transport = sample_transport("codex", "openai:image", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_compact_endpoint_does_not_get_default_request_gzip_policy() {
let transport = sample_transport("codex", "openai:responses:compact", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_compact_endpoint_rejects_an_explicit_request_gzip_policy() {
let transport = sample_transport(
"codex",
"openai:responses:compact",
None,
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 2048}})),
);
assert_eq!(resolved_gzip_policy(&transport), None);
assert_eq!(resolved_content_encoding(&transport), None);
}
#[test]
fn non_codex_endpoint_does_not_get_default_request_gzip_policy() {
let transport = sample_transport("openai", "openai:responses", None, None);
assert_eq!(resolve_transport_request_gzip_policy(&transport), None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
}
@@ -1,8 +1,8 @@
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{
is_matching_stream_http_request as is_matching_stream_http_request_impl,
resolve_execution_runtime_stream_plan_kind as resolve_execution_runtime_stream_plan_kind_impl,
resolve_execution_runtime_sync_plan_kind as resolve_execution_runtime_sync_plan_kind_impl,
resolve_execution_runtime_stream_plan_kind_with_client_surface as resolve_execution_runtime_stream_plan_kind_impl,
resolve_execution_runtime_sync_plan_kind_with_client_surface as resolve_execution_runtime_sync_plan_kind_impl,
supports_stream_execution_decision_kind as supports_stream_execution_decision_kind_impl,
supports_sync_execution_decision_kind as supports_sync_execution_decision_kind_impl,
};
@@ -11,28 +11,34 @@ pub(crate) fn resolve_execution_runtime_stream_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
resolve_execution_runtime_stream_plan_kind_impl(
let plan_kind = resolve_execution_runtime_stream_plan_kind_impl(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, true, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn resolve_execution_runtime_sync_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
resolve_execution_runtime_sync_plan_kind_impl(
let plan_kind = resolve_execution_runtime_sync_plan_kind_impl(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, false, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn is_matching_stream_request(
@@ -62,7 +68,7 @@ mod tests {
resolve_execution_runtime_sync_plan_kind, supports_stream_execution_decision_kind,
supports_sync_execution_decision_kind,
};
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{ApiOperation, ClientSurface, GatewayControlDecision};
fn sample_decision(route_family: &str, route_kind: &str) -> GatewayControlDecision {
GatewayControlDecision {
@@ -71,12 +77,16 @@ mod tests {
route_class: Some("ai_public".to_string()),
route_family: Some(route_family.to_string()),
route_kind: Some(route_kind.to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_context: None,
admin_principal: None,
auth_endpoint_signature: None,
execution_runtime_candidate: true,
local_auth_rejection: None,
model_directive_policy: Default::default(),
}
}
@@ -120,7 +130,9 @@ mod tests {
let (claude_parts, _) = claude_request.into_parts();
let claude_api_key = sample_decision_with_auth_channel("claude", "messages", "api_key");
let claude_bearer = sample_decision_with_auth_channel("claude", "messages", "bearer_like");
let mut claude_bearer =
sample_decision_with_auth_channel("claude", "messages", "bearer_like");
claude_bearer.client_surface = Some(ClientSurface::ClaudeCode);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&claude_parts, &claude_api_key),
Some("claude_chat_sync")
@@ -130,6 +142,13 @@ mod tests {
Some("claude_cli_stream")
);
let claude_sdk_bearer =
sample_decision_with_auth_channel("claude", "messages", "bearer_like");
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&claude_parts, &claude_sdk_bearer),
Some("claude_chat_sync")
);
let gemini_request = Request::builder()
.method(Method::POST)
.uri("/v1beta/models/gemini-2.5-pro:generateContent")
@@ -151,6 +170,36 @@ mod tests {
);
}
#[test]
fn resolves_claude_count_tokens_as_native_sync_operation() {
let request = Request::builder()
.method(Method::POST)
.uri("/v1/messages/count_tokens")
.body(())
.expect("request should build");
let (parts, _) = request.into_parts();
let mut decision = sample_decision("claude", "count_tokens");
decision.api_operation = Some(ApiOperation::ClaudeCountTokens);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&parts, &decision),
Some("claude_count_tokens_sync")
);
assert!(supports_sync_execution_decision_kind(
"claude_count_tokens_sync"
));
decision.api_operation = Some(ApiOperation::ClaudeMessagesCreate);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&parts, &decision),
None
);
assert_eq!(
resolve_execution_runtime_stream_plan_kind(&parts, &decision),
None
);
}
#[test]
fn stream_matching_uses_surface_route_logic() {
let request = Request::builder()
@@ -175,7 +175,7 @@ pub(crate) async fn build_local_gemini_files_stream_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalGeminiFilesSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -193,12 +193,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalGeminiFilesSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalGeminiFilesStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -216,6 +231,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalGeminiFilesStreamAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalGeminiFilesSyncAttemptSource<'_> {
@@ -323,7 +353,7 @@ pub(crate) async fn maybe_build_sync_local_gemini_files_decision_payload(
let (mut source, _) =
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -365,7 +395,7 @@ pub(crate) async fn maybe_build_stream_local_gemini_files_decision_payload(
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
let empty_body_json = serde_json::Value::Null;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -414,7 +444,7 @@ async fn build_local_sync_plan_and_reports(
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -467,7 +497,7 @@ async fn build_local_stream_plan_and_reports(
let mut plans = Vec::new();
let empty_body_json = serde_json::Value::Null;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -7,14 +7,14 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_gemini_files_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{AiExecutionDecision, AppState, GatewayError};
use crate::{append_local_failover_policy_to_value, AiExecutionDecision, AppState, GatewayError};
use super::request::resolve_local_gemini_files_candidate_payload_parts;
use super::support::{
@@ -107,6 +107,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
original_request_body_json: Some(body_json),
original_request_body_base64: resolved.provider_request_body_base64.as_deref(),
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: spec_metadata.require_streaming,
upstream_is_stream: spec_metadata.require_streaming,
@@ -114,6 +115,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
needs_conversion: false,
extra_fields,
});
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let super::request::LocalGeminiFilesCandidatePayloadParts {
transport: _,
auth_header,
@@ -124,7 +126,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
upstream_url,
file_name: _,
} = resolved;
let request_gzip = resolve_transport_request_gzip_policy(&transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -134,6 +136,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -156,8 +159,8 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -166,6 +169,10 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -53,7 +53,9 @@ pub(super) async fn resolve_local_gemini_files_decision_input(
state,
auth_context,
None,
decision.auth_endpoint_signature.as_deref(),
Some(&explicit_required_capabilities),
&decision.model_directive_policy,
)
.await
{
@@ -253,7 +253,7 @@ pub(crate) async fn build_local_image_stream_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiImageSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -271,12 +271,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiImageSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiImageStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -294,6 +309,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiImageStreamAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiImageSyncAttemptSource<'_> {
@@ -421,7 +451,7 @@ pub(crate) async fn maybe_build_sync_local_image_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -482,7 +512,7 @@ pub(crate) async fn maybe_build_stream_local_image_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -540,7 +570,7 @@ async fn build_local_sync_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -617,7 +647,7 @@ async fn build_local_stream_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -5,7 +5,7 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_openai_image_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
@@ -13,7 +13,8 @@ use crate::ai_serving::transport::{
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{
append_execution_contract_fields_to_value, AiExecutionDecision, AppState, GatewayError,
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_image_candidate_payload_parts;
@@ -84,11 +85,7 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
serde_json::Value::Bool(true),
);
}
let upstream_is_stream = resolved
.provider_request_body
.get("stream")
.and_then(serde_json::Value::as_bool)
.unwrap_or(spec_metadata.require_streaming);
let upstream_is_stream = resolved.upstream_is_stream;
let effective_headers = input.effective_headers(&parts.headers);
let report_context = append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -123,6 +120,7 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
original_request_body_json: Some(body_json),
original_request_body_base64: body_base64,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: spec_metadata.require_streaming,
upstream_is_stream,
@@ -135,7 +133,8 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
spec_metadata.api_format,
provider_api_format.as_str(),
);
let request_gzip = resolve_transport_request_gzip_policy(&transport);
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -145,6 +144,7 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -162,8 +162,8 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
provider_request_body: Some(resolved.provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -172,6 +172,10 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -16,8 +16,8 @@ use crate::ai_serving::transport::{
ProviderOpenAiImageHeadersInput, StandardProviderRequestHeadersInput, GROK_CHAT_PATH,
};
use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
build_chatgpt_web_image_request_body,
apply_codex_openai_special_headers, build_chatgpt_web_image_request_body,
build_codex_openai_image_api_provider_request_body,
build_gemini_image_request_body_from_openai_image_request,
build_openai_image_api_provider_request_body, build_openai_image_provider_request_body,
default_model_for_openai_image_operation, normalize_openai_image_request,
@@ -48,6 +48,7 @@ pub(super) struct LocalOpenAiImageCandidatePayloadParts {
pub(super) upstream_url: String,
pub(super) input_summary: Value,
pub(super) transport_profile: Option<ResolvedTransportProfile>,
pub(super) upstream_is_stream: bool,
}
pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
@@ -130,7 +131,10 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
parts,
body_json,
body_base64,
openai_image_normalize_options_for_provider(&transport.provider.provider_type),
openai_image_normalize_options_for_provider(
&transport.provider.provider_type,
Some(prepared_candidate.mapped_model.as_str()),
),
);
let Some(normalized_request) = normalized_request else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
@@ -174,29 +178,56 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
} else {
build_openai_image_upstream_url(transport, Some(parts.uri.path()), parts.uri.query())
};
let mut provider_request_body = if is_chatgpt_web {
match build_chatgpt_web_image_request_body(parts, body_json, body_base64) {
Ok(body) => body,
Err(err) => err.to_error_json(),
}
} else if is_codex || is_grok {
build_openai_image_provider_request_body(&normalized_request)
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
spec_metadata.api_format,
spec_metadata.require_streaming && candidate.supports_streaming,
false,
);
let provider_request_body = if is_chatgpt_web {
Some(
match build_chatgpt_web_image_request_body(parts, body_json, body_base64) {
Ok(body) => body,
Err(err) => err.to_error_json(),
},
)
} else if is_codex {
build_codex_openai_image_api_provider_request_body(
&normalized_request,
Some(prepared_candidate.mapped_model.as_str()),
upstream_is_stream,
)
} else if is_grok {
Some(build_openai_image_provider_request_body(
&normalized_request,
))
} else {
build_openai_image_api_provider_request_body(
&normalized_request,
Some(prepared_candidate.mapped_model.as_str()),
upstream_is_stream,
)
};
if !is_chatgpt_web {
apply_codex_openai_responses_special_body_edits(
&mut provider_request_body,
transport.provider.provider_type.as_str(),
spec_metadata.api_format,
transport.endpoint.body_rules.as_ref(),
Some(candidate.key_id.as_str()),
);
}
let Some(provider_request_body) = provider_request_body else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_missing",
CandidateFailureDiagnostic::provider_request_body_missing(
spec_metadata.api_format,
spec_metadata.api_format,
"codex_openai_images_request_contract",
),
)
.await;
return None;
};
let Some(mut provider_request_headers) = (if is_grok {
build_grok_browser_headers(GrokHeaderInput {
transport,
@@ -210,9 +241,17 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
})
} else {
build_openai_image_headers(ProviderOpenAiImageHeadersInput {
transport,
headers: effective_headers,
auth_header: &auth_header,
auth_value: &auth_value,
accept: if is_codex {
None
} else if upstream_is_stream {
Some("text/event-stream")
} else {
Some("application/json")
},
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &provider_request_body,
original_request_body: body_json,
@@ -239,7 +278,7 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
provider_request_headers.insert("x-aether-chatgpt-web-image".to_string(), "1".to_string());
} else if is_grok {
} else {
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
@@ -281,6 +320,7 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
upstream_url,
input_summary,
transport_profile,
upstream_is_stream,
})
}
@@ -397,7 +437,14 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
return None;
}
};
let upstream_is_stream = spec_metadata.require_streaming;
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
provider_api_format,
spec_metadata.require_streaming && candidate.supports_streaming,
false,
);
let Some(upstream_url) = crate::ai_serving::planner::standard::build_standard_upstream_url(
parts,
transport,
@@ -468,6 +515,7 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
upstream_url,
input_summary: converted.summary_json,
transport_profile: None,
upstream_is_stream,
})
}
@@ -58,7 +58,9 @@ pub(super) async fn resolve_local_openai_image_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -124,6 +126,7 @@ pub(super) async fn list_local_openai_image_candidate_attempts(
matches_client_format.then_some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -144,8 +147,8 @@ pub(super) async fn list_local_openai_image_candidate_attempts(
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
candidate,
false,
)
});
}
@@ -197,6 +200,7 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
matches_client_format.then_some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -206,16 +210,16 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
candidate,
false,
)
});
format_skipped.retain(|candidate| {
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
&candidate.candidate,
false,
)
});
}
@@ -105,7 +105,7 @@ pub(crate) async fn build_local_video_sync_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalVideoCreateSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -123,6 +123,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalVideoCreateSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalVideoCreateSyncAttemptSource<'_> {
@@ -195,7 +210,7 @@ pub(crate) async fn maybe_build_sync_local_video_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
state, parts, body_json, trace_id, &input, attempt, spec,
)
@@ -240,7 +255,7 @@ async fn build_local_sync_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
state, parts, body_json, trace_id, &input, attempt, spec,
)
@@ -5,14 +5,14 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_video_create_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{AiExecutionDecision, AppState, GatewayError};
use crate::{append_local_failover_policy_to_value, AiExecutionDecision, AppState, GatewayError};
use super::request::resolve_local_video_create_candidate_payload_parts;
use super::support::{LocalVideoCreateCandidateAttempt, LocalVideoCreateDecisionInput};
@@ -88,6 +88,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
original_request_body_json: Some(body_json),
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: false,
upstream_is_stream: false,
@@ -95,6 +96,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
needs_conversion: false,
extra_fields,
});
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let super::request::LocalVideoCreateCandidatePayloadParts {
transport: _,
auth_header,
@@ -104,7 +106,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
provider_request_body,
upstream_url,
} = resolved;
let request_gzip = resolve_transport_request_gzip_policy(&transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: false,
@@ -114,6 +116,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -137,8 +140,8 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -147,6 +150,10 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -62,7 +62,9 @@ pub(super) async fn resolve_local_video_create_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -130,6 +132,7 @@ pub(super) async fn list_local_video_create_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -186,6 +189,7 @@ pub(super) async fn build_local_video_create_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -3,5 +3,43 @@
mod tests;
pub(crate) use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_identity_headers, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_special_headers,
};
pub(crate) fn codex_model_capabilities_for_transport(
transport: &crate::ai_serving::GatewayProviderTransportSnapshot,
provider_api_format: &str,
provider_model: &str,
source_model: &str,
) -> Option<crate::ai_serving::CodexResponsesModelCapabilities> {
codex_model_capabilities(
&transport.provider.provider_type,
provider_api_format,
provider_model,
source_model,
transport.key.upstream_metadata.as_ref(),
)
}
fn codex_model_capabilities(
provider_type: &str,
provider_api_format: &str,
provider_model: &str,
source_model: &str,
upstream_metadata: Option<&serde_json::Value>,
) -> Option<crate::ai_serving::CodexResponsesModelCapabilities> {
let uses_codex_model_catalog =
crate::ai_serving::is_openai_responses_family_format(provider_api_format)
|| crate::ai_serving::api_format_alias_matches(provider_api_format, "openai:search");
if !provider_type.trim().eq_ignore_ascii_case("codex") || !uses_codex_model_catalog {
return None;
}
Some(
crate::ai_serving::resolve_codex_responses_model_capabilities(
provider_model,
source_model,
upstream_metadata,
),
)
}
@@ -1,16 +1,58 @@
use std::collections::BTreeMap;
use super::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_identity_headers, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_special_headers, codex_model_capabilities,
};
use crate::ai_serving::planner::standard::{
build_cross_format_openai_responses_request_body, build_local_openai_responses_request_body,
};
use crate::ai_serving::planner::standard::build_local_openai_responses_request_body;
use http::{HeaderMap, HeaderValue};
use serde_json::json;
#[test]
fn search_uses_live_codex_model_catalog_capabilities() {
let metadata = crate::ai_serving::build_codex_model_catalog_metadata(&[json!({
"slug": "gpt-search-custom",
"default_reasoning_level": "low",
"supported_reasoning_levels": [
{"effort": "low"},
{"effort": "max"}
],
"supports_parallel_tool_calls": true
})]);
let capabilities = codex_model_capabilities(
"codex",
"openai:search",
"gpt-search-custom",
"gpt-search-custom",
Some(&metadata),
)
.expect("Search should resolve capabilities from the Codex model catalog");
assert_eq!(
capabilities.default_reasoning_effort.as_deref(),
Some("low")
);
assert_eq!(
capabilities.supported_reasoning_efforts,
vec!["low".to_string(), "max".to_string()]
);
assert!(codex_model_capabilities(
"codex",
"openai:chat",
"gpt-search-custom",
"gpt-search-custom",
Some(&metadata),
)
.is_none());
}
#[test]
fn applies_codex_defaults_when_body_rules_do_not_handle_fields() {
let mut body = json!({
"model": "gpt-5",
"model": "gpt-5.4",
"max_output_tokens": 128,
"temperature": 0.3,
"top_p": 0.9,
@@ -31,10 +73,11 @@ fn applies_codex_defaults_when_body_rules_do_not_handle_fields() {
assert!(body.get("top_p").is_none());
assert!(body.get("metadata").is_none());
assert_eq!(body["store"], false);
assert_eq!(body["instructions"], "");
assert!(body.get("instructions").is_none());
assert_eq!(body["include"], json!(["reasoning.encrypted_content"]));
assert_eq!(body["parallel_tool_calls"], true);
assert!(body.get("reasoning").is_none());
assert_eq!(body["reasoning"]["effort"], "medium");
assert!(body["reasoning"].get("summary").is_none());
}
#[test]
@@ -80,7 +123,7 @@ fn strips_store_for_compact_even_when_body_rules_handle_it() {
{"action":"set","path":"top_p","value":0.5}
]);
let mut body = json!({
"model": "gpt-5",
"model": "gpt-5.4",
"max_output_tokens": 128,
"metadata": {"client": "desktop", "mode": "custom"},
"store": true,
@@ -99,12 +142,29 @@ fn strips_store_for_compact_even_when_body_rules_handle_it() {
assert!(body.get("max_output_tokens").is_none());
assert!(body.get("store").is_none());
assert_eq!(body["instructions"], "Keep custom");
assert_eq!(body["metadata"]["mode"], "custom");
assert_eq!(body["top_p"], 0.5);
assert!(body.get("metadata").is_none());
assert!(body.get("top_p").is_none());
assert_eq!(body["parallel_tool_calls"], true);
assert!(body.as_object().is_some_and(|object| {
object.keys().all(|field| {
matches!(
field.as_str(),
"model"
| "input"
| "instructions"
| "tools"
| "parallel_tool_calls"
| "reasoning"
| "service_tier"
| "prompt_cache_key"
| "text"
)
})
}));
}
#[test]
fn injects_stable_prompt_cache_key_for_codex_requests() {
fn does_not_synthesize_prompt_cache_key_from_api_key_identity() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
@@ -118,83 +178,478 @@ fn injects_stable_prompt_cache_key_for_codex_requests() {
Some("key-123"),
);
assert!(body.get("prompt_cache_key").is_none());
}
#[test]
fn adapts_generic_prompt_cache_key_to_codex_native_identity() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
"prompt_cache_key": "ltm-pc-v2-5557e02f5c9b447a97673ba330dbe77a",
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
let expected_identity = "d9c5d122-7c1c-5fb1-ba9d-656062eda44e";
assert_eq!(body["prompt_cache_key"], expected_identity);
assert_eq!(body["client_metadata"]["session_id"], expected_identity);
assert_eq!(body["client_metadata"]["thread_id"], expected_identity);
}
#[test]
fn preserves_native_codex_cache_identity_and_metadata() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
"prompt_cache_key": "guardian:parent-thread",
"client_metadata": {
"session_id": "native-session",
"thread_id": "native-thread",
"turn_id": "native-turn"
}
});
let expected = body.clone();
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
assert_eq!(body["prompt_cache_key"], expected["prompt_cache_key"]);
assert_eq!(body["client_metadata"], expected["client_metadata"]);
}
#[test]
fn preserves_uuid_prompt_cache_key_while_completing_codex_identity() {
let identity = "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3";
let mut body = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": identity
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
None,
);
assert_eq!(body["prompt_cache_key"], identity);
assert_eq!(body["client_metadata"]["session_id"], identity);
assert_eq!(body["client_metadata"]["thread_id"], identity);
}
#[test]
fn keeps_codex_prompt_cache_domains_distinct() {
let mut first = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "tenant-a"
});
let mut second = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "tenant-b"
});
for body in [&mut first, &mut second] {
apply_codex_openai_responses_special_body_edits(
body,
"codex",
"openai:responses",
None,
None,
);
}
assert_ne!(first["prompt_cache_key"], second["prompt_cache_key"]);
assert_eq!(
body["prompt_cache_key"],
"53363264-dbb0-5f9d-b9c7-3e92c45c5bdf"
first["prompt_cache_key"],
first["client_metadata"]["session_id"]
);
assert_eq!(
second["prompt_cache_key"],
second["client_metadata"]["session_id"]
);
}
#[test]
fn keeps_existing_prompt_cache_key_for_codex_requests() {
let mut body = json!({
"model": "gpt-5",
fn completes_partial_and_null_codex_client_metadata() {
let mut partial = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "existing-key",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"thread_id": "native-thread",
"caller": "sdk"
}
});
let mut null_metadata = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": null
});
let mut null_session = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"session_id": null,
"thread_id": null,
"caller": "sdk"
}
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
for body in [&mut partial, &mut null_metadata, &mut null_session] {
apply_codex_openai_responses_special_body_edits(
body,
"codex",
"openai:responses",
None,
None,
);
}
assert_eq!(body["prompt_cache_key"], "existing-key");
assert_eq!(partial["client_metadata"]["thread_id"], "native-thread");
assert_eq!(partial["client_metadata"]["caller"], "sdk");
assert_eq!(
partial["client_metadata"]["session_id"],
partial["prompt_cache_key"]
);
assert_eq!(
null_metadata["client_metadata"]["session_id"],
null_metadata["prompt_cache_key"]
);
assert_eq!(
null_metadata["client_metadata"]["thread_id"],
null_metadata["prompt_cache_key"]
);
assert_eq!(
null_session["client_metadata"]["session_id"],
null_session["prompt_cache_key"]
);
assert_eq!(
null_session["client_metadata"]["thread_id"],
null_session["prompt_cache_key"]
);
assert_eq!(null_session["client_metadata"]["caller"], "sdk");
}
#[test]
fn injects_chatgpt_account_id_and_session_headers_for_codex_requests() {
fn leaves_malformed_codex_client_metadata_unchanged() {
let mut body = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": "invalid"
});
let mut malformed_fields = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"session_id": 42,
"thread_id": ""
}
});
let expected_malformed_metadata = malformed_fields["client_metadata"].clone();
for candidate in [&mut body, &mut malformed_fields] {
apply_codex_openai_responses_special_body_edits(
candidate,
"codex",
"openai:responses",
None,
None,
);
}
assert_eq!(body["prompt_cache_key"], "generic-affinity");
assert_eq!(body["client_metadata"], "invalid");
assert_eq!(malformed_fields["prompt_cache_key"], "generic-affinity");
assert_eq!(
malformed_fields["client_metadata"],
expected_malformed_metadata
);
}
#[test]
fn limits_prompt_cache_identity_adaptation_to_codex_responses_family() {
let original = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity"
});
let mut standard_openai = original.clone();
let mut codex_compact = original.clone();
apply_codex_openai_responses_special_body_edits(
&mut standard_openai,
"openai",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_special_body_edits(
&mut codex_compact,
"codex",
"openai:responses:compact",
None,
None,
);
assert_eq!(standard_openai, original);
assert_ne!(codex_compact["prompt_cache_key"], "generic-affinity");
assert!(codex_compact.get("client_metadata").is_none());
}
#[test]
fn chat_to_codex_responses_adapts_prompt_cache_identity_end_to_end() {
let body = json!({
"model": "gpt-5.6-luna",
"messages": [{"role": "user", "content": "hello"}],
"prompt_cache_key": "ltm-pc-v2-5557e02f5c9b447a97673ba330dbe77a"
});
let provider_request_body = build_cross_format_openai_responses_request_body(
&body,
"gpt-5.6-luna",
"openai:chat",
"openai:responses",
true,
false,
"codex",
None,
None,
&HeaderMap::new(),
false,
)
.expect("chat to Codex Responses request should build");
let expected_identity = "d9c5d122-7c1c-5fb1-ba9d-656062eda44e";
assert_eq!(provider_request_body["prompt_cache_key"], expected_identity);
assert_eq!(
provider_request_body["client_metadata"]["session_id"],
expected_identity
);
assert_eq!(
provider_request_body["client_metadata"]["thread_id"],
expected_identity
);
let mut provider_request_headers = BTreeMap::new();
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
&HeaderMap::new(),
"codex",
"openai:responses",
Some("trace-codex-cache-identity"),
None,
);
apply_codex_openai_responses_identity_headers(
&mut provider_request_headers,
&provider_request_body,
"codex",
"openai:responses",
);
assert_eq!(
provider_request_headers
.get("session-id")
.map(String::as_str),
Some(expected_identity)
);
assert_eq!(
provider_request_headers
.get("thread-id")
.map(String::as_str),
Some(expected_identity)
);
}
#[test]
fn projects_uuid_prompt_cache_identity_into_missing_session_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
});
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
Some("trace-codex-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(headers.get("x-client-request-id"), None);
assert_eq!(
headers.get("user-agent"),
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
}
#[test]
fn projects_native_codex_metadata_for_non_uuid_cache_overrides() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "guardian:parent-thread",
"client_metadata": {
"session_id": "019f687b-8e92-7842-9631-d5bf0dba0a3b",
"thread_id": "019f6d20-1111-7222-8333-444455556666"
}
});
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("session-id").map(String::as_str),
Some("019f687b-8e92-7842-9631-d5bf0dba0a3b")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("019f6d20-1111-7222-8333-444455556666")
);
}
#[test]
fn leaves_non_native_cache_keys_out_of_identity_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "generic-cache-key"
});
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert!(!headers.contains_key("session-id"));
assert!(!headers.contains_key("thread-id"));
}
#[test]
fn leaves_malformed_native_metadata_out_of_identity_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
"client_metadata": {
"session_id": 42,
"thread_id": ""
}
});
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert!(!headers.contains_key("session-id"));
assert!(!headers.contains_key("thread-id"));
}
#[test]
fn injects_only_codex_client_headers_for_images_requests() {
let mut headers = BTreeMap::new();
apply_codex_openai_special_headers(
&mut headers,
&json!({
"model": "gpt-image-2",
"prompt": "draw a city"
}),
&HeaderMap::new(),
"codex",
"openai:image",
Some("trace-codex-image-123"),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(
headers.get("x-client-request-id"),
Some(&"trace-codex-123".to_string())
);
assert_eq!(
headers.get("user-agent"),
Some(
&"codex-tui/0.122.0 (Mac OS 15.2.0; arm64) vscode/2.6.11 (codex-tui; 0.122.0)"
.to_string()
)
);
assert_eq!(headers.get("originator"), Some(&"codex-tui".to_string()));
assert_eq!(
headers.get("session_id"),
Some(&"ab5ecce4f0d110fe".to_string())
);
assert_eq!(
headers.get("conversation_id"),
Some(&"ab5ecce4f0d110fe".to_string())
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
for name in ["x-client-request-id", "session-id", "thread-id"] {
assert!(
!headers.contains_key(name),
"unexpected Images header: {name}"
);
}
}
#[test]
fn respects_existing_codex_request_and_session_headers() {
fn preserves_client_context_headers_and_enforces_codex_provider_identity() {
let mut headers = BTreeMap::new();
headers.insert(
"x-client-request-id".to_string(),
"kept-by-rule-request".to_string(),
);
headers.insert("session_id".to_string(), "kept-by-rule".to_string());
headers.insert("session-id".to_string(), "kept-by-rule-session".to_string());
headers.insert("thread-id".to_string(), "kept-by-rule-thread".to_string());
headers.insert(
"chatgpt-account-id".to_string(),
"configured-spoof".to_string(),
);
headers.insert(
"x-openai-fedramp".to_string(),
"configured-false".to_string(),
);
headers.insert(
"User-Agent".to_string(),
"AsyncOpenAI/Python 2.44.0".to_string(),
);
headers.insert("ORIGINATOR".to_string(), "sdk-client".to_string());
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
@@ -205,12 +660,12 @@ fn respects_existing_codex_request_and_session_headers() {
HeaderValue::from_static("user-specified-request"),
);
original_headers.insert(
"session_id",
"session-id",
HeaderValue::from_static("user-specified-session"),
);
original_headers.insert(
"conversation_id",
HeaderValue::from_static("user-specified-conversation"),
"thread-id",
HeaderValue::from_static("user-specified-thread"),
);
original_headers.insert(
"user-agent",
@@ -220,64 +675,105 @@ fn respects_existing_codex_request_and_session_headers() {
"originator",
HeaderValue::from_static("user-specified-originator"),
);
original_headers.insert("version", HeaderValue::from_static("user-version"));
original_headers.insert("x-openai-fedramp", HeaderValue::from_static("user-fedramp"));
original_headers.insert(
"chatgpt-account-id",
HeaderValue::from_static("user-account"),
);
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&original_headers,
"codex",
"openai:responses",
Some("trace-codex-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("x-client-request-id"),
Some(&"kept-by-rule-request".to_string())
);
assert!(!headers.contains_key("user-agent"));
assert!(!headers.contains_key("originator"));
assert_eq!(headers.get("session_id"), Some(&"kept-by-rule".to_string()));
assert!(!headers.contains_key("conversation_id"));
assert_eq!(
headers.get("user-agent"),
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("user-agent"))
.count(),
1
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("originator"))
.count(),
1
);
assert!(!headers.contains_key("version"));
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session-id"),
Some(&"kept-by-rule-session".to_string())
);
assert_eq!(
headers.get("thread-id"),
Some(&"kept-by-rule-thread".to_string())
);
}
#[test]
fn skips_conversation_id_for_compact_codex_requests() {
fn compact_projects_uuid_prompt_cache_identity_into_session_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
});
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses:compact",
Some("trace-codex-compact-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(
&mut headers,
&body,
"codex",
"openai:responses:compact",
);
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(
headers.get("x-client-request-id"),
Some(&"trace-codex-compact-123".to_string())
);
assert_eq!(headers.get("x-client-request-id"), None);
assert_eq!(
headers.get("user-agent"),
Some(
&"codex-tui/0.122.0 (Mac OS 15.2.0; arm64) vscode/2.6.11 (codex-tui; 0.122.0)"
.to_string()
)
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex-tui".to_string()));
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session_id"),
Some(&"ab5ecce4f0d110fe".to_string())
headers.get("session-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert!(!headers.contains_key("conversation_id"));
}
@@ -178,7 +178,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalStandardSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -201,12 +201,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalStandardSyncAttemptSour
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalStandardStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -229,6 +244,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalStandardStreamAttempt
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalStandardSyncAttemptSource<'_> {
@@ -340,7 +370,7 @@ pub(crate) async fn maybe_build_sync_via_standard_family_payload(
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -390,7 +420,7 @@ pub(crate) async fn maybe_build_stream_via_standard_family_payload(
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -449,7 +479,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
return Ok(Vec::new());
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -524,7 +554,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
return Ok(Vec::new());
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -60,7 +60,9 @@ pub(super) async fn resolve_local_standard_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -119,8 +121,10 @@ pub(super) async fn materialize_local_standard_candidate_attempts(
);
let preselection = preselect_local_execution_candidates_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
false,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -243,9 +247,11 @@ pub(super) async fn build_local_standard_candidate_attempt_source<'a>(
let (source, candidate_count) =
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
spec_metadata.api_format,
&input.requested_model,
None,
spec_metadata.require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
@@ -338,8 +344,10 @@ async fn maybe_append_gemini_image_openai_image_preselection(
let image_preselection = preselect_local_execution_candidates_for_api_formats_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -15,7 +15,7 @@ use crate::ai_serving::planner::report_context::{
use crate::ai_serving::planner::spec_metadata::local_standard_spec_metadata;
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
@@ -140,6 +140,7 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -176,7 +177,7 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
transport_profile: _,
request_redacted: _,
} = resolved;
let request_gzip = resolve_transport_request_gzip_policy(&transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -186,6 +187,7 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: candidate.provider_name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -203,8 +205,8 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
@@ -213,7 +215,11 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -368,10 +374,13 @@ mod tests {
auth_snapshot: sample_auth_snapshot(),
required_capabilities: None,
request_auth_channel: None,
client_surface: None,
gateway_credential_carrier: None,
client_session_affinity: None,
routing_policy: None,
routing_trace_seed: None,
routing_context: None,
model_directive_policy: Default::default(),
}
}
@@ -475,6 +484,7 @@ mod tests {
} else {
"gpt-4o-upstream".to_string()
},
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -21,8 +21,9 @@ use crate::ai_serving::planner::redaction::{
};
use crate::ai_serving::planner::spec_metadata::local_standard_spec_metadata;
use crate::ai_serving::planner::standard::{
apply_codex_openai_responses_special_headers, apply_deepseek_tool_call_thinking_compat,
is_deepseek_provider, request_body_build_failure_extra_data,
apply_codex_openai_special_headers, apply_deepseek_tool_call_thinking_compat,
codex_model_capabilities_for_transport, is_deepseek_provider,
openai_provider_request_contract_failure_extra_data, request_body_build_failure_extra_data,
request_conversion_failure_extra_data,
};
use crate::ai_serving::transport::kiro::{
@@ -44,7 +45,9 @@ use crate::ai_serving::transport::{
};
use crate::ai_serving::{
build_openai_image_request_body_from_gemini_image_request, gemini_request_is_image_generation,
project_codex_openai_image_api_request_body, project_openai_image_api_request_body,
CandidateFailureDiagnostic, GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
OpenAiImageOperation,
};
use crate::{AppState, GatewayError};
@@ -313,7 +316,13 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
{
return Ok(
resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
state, parts, trace_id, body_json, input, attempt,
state,
parts,
trace_id,
body_json,
input,
attempt,
spec_metadata.require_streaming,
)
.await,
);
@@ -555,13 +564,35 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
);
let force_body_stream_field =
endpoint_config_forces_body_stream_field(transport.endpoint.config.as_ref());
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
state,
provider_api_format,
Some(&input.requested_model),
)
.await;
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(provider_api_format, Some(&input.requested_model));
let model_directive_mapping = match model_directive_resolution
.mapping_patch_for_mapped_model(&prepared_candidate.mapped_model)
{
Ok(mapping) => mapping,
Err(skip_reason) => {
mark_skipped_local_standard_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
};
crate::ai_serving::hydrate_openai_response_history(
state.runtime_state(),
body_json,
spec_metadata.api_format,
provider_api_format,
input.auth_context.api_key_id.as_str(),
)
.await?;
let redaction = resolve_provider_chat_pii_redaction(
state,
parts,
@@ -588,7 +619,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
},
Some(input.auth_context.api_key_id.as_str()),
Some(effective_headers),
enable_model_directives,
false,
) {
Some(body) => body,
None => {
@@ -655,18 +686,8 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
provider_api_format,
Some(body_json),
);
if let Some(mapping) =
crate::system_features::reasoning_model_directive_mapping_for_api_format_and_model(
state,
provider_api_format,
Some(&input.requested_model),
)
.await
{
crate::ai_serving::apply_model_directive_mapping_patch(
&mut provider_request_body,
&mapping,
);
if let Some(mapping) = model_directive_mapping.as_ref() {
crate::ai_serving::apply_model_directive_mapping_patch(&mut provider_request_body, mapping);
// Directive mapping is a deep-merge patch and may overwrite/add `stream`;
// re-enforce stream-field policy afterward.
enforce_provider_body_stream_policy(
@@ -712,6 +733,61 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
);
}
let normalized_provider_api_format =
crate::ai_serving::normalize_api_format_alias(provider_api_format);
if matches!(
normalized_provider_api_format.as_str(),
"openai:chat" | "openai:responses" | "openai:responses:compact"
) {
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(input.requested_model.as_str());
let codex_model_capabilities = codex_model_capabilities_for_transport(
transport,
provider_api_format,
prepared_candidate.mapped_model.as_str(),
source_model,
);
if let Err(violation) =
crate::ai_serving::finalize_openai_provider_request_with_codex_model_capabilities(
&mut provider_request_body,
crate::ai_serving::OpenAiProviderRequestFinalization {
source_api_format: spec_metadata.api_format,
provider_api_format,
provider_type: transport.provider.provider_type.as_str(),
provider_model: prepared_candidate.mapped_model.as_str(),
source_model,
body_rules: transport.endpoint.body_rules.as_ref(),
upstream_is_stream,
require_body_stream_field: request_requires_body_stream_field(
body_json,
force_body_stream_field,
),
},
codex_model_capabilities.as_ref(),
)
{
mark_skipped_local_standard_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
Some(openai_provider_request_contract_failure_extra_data(
&violation,
spec_metadata.api_format,
provider_api_format,
"standard_family_request_finalization",
)),
)
.await;
return Ok(None);
}
}
if let Some(kiro_auth) = kiro_auth.as_ref() {
return Ok(build_kiro_cross_format_payload_parts(
state,
@@ -752,8 +828,6 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
.await);
}
let normalized_provider_api_format =
crate::ai_serving::normalize_api_format_alias(provider_api_format);
if normalized_provider_api_format == "gemini:generate_content"
&& is_gemini_cli_provider_transport(transport)
{
@@ -838,7 +912,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
return Ok(None);
};
let mut provider_request_headers = resolved_headers.headers;
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
@@ -988,7 +1062,7 @@ async fn build_gemini_cli_cross_format_payload_parts(
};
let mut provider_request_headers = resolved.headers.headers;
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&resolved.body,
effective_headers,
@@ -1146,6 +1220,7 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
body_json: &serde_json::Value,
input: &LocalStandardDecisionInput,
attempt: &LocalStandardCandidateAttempt,
client_requires_streaming: bool,
) -> Option<LocalStandardCandidatePayloadParts> {
let client_api_format = "gemini:generate_content";
let provider_api_format = "openai:image";
@@ -1221,17 +1296,60 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
return None;
};
let upstream_is_stream = true;
let upstream_url =
build_openai_image_upstream_url(transport, Some("/v1/images/generations"), None);
let upstream_is_stream = resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
provider_api_format,
client_requires_streaming && candidate.supports_streaming,
false,
);
let is_codex = transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex");
let mut provider_request_body = converted.body_json;
if upstream_is_stream {
provider_request_body
.as_object_mut()?
.insert("stream".to_string(), Value::Bool(true));
}
provider_request_body = project_openai_image_api_request_body(
&provider_request_body,
&prepared_candidate.mapped_model,
converted.operation,
crate::image_capabilities::openai_image_provider_max_generation_count_for_model(
transport.provider.provider_type.as_str(),
Some(prepared_candidate.mapped_model.as_str()),
),
)?;
if is_codex {
provider_request_body = project_codex_openai_image_api_request_body(
&provider_request_body,
converted.operation,
)?;
}
let request_path = match converted.operation {
OpenAiImageOperation::Generate => "/v1/images/generations",
OpenAiImageOperation::Edit => "/v1/images/edits",
};
let upstream_url = build_openai_image_upstream_url(transport, Some(request_path), None);
let effective_headers = input.effective_headers(&parts.headers);
let Some(mut provider_request_headers) =
build_openai_image_headers(ProviderOpenAiImageHeadersInput {
transport,
headers: effective_headers,
auth_header: &prepared_candidate.auth_header,
auth_value: &prepared_candidate.auth_value,
accept: if is_codex {
None
} else if upstream_is_stream {
Some("text/event-stream")
} else {
Some("application/json")
},
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &converted.body_json,
provider_request_body: &provider_request_body,
original_request_body: body_json,
})
else {
@@ -1252,9 +1370,9 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
.await;
return None;
};
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&converted.body_json,
&provider_request_body,
effective_headers,
transport.provider.provider_type.as_str(),
provider_api_format,
@@ -1267,7 +1385,7 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
auth_value: prepared_candidate.auth_value,
mapped_model: converted.mapped_model,
provider_api_format: provider_api_format.to_string(),
provider_request_body: converted.body_json,
provider_request_body,
provider_request_headers,
upstream_url,
upstream_is_stream,
@@ -1,12 +1,10 @@
use std::collections::BTreeMap;
use aether_contracts::RequestBody;
use super::{
augment_sync_report_context, build_ai_execution_plan_from_decision,
generic_decision_missing_exact_provider_request, take_ai_decision_plan_core,
take_ai_upstream_auth_pair, take_non_empty_string, AiExecutionPlanFromDecisionParts,
AiStreamAttempt, AiSyncAttempt,
generic_decision_missing_exact_provider_request, resolve_ai_passthrough_sync_request_body,
take_ai_decision_plan_core, take_ai_upstream_auth_pair, take_non_empty_string,
AiExecutionPlanFromDecisionParts, AiStreamAttempt, AiSyncAttempt,
};
use crate::ai_serving::transport::{
build_standard_plan_fallback_headers, StandardPlanFallbackAcceptPolicy,
@@ -61,6 +59,10 @@ pub(crate) fn build_gemini_sync_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
@@ -70,7 +72,7 @@ pub(crate) fn build_gemini_sync_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream,
},
);
@@ -129,6 +131,10 @@ pub(crate) fn build_gemini_stream_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -137,7 +143,7 @@ pub(crate) fn build_gemini_stream_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream: true,
},
);
@@ -15,7 +15,8 @@ mod normalize;
mod openai;
pub(crate) use self::codex::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_special_body_edits, apply_codex_openai_special_headers,
codex_model_capabilities_for_transport,
};
pub(crate) use self::deepseek::{apply_deepseek_tool_call_thinking_compat, is_deepseek_provider};
pub(crate) use self::family::{
@@ -25,9 +26,11 @@ pub(crate) use self::family::{
pub(crate) use self::normalize::{
build_cross_format_openai_chat_request_body, build_cross_format_openai_chat_upstream_url,
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_codex_model_capabilities,
build_cross_format_openai_responses_upstream_url, build_local_openai_chat_request_body,
build_local_openai_chat_upstream_url, build_local_openai_responses_request_body,
build_local_openai_responses_upstream_url,
build_local_openai_responses_request_body_with_codex_model_capabilities,
build_local_openai_responses_upstream_url, validate_final_openai_provider_request,
};
pub(crate) use self::openai::{
build_local_openai_chat_stream_attempt_source_for_kind,
@@ -62,8 +65,8 @@ pub(crate) use crate::ai_serving::{
normalize_openai_responses_request_to_openai_chat_request, parse_openai_tool_result_content,
};
pub(crate) use aether_ai_serving::{
request_body_build_failure_extra_data, request_conversion_failure_extra_data,
same_format_provider_request_body_failure_extra_data,
openai_provider_request_contract_failure_extra_data, request_body_build_failure_extra_data,
request_conversion_failure_extra_data, same_format_provider_request_body_failure_extra_data,
};
pub(crate) fn build_standard_upstream_url(
@@ -81,6 +84,7 @@ pub(crate) fn build_standard_upstream_url(
upstream_is_stream,
parts.uri.query(),
None,
None,
provider_request_body,
)
}
@@ -297,7 +301,7 @@ mod tests {
let converted = build_standard_request_body(
&request,
"claude:messages",
"gpt-5",
"gpt-5.4",
"codex",
"openai:responses",
"/v1/messages",
@@ -309,7 +313,7 @@ mod tests {
assert!(converted.get("metadata").is_none());
assert_eq!(converted["store"], false);
assert_eq!(converted["instructions"], "");
assert!(converted.get("instructions").is_none());
assert_eq!(converted["include"], json!(["reasoning.encrypted_content"]));
assert_eq!(converted["parallel_tool_calls"], true);
assert_eq!(converted["reasoning"]["effort"], "medium");
@@ -12,9 +12,38 @@ pub(crate) use self::chat::{
};
pub(crate) use self::responses::{
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_codex_model_capabilities,
build_cross_format_openai_responses_upstream_url, build_local_openai_responses_request_body,
build_local_openai_responses_request_body_with_codex_model_capabilities,
build_local_openai_responses_upstream_url,
};
pub(super) use crate::ai_serving::planner::common::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
};
pub(crate) fn validate_final_openai_provider_request(
provider_api_format: &str,
mapped_model: &str,
source_request_body: &serde_json::Value,
provider_request_body: &serde_json::Value,
) -> Option<()> {
let provider_model = provider_request_body
.get("model")
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.unwrap_or(mapped_model);
let source_model = source_request_body
.get("model")
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.unwrap_or(mapped_model);
crate::ai_serving::validate_openai_provider_request_contract(
provider_api_format,
provider_model,
source_model,
provider_request_body,
)
.ok()
}
@@ -9,7 +9,10 @@ use crate::ai_serving::{
GatewayProviderTransportSnapshot,
};
use super::{enforce_provider_body_stream_policy, request_requires_body_stream_field};
use super::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
validate_final_openai_provider_request,
};
pub(crate) fn build_local_openai_chat_request_body(
body_json: &Value,
@@ -39,6 +42,12 @@ pub(crate) fn build_local_openai_chat_request_body(
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
"openai:chat",
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -92,6 +101,12 @@ pub(crate) fn build_cross_format_openai_chat_request_body(
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -2,14 +2,16 @@ use serde_json::Value;
use crate::ai_serving::transport::apply_standard_provider_request_body_rules_with_request_headers;
use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits,
apply_openai_responses_compact_special_body_edits,
build_cross_format_openai_responses_request_body_with_model_directives as surface_build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_model_directives_and_history_scope as surface_build_cross_format_openai_responses_request_body,
build_local_openai_responses_request_body_with_model_directives as surface_build_local_openai_responses_request_body,
GatewayProviderTransportSnapshot,
};
use super::{enforce_provider_body_stream_policy, request_requires_body_stream_field};
use super::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
validate_final_openai_provider_request,
};
pub(crate) fn build_local_openai_responses_request_body(
body_json: &Value,
@@ -19,9 +21,35 @@ pub(crate) fn build_local_openai_responses_request_body(
provider_type: &str,
provider_api_format: &str,
body_rules: Option<&Value>,
user_api_key_id: Option<&str>,
_user_api_key_id: Option<&str>,
request_headers: &http::HeaderMap,
enable_model_directives: bool,
) -> Option<Value> {
build_local_openai_responses_request_body_with_codex_model_capabilities(
body_json,
mapped_model,
require_streaming,
force_body_stream_field,
provider_type,
provider_api_format,
body_rules,
request_headers,
None,
enable_model_directives,
)
}
pub(crate) fn build_local_openai_responses_request_body_with_codex_model_capabilities(
body_json: &Value,
mapped_model: &str,
require_streaming: bool,
force_body_stream_field: bool,
provider_type: &str,
provider_api_format: &str,
body_rules: Option<&Value>,
request_headers: &http::HeaderMap,
model_capabilities: Option<&crate::ai_serving::CodexResponsesModelCapabilities>,
enable_model_directives: bool,
) -> Option<Value> {
let provider_request_body = surface_build_local_openai_responses_request_body(
body_json,
@@ -36,23 +64,39 @@ pub(crate) fn build_local_openai_responses_request_body(
body_json,
request_headers,
)?;
apply_codex_openai_responses_special_body_edits(
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(mapped_model);
crate::ai_serving::apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities(
&mut provider_request_body,
provider_type,
provider_api_format,
mapped_model,
source_model,
model_capabilities,
body_rules,
user_api_key_id,
);
apply_openai_responses_compact_special_body_edits(
&mut provider_request_body,
provider_api_format,
);
crate::ai_serving::strip_incompatible_openai_responses_reasoning_items(
&mut provider_request_body,
provider_api_format,
);
enforce_provider_body_stream_policy(
&mut provider_request_body,
provider_api_format,
require_streaming,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -68,6 +112,36 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
user_api_key_id: Option<&str>,
request_headers: &http::HeaderMap,
enable_model_directives: bool,
) -> Option<Value> {
build_cross_format_openai_responses_request_body_with_codex_model_capabilities(
body_json,
mapped_model,
client_api_format,
provider_api_format,
upstream_is_stream,
force_body_stream_field,
provider_type,
body_rules,
request_headers,
user_api_key_id,
None,
enable_model_directives,
)
}
pub(crate) fn build_cross_format_openai_responses_request_body_with_codex_model_capabilities(
body_json: &Value,
mapped_model: &str,
client_api_format: &str,
provider_api_format: &str,
upstream_is_stream: bool,
force_body_stream_field: bool,
provider_type: &str,
body_rules: Option<&Value>,
request_headers: &http::HeaderMap,
history_scope: Option<&str>,
model_capabilities: Option<&crate::ai_serving::CodexResponsesModelCapabilities>,
enable_model_directives: bool,
) -> Option<Value> {
let provider_request_body = surface_build_cross_format_openai_responses_request_body(
body_json,
@@ -76,6 +150,7 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
provider_api_format,
upstream_is_stream,
enable_model_directives,
history_scope,
)?;
let mut provider_request_body =
apply_standard_provider_request_body_rules_with_request_headers(
@@ -84,23 +159,39 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
body_json,
request_headers,
)?;
apply_codex_openai_responses_special_body_edits(
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(mapped_model);
crate::ai_serving::apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities(
&mut provider_request_body,
provider_type,
provider_api_format,
mapped_model,
source_model,
model_capabilities,
body_rules,
user_api_key_id,
);
apply_openai_responses_compact_special_body_edits(
&mut provider_request_body,
provider_api_format,
);
crate::ai_serving::strip_incompatible_openai_responses_reasoning_items(
&mut provider_request_body,
provider_api_format,
);
enforce_provider_body_stream_policy(
&mut provider_request_body,
provider_api_format,
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -6,8 +6,8 @@ use http::Request;
use serde_json::{json, Value};
use super::{
build_cross_format_openai_responses_request_body, build_local_openai_responses_request_body,
build_local_openai_responses_upstream_url,
build_cross_format_openai_responses_request_body, build_local_openai_chat_request_body,
build_local_openai_responses_request_body, build_local_openai_responses_upstream_url,
};
fn object_keys(value: &Value) -> Vec<&str> {
@@ -146,12 +146,47 @@ fn local_openai_responses_wrapper_preserves_body_order_after_edits() {
"reasoning",
"tool_choice",
"parallel_tool_calls",
"instructions",
"prompt_cache_key",
]
);
assert_eq!(provider_request_body["parallel_tool_calls"], json!(true));
assert_eq!(provider_request_body["instructions"], json!(""));
assert!(provider_request_body.get("instructions").is_none());
}
#[test]
fn local_openai_responses_wrapper_strips_foreign_reasoning_item_ids() {
let body_json = json!({
"model": "gpt-5.4",
"input": [
{"type": "reasoning", "id": "rs_provider_123", "summary": []},
{
"type": "reasoning",
"id": "item_72d3bd8d367d01977ace23f1",
"summary": []
},
{"type": "message", "role": "user", "content": "continue"}
]
});
let provider_request_body = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
false,
false,
"codex",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.expect("local OpenAI Responses body should build");
let input = provider_request_body["input"]
.as_array()
.expect("input array");
assert_eq!(input.len(), 2);
assert_eq!(input[0]["id"], "rs_provider_123");
assert_eq!(input[1]["type"], "message");
}
#[test]
@@ -181,18 +216,49 @@ fn local_openai_responses_compact_wrapper_strips_store_for_same_format_requests(
}
#[test]
fn local_openai_responses_compact_wrapper_strips_include_for_codex_requests() {
fn local_codex_compact_wrapper_applies_the_complete_request_projection() {
let body_json = json!({
"model": "gpt-5.4",
"input": [],
"model": "gpt-5.6-sol",
"input": [{
"type": "message",
"role": "user",
"content": [{"type": "input_text", "text": "hello"}]
}],
"instructions": "Work carefully",
"client_metadata": {"origin": "codex"},
"include": ["reasoning.encrypted_content"],
"store": true,
"stream": true
"stream": true,
"stream_options": {"reasoning_summary_delivery": "sequential_cutoff"},
"tool_choice": "auto",
"parallel_tool_calls": true,
"reasoning": {"effort": "max", "summary": "auto", "context": "all_turns"},
"text": {"verbosity": "medium"},
"tools": [{
"type": "function",
"name": "lookup",
"parameters": {"type": "object", "properties": {}}
}],
"service_tier": "priority",
"prompt_cache_key": "thread-compact"
});
let provider_request_body = build_local_openai_responses_request_body(
let regular = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
"gpt-5.6-sol",
true,
false,
"codex",
"openai:responses",
None,
Some("key-123"),
&http::HeaderMap::new(),
false,
)
.expect("local Codex Responses body should build");
let compact = build_local_openai_responses_request_body(
&body_json,
"gpt-5.6-sol",
false,
false,
"codex",
@@ -202,22 +268,44 @@ fn local_openai_responses_compact_wrapper_strips_include_for_codex_requests() {
&http::HeaderMap::new(),
false,
)
.expect("local codex compact body should build");
.expect("local Codex Compact body should build");
assert!(provider_request_body.get("include").is_none());
assert!(provider_request_body.get("store").is_none());
assert!(provider_request_body.get("stream").is_none());
assert_eq!(provider_request_body["instructions"], "");
assert_eq!(
provider_request_body["prompt_cache_key"],
"3d2e2842-74cb-55dd-803a-b8940b3500c2"
);
for field in [
"client_metadata",
"include",
"store",
"stream",
"stream_options",
"tool_choice",
] {
assert!(
regular.get(field).is_some(),
"Responses should contain {field}"
);
assert!(compact.get(field).is_none(), "Compact should omit {field}");
}
for field in [
"model",
"input",
"instructions",
"parallel_tool_calls",
"reasoning",
"text",
"tools",
"service_tier",
"prompt_cache_key",
] {
assert_eq!(
compact[field], regular[field],
"Compact should preserve {field}"
);
}
}
#[test]
fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
let body_json = json!({
"model": "gpt-5.4-max",
"model": "gpt-5.6-sol-max",
"input": "hello",
"reasoning": {"effort": "low", "summary": "auto"}
});
@@ -227,7 +315,7 @@ fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
let provider_request_body = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
"gpt-5.6-sol",
false,
false,
"openai",
@@ -239,11 +327,137 @@ fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
)
.expect("local openai responses body should build");
assert_eq!(provider_request_body["reasoning"]["effort"], "xhigh");
assert_eq!(provider_request_body["reasoning"]["effort"], "max");
assert_eq!(provider_request_body["reasoning"]["summary"], "auto");
assert_eq!(provider_request_body["metadata"]["override_seen"], true);
}
#[test]
fn final_openai_provider_contract_uses_the_mapped_model_for_reasoning() {
let alias = json!({
"model": "deployment-alias",
"input": "hello",
"reasoning": {"effort": "max"}
});
assert!(build_local_openai_responses_request_body(
&alias,
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_some());
assert!(build_local_openai_responses_request_body(
&alias,
"gpt-5.4",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let minimal = json!({
"model": "deployment-alias",
"messages": [{"role": "user", "content": "hello"}],
"reasoning_effort": "minimal"
});
assert!(build_local_openai_chat_request_body(
&minimal,
"gpt-5.6-terra",
false,
false,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let opaque_mapping = json!({
"model": "gpt-5.6-sol-max",
"input": "hello",
"reasoning": {"effort": "max", "mode": "pro"},
"prompt_cache_options": {"mode": "explicit", "ttl": "30m"}
});
assert!(build_local_openai_responses_request_body(
&opaque_mapping,
"azure-production",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_some());
assert!(build_local_openai_responses_request_body(
&opaque_mapping,
"gpt-5.4",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
}
#[test]
fn final_openai_provider_contract_validates_body_rule_output() {
let body = json!({
"model": "gpt-5.6-sol",
"input": "hello",
"reasoning": {"effort": "max"}
});
let model_override = json!([
{"action":"set","path":"model","value":"gpt-5.4"}
]);
assert!(build_local_openai_responses_request_body(
&body,
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
Some(&model_override),
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let cache_override = json!([
{"action":"set","path":"prompt_cache_options.ttl","value":"1h"}
]);
assert!(build_local_openai_responses_request_body(
&json!({"model":"gpt-5.6-sol","input":"hello"}),
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
Some(&cache_override),
None,
&http::HeaderMap::new(),
false,
)
.is_none());
}
#[test]
fn local_openai_responses_upstream_url_preserves_codex_base_path() {
let request = Request::builder()
@@ -371,7 +585,7 @@ fn applies_codex_defaults_unless_body_rules_handle_the_field() {
}
#[test]
fn injects_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
fn omits_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
let body_json = json!({
"model": "claude-sonnet-4-5",
"messages": [{
@@ -395,14 +609,11 @@ fn injects_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
)
.expect("claude cli to codex request should build");
assert_eq!(
provider_request_body["prompt_cache_key"],
"4ee6ea6e-3ac6-5a18-8cb8-1f8b956419e5"
);
assert!(provider_request_body.get("prompt_cache_key").is_none());
}
#[test]
fn injects_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
fn omits_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
let body_json = json!({
"model": "gpt-5",
"messages": [{
@@ -425,8 +636,5 @@ fn injects_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
)
.expect("openai chat to codex request should build");
assert_eq!(
provider_request_body["prompt_cache_key"],
"4ee6ea6e-3ac6-5a18-8cb8-1f8b956419e5"
);
assert!(provider_request_body.get("prompt_cache_key").is_none());
}
@@ -6,6 +6,7 @@ mod request;
mod support;
pub(super) use self::payload::maybe_build_local_openai_chat_decision_payload_for_candidate;
pub(super) use self::request::LocalOpenAiChatRequestPreparation;
pub(super) use self::support::{
build_lazy_local_openai_chat_candidate_attempt_source,
build_local_openai_chat_candidate_attempt_source,
@@ -6,18 +6,21 @@ use crate::ai_serving::planner::report_context::{
insert_provider_stream_event_api_format, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_chat_candidate_payload_parts;
use super::request::{
resolve_local_openai_chat_candidate_payload_parts, LocalOpenAiChatRequestPreparation,
};
use super::support::{LocalOpenAiChatCandidateAttempt, LocalOpenAiChatDecisionInput};
#[allow(clippy::too_many_arguments)]
@@ -27,6 +30,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
trace_id: &str,
body_json: &serde_json::Value,
input: &LocalOpenAiChatDecisionInput,
preparation: Option<&mut LocalOpenAiChatRequestPreparation>,
attempt: LocalOpenAiChatCandidateAttempt,
decision_kind: &str,
report_kind: &str,
@@ -40,12 +44,15 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
candidate_id,
..
} = attempt;
let upstream_is_stream = upstream_is_stream && eligible.candidate.supports_streaming;
let payload_started_at = std::time::Instant::now();
let Some(resolved) = resolve_local_openai_chat_candidate_payload_parts(
state,
parts,
trace_id,
body_json,
input,
preparation,
&eligible,
candidate_index,
&candidate_id,
@@ -55,9 +62,25 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
)
.await?
else {
observe_gateway_stage_ms(
"stream_candidate_payload_parts",
payload_started_at.elapsed().as_millis() as u64,
);
return Ok(None);
};
observe_gateway_stage_ms(
"stream_candidate_payload_parts",
payload_started_at.elapsed().as_millis() as u64,
);
let candidate = &eligible.candidate;
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
resolved.transport.endpoint.config.as_ref(),
resolved.transport.provider.provider_type.as_str(),
resolved.provider_api_format.as_str(),
upstream_is_stream,
false,
);
let prompt_cache_key = resolved
.provider_request_body
@@ -66,9 +89,14 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned);
let proxy_started_at = std::time::Instant::now();
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
observe_gateway_stage_ms(
"stream_candidate_proxy",
proxy_started_at.elapsed().as_millis() as u64,
);
let transport_profile = resolved
.transport_profile
.clone()
@@ -130,6 +158,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
Some(body_json)
};
let effective_headers = input.effective_headers(&parts.headers);
let report_context_started_at = std::time::Instant::now();
let report_context = append_local_failover_policy_to_value(
append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -164,6 +193,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -184,8 +214,13 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
),
&transport,
);
let request_gzip = resolve_transport_request_gzip_policy(&transport);
observe_gateway_stage_ms(
"stream_candidate_report_context",
report_context_started_at.elapsed().as_millis() as u64,
);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let decision_started_at = std::time::Instant::now();
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream,
decision_kind: decision_kind.to_string(),
@@ -194,6 +229,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -211,8 +247,8 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
@@ -221,6 +257,14 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
observe_gateway_stage_ms(
"stream_candidate_decision_build",
decision_started_at.elapsed().as_millis() as u64,
);
Ok(Some(decision))
}
File diff suppressed because it is too large Load Diff
@@ -293,9 +293,11 @@ pub(crate) async fn build_lazy_local_openai_chat_candidate_attempt_source<'a>(
);
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
"openai:chat",
&input.requested_model,
None,
require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
@@ -13,6 +13,7 @@ use self::decision::{
build_lazy_local_openai_chat_candidate_attempt_source,
maybe_build_local_openai_chat_decision_payload_for_candidate, LocalOpenAiChatCandidateAttempt,
LocalOpenAiChatCandidateAttemptSource, LocalOpenAiChatDecisionInput,
LocalOpenAiChatRequestPreparation,
};
use self::plans::{
build_local_openai_chat_stream_attempt_source, build_local_openai_chat_stream_plan_and_reports,
@@ -147,7 +148,7 @@ pub(crate) async fn maybe_build_sync_local_decision_payload(
)
.await;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let upstream_is_stream = self::plans::openai_chat_upstream_is_stream_for_candidate(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
@@ -159,6 +160,7 @@ pub(crate) async fn maybe_build_sync_local_decision_payload(
trace_id,
body_json,
&input,
None,
attempt,
OPENAI_CHAT_SYNC_PLAN_KIND,
"openai_chat_sync_success",
@@ -199,7 +201,7 @@ pub(crate) async fn maybe_build_stream_local_decision_payload(
)
.await;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let upstream_is_stream = self::plans::openai_chat_upstream_is_stream_for_candidate(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
@@ -211,6 +213,7 @@ pub(crate) async fn maybe_build_stream_local_decision_payload(
trace_id,
body_json,
&input,
None,
attempt,
OPENAI_CHAT_STREAM_PLAN_KIND,
"openai_chat_stream_success",
@@ -31,11 +31,12 @@ pub(super) fn openai_chat_upstream_is_stream_for_candidate(
crate::ai_serving::transport::kiro::is_kiro_claude_messages_transport(
transport,
provider_api_format,
) || openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport,
provider_api_format,
client_is_stream,
);
) || openai_chat_antigravity_requires_upstream_streaming(transport, provider_api_format)
|| openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport,
provider_api_format,
client_is_stream,
);
resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
@@ -45,6 +46,15 @@ pub(super) fn openai_chat_upstream_is_stream_for_candidate(
)
}
fn openai_chat_antigravity_requires_upstream_streaming(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
) -> bool {
crate::ai_serving::transport::antigravity::is_antigravity_provider_transport(transport)
&& crate::ai_serving::normalize_api_format_alias(provider_api_format)
== "gemini:generate_content"
}
fn openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
@@ -201,4 +211,24 @@ mod tests {
false,
));
}
#[test]
fn openai_chat_policy_resolver_preserves_antigravity_streaming_envelope() {
let antigravity = sample_transport(
"antigravity",
"gemini:generate_content",
Some(json!({"upstream_stream_policy": "force_non_stream"})),
);
assert!(openai_chat_upstream_is_stream_for_candidate(
&antigravity,
"gemini:generate_content",
false,
));
assert!(openai_chat_upstream_is_stream_for_candidate(
&antigravity,
"gemini:generate_content",
true,
));
}
}
@@ -21,8 +21,10 @@ pub(crate) async fn list_local_openai_chat_candidates(
> {
let outcome = preselect_local_execution_candidates_with_serving(
PlannerAppState::new(state),
&input.model_directive_policy,
"openai:chat",
&input.requested_model,
None,
require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -9,6 +9,7 @@ use crate::ai_serving::planner::decision_input::{
};
use crate::ai_serving::resolve_local_decision_execution_runtime_auth_context;
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{AppState, GatewayError};
pub(crate) async fn resolve_local_openai_chat_decision_input(
@@ -59,11 +60,14 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
return Ok(None);
};
let auth_started_at = std::time::Instant::now();
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context.clone(),
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -106,10 +110,20 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
return Err(err);
}
};
observe_gateway_stage_ms(
"openai_chat_decision_input_auth",
auth_started_at.elapsed().as_millis() as u64,
);
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
let affinity_started_at = std::time::Instant::now();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
observe_gateway_stage_ms(
"openai_chat_decision_input_affinity",
affinity_started_at.elapsed().as_millis() as u64,
);
let routing_started_at = std::time::Instant::now();
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
@@ -126,5 +140,9 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
);
return Err(err);
}
observe_gateway_stage_ms(
"openai_chat_decision_input_routing",
routing_started_at.elapsed().as_millis() as u64,
);
Ok(Some(input))
}
@@ -1,11 +1,12 @@
use async_trait::async_trait;
use std::collections::VecDeque;
use tracing::warn;
use super::super::{
build_lazy_local_openai_chat_candidate_attempt_source,
maybe_build_local_openai_chat_decision_payload_for_candidate, AppState, GatewayControlDecision,
GatewayError, LocalOpenAiChatCandidateAttempt, LocalOpenAiChatCandidateAttemptSource,
LocalOpenAiChatDecisionInput,
LocalOpenAiChatDecisionInput, LocalOpenAiChatRequestPreparation,
};
use super::diagnostic::{
set_local_openai_chat_candidate_evaluation_diagnostic, set_local_openai_chat_miss_diagnostic,
@@ -18,6 +19,23 @@ use crate::ai_serving::planner::plan_builders::{
build_openai_chat_stream_plan_from_decision, AiStreamAttempt,
};
use crate::ai_serving::planner::runtime_miss::apply_local_runtime_candidate_terminal_reason;
use crate::ai_serving::planner::standard::build_local_openai_chat_upstream_url;
use crate::ai_serving::transport::{
is_windsurf_provider_transport, local_openai_chat_transport_unsupported_reason,
};
use crate::clock::request_distribution_seed;
use crate::stage_metrics::{
observe_gateway_stage_ms, record_openai_chat_stream_payload_build_prefetch_avoided,
record_openai_chat_stream_payload_build_selected,
record_openai_chat_stream_raw_candidates_scanned,
record_openai_chat_stream_target_select_selected_rank,
};
use crate::upstream_admission::upstream_target_key_from_url;
const OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW_ENV: &str =
"AETHER_GATEWAY_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW";
const DEFAULT_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW: usize = 2;
const MAX_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW: usize = 8;
pub(crate) struct LocalOpenAiChatStreamAttemptSource<'a> {
state: &'a AppState,
@@ -26,6 +44,8 @@ pub(crate) struct LocalOpenAiChatStreamAttemptSource<'a> {
body_json: serde_json::Value,
input: LocalOpenAiChatDecisionInput,
candidates: LocalOpenAiChatCandidateAttemptSource<'a>,
prefetched_attempts: VecDeque<LocalOpenAiChatCandidateAttempt>,
request_preparation: LocalOpenAiChatRequestPreparation,
}
pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
@@ -40,6 +60,7 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
return Ok(None);
}
let attempt_source_started_at = std::time::Instant::now();
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, true,
)
@@ -76,6 +97,10 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
Some(input.requested_model.as_str()),
candidate_count,
);
observe_gateway_stage_ms(
"openai_chat_attempt_source_build",
attempt_source_started_at.elapsed().as_millis() as u64,
);
Ok(Some((
LocalOpenAiChatStreamAttemptSource {
@@ -85,6 +110,8 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
body_json: effective_body_json,
input,
candidates,
prefetched_attempts: VecDeque::new(),
request_preparation: LocalOpenAiChatRequestPreparation,
},
candidate_count,
)))
@@ -93,18 +120,13 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiChatStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
}
}
apply_local_runtime_candidate_terminal_reason(
self.state,
self.trace_id,
"no_local_stream_plans",
let select_started_at = std::time::Instant::now();
let selected = self.next_execution_attempt_with_target_select().await?;
observe_gateway_stage_ms(
"openai_chat_stream_target_select",
select_started_at.elapsed().as_millis() as u64,
);
Ok(None)
Ok(selected)
}
async fn drain_execution_attempts(&mut self) -> Result<Vec<AiStreamAttempt>, GatewayError> {
@@ -116,11 +138,178 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiChatStreamAttem
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.key_id != key_id);
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.endpoint_id != endpoint_id);
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.provider_id != provider_id);
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiChatStreamAttemptSource<'_> {
async fn build_stream_attempt(
async fn next_execution_attempt_with_target_select(
&mut self,
) -> Result<Option<AiStreamAttempt>, GatewayError> {
loop {
let Some(attempt) = self.next_raw_attempt_with_target_select().await? else {
apply_local_runtime_candidate_terminal_reason(
self.state,
self.trace_id,
"no_local_stream_plans",
);
return Ok(None);
};
let plan_started_at = std::time::Instant::now();
record_openai_chat_stream_payload_build_selected();
match self.build_stream_attempt(attempt).await? {
Some(attempt) => {
observe_gateway_stage_ms(
"stream_candidate_plan_build",
plan_started_at.elapsed().as_millis() as u64,
);
return Ok(Some(attempt));
}
None => {
observe_gateway_stage_ms(
"stream_candidate_plan_build",
plan_started_at.elapsed().as_millis() as u64,
);
continue;
}
}
}
}
async fn next_raw_attempt_with_target_select(
&mut self,
) -> Result<Option<LocalOpenAiChatCandidateAttempt>, GatewayError> {
let select_window = openai_chat_stream_target_select_window();
if select_window <= 1 {
return self.next_raw_attempt_linear().await;
}
let mut attempts = Vec::with_capacity(select_window);
for _ in 0..select_window {
match self.next_raw_attempt_linear().await? {
Some(attempt) => attempts.push(attempt),
None => break,
}
}
if attempts.is_empty() {
return Ok(None);
}
record_openai_chat_stream_raw_candidates_scanned(attempts.len());
let seed = request_distribution_seed();
let target_keys = attempts
.iter()
.map(|attempt| self.lightweight_target_key_for_attempt(attempt))
.collect::<Vec<_>>();
for target_key in target_keys.iter().flatten() {
self.state
.upstream_target_admission
.record_raw_seen_for_target_key(target_key);
}
let selected_index = if target_keys.iter().all(Option::is_some) {
let choices = attempts
.iter()
.zip(target_keys.iter())
.map(|(attempt, target_key)| {
let target_key = target_key.as_deref().unwrap_or("-");
let snapshot = self
.state
.upstream_target_admission
.snapshot_for_target_key(target_key);
TargetSelectChoice {
target_key,
identity: target_select_candidate_identity(attempt),
in_flight: snapshot
.as_ref()
.map(|snapshot| snapshot.in_flight)
.unwrap_or(0),
selection_pressure_total: snapshot
.as_ref()
.map(|snapshot| snapshot.selection_pressure_total)
.unwrap_or(0),
}
})
.collect::<Vec<_>>();
select_target_index(seed, &choices)
} else {
0
};
record_openai_chat_stream_target_select_selected_rank(selected_index);
record_openai_chat_stream_payload_build_prefetch_avoided(attempts.len().saturating_sub(1));
if let Some(Some(target_key)) = target_keys.get(selected_index) {
self.state
.upstream_target_admission
.record_preselect_for_target_key(target_key);
}
let selected = attempts.remove(selected_index);
self.prefetched_attempts.extend(attempts);
Ok(Some(selected))
}
async fn next_raw_attempt_linear(
&mut self,
) -> Result<Option<LocalOpenAiChatCandidateAttempt>, GatewayError> {
if let Some(attempt) = self.prefetched_attempts.pop_front() {
return Ok(Some(attempt));
}
let source_started_at = std::time::Instant::now();
let attempt = self.candidates.next_attempt().await?;
observe_gateway_stage_ms(
"stream_candidate_source_next",
source_started_at.elapsed().as_millis() as u64,
);
Ok(attempt)
}
fn lightweight_target_key_for_attempt(
&self,
attempt: &LocalOpenAiChatCandidateAttempt,
) -> Option<String> {
let provider_api_format = attempt.eligible.provider_api_format.trim();
if !provider_api_format.eq_ignore_ascii_case("openai:chat") {
return None;
}
let transport = &attempt.eligible.transport;
if transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("grok")
|| is_windsurf_provider_transport(transport)
|| local_openai_chat_transport_unsupported_reason(transport).is_some()
{
return None;
}
if transport.provider.proxy.is_some()
|| transport.endpoint.proxy.is_some()
|| transport.key.proxy.is_some()
{
return None;
}
let upstream_url = build_local_openai_chat_upstream_url(self.parts, transport)?;
upstream_target_key_from_url(upstream_url.as_str(), None)
}
async fn build_stream_attempt(
&mut self,
attempt: LocalOpenAiChatCandidateAttempt,
) -> Result<Option<AiStreamAttempt>, GatewayError> {
let upstream_is_stream = openai_chat_upstream_is_stream_for_candidate(
@@ -134,6 +323,7 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
self.trace_id,
&self.body_json,
&self.input,
Some(&mut self.request_preparation),
attempt,
OPENAI_CHAT_STREAM_PLAN_KIND,
"openai_chat_stream_success",
@@ -158,6 +348,97 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
}
}
fn openai_chat_stream_target_select_window() -> usize {
std::env::var(OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW_ENV)
.ok()
.and_then(|value| value.trim().parse::<usize>().ok())
.filter(|value| *value > 0)
.unwrap_or(DEFAULT_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW)
.clamp(1, MAX_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW)
}
#[derive(Clone, Copy)]
struct TargetSelectCandidateIdentity<'a> {
provider_id: &'a str,
endpoint_id: &'a str,
key_id: &'a str,
candidate_id: &'a str,
}
#[derive(Clone, Copy)]
struct TargetSelectChoice<'a> {
target_key: &'a str,
identity: TargetSelectCandidateIdentity<'a>,
in_flight: usize,
selection_pressure_total: u64,
}
fn select_target_index(seed: u64, choices: &[TargetSelectChoice<'_>]) -> usize {
choices
.iter()
.enumerate()
.min_by_key(|(index, choice)| {
target_select_score(
seed,
choice.target_key,
&choice.identity,
*index,
choice.in_flight,
choice.selection_pressure_total,
)
})
.map(|(index, _)| index)
.unwrap_or(0)
}
fn target_select_candidate_identity(
attempt: &LocalOpenAiChatCandidateAttempt,
) -> TargetSelectCandidateIdentity<'_> {
TargetSelectCandidateIdentity {
provider_id: &attempt.eligible.candidate.provider_id,
endpoint_id: &attempt.eligible.candidate.endpoint_id,
key_id: &attempt.eligible.candidate.key_id,
candidate_id: &attempt.candidate_id,
}
}
fn target_select_tie_break(
seed: u64,
target_key: &str,
identity: &TargetSelectCandidateIdentity<'_>,
index: usize,
) -> u64 {
let mut hash = seed ^ ((index as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15));
hash = hash_string(hash, target_key);
hash = hash_string(hash, identity.provider_id);
hash = hash_string(hash, identity.endpoint_id);
hash = hash_string(hash, identity.key_id);
hash_string(hash, identity.candidate_id)
}
fn hash_string(mut hash: u64, value: &str) -> u64 {
for byte in value.as_bytes() {
hash ^= u64::from(*byte);
hash = hash.wrapping_mul(0x100_0000_01B3);
}
hash
}
fn target_select_score(
seed: u64,
target_key: &str,
identity: &TargetSelectCandidateIdentity<'_>,
index: usize,
in_flight: usize,
selected_total: u64,
) -> (usize, u64, u64) {
(
in_flight,
selected_total,
target_select_tie_break(seed, target_key, identity, index),
)
}
pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
state: &AppState,
parts: &http::request::Parts,
@@ -207,3 +488,82 @@ pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
Ok(plans)
}
#[cfg(test)]
mod tests {
use super::*;
fn identity<'a>(
endpoint_id: &'a str,
candidate_id: &'a str,
) -> TargetSelectCandidateIdentity<'a> {
TargetSelectCandidateIdentity {
provider_id: "provider",
endpoint_id,
key_id: "key",
candidate_id,
}
}
#[test]
fn target_select_score_prefers_lower_in_flight() {
let busy = identity("endpoint-a", "candidate-a");
let idle = identity("endpoint-b", "candidate-b");
assert!(
target_select_score(7, "http://127.0.0.1:18182|proxy=-", &idle, 1, 0, 10)
< target_select_score(7, "http://127.0.0.1:18181|proxy=-", &busy, 0, 5, 0)
);
}
#[test]
fn target_select_tie_break_distinguishes_equivalent_targets() {
let left = identity("endpoint-a", "candidate-a");
let right = identity("endpoint-b", "candidate-b");
assert_ne!(
target_select_tie_break(11, "http://127.0.0.1:18181|proxy=-", &left, 0),
target_select_tie_break(11, "http://127.0.0.1:18182|proxy=-", &right, 1)
);
}
#[test]
fn select_target_index_prefers_lower_in_flight_target() {
let choices = [
TargetSelectChoice {
target_key: "http://127.0.0.1:18181|proxy=-",
identity: identity("endpoint-a", "candidate-a"),
in_flight: 8,
selection_pressure_total: 0,
},
TargetSelectChoice {
target_key: "http://127.0.0.1:18182|proxy=-",
identity: identity("endpoint-b", "candidate-b"),
in_flight: 1,
selection_pressure_total: 100,
},
];
assert_eq!(select_target_index(17, &choices), 1);
}
#[test]
fn select_target_index_uses_selection_pressure_before_tie_break() {
let choices = [
TargetSelectChoice {
target_key: "http://127.0.0.1:18181|proxy=-",
identity: identity("endpoint-a", "candidate-a"),
in_flight: 0,
selection_pressure_total: 20,
},
TargetSelectChoice {
target_key: "http://127.0.0.1:18182|proxy=-",
identity: identity("endpoint-b", "candidate-b"),
in_flight: 0,
selection_pressure_total: 1,
},
];
assert_eq!(select_target_index(19, &choices), 1);
}
}
@@ -93,7 +93,7 @@ pub(crate) async fn build_local_openai_chat_sync_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiChatSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -116,6 +116,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiChatSyncAttemptSo
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiChatSyncAttemptSource<'_> {
@@ -134,6 +149,7 @@ impl LocalOpenAiChatSyncAttemptSource<'_> {
self.trace_id,
&self.body_json,
&self.input,
None,
attempt,
OPENAI_CHAT_SYNC_PLAN_KIND,
"openai_chat_sync_success",
@@ -129,7 +129,7 @@ pub(crate) fn build_openai_chat_stream_plan_from_decision(
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
stream: effective_upstream_is_stream,
},
);
@@ -229,7 +229,7 @@ pub(crate) fn build_openai_responses_stream_plan_from_decision(
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
stream: effective_upstream_is_stream,
},
);
@@ -291,6 +291,7 @@ mod tests {
request_id: Some("req_123".to_string()),
candidate_id: Some("cand_123".to_string()),
provider_name: Some("Codex".to_string()),
provider_type: Some("codex".to_string()),
provider_id: Some("prov_123".to_string()),
endpoint_id: Some("ep_123".to_string()),
key_id: Some("key_123".to_string()),
@@ -385,6 +386,46 @@ mod tests {
);
}
#[test]
fn build_compact_stream_plan_preserves_non_stream_upstream_mode() {
let parts = http::Request::builder()
.uri("http://localhost/v1/responses/compact")
.body(())
.expect("request should build")
.into_parts()
.0;
let mut payload = sample_responses_payload();
payload.decision_kind = Some("openai_responses_compact_stream".to_string());
payload.upstream_url = Some("https://example.com/v1/responses/compact".to_string());
payload.provider_api_format = Some("openai:responses:compact".to_string());
payload.client_api_format = Some("openai:responses:compact".to_string());
payload.upstream_is_stream = false;
payload.provider_request_body = Some(json!({
"model": "gpt-5.6-sol",
"input": [],
"instructions": "You are Codex.",
"tools": [],
"parallel_tool_calls": true,
"reasoning": {"effort": "high"},
"prompt_cache_key": "cache-key",
"text": {"verbosity": "low"}
}));
let built =
build_openai_responses_stream_plan_from_decision(&parts, &json!({}), payload, true)
.expect("plan build should succeed")
.expect("plan should be produced");
assert!(!built.plan.stream);
assert!(built
.plan
.body
.json_body
.as_ref()
.is_some_and(|body| body.get("stream").is_none()));
assert!(!built.plan.headers.contains_key("accept"));
}
#[test]
fn build_openai_chat_stream_plan_fallback_preserves_complete_same_format_headers() {
let parts = http::Request::builder()
@@ -404,6 +445,7 @@ mod tests {
request_id: Some("req_stream_456".to_string()),
candidate_id: Some("cand_stream_456".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_stream_456".to_string()),
endpoint_id: Some("ep_stream_456".to_string()),
key_id: Some("key_stream_456".to_string()),
@@ -462,7 +504,7 @@ mod tests {
}
#[test]
fn build_openai_chat_stream_plan_keeps_downstream_stream_for_force_non_stream_upstream() {
fn build_openai_chat_stream_plan_preserves_force_non_stream_upstream_mode() {
fn force_non_stream_payload(provider_request_body: Option<Value>) -> AiExecutionDecision {
AiExecutionDecision {
action: "stream".to_string(),
@@ -472,6 +514,7 @@ mod tests {
request_id: Some("req_force_non_stream".to_string()),
candidate_id: Some("cand_force_non_stream".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_force_non_stream".to_string()),
endpoint_id: Some("ep_force_non_stream".to_string()),
key_id: Some("key_force_non_stream".to_string()),
@@ -523,7 +566,7 @@ mod tests {
.expect("plan build should succeed")
.expect("plan should be produced");
assert!(built.plan.stream);
assert!(!built.plan.stream);
assert_eq!(
built
.plan
@@ -548,7 +591,7 @@ mod tests {
.expect("fallback plan build should succeed")
.expect("fallback plan should be produced");
assert!(built.plan.stream);
assert!(!built.plan.stream);
assert_eq!(
built
.plan
@@ -579,6 +622,7 @@ mod tests {
request_id: Some("req_stream_789".to_string()),
candidate_id: Some("cand_stream_789".to_string()),
provider_name: Some("Claude".to_string()),
provider_type: Some("anthropic".to_string()),
provider_id: Some("prov_stream_789".to_string()),
endpoint_id: Some("ep_stream_789".to_string()),
key_id: Some("key_stream_789".to_string()),
@@ -257,6 +257,7 @@ mod tests {
request_id: Some("req_123".to_string()),
candidate_id: Some("cand_123".to_string()),
provider_name: Some("Codex".to_string()),
provider_type: Some("codex".to_string()),
provider_id: Some("prov_123".to_string()),
endpoint_id: Some("ep_123".to_string()),
key_id: Some("key_123".to_string()),
@@ -369,6 +370,7 @@ mod tests {
request_id: Some("req_456".to_string()),
candidate_id: Some("cand_456".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_456".to_string()),
endpoint_id: Some("ep_456".to_string()),
key_id: Some("key_456".to_string()),
@@ -440,6 +442,7 @@ mod tests {
request_id: Some("req_789".to_string()),
candidate_id: Some("cand_789".to_string()),
provider_name: Some("Claude".to_string()),
provider_type: Some("anthropic".to_string()),
provider_id: Some("prov_789".to_string()),
endpoint_id: Some("ep_789".to_string()),
key_id: Some("key_789".to_string()),
@@ -9,7 +9,7 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_openai_responses_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, resolve_transport_request_gzip_policy,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
@@ -142,6 +142,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -204,7 +205,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
image_request_summary: _,
request_redacted: _,
} = resolved;
let request_gzip = resolve_transport_request_gzip_policy(&transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -214,6 +215,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -231,8 +233,8 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip,
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
@@ -241,6 +243,10 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -31,8 +31,8 @@ use crate::ai_serving::planner::spec_metadata::local_openai_responses_spec_metad
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::{
ai_local_execution_contract_for_formats, extract_pool_sticky_session_token,
resolve_local_decision_execution_runtime_auth_context, ExecutionRuntimeAuthContext,
GatewayControlDecision, PlannerAppState,
openai_responses_request_operation, resolve_local_decision_execution_runtime_auth_context,
ExecutionRuntimeAuthContext, GatewayControlDecision, PlannerAppState,
};
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::{AppState, GatewayError};
@@ -91,7 +91,9 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
state,
auth_context.clone(),
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -161,6 +163,7 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
spec: LocalOpenAiResponsesSpec,
) -> Result<(Vec<LocalOpenAiResponsesCandidateAttempt>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let request_operation = openai_responses_request_operation(spec_metadata.api_format, body_json);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
@@ -171,8 +174,10 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
);
let preselection = preselect_local_execution_candidates_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
request_operation,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -259,6 +264,7 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
spec: LocalOpenAiResponsesSpec,
) -> Result<(LocalOpenAiResponsesCandidateAttemptSource<'a>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let request_operation = openai_responses_request_operation(spec_metadata.api_format, body_json);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
@@ -280,9 +286,11 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
Ok(
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
spec_metadata.api_format,
&input.requested_model,
request_operation,
spec_metadata.require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
@@ -365,8 +373,10 @@ pub(crate) async fn build_local_openai_responses_image_candidate_attempt_source<
);
let preselection = preselect_local_execution_candidates_for_api_formats_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
false,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -114,7 +114,7 @@ pub(crate) async fn maybe_build_sync_local_openai_responses_decision_payload(
)
.await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -153,7 +153,7 @@ pub(crate) async fn maybe_build_stream_local_openai_responses_decision_payload(
)
.await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -162,7 +162,7 @@ pub(super) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiResponsesSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -185,12 +185,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiResponsesSyncAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiResponsesStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -213,6 +228,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiResponsesStream
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiResponsesSyncAttemptSource<'_> {
@@ -331,7 +361,7 @@ pub(super) async fn build_local_sync_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -403,7 +433,7 @@ pub(super) async fn build_local_stream_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -1,9 +1,8 @@
use std::collections::BTreeMap;
use aether_contracts::RequestBody;
use super::{
augment_sync_report_context, build_ai_execution_plan_from_decision, take_ai_decision_plan_core,
augment_sync_report_context, build_ai_execution_plan_from_decision,
resolve_ai_passthrough_sync_request_body, take_ai_decision_plan_core,
take_ai_upstream_auth_pair, take_non_empty_string, AiExecutionPlanFromDecisionParts,
AiStreamAttempt, AiSyncAttempt,
};
@@ -63,6 +62,10 @@ pub(crate) fn build_standard_sync_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
@@ -72,7 +75,7 @@ pub(crate) fn build_standard_sync_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream,
},
);
@@ -146,6 +149,11 @@ pub(crate) fn build_standard_stream_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -154,8 +162,8 @@ pub(crate) fn build_standard_stream_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
body: request_body,
stream,
},
);
@@ -165,3 +173,88 @@ pub(crate) fn build_standard_stream_plan_from_decision(
report_context,
}))
}
#[cfg(test)]
mod tests {
use aether_contracts::{ExecutionResponseBodyMode, EXECUTION_RESPONSE_BODY_MODE_HEADER};
use serde_json::json;
use super::{
build_standard_stream_plan_from_decision, build_standard_sync_plan_from_decision,
AiExecutionDecision,
};
fn decision_with_raw_body(upstream_is_stream: bool) -> AiExecutionDecision {
serde_json::from_value(json!({
"action": if upstream_is_stream { "stream" } else { "sync" },
"request_id": "req-raw",
"provider_id": "provider-raw",
"endpoint_id": "endpoint-raw",
"key_id": "key-raw",
"upstream_url": "https://api.anthropic.test/v1/messages",
"provider_api_format": "claude:messages",
"client_api_format": "claude:messages",
"provider_request_headers": {
"content-type": "application/json",
(EXECUTION_RESPONSE_BODY_MODE_HEADER): ExecutionResponseBodyMode::PreserveBytes.as_str()
},
"provider_request_body": {
"model": "claude-sonnet-4",
"messages": []
},
"provider_request_body_base64": "eyAibW9kZWwiOiAiY2xhdWRlLXNvbm5ldC00IiwgIm1lc3NhZ2VzIjogW10gfQ==",
"content_type": "application/json",
"upstream_is_stream": upstream_is_stream
}))
.expect("decision should deserialize")
}
fn request_parts() -> http::request::Parts {
http::Request::builder()
.uri("http://localhost/v1/messages")
.body(())
.expect("request should build")
.into_parts()
.0
}
#[test]
fn standard_sync_plan_prefers_exact_request_body_bytes() {
let built = build_standard_sync_plan_from_decision(
&request_parts(),
&json!({}),
decision_with_raw_body(false),
)
.expect("plan should build")
.expect("plan should exist");
assert!(built.plan.body.json_body.is_none());
assert_eq!(
built.plan.body.body_bytes_b64.as_deref(),
Some("eyAibW9kZWwiOiAiY2xhdWRlLXNvbm5ldC00IiwgIm1lc3NhZ2VzIjogW10gfQ==")
);
assert_eq!(
built
.plan
.headers
.get(EXECUTION_RESPONSE_BODY_MODE_HEADER)
.map(String::as_str),
Some(ExecutionResponseBodyMode::PreserveBytes.as_str())
);
}
#[test]
fn standard_stream_plan_prefers_exact_request_body_bytes() {
let built = build_standard_stream_plan_from_decision(
&request_parts(),
&json!({}),
decision_with_raw_body(true),
false,
)
.expect("plan should build")
.expect("plan should exist");
assert!(built.plan.body.json_body.is_none());
assert!(built.plan.body.body_bytes_b64.is_some());
}
}
@@ -10,9 +10,7 @@ impl<'a> PlannerAppState<'a> {
now_unix_secs: u64,
) -> Result<Option<GatewayAuthApiKeySnapshot>, GatewayError> {
self.app()
.data
.read_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
.read_cached_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
.await
.map_err(|err| GatewayError::Internal(err.to_string()))
}
}
@@ -10,16 +10,15 @@ impl<'a> PlannerAppState<'a> {
api_key_id: &str,
requested_model: Option<&str>,
explicit_required_capabilities: Option<&Value>,
model_directive_base_model: Option<&str>,
) -> Option<Value> {
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled(self.app()).await;
crate::request_candidate_runtime::resolve_request_candidate_required_capabilities(
self.app(),
user_id,
api_key_id,
requested_model,
explicit_required_capabilities,
enable_model_directives,
model_directive_base_model,
)
.await
}
@@ -20,14 +20,8 @@ impl<'a> PlannerAppState<'a> {
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
) -> Result<Vec<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
self.app(),
api_format,
Some(global_model_name),
)
.await;
crate::scheduler::candidate::list_selectable_candidates(
self.app().data.as_ref(),
self.app(),
@@ -52,6 +46,39 @@ impl<'a> PlannerAppState<'a> {
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
Vec<SchedulerSkippedCandidate>,
),
GatewayError,
> {
self.list_selectable_candidates_with_skip_reasons_for_request_operation(
api_format,
global_model_name,
require_streaming,
required_capabilities,
auth_snapshot,
client_session_affinity,
now_unix_secs,
enable_model_directives,
None,
)
.await
}
pub(crate) async fn list_selectable_candidates_with_skip_reasons_for_request_operation(
self,
api_format: &str,
global_model_name: &str,
require_streaming: bool,
required_capabilities: Option<&serde_json::Value>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
request_operation: Option<&str>,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
@@ -63,16 +90,8 @@ impl<'a> PlannerAppState<'a> {
let wait_interval = Duration::from_millis(API_KEY_CONCURRENCY_WAIT_POLL_INTERVAL_MS.max(1));
let wait_deadline = Instant::now() + wait_timeout;
let mut attempt_now_unix_secs = now_unix_secs;
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
self.app(),
api_format,
Some(global_model_name),
)
.await;
loop {
let result = crate::scheduler::candidate::list_selectable_candidates_with_skip_reasons(
let result = crate::scheduler::candidate::list_selectable_candidates_with_skip_reasons_for_request_operation(
self.app().data.as_ref(),
self.app(),
api_format,
@@ -83,6 +102,7 @@ impl<'a> PlannerAppState<'a> {
client_session_affinity,
attempt_now_unix_secs,
enable_model_directives,
request_operation,
)
.await?;
@@ -3,8 +3,20 @@ pub(crate) use crate::ai_serving::transport::{
GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
};
use crate::GatewayError;
use std::sync::Arc;
impl<'a> PlannerAppState<'a> {
pub(crate) async fn read_provider_transport_snapshot_arc(
self,
provider_id: &str,
endpoint_id: &str,
key_id: &str,
) -> Result<Option<Arc<GatewayProviderTransportSnapshot>>, GatewayError> {
self.app()
.read_provider_transport_snapshot_arc(provider_id, endpoint_id, key_id)
.await
}
pub(crate) async fn read_provider_transport_snapshot(
self,
provider_id: &str,
+111 -56
View File
@@ -3,14 +3,21 @@ pub(crate) use aether_ai_formats::api::{
aggregate_openai_chat_stream_sync_response, aggregate_openai_responses_stream_sync_response,
aggregate_standard_chat_stream_sync_response, aggregate_standard_cli_stream_sync_response,
api_format_alias_matches, api_format_storage_aliases,
apply_codex_openai_responses_chat_body_edits, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_responses_special_headers, apply_model_directive_mapping_patch,
apply_codex_openai_compact_terminal_headers, apply_codex_openai_responses_chat_body_edits,
apply_codex_openai_responses_identity_headers,
apply_codex_openai_responses_lite_header_for_request_body_with_capabilities,
apply_codex_openai_responses_lite_header_with_capabilities,
apply_codex_openai_responses_special_body_edits,
apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities,
apply_codex_openai_special_headers, apply_model_directive_mapping_patch,
apply_model_directive_overrides_from_model, apply_model_directive_overrides_from_request,
apply_openai_responses_compact_special_body_edits, build_chatgpt_web_image_request_body,
build_codex_model_catalog_metadata, build_codex_openai_image_api_provider_request_body,
build_core_error_body_for_client_format, build_cross_format_openai_chat_request_body,
build_cross_format_openai_chat_request_body_with_model_directives,
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_model_directives,
build_cross_format_openai_responses_request_body_with_model_directives_and_history_scope,
build_gemini_image_request_body_from_openai_image_request,
build_gemini_image_response_from_openai_image_response,
build_gemini_image_response_from_openai_responses_image_response, build_generated_tool_call_id,
@@ -41,17 +48,21 @@ pub(crate) use aether_ai_formats::api::{
convert_standard_chat_response, convert_standard_cli_response, copy_request_number_field,
copy_request_number_field_as, core_error_background_report_kind,
core_error_default_client_api_format, core_success_background_report_kind,
default_model_directive_mapping_patch, default_model_directive_suffixes,
default_model_for_openai_image_operation, encode_done_sse, encode_json_sse,
encode_kiro_sse_events, endpoint_config_forces_upstream_stream_policy,
enforce_request_body_stream_field, estimate_kiro_tokens, extract_openai_text_content,
finalize_openai_provider_request,
finalize_openai_provider_request_with_codex_model_capabilities,
find_kiro_real_thinking_end_tag, find_kiro_real_thinking_end_tag_at_buffer_end,
find_kiro_real_thinking_start_tag, force_upstream_streaming_for_provider,
gemini_request_is_image_generation, implicit_sync_finalize_report_kind,
is_core_error_finalize_kind, is_matching_stream_http_request, is_matching_stream_request,
is_openai_image_stream_request, is_openai_responses_family_format, is_openai_responses_format,
kiro_crc32, map_claude_stop_reason, map_openai_reasoning_effort_to_claude_output,
map_openai_reasoning_effort_to_gemini_budget, maybe_bridge_standard_sync_json_to_stream,
maybe_build_ai_surface_stream_rewriter,
find_kiro_real_thinking_start_tag, forbid_upstream_streaming_for_provider,
force_upstream_streaming_for_provider, gemini_request_is_image_generation,
hydrate_response_history, implicit_sync_finalize_report_kind, is_core_error_finalize_kind,
is_matching_stream_http_request, is_matching_stream_request, is_openai_image_stream_request,
is_openai_responses_compact_format, is_openai_responses_family_format,
is_openai_responses_format, kiro_crc32, map_claude_stop_reason,
map_openai_reasoning_effort_to_claude_output, map_openai_reasoning_effort_to_gemini_budget,
maybe_bridge_standard_sync_json_to_stream, maybe_build_ai_surface_stream_rewriter,
maybe_build_openai_chat_cross_format_sync_product_from_normalized_payload,
maybe_build_openai_image_sync_finalize_product,
maybe_build_openai_responses_cross_format_sync_product_from_normalized_payload,
@@ -60,23 +71,31 @@ pub(crate) use aether_ai_formats::api::{
maybe_build_standard_cross_format_sync_product_from_normalized_payload,
maybe_build_standard_same_format_sync_body_from_normalized_payload,
maybe_build_standard_sync_finalize_product_from_normalized_payload, model_directive_base_model,
normalize_api_format_alias, normalize_claude_request_to_openai_chat_request,
normalize_gemini_request_to_openai_chat_request, normalize_openai_image_request,
normalize_openai_image_request_with_options,
model_directive_builtin_suffix_supported_for_source_model,
model_directive_suffix_has_builtin_mapping, normalize_api_format_alias,
normalize_claude_request_to_openai_chat_request,
normalize_gemini_request_to_openai_chat_request, normalize_openai_image_quality,
normalize_openai_image_request, normalize_openai_image_request_with_options,
normalize_openai_responses_request_to_openai_chat_request,
normalize_provider_private_report_context, normalize_provider_private_response_value,
normalize_standard_request_to_openai_chat_request, openai_image_operation_from_path,
parse_direct_request_body, parse_openai_stop_sequences, parse_openai_tool_result_content,
prepare_local_success_response_parts, prepare_local_success_response_parts_owned,
provider_adaptation_allows_sync_finalize_envelope, provider_adaptation_anchor_api_format,
provider_adaptation_descriptor_for_envelope, provider_adaptation_descriptor_for_provider_type,
parse_codex_auth_identity, parse_direct_request_body, parse_model_directive,
parse_model_directive_with_suffixes, parse_openai_stop_sequences,
parse_openai_tool_result_content, prepare_local_success_response_parts,
prepare_local_success_response_parts_owned, project_codex_openai_image_api_request_body,
project_openai_image_api_request_body, provider_adaptation_allows_sync_finalize_envelope,
provider_adaptation_anchor_api_format, provider_adaptation_descriptor_for_envelope,
provider_adaptation_descriptor_for_provider_type,
provider_adaptation_requires_eventstream_accept,
provider_adaptation_should_unwrap_stream_envelope,
provider_private_response_allows_sync_finalize, request_candidate_api_format_preference,
request_candidate_api_formats, request_conversion_kind,
request_conversion_requires_enable_flag, request_path_implies_stream_request,
resolve_claude_stream_spec, resolve_claude_sync_spec,
resolve_execution_runtime_stream_plan_kind, resolve_execution_runtime_sync_plan_kind,
provider_private_response_allows_sync_finalize, record_converted_response_history,
request_candidate_api_format_preference, request_candidate_api_formats,
request_conversion_kind, request_conversion_requires_enable_flag,
request_path_implies_stream_request, resolve_claude_stream_spec, resolve_claude_sync_spec,
resolve_codex_responses_model_capabilities, resolve_execution_runtime_stream_plan_kind,
resolve_execution_runtime_stream_plan_kind_with_client_surface,
resolve_execution_runtime_sync_plan_kind,
resolve_execution_runtime_sync_plan_kind_with_client_surface,
resolve_finalize_stream_rewrite_mode, resolve_gemini_files_stream_spec,
resolve_gemini_files_sync_spec, resolve_gemini_stream_spec, resolve_gemini_sync_spec,
resolve_local_image_stream_spec, resolve_local_image_sync_spec,
@@ -85,24 +104,28 @@ pub(crate) use aether_ai_formats::api::{
resolve_openai_embedding_sync_spec, resolve_openai_responses_stream_spec,
resolve_openai_responses_sync_spec, resolve_requested_gemini_image_model_for_request,
resolve_requested_openai_image_model_for_request,
resolve_upstream_is_stream_from_endpoint_config, sanitize_request_path,
sanitize_request_path_and_query, sanitize_request_query_string,
stream_body_contains_error_event, supports_stream_execution_decision_kind,
supports_sync_execution_decision_kind, sync_chat_response_conversion_kind,
sync_cli_response_conversion_kind, transform_provider_private_stream_line, value_as_u64,
AiControlPlanRequest, AiSurfaceFinalizeError, AiSurfaceStreamRewriter, CanonicalStreamFrame,
ChatGptWebImageRequestError, ClaudeClientEmitter, ClaudeProviderState,
ExecutionRuntimeAuthContext, FinalizeStreamRewriteMode, FormatContext, GeminiClientEmitter,
GeminiImageRequestForOpenAi, GeminiProviderState, KiroToClaudeCliStreamState,
LocalCoreSyncErrorKind, LocalGeminiFilesSpec, LocalOpenAiImageSpec, LocalOpenAiResponsesSpec,
LocalSameFormatProviderFamily, LocalSameFormatProviderSpec, LocalStandardSourceFamily,
LocalStandardSourceMode, LocalStandardSpec, LocalSyncReportParts, LocalVideoCreateFamily,
LocalVideoCreateSpec, NormalizedOpenAiImageRequest, OpenAIChatClientEmitter,
OpenAIChatProviderState, OpenAIResponsesClientEmitter, OpenAIResponsesProviderState,
OpenAiImageNormalizeOptions, OpenAiImageOperation, OpenAiImageRequestForGemini,
OpenAiImageResponseFormat, OpenAiImageStreamState, OpenAiImageSyncFinalizeProduct,
resolve_upstream_is_stream_for_provider as resolve_format_upstream_is_stream_for_provider,
resolve_upstream_is_stream_from_endpoint_config, response_history_is_loaded,
response_history_storage_key, sanitize_request_path, sanitize_request_path_and_query,
sanitize_request_query_string, stream_body_contains_error_event,
supports_stream_execution_decision_kind, supports_sync_execution_decision_kind,
sync_chat_response_conversion_kind, sync_cli_response_conversion_kind,
transform_provider_private_stream_line, validate_openai_provider_request_contract,
value_as_u64, AiControlPlanRequest, AiSurfaceFinalizeError, AiSurfaceStreamRewriter,
CanonicalStreamFrame, ChatGptWebImageRequestError, ClaudeClientEmitter, ClaudeProviderState,
CodexResponsesModelCapabilities, ExecutionRuntimeAuthContext, FinalizeStreamRewriteMode,
FormatContext, GeminiClientEmitter, GeminiImageRequestForOpenAi, GeminiProviderState,
KiroToClaudeCliStreamState, LocalCoreSyncErrorKind, LocalGeminiFilesSpec, LocalOpenAiImageSpec,
LocalOpenAiResponsesSpec, LocalSameFormatProviderFamily, LocalSameFormatProviderSpec,
LocalStandardSourceFamily, LocalStandardSourceMode, LocalStandardSpec, LocalSyncReportParts,
LocalVideoCreateFamily, LocalVideoCreateSpec, NormalizedOpenAiImageRequest,
OpenAIChatClientEmitter, OpenAIChatProviderState, OpenAIResponsesClientEmitter,
OpenAIResponsesProviderState, OpenAiImageNormalizeOptions, OpenAiImageOperation,
OpenAiImageRequestForGemini, OpenAiImageResponseFormat, OpenAiImageStreamState,
OpenAiImageSyncFinalizeProduct, OpenAiProviderRequestFinalization,
ProviderAdaptationDescriptor, ProviderAdaptationSurface, ProviderPrivateStreamNormalizer,
RequestConversionKind, StandardCrossFormatSyncProduct, StandardSyncFinalizeNormalizedProduct,
ReasoningEffort, RequestConversionKind, ResponseHistoryRecord, ServiceTier,
StandardCrossFormatSyncProduct, StandardSyncFinalizeNormalizedProduct,
StreamingStandardFormatMatrix, SyncChatResponseConversionKind, SyncCliResponseConversionKind,
SyncToStreamBridgeOutcome, ANTIGRAVITY_V1INTERNAL_ENVELOPE_NAME, CLAUDE_CHAT_STREAM_PLAN_KIND,
CLAUDE_CHAT_STREAM_SUCCESS_REPORT_KIND, CLAUDE_CHAT_SYNC_ERROR_REPORT_KIND,
@@ -110,14 +133,15 @@ pub(crate) use aether_ai_formats::api::{
CLAUDE_CHAT_SYNC_SUCCESS_REPORT_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_STREAM_SUCCESS_REPORT_KIND, CLAUDE_CLI_SYNC_ERROR_REPORT_KIND,
CLAUDE_CLI_SYNC_FINALIZE_REPORT_KIND, CLAUDE_CLI_SYNC_PLAN_KIND,
CLAUDE_CLI_SYNC_SUCCESS_REPORT_KIND, CODEX_OPENAI_IMAGE_DEFAULT_MODEL,
CODEX_OPENAI_IMAGE_DEFAULT_OUTPUT_FORMAT, CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_MODEL,
CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_PROMPT, CODEX_OPENAI_IMAGE_INTERNAL_MODEL,
EXECUTION_RUNTIME_STREAM_ACTION, EXECUTION_RUNTIME_STREAM_DECISION_ACTION,
EXECUTION_RUNTIME_SYNC_ACTION, EXECUTION_RUNTIME_SYNC_DECISION_ACTION,
GEMINI_CHAT_STREAM_PLAN_KIND, GEMINI_CHAT_STREAM_SUCCESS_REPORT_KIND,
GEMINI_CHAT_SYNC_ERROR_REPORT_KIND, GEMINI_CHAT_SYNC_FINALIZE_REPORT_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CHAT_SYNC_SUCCESS_REPORT_KIND, GEMINI_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_SUCCESS_REPORT_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND,
CODEX_OPENAI_IMAGE_DEFAULT_MODEL, CODEX_OPENAI_IMAGE_DEFAULT_OUTPUT_FORMAT,
CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_MODEL, CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_PROMPT,
CODEX_OPENAI_IMAGE_INTERNAL_MODEL, EXECUTION_RUNTIME_STREAM_ACTION,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_STREAM_SUCCESS_REPORT_KIND, GEMINI_CHAT_SYNC_ERROR_REPORT_KIND,
GEMINI_CHAT_SYNC_FINALIZE_REPORT_KIND, GEMINI_CHAT_SYNC_PLAN_KIND,
GEMINI_CHAT_SYNC_SUCCESS_REPORT_KIND, GEMINI_CLI_STREAM_PLAN_KIND,
GEMINI_CLI_STREAM_SUCCESS_REPORT_KIND, GEMINI_CLI_SYNC_ERROR_REPORT_KIND,
GEMINI_CLI_SYNC_FINALIZE_REPORT_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
GEMINI_CLI_SYNC_SUCCESS_REPORT_KIND, GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME,
@@ -125,22 +149,53 @@ pub(crate) use aether_ai_formats::api::{
GEMINI_FILES_DELETE_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND,
GEMINI_FILES_LIST_PLAN_KIND, GEMINI_FILES_UPLOAD_PLAN_KIND, GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND,
GEMINI_VIDEO_CREATE_SYNC_FINALIZE_REPORT_KIND, GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND,
KIRO_ENVELOPE_NAME, KIRO_MAX_THINKING_BUFFER, OPENAI_CHAT_STREAM_PLAN_KIND,
OPENAI_CHAT_STREAM_SUCCESS_REPORT_KIND, OPENAI_CHAT_SYNC_ERROR_REPORT_KIND,
OPENAI_CHAT_SYNC_FINALIZE_REPORT_KIND, OPENAI_CHAT_SYNC_PLAN_KIND,
OPENAI_CHAT_SYNC_SUCCESS_REPORT_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_STREAM_SUCCESS_REPORT_KIND,
OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_IMAGE_SYNC_SUCCESS_REPORT_KIND, OPENAI_RERANK_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_SUCCESS_REPORT_KIND,
KIRO_ENVELOPE_NAME, KIRO_MAX_THINKING_BUFFER, MODEL_DIRECTIVE_API_FORMATS,
OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_STREAM_SUCCESS_REPORT_KIND,
OPENAI_CHAT_SYNC_ERROR_REPORT_KIND, OPENAI_CHAT_SYNC_FINALIZE_REPORT_KIND,
OPENAI_CHAT_SYNC_PLAN_KIND, OPENAI_CHAT_SYNC_SUCCESS_REPORT_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND,
OPENAI_IMAGE_STREAM_SUCCESS_REPORT_KIND, OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_IMAGE_SYNC_SUCCESS_REPORT_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_SUCCESS_REPORT_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_ERROR_REPORT_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_FINALIZE_REPORT_KIND, OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_SUCCESS_REPORT_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_STREAM_SUCCESS_REPORT_KIND, OPENAI_RESPONSES_SYNC_ERROR_REPORT_KIND,
OPENAI_RESPONSES_SYNC_FINALIZE_REPORT_KIND, OPENAI_RESPONSES_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_SUCCESS_REPORT_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_SUCCESS_REPORT_KIND, OPENAI_SEARCH_SYNC_PLAN_KIND,
OPENAI_SEARCH_SYNC_SUCCESS_REPORT_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_formats::{is_embedding_api_format, is_rerank_api_format};
pub(crate) use aether_ai_formats::{
api_format_defaults_to_client_error_failover, api_format_defaults_to_non_stream,
api_format_permission_covers, intersect_api_format_allowed_lists, is_embedding_api_format,
is_rerank_api_format, openai_responses_request_operation,
openai_responses_synthetic_reasoning_item_id,
strip_incompatible_openai_responses_reasoning_items, ApiOperation, ClientSurface,
};
pub(crate) fn plan_kind_matches_api_operation(
plan_kind: &str,
require_streaming: bool,
expected_operation: Option<ApiOperation>,
) -> bool {
let Some(expected_operation) = expected_operation else {
return true;
};
if expected_operation == ApiOperation::OpenAiResponsesCompact {
return if require_streaming {
plan_kind == OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND
} else {
plan_kind == OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND
};
}
let resolved_operation = if require_streaming {
resolve_local_same_format_stream_spec(plan_kind).and_then(|spec| spec.operation)
} else {
resolve_local_same_format_sync_spec(plan_kind).and_then(|spec| spec.operation)
};
resolved_operation == Some(expected_operation)
}
@@ -0,0 +1,96 @@
use crate::ai_serving::{
hydrate_response_history, normalize_api_format_alias, record_converted_response_history,
response_history_is_loaded, response_history_storage_key, ResponseHistoryRecord,
};
use aether_runtime_state::RuntimeState;
use serde_json::Value;
use tracing::warn;
use crate::GatewayError;
pub(crate) async fn hydrate_openai_response_history(
runtime_state: &RuntimeState,
request: &Value,
client_api_format: &str,
provider_api_format: &str,
history_scope: &str,
) -> Result<(), GatewayError> {
if normalize_api_format_alias(client_api_format) != "openai:responses"
|| normalize_api_format_alias(provider_api_format) != "openai:chat"
{
return Ok(());
}
let Some(previous_response_id) = request
.get("previous_response_id")
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
else {
return Ok(());
};
if response_history_is_loaded(previous_response_id, Some(history_scope)) {
return Ok(());
}
let storage_key = response_history_storage_key(previous_response_id, Some(history_scope));
let payload = runtime_state.kv_get(&storage_key).await.map_err(|error| {
warn!(
event_name = "openai_response_history_read_failed",
log_type = "ops",
backend = runtime_state.backend_kind().as_str(),
error = ?error,
"gateway failed to read shared OpenAI response history"
);
GatewayError::Internal("OpenAI response history lookup failed".to_string())
})?;
let Some(payload) = payload else {
return Ok(());
};
if let Err(error) =
hydrate_response_history(previous_response_id, Some(history_scope), &payload)
{
let _ = runtime_state.kv_delete(&storage_key).await;
warn!(
event_name = "openai_response_history_invalid",
log_type = "ops",
backend = runtime_state.backend_kind().as_str(),
error = %error,
"gateway rejected invalid shared OpenAI response history"
);
return Err(GatewayError::Internal(
"OpenAI response history validation failed".to_string(),
));
}
Ok(())
}
pub(crate) async fn persist_response_history_record(
runtime_state: &RuntimeState,
record: ResponseHistoryRecord,
) {
if let Err(error) = runtime_state
.kv_set(&record.storage_key, record.payload, Some(record.ttl))
.await
{
warn!(
event_name = "openai_response_history_write_failed",
log_type = "ops",
backend = runtime_state.backend_kind().as_str(),
error = ?error,
"gateway failed to persist shared OpenAI response history"
);
}
}
pub(crate) async fn persist_converted_response_history(
runtime_state: &RuntimeState,
report_context: &Value,
response: Option<&Value>,
) {
let Some(response) = response else {
return;
};
if let Some(record) = record_converted_response_history(report_context, response) {
persist_response_history_record(runtime_state, record).await;
}
}
+21 -20
View File
@@ -59,8 +59,8 @@ pub(crate) mod windsurf {
}
pub(crate) use aether_provider_transport::{
append_transport_diagnostics_to_value, apply_local_body_rules,
apply_local_body_rules_with_request_headers, apply_local_header_rules,
append_transport_diagnostics_to_value, apply_local_auth_config_header_overrides,
apply_local_body_rules, apply_local_body_rules_with_request_headers, apply_local_header_rules,
apply_local_header_rules_with_request_headers, apply_standard_provider_request_body_rules,
apply_standard_provider_request_body_rules_with_request_headers,
apply_transport_request_body_semantics, body_rules_are_locally_supported,
@@ -82,10 +82,11 @@ pub(crate) use aether_provider_transport::{
build_windsurf_cascade_headers, build_windsurf_cascade_request_body,
build_windsurf_cascade_upstream_url, candidate_common_transport_skip_reason,
candidate_transport_pair_skip_reason, classify_same_format_provider_request_behavior,
ensure_upstream_auth_header, gemini_files_transport_unsupported_reason,
header_rules_are_locally_supported, header_rules_have_enabled_rules,
is_gemini_cli_provider_transport, is_windsurf_provider_transport,
local_gemini_transport_unsupported_reason_with_network,
classify_same_format_provider_request_behavior_for_operation,
enforce_same_format_provider_api_operation_body_policy, ensure_upstream_auth_header,
gemini_files_transport_unsupported_reason, header_rules_are_locally_supported,
header_rules_have_enabled_rules, is_gemini_cli_provider_transport,
is_windsurf_provider_transport, local_gemini_transport_unsupported_reason_with_network,
local_openai_chat_transport_unsupported_reason,
local_standard_transport_unsupported_reason_with_network,
local_windsurf_request_transport_unsupported_reason_with_network,
@@ -93,22 +94,22 @@ pub(crate) use aether_provider_transport::{
request_conversion_enabled_for_transport, request_conversion_transport_supported,
request_conversion_transport_unsupported_reason, request_pair_allowed_for_transport,
request_pair_direct_auth, request_pair_transport_unsupported_reason,
resolve_gemini_cli_project_id, resolve_gemini_files_auth, resolve_grok_session_auth,
resolve_local_gemini_cli_request_auth, resolve_openai_image_auth,
resolve_same_format_provider_direct_auth, resolve_transport_execution_timeouts,
resolve_transport_profile, resolve_transport_proxy_snapshot,
resolve_transport_proxy_snapshot_with_tunnel_affinity, resolve_video_create_auth,
same_format_provider_transport_supported, same_format_provider_transport_unsupported_reason,
should_skip_upstream_passthrough_header, should_try_same_format_provider_oauth_auth,
supports_local_gemini_transport_with_network,
resolve_anthropic_compatibility_profile, resolve_gemini_cli_project_id,
resolve_gemini_files_auth, resolve_grok_session_auth, resolve_local_gemini_cli_request_auth,
resolve_openai_image_auth, resolve_same_format_provider_direct_auth,
resolve_transport_execution_timeouts, resolve_transport_profile,
resolve_transport_proxy_snapshot, resolve_transport_proxy_snapshot_with_tunnel_affinity,
resolve_video_create_auth, same_format_provider_transport_supported,
same_format_provider_transport_unsupported_reason, should_skip_upstream_passthrough_header,
should_try_same_format_provider_oauth_auth, supports_local_gemini_transport_with_network,
supports_local_generic_oauth_request_auth_resolution,
supports_local_oauth_request_auth_resolution, transport_proxy_is_locally_supported,
video_create_transport_unsupported_reason, CandidateTransportPolicyFacts,
GatewayProviderTransportSnapshot, GeminiCliRequestAuth, GeminiCliRequestAuthSupport,
GeminiCliRequestAuthUnsupportedReason, GeminiCliRequestEnvelopeSupport,
GeminiFilesHeadersInput, GeminiFilesRequestBodyError, GeminiFilesRequestBodyParts,
GrokHeaderInput, LocalResolvedOAuthRequestAuth, ProviderOpenAiImageHeadersInput,
ProviderVideoCreateFamily, ProviderVideoCreateHeadersInput,
transport_supports_api_operation, video_create_transport_unsupported_reason,
AnthropicCompatibilityProfile, CandidateTransportPolicyFacts, GatewayProviderTransportSnapshot,
GeminiCliRequestAuth, GeminiCliRequestAuthSupport, GeminiCliRequestAuthUnsupportedReason,
GeminiCliRequestEnvelopeSupport, GeminiFilesHeadersInput, GeminiFilesRequestBodyError,
GeminiFilesRequestBodyParts, GrokHeaderInput, LocalResolvedOAuthRequestAuth,
ProviderOpenAiImageHeadersInput, ProviderVideoCreateFamily, ProviderVideoCreateHeadersInput,
SameFormatProviderCompatibilityEdit, SameFormatProviderCompatibilityEditAction,
SameFormatProviderFamily, SameFormatProviderHeadersInput, SameFormatProviderRequestBehavior,
SameFormatProviderRequestBehaviorParams, SameFormatProviderRequestBodyInput,
@@ -0,0 +1,218 @@
use aether_runtime::{MetricKind, MetricSample};
pub(crate) fn gateway_allocator_metric_samples() -> Vec<MetricSample> {
match allocator_snapshot() {
Some(snapshot) => snapshot.to_metric_samples(),
None => unavailable_metric_samples(),
}
}
#[derive(Debug, Clone, Copy, Default)]
struct AllocatorSnapshot {
allocated_bytes: u64,
active_bytes: u64,
resident_bytes: u64,
mapped_bytes: u64,
retained_bytes: u64,
metadata_bytes: u64,
}
impl AllocatorSnapshot {
fn to_metric_samples(self) -> Vec<MetricSample> {
vec![
gauge(
"gateway_allocator_observability_available",
"Whether gateway allocator heap metrics were available for this scrape.",
1,
),
gauge(
"gateway_allocator_allocated_bytes",
"Bytes currently allocated by the gateway allocator.",
self.allocated_bytes,
),
gauge(
"gateway_allocator_active_bytes",
"Bytes in active pages managed by the gateway allocator.",
self.active_bytes,
),
gauge(
"gateway_allocator_resident_bytes",
"Bytes resident in physical memory for the gateway allocator.",
self.resident_bytes,
),
gauge(
"gateway_allocator_mapped_bytes",
"Bytes mapped by the gateway allocator.",
self.mapped_bytes,
),
gauge(
"gateway_allocator_retained_bytes",
"Bytes retained by the gateway allocator for future use.",
self.retained_bytes,
),
gauge(
"gateway_allocator_metadata_bytes",
"Bytes used for allocator metadata.",
self.metadata_bytes,
),
gauge(
"gateway_allocator_active_to_allocated_basis_points",
"Active allocator bytes divided by allocated bytes in basis points.",
ratio_basis_points(self.active_bytes, self.allocated_bytes),
),
gauge(
"gateway_allocator_resident_to_allocated_basis_points",
"Resident allocator bytes divided by allocated bytes in basis points.",
ratio_basis_points(self.resident_bytes, self.allocated_bytes),
),
]
}
}
fn unavailable_metric_samples() -> Vec<MetricSample> {
vec![
gauge(
"gateway_allocator_observability_available",
"Whether gateway allocator heap metrics were available for this scrape.",
0,
),
gauge(
"gateway_allocator_allocated_bytes",
"Bytes currently allocated by the gateway allocator.",
0,
),
gauge(
"gateway_allocator_active_bytes",
"Bytes in active pages managed by the gateway allocator.",
0,
),
gauge(
"gateway_allocator_resident_bytes",
"Bytes resident in physical memory for the gateway allocator.",
0,
),
gauge(
"gateway_allocator_mapped_bytes",
"Bytes mapped by the gateway allocator.",
0,
),
gauge(
"gateway_allocator_retained_bytes",
"Bytes retained by the gateway allocator for future use.",
0,
),
gauge(
"gateway_allocator_metadata_bytes",
"Bytes used for allocator metadata.",
0,
),
gauge(
"gateway_allocator_active_to_allocated_basis_points",
"Active allocator bytes divided by allocated bytes in basis points.",
0,
),
gauge(
"gateway_allocator_resident_to_allocated_basis_points",
"Resident allocator bytes divided by allocated bytes in basis points.",
0,
),
]
}
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
fn allocator_snapshot() -> Option<AllocatorSnapshot> {
refresh_jemalloc_epoch()?;
Some(AllocatorSnapshot {
allocated_bytes: read_jemalloc_stat("stats.allocated\0")?,
active_bytes: read_jemalloc_stat("stats.active\0")?,
resident_bytes: read_jemalloc_stat("stats.resident\0")?,
mapped_bytes: read_jemalloc_stat("stats.mapped\0")?,
retained_bytes: read_jemalloc_stat("stats.retained\0")?,
metadata_bytes: read_jemalloc_stat("stats.metadata\0")?,
})
}
#[cfg(not(all(feature = "jemalloc", not(target_env = "msvc"))))]
fn allocator_snapshot() -> Option<AllocatorSnapshot> {
None
}
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
fn refresh_jemalloc_epoch() -> Option<()> {
let mut epoch = 1_u64;
let result = unsafe {
tikv_jemalloc_sys::mallctl(
c"epoch".as_ptr(),
std::ptr::null_mut(),
std::ptr::null_mut(),
(&mut epoch as *mut u64).cast(),
std::mem::size_of::<u64>(),
)
};
if result == 0 {
Some(())
} else {
None
}
}
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
fn read_jemalloc_stat(name: &str) -> Option<u64> {
let mut value = 0_usize;
let mut size = std::mem::size_of::<usize>();
let result = unsafe {
tikv_jemalloc_sys::mallctl(
name.as_ptr().cast(),
(&mut value as *mut usize).cast(),
&mut size,
std::ptr::null_mut(),
0,
)
};
if result == 0 {
Some(u64_from_usize(value))
} else {
None
}
}
fn gauge(name: &'static str, help: &'static str, value: u64) -> MetricSample {
MetricSample::new(name, help, MetricKind::Gauge, value)
}
fn ratio_basis_points(numerator: u64, denominator: u64) -> u64 {
if denominator == 0 {
return 0;
}
numerator.saturating_mul(10_000) / denominator
}
fn u64_from_usize(value: usize) -> u64 {
u64::try_from(value).unwrap_or(u64::MAX)
}
#[cfg(test)]
mod tests {
use super::{gateway_allocator_metric_samples, ratio_basis_points};
#[test]
fn renders_allocator_metric_samples() {
let samples = gateway_allocator_metric_samples();
assert!(samples
.iter()
.any(|sample| sample.name == "gateway_allocator_observability_available"));
assert!(samples
.iter()
.any(|sample| sample.name == "gateway_allocator_allocated_bytes"));
assert!(samples
.iter()
.any(|sample| sample.name == "gateway_allocator_active_to_allocated_basis_points"));
}
#[test]
fn computes_ratio_basis_points() {
assert_eq!(ratio_basis_points(150, 100), 15_000);
assert_eq!(ratio_basis_points(1, 0), 0);
}
}
+2
View File
@@ -1,6 +1,7 @@
pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
match crate::ai_serving::normalize_api_format_alias(api_format).as_str() {
"gemini:generate_content" => Some("gemini:generate_content"),
"gemini:interactions" => Some("gemini:interactions"),
"gemini:embedding" => Some("gemini:embedding"),
"gemini:video" => Some("gemini:video"),
"gemini:files" => Some("gemini:files"),
@@ -11,6 +12,7 @@ pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
pub(crate) fn local_path(api_format: &str) -> Option<&'static str> {
match crate::ai_serving::normalize_api_format_alias(api_format).as_str() {
"gemini" | "gemini:generate_content" => Some("/v1beta/models/{model}:{action}"),
"gemini:interactions" => Some("/v1/interactions"),
"gemini:embedding" => Some("/v1beta/models/{model}:{action}"),
"gemini:video" => Some("/v1beta/models/{model}:predictLongRunning"),
"gemini:files" => Some("/v1beta/files"),
+2
View File
@@ -5,6 +5,7 @@ pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
"openai:rerank" => Some("openai:rerank"),
"openai:responses" => Some("openai:responses"),
"openai:responses:compact" => Some("openai:responses:compact"),
"openai:search" => Some("openai:search"),
"openai:image" => Some("openai:image"),
"openai:video" => Some("openai:video"),
_ => None,
@@ -18,6 +19,7 @@ pub(crate) fn local_path(api_format: &str) -> Option<&'static str> {
"openai:rerank" => Some("/v1/rerank"),
"openai:responses" => Some("/v1/responses"),
"openai:responses:compact" => Some("/v1/responses/compact"),
"openai:search" => Some("/v1/alpha/search"),
"openai:image" => Some("/v1/images/generations"),
"openai:video" => Some("/v1/videos"),
_ => None,
+41 -3
View File
@@ -1,8 +1,13 @@
use axum::body::Body;
use axum::extract::Request;
use axum::http::{header, HeaderValue, Response, StatusCode};
use axum::routing::{any, post};
use axum::Router;
use super::{aliyun, claude, doubao, gemini, jina, openai};
use crate::{handlers::proxy::proxy_request, state::AppState};
use crate::api::response::build_local_http_error_response_with_request_path;
use crate::headers::extract_or_generate_trace_id;
use crate::{handlers::proxy::proxy_request, state::AppState, GatewayError};
// Router registration patterns live here so AI public ingress has a single mount registry.
// They intentionally stay separate from manifest-facing route inventories in constants.rs,
@@ -11,22 +16,27 @@ const AI_POST_ROUTE_PATTERNS: &[&str] = &[
"/v1/chat/completions",
"/v1/embeddings",
"/v1/rerank",
"/v1/messages",
"/v1/messages/count_tokens",
"/v1/responses",
"/v1/responses/compact",
"/v1/alpha/search",
"/v1/images/generations",
"/v1/images/edits",
"/v1/interactions",
"/v1beta/interactions",
"/v1internal:loadCodeAssist",
"/v1internal:fetchAvailableModels",
"/v1internal:retrieveUserQuotaSummary",
"/v1internal:fetchUserInfo",
"/v1internal:fetchAdminControls",
"/v1internal:setUserSettings",
"/v1internal:listExperiments",
"/v1internal:recordCodeAssistMetrics",
"/v1internal:writeTrajectoryAcls",
"/v1internal:streamGenerateContent",
];
const CLAUDE_POST_ROUTE_PATTERNS: &[&str] = &["/v1/messages", "/v1/messages/count_tokens"];
const AI_ANY_ROUTE_PATTERNS: &[&str] = &[
"/v1/models/{*gemini_path}",
"/v1beta/models/{*gemini_path}",
@@ -43,12 +53,33 @@ pub(crate) fn mount_ai_routes(mut router: Router<AppState>) -> Router<AppState>
for path in AI_POST_ROUTE_PATTERNS {
router = router.route(path, post(proxy_request));
}
for path in CLAUDE_POST_ROUTE_PATTERNS {
router = router.route(
path,
post(proxy_request).fallback(claude_method_not_allowed),
);
}
for path in AI_ANY_ROUTE_PATTERNS {
router = router.route(path, any(proxy_request));
}
router
}
async fn claude_method_not_allowed(request: Request) -> Result<Response<Body>, GatewayError> {
let trace_id = extract_or_generate_trace_id(request.headers());
let mut response = build_local_http_error_response_with_request_path(
&trace_id,
None,
Some(request.uri().path()),
StatusCode::METHOD_NOT_ALLOWED,
"Method not allowed",
)?;
response
.headers_mut()
.insert(header::ALLOW, HeaderValue::from_static("POST"));
Ok(response)
}
pub(crate) fn public_api_format_local_path(api_format: &str) -> &'static str {
let normalized = api_format.trim().to_ascii_lowercase();
openai::local_path(&normalized)
@@ -95,6 +126,12 @@ mod tests {
fn supports_data_api_endpoint_signatures_and_public_paths() {
for (api_format, family, kind, path) in [
("openai:embedding", "openai", "embedding", "/v1/embeddings"),
(
"gemini:interactions",
"gemini",
"interactions",
"/v1/interactions",
),
(
"gemini:embedding",
"gemini",
@@ -110,6 +147,7 @@ mod tests {
"/api/v1/services/embeddings/multimodal-embedding/multimodal-embedding",
),
("openai:rerank", "openai", "rerank", "/v1/rerank"),
("openai:search", "openai", "search", "/v1/alpha/search"),
("jina:rerank", "jina", "rerank", "/v1/rerank"),
] {
assert_eq!(
@@ -21,6 +21,7 @@ pub(crate) fn mount_public_support_routes(router: Router<AppState>) -> Router<Ap
.route("/api/public/global-models", get(proxy_request))
.route("/api/public/health/api-formats", get(proxy_request))
.route("/api/public/health/models", get(proxy_request))
.route("/api/public/health/related", get(proxy_request))
.route("/api/modules/auth-status", get(proxy_request))
.route("/api/capabilities", get(proxy_request))
.route("/api/capabilities/user-configurable", get(proxy_request))
+203 -8
View File
@@ -6,6 +6,7 @@ use axum::http::Response;
use axum::http::StatusCode;
use serde_json::json;
use crate::ai_serving::{build_core_error_body_for_client_format, LocalCoreSyncErrorKind};
use crate::constants::*;
use crate::control::GatewayControlDecision;
use crate::control::GatewayLocalAuthRejection;
@@ -191,7 +192,7 @@ pub(crate) fn build_local_balance_denied_response(
Some(remaining) => format!("余额不足(剩余: ${remaining:.2})"),
None => "余额不足".to_string(),
};
let payload = json!({
let fallback_payload = json!({
"error": {
"type": "balance_exceeded",
"message": message,
@@ -201,6 +202,13 @@ pub(crate) fn build_local_balance_denied_response(
}
}
});
let payload = build_local_error_payload(
control_decision,
None,
&message,
LocalCoreSyncErrorKind::RateLimit,
fallback_payload,
);
let body =
serde_json::to_vec(&payload).map_err(|err| GatewayError::Internal(err.to_string()))?;
let headers = BTreeMap::from([("content-type".to_string(), "application/json".to_string())]);
@@ -218,12 +226,20 @@ pub(crate) fn build_local_user_rpm_limited_response(
control_decision: Option<&GatewayControlDecision>,
rejection: &FrontdoorUserRpmRejection,
) -> Result<Response<Body>, GatewayError> {
let payload = json!({
let message = "请求过于频繁,请稍后重试";
let fallback_payload = json!({
"error": {
"type": "rate_limit_exceeded",
"message": "请求过于频繁,请稍后重试",
"message": message,
}
});
let payload = build_local_error_payload(
control_decision,
None,
message,
LocalCoreSyncErrorKind::RateLimit,
fallback_payload,
);
let body =
serde_json::to_vec(&payload).map_err(|err| GatewayError::Internal(err.to_string()))?;
let headers = BTreeMap::from([
@@ -248,12 +264,35 @@ pub(crate) fn build_local_http_error_response(
status_code: StatusCode,
message: &str,
) -> Result<Response<Body>, GatewayError> {
let payload = json!({
build_local_http_error_response_with_request_path(
trace_id,
control_decision,
None,
status_code,
message,
)
}
pub(crate) fn build_local_http_error_response_with_request_path(
trace_id: &str,
control_decision: Option<&GatewayControlDecision>,
request_path: Option<&str>,
status_code: StatusCode,
message: &str,
) -> Result<Response<Body>, GatewayError> {
let fallback_payload = json!({
"error": {
"type": "http_error",
"message": message,
}
});
let payload = build_local_error_payload(
control_decision,
request_path,
message,
local_error_kind_for_status(status_code),
fallback_payload,
);
let body =
serde_json::to_vec(&payload).map_err(|err| GatewayError::Internal(err.to_string()))?;
let headers = BTreeMap::from([("content-type".to_string(), "application/json".to_string())]);
@@ -329,19 +368,28 @@ pub(crate) fn build_local_auth_rejection_response(
pub(crate) fn build_local_overloaded_response(
trace_id: &str,
control_decision: Option<&GatewayControlDecision>,
request_path: Option<&str>,
gate: &str,
limit: usize,
) -> Result<Response<Body>, GatewayError> {
let payload = json!({
let message = "服务繁忙,请稍后重试";
let fallback_payload = json!({
"error": {
"type": "overloaded",
"message": "服务繁忙,请稍后重试",
"message": message,
"details": {
"gate": gate,
"limit": limit,
}
}
});
let payload = build_local_error_payload(
control_decision,
request_path,
message,
LocalCoreSyncErrorKind::Overloaded,
fallback_payload,
);
let body =
serde_json::to_vec(&payload).map_err(|err| GatewayError::Internal(err.to_string()))?;
let headers = BTreeMap::from([("content-type".to_string(), "application/json".to_string())]);
@@ -354,10 +402,65 @@ pub(crate) fn build_local_overloaded_response(
)
}
fn build_local_error_payload(
control_decision: Option<&GatewayControlDecision>,
request_path: Option<&str>,
message: &str,
kind: LocalCoreSyncErrorKind,
fallback_payload: serde_json::Value,
) -> serde_json::Value {
if !local_error_uses_claude_format(control_decision, request_path) {
return fallback_payload;
}
build_core_error_body_for_client_format("claude:messages", message, None, kind)
.unwrap_or(fallback_payload)
}
fn local_error_uses_claude_format(
control_decision: Option<&GatewayControlDecision>,
request_path: Option<&str>,
) -> bool {
control_decision.is_some_and(|decision| {
decision.route_family.as_deref() == Some("claude")
|| decision
.auth_endpoint_signature
.as_deref()
.is_some_and(|format| {
crate::ai_serving::normalize_api_format_alias(format)
.eq_ignore_ascii_case("claude:messages")
})
}) || request_path.is_some_and(|path| {
matches!(
path.trim_end_matches('/'),
"/v1/messages" | "/v1/messages/count_tokens"
)
})
}
fn local_error_kind_for_status(status: StatusCode) -> LocalCoreSyncErrorKind {
match status.as_u16() {
400 | 405 | 422 => LocalCoreSyncErrorKind::InvalidRequest,
401 => LocalCoreSyncErrorKind::Authentication,
403 => LocalCoreSyncErrorKind::PermissionDenied,
404 => LocalCoreSyncErrorKind::NotFound,
413 => LocalCoreSyncErrorKind::RequestTooLarge,
429 => LocalCoreSyncErrorKind::RateLimit,
503 | 529 => LocalCoreSyncErrorKind::Overloaded,
_ => LocalCoreSyncErrorKind::ServerError,
}
}
#[cfg(test)]
mod tests {
use super::build_client_response_from_parts;
use axum::body::Body;
use super::{
build_client_response_from_parts, build_local_auth_rejection_response,
build_local_http_error_response_with_request_path, build_local_overloaded_response,
build_local_user_rpm_limited_response,
};
use crate::control::{GatewayControlDecision, GatewayLocalAuthRejection};
use crate::rate_limit::FrontdoorUserRpmRejection;
use axum::body::{to_bytes, Body};
use std::collections::BTreeMap;
#[test]
@@ -386,4 +489,96 @@ mod tests {
Some("no")
);
}
fn claude_decision() -> GatewayControlDecision {
GatewayControlDecision::synthetic(
"/v1/messages",
Some("ai_public".to_string()),
Some("claude".to_string()),
Some("messages".to_string()),
Some("claude:messages".to_string()),
)
}
async fn response_json(response: http::Response<Body>) -> serde_json::Value {
let body = to_bytes(response.into_body(), usize::MAX)
.await
.expect("response body should read");
serde_json::from_slice(&body).expect("response body should be JSON")
}
#[tokio::test]
async fn claude_local_errors_use_anthropic_envelopes() {
let decision = claude_decision();
let invalid_key = build_local_auth_rejection_response(
"trace-auth",
Some(&decision),
&GatewayLocalAuthRejection::InvalidApiKey,
)
.expect("invalid-key response should build");
let invalid_key = response_json(invalid_key).await;
assert_eq!(invalid_key["type"], "error");
assert_eq!(invalid_key["error"]["type"], "authentication_error");
let rpm = build_local_user_rpm_limited_response(
"trace-rpm",
Some(&decision),
&FrontdoorUserRpmRejection {
scope: "api_key",
limit: 1,
retry_after: 60,
},
)
.expect("RPM response should build");
let rpm = response_json(rpm).await;
assert_eq!(rpm["type"], "error");
assert_eq!(rpm["error"]["type"], "rate_limit_error");
let overloaded = build_local_overloaded_response(
"trace-overload",
None,
Some("/v1/messages/count_tokens"),
"requests",
10,
)
.expect("overload response should build");
let overloaded = response_json(overloaded).await;
assert_eq!(overloaded["type"], "error");
assert_eq!(overloaded["error"]["type"], "overloaded_error");
}
#[tokio::test]
async fn claude_path_shapes_pre_control_http_errors_and_413() {
for path in ["/v1/messages", "/v1/messages/count_tokens"] {
let forbidden = build_local_http_error_response_with_request_path(
"trace-pre-control",
None,
Some(path),
http::StatusCode::FORBIDDEN,
"blocked",
)
.expect("forbidden response should build");
let forbidden = response_json(forbidden).await;
assert_eq!(forbidden["type"], "error", "path: {path}");
assert_eq!(
forbidden["error"]["type"], "permission_error",
"path: {path}"
);
let too_large = build_local_http_error_response_with_request_path(
"trace-too-large",
None,
Some(path),
http::StatusCode::PAYLOAD_TOO_LARGE,
"too large",
)
.expect("payload-too-large response should build");
let too_large = response_json(too_large).await;
assert_eq!(too_large["type"], "error", "path: {path}");
assert_eq!(
too_large["error"]["type"], "request_too_large",
"path: {path}"
);
}
}
}
+30 -26
View File
@@ -144,34 +144,38 @@ pub(crate) fn spawn_video_task_poller(state: AppState) -> Option<JoinHandle<()>>
return None;
}
Some(tokio::spawn(async move {
let mut interval = tokio::time::interval(config.interval);
interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
interval.tick().await;
let mut deferred_since = None;
loop {
Some(crate::task_runtime::spawn_singleton_worker(
state,
crate::task_runtime::TASK_KEY_VIDEO_TASK_POLLER,
move |state| async move {
let mut interval = tokio::time::interval(config.interval);
interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
interval.tick().await;
if state
.data
.should_defer_maintenance_for_database_pool_pressure(&mut deferred_since)
{
debug!(
event_name = "video_task_poller_deferred",
log_type = "event",
"gateway video task poller deferred because database pool has no idle reserve"
);
continue;
let mut deferred_since = None;
loop {
interval.tick().await;
if state
.data
.should_defer_maintenance_for_database_pool_pressure(&mut deferred_since)
{
debug!(
event_name = "video_task_poller_deferred",
log_type = "event",
"gateway video task poller deferred because database pool has no idle reserve"
);
continue;
}
if let Err(err) = poll_video_tasks_once(&state, config.batch_size).await {
warn!(
event_name = "video_task_poller_tick_failed",
log_type = "event",
error = ?err,
"gateway video task poller tick failed"
);
}
}
if let Err(err) = poll_video_tasks_once(&state, config.batch_size).await {
warn!(
event_name = "video_task_poller_tick_failed",
log_type = "event",
error = ?err,
"gateway video task poller tick failed"
);
}
}
}))
},
))
}
async fn fetch_video_task_refresh_attempt(
+5
View File
@@ -11,6 +11,7 @@ pub(crate) struct S3BackupConfig {
pub(crate) scope: BackupScope,
pub(crate) endpoint: String,
pub(crate) region: String,
pub(crate) user_agent: String,
pub(crate) bucket: String,
pub(crate) prefix: String,
pub(crate) access_key_id: String,
@@ -104,6 +105,8 @@ impl S3BackupConfig {
endpoint,
region: optional_string(entries, "backup_s3_region")?
.unwrap_or_else(|| "auto".to_string()),
user_agent: optional_string(entries, "backup_s3_user_agent")?
.unwrap_or_else(|| "rclone/v1.68.0".to_string()),
bucket,
prefix: optional_string(entries, "backup_s3_prefix")?
.unwrap_or_else(|| "aether/backups/".to_string()),
@@ -179,6 +182,7 @@ fn config_label(key: &str) -> &str {
match key {
"backup_s3_endpoint" => "Endpoint(S3 地址)",
"backup_s3_region" => "Region(S3 区域)",
"backup_s3_user_agent" => "User-Agent(请求头标识)",
"backup_s3_bucket" => "Bucket(存储桶)",
"backup_s3_prefix" => "Prefix(备份前缀)",
"backup_s3_access_key_id" => "Access Key ID(访问密钥 ID)",
@@ -384,6 +388,7 @@ mod tests {
assert_eq!(config.scope, BackupScope::Data);
assert_eq!(config.region, "auto");
assert_eq!(config.user_agent, "rclone/v1.68.0");
assert_eq!(config.prefix, "aether/backups/");
assert_eq!(config.path_style, true);
assert_eq!(config.compression, "zstd");
@@ -132,6 +132,7 @@ mod tests {
scope,
endpoint: "https://example.com".to_string(),
region: "auto".to_string(),
user_agent: "rclone/v1.68.0".to_string(),
bucket: "aether-backups".to_string(),
prefix: "prod/".to_string(),
access_key_id: "test-access-key".to_string(),
+14 -1
View File
@@ -6,7 +6,8 @@ use bytes::Bytes;
use futures_util::TryStreamExt;
use object_store::aws::AmazonS3Builder;
use object_store::path::Path;
use object_store::ObjectStore;
use object_store::{ClientOptions, ObjectStore};
use reqwest::header::HeaderValue;
use tokio::sync::RwLock;
use super::config::S3BackupConfig;
@@ -84,7 +85,19 @@ pub(crate) struct ObjectStoreS3BackupStore {
impl ObjectStoreS3BackupStore {
pub(crate) fn from_config(config: &S3BackupConfig) -> Result<Self, BackupStoreError> {
// 部分 S3 兼容网关(如中国科技云 s3.cstcloud.cn)按 User-Agent 放行请求,
// object_store 默认 UA 会被拒,因此允许自定义 User-Agent。
let client_options = if config.user_agent.trim().is_empty() {
ClientOptions::new()
} else {
ClientOptions::new().with_user_agent(
HeaderValue::from_str(config.user_agent.trim()).map_err(|error| {
BackupStoreError::new(format!("S3 备份 User-Agent 配置无效: {error}"))
})?,
)
};
let store = AmazonS3Builder::new()
.with_client_options(client_options)
.with_endpoint(config.endpoint.clone())
.with_region(config.region.clone())
.with_bucket_name(config.bucket.clone())
+1
View File
@@ -31,6 +31,7 @@ const S3_BACKUP_CONFIG_KEYS: &[&str] = &[
"backup_s3_scope",
"backup_s3_endpoint",
"backup_s3_region",
"backup_s3_user_agent",
"backup_s3_bucket",
"backup_s3_prefix",
"backup_s3_access_key_id",

Some files were not shown because too many files have changed in this diff Show More