Compare commits

..
542 Commits
Author SHA1 Message Date
elky 06f5d3c8c0 fix(gateway): complete worker registration cleanup 2026-07-31 11:32:07 +08:00
elky 082407fa51 Merge PR #697: prevent duplicate worker registrations 2026-07-31 11:11:14 +08:00
fawney19 6688ee26db Merge pull request #702 from MMEXA/fix/reconcile-auth-channel-mismatch-formats
fix(gateway): 修复批量更新 API 格式时的认证通道状态冲突
2026-07-31 10:28:37 +08:00
elky beb003b7ad feat(models): add external catalog proxy selection 2026-07-31 09:32:25 +08:00
MMEXA 6ecfe0f0a1 fix(gateway): reconcile auth mismatch formats on key update 2026-07-30 22:14:08 +08:00
ZheFox 12057db476 Merge pull request #701 from zhefox/main
Persist OpenAI Responses continuation history across instances
2026-07-30 21:08:34 +08:00
ZheFox ff47d8d48a fix(gateway): route response history through ai seam 2026-07-30 20:34:41 +08:00
ZheFox ef5f36cc2b fix(ai): satisfy response history clippy checks 2026-07-30 20:13:36 +08:00
ZheFox 84022c4d48 Merge upstream/main into main 2026-07-30 19:40:39 +08:00
ZheFox 118f441029 feat(gateway): persist OpenAI Responses continuation history 2026-07-30 19:26:52 +08:00
elky 20399b004d Merge PR #700: fix admin pool batch update body buffering
Preserve main's failover and usage metadata fixes, restore default tunnel regression coverage, and satisfy current Clippy.
2026-07-30 17:56:37 +08:00
elky 050eb77508 fix(ai): harden responses replay and failure diagnostics 2026-07-30 17:19:54 +08:00
elky 1ab4f079c9 fix(gateway): restore failover and usage diagnostics 2026-07-30 09:12:11 +08:00
MMEXA 6c733f7590 fix(usage): preserve request diagnostics in event seeds 2026-07-30 06:44:59 +08:00
MMEXA d7d8db45ba test(gateway): align tunnel error fixture with failover policy 2026-07-30 06:44:59 +08:00
MMEXA 8cf9af79da fix(ci): remove redundant usage policy update 2026-07-30 05:45:38 +08:00
MMEXA e55793c765 fix(ci): satisfy gateway clippy on upstream baseline 2026-07-30 05:14:34 +08:00
MMEXA d8902ea612 fix(gateway): buffer admin pool batch update bodies 2026-07-30 05:14:34 +08:00
elky a04673a90d feat(gateway): harden failover and payload handling
Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
2026-07-30 01:03:27 +08:00
ZheFox a97acc07fc Merge pull request #698 from zhefox/main
fix(ci): stabilize cross-platform workflow checks
2026-07-29 22:16:17 +08:00
zhefox f8000012f7 fix(ci): stabilize cross-platform workflow checks 2026-07-29 21:55:43 +08:00
worker-2 6080f8cc88 fix(gateway): stabilize worker task records
Key worker boot records by task so process restarts update the existing
row instead of registering another row for each gateway instance.

Closes #693
Confidence: high
Scope-risk: narrow
2026-07-29 17:27:51 +08:00
ZheFox 37df5b93b1 Merge pull request #696 from zhefox/main
Fix client metadata handling across formats
2026-07-28 18:12:59 +08:00
ZheFox e53abdaec2 Merge branch 'fawney19:main' into main 2026-07-28 18:12:28 +08:00
ZheFox 2db32ea97e Merge branch 'main' of https://github.com/zhefox/Aether 2026-07-28 17:40:46 +08:00
ZheFox 581897ee74 fix(formats): ignore responses client metadata across targets 2026-07-28 17:40:41 +08:00
ZheFox 9a88f966d8 Merge pull request #695 from zhefox/main
Enhance provider capabilities and clean up OAuth keys
2026-07-28 13:58:33 +08:00
ZheFox 9d9316e434 Merge branch 'fawney19:main' into main 2026-07-28 13:56:34 +08:00
ZheFox 1b697b1111 feat(providers): support FedRAMP Codex agent identity registration 2026-07-28 13:29:30 +08:00
ZheFox 3043982486 fix(providers): derive Codex primary quota label from window 2026-07-28 12:57:45 +08:00
ZheFox 0bf92ffffc feat(providers): advertise responses API agent capability 2026-07-28 12:02:03 +08:00
ZheFox f0f87b56a3 feat(providers): add credential-fenced OAuth key cleanup 2026-07-28 11:32:11 +08:00
elky 4148ab1931 fix(routing): harden routed pool scheduling 2026-07-27 22:06:28 +08:00
elky 550cc36760 feat(providers): expand OAuth account management
Add Claude Code manual and cookie authorization, including redacted batch tasks. Harden OAuth imports, duplicate replacement, provider dialogs, and related account-management tests.
2026-07-27 15:53:28 +08:00
elky 531cf11025 feat(gateway): harden provider request execution
Preserve exact request payloads and model client surface and API operation explicitly.

Add Anthropic compatibility profiles, bounded stream commitment, and scoped OAuth retry behavior across provider transports.
2026-07-27 09:36:31 +08:00
elky 79b70f7b5c fix(frontend): align sidebar collapse button 2026-07-26 15:07:51 +08:00
elky 10d369f59c feat(providers): add provider transfer limits 2026-07-26 15:06:56 +08:00
elky 2ef7ac79bc feat(frontend): add collapsible navigation sidebar
Persist the desktop sidebar state, provide accessible compact navigation tooltips, and cover the collapsed navigation markup with a focused component test.
2026-07-25 21:28:51 +08:00
elky 778cfb1a5c feat(data): complete portable SQL backend parity
Align MySQL and SQLite schemas, migrations, usage, stats, export, and backfill behavior with the shared data contracts. Extend gateway startup and maintenance support across all SQL drivers.
2026-07-25 21:28:21 +08:00
elky 764e9fd131 feat(frontend): improve provider detail drawer and pool actions 2026-07-25 11:16:59 +08:00
elky 387134ca87 fix(models): correct fast pricing and online sync 2026-07-24 01:45:38 +08:00
elky a0767d957c fix(frontend): synchronize pool account state 2026-07-23 16:20:18 +08:00
ZheFox b94ef91d07 Merge pull request #692 from zhefox/main
Sync global model prices and track online pricing sources
2026-07-23 16:02:17 +08:00
ZheFox e7910751d9 Merge branch 'fawney19:main' into main 2026-07-23 15:19:48 +08:00
ZheFox 1d2655432d feat(models): track online pricing sources and unsupported fields 2026-07-23 15:18:08 +08:00
ZheFox 323273ff30 feat(models): sync global model prices from online catalog 2026-07-23 13:29:28 +08:00
ZheFox fb2009c65b Merge pull request #691 from zhefox/main
fix(formats): ignore Responses client transport metadata
2026-07-23 12:18:40 +08:00
ZheFox e186cc6848 Merge branch 'main' of https://github.com/zhefox/Aether 2026-07-23 12:17:46 +08:00
ZheFox 615ac99ad7 fix(formats): ignore Responses client transport metadata 2026-07-23 12:16:44 +08:00
ZheFox ec36cfbf75 Merge pull request #690 from zhefox/main
fix(provider): classify deleted Codex agent runtime as invalid
2026-07-23 11:22:30 +08:00
ZheFox 7bf228a33c fix(provider): classify deleted Codex agent runtime as invalid 2026-07-23 11:21:55 +08:00
elky 3606290ac8 fix(provider): harden Agent Identity OAuth lifecycle 2026-07-23 09:33:00 +08:00
elky e49024d33b fix(frontend): shorten Agent Identity tab label 2026-07-22 20:26:49 +08:00
elky fdbc2607ec feat(provider): add dedicated Codex Agent Identity flow 2026-07-22 20:19:29 +08:00
elky c7cc8fd7db test(provider): simplify agent identity assertions 2026-07-22 14:19:58 +08:00
elky 07efcb5146 fix(data): repair legacy active flag synchronization 2026-07-22 14:19:34 +08:00
elky 856605defa fix(model-directives): harden suffix configuration 2026-07-22 14:19:09 +08:00
elky 713010fa0a fix(gateway): restore auth role refresh and Rust checks
Refresh the resolved user role without bypassing owner group and key policies. Resolve Rust 1.95 Clippy failures and make the pending persistence bound test scheduler-independent.
2026-07-22 11:25:24 +08:00
ZheFox cd2fbeeead Merge pull request #689 from AAEE86/feat/agent-identity-support
feat(codex): enroll agent identity from session token
2026-07-22 10:32:06 +08:00
AAEE86 a4350a482a feat(codex): enroll agent identity from session token 2026-07-22 10:21:11 +08:00
ZheFox c825375367 Merge pull request #688 from AAEE86/feat/agent-identity-support
feat(codex): support agent identity accounts
2026-07-22 09:20:11 +08:00
elky fc92c4f431 perf(gateway): scale request hot paths for 20k streams
Shard and singleflight hot-path caches, batch and prioritize candidate and usage lifecycle persistence, and extend database and pressure-test instrumentation for 20k concurrent streams.
2026-07-22 02:11:08 +08:00
AAEE86 b61c590bdb feat(codex): support agent identity accounts 2026-07-21 20:58:49 +08:00
ZheFox 7756c0913f Merge pull request #685 from zhefox/main
fix(gateway): apply group policy to admin-owned keys
2026-07-20 15:52:06 +08:00
ZheFox c34ec7c1ee fix(gateway): apply group policy to admin-owned keys 2026-07-20 15:51:01 +08:00
elky f8778c4a23 feat(gateway): configure cyber policy failover 2026-07-19 23:27:19 +08:00
elky e0dbb233f7 fix(frontend): avoid misleading cache TTL fallback label 2026-07-19 22:21:46 +08:00
elky 9725f9abae fix(frontend): clarify processing tier pricing 2026-07-19 22:00:12 +08:00
elky 5d575f1590 test(stats): treat bulk API key snapshots as authoritative 2026-07-19 19:07:04 +08:00
elky d562c594c3 fix(frontend): preserve compact scope and detail badge 2026-07-19 16:42:32 +08:00
MMEXA ce226a3010 Merge 0c3f51bcec into 644ae9c1bf 2026-07-19 16:12:37 +08:00
elky 644ae9c1bf feat(pool): add table-driven account batch actions 2026-07-19 16:09:20 +08:00
elky 95053f9502 Merge PR #672: 支持账号批量配置与可用模型管理 2026-07-18 22:06:32 +08:00
elky 8fbda84acb fix(data): preserve API key history end to end 2026-07-18 21:58:21 +08:00
elky 03b7d573e0 Merge PR #683: decouple API key historical identity 2026-07-18 21:20:35 +08:00
MMEXA 0c3f51bcec merge(main): 解决 usage 模型展示契约冲突 2026-07-18 19:26:14 +08:00
elky e3d97b573b fix(usage): align fast-tier pricing and model metadata 2026-07-18 16:57:04 +08:00
MMEXA ac3796af84 fix(gateway): 恢复响应边界并统一格式入口 2026-07-18 06:45:37 +08:00
MMEXA f9c343eb07 fix(gateway): 适配 Rust 1.95 整除检查 2026-07-18 05:55:06 +08:00
MMEXA e31df5989a merge(main): 解决 usage 展示与生命周期同步冲突 2026-07-18 05:38:38 +08:00
MMEXA 98fbf029fc fix(data): 解耦 API Key 历史统计身份 2026-07-18 04:42:45 +08:00
MMEXA 4d9a648202 test(gateway): 统一流错误测试的格式层入口 2026-07-18 03:37:31 +08:00
MMEXA 405ca3e66a fix(ci): 恢复非流式错误体边界并适配新版 Clippy 2026-07-18 03:26:53 +08:00
MMEXA 0355c28683 fix(data): 解耦候选记录的 API Key 历史身份 2026-07-18 02:40:34 +08:00
fawney19 6c33b8d8fb Merge pull request #682 from MMEXA/codex/codex-prompt-cache-identity-20260717
fix(codex): 统一通用缓存键与原生会话身份
2026-07-18 00:33:20 +08:00
elky a6c6f14b09 style(frontend): align pool cycle stats values 2026-07-18 00:13:35 +08:00
elky e558f55cd9 style(frontend): refine badges and cycle stats 2026-07-18 00:05:22 +08:00
elky 88a057b8d9 fix(usage): force fast badge background transparent 2026-07-17 22:56:54 +08:00
elky ed27d404ac style(usage): make fast badge background transparent 2026-07-17 22:52:47 +08:00
elky 5dda34c66e style(usage): give fast tier an amber accent 2026-07-17 22:36:55 +08:00
elky 373ebf26d6 fix(pricing): default zero tier ratios to one 2026-07-17 21:33:21 +08:00
elky f65ed2795c fix(codex): support dynamic quota windows 2026-07-17 20:18:04 +08:00
elky 664c063a06 feat(usage): enrich audit metadata and detail views 2026-07-17 19:20:16 +08:00
MMEXA 75795c6fbc test(codex): 对齐 Compact 确定性缓存身份 2026-07-17 08:49:35 +08:00
MMEXA 5b332da7d7 fix(codex): 补齐缓存身份终态请求头 2026-07-17 08:11:16 +08:00
MMEXA d9796d502b fix(codex): 统一通用缓存键与原生会话身份 2026-07-17 06:13:09 +08:00
MMEXA 3b0d87b0fd Merge remote-tracking branch 'origin/main' into codex/pool-key-bulk-management-20260714 2026-07-17 00:17:02 +08:00
MMEXA 3c348dff3a Merge remote-tracking branch 'origin/main' into codex/usage-pending-reasoning-reset-expiry-20260712
# Conflicts:
#	frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts
2026-07-17 00:16:58 +08:00
MMEXA ec1783a35c Merge remote-tracking branch 'origin/main' into codex/pool-key-bulk-management-20260714
# Conflicts:
#	apps/aether-gateway/src/handlers/admin/request/provider/tasks.rs
#	frontend/src/api/endpoints/pool.ts
2026-07-16 23:43:04 +08:00
MMEXA 427030c5de Merge remote-tracking branch 'origin/main' into codex/usage-pending-reasoning-reset-expiry-20260712
# Conflicts:
#	crates/aether-ai-formats/src/formats/openai/responses/mod.rs
#	crates/aether-usage/runtime/src/runtime.rs
#	frontend/src/features/usage/components/UsageRecordsTable.vue
#	frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts
2026-07-16 23:41:58 +08:00
elky 0be380243b feat(pricing): support processing tier multipliers 2026-07-16 23:30:42 +08:00
fawney19 312583f055 Merge pull request #680 from Kayphoon/codex/s3-backup-user-agent
feat(admin): configure S3 backup User-Agent
2026-07-16 23:30:30 +08:00
fawney19 33f49ea9b0 Merge pull request #678 from AAEE86/fix
fix: map Developer role to "system" in OpenAI Chat Completions output
2026-07-16 23:29:55 +08:00
fawney19 470cef17cf Merge pull request #676 from MMEXA/codex/sync-capture-envelope-finalize-20260716
修复同步 finalize 的 Responses 流聚合与转换
2026-07-16 23:29:38 +08:00
ZheFox 3f5f65eb9a Merge pull request #681 from zhefox/main
Codex 重置功能和显示缓存修复以及批量key的导入和管理功能
2026-07-16 19:48:33 +08:00
ZheFox 6664c2dbb8 feat(pool): 支持批量导入 Key 和选择性更新设置 2026-07-16 19:31:07 +08:00
ZheFox 0099167a6d fix(codex): 避免重置机会缺失触发配额刷新 2026-07-16 19:13:09 +08:00
ZheFox f009fb73c3 缓存问题修复 2026-07-16 18:55:28 +08:00
ZheFox 715a5ed626 修复重置次数缓存问题 2026-07-16 18:09:10 +08:00
ZheFox 5cf38d1b35 Codex 重置功能和显示修复 2026-07-16 17:23:41 +08:00
Kayphoon 6b707f29a2 feat(admin): configure S3 backup User-Agent 2026-07-16 08:56:56 +00:00
elky 9ea84f9748 fix(frontend): show service tier transitions 2026-07-16 16:38:30 +08:00
elky c32d043afb fix(frontend): preserve fetched model preset pricing 2026-07-16 16:38:30 +08:00
elky 8fe4d24408 fix(usage): canonicalize cached token totals 2026-07-16 16:38:30 +08:00
elky e369e4aab1 fix(formats): preserve chat-backed Responses metadata 2026-07-16 16:38:30 +08:00
ZheFox 7dc919e8e3 Merge pull request #679 from zhefox/main
fix(frontend): 优化移动端弹窗并完善提供商配额刷新
2026-07-16 15:48:49 +08:00
ZheFox 1333efdad5 fix(frontend): 优化移动端弹窗并完善提供商配额刷新 2026-07-16 15:21:11 +08:00
AAEE86 cd8de1aa13 fix: map Developer role to "system" in OpenAI Chat Completions output 2026-07-16 14:29:08 +08:00
elky d6215d9dec ci(tunnel): reduce artifact retention 2026-07-16 13:12:30 +08:00
elky 9a47267545 fix(usage): bound terminal event persistence
Add end-to-end terminal admission, bounded database fallback, and observable overload handling. Preserve first-byte lifecycle state across asynchronous runtime and frontend updates.
2026-07-16 13:12:30 +08:00
MMEXA 7851503fbc fix(finalize): 严格聚合并投影同步 Responses 流 2026-07-16 12:50:00 +08:00
ZheFox c6d373e6aa Merge pull request #677 from zhefox/main
fix(codex): 移除 Responses Lite 请求中的 context_management
2026-07-16 12:28:34 +08:00
ZheFox 71fcb9c168 fix(codex): 服务端压缩使用标准 Responses 合约 2026-07-16 12:02:33 +08:00
ZheFox 3976652942 fix(codex): 移除 Responses Lite 请求中的 context_management 2026-07-16 11:25:11 +08:00
MMEXA 7b56546e21 fix(finalize): 聚合同步流捕获包装 2026-07-16 10:24:32 +08:00
fawney19 85854e4476 Merge pull request #675 from fawney19/fix/pr-669-tail
feat(codex): complete PR #669 protocol follow-up
2026-07-16 08:54:29 +08:00
elky b50242ab9f fix(test): handle absent empty testkit bin directory 2026-07-16 01:29:51 +08:00
MMEXA 20b27a13b2 feat(codex): 按操作语义路由 Responses V2 压缩
(cherry picked from commit 2fc604e047)
2026-07-16 00:34:42 +08:00
MMEXA 598b2fb374 fix(auth): 授权 Responses Compact 伴随端点
(cherry picked from commit e8afa03e45)
2026-07-16 00:32:14 +08:00
MMEXA ff7988430d fix(openai): encode tool errors in Responses output
(cherry picked from commit f127b67e73)
2026-07-16 00:31:23 +08:00
MMEXA 25da99fac2 fix(codex): preserve reset consume request body
(cherry picked from commit fc2dfb82d2)
2026-07-16 00:27:28 +08:00
elky 8616fe6ee2 refactor(workspace): enforce layered crate boundaries 2026-07-15 23:47:19 +08:00
MMEXA 9b8724453b test(pool): 使用正式 Gemini API 格式 2026-07-14 08:39:41 +08:00
MMEXA 01e104d86a fix(pool): 对齐账号批量配置语义 2026-07-14 08:07:28 +08:00
MMEXA 0acd1de29c fix(gateway): 保持密钥更新模块显式所有权 2026-07-14 05:19:04 +08:00
MMEXA a25fab371a feat(pool): add bulk key configuration management 2026-07-14 04:57:05 +08:00
MMEXA 93e2f95c47 fix(frontend): 按端点能力约束会话压缩映射 2026-07-14 02:05:10 +08:00
MMEXA f10d631a9c feat(frontend): 澄清模型映射适用范围 2026-07-14 00:30:39 +08:00
MMEXA cfc4894dab fix(usage): 保留最新进行态生命周期事件 2026-07-14 00:30:24 +08:00
MMEXA 3f86fdd6bc feat(usage): 展示压缩操作与进行态请求语义 2026-07-13 22:03:44 +08:00
MMEXA b09d1f1c33 fix(usage): expose pending reasoning and exact reset expiry 2026-07-13 22:03:44 +08:00
MMEXA 2fc604e047 feat(codex): 按操作语义路由 Responses V2 压缩 2026-07-13 22:03:34 +08:00
MMEXA e8afa03e45 fix(auth): 授权 Responses Compact 伴随端点 2026-07-13 06:07:54 +08:00
MMEXA fc2dfb82d2 fix(codex): preserve reset consume request body 2026-07-12 23:04:33 +08:00
elky a728c090a9 fix(gateway): scope concurrency helper to tests 2026-07-12 22:36:48 +08:00
elky e58621a735 Merge PR #669: align GPT-5.6 and Codex request protocols 2026-07-12 21:50:20 +08:00
MMEXA b1be370b2e fix(gateway): scope concurrency test helper to tests 2026-07-12 21:08:30 +08:00
MMEXA cf0d957ac7 Merge f127b67e73 into 7f61bb43c7 2026-07-12 20:34:39 +08:00
MMEXA f127b67e73 fix(openai): encode tool errors in Responses output 2026-07-12 20:34:31 +08:00
elky 7f61bb43c7 feat(security): harden gateway request and runtime controls 2026-07-12 14:10:54 +08:00
MMEXA 25c49dd804 fix(data): keep terminal usage state monotonic 2026-07-12 05:45:37 +08:00
MMEXA 72222d935c test(gateway): use valid tunnel relay envelopes 2026-07-12 05:45:32 +08:00
MMEXA 063d517306 test(gateway): compare timeout response numerically 2026-07-12 04:43:13 +08:00
MMEXA 02495ce28e fix(admin): preserve inactive endpoint key counts 2026-07-12 04:19:06 +08:00
MMEXA 63936aa110 fix(gateway): route format rules through serving facade 2026-07-12 03:56:53 +08:00
MMEXA 8d4d42a887 fix(auth): resolve group policy before key intersection 2026-07-12 03:35:42 +08:00
MMEXA 2316df5c9a feat(codex): align Search and execution protocol 2026-07-12 03:04:15 +08:00
MMEXA 59d37ae1dd fix(frontend): import structured models.dev pricing 2026-07-11 18:12:55 +08:00
MMEXA 3014fd50c6 fix(billing): preserve effective cache and tier facts 2026-07-11 18:12:55 +08:00
MMEXA 14c4e3a04e fix(codex): enforce provider request identity 2026-07-11 18:09:14 +08:00
MMEXA 8f1070a451 feat(frontend): expose processing tier pricing 2026-07-11 12:27:09 +08:00
MMEXA 0b30cc6b0f feat(openai): unify tier authorization and settlement 2026-07-11 12:27:05 +08:00
MMEXA b2f596b8f0 fix(gateway): route Codex header through serving facade 2026-07-11 09:43:12 +08:00
MMEXA 01a96fed74 fix(codex): simplify summary normalization 2026-07-11 09:20:07 +08:00
MMEXA 46a903aada fix(codex): align current reasoning request semantics 2026-07-11 09:15:08 +08:00
MMEXA dfa121dd5b feat(openai): align GPT-5.6 and Codex request contracts 2026-07-11 07:40:12 +08:00
elky bc1da3bf3f feat(security): harden client IP and admin controls 2026-07-10 15:13:12 +08:00
elky 6e0dc3b59e feat(frontend): refine global model pricing dialog 2026-07-10 15:13:12 +08:00
elky 4bf5d4c044 Fix cache token accounting and tiered pricing 2026-07-10 15:13:12 +08:00
fawney19 736fc76345 Merge pull request #668 from MMEXA/codex/antigravity-empty-output-retry-20260709
修复 Gemini 空输出按候选重试处理
2026-07-10 09:16:56 +08:00
MMEXA b6b2ca38f4 触发 CI 重跑 2026-07-10 00:41:22 +08:00
MMEXA f07eb25cfc 修复 usage 详情 body 引用解包 2026-07-10 00:23:31 +08:00
MMEXA d2ea437c1c 修复 Gemini 空输出按候选重试处理 2026-07-09 23:10:30 +08:00
fawney19 7bc7d0f8d8 Merge pull request #666 from xixiknow/main
Fix provider key response time counter overflow
2026-07-09 18:04:11 +08:00
fawney19 14cf639aba Merge pull request #667 from MMEXA/codex/antigravity-v1internal-query-20260709
修复 Antigravity v1internal 查询参数透传
2026-07-09 17:50:38 +08:00
yangrs 55cdab592c Remove redundant response time conversion 2026-07-09 16:26:09 +08:00
MMEXA ee0ec18283 修复 Antigravity v1internal 查询参数透传 2026-07-09 16:03:29 +08:00
Start f31c9e03e2 Merge branch 'fawney19:main' into main 2026-07-09 15:11:44 +08:00
fawney19 e50db10439 Merge pull request #663 from MMEXA/codex/gemini-interactions-antigravity-20260705
完善 Gemini Interactions 与 Antigravity 全链路兼容
2026-07-09 14:55:51 +08:00
yangrs 192dc6c20d Fix provider key response time overflow 2026-07-09 14:40:48 +08:00
elky 5e1d14f19b Fix timeline duration display from latency 2026-07-09 11:46:45 +08:00
MMEXA b8b89d21b7 fix: 同步提交本地 sync 错误上报 2026-07-08 23:49:18 +08:00
MMEXA 5eddf4f9ee 细化 Antigravity 测试模型项目元数据补全 2026-07-08 22:45:30 +08:00
MMEXA c7186e1720 完善 Antigravity 配额展示与 CI 断言 2026-07-08 22:34:29 +08:00
MMEXA 4866509938 移除 Antigravity 未知重置时间噪音 2026-07-08 22:34:29 +08:00
MMEXA 2122660a5c 对齐原生 Antigravity 控制面与显示模型 2026-07-08 22:34:29 +08:00
MMEXA 5c68ab896a 恢复历史 backfill 兼容 live 账本 2026-07-08 22:34:29 +08:00
MMEXA f9c8ec41f4 完善 Antigravity 与 Gemini 跨格式兼容 2026-07-08 22:34:29 +08:00
MMEXA b1ed6b24b0 触发 CI 复跑 2026-07-08 22:34:29 +08:00
MMEXA c17c78ad4b 修正 Antigravity Gemini 3.5 Flash 档位展示 2026-07-08 22:34:29 +08:00
MMEXA 9ec48ab6b9 优化 Antigravity 配额展示顺序 2026-07-08 22:34:29 +08:00
MMEXA accd250226 修正 Antigravity 配额模型标签 2026-07-08 22:34:29 +08:00
MMEXA 80a6579766 支持 Gemini Interactions 与 Antigravity 配额精细化 2026-07-08 22:34:29 +08:00
fawney19 1ca83ca3fb Merge pull request #664 from MMEXA/codex/wallet-auth-cache-delay-20260706
修复钱包余额变更后的鉴权缓存延迟
2026-07-07 01:59:44 +08:00
fawney19 a931da0764 Merge pull request #662 from MMEXA/codex/reset-credit-20260704
增加 Codex 重置次数功能
2026-07-07 01:58:44 +08:00
fawney19 a61374c595 Merge pull request #661 from MMEXA/codex/frontend-debug-20260704
修复前端调试与基础交互问题
2026-07-07 01:57:41 +08:00
MMEXA c3136126e5 修复钱包余额变更后的鉴权缓存延迟 2026-07-06 06:15:31 +08:00
MMEXA b23d299533 重跑 Codex 重置次数 CI 2026-07-05 01:37:31 +08:00
MMEXA b03aae18c3 修复 Codex 重置次数 CI 检查 2026-07-04 15:47:01 +08:00
MMEXA 99b6fe468f 简化 Codex 重置机会展示标签 2026-07-04 15:16:38 +08:00
MMEXA ef77ec04ca 增加 Codex 重置次数功能 2026-07-04 06:10:53 +08:00
MMEXA 242081433e 修复前端调试与基础交互问题 2026-07-04 05:24:40 +08:00
ZheFox b86d4e1f0c Merge pull request #660 from zhefox/main
refactor(frontend): unify mobile menu background styles
2026-07-03 13:17:59 +08:00
ZheFox a151f37d63 refactor(frontend): unify mobile menu background styles 2026-07-03 13:17:18 +08:00
ZheFox 42f7907740 Merge pull request #659 from zhefox/main
refactor(frontend): improve mobile overflow handling
2026-07-03 12:54:01 +08:00
ZheFox e72e25c59c refactor(frontend): improve mobile overflow handling 2026-07-03 12:53:22 +08:00
ZheFox 1b0440481b Merge pull request #658 from zhefox/main
修复管理端额度显示、节点表格显示与移动端滚动问题
2026-07-03 12:49:15 +08:00
ZheFox 26d85681f0 refactor(frontend): improve mobile overflow and proxy node table 2026-07-03 12:25:53 +08:00
ZheFox 1dcee77055 refactor(frontend): improve dialog and mobile overflow handling 2026-07-03 11:26:40 +08:00
elky 1ac16005f9 Stabilize usage worker autoscale tests 2026-07-02 17:28:25 +08:00
elky 2f1cdb6a0b Record exhausted usage failures synchronously 2026-07-02 16:08:04 +08:00
elky ac93851b2a Stabilize Gateway h2c transport test 2026-07-02 14:02:26 +08:00
elky 400b3125a4 Preserve terminal request candidate state 2026-07-02 01:40:57 +08:00
elky 2e5ff32e1a perf(frontend): 收敛导航预取并去重首屏请求
- 导航预取仅保留 pointerdown 触发,移除 mouseenter/focus,避免鼠标划过误触发
- 后台预取只做组件懒加载,不再预取各页业务数据,减少首屏资源争抢
- 版本状态检查增加 sessionStorage 缓存(正常 20 分钟 / 错误 5 分钟 TTL)
- fetchModules、必读公告拉取增加请求去重,避免并发重复请求
- 更新检查改用可清理的定时器,组件卸载时清理
- UsageRecordsTable 搜索防抖改为自定义实现,卸载时取消挂起 emit 并补充测试
2026-07-01 20:42:52 +08:00
elky a0f7074e59 chore: disable Redis persistence by default, document triage and policy 2026-07-01 14:15:16 +08:00
elky 7c32be46ca Mark sync usage active earlier 2026-07-01 02:21:20 +08:00
Entropy.Xu 6ed2f9bd0a fix: apply actual billing cost to wallet settlement 2026-07-01 01:12:40 +08:00
elky 778b106023 test: stabilize gateway nextest timing 2026-06-30 18:42:57 +08:00
elky f179ee72f9 chore: update gateway pressure observability 2026-06-30 17:01:39 +08:00
elky 974def5fef refactor(frontend): extract provider key identity block 2026-06-30 17:01:39 +08:00
elky e5351b7d9d refactor(frontend): extract provider key actions 2026-06-30 17:01:39 +08:00
elky ed83184d55 refactor(frontend): extract provider quota display components 2026-06-30 17:01:39 +08:00
elky 15b6606c82 refactor(frontend): extract pool key display panels 2026-06-30 17:01:39 +08:00
elky d7411a3104 refactor(frontend): extract pool header and theme toggle 2026-06-30 17:01:39 +08:00
elky 9f138d09e6 refactor(frontend): modularize i18n architecture 2026-06-30 17:01:39 +08:00
ZheFox bf29129a4b Merge pull request #654 from zhefox/main
Cancel upstream streams on client disconnect and void cancelled usage billing
2026-06-29 01:14:11 +08:00
zhefox f6293b6812 fix(usage): void cancelled usage and cancel dropped streams 2026-06-29 00:30:41 +08:00
elky 7e9424008f Add usage queue worker autoscaling 2026-06-26 14:02:57 +08:00
elky 6c5e70ccb1 fix monitoring error totals and counter health 2026-06-26 10:48:45 +08:00
elky 063834e95b Split admin operations dashboard route 2026-06-26 01:48:37 +08:00
elky c76d6b6396 Add admin operations dashboard and usage state fixes 2026-06-26 01:32:45 +08:00
elky 6f00e9fc67 Improve gateway transport and usage runtime 2026-06-25 22:36:27 +08:00
elky d336d1a7fa Improve gateway scheduling and runtime admission 2026-06-24 01:53:45 +08:00
ZheFox cf0af8fa1e Merge pull request #652 from zhefox/main
fix(usage): preserve token counts in body redaction
2026-06-23 14:40:59 +08:00
zhefox fd220b6c42 fix(usage): preserve token counts in body redaction 2026-06-23 14:38:39 +08:00
zhefox 3472bb75e7 ci: combine gateway clippy and nextest jobs 2026-06-23 14:06:43 +08:00
zhefox ba65c96c74 Merge branch 'main' of https://github.com/zhefox/Aether 2026-06-23 13:36:00 +08:00
zhefox c54b214657 ci: shard gateway tests and disable debug info in rust ci 2026-06-23 13:35:56 +08:00
ZheFox 4fcc17114f Merge pull request #651 from zhefox/main
fix(ai-formats): accept Claude context_management in responses conversion
2026-06-23 10:43:26 +08:00
zhefox deb5f55786 fix(ai-formats): clean up cross-format safety rules for Gemini requests 2026-06-23 10:30:13 +08:00
zhefox 1836c2b652 fix(ai-formats): accept Claude context_management in responses conversion 2026-06-23 10:03:55 +08:00
elky 5b7805181b perf: queue request candidate persistence 2026-06-22 02:49:17 +08:00
elky f75894acbb perf: reduce gateway db pressure under load 2026-06-22 00:08:48 +08:00
elky 541cc197c4 fix: preserve in-memory user export fallback 2026-06-22 00:08:48 +08:00
fawney19 363d1aba9a Merge pull request #615 from AAEE86/main
feat: 健康监控仪表盘与关联下钻优化,完善使用记录展示
2026-06-21 12:48:13 +08:00
fawney19 eb2cf662b7 Merge pull request #650 from stabey/pr/claude-system-responses-20260620
fix(ai-formats): preserve Claude in-message system guidance in Responses
2026-06-21 12:47:26 +08:00
elky 900f8a7163 fix(pool): allow zero cooldown settings 2026-06-21 12:20:08 +08:00
elky 61bdd304b7 Handle inactive PAT owner as invalid OAuth token 2026-06-21 11:39:20 +08:00
elky 279735ae7f Auto-size SQL pool defaults 2026-06-21 11:15:40 +08:00
elky cc2830f6ec Merge branch 'review/pr-639' 2026-06-21 10:48:49 +08:00
elky 8dbd730568 fix: respect imported oauth authorization headers 2026-06-21 02:27:06 +08:00
stabey bb6aa03485 fix(ai-formats): strip Claude billing headers from preserved guidance 2026-06-21 00:22:57 +08:00
stabey 6a22488698 fix(ai-formats): preserve Claude in-message system guidance in responses 2026-06-21 00:07:57 +08:00
elky f1c30439ff fix: preserve provider auth metadata 2026-06-20 22:11:42 +08:00
fawney19 1123095bb7 Merge pull request #624 from MMEXA/codex/fix-antigravity-oauth-quota
修复 Antigravity OAuth 导入后配额复检缺 project
2026-06-19 23:11:38 +08:00
MMEXA 938f11981d fix(ai-serving): route Antigravity auth enum through facade 2026-06-19 22:25:52 +08:00
MMEXA 6c4e730e60 修复 Antigravity OAuth 配额复检缺 project 2026-06-19 22:21:22 +08:00
elky 16584067d7 Add route-backed routing profile views 2026-06-18 02:06:34 +08:00
fawney19 6de0fe75a4 Merge pull request #641 from Kayphoon/codex/usage-cleanup-break-condition
fix(usage): align cleanup loop break conditions with candidate row count
2026-06-17 11:04:55 +08:00
fawney19 34f0913ed0 Merge pull request #645 from zhefox/main
修复 OpenAI Chat/Responses/Messages 转换兼容性并透传 Codex cyber_policy 错误
2026-06-17 11:03:29 +08:00
zhefox 5b305c64e1 fix(ai-formats): omit request tool call ids in OpenAI Responses input 2026-06-17 09:15:38 +08:00
zhefox 8ad97761e8 fix(ai-formats): preserve OpenAI Responses tool call item ids 2026-06-17 08:29:25 +08:00
zhefox 0f92ef664d fix(ai-formats): support OpenAI Responses custom tool/raw passthrough 2026-06-17 04:24:56 +08:00
zhefox 16a4fd3687 Merge branch 'main' of https://github.com/zhefox/Aether 2026-06-17 04:07:53 +08:00
zhefox 3a3fcbe46a fix(ai-formats): preserve OpenAI tool call item ids 2026-06-17 04:05:09 +08:00
zhefox 18d8ea2052 fix(ai-formats): preserve OpenAI tool call item ids 2026-06-17 04:03:40 +08:00
zhefox 6ab08f4014 fix(ai-formats): preserve Claude raw blocks, reasoning tokens, and test stack safety 2026-06-17 03:45:23 +08:00
zhefox 628a3a0d8d fix(ai-formats): support cyber policy failover and custom tool/audio passthrough 2026-06-17 02:33:14 +08:00
elky f52628e00b Handle OpenAI Responses keepalive stream events 2026-06-16 22:52:06 +08:00
zhefox f9d97ececb fix(ai-formats): ignore OpenAI Responses metadata events 2026-06-16 22:34:38 +08:00
fawney19 803e555022 Merge pull request #635 from zhefox/main
fix(gateway): 支持 OpenAI 图片编辑端点请求
2026-06-16 22:31:52 +08:00
zhefox b1bd727978 将 JSON 提示注入为 developer 输入 2026-06-16 21:40:15 +08:00
zhefox c2748dc868 忽略 OpenAI Responses keepalive 事件 2026-06-16 20:00:46 +08:00
ZheFox 302620cb94 Merge branch 'fawney19:main' into main 2026-06-16 12:17:28 +08:00
AAEE86 c255f29e98 Merge remote-tracking branch 'upstream/main' 2026-06-16 10:53:47 +08:00
Kayphoon 6d1b818414 fix(usage): align cleanup loop break conditions with candidate row count
The cleanup loop break condition used rows_affected() from the UPDATE
statement, but for rows that only had blob/audit refs (no inline
compressed body data), the UPDATE reported 0 affected rows. This caused
the loop to exit after the first batch, skipping the majority of
candidates.

Change the break condition in all 4 cleanup functions from:
  if cleaned == 0 || cleaned < batch_size
to:
  if rows.len() < batch_size

This ensures the loop continues as long as SELECT returns a full batch,
regardless of how many rows the UPDATE actually modified.

Affected functions:
- cleanup_usage_raw_body_fields
- cleanup_usage_compressed_body_fields
- cleanup_usage_header_fields
- cleanup_usage_stale_body_fields
2026-06-16 04:19:58 +08:00
elky 669636d3e4 Harden PII redaction format conversion 2026-06-14 20:36:57 +08:00
elky 68038c182b Distinguish expired OAuth token status 2026-06-12 19:43:59 +08:00
elky 308cc88ef7 Fix provider deletion cleanup 2026-06-12 16:25:11 +08:00
elky 30b545785f feat: improve failover rules and request timeline 2026-06-11 00:49:29 +08:00
elky 31fade82f6 Preserve OpenAI encrypted reasoning blocks 2026-06-10 20:02:03 +08:00
elky 0246ba93dd fix(transport): preserve safe accept encoding 2026-06-10 19:58:57 +08:00
elky ff7ec8575c fix(ai-formats): ignore null stream errors 2026-06-10 18:44:46 +08:00
elky aa58cb4a05 build: speed up release image linking 2026-06-10 18:37:23 +08:00
elky e9b4efc2d4 fix(ai-serving): preserve explicit request encoding 2026-06-10 18:16:55 +08:00
elky ea76f7bb0b Support OpenAI Responses builtin tool stream items 2026-06-10 14:49:13 +08:00
elky 8edcbdcb29 feat(usage): expose request timing details 2026-06-10 09:16:15 +08:00
ndllz 5249660e07 fix: respect oauth module disabled state 2026-06-09 18:13:07 +08:00
ndllz 84b99a641a fix: speed up usage activity heatmap render 2026-06-09 16:52:58 +08:00
AAEE86 4824e4a487 fix(frontend): 移除账号导入重复处理中提示 2026-06-09 16:40:45 +08:00
zhefox ba723ebe48 fix(usage): always use truncated body placeholder when limit exceeded 2026-06-09 09:59:10 +08:00
zhefox 04ba8cbe9e fix(gateway): support openai image accept negotiation 2026-06-08 20:40:30 +08:00
elky 84f41dae77 feat(format): audit same-format compatibility rewrites 2026-06-08 16:12:37 +08:00
zhefox 82040bfc21 fix(gateway): support OpenAI image edit requests 2026-06-08 13:22:08 +08:00
elky 6155ffefcc fix(format): avoid false cache-control conversion blocks 2026-06-08 00:52:48 +08:00
elky bf4279a590 Merge remote-tracking branch 'origin/main' into dev
# Conflicts:
#	crates/aether-ai-formats/src/formats/openai/chat/stream.rs
#	crates/aether-ai-formats/src/formats/openai/responses/response.rs
#	crates/aether-ai-formats/src/formats/shared/sync_products.rs
2026-06-08 00:17:30 +08:00
elky 77759fac54 feat: 新增提供商批量处理功能 2026-06-07 22:57:02 +08:00
elky 63a2fd4dcf Merge origin/main into dev 2026-06-06 03:11:38 +08:00
elky 7a19891c60 fix: distinguish unaudited conversion fields 2026-06-06 00:35:27 +08:00
AAEE86 85573d7980 Merge remote-tracking branch 'upstream/main' 2026-06-05 08:28:43 +08:00
zhefox ebd59246a8 fix(test): assert responses timestamps and output text in finalize tests 2026-06-04 17:35:10 +08:00
zhefox fd27f55fe5 fix(provider): normalize OpenAI Responses modern fields and stream events 2026-06-04 15:45:37 +08:00
fawney19 69b8b96fb8 Merge pull request #625 from stabey/pr/responses-call-items-cache-control-20260604
fix: 剥离 Codex cache_control 并完善 Responses 工具调用展示
2026-06-04 13:59:56 +08:00
fawney19 19d1d36043 Merge pull request #620 from zhefox/main
fix(provider): 修复 Chat reasoning_effort 值域与 Responses 扩展透传
2026-06-04 13:59:42 +08:00
stabey 9f19ca5754 fix(usage): keep streamed call args in responses completion
The response.completed fallback rebuilt every call item with responsesCallInput(), which returns '{}' for a function_call lacking arguments. Since '{}' is truthy, ensureToolCall overwrote arguments already collected from streamed delta events. Guard the completed branch with responsesCallHasInput (matching the output_item.done branch) so empty/default inputs no longer clobber streamed args, and align its dedupe key with the streaming phase to avoid duplicate tool-call rendering when an item has no id. Drop the now-dead '工具调用' fallbacks since responsesCallName never returns empty.
2026-06-04 13:30:20 +08:00
stabey 2de2a792f6 fix(codex): strip cache_control before responses upstream 2026-06-04 12:01:36 +08:00
stabey ada690624b fix(usage): render responses call items in conversation view 2026-06-04 11:06:36 +08:00
elky 465476985b fix: preserve provider schema drift safely 2026-06-03 22:29:24 +08:00
elky da5624c98e chore: add format field coverage generator 2026-06-03 21:44:33 +08:00
elky b2f68bbaf7 feat: enforce full format field coverage audit 2026-06-03 21:18:49 +08:00
elky 7507af5829 feat: audit strict format conversion contracts 2026-06-03 20:27:15 +08:00
zhefox 5e39801bba fix(provider): clamp reasoning effort and filter chat extensions 2026-06-03 10:52:26 +08:00
elky 5ac153a0bb Fix gateway nextest stack limit 2026-06-03 01:32:25 +08:00
elky c7a5155ce4 Fix sync CLI test stack overflow 2026-06-03 01:12:40 +08:00
elky 869c3d3037 Fix finalize local test stack overflow 2026-06-03 00:48:36 +08:00
elky ef6a11c146 fix(gateway): preserve heartbeat no-path fallback 2026-06-03 00:25:01 +08:00
elky 21432911de Merge remote-tracking branch 'origin/pr/605' 2026-06-03 00:16:22 +08:00
elky eb98340924 Merge remote-tracking branch 'origin/pr/604' 2026-06-02 23:34:59 +08:00
elky 746af0d93e Fix Kiro cache usage reporting 2026-06-02 23:06:29 +08:00
elky 08ac9c5c58 Merge remote-tracking branch 'origin/pr/614' 2026-06-02 22:10:41 +08:00
elky bce3bf2b6e Merge remote-tracking branch 'origin/pr/593' 2026-06-02 21:40:59 +08:00
elky 4ec9ca61cf Merge remote-tracking branch 'origin/pr/613'
# Conflicts:
#	apps/aether-gateway/src/tests/usage/direct.rs
2026-06-02 19:27:50 +08:00
elky 657e6aa672 Merge remote-tracking branch 'origin/pr/619' 2026-06-02 19:23:44 +08:00
AAEE86 7835840ebd feat(dashboard): 增加全站实时指标和自动刷新
- 管理员仪表盘新增全站 RPM/TPM 与在线/启用用户指标
- 合并今日请求/费用、全站 RPM/TPM、在线/启用用户卡片展示
- 在线用户按最近 5 分钟活跃请求去重统计
- 全站 RPM/TPM 按最近 60 秒请求与 Token 统计
- 新增仪表盘自动刷新按钮,开启后每 10 秒静默刷新数据
- 同步前端类型、空态占位和仪表盘测试
2026-06-02 18:16:24 +08:00
zhefox 6cabcd85aa fix(provider): preserve Claude messages defaults in responses conversion 2026-06-02 17:14:26 +08:00
elky 03e436707d Fix gateway usage nextest stack overflow 2026-06-02 16:59:13 +08:00
elky 781bc5ac58 Merge branch 'pr-617' 2026-06-02 10:40:34 +08:00
AAEE86 86f72da3d9 feat(health): 增加历史状态条指标 Tooltip
- 为健康监控时间轴返回 timeline_details 分段指标
- Hover 历史状态柱时展示总请求/成功/失败/可用率/状态
- 展示平均耗时/TTFB/速度和完整时间范围
- 修复历史状态柱 Tooltip 触发区域不可用的问题
- 补齐前端类型、详情抽屉透传和 mock 数据
2026-06-02 10:36:54 +08:00
elky 0a2c674ad8 Ignore tunnel release tags for app build version 2026-06-02 09:43:12 +08:00
zhefox 0daa8c196b fix(provider): preserve reasoning and Claude tool results in responses conversion 2026-06-02 09:04:55 +08:00
zhefox 98dc5925a5 fix(provider): preserve openai responses tool history in chat conversion 2026-06-02 00:35:19 +08:00
AAEE86 d5d3f09846 refactor(health): add dashboard overview and related drill-down
- Replace health monitor tabs with a dashboard layout
- Add related health drill-down for endpoint, model, and provider cards
- Render provider health as cards and hide empty monitors
2026-06-02 00:10:59 +08:00
AAEE86 0e6fc96eb1 test(gateway): run wallet usage settlement test on larger stack
Wrap the wallet settlement usage test with the large-stack async test helper to
avoid stack overflow in the default test thread.
2026-06-01 22:40:22 +08:00
AAEE86 b052f40ffb test(gateway): run base usage body capture test on larger stack
Wrap the request_record_level=base local gateway usage test with the existing
large-stack async test helper to avoid stack overflow in the default test thread.
2026-06-01 22:23:54 +08:00
AAEE86 2aef9d2478 Refine mobile usage record metadata layout 2026-06-01 21:59:45 +08:00
AAEE86 8627a18f2e test(gateway): run local usage report test on large stack
Wrap the local OpenAI chat sync usage-reporting test in the existing
large-stack harness to avoid stack overflows under nextest suite load.
2026-06-01 21:45:44 +08:00
AAEE86 21c478be22 fix(health): hide empty endpoint monitors
- Remove raw API format label from endpoint health cards
- Hide endpoint health cards with no requests
2026-06-01 21:24:20 +08:00
AAEE86 9d8f7d158b Refine mobile usage record details
- Move mobile usage actions into the card header
- Add compact user/provider metadata line on mobile
- Preserve hidden unknown toggle and auto refresh controls
2026-06-01 21:13:10 +08:00
AAEE86 c1649fe837 refactor(health): consolidate monitor components 2026-06-01 20:49:38 +08:00
AAEE86 d3c8317939 fix(health): align model health card layout 2026-06-01 18:54:56 +08:00
AAEE86 7ffe33f867 feat(health): refine health monitor metrics
- add TPS to model and provider health payloads

- exclude user-cancelled 499 requests from health statistics

- update model/provider health cards with average latency, average TTFB, TPS, and availability
2026-06-01 18:34:51 +08:00
elky 6c2a57f237 fix rust ci failures 2026-06-01 02:42:10 +08:00
github-actions[bot] 0f4141ef3f chore(tunnel): update download links for tunnel-v0.3.16 2026-05-31 17:46:42 +00:00
elky 37413c0211 Refactor tunnel stability protocol 2026-06-01 01:36:49 +08:00
Entropy.Xu d1b64b6748 修复:完善 Kiro 模拟缓存共享回收 2026-05-31 22:43:27 +08:00
Entropy.Xu c2bcfab7d4 修复:Kiro 模拟缓存接入共享运行时 2026-05-31 22:13:45 +08:00
elky 392353ffff Merge remote-tracking branch 'entropy-xu/codex/ccswitch-import' 2026-05-31 20:52:26 +08:00
Entropy.Xu 9734be31cf 修复:收敛 Kiro 模拟缓存断点语义 2026-05-31 20:45:16 +08:00
elky 905453d62b revert: remove usage elapsed clock calibration 2026-05-31 20:30:51 +08:00
Entropy.Xu a3b8a99709 修复:补齐 Kiro 模拟缓存 TTL 和消息级断点 2026-05-31 20:22:18 +08:00
Entropy.Xu 2e24e5f358 修复:扩大 Kiro 模拟缓存前缀读取范围 2026-05-31 19:52:18 +08:00
github-actions[bot] 40eb3cf6e1 chore(tunnel): update download links for tunnel-v0.3.15 2026-05-31 11:17:51 +00:00
elky 549463088c chore: bump aether-tunnel version to 0.3.15 2026-05-31 19:09:56 +08:00
elky f8b5651883 Support encoded tunnel node names 2026-05-31 16:33:06 +08:00
stabey de0a880ca6 test(gateway): 修复 usage wallet 测试栈溢出 2026-05-31 03:35:01 +08:00
stabey ba4e194cb5 test(gateway): 修复 usage base 记录测试栈溢出 2026-05-31 03:21:29 +08:00
stabey 1c05a722c1 test(gateway): 修复 usage local 同步测试栈溢出 2026-05-31 03:09:11 +08:00
stabey eda94913cf test(gateway): 修复 usage 同步测试栈溢出
CI 中 gateway_records_pending_usage_before_execution_runtime_sync_result_arrives 仍会在默认测试栈上溢出。

复用 large-stack tokio runtime 包装该测试,避免 gateway 全量测试在无业务失败时被 SIGABRT 中断。
2026-05-31 02:54:11 +08:00
stabey 3dfafbc379 fix(ai): 按 Responses 文本分片去重快照
upstream 已有 8abedecb 处理单个 OpenAI Responses 文本流中 delta 与 done/completed 快照重复输出的问题。

本提交保留该方向,并把去重状态从全局文本扩展为按 output_index/item_id 与 content_index 分片记录,避免多个 message item 或多个 text content part 共用同一段快照状态。
2026-05-31 02:54:11 +08:00
stabey 8d1e54eba6 fix(stream): 中途失败时不合成正常收尾
上游流式读取失败后,已经缓冲的局部转换状态可能是不完整的工具调用。

在 terminal failure 存在时跳过 normalizer 和 rewriter 的 finish 路径,避免把半截 tool_use 补成正常的 Claude message_stop。
2026-05-31 02:54:11 +08:00
stabey 6bfd56b54f fix(usage): 避免上游流式错误误记为成功
当上游流式响应中途失败时,sync error payload 可能同时包含合成错误体和部分上游流 body。

优先使用合成错误体生成 usage 终态,避免只因为上游先返回过 200 和部分 SSE 内容就把失败请求记录为 completed/settled。
2026-05-31 02:54:11 +08:00
elky 49f952692b Fix remaining sync chat stack overflows 2026-05-31 02:12:22 +08:00
elky fde15c9b60 Fix sync chat test stack overflow 2026-05-31 01:13:09 +08:00
elky 06f26cfacf Merge remote-tracking branch 'origin/pr/597' 2026-05-31 00:09:42 +08:00
elky 5360665432 test(gateway): avoid stack overflow in cors proxy test 2026-05-30 22:48:19 +08:00
elky a20ac1d31f fix(pool): align oauth status filter with visible state 2026-05-30 21:37:21 +08:00
elky 1bdd300606 Merge branch 'review-pr-612' 2026-05-30 21:26:18 +08:00
elky c56f0198ff Merge branch 'review-pr-611' 2026-05-30 21:26:12 +08:00
elky 02fc6bd4ef Merge branch 'review-pr-610' 2026-05-30 21:26:07 +08:00
elky c3a8352d76 Merge branch 'review-pr-603' 2026-05-30 21:25:58 +08:00
elky 463576915f Merge branch 'review-pr-602' 2026-05-30 21:25:52 +08:00
elky 1db6b9d307 Merge branch 'review-pr-596' 2026-05-30 21:25:46 +08:00
elky b5a02a118f Merge branch 'review-pr-595' 2026-05-30 21:25:39 +08:00
elky ae96d5d61b fix usage trace active key selection 2026-05-30 19:48:18 +08:00
cym ce1d532e3c fix(pool): align status filters with visible key state 2026-05-30 18:49:26 +08:00
Entropy.Xu f27485ec05 fix(kiro): 忽略图片 base64 token 估算 2026-05-30 02:59:04 +08:00
Entropy.Xu 9616f458de fix(kiro): 模拟缓存读取移动断点前缀 2026-05-30 00:42:14 +08:00
MMEXA 3455faf7da 修复格式转换优先级保持的首轮候选排序
让开启格式转换优先级保持的跨格式候选进入首轮候选页。

普通跨格式候选仍延后到后续页,保持原有兜底语义。
2026-05-29 22:01:11 +08:00
Entropy.Xu 7ed4b84654 feat(ccswitch): 添加一键导入和用量查询 2026-05-29 21:39:45 +08:00
github-actions[bot] 0d76a8e478 chore(tunnel): update download links for tunnel-v0.3.14 2026-05-29 13:32:35 +00:00
elky b9612fef9b chore: bump aether-tunnel version to 0.3.14 2026-05-29 21:21:30 +08:00
fawney19 92ae88f1be fix: avoid postgres migration version collision 2026-05-29 16:18:59 +08:00
ZheFox 91a5e58cec Merge branch 'fawney19:main' into main 2026-05-29 15:27:46 +08:00
fawney19 1658925f52 Disable key circuit breaker for pool providers 2026-05-29 15:16:06 +08:00
Entropy.Xu bb5a4454a5 feat(gateway): 添加标准文本非流式心跳 2026-05-29 14:35:16 +08:00
fawney19 9fb600df1b Fix PR 599 check regressions 2026-05-29 12:30:56 +08:00
ZheFox fff4fe4e20 Merge branch 'fawney19:main' into main 2026-05-29 12:27:38 +08:00
fawney19 3e4dfd2bac Merge branch 'pr-599' 2026-05-29 02:46:07 +08:00
fawney19 8bd82c8c95 Update endpoint base URL placeholders 2026-05-29 02:44:03 +08:00
fawney19 b59c724455 Normalize endpoint API root handling 2026-05-29 02:29:33 +08:00
ZheFox 3ee272fd53 Merge branch 'fawney19:main' into main 2026-05-29 01:48:50 +08:00
AAEE86 ab5d1f266f fix(usage): Optimize the billing layout of the request details page for mobile devices 2026-05-29 00:01:51 +08:00
Entropy.Xu 906742e3c4 fix(billing): 复用待支付套餐订单 2026-05-28 23:09:36 +08:00
AAEE86 0ee45f41e1 feat(mobile): Refine mobile usage record layout 2026-05-28 22:43:20 +08:00
RWDai 6d285410c2 Preserve dashboard daily breakdown rows 2026-05-28 22:02:42 +08:00
Entropy.Xu eaabfb83ed fix(tunnel): bound upstream clients and heartbeat deltas 2026-05-28 20:34:08 +08:00
fawney19 ef2953038e Fix usage records filtering and pool trace display 2026-05-28 20:24:48 +08:00
zhefox cc1a63bf01 fix: repair missing routing profiles snapshot 2026-05-28 18:59:04 +08:00
Novick Yuan 4b2d8cef3c fix(usage): calibrate active elapsed clock efficiently 2026-05-28 18:45:16 +08:00
fawney19 df518ad668 fix contracts usage server time header 2026-05-28 17:54:56 +08:00
fawney19 37b0c00701 Merge remote-tracking branch 'origin/main' 2026-05-28 17:19:04 +08:00
fawney19 88f03aaef2 Keep key circuit breaker out of pool scoring 2026-05-28 17:18:42 +08:00
fawney19 ffd8d273c4 Revert "Merge remote-tracking branch 'origin/pr/592'"
This reverts commit 3504875922, reversing
changes made to 5c3a1aecbe.
2026-05-28 17:10:27 +08:00
fawney19 ef2a96bcc4 Remove default hot pool size cap 2026-05-28 17:01:03 +08:00
Novick Yuan 734717899b Invalidate model routing cache after admin model writes 2026-05-28 16:57:28 +08:00
RWDai d2d28c30d9 Use bearer auth for OpenAI embedding passthrough 2026-05-28 16:53:00 +08:00
fawney19 10532e1a55 Merge pull request #594 from AAEE86/main
feat(usage): support output_config effort badge source
2026-05-28 16:48:02 +08:00
fawney19 47886abd2b Preserve streaming usage timing on refresh 2026-05-28 16:33:10 +08:00
fawney19 d076f64db3 Harden usage server timing header passthrough 2026-05-28 16:28:19 +08:00
AAEE86 60e3ffc402 feat(usage): support output_config effort badge source
- extract reasoning effort from provider request body output_config.effort
- include output_config.effort in usage list fallback SQL
- cover the new request body shape in usage metadata tests
2026-05-28 16:14:37 +08:00
fawney19 b21be24faa Merge remote-tracking branch 'origin/pr/591' 2026-05-28 16:07:08 +08:00
fawney19 3504875922 Merge remote-tracking branch 'origin/pr/592' 2026-05-28 16:07:07 +08:00
Entropy.Xu 0f6d4b9146 feat(embedding): 接入阿里云多模态向量端点 2026-05-28 16:05:36 +08:00
Mas0nShi 6ebd39ed0b Stabilize stream first-byte usage test 2026-05-28 15:33:28 +08:00
fawney19 5c3a1aecbe Merge remote-tracking branch 'origin/pr/591' 2026-05-28 15:23:26 +08:00
Mas0nShi 1a45ec9386 Fix Gemini CLI streaming policy for OpenAI chat 2026-05-28 15:05:05 +08:00
fawney19 97133f657f style: 突出路由策略选中标签样式 2026-05-28 15:03:14 +08:00
Novick Yuan b108dc5ea6 Assert usage server timing over HTTP 2026-05-28 15:01:59 +08:00
Novick Yuan 35cf44b38e Extract active usage elapsed clock 2026-05-28 14:30:15 +08:00
Novick Yuan 6412294262 Use header-only usage server timing 2026-05-28 14:30:15 +08:00
Novick Yuan 01c8592ca6 Align admin user usage timing samples 2026-05-28 14:30:15 +08:00
Novick Yuan 9b5c3ecd23 Use shared clock for active usage timers 2026-05-28 14:30:15 +08:00
Novick Yuan 6de684df59 Track server clock offset for usage data 2026-05-28 14:30:15 +08:00
Novick Yuan 8aca1f8b93 Add server time to usage responses 2026-05-28 14:30:15 +08:00
fawney19 bb2fc2ec00 Optimize health monitor database reads 2026-05-28 14:24:37 +08:00
fawney19 18566b5837 Merge remote-tracking branch 'origin/pr/587' 2026-05-28 13:56:04 +08:00
fawney19 069e1c1e60 Fix merged PR check regressions 2026-05-28 13:53:04 +08:00
fawney19 2c28d9979c Merge commit 'refs/pr/585'
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/request.rs
#	apps/aether-gateway/src/tests/ai_execute/stream_provider_gemini/local_cli.rs
#	apps/aether-gateway/src/tests/ai_execute/sync/gemini/cli.rs
#	apps/aether-gateway/src/tests/control/admin/provider_query.rs
#	crates/aether-provider-transport/src/gemini_cli/mod.rs
#	crates/aether-provider-transport/src/gemini_cli/request.rs
#	crates/aether-provider-transport/src/gemini_cli/url.rs
#	crates/aether-provider-transport/src/lib.rs
2026-05-28 13:19:42 +08:00
fawney19 0efb3d340d Merge commit 'refs/pr/530' 2026-05-28 12:53:45 +08:00
fawney19 535039c29e Merge remote-tracking branch 'origin/pr/584' 2026-05-28 12:16:51 +08:00
fawney19 93d3de1644 feat: improve routing policy diagnostics 2026-05-28 12:11:43 +08:00
AAEE86 4ce056fe45 feat(health): add model and provider health monitoring
- Rename the original health monitor to endpoint health monitor
- Add tab navigation for endpoint, model, and provider health views
- Add model health monitor cards with availability, latency, first-byte latency, and 60-point history
- Add admin-only provider health monitor with collapsible active-provider sections
- Show per-provider model health cards after expanding a provider
- Add backend model health and provider health monitor payload builders
- Add admin endpoint for provider health monitoring
- Add provider-scoped usage breakdown filtering for per-provider model statistics
- Add frontend API types and request helpers for model/provider health data
- Add demo mock data for model and provider health monitoring
- Fix model health timeline time-unit handling so request history segments render correctly

Verification:
- cargo fmt
- npm run type-check
- npm run build
- cargo test -p aether-gateway health_models
- cargo test -p aether-gateway health_providers
- cargo test -p aether-gateway gateway_exposes_frontdoor_manifest_without_proxying_upstream
2026-05-28 12:00:20 +08:00
Mas0nShi 9ad9858ac2 Merge origin/main into fix/gemini-cli-v1internal 2026-05-28 11:58:00 +08:00
MMEXA adca142d1e fix: adapt gemini cli to v1internal endpoint 2026-05-28 00:24:56 +08:00
stabey 739e39e1ca fix: 修复缓存 token usage 转换语义
统一 OpenAI、Gemini、Claude 之间缓存 token 的 usage 语义,避免 Claude 侧重复统计缓存输入 token。

同时补充 stream 合并逻辑、字段注释和覆盖转换链路的测试。
2026-05-27 23:49:07 +08:00
fawney19 14ad6e9b75 Merge remote-tracking branch 'origin/pr/583' 2026-05-27 18:42:57 +08:00
fawney19 d46d225a90 暗色模式下交换流式徽章填充与描边样式 2026-05-27 18:38:59 +08:00
ZheFox c05d227df2 Merge branch 'fawney19:main' into main 2026-05-27 17:06:31 +08:00
fawney19 42e723ff7c Clarify API key concurrency skip reasons 2026-05-27 17:00:09 +08:00
ZheFox b02d62642a Merge branch 'fawney19:main' into main 2026-05-27 16:18:21 +08:00
zhefox 8abedecb16 fix(ai): dedupe OpenAI responses text snapshot deltas 2026-05-27 16:11:02 +08:00
fawney19 d77a572dc7 fix: cast usage provider body before jsonb type checks 2026-05-27 15:48:12 +08:00
fawney19 8606455355 Merge pull request #581 from zhefox/main
fix(gateway): preserve JSON mode chat hints in responses normalization
2026-05-27 15:40:17 +08:00
fawney19 21e52722e6 Prefer provider request body for usage badges 2026-05-27 15:39:39 +08:00
fawney19 6673ab6d4a Add fast model directive service tier 2026-05-27 15:08:23 +08:00
fawney19 d488b1a680 fix auth refresh request body 2026-05-27 15:06:51 +08:00
fawney19 b9ac97ebc3 Simplify request detail cost overview 2026-05-27 14:24:57 +08:00
zhefox e09d3199c1 fix(gateway): update codex prompt cache key test 2026-05-27 13:57:48 +08:00
fawney19 ccfc4cbddc Cache provider catalog lookups 2026-05-27 13:56:39 +08:00
zhefox 41ad422002 fix(gateway): preserve JSON mode chat hints in responses normalization 2026-05-27 13:35:16 +08:00
fawney19 674cc85005 Stabilize stream runtime nextest timing 2026-05-27 11:00:04 +08:00
fawney19 dd2da69361 Merge pull request #580 from AAEE86/main
fix(mobile): improve usage and pool management layouts
2026-05-27 10:23:17 +08:00
fawney19 0ee6e393ce Record stream first byte on upstream event 2026-05-27 10:19:49 +08:00
fawney19 433a4d3c7d test(gateway): run claude pii redaction cases on large stack 2026-05-27 09:21:57 +08:00
AAEE86 049f26c03b fix(mobile): improve usage and pool management layouts
- Fix pool account batch dialog scrolling on mobile
- Rework usage records mobile filters into clearer rows
- Align user filter styling with other select filters
- Improve request detail drawer metric layout on mobile
2026-05-27 09:20:10 +08:00
fawney19 cf8372c8cb fix(usage): record visible stream first byte timing 2026-05-27 02:48:01 +08:00
fawney19 f03550415b style(usage): make fast badge white 2026-05-27 02:11:59 +08:00
fawney19 5a710c4f5e Merge remote-tracking branch 'origin/pr/578' 2026-05-27 02:07:52 +08:00
fawney19 56901f91ce Merge remote-tracking branch 'origin/pr/576' 2026-05-27 02:06:22 +08:00
fawney19 1109c3547c Merge branch 'pr-577' 2026-05-27 01:35:25 +08:00
fawney19 d24ead234d Merge branch 'pr-575'
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/family/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/family/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/chat/decision/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/request.rs
2026-05-27 01:34:19 +08:00
fawney19 d816ae5c88 Merge remote-tracking branch 'origin/pr/573' 2026-05-27 01:06:57 +08:00
fawney19 8c6e586063 Merge remote-tracking branch 'zhefox/main' 2026-05-27 01:01:34 +08:00
fawney19 c632ec616d Merge remote-tracking branch 'origin/pr/564' 2026-05-27 00:52:15 +08:00
fawney19 bd71a46c25 Merge pull request #561 from Kayphoon/codex/s3-integrated-backup 2026-05-27 00:48:59 +08:00
fawney19 e2b5c3acc8 docs: remove simple query inventory 2026-05-27 00:45:03 +08:00
AAEE86 e27ca671fd feat(usage): show reasoning and fast badges in usage records
- extract provider reasoning effort from request body metadata
- extract priority service tier and expose it as service_tier
- show reasoning level and fast badges after model names
- include badges in active request updates and usage list payloads
- add targeted backend and frontend coverage
2026-05-27 00:36:52 +08:00
fawney19 614c999871 feat(admin): expose s3 backup as module 2026-05-27 00:30:37 +08:00
fawney19 42693c2c52 chore: remove s3 backup docs 2026-05-27 00:07:37 +08:00
fawney19 e21cd72181 fix: preserve pool scan budget for exhausted accounts 2026-05-26 23:49:26 +08:00
MMEXA a9e6a7d644 fix(frontend): recover login redirect navigation 2026-05-26 23:22:30 +08:00
yangrs ba72770cab fix: wire windsurf oauth runtime scheduling 2026-05-26 22:40:46 +08:00
Kayphoon 7530bec7de test(gateway): cover chat pii redaction formats 2026-05-26 22:29:45 +08:00
zhefox 1173a4d9d5 fix(provider): split partial model fetch warnings from errors 2026-05-26 17:45:06 +08:00
zhefox aa409a8a9c fix(provider): refine endpoint default paths for openai and claude roots 2026-05-26 16:50:36 +08:00
AAEE86 949e251b2e fix(provider): 模型测试按 Key 模型权限过滤
测试模型前检查 provider key 的 allowed_models:
- 空权限视为允许所有模型
- 非空权限需匹配请求模型或映射后的实际模型
- 不匹配的 key 标记为跳过,避免发起测试请求

同时补充相关单测和前端跳过原因文案。
2026-05-26 16:46:21 +08:00
fawney19 4933ae9014 Merge pull request #572 from AAEE86/main
fix(admin): add User-Agent for Done-hub provider ops
2026-05-26 16:19:02 +08:00
AAEE86 1793443b09 fix(admin): add User-Agent for Done-hub provider ops
- Done-hub Cookie 请求增加浏览器 User-Agent
- 覆盖认证验证和余额查询的共享请求头
- 补充请求头测试,确认 Cookie 与 User-Agent 同时发送
2026-05-26 16:07:02 +08:00
ZheFox 23a36e37bb Merge branch 'fawney19:main' into main 2026-05-26 15:56:17 +08:00
zhefox c9cf1d458a fix(provider): factor model fetch route test type alias 2026-05-26 15:56:03 +08:00
zhefox d28a389a93 fix(provider): support unversioned API roots in model fetch 2026-05-26 15:39:56 +08:00
fawney19 7e76c9763d Clarify stream first byte timeout message 2026-05-26 15:02:50 +08:00
Kayphoon 9e029462aa test(gateway): stabilize stream timeout regression 2026-05-26 14:41:56 +08:00
Kayphoon 4523a2c67b fix(data): cast MySQL usage aggregates 2026-05-26 14:41:56 +08:00
Kayphoon 84c8bc960e feat(admin): add configurable S3 backups 2026-05-26 14:41:56 +08:00
zhefox c4927162b7 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-26 13:57:27 +08:00
zhefox 1ebe0aeadf fix(users): allow clearing explicit admin group memberships 2026-05-26 13:57:22 +08:00
ZheFox 992c58f2bd Merge branch 'fawney19:main' into main 2026-05-26 13:08:59 +08:00
zhefox 0bf63cc80e fix(provider): support multi-key selection in model tests 2026-05-26 13:08:35 +08:00
zhefox 5fc6dc8019 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-26 12:31:03 +08:00
zhefox 96184caa48 fix(codex): strip unsupported OpenAI responses body fields 2026-05-26 12:30:58 +08:00
fawney19 12ff87949d fix(ai): preserve combined Gemini builtin tools 2026-05-26 11:14:24 +08:00
fawney19 b75953bf4c Merge remote-tracking branch 'origin/pr/569' 2026-05-26 11:11:13 +08:00
MMEXA c733139091 fix(ai): normalize Gemini search grounding tools 2026-05-26 05:33:34 +08:00
fawney19 331d37be26 test: run sub2api balance provider ops on larger stack 2026-05-26 02:30:39 +08:00
fawney19 57ccd44b89 fix: fill provider quota execution timeout defaults 2026-05-26 02:11:10 +08:00
fawney19 d60b6e7454 test: run gemini image bridge case on larger stack 2026-05-26 01:52:01 +08:00
fawney19 50e4f27276 Merge remote-tracking branch 'origin/pr/567' 2026-05-26 01:21:03 +08:00
fawney19 a0f22ae659 fix: tighten pr 566 claude and deepseek handling 2026-05-26 00:57:28 +08:00
fawney19 235f32e10e Merge remote-tracking branch 'origin/pr/566' into review/pr-566-fix 2026-05-26 00:43:45 +08:00
fawney19 e7b3acdec3 fix(provider): format oauth import tests 2026-05-25 23:48:04 +08:00
fawney19 f3a367b02d Merge commit 'refs/pull/563/head' of github-fawney19:fawney19/Aether into review/pr-562 2026-05-25 23:44:02 +08:00
fawney19 c03aebba3f Merge branch 'pr-562' into review/pr-562 2026-05-25 23:10:29 +08:00
fawney19 4fb8955bc2 fix(provider): move key model auto-match into dialog 2026-05-25 23:03:52 +08:00
Novick Yuan 5dfccdec3e Fix stream candidate watchdog timeout semantics 2026-05-25 21:43:13 +08:00
fawney19 8b386b0aac Merge remote-tracking branch 'origin/pr/555' 2026-05-25 21:14:45 +08:00
hemo94931 9010f0806a fix(ai): sanitize Claude Read pages passthrough 2026-05-25 21:01:10 +08:00
hemo94931 495795327c fix(ai): sanitize empty Read pages for Claude tools 2026-05-25 21:01:10 +08:00
root 686311eabf test: align OpenAI image stream keepalive expectation 2026-05-25 21:01:10 +08:00
root e68b843875 Add DeepSeek thinking compatibility 2026-05-25 21:00:08 +08:00
root b46028cb85 refactor: clarify OpenAI SSE control policy 2026-05-25 21:00:08 +08:00
root fa172ecb95 fix: avoid synthetic keepalive for OpenAI streams 2026-05-25 21:00:08 +08:00
root 431311979a fix: handle split streaming terminal events 2026-05-25 20:58:17 +08:00
fawney19 d3249485fa Fix stream timeout semantics 2026-05-25 20:09:37 +08:00
zhefox 4c22a819f9 fix(provider): always show batch assign models action 2026-05-25 20:05:13 +08:00
zhefox f4d66021e4 Merge remote-tracking branch 'upstream/main'
# Conflicts:
#	crates/aether-data/src/repository/usage/mysql.rs
2026-05-25 17:29:12 +08:00
zhefox b72abec2fc fix(gateway): spawn oauth account refresh asynchronously 2026-05-25 15:09:41 +08:00
zhefox db4f3fd210 fix(usage): cast mysql usage aggregates to numeric types 2026-05-25 13:56:12 +08:00
zhefox 230ce5df5f Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-25 13:38:38 +08:00
zhefox ec681335e8 fix(usage): treat empty body_state as missing terminal event 2026-05-25 13:38:18 +08:00
Entropy.Xu aaad113190 feat(admin-users): 支持按创建时间排序 2026-05-25 12:21:58 +08:00
calida-tec 63681b4be3 fix(provider): accept common OAuth token JSON aliases 2026-05-25 10:02:47 +08:00
MMEXA b347f1816d Fix native Antigravity stream envelope handling 2026-05-25 09:39:55 +08:00
ZheFox e9efc5c42a Merge branch 'fawney19:main' into main 2026-05-25 09:15:13 +08:00
MMEXA 28c3a5dbe4 Add Antigravity v1internal gateway adapter 2026-05-25 06:58:15 +08:00
ZheFox ba188aea92 Merge branch 'fawney19:main' into main 2026-05-24 23:40:40 +08:00
zhefox c92bdfba16 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-24 18:51:28 +08:00
ZheFox 83a2609344 Merge branch 'fawney19:main' into main 2026-05-24 18:51:00 +08:00
zhefox 2207b60834 fix(usage): preserve failed status for active request refreshes 2026-05-24 18:48:50 +08:00
ZheFox c09bb28d16 Merge branch 'fawney19:main' into main 2026-05-24 18:07:09 +08:00
zhefox bbd4338e8e Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-24 15:21:27 +08:00
zhefox e6423a91aa feat(provider): auto-match batch assign models from key 2026-05-24 15:19:37 +08:00
Mas0nShi 0d80db8a8d Refactor Gemini CLI v1internal planner request builder 2026-05-22 18:57:03 +08:00
Mas0nShi 8cc6888c5b Fix Gemini CLI OpenAI conversion envelope 2026-05-22 18:40:11 +08:00
Mas0nShi c67818ee86 Fix Gemini CLI standard conversion envelope 2026-05-22 18:24:40 +08:00
Mas0nShi 8df0e1790d Fix Gemini CLI batch import parse error entry 2026-05-22 17:25:13 +08:00
Mas0nShi 8b7643e150 Merge remote-tracking branch 'origin/main' into fix/gemini-cli-v1internal
# Conflicts:
#	apps/aether-gateway/src/ai_serving/transport.rs
#	apps/aether-gateway/src/handlers/admin/provider/oauth/dispatch/batch/parse.rs
#	apps/aether-gateway/src/handlers/shared/catalog.rs
#	crates/aether-admin/src/provider/quota.rs
#	crates/aether-model-fetch/src/strategy.rs
#	crates/aether-provider-pool/src/lib.rs
#	crates/aether-provider-pool/src/service.rs
#	crates/aether-provider-transport/src/provider_types.rs
#	frontend/src/features/providers/components/ProviderDetailDrawer.vue
#	frontend/src/utils/__tests__/providerKeyQuota.spec.ts
#	frontend/src/utils/providerKeyQuota.ts
#	frontend/src/views/admin/PoolManagement.vue
2026-05-22 17:13:57 +08:00
Mas0nShi ce02f1ae8c Add Gemini CLI v1internal quota support 2026-05-22 16:58:04 +08:00
Mas0nShi 9533bd7043 fix: route Gemini CLI generateContent through stream 2026-05-22 09:30:55 +08:00
Mas0nShi 66f21de50e fix: unwrap Gemini CLI model test envelopes 2026-05-21 17:39:10 +08:00
Mas0nShi 3e6ce6cf4a fix: hydrate Gemini CLI project metadata 2026-05-21 17:10:17 +08:00
Mas0nShi e53d5f07e8 feat: support Gemini CLI v1internal quota 2026-05-21 15:38:10 +08:00
1990 changed files with 378472 additions and 79289 deletions
+29 -55
View File
@@ -4,6 +4,10 @@
# 应用端口(默认 8084)
APP_PORT=8084
# 对外访问地址,用于一键安装、CC Switch 导入、支付回调等需要生成公网 URL 的场景。
# 生产环境建议显式配置为不带内部端口的公网域名,例如 https://aether.example.com
# AETHER_PUBLIC_BASE_URL=https://aether.example.com
# Docker Compose 镜像(默认正式版 latest;提前测试可改 rc/beta;也可固定具体版本)
# 示例:
# APP_IMAGE=ghcr.io/fawney19/aether:latest
@@ -54,63 +58,33 @@ ADMIN_USERNAME=admin123456
# ==================== 可选配置(有默认值) ====================
# 可信反向代理 IP/CIDR,只有这些来源发送的 X-Real-IP / X-Forwarded-For 会被采用。
# 默认仅信任本机回环代理:127.0.0.0/8,::1/128。
# Docker/Nginx 位于独立容器时,请按实际容器网络设置,例如:172.16.0.0/12。
# AETHER_TRUSTED_PROXY_CIDRS=127.0.0.0/8,::1/128,172.16.0.0/12
# docker compose 下 app 启动前自动执行 pending migration/backfill(默认 true)
# AETHER_GATEWAY_AUTO_PREPARE_DATABASE=true
# 管理后台更新策略:
# - systemd/二进制部署使用 self:下载 GitHub Release 包,校验 SHA256 后切换 current 并重启。
# - Docker Compose 使用 docker:后台只提示版本,实际更新请在 compose 目录执行 ./update.sh。
# - 源码/本地构建使用 manual:手动拉取源码或下载 release。
# Compose 默认把持久化文件放在 ./datas/{postgres,mysql,sqlite,redis},日志放在 ./logs。
# 分布式/多节点部署不要使用 ./datas 作为共享数据目录;应使用外部共享 Postgres/MySQL 和 Redis。
# 多节点不要从管理后台一键更新单个节点,应使用镜像滚动更新、systemd 分批发布或外部编排。
# AETHER_BASE_DIR=/opt/aether
# AETHER_UPDATE_STRATEGY=docker
# AETHER_DOCKER_UPDATE_COMMAND=./update.sh
# AETHER_GATEWAY_DEPLOYMENT_TOPOLOGY=single-node
# AETHER_GATEWAY_NODE_ROLE=all
# Docker Compose 默认强制把应用日志输出到 stdout/stderr,并由 Docker 轮转日志。
# 如需文件日志,需要在 compose 里把 AETHER_LOG_DESTINATION 改成 file 或 both,
# 并把容器用户可写目录挂载到 /opt/aether/logs。
# AETHER_LOG_DESTINATION=stdout
# AETHER_LOG_FORMAT=pretty
# AETHER_LOG_DIR=/opt/aether/logs
# 服务器访问 GitHub 需要代理时可配置;也兼容 UPDATE_PROXY_URL / HTTPS_PROXY / ALL_PROXY / HTTP_PROXY。
# 如果 Aether 跑在 Docker 容器里,想走宿主机代理时请写 host.docker.internal,不要写 127.0.0.1。
# AETHER_UPDATE_PROXY_URL=http://host.docker.internal:7890
# 共享出口触发 GitHub API 限流时可配置只读 token;也兼容 GITHUB_TOKEN / GH_TOKEN。
# AETHER_UPDATE_GITHUB_TOKEN=
# 下载超时控制:总超时默认 600 秒;连续无响应/无数据默认 30 秒。
# AETHER_UPDATE_DOWNLOAD_TIMEOUT_SECS=600
# AETHER_UPDATE_DOWNLOAD_IDLE_TIMEOUT_SECS=30
# 本地联调后台在线更新(配合 docker-compose.release-local.yml):
# 会用当前源码构建 release-layout 测试镜像,并伪装成较旧版本以触发升级入口。
# AETHER_RELEASE_LOCAL_VERSION=v0.7.0
# AETHER_RELEASE_LOCAL_PORT=18085
# LOCAL_RELEASE_APP_IMAGE=aether-app:release-local
# PostgreSQL 连接池配置(默认按 CPU 自动计算;正式高并发环境可显式预算)
# AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS=12
# AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS=80
# AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS=2048
# AETHER_GATEWAY_REQUEST_BODY_BUFFER_BUDGET_MB=256
# AETHER_GATEWAY_REQUEST_BODY_READ_TIMEOUT_MS=120000
# 可选的 Payload 上限(MiB);默认及 0 均表示不限制。
# AETHER_MAX_REQUEST_BODY_MB=0
# AETHER_GATEWAY_SECURITY_CACHE_TTL_MS=1000
# AETHER_MAX_REDACTED_SYNC_RESPONSE_BODY_MB=0
# AETHER_MAX_INTERNAL_BUFFERED_BODY_MB=0
# AETHER_TUNNEL_NODE_STATUS_QUEUE_CAPACITY=1024
# PostgreSQL 连接池配置(默认适合单实例/小型部署;高并发可按需调大)
# 推荐计算方式(单实例):
# MAX = CPU 核数 × 10(AI 网关偏 IO 等待,可激进些;纯 OLTP 用 × 4)
# MIN = MAX × 0.2(保留常驻连接应对突发流量,避免冷启动握手开销)
# 多实例部署时请按 实例数 × MAX 控制总和,PG 端 max_connections 至少为该总和 + 20 余量
# AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS=4
# AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS=20
# AETHER_GATEWAY_DATA_POSTGRES_STATEMENT_CACHE_CAPACITY=100
# AETHER_GATEWAY_DATA_POSTGRES_ACQUIRE_TIMEOUT_MS=3000
# PostgreSQL 性能调优(默认值适合 2核4GB 机器,按实际配置覆盖)
# 参考:shared_buffers ≈ 可用内存 25%,effective_cache_size ≈ 可用内存 50-75%
# POSTGRES_SHM_SIZE 控制 Docker 容器 /dev/shm;仪表盘统计等并行查询会使用它。
# work_mem 是每个连接每个排序操作的内存,不要设太大(并发数 × work_mem 是实际占用)
# | 系统内存 | shared_buffers | effective_cache_size | work_mem |
# | 2GB | 256MB | 768MB | 4MB |
# | 4GB | 1GB | 3GB | 16MB |
# | 8GB | 2GB | 6GB | 16MB |
# | 16GB | 4GB | 12GB | 32MB |
# | 32GB+ | 8GB | 24GB | 32MB |
# POSTGRES_SHARED_BUFFERS=1GB
# POSTGRES_EFFECTIVE_CACHE_SIZE=3GB
# POSTGRES_SHM_SIZE=512mb
# PostgreSQL 容器调优:docker-compose.yml 已内置通用默认值,通常不用配置。
# 只有在 Postgres 独占大内存、或压测显示 DB 缓存/排序/维护任务成为瓶颈时再覆盖。
# 内置默认:shared_buffers=1GB, effective_cache_size=3GB, shm_size=512mb,
# work_mem=16MB, maintenance_work_mem=256MB。
# POSTGRES_SHARED_BUFFERS=8GB
# POSTGRES_EFFECTIVE_CACHE_SIZE=24GB
# POSTGRES_SHM_SIZE=2gb
# POSTGRES_WORK_MEM=16MB
# POSTGRES_MAINTENANCE_WORK_MEM=256MB
# POSTGRES_MAINTENANCE_WORK_MEM=1GB
+1
View File
@@ -131,6 +131,7 @@ jobs:
aether-tunnel-*.tar.gz
aether-tunnel-*.zip
if-no-files-found: error
retention-days: 1
release:
needs: build
+172 -18
View File
@@ -25,6 +25,8 @@ concurrency:
env:
CARGO_INCREMENTAL: 0
CARGO_PROFILE_DEV_DEBUG: 0
CARGO_PROFILE_TEST_DEBUG: 0
CARGO_TERM_COLOR: always
jobs:
@@ -68,7 +70,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo clippy -p aether-gateway --all-targets -- -D warnings
run: cargo clippy -p aether-gateway --lib --bins --examples -- -D warnings
- name: Show sccache stats
if: always()
@@ -136,7 +138,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo clippy --workspace --exclude aether-gateway --exclude aether-data --all-targets -- -D warnings
run: cargo clippy --workspace --exclude aether-gateway --exclude aether-data --exclude aether-integration-tests --all-targets -- -D warnings
- name: Show sccache stats
if: always()
@@ -184,14 +186,27 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Setup mold
uses: rui314/setup-mold@v1
- name: Install nextest
uses: taiki-e/install-action@nextest
- name: Test
- name: Test lib
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run -p aether-gateway
RUST_MIN_STACK: "16777216"
RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
run: cargo nextest run -p aether-gateway --lib
- name: Test bin
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
RUST_MIN_STACK: "16777216"
RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
run: cargo nextest run -p aether-gateway --bin aether-gateway
- name: Show sccache stats
if: always()
@@ -237,6 +252,45 @@ jobs:
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
check_data_features:
name: Check (Data Feature - ${{ matrix.feature }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
feature:
- postgres
- mysql
- sqlite
- all-drivers
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Check selected data driver
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo check -p aether-data --no-default-features --features ${{ matrix.feature }}
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
test_rest:
name: Test (Workspace Rest)
runs-on: ubuntu-latest
@@ -265,7 +319,79 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run --workspace --exclude aether-gateway --exclude aether-data
run: cargo nextest run --workspace --exclude aether-gateway --exclude aether-data --exclude aether-integration-tests
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
test_data_adapters:
name: Test (Data Adapter - ${{ matrix.package }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
package:
- aether-data-postgres
- aether-data-mysql
- aether-data-sqlite
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Install nextest
uses: taiki-e/install-action@nextest
- name: Test adapter
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run -p ${{ matrix.package }}
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
check_integration_scenarios:
name: Test (Integration Scenarios)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Test scenario binaries
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo test -p aether-integration-tests --bins
- name: Show sccache stats
if: always()
@@ -280,14 +406,20 @@ jobs:
needs:
- test_gateway
- test_data
- check_data_features
- test_rest
- test_data_adapters
- check_integration_scenarios
if: ${{ always() }}
steps:
- name: Verify test jobs
run: |
if [ "${{ needs.test_gateway.result }}" != "success" ] || \
[ "${{ needs.test_data.result }}" != "success" ] || \
[ "${{ needs.test_rest.result }}" != "success" ]; then
[ "${{ needs.check_data_features.result }}" != "success" ] || \
[ "${{ needs.test_rest.result }}" != "success" ] || \
[ "${{ needs.test_data_adapters.result }}" != "success" ] || \
[ "${{ needs.check_integration_scenarios.result }}" != "success" ]; then
echo "Tests failed"
exit 1
fi
@@ -317,7 +449,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo test -p aether-data sqlite --lib
run: cargo test -p aether-data --all-features sqlite --lib
- name: Show sccache stats
if: always()
@@ -361,26 +493,48 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Add PostgreSQL server binaries to PATH
run: echo "$(pg_config --bindir)" >> "$GITHUB_PATH"
- name: Run Postgres migration smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data postgres_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features postgres_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
- name: Run Postgres provider metadata migration smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data --all-features postgres_provider_upstream_metadata_migration_preserves_json_when_url_is_set --lib -- --nocapture
- name: Run Postgres API key lifecycle tests
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_REQUIRE_LOCAL_POSTGRES_TESTS: "true"
run: |
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_request_candidates_preserve_deleted_api_key_identity --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_request_candidate_migration_decouples_legacy_api_key_foreign_key --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_stats_daily_api_key_migration_decouples_legacy_foreign_key --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_expired_api_key_cleanup_preserves_historical_identity --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_api_key_leaderboard_user_filter_preserves_aggregate_history --lib -- --exact --nocapture
- name: Run Postgres core export smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data postgres_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features postgres_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
- name: Run SQLite-to-Postgres import smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data sqlite_core_export_reads_migrated_database_rows --lib -- --nocapture
run: cargo test -p aether-data --all-features sqlite_core_export_reads_migrated_database_rows --lib -- --nocapture
- name: Show sccache stats
if: always()
@@ -430,56 +584,56 @@ jobs:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
- name: Run MySQL usage write smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_usage_write_repository_upserts_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_usage_write_repository_upserts_and_flushes_counters_when_url_is_set --lib -- --nocapture
- name: Run MySQL usage read smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_usage_read_repository_reads_usage_contract_views_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_usage_read_repository_reads_usage_contract_views_when_url_is_set --lib -- --nocapture
- name: Run MySQL provider catalog smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_provider_catalog_repository_round_trips_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_provider_catalog_repository_round_trips_when_url_is_set --lib -- --nocapture
- name: Run MySQL core export smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
- name: Run MySQL wallet read smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_wallet_read_repository_reads_wallet_contract_views --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_wallet_read_repository_reads_wallet_contract_views --lib -- --nocapture
- name: Run MySQL wallet daily usage aggregation smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_wallet_daily_usage_aggregation_uses_settlement_wallets_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_wallet_daily_usage_aggregation_uses_settlement_wallets_when_url_is_set --lib -- --nocapture
- name: Run MySQL stats aggregation smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_stats_aggregation_runs_after_mysql_migrations_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_stats_aggregation_runs_after_mysql_migrations_when_url_is_set --lib -- --nocapture
- name: Show sccache stats
if: always()
+1
View File
@@ -2,6 +2,7 @@
# Edit at https://www.toptal.com/developers/gitignore?templates=python
*.rsa
*_rsa
# AI Assistant Configuration
.codex/
Generated
+459 -20
View File
@@ -69,6 +69,14 @@ dependencies = [
"uuid",
]
[[package]]
name = "aether-admission-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-ai-formats"
version = "0.1.0"
@@ -95,6 +103,7 @@ dependencies = [
"aether-pool-core",
"aether-scheduler-core",
"async-trait",
"base64 0.22.1",
"http",
"serde",
"serde_json",
@@ -155,7 +164,9 @@ dependencies = [
"aether-ai-formats",
"aether-cache",
"aether-data-contracts",
"aether-data-query",
"aether-data-mysql",
"aether-data-postgres",
"aether-data-sqlite",
"aether-wallet",
"async-trait",
"chrono",
@@ -183,7 +194,48 @@ dependencies = [
"chrono",
"serde",
"serde_json",
"sha2",
"thiserror 2.0.18",
"tokio",
]
[[package]]
name = "aether-data-mysql"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"aether-data-contracts",
"aether-data-query",
"async-trait",
"chrono",
"chrono-tz",
"flate2",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tracing",
"uuid",
]
[[package]]
name = "aether-data-postgres"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"aether-data-contracts",
"aether-data-query",
"async-trait",
"chrono",
"chrono-tz",
"flate2",
"futures-util",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tracing",
"uuid",
]
[[package]]
@@ -203,6 +255,25 @@ dependencies = [
"toml",
]
[[package]]
name = "aether-data-sqlite"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"aether-data-contracts",
"aether-data-query",
"async-trait",
"chrono",
"chrono-tz",
"flate2",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tracing",
"uuid",
]
[[package]]
name = "aether-dispatch-core"
version = "0.1.0"
@@ -229,6 +300,11 @@ dependencies = [
"aether-data",
"aether-data-contracts",
"aether-dispatch-core",
"aether-gateway-control",
"aether-gateway-execution",
"aether-gateway-frontdoor",
"aether-gateway-tunnel",
"aether-gateway-workers",
"aether-http",
"aether-model-fetch",
"aether-oauth",
@@ -240,7 +316,7 @@ dependencies = [
"aether-runtime-state",
"aether-scheduler-core",
"aether-task-runtime",
"aether-testkit",
"aether-test-support",
"aether-usage-runtime",
"aether-video-tasks-core",
"aether-wallet",
@@ -258,8 +334,13 @@ dependencies = [
"futures-util",
"hmac",
"http",
"http-body-util",
"hyper",
"hyper-util",
"ldap3",
"libc",
"md-5",
"object_store",
"parking_lot",
"regex",
"reqwest",
@@ -269,9 +350,12 @@ dependencies = [
"serde_json",
"sha1",
"sha2",
"socket2 0.6.3",
"sqlx",
"sysinfo",
"tar",
"thiserror 2.0.18",
"tikv-jemalloc-sys",
"tikv-jemallocator",
"tokio",
"tokio-util",
@@ -287,6 +371,63 @@ dependencies = [
"zstd",
]
[[package]]
name = "aether-gateway-control"
version = "0.1.0"
dependencies = [
"http",
]
[[package]]
name = "aether-gateway-execution"
version = "0.1.0"
dependencies = [
"aether-contracts",
"bytes",
"serde_json",
]
[[package]]
name = "aether-gateway-frontdoor"
version = "0.1.0"
dependencies = [
"aether-ai-formats",
"axum",
"bytes",
"futures-util",
"http",
"serde_json",
"tokio",
"tower",
"tracing",
"tracing-subscriber",
"uuid",
]
[[package]]
name = "aether-gateway-tunnel"
version = "0.1.0"
dependencies = [
"aether-admission-core",
"aether-contracts",
"base64 0.22.1",
"bytes",
"http",
"serde",
"serde_json",
]
[[package]]
name = "aether-gateway-workers"
version = "0.1.0"
dependencies = [
"aether-runtime-state",
"aether-task-runtime",
"aether-test-support",
"tokio",
"tracing",
]
[[package]]
name = "aether-http"
version = "0.1.0"
@@ -295,6 +436,48 @@ dependencies = [
"serde",
]
[[package]]
name = "aether-integration-tests"
version = "0.1.0"
dependencies = [
"aether-contracts",
"aether-data",
"aether-data-contracts",
"aether-gateway",
"aether-runtime-state",
"aether-testkit",
"async-stream",
"axum",
"futures-util",
"http",
"reqwest",
"serde",
"serde_json",
"sha2",
"sqlx",
"tokio",
"tokio-tungstenite 0.28.0",
]
[[package]]
name = "aether-loadtools"
version = "0.1.0"
dependencies = [
"aether-http",
"aether-runtime",
"aether-runtime-state",
"aether-test-support",
"bytes",
"futures-util",
"http",
"libc",
"reqwest",
"serde",
"serde_json",
"sysinfo",
"tokio",
]
[[package]]
name = "aether-model-fetch"
version = "0.1.0"
@@ -340,12 +523,21 @@ dependencies = [
"serde_json",
]
[[package]]
name = "aether-provider-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-provider-pool"
version = "0.1.0"
dependencies = [
"aether-data-contracts",
"aether-pool-core",
"aether-provider-transport",
"serde_json",
"url",
"uuid",
@@ -365,6 +557,9 @@ dependencies = [
"async-trait",
"axum",
"base64 0.22.1",
"chrono",
"crypto_box",
"ed25519-dalek",
"http",
"regex",
"reqwest",
@@ -439,11 +634,20 @@ dependencies = [
"sha2",
]
[[package]]
name = "aether-task-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-task-runtime"
version = "0.1.0"
dependencies = [
"aether-runtime",
"aether-task-core",
"serde",
"serde_json",
"tokio",
@@ -451,38 +655,34 @@ dependencies = [
"tracing",
]
[[package]]
name = "aether-test-support"
version = "0.1.0"
dependencies = [
"tokio",
]
[[package]]
name = "aether-testkit"
version = "0.1.0"
dependencies = [
"aether-contracts",
"aether-data",
"aether-data-contracts",
"aether-gateway",
"aether-http",
"aether-loadtools",
"aether-runtime",
"aether-runtime-state",
"async-stream",
"axum",
"bytes",
"futures-util",
"http",
"libc",
"reqwest",
"serde",
"serde_json",
"sqlx",
"sysinfo",
"tokio",
"tokio-tungstenite 0.28.0",
]
[[package]]
name = "aether-tunnel"
version = "0.3.13"
version = "0.3.16"
dependencies = [
"aether-contracts",
"aether-gateway",
"aether-gateway-tunnel",
"aether-http",
"aether-runtime",
"aether-runtime-state",
@@ -521,6 +721,14 @@ dependencies = [
"webpki-roots 0.26.11",
]
[[package]]
name = "aether-usage-core"
version = "0.1.0"
dependencies = [
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "aether-usage-runtime"
version = "0.1.0"
@@ -532,6 +740,7 @@ dependencies = [
"aether-runtime-state",
"async-trait",
"base64 0.22.1",
"futures-util",
"serde",
"serde_json",
"tokio",
@@ -905,6 +1114,19 @@ dependencies = [
"zeroize",
]
[[package]]
name = "bigdecimal"
version = "0.4.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4d6867f1565b3aad85681f1015055b087fcfd840d6aeee6eee7f2da317603695"
dependencies = [
"autocfg",
"libm",
"num-bigint",
"num-integer",
"num-traits",
]
[[package]]
name = "bindgen"
version = "0.72.1"
@@ -953,6 +1175,15 @@ dependencies = [
"serde_core",
]
[[package]]
name = "blake2"
version = "0.10.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "46502ad458c9a52b69d4d4d32775c788b7a1b85e8bc9d482d92250fc0e3f8efe"
dependencies = [
"digest",
]
[[package]]
name = "block-buffer"
version = "0.10.4"
@@ -1134,6 +1365,7 @@ checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad"
dependencies = [
"crypto-common",
"inout",
"zeroize",
]
[[package]]
@@ -1281,6 +1513,16 @@ dependencies = [
"libc",
]
[[package]]
name = "core-foundation"
version = "0.10.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b2a6cd9ae233e7f62ba4e9353e81a88df7fc8a5987b8d445b4d90c879bd156f6"
dependencies = [
"core-foundation-sys",
"libc",
]
[[package]]
name = "core-foundation-sys"
version = "0.8.7"
@@ -1408,6 +1650,36 @@ dependencies = [
"typenum",
]
[[package]]
name = "crypto_box"
version = "0.9.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "16182b4f39a82ec8a6851155cc4c0cda3065bb1db33651726a29e1951de0f009"
dependencies = [
"aead",
"blake2",
"crypto_secretbox",
"curve25519-dalek",
"salsa20",
"subtle",
"zeroize",
]
[[package]]
name = "crypto_secretbox"
version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b9d6cf87adf719ddf43a805e92c6870a531aedda35ff640442cbaf8674e141e1"
dependencies = [
"aead",
"cipher",
"generic-array",
"poly1305",
"salsa20",
"subtle",
"zeroize",
]
[[package]]
name = "csscolorparser"
version = "0.6.2"
@@ -1427,6 +1699,33 @@ dependencies = [
"cipher",
]
[[package]]
name = "curve25519-dalek"
version = "4.1.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "97fb8b7c4503de7d6ae7b42ab72a5a59857b4c937ec27a3d4539dba95b5ab2be"
dependencies = [
"cfg-if",
"cpufeatures",
"curve25519-dalek-derive",
"digest",
"fiat-crypto",
"rustc_version",
"subtle",
"zeroize",
]
[[package]]
name = "curve25519-dalek-derive"
version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f46882e17999c6cc590af592290432be3bce0428cb0d5f8b6715e4dc7b383eb3"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.117",
]
[[package]]
name = "darling"
version = "0.23.0"
@@ -1587,6 +1886,30 @@ version = "1.0.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813"
[[package]]
name = "ed25519"
version = "2.2.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "115531babc129696a58c64a4fef0a8bf9e9698629fb97e9e40767d235cfbcd53"
dependencies = [
"pkcs8",
"signature",
]
[[package]]
name = "ed25519-dalek"
version = "2.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "70e796c081cee67dc755e1a36a0a172b897fab85fc3f6bc48307991f64e4eca9"
dependencies = [
"curve25519-dalek",
"ed25519",
"serde",
"sha2",
"subtle",
"zeroize",
]
[[package]]
name = "either"
version = "1.15.0"
@@ -1659,6 +1982,12 @@ version = "2.4.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6"
[[package]]
name = "fiat-crypto"
version = "0.2.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "28dea519a9695b9977216879a3ebfddf92f1c08c05d984f8996aecd6ecdc811d"
[[package]]
name = "filedescriptor"
version = "0.8.3"
@@ -1897,6 +2226,7 @@ checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
dependencies = [
"typenum",
"version_check",
"zeroize",
]
[[package]]
@@ -2127,6 +2457,12 @@ version = "1.0.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9"
[[package]]
name = "humantime"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424"
[[package]]
name = "hyper"
version = "1.8.1"
@@ -2160,6 +2496,7 @@ dependencies = [
"hyper",
"hyper-util",
"rustls 0.23.37",
"rustls-native-certs 0.8.3",
"rustls-pki-types",
"tokio",
"tokio-rustls 0.26.4",
@@ -2186,6 +2523,7 @@ dependencies = [
"pin-project-lite",
"socket2 0.6.3",
"tokio",
"tower-layer",
"tower-service",
"tracing",
]
@@ -2491,7 +2829,7 @@ dependencies = [
"percent-encoding",
"ring 0.16.20",
"rustls 0.21.12",
"rustls-native-certs",
"rustls-native-certs 0.6.3",
"thiserror 1.0.69",
"tokio",
"tokio-rustls 0.24.1",
@@ -2838,6 +3176,41 @@ dependencies = [
"libc",
]
[[package]]
name = "object_store"
version = "0.12.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fbfbfff40aeccab00ec8a910b57ca8ecf4319b335c542f2edcd19dd25a1e2a00"
dependencies = [
"async-trait",
"base64 0.22.1",
"bytes",
"chrono",
"form_urlencoded",
"futures",
"http",
"http-body-util",
"humantime",
"hyper",
"itertools 0.14.0",
"md-5",
"parking_lot",
"percent-encoding",
"quick-xml",
"rand 0.9.2",
"reqwest",
"ring 0.17.14",
"serde",
"serde_json",
"serde_urlencoded",
"thiserror 2.0.18",
"tokio",
"tracing",
"url",
"wasm-bindgen-futures",
"web-time",
]
[[package]]
name = "oid-registry"
version = "0.6.1"
@@ -2882,6 +3255,12 @@ version = "0.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d05e27ee213611ffe7d6348b942e8f942b37114c00cc03cec254295a4a17852e"
[[package]]
name = "openssl-probe"
version = "0.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe"
[[package]]
name = "ordered-float"
version = "4.6.0"
@@ -3103,6 +3482,17 @@ version = "0.2.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b4596b6d070b27117e987119b4dac604f3c58cfb0b191112e24771b2faeac1a6"
[[package]]
name = "poly1305"
version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8159bd90725d2df49889a078b54f4f79e87f1f8a8444194cdca81d38f5393abf"
dependencies = [
"cpufeatures",
"opaque-debug",
"universal-hash",
]
[[package]]
name = "polyval"
version = "0.6.2"
@@ -3164,6 +3554,16 @@ dependencies = [
"unicode-ident",
]
[[package]]
name = "quick-xml"
version = "0.38.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b66c2058c55a409d601666cffe35f04333cf1013010882cec174a7467cd4e21c"
dependencies = [
"memchr",
"serde",
]
[[package]]
name = "quinn"
version = "0.11.9"
@@ -3497,6 +3897,7 @@ dependencies = [
"pin-project-lite",
"quinn",
"rustls 0.23.37",
"rustls-native-certs 0.8.3",
"rustls-pki-types",
"serde",
"serde_json",
@@ -3649,10 +4050,22 @@ version = "0.6.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a9aace74cb666635c918e9c12bc0d348266037aa8eb599b5cba565709a8dff00"
dependencies = [
"openssl-probe",
"openssl-probe 0.1.6",
"rustls-pemfile",
"schannel",
"security-framework",
"security-framework 2.11.1",
]
[[package]]
name = "rustls-native-certs"
version = "0.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "612460d5f7bea540c490b2b6395d8e34a953e52b491accd6c86c8164c5932a63"
dependencies = [
"openssl-probe 0.2.1",
"rustls-pki-types",
"schannel",
"security-framework 3.7.0",
]
[[package]]
@@ -3708,6 +4121,15 @@ version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f"
[[package]]
name = "salsa20"
version = "0.10.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "97a22f5af31f73a954c10289c93e8a50cc23d971e80ee446f1f6f7137a088213"
dependencies = [
"cipher",
]
[[package]]
name = "schannel"
version = "0.1.29"
@@ -3751,7 +4173,20 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "897b2245f0b511c87893af39b033e5ca9cce68824c4d7e7630b5a1d339658d02"
dependencies = [
"bitflags 2.11.0",
"core-foundation",
"core-foundation 0.9.4",
"core-foundation-sys",
"libc",
"security-framework-sys",
]
[[package]]
name = "security-framework"
version = "3.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d"
dependencies = [
"bitflags 2.11.0",
"core-foundation 0.10.1",
"core-foundation-sys",
"libc",
"security-framework-sys",
@@ -4025,6 +4460,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ee6798b1838b6a0f69c007c133b8df5866302197e404e8b6ee8ed3e3a5e68dc6"
dependencies = [
"base64 0.22.1",
"bigdecimal",
"bytes",
"chrono",
"crc",
@@ -4101,6 +4537,7 @@ checksum = "aa003f0038df784eb8fecbbac13affe3da23b45194bd57dba231c8f48199c526"
dependencies = [
"atoi",
"base64 0.22.1",
"bigdecimal",
"bitflags 2.11.0",
"byteorder",
"bytes",
@@ -4144,6 +4581,7 @@ checksum = "db58fcd5a53cf07c184b154801ff91347e4c30d17a3562a635ff028ad5deda46"
dependencies = [
"atoi",
"base64 0.22.1",
"bigdecimal",
"bitflags 2.11.0",
"byteorder",
"chrono",
@@ -4161,6 +4599,7 @@ dependencies = [
"log",
"md-5",
"memchr",
"num-bigint",
"once_cell",
"rand 0.8.5",
"serde",
+64 -27
View File
@@ -1,34 +1,52 @@
[workspace]
members = [
"apps/aether-tunnel",
"crates/aether-ai-formats",
"crates/aether-ai/formats",
"crates/aether-admin",
"crates/aether-ai-serving",
"crates/aether-admission-core",
"crates/aether-ai/serving",
"crates/aether-pool-core",
"crates/aether-provider-pool",
"crates/aether-provider/core",
"crates/aether-provider/pool",
"crates/aether-routing-core",
"crates/aether-data-contracts",
"crates/aether-data-query",
"crates/aether-data-schema",
"crates/aether-data/contracts",
"crates/aether-data/adapters/postgres",
"crates/aether-data/adapters/mysql",
"crates/aether-data/adapters/sqlite",
"crates/aether-data/query",
"crates/aether-data/schema",
"crates/aether-dispatch-core",
"crates/aether-cache",
"crates/aether-billing",
"crates/aether-wallet",
"crates/aether-crypto",
"crates/aether-contracts",
"crates/aether-data",
"crates/aether-data/runtime",
"crates/aether-model-fetch",
"crates/aether-oauth",
"crates/aether-provider-transport",
"crates/aether-provider/transport",
"crates/aether-scheduler-core",
"crates/aether-runtime-state",
"crates/aether-task-runtime",
"crates/aether-usage-runtime",
"crates/aether-runtime/state",
"crates/aether-task/runtime",
"crates/aether-task/core",
"crates/aether-gateway/frontdoor",
"crates/aether-gateway/control",
"crates/aether-gateway/execution",
"crates/aether-gateway/workers",
"crates/aether-gateway/tunnel",
"crates/aether-testing/loadtools",
"crates/aether-testing/integration",
"crates/aether-usage/core",
"crates/aether-testing/support",
"crates/aether-usage/runtime",
"crates/aether-video-tasks-core",
"apps/aether-gateway",
"crates/aether-http",
"crates/aether-runtime",
"crates/aether-testkit",
"crates/aether-runtime/base",
"crates/aether-testing/testkit",
]
default-members = [
"apps/aether-gateway",
]
resolver = "2"
@@ -39,33 +57,48 @@ repository = "https://github.com/fawney19/Aether.git"
[workspace.dependencies]
aether-admin = { path = "crates/aether-admin" }
aether-ai-formats = { path = "crates/aether-ai-formats" }
aether-ai-serving = { path = "crates/aether-ai-serving" }
aether-admission-core = { path = "crates/aether-admission-core" }
aether-ai-formats = { path = "crates/aether-ai/formats" }
aether-ai-serving = { path = "crates/aether-ai/serving" }
aether-pool-core = { path = "crates/aether-pool-core" }
aether-provider-pool = { path = "crates/aether-provider-pool" }
aether-provider-core = { path = "crates/aether-provider/core" }
aether-provider-pool = { path = "crates/aether-provider/pool" }
aether-routing-core = { path = "crates/aether-routing-core" }
aether-data-contracts = { path = "crates/aether-data-contracts" }
aether-data-query = { path = "crates/aether-data-query" }
aether-data-schema = { path = "crates/aether-data-schema" }
aether-data-contracts = { path = "crates/aether-data/contracts" }
aether-data-postgres = { path = "crates/aether-data/adapters/postgres" }
aether-data-mysql = { path = "crates/aether-data/adapters/mysql" }
aether-data-sqlite = { path = "crates/aether-data/adapters/sqlite" }
aether-data-query = { path = "crates/aether-data/query" }
aether-data-schema = { path = "crates/aether-data/schema" }
aether-dispatch-core = { path = "crates/aether-dispatch-core" }
aether-cache = { path = "crates/aether-cache" }
aether-billing = { path = "crates/aether-billing" }
aether-wallet = { path = "crates/aether-wallet" }
aether-crypto = { path = "crates/aether-crypto" }
aether-contracts = { path = "crates/aether-contracts" }
aether-data = { path = "crates/aether-data" }
aether-data = { path = "crates/aether-data/runtime" }
aether-model-fetch = { path = "crates/aether-model-fetch" }
aether-oauth = { path = "crates/aether-oauth" }
aether-provider-transport = { path = "crates/aether-provider-transport" }
aether-provider-transport = { path = "crates/aether-provider/transport" }
aether-scheduler-core = { path = "crates/aether-scheduler-core" }
aether-runtime-state = { path = "crates/aether-runtime-state" }
aether-task-runtime = { path = "crates/aether-task-runtime" }
aether-usage-runtime = { path = "crates/aether-usage-runtime" }
aether-runtime-state = { path = "crates/aether-runtime/state" }
aether-task-runtime = { path = "crates/aether-task/runtime" }
aether-task-core = { path = "crates/aether-task/core" }
aether-gateway-frontdoor = { path = "crates/aether-gateway/frontdoor" }
aether-gateway-control = { path = "crates/aether-gateway/control" }
aether-gateway-execution = { path = "crates/aether-gateway/execution" }
aether-gateway-workers = { path = "crates/aether-gateway/workers" }
aether-gateway-tunnel = { path = "crates/aether-gateway/tunnel" }
aether-loadtools = { path = "crates/aether-testing/loadtools" }
aether-integration-tests = { path = "crates/aether-testing/integration" }
aether-test-support = { path = "crates/aether-testing/support" }
aether-usage-core = { path = "crates/aether-usage/core" }
aether-usage-runtime = { path = "crates/aether-usage/runtime" }
aether-video-tasks-core = { path = "crates/aether-video-tasks-core" }
aether-gateway = { path = "apps/aether-gateway" }
aether-http = { path = "crates/aether-http" }
aether-runtime = { path = "crates/aether-runtime" }
aether-testkit = { path = "crates/aether-testkit" }
aether-runtime = { path = "crates/aether-runtime/base" }
aether-testkit = { path = "crates/aether-testing/testkit" }
aes = "0.8"
aes-gcm = "0.10"
async-stream = "0.3"
@@ -77,10 +110,13 @@ bytes = "1"
cbc = "0.1"
chrono = { version = "0.4", features = ["serde"] }
chrono-tz = "0.10"
crypto_box = { version = "0.9", features = ["seal"] }
ed25519-dalek = { version = "2.2", features = ["pkcs8"] }
flate2 = "1"
futures-util = "0.3"
hmac = "0.12"
http = "1"
object_store = { version = "0.12", default-features = false, features = ["aws"] }
pbkdf2 = { version = "0.12", default-features = false, features = ["hmac"] }
reqwest = { version = "0.12", default-features = false, features = ["json", "stream", "rustls-tls", "http2", "socks"] }
redis = { version = "0.28", default-features = false, features = ["tokio-comp", "script", "streams", "connection-manager"] }
@@ -91,8 +127,9 @@ serde = { version = "1", features = ["derive"] }
serde_json = { version = "1", features = ["preserve_order"] }
serde_path_to_error = "0.1"
sha2 = "0.10"
socket2 = "0.6"
tar = "0.4"
sqlx = { version = "0.8", default-features = false, features = ["postgres", "mysql", "sqlite", "runtime-tokio-rustls", "chrono"] }
sqlx = { version = "0.8", default-features = false, features = ["runtime-tokio-rustls", "chrono"] }
thiserror = "2"
tokio = { version = "1", features = ["macros", "net", "rt-multi-thread", "signal", "sync", "time"] }
tokio-util = { version = "0.7", features = ["codec", "io-util"] }
+8 -4
View File
@@ -23,10 +23,11 @@ RUN npm run build
FROM ${RUST_BASE_IMAGE} AS gateway-base
WORKDIR /build
# 本地镜像优先缩短构建时间,保留 release 语义,但改用更快的 thin LTO。
# 生产级 release 构建:保留 thin LTO,同时用 lld 缩短最终链接阶段。
ENV CARGO_REGISTRIES_CRATES_IO_PROTOCOL=sparse \
CARGO_PROFILE_RELEASE_LTO=thin \
CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16 \
RUSTFLAGS="-C linker=clang -C link-arg=-fuse-ld=lld"
RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
--mount=type=cache,target=/var/lib/apt,sharing=locked \
@@ -34,10 +35,12 @@ RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
apt-get update && apt-get install -y --no-install-recommends \
build-essential \
ca-certificates \
clang \
cmake \
git \
libclang-dev \
libssl-dev \
lld \
pkg-config \
perl
@@ -59,7 +62,7 @@ COPY --from=gateway-planner /build/recipe.json ./recipe.json
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-local,target=/build/target,sharing=locked \
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --recipe-path recipe.json
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --features jemalloc --recipe-path recipe.json
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
@@ -67,7 +70,8 @@ COPY crates/ ./crates/
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-local,target=/build/target,sharing=locked \
cargo build --release --locked -p aether-gateway && \
set -eux; \
cargo build --release --locked -p aether-gateway --bin aether-gateway --features jemalloc; \
cp target/release/aether-gateway /tmp/aether-gateway
# ==================== 最小运行时打包 ====================
+2 -2
View File
@@ -60,7 +60,7 @@ COPY --from=gateway-planner /build/recipe.json ./recipe.json
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-release-local,target=/build/target,sharing=locked \
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --recipe-path recipe.json
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --features jemalloc --recipe-path recipe.json
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
@@ -68,7 +68,7 @@ COPY crates/ ./crates/
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-release-local,target=/build/target,sharing=locked \
cargo build --release --locked -p aether-gateway && \
cargo build --release --locked -p aether-gateway --features jemalloc && \
cp target/release/aether-gateway /tmp/aether-gateway
# ==================== 最小运行时打包 ====================
+11 -2
View File
@@ -142,8 +142,17 @@ Aether Tunnel 是配套的正向代理节点,部署在海外 VPS 上,为墙
- `APP_PORT`:`aether-gateway` 唯一监听端口,固定绑定 `0.0.0.0:${APP_PORT}`
- `DATABASE_URL`:数据库连接串;SQLite 例如 `sqlite:///opt/aether/data/aether.db`,Postgres 例如 `postgresql://postgres:aether@postgres:5432/aether`
- `AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS` / `AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS`:数据库连接池手动覆盖值;未配置时会自动推导,SQLite 固定 `1/1`,Postgres/MySQL 按 CPU 核心数计算并默认封顶 `100`
- `AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS`:单实例请求并发上限;未配置时按 CPU 自动推导(基础范围 `512-65536`),低文件描述符预算时会进一步下调
- `AETHER_GATEWAY_REQUEST_BODY_BUFFER_BUDGET_MB`:单实例同时读取和解压请求体的加权内存预算,默认 `256MB`
- `AETHER_GATEWAY_REQUEST_BODY_READ_TIMEOUT_MS`:请求体完整读取超时,默认 `120000ms`
- `AETHER_MAX_REQUEST_BODY_MB`:可选的单请求解压后请求体上限;未配置或设为 `0` 时不限制
- `AETHER_MAX_INTERNAL_BUFFERED_BODY_MB`:可选的 heartbeat、管理探测等内部整包响应体上限;未配置或设为 `0` 时不限制
- `AETHER_TUNNEL_NODE_STATUS_QUEUE_CAPACITY`:隧道节点状态上报队列容量,默认 `1024`;满载时拒绝新事件,避免控制面故障导致无界内存增长
- `AETHER_GATEWAY_SECURITY_CACHE_TTL_MS`:IP 黑白名单本地缓存时间,默认 `1000ms`,写操作会主动失效相关缓存
- `AETHER_MAX_REDACTED_SYNC_RESPONSE_BODY_MB`:可选的 PII 恢复同步响应缓冲上限;未配置或设为 `0` 时不限制
- `REDIS_URL`:Redis 连接串;仅 Postgres + Redis 的 Docker Compose 部署需要配置
- `AETHER_RUNTIME_BACKEND=memory|redis`:运行时缓存/协调后端。SQLite 默认用 `memory`,不会连接 Redis
- `AETHER_RUNTIME_BACKEND=memory|redis`:运行时缓存/协调后端。SQLite 默认用 `memory`,不会连接 Redis;多节点部署和需要跨 gateway 重启恢复 OpenAI Responses continuation history 的部署必须使用共享 Redis
- `AETHER_GATEWAY_AUTO_PREPARE_DATABASE`:常规启动前自动执行挂起的 schema migration 和 backfill;仓库自带的 `docker-compose.yml` 默认开启
- `JWT_SECRET_KEY` / `ENCRYPTION_KEY`:认证和敏感数据加密所需密钥
- `API_KEY_PREFIX`:用户和管理员新建 API Key 时使用的前缀,默认 `sk`
@@ -168,4 +177,4 @@ Aether Tunnel 是配套的正向代理节点,部署在海外 VPS 上,为墙
## Star History
[![Star History Chart](https://api.star-history.com/svg?repos=fawney19/Aether&type=Date)](https://star-history.com/#fawney19/Aether&Date)
[![Star History Chart](https://api.star-history.com/svg?repos=fawney19/Aether&type=date&legend=top-left)](https://www.star-history.com/?repos=fawney19%2FAether&type=date&legend=top-left)
+27 -4
View File
@@ -6,6 +6,16 @@ license.workspace = true
repository.workspace = true
description = "Rust ingress gateway for Aether phase 3a transparent proxy"
[features]
default = []
jemalloc = [
"dep:tikv-jemallocator",
"dep:tikv-jemalloc-sys",
"tikv-jemallocator/stats",
"tikv-jemalloc-sys/stats",
]
testkit = []
[dependencies]
aether-admin.workspace = true
aether-ai-formats.workspace = true
@@ -14,10 +24,15 @@ aether-billing.workspace = true
aether-cache.workspace = true
aether-contracts.workspace = true
aether-crypto.workspace = true
aether-data.workspace = true
aether-data = { workspace = true, features = ["all-drivers"] }
aether-data-contracts.workspace = true
aether-dispatch-core.workspace = true
aether-gateway-frontdoor.workspace = true
aether-gateway-control.workspace = true
aether-gateway-execution.workspace = true
aether-http.workspace = true
aether-gateway-workers.workspace = true
aether-gateway-tunnel.workspace = true
aether-model-fetch.workspace = true
aether-oauth.workspace = true
aether-pool-core.workspace = true
@@ -46,8 +61,13 @@ flate2.workspace = true
futures-util.workspace = true
hmac.workspace = true
http.workspace = true
http-body-util = "0.1"
hyper = { version = "1", features = ["client", "server", "http1", "http2"] }
hyper-util = { version = "0.1", features = ["client-legacy", "client-pool", "server-auto", "service", "tokio"] }
ldap3 = { version = "0.11", default-features = false, features = ["sync", "tls-rustls"] }
libc = "0.2"
md-5 = "0.10"
object_store.workspace = true
parking_lot = "0.12"
regex.workspace = true
reqwest.workspace = true
@@ -57,8 +77,10 @@ serde.workspace = true
serde_json.workspace = true
sha1 = "0.10"
sha2 = { workspace = true, features = ["oid"] }
socket2.workspace = true
tar.workspace = true
sqlx.workspace = true
sqlx = { workspace = true, features = ["postgres", "mysql", "sqlite", "migrate"] }
sysinfo = "0.32"
thiserror.workspace = true
tokio.workspace = true
tokio-util.workspace = true
@@ -73,8 +95,9 @@ wreq-util.workspace = true
zstd.workspace = true
[target.'cfg(not(target_env = "msvc"))'.dependencies]
tikv-jemallocator = "0.6"
tikv-jemallocator = { version = "0.6", optional = true }
tikv-jemalloc-sys = { version = "0.6", optional = true }
[dev-dependencies]
aether-testkit.workspace = true
aether-test-support.workspace = true
tracing-subscriber.workspace = true
+13 -16
View File
@@ -11,19 +11,18 @@ fn main() {
let package_version = env::var("CARGO_PKG_VERSION").unwrap_or_else(|_| "unknown".to_string());
let version = env::var("AETHER_BUILD_VERSION")
.ok()
.filter(|value| !value.trim().is_empty())
.and_then(|value| normalize_gateway_version_source(&value))
.or_else(|| {
env::var("AETHER_VERSION")
.ok()
.filter(|value| !value.trim().is_empty())
.and_then(|value| normalize_gateway_version_source(&value))
})
.or_else(|| {
env::var("GITHUB_REF_NAME")
.ok()
.filter(|value| value.trim().starts_with('v'))
.and_then(|value| normalize_gateway_version_source(&value))
})
.or_else(git_describe_version)
.map(|value| normalize_version(&value))
.filter(|value| !value.is_empty())
.unwrap_or(package_version);
@@ -38,7 +37,9 @@ fn main() {
fn git_describe_version() -> Option<String> {
let output = Command::new("git")
.args(["describe", "--tags", "--always", "--dirty"])
.args([
"describe", "--tags", "--match", "v[0-9]*", "--always", "--dirty",
])
.output()
.ok()?;
if !output.status.success() {
@@ -46,17 +47,13 @@ fn git_describe_version() -> Option<String> {
}
let version = String::from_utf8(output.stdout).ok()?;
let version = version.trim();
if version.is_empty() {
None
} else {
Some(version.to_string())
}
normalize_gateway_version_source(version)
}
fn normalize_version(value: &str) -> String {
value
.trim()
.strip_prefix('v')
.unwrap_or(value.trim())
.to_string()
fn normalize_gateway_version_source(value: &str) -> Option<String> {
let trimmed = value.trim();
if trimmed.is_empty() || trimmed.starts_with("tunnel-v") {
return None;
}
Some(trimmed.strip_prefix('v').unwrap_or(trimmed).to_string())
}
@@ -28,6 +28,12 @@ pub(crate) fn maybe_normalize_provider_private_sync_report_payload(
let mut normalized = payload.clone();
normalized.report_context = normalize_provider_private_report_context(Some(report_context));
if let (Some(body_json), Some(context)) = (
payload.body_json.as_ref(),
normalized.report_context.as_mut(),
) {
maybe_attach_gemini_cli_v1internal_credits_context(report_context, body_json, context);
}
if let Some(body_json) = payload.body_json.clone() {
normalized.body_json = normalize_provider_private_response_value(body_json, report_context);
@@ -55,6 +61,44 @@ pub(crate) fn maybe_normalize_provider_private_sync_report_payload(
Ok(Some(normalized))
}
fn maybe_attach_gemini_cli_v1internal_credits_context(
original_report_context: &Value,
body_json: &Value,
normalized_report_context: &mut Value,
) {
if !original_report_context
.get("envelope_name")
.and_then(Value::as_str)
.is_some_and(|value| value.eq_ignore_ascii_case("gemini_cli:v1internal"))
{
return;
}
let mut credits = serde_json::Map::new();
for (source, target) in [
("remainingCredits", "remainingCredits"),
("consumedCredits", "consumedCredits"),
("traceId", "traceId"),
] {
if let Some(value) = body_json
.get(source)
.cloned()
.filter(|value| !value.is_null())
{
credits.insert(target.to_string(), value);
}
}
if credits.is_empty() {
return;
}
if let Some(object) = normalized_report_context.as_object_mut() {
object.insert(
"gemini_cli_v1internal_credits".to_string(),
Value::Object(credits),
);
}
}
fn normalize_provider_private_stream_bytes(
report_context: &Value,
body: &[u8],
+34 -22
View File
@@ -55,16 +55,20 @@ pub(crate) use aether_ai_formats::api::{
ExecutionRuntimeAuthContext, LocalCoreSyncErrorKind, LocalOpenAiImageSpec,
LocalSameFormatProviderFamily, LocalSameFormatProviderSpec, LocalStandardSourceFamily,
LocalStandardSourceMode, LocalStandardSpec, OpenAIChatClientEmitter,
OpenAIResponsesClientEmitter, StreamingStandardTerminalObserver,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_DECISION_ACTION,
GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OpenAIResponsesClientEmitter, StreamingStandardTerminalObserver, CLAUDE_CHAT_STREAM_PLAN_KIND,
CLAUDE_CLI_STREAM_PLAN_KIND, EXECUTION_RUNTIME_STREAM_DECISION_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_CHAT_STREAM_PLAN_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND,
OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_STREAM_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_formats::protocol::stream::CanonicalUsage as StreamingCanonicalUsage;
pub(crate) use aether_ai_formats::CODEX_RESPONSES_LITE_HEADER;
pub(crate) fn parse_direct_request_body(
parts: &http::request::Parts,
@@ -83,28 +87,36 @@ pub(crate) fn resolve_execution_runtime_stream_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
aether_ai_formats::api::resolve_execution_runtime_stream_plan_kind(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
let plan_kind =
aether_ai_formats::api::resolve_execution_runtime_stream_plan_kind_with_client_surface(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, true, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn resolve_execution_runtime_sync_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
aether_ai_formats::api::resolve_execution_runtime_sync_plan_kind(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
let plan_kind =
aether_ai_formats::api::resolve_execution_runtime_sync_plan_kind_with_client_surface(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, false, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn is_matching_stream_request(
@@ -2,6 +2,7 @@ use serde_json::Value;
use crate::ai_serving::{
maybe_build_ai_surface_stream_rewriter, AiSurfaceFinalizeError, AiSurfaceStreamRewriter,
ResponseHistoryRecord,
};
use crate::GatewayError;
@@ -24,6 +25,10 @@ impl LocalStreamRewriter<'_> {
pub(crate) fn finish(&mut self) -> Result<Vec<u8>, GatewayError> {
self.inner.finish().map_err(map_surface_error)
}
pub(crate) fn take_response_history_record(&mut self) -> Option<ResponseHistoryRecord> {
self.inner.take_response_history_record()
}
}
fn map_surface_error(error: AiSurfaceFinalizeError) -> GatewayError {
@@ -8,6 +8,45 @@ fn utf8(bytes: Vec<u8>) -> String {
String::from_utf8(bytes).expect("utf8 should decode")
}
#[test]
fn same_format_claude_local_stream_rewriter_sanitizes_read_input_json_delta() {
let report_context = json!({
"provider_api_format": "claude:messages",
"client_api_format": "claude:messages",
"anthropic_compatibility_profile": "claude_code_legacy",
"needs_conversion": false,
});
let mut rewriter =
maybe_build_local_stream_rewriter(Some(&report_context)).expect("rewriter should exist");
let mut output = rewriter
.push_chunk(
b"event: content_block_start\n\
data: {\"type\":\"content_block_start\",\"index\":0,\"content_block\":{\"type\":\"tool_use\",\"id\":\"call_read_1\",\"name\":\"Read\",\"input\":{}}}\n\n",
)
.expect("start should be accepted");
output.extend(
rewriter
.push_chunk(
b"event: content_block_delta\n\
data: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"input_json_delta\",\"partial_json\":\"{\\\"file_path\\\":\\\"/tmp/a.txt\\\",\\\"pages\\\":\\\"\\\"}\"}}\n\n",
)
.expect("delta should be accepted"),
);
output.extend(
rewriter
.push_chunk(
b"event: content_block_stop\n\
data: {\"type\":\"content_block_stop\",\"index\":0}\n\n",
)
.expect("stop should flush sanitized delta"),
);
let output_text = utf8(output);
assert!(output_text.contains("\"name\":\"Read\""));
assert!(output_text.contains("\\\"file_path\\\":\\\"/tmp/a.txt\\\""));
assert!(!output_text.contains("\\\"pages\\\":\\\"\\\""));
}
#[test]
fn standard_sync_bridge_converts_openai_chat_sync_json_to_openai_chat_sse() {
let outcome = maybe_bridge_standard_sync_json_to_stream(
@@ -25,12 +25,16 @@ fn test_decision() -> GatewayControlDecision {
route_class: Some("ai_public".to_string()),
route_family: Some("openai".to_string()),
route_kind: Some("compact".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("openai:responses:compact".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
}
}
@@ -172,6 +176,9 @@ fn aggregates_openai_responses_stream_completed_event_to_final_response() {
let result = aggregate_openai_responses_stream_sync_response(body.as_bytes())
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -180,6 +187,9 @@ fn aggregates_openai_responses_stream_completed_event_to_final_response() {
"object": "response",
"model": "gpt-5",
"status": "completed",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Hello",
"output": [{
"type": "message",
"id": "resp_123_msg",
@@ -215,6 +225,9 @@ fn aggregates_openai_responses_stream_tool_call_events_to_final_response() {
let result = aggregate_openai_responses_stream_sync_response(body.as_bytes())
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -223,6 +236,9 @@ fn aggregates_openai_responses_stream_tool_call_events_to_final_response() {
"object": "response",
"model": "gpt-5",
"status": "completed",
"created_at": created_at,
"completed_at": created_at,
"output_text": "",
"output": [{
"type": "function_call",
"id": "call_123",
@@ -811,6 +827,9 @@ fn converts_claude_cli_response_to_openai_responses_response() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -819,6 +838,9 @@ fn converts_claude_cli_response_to_openai_responses_response() {
"object": "response",
"status": "completed",
"model": "claude-code-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Hello Claude CLI",
"output": [{
"type": "message",
"id": "msg_cli_123_msg",
@@ -868,6 +890,9 @@ fn converts_claude_cli_tool_use_to_openai_responses_function_call() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -876,6 +901,9 @@ fn converts_claude_cli_tool_use_to_openai_responses_function_call() {
"object": "response",
"status": "completed",
"model": "claude-code-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Running tool.",
"output": [
{
"type": "message",
@@ -933,6 +961,9 @@ fn converts_gemini_cli_response_to_openai_responses_response() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -941,6 +972,9 @@ fn converts_gemini_cli_response_to_openai_responses_response() {
"object": "response",
"status": "completed",
"model": "gemini-cli-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Hello Gemini CLI",
"output": [{
"type": "message",
"id": "resp_cli_123_msg",
@@ -995,6 +1029,9 @@ fn converts_gemini_cli_function_call_to_openai_responses_function_call() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -1003,6 +1040,9 @@ fn converts_gemini_cli_function_call_to_openai_responses_function_call() {
"object": "response",
"status": "completed",
"model": "gemini-cli-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Need a tool.",
"output": [
{
"type": "message",
@@ -1130,7 +1170,7 @@ fn local_finalize_handles_openai_responses_compact_cross_format_sync_response()
assert_eq!(report.report_kind, "openai_responses_compact_sync_success");
assert_eq!(
report.client_body_json.expect("client body should exist")["object"],
"response"
"response.compaction"
);
}
@@ -1186,7 +1226,7 @@ fn local_finalize_handles_openai_responses_compact_cross_format_function_call_re
.background_report
.expect("compact tool-call should downgrade to success report");
let client_body = report.client_body_json.expect("client body should exist");
assert_eq!(client_body["object"], "response");
assert_eq!(client_body["object"], "response.compaction");
assert_eq!(client_body["output"][1]["type"], "function_call");
}
@@ -1403,6 +1443,57 @@ fn local_finalize_handles_openai_responses_cross_format_stream_response_from_gem
);
}
#[test]
fn local_finalize_rejects_antigravity_usage_only_gemini_wrapper() {
let payload = GatewaySyncReportRequest {
trace_id: "trace-antigravity-empty-gemini-wrapper".to_string(),
report_kind: "gemini_chat_sync_finalize".to_string(),
report_context: Some(json!({
"client_api_format": "gemini:generate_content",
"provider_api_format": "gemini:generate_content",
"model": "gemini-3.5-flash",
"mapped_model": "gemini-3-flash-agent",
"needs_conversion": false,
"has_envelope": true,
"envelope_name": "antigravity:v1internal",
"upstream_is_stream": true,
})),
status_code: 200,
headers: BTreeMap::from([("content-type".to_string(), "application/json".to_string())]),
body_json: Some(json!({
"chunks": [{
"response": {
"responseId": "resp-usage-only",
"modelVersion": "gemini-3-flash-agent",
"usageMetadata": {
"promptTokenCount": 5528,
"totalTokenCount": 5528
}
},
"metadata": {},
"traceId": "trace-antigravity-empty-gemini-wrapper"
}],
"metadata": {
"stream": true,
"stored_chunks": 1,
"total_chunks": 1
}
})),
client_body_json: None,
body_base64: None,
telemetry: None,
};
let outcome = maybe_build_local_core_sync_finalize_response(
"trace-antigravity-empty-gemini-wrapper",
&test_decision(),
&payload,
)
.expect("local finalize should evaluate payload");
assert!(outcome.is_none());
}
#[test]
fn local_finalize_handles_openai_responses_compact_openai_family_stream_response_even_when_conversion_flagged(
) {
@@ -1702,6 +1793,93 @@ fn local_finalize_handles_openai_chat_cross_format_sync_response_from_openai_res
assert_eq!(client_body["usage"]["total_tokens"], 5);
}
#[test]
fn local_finalize_aggregates_openai_responses_capture_envelope_before_chat_conversion() {
let payload = GatewaySyncReportRequest {
trace_id: "trace-openai-chat-capture-envelope-sync-123".to_string(),
report_kind: "openai_chat_sync_finalize".to_string(),
report_context: Some(json!({
"client_api_format": "openai:chat",
"provider_api_format": "openai:responses",
"model": "gpt-5.6-luna",
"mapped_model": "gpt-5.6-luna",
"needs_conversion": true,
"has_envelope": false,
})),
status_code: 200,
headers: BTreeMap::from([("content-type".to_string(), "application/json".to_string())]),
body_json: Some(json!({
"chunks": [
{
"type": "response.output_text.delta",
"response_id": "resp_capture_gateway_123",
"output_index": 0,
"content_index": 0,
"delta": "Gateway "
},
{
"type": "response.output_text.done",
"response_id": "resp_capture_gateway_123",
"output_index": 0,
"content_index": 0,
"text": "Gateway capture"
},
{
"type": "response.completed",
"response": {
"id": "resp_capture_gateway_123",
"object": "response",
"status": "completed",
"model": "gpt-5.6-luna",
"output": [],
"usage": {
"input_tokens": 2,
"output_tokens": 3,
"total_tokens": 5
}
}
}
],
"metadata": {}
})),
client_body_json: None,
body_base64: None,
telemetry: None,
};
let outcome = maybe_build_local_core_sync_finalize_response(
"trace-openai-chat-capture-envelope-sync-123",
&test_decision(),
&payload,
)
.expect("capture envelope finalize should succeed")
.expect("capture envelope finalize should match");
let report = outcome
.background_report
.expect("capture envelope conversion should produce a success report");
assert_eq!(report.report_kind, "openai_chat_sync_success");
let provider_body = report
.body_json
.expect("aggregated provider body should exist");
assert_eq!(provider_body["id"], "resp_capture_gateway_123");
assert_eq!(
provider_body["output"][0]["content"][0]["text"],
"Gateway capture"
);
assert!(provider_body.get("chunks").is_none());
let client_body = report
.client_body_json
.expect("converted client body should exist");
assert_eq!(
client_body["choices"][0]["message"]["content"],
"Gateway capture"
);
assert_eq!(client_body["usage"]["prompt_tokens"], 2);
assert_eq!(client_body["usage"]["completion_tokens"], 3);
assert_eq!(client_body["usage"]["total_tokens"], 5);
}
#[test]
fn local_finalize_handles_claude_chat_cross_format_sync_response_from_openai_chat() {
let payload = GatewaySyncReportRequest {
@@ -1748,12 +1926,16 @@ fn local_finalize_handles_claude_chat_cross_format_sync_response_from_openai_cha
route_class: Some("ai_public".to_string()),
route_family: Some("claude".to_string()),
route_kind: Some("chat".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("claude:messages".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
},
&payload,
)
@@ -1815,12 +1997,16 @@ fn local_finalize_handles_gemini_cli_cross_format_sync_response_from_claude_cli(
route_class: Some("ai_public".to_string()),
route_family: Some("gemini".to_string()),
route_kind: Some("cli".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("gemini:generate_content".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
},
&payload,
)
+24 -11
View File
@@ -3,6 +3,7 @@ pub(crate) mod api;
mod finalize;
mod planner;
mod pure;
mod response_history;
pub(crate) mod transport;
use axum::body::Body;
@@ -13,7 +14,9 @@ use crate::{usage::GatewaySyncReportRequest, AppState, GatewayError};
pub(crate) use self::adaptation::{
maybe_build_provider_private_stream_normalizer, ProviderPrivateStreamNormalizer,
};
pub(crate) use self::api::gemini_generate_content_response_has_visible_output;
pub(crate) use self::api::{
gemini_generate_content_response_has_visible_output, CODEX_RESPONSES_LITE_HEADER,
};
pub(crate) use self::finalize::common::LocalCoreSyncFinalizeOutcome;
pub(crate) use self::finalize::internal::{
maybe_bridge_standard_sync_json_to_stream, maybe_build_stream_response_rewriter,
@@ -48,18 +51,24 @@ pub(crate) use self::planner::{
build_standard_family_stream_plan_and_reports, build_standard_family_sync_attempt_source,
build_standard_family_sync_plan_and_reports, build_standard_stream_plan_from_decision,
build_standard_sync_plan_from_decision, candidate_auth_channel_skip_reason,
extract_pool_sticky_session_token, maybe_build_stream_decision_payload,
maybe_build_stream_plan_payload, maybe_build_sync_decision_payload,
maybe_build_sync_plan_payload, planner_is_matching_stream_request, provider_key_pool_score_id,
provider_key_pool_score_scope, read_candidate_transport_snapshot,
record_local_runtime_candidate_skip_reason,
codex_model_capabilities_for_transport, extract_pool_sticky_session_token,
maybe_build_stream_decision_payload, maybe_build_stream_plan_payload,
maybe_build_sync_decision_payload, maybe_build_sync_plan_payload,
planner_is_matching_stream_request, provider_key_pool_score_id, provider_key_pool_score_scope,
read_candidate_transport_snapshot, record_local_runtime_candidate_skip_reason,
resolve_tunnel_scheduler_affinity_context, resolve_upstream_is_stream_for_provider,
set_local_openai_chat_execution_exhausted_diagnostic,
set_local_openai_image_execution_exhausted_diagnostic, CandidateFailureDiagnostic,
CandidateFailureDiagnosticKind, EligibleLocalExecutionCandidate, GatewayAuthApiKeySnapshot,
GatewayProviderTransportSnapshot, LocalExecutionAttemptSource, LocalExecutionCandidateKind,
LocalResolvedOAuthRequestAuth, PlannerAppState, SkippedLocalExecutionCandidate,
set_local_openai_image_execution_exhausted_diagnostic, validate_final_openai_provider_request,
CandidateFailureDiagnostic, CandidateFailureDiagnosticKind, EligibleLocalExecutionCandidate,
GatewayAuthApiKeySnapshot, GatewayProviderTransportSnapshot, LocalExecutionAttemptSource,
LocalExecutionCandidateKind, LocalResolvedOAuthRequestAuth, PlannerAppState,
SkippedLocalExecutionCandidate,
};
pub(crate) use self::pure::*;
pub(crate) use self::response_history::{
hydrate_openai_response_history, persist_converted_response_history,
persist_response_history_record,
};
pub(crate) use self::transport::{
append_transport_diagnostics_to_value, build_request_trace_proxy_value,
candidate_common_transport_skip_reason, candidate_transport_pair_skip_reason,
@@ -68,7 +77,7 @@ pub(crate) use self::transport::{
request_pair_allowed_for_transport, request_pair_direct_auth,
request_pair_transport_unsupported_reason, CandidateTransportPolicyFacts,
};
pub(crate) use crate::control::GatewayControlDecision;
pub(crate) use crate::control::{GatewayControlDecision, GatewayCredentialCarrier};
pub(crate) use crate::execution_runtime::{ConversionMode, ExecutionStrategy};
pub(crate) use crate::headers::RequestOrigin;
pub(crate) use aether_ai_serving::{
@@ -86,6 +95,7 @@ pub(crate) fn build_provider_transport_request_url(
upstream_is_stream: bool,
request_query: Option<&str>,
kiro_api_region: Option<&str>,
api_operation: Option<ApiOperation>,
) -> Option<String> {
self::transport::build_transport_request_url(
transport,
@@ -95,6 +105,7 @@ pub(crate) fn build_provider_transport_request_url(
upstream_is_stream,
request_query,
kiro_api_region,
api_operation,
},
)
}
@@ -106,6 +117,7 @@ pub(crate) fn build_provider_transport_request_url_for_request_body(
upstream_is_stream: bool,
request_query: Option<&str>,
kiro_api_region: Option<&str>,
api_operation: Option<ApiOperation>,
provider_request_body: Option<&serde_json::Value>,
) -> Option<String> {
self::transport::build_transport_request_url_for_request_body(
@@ -116,6 +128,7 @@ pub(crate) fn build_provider_transport_request_url_for_request_body(
upstream_is_stream,
request_query,
kiro_api_region,
api_operation,
},
provider_request_body,
)
@@ -0,0 +1,170 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use serde_json::Value;
use crate::ai_serving::transport::antigravity::{
build_antigravity_safe_v1internal_request, build_antigravity_static_identity_headers,
classify_local_antigravity_request_support, AntigravityEnvelopeRequestType,
AntigravityRequestAuth, AntigravityRequestAuthUnsupportedReason,
AntigravityRequestEnvelopeSupport, AntigravityRequestSideSupport,
AntigravityRequestSideUnsupportedReason,
};
use crate::ai_serving::transport::{
build_standard_provider_request_headers, GatewayProviderTransportSnapshot,
StandardProviderRequestHeaders, StandardProviderRequestHeadersInput,
};
use crate::AppState;
pub(crate) const ANTIGRAVITY_V1INTERNAL_ENVELOPE_NAME: &str = "antigravity:v1internal";
pub(crate) enum AntigravityV1InternalRequestError {
TransportUnsupported,
EnvelopeUnsupported,
UpstreamUrlUnavailable,
HeaderRulesApplyFailed,
}
pub(crate) struct AntigravityV1InternalRequestInput<'a> {
pub(crate) state: &'a AppState,
pub(crate) parts: &'a http::request::Parts,
pub(crate) transport: &'a Arc<GatewayProviderTransportSnapshot>,
pub(crate) trace_id: &'a str,
pub(crate) mapped_model: &'a str,
pub(crate) provider_api_format: &'a str,
pub(crate) auth_header: &'a str,
pub(crate) auth_value: &'a str,
pub(crate) request_headers: &'a http::HeaderMap,
pub(crate) original_request_body: &'a Value,
pub(crate) gemini_request_body: &'a Value,
pub(crate) upstream_is_stream: bool,
pub(crate) same_format: bool,
}
pub(crate) struct AntigravityV1InternalRequest {
pub(crate) transport: Arc<GatewayProviderTransportSnapshot>,
pub(crate) body: Value,
pub(crate) headers: StandardProviderRequestHeaders,
pub(crate) upstream_url: String,
}
pub(crate) async fn build_antigravity_v1internal_provider_request(
input: AntigravityV1InternalRequestInput<'_>,
) -> Result<AntigravityV1InternalRequest, AntigravityV1InternalRequestError> {
let payload = build_antigravity_v1internal_payload(
input.state,
input.transport,
input.trace_id,
input.mapped_model,
input.gemini_request_body,
)
.await?;
let upstream_url = crate::ai_serving::build_provider_transport_request_url_for_request_body(
&payload.transport,
input.provider_api_format,
Some(input.mapped_model),
input.upstream_is_stream,
input.parts.uri.query(),
None,
None,
Some(&payload.body),
)
.ok_or(AntigravityV1InternalRequestError::UpstreamUrlUnavailable)?;
let extra_headers: BTreeMap<String, String> =
build_antigravity_static_identity_headers(&payload.auth);
let mut headers =
build_standard_provider_request_headers(StandardProviderRequestHeadersInput {
transport: &payload.transport,
provider_api_format: input.provider_api_format,
same_format: input.same_format,
headers: input.request_headers,
auth_header: input.auth_header,
auth_value: input.auth_value,
extra_headers: &extra_headers,
header_rules: payload.transport.endpoint.header_rules.as_ref(),
provider_request_body: &payload.body,
original_request_body: input.original_request_body,
upstream_is_stream: input.upstream_is_stream,
})
.ok_or(AntigravityV1InternalRequestError::HeaderRulesApplyFailed)?;
headers
.headers
.insert("accept".to_string(), "text/event-stream".to_string());
Ok(AntigravityV1InternalRequest {
transport: payload.transport,
body: payload.body,
headers,
upstream_url,
})
}
struct AntigravityV1InternalPayload {
transport: Arc<GatewayProviderTransportSnapshot>,
auth: AntigravityRequestAuth,
body: Value,
}
async fn build_antigravity_v1internal_payload(
state: &AppState,
transport: &Arc<GatewayProviderTransportSnapshot>,
trace_id: &str,
mapped_model: &str,
gemini_request_body: &Value,
) -> Result<AntigravityV1InternalPayload, AntigravityV1InternalRequestError> {
let mut resolved_transport = Arc::clone(transport);
let mut antigravity_support = classify_local_antigravity_request_support(
&resolved_transport,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
);
if matches!(
antigravity_support,
AntigravityRequestSideSupport::Unsupported(
AntigravityRequestSideUnsupportedReason::UnsupportedAuth(
AntigravityRequestAuthUnsupportedReason::MissingProjectId
)
)
) {
if let Some(hydrated) = state
.hydrate_antigravity_project_metadata_for_transport(&resolved_transport)
.await
{
resolved_transport = Arc::new(hydrated);
antigravity_support = classify_local_antigravity_request_support(
&resolved_transport,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
);
}
}
let auth = match antigravity_support {
AntigravityRequestSideSupport::Supported(spec) => spec.auth,
AntigravityRequestSideSupport::Unsupported(_) => {
return Err(AntigravityV1InternalRequestError::TransportUnsupported);
}
};
let body = match build_antigravity_safe_v1internal_request(
&auth,
trace_id,
mapped_model,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
) {
AntigravityRequestEnvelopeSupport::Supported(envelope) => envelope,
AntigravityRequestEnvelopeSupport::Unsupported(_) => {
return Err(AntigravityV1InternalRequestError::EnvelopeUnsupported);
}
};
Ok(AntigravityV1InternalPayload {
transport: resolved_transport,
auth,
body,
})
}
@@ -1,6 +1,8 @@
use aether_routing_core::ResolvedRoutingPolicy;
use aether_scheduler_core::{
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session, ClientSessionAffinity,
SchedulerAffinityTarget, SchedulerMinimalCandidateSelectionCandidate,
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope,
ClientSessionAffinity, SchedulerAffinityScope, SchedulerAffinityTarget,
SchedulerMinimalCandidateSelectionCandidate,
};
use crate::ai_serving::{GatewayAuthApiKeySnapshot, PlannerAppState};
@@ -8,25 +10,38 @@ use crate::scheduler::affinity::SCHEDULER_AFFINITY_TTL;
const PLANNER_SCHEDULER_AFFINITY_MAX_ENTRIES: usize = 10_000;
pub(crate) fn has_explicit_session_affinity(
client_session_affinity: Option<&ClientSessionAffinity>,
) -> bool {
client_session_affinity.is_some_and(ClientSessionAffinity::has_session_key)
}
pub(crate) fn read_cached_scheduler_affinity_target(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: Option<&str>,
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Option<SchedulerAffinityTarget> {
if !has_explicit_session_affinity(client_session_affinity) {
return None;
}
let requested_model = requested_model
.map(str::trim)
.filter(|value| !value.is_empty())?;
let api_key_id = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())?;
let cache_key = build_scheduler_affinity_cache_key_for_api_key_id_with_client_session(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
)?;
let affinity_scope = scheduler_affinity_scope_for_routing_policy(routing_policy);
let cache_key =
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
affinity_scope.as_ref(),
)?;
state
.app()
@@ -41,6 +56,9 @@ pub(crate) fn remember_scheduler_affinity_for_candidate(
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) {
if !has_explicit_session_affinity(client_session_affinity) {
return;
}
remember_scheduler_affinity_for_candidate_at_epoch(
state,
auth_snapshot,
@@ -61,18 +79,71 @@ pub(crate) fn remember_scheduler_affinity_for_candidate_at_epoch(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
expected_epoch: Option<u64>,
) {
remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state,
auth_snapshot,
client_session_affinity,
client_api_format,
requested_model,
candidate,
None,
expected_epoch,
);
}
#[allow(clippy::too_many_arguments)]
pub(crate) fn remember_scheduler_affinity_for_candidate_with_routing_policy_at_epoch(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
routing_policy: Option<&ResolvedRoutingPolicy>,
expected_epoch: Option<u64>,
) {
let affinity_scope = scheduler_affinity_scope_for_routing_policy(routing_policy);
remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state,
auth_snapshot,
client_session_affinity,
client_api_format,
requested_model,
candidate,
affinity_scope.as_ref(),
expected_epoch,
);
}
#[allow(clippy::too_many_arguments)]
fn remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
affinity_scope: Option<&SchedulerAffinityScope>,
expected_epoch: Option<u64>,
) {
if !has_explicit_session_affinity(client_session_affinity) {
return;
}
let Some(api_key_id) = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())
else {
return;
};
let Some(cache_key) = build_scheduler_affinity_cache_key_for_api_key_id_with_client_session(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
) else {
let Some(cache_key) =
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
affinity_scope,
)
else {
return;
};
@@ -88,3 +159,15 @@ pub(crate) fn remember_scheduler_affinity_for_candidate_at_epoch(
expected_epoch,
);
}
fn scheduler_affinity_scope_for_routing_policy(
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Option<SchedulerAffinityScope> {
let policy = routing_policy?;
let group_id = policy
.group_id
.as_deref()
.map(str::trim)
.filter(|group_id| !group_id.is_empty())?;
Some(SchedulerAffinityScope::new(group_id, policy.group_version))
}
File diff suppressed because it is too large Load Diff
@@ -134,6 +134,7 @@ mod tests {
global_model_id: "global-1".to_string(),
global_model_name: "gpt-5.4".to_string(),
selected_provider_model_name: "gpt-5.4".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -198,6 +199,7 @@ mod tests {
}
}
})),
upstream_metadata: None,
decrypted_api_key: "sk-test".to_string(),
decrypted_auth_config: None,
},
@@ -254,6 +256,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "__placeholder__".to_string(),
decrypted_auth_config: None,
},
@@ -144,6 +144,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: String::new(),
decrypted_auth_config: None,
},
@@ -168,6 +169,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-test".to_string(),
selected_provider_model_name: "gpt-test-upstream".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -1,5 +1,3 @@
use std::collections::BTreeMap;
use aether_ai_serving::{
ai_ranking_context, build_ai_rankable_candidate, run_ai_candidate_ranking,
AiCandidateRankingPort, AiRankableCandidateParts, AiRankingContextConfig,
@@ -7,6 +5,7 @@ use aether_ai_serving::{
};
use aether_routing_core::{ResolvedRoutingPolicy, RoutingSchedulingMode, RoutingSetPriorityMode};
use async_trait::async_trait;
use tokio::sync::Mutex;
use tracing::warn;
use crate::ai_serving::{GatewayAuthApiKeySnapshot, PlannerAppState};
@@ -24,7 +23,7 @@ use aether_scheduler_core::{
use super::candidate_affinity_cache::read_cached_scheduler_affinity_target;
use super::candidate_resolution::{EligibleLocalExecutionCandidate, LocalExecutionCandidateKind};
use super::candidate_transport_ranking_facts::{
resolve_cached_transport_ranking_facts, CandidateTransportRankingFacts,
resolve_cached_transport_ranking_facts, CandidateTransportRankingFactsCache,
};
struct GatewayLocalCandidateRankingPort<'a> {
@@ -35,6 +34,7 @@ struct GatewayLocalCandidateRankingPort<'a> {
required_capabilities: Option<&'a serde_json::Value>,
ordering_config: SchedulerOrderingConfig,
routing_policy: Option<&'a ResolvedRoutingPolicy>,
transport_ranking_facts_cache: Mutex<CandidateTransportRankingFactsCache>,
}
#[async_trait]
@@ -66,6 +66,7 @@ impl AiCandidateRankingPort for GatewayLocalCandidateRankingPort<'_> {
self.client_session_affinity,
normalized_client_api_format,
affinity_requested_model,
self.routing_policy,
))
}
@@ -84,13 +85,17 @@ impl AiCandidateRankingPort for GatewayLocalCandidateRankingPort<'_> {
normalized_client_api_format: &str,
cached_affinity_match: bool,
) -> Result<SchedulerRankableCandidate, Self::Error> {
let ranking_facts = resolve_transport_ranking_facts_for_candidate(
self.state,
&candidate.candidate,
candidate.transport.as_ref(),
self.ordering_config,
)
.await;
let ranking_facts = {
let mut cache = self.transport_ranking_facts_cache.lock().await;
resolve_cached_transport_ranking_facts(
self.state,
&mut cache,
&candidate.candidate,
candidate.transport.as_ref(),
self.ordering_config,
)
.await
};
let routing_overlaid_candidate =
routing_overlaid_candidate(self.routing_policy, candidate.kind, &candidate.candidate);
Ok(build_ai_rankable_candidate(AiRankableCandidateParts {
@@ -137,6 +142,7 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
required_capabilities,
ordering_config,
routing_policy,
transport_ranking_facts_cache: Mutex::new(CandidateTransportRankingFactsCache::default()),
};
match run_ai_candidate_ranking(&port, candidates, normalized_client_api_format).await {
@@ -145,23 +151,6 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
}
}
async fn resolve_transport_ranking_facts_for_candidate(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &crate::ai_serving::GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let mut ordering_cache = BTreeMap::new();
resolve_cached_transport_ranking_facts(
state,
&mut ordering_cache,
candidate,
transport,
ordering_config,
)
.await
}
fn cached_affinity_matches_local_execution_scope(
eligible: &EligibleLocalExecutionCandidate,
target: &SchedulerAffinityTarget,
@@ -233,17 +222,21 @@ fn routing_overlaid_candidate(
let mut overlaid = candidate.clone();
overlaid.provider_priority = policy
.ranking_overlay
.provider_priority_or_unspecified(candidate.provider_id.as_str());
.provider_priority(candidate.provider_id.as_str(), candidate.provider_priority);
let overlaid_key_priority = match kind {
LocalExecutionCandidateKind::SingleKey => policy
.ranking_overlay
.key_priority_or_unspecified(candidate.key_id.as_str()),
.key_priority_overrides
.get(candidate.key_id.as_str()),
LocalExecutionCandidateKind::PoolGroup => policy
.ranking_overlay
.pool_priority_or_unspecified(candidate.provider_id.as_str()),
.pool_priority_overrides
.get(candidate.provider_id.as_str()),
};
overlaid.key_internal_priority = overlaid_key_priority;
overlaid.key_global_priority_for_format = Some(overlaid_key_priority);
if let Some(overlaid_key_priority) = overlaid_key_priority.copied() {
overlaid.key_internal_priority = overlaid_key_priority;
overlaid.key_global_priority_for_format = Some(overlaid_key_priority);
}
overlaid
}
@@ -284,7 +277,9 @@ mod tests {
use serde_json::json;
use super::super::candidate_affinity_cache::remember_scheduler_affinity_for_candidate;
use super::super::candidate_transport_ranking_facts::resolve_cached_candidate_transport_ranking_facts;
use super::super::candidate_transport_ranking_facts::{
resolve_cached_candidate_transport_ranking_facts, CandidateTransportRankingFactsCache,
};
use super::{PlannerAppState, SchedulerMinimalCandidateSelectionCandidate};
use crate::ai_serving::planner::candidate_resolution::{
resolve_and_rank_local_execution_candidates,
@@ -306,7 +301,7 @@ mod tests {
let ordering_config = super::read_scheduler_ordering_config_or_default(state).await;
let mut candidates = candidates;
let mut rankables = Vec::with_capacity(candidates.len());
let mut ordering_cache = BTreeMap::new();
let mut ordering_cache = CandidateTransportRankingFactsCache::default();
for (original_index, candidate) in candidates.iter().enumerate() {
let ranking_facts = resolve_cached_candidate_transport_ranking_facts(
@@ -358,12 +353,13 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-4.1".to_string(),
selected_provider_model_name: "gpt-4.1".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
#[test]
fn routing_policy_priorities_do_not_fall_back_to_candidate_priorities() {
fn routing_policy_priorities_fall_back_to_candidate_priorities() {
let mut candidate = sample_candidate("endpoint-1", "key-1");
candidate.provider_priority = 7;
candidate.key_internal_priority = 3;
@@ -389,18 +385,9 @@ mod tests {
&candidate,
);
assert_eq!(
overlaid.provider_priority,
aether_routing_core::ROUTING_PRIORITY_UNSPECIFIED
);
assert_eq!(
overlaid.key_internal_priority,
aether_routing_core::ROUTING_PRIORITY_UNSPECIFIED
);
assert_eq!(
overlaid.key_global_priority_for_format,
Some(aether_routing_core::ROUTING_PRIORITY_UNSPECIFIED)
);
assert_eq!(overlaid.provider_priority, 7);
assert_eq!(overlaid.key_internal_priority, 3);
assert_eq!(overlaid.key_global_priority_for_format, Some(2));
}
#[test]
@@ -634,6 +621,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-4.1".to_string(),
selected_provider_model_name: "gpt-4.1".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -1540,6 +1528,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-cached",
"endpoint-cached",
@@ -1551,7 +1540,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1573,7 +1562,7 @@ mod tests {
"openai:chat",
"gpt-4.1",
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -1714,6 +1703,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_cross_format = sample_priority_candidate(
"provider-shared",
"endpoint-openai",
@@ -1725,7 +1715,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"claude:messages",
"gpt-4.1",
&cached_cross_format,
@@ -1747,7 +1737,7 @@ mod tests {
"claude:messages",
"gpt-4.1",
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -1915,6 +1905,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-pool",
"endpoint-pool",
@@ -1926,7 +1917,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1948,7 +1939,7 @@ mod tests {
"openai:chat",
Some("gpt-4.1"),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -2008,6 +1999,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-pool",
"endpoint-pool",
@@ -2019,7 +2011,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -2041,7 +2033,7 @@ mod tests {
"openai:chat",
Some("gpt-4.1"),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
None,
None,
None,
@@ -2064,7 +2056,7 @@ mod tests {
}
#[tokio::test]
async fn remembers_scheduler_affinity_for_candidate_using_requested_model_key() {
async fn ignores_scheduler_affinity_without_client_session_scope() {
let state = AppState::new().expect("state should build");
let auth_snapshot = sample_auth_snapshot();
let candidate = sample_candidate("endpoint-1", "key-1");
@@ -2078,15 +2070,12 @@ mod tests {
&candidate,
);
let remembered = state
assert!(state
.read_scheduler_affinity_target(
"scheduler_affinity:api-key-1:openai:chat:gpt-5",
SCHEDULER_AFFINITY_TTL,
)
.expect("affinity target should be cached");
assert_eq!(remembered.provider_id, "provider-1");
assert_eq!(remembered.endpoint_id, "endpoint-1");
assert_eq!(remembered.key_id, "key-1");
.is_none());
}
#[tokio::test]
@@ -7,6 +7,7 @@ use aether_ai_serving::{
use aether_routing_core::ResolvedRoutingPolicy;
use async_trait::async_trait;
use std::convert::Infallible;
use std::time::Instant;
use tracing::warn;
use aether_scheduler_core::{
@@ -20,6 +21,7 @@ use crate::ai_serving::{
PlannerAppState,
};
use crate::orchestration::LocalExecutionCandidateMetadata;
use crate::stage_metrics::observe_gateway_stage_ms;
use super::candidate_ranking::rank_eligible_local_execution_candidates;
@@ -68,7 +70,7 @@ struct GatewayLocalCandidateResolutionPort<'a> {
#[async_trait]
impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
type Candidate = SchedulerMinimalCandidateSelectionCandidate;
type Transport = GatewayProviderTransportSnapshot;
type Transport = Arc<GatewayProviderTransportSnapshot>;
type Eligible = EligibleLocalExecutionCandidate;
type Skipped = SkippedLocalExecutionCandidate;
type Error = Infallible;
@@ -77,7 +79,12 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
&self,
candidate: &Self::Candidate,
) -> Result<Option<Self::Transport>, Self::Error> {
Ok(read_candidate_transport_snapshot(self.state, candidate).await)
let started_at = Instant::now();
let transport = read_candidate_transport_snapshot_arc(self.state, candidate).await;
let elapsed_ms = started_at.elapsed().as_millis() as u64;
observe_gateway_stage_ms("candidate_transport_snapshot", elapsed_ms);
observe_gateway_stage_ms("candidate_resolution_transport_read", elapsed_ms);
Ok(transport)
}
fn build_missing_transport_skipped_candidate(
@@ -139,7 +146,7 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
SkippedLocalExecutionCandidate {
candidate,
skip_reason,
transport: Some(Arc::new(transport)),
transport: Some(transport),
ranking: None,
extra_data: None,
}
@@ -159,7 +166,7 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
EligibleLocalExecutionCandidate {
kind,
candidate,
transport: Arc::new(transport),
transport,
provider_api_format,
orchestration: LocalExecutionCandidateMetadata::default(),
ranking: None,
@@ -171,7 +178,8 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
candidates: Vec<Self::Eligible>,
normalized_client_api_format: &str,
) -> Result<Vec<Self::Eligible>, Self::Error> {
Ok(rank_eligible_local_execution_candidates(
let started_at = Instant::now();
let ranked = rank_eligible_local_execution_candidates(
self.state,
candidates,
normalized_client_api_format,
@@ -181,7 +189,12 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
self.required_capabilities,
self.routing_policy,
)
.await)
.await;
observe_gateway_stage_ms(
"candidate_resolution_rank",
started_at.elapsed().as_millis() as u64,
);
Ok(ranked)
}
async fn apply_pool_scheduler(
@@ -358,8 +371,13 @@ async fn resolve_and_rank_local_execution_candidates_with_pool_expansion(
expand_pool_groups,
};
let started_at = Instant::now();
match run_ai_candidate_resolution(&port, candidates, request).await {
Ok(mut outcome) => {
observe_gateway_stage_ms(
"candidate_resolution_core",
started_at.elapsed().as_millis() as u64,
);
for candidate in &mut outcome.eligible_candidates {
candidate.orchestration.scheduler_affinity_epoch = Some(scheduler_affinity_epoch);
}
@@ -506,8 +524,17 @@ pub(crate) async fn read_candidate_transport_snapshot(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> Option<GatewayProviderTransportSnapshot> {
read_candidate_transport_snapshot_arc(state, candidate)
.await
.map(|transport| (*transport).clone())
}
pub(crate) async fn read_candidate_transport_snapshot_arc(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> Option<Arc<GatewayProviderTransportSnapshot>> {
match state
.read_provider_transport_snapshot(
.read_provider_transport_snapshot_arc(
&candidate.provider_id,
&candidate.endpoint_id,
&candidate.key_id,
@@ -591,6 +618,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "secret".to_string(),
decrypted_auth_config: None,
},
@@ -615,6 +643,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "claude-sonnet".to_string(),
selected_provider_model_name: "claude-sonnet".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
File diff suppressed because it is too large Load Diff
@@ -1,8 +1,10 @@
use std::collections::BTreeMap;
use aether_contracts::ProxySnapshot;
use aether_scheduler_core::{
SchedulerMinimalCandidateSelectionCandidate, SchedulerTunnelAffinityBucket,
};
use serde_json::Value;
use tracing::warn;
use crate::ai_serving::{GatewayProviderTransportSnapshot, PlannerAppState};
@@ -10,7 +12,9 @@ use crate::scheduler::config::SchedulerOrderingConfig;
use super::candidate_resolution::read_candidate_transport_snapshot;
pub(super) type CandidateTransportIdentity<'a> = (&'a str, &'a str, &'a str);
const TUNNEL_OWNER_INSTANCE_ID_EXTRA_KEY: &str = "tunnel_owner_instance_id";
pub(super) type CandidateTransportIdentity = (String, String, String);
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(super) struct CandidateTransportRankingFacts {
@@ -18,43 +22,68 @@ pub(super) struct CandidateTransportRankingFacts {
pub(super) keep_priority_on_conversion: bool,
}
pub(super) async fn resolve_cached_candidate_transport_ranking_facts<'a>(
state: PlannerAppState<'_>,
cache: &mut BTreeMap<CandidateTransportIdentity<'a>, CandidateTransportRankingFacts>,
candidate: &'a SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.get(&identity).copied() {
return facts;
}
let facts = resolve_candidate_transport_ranking_facts(state, candidate, ordering_config).await;
cache.insert(identity, facts);
facts
#[derive(Debug, Default)]
pub(super) struct CandidateTransportRankingFactsCache {
candidate_facts: BTreeMap<CandidateTransportIdentity, CandidateTransportRankingFacts>,
configured_proxy_snapshots: BTreeMap<String, Option<ProxySnapshot>>,
system_proxy_snapshot: Option<Option<ProxySnapshot>>,
tunnel_buckets_by_node_id: BTreeMap<String, SchedulerTunnelAffinityBucket>,
}
pub(super) async fn resolve_cached_transport_ranking_facts<'a>(
pub(super) async fn resolve_cached_candidate_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut BTreeMap<CandidateTransportIdentity<'a>, CandidateTransportRankingFacts>,
candidate: &'a SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.get(&identity).copied() {
if let Some(facts) = cache.candidate_facts.get(&identity).copied() {
return facts;
}
let facts =
resolve_candidate_transport_ranking_facts_from_transport(state, transport, ordering_config)
.await;
cache.insert(identity, facts);
resolve_candidate_transport_ranking_facts(state, cache, candidate, ordering_config).await;
cache.candidate_facts.insert(identity, facts);
facts
}
pub(super) async fn resolve_cached_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.candidate_facts.get(&identity).copied() {
return facts;
}
let facts = resolve_candidate_transport_ranking_facts_from_transport(
state,
cache,
transport,
ordering_config,
)
.await;
cache.candidate_facts.insert(identity, facts);
facts
}
pub(super) async fn candidate_keeps_priority_on_conversion(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> bool {
let mut cache = CandidateTransportRankingFactsCache::default();
resolve_candidate_transport_ranking_facts(state, &mut cache, candidate, ordering_config)
.await
.keep_priority_on_conversion
}
async fn resolve_candidate_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
@@ -65,17 +94,23 @@ async fn resolve_candidate_transport_ranking_facts(
};
};
resolve_candidate_transport_ranking_facts_from_transport(state, &transport, ordering_config)
.await
resolve_candidate_transport_ranking_facts_from_transport(
state,
cache,
&transport,
ordering_config,
)
.await
}
async fn resolve_candidate_transport_ranking_facts_from_transport(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
CandidateTransportRankingFacts {
tunnel_bucket: resolve_tunnel_owner_affinity_from_transport(state, transport).await,
tunnel_bucket: resolve_tunnel_owner_affinity_from_transport(state, cache, transport).await,
keep_priority_on_conversion: ordering_config.keep_priority_on_conversion
|| transport.provider.keep_priority_on_conversion,
}
@@ -83,12 +118,11 @@ async fn resolve_candidate_transport_ranking_facts_from_transport(
async fn resolve_tunnel_owner_affinity_from_transport(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
) -> SchedulerTunnelAffinityBucket {
let Some(proxy) = state
.app()
.resolve_transport_proxy_snapshot_with_tunnel_affinity(transport)
.await
let Some(proxy) =
resolve_transport_proxy_snapshot_with_tunnel_affinity_cached(state, cache, transport).await
else {
return SchedulerTunnelAffinityBucket::Neutral;
};
@@ -104,10 +138,74 @@ async fn resolve_tunnel_owner_affinity_from_transport(
return SchedulerTunnelAffinityBucket::Neutral;
};
if let Some(bucket) = cache.tunnel_buckets_by_node_id.get(node_id).copied() {
return bucket;
}
let bucket = resolve_tunnel_owner_affinity_from_proxy(state, &proxy, node_id).await;
cache
.tunnel_buckets_by_node_id
.insert(node_id.to_string(), bucket);
bucket
}
async fn resolve_transport_proxy_snapshot_with_tunnel_affinity_cached(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
) -> Option<ProxySnapshot> {
for raw in [
transport.key.proxy.as_ref(),
transport.endpoint.proxy.as_ref(),
transport.provider.proxy.as_ref(),
]
.into_iter()
.flatten()
{
let cache_key = proxy_config_cache_key(raw);
if let Some(snapshot) = cache.configured_proxy_snapshots.get(&cache_key) {
if snapshot.is_some() {
return snapshot.clone();
}
continue;
}
let snapshot = state
.app()
.resolve_configured_proxy_snapshot_with_tunnel_affinity(Some(raw))
.await;
cache
.configured_proxy_snapshots
.insert(cache_key, snapshot.clone());
if snapshot.is_some() {
return snapshot;
}
}
if let Some(snapshot) = cache.system_proxy_snapshot.as_ref() {
return snapshot.clone();
}
let snapshot = state.app().resolve_system_proxy_snapshot().await;
cache.system_proxy_snapshot = Some(snapshot.clone());
snapshot
}
async fn resolve_tunnel_owner_affinity_from_proxy(
state: PlannerAppState<'_>,
proxy: &ProxySnapshot,
node_id: &str,
) -> SchedulerTunnelAffinityBucket {
if state.app().tunnel.has_local_proxy(node_id) {
return SchedulerTunnelAffinityBucket::LocalTunnel;
}
if let Some(owner_instance_id) = proxy_tunnel_owner_instance_id(proxy) {
return if owner_instance_id == state.app().tunnel.local_instance_id() {
SchedulerTunnelAffinityBucket::LocalTunnel
} else {
SchedulerTunnelAffinityBucket::RemoteTunnel
};
}
match state
.app()
.tunnel
@@ -134,10 +232,25 @@ async fn resolve_tunnel_owner_affinity_from_transport(
fn candidate_transport_identity(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> CandidateTransportIdentity<'_> {
) -> CandidateTransportIdentity {
(
candidate.provider_id.as_str(),
candidate.endpoint_id.as_str(),
candidate.key_id.as_str(),
candidate.provider_id.clone(),
candidate.endpoint_id.clone(),
candidate.key_id.clone(),
)
}
fn proxy_config_cache_key(raw: &Value) -> String {
serde_json::to_string(raw).unwrap_or_else(|_| raw.to_string())
}
fn proxy_tunnel_owner_instance_id(proxy: &ProxySnapshot) -> Option<&str> {
proxy
.extra
.as_ref()
.and_then(Value::as_object)
.and_then(|extra| extra.get(TUNNEL_OWNER_INSTANCE_ID_EXTRA_KEY))
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
}
@@ -1,15 +1,16 @@
use axum::body::Bytes;
use crate::ai_serving::is_json_request;
use crate::ai_serving::{
endpoint_config_forces_upstream_stream_policy as endpoint_config_forces_upstream_stream_policy_impl,
enforce_request_body_stream_field as enforce_request_body_stream_field_impl,
force_upstream_streaming_for_provider as force_upstream_streaming_for_provider_impl,
is_json_request, parse_direct_request_body as parse_direct_request_body_impl,
resolve_upstream_is_stream_from_endpoint_config as resolve_upstream_is_stream_from_endpoint_config_impl,
parse_direct_request_body as parse_direct_request_body_impl,
resolve_format_upstream_is_stream_for_provider as resolve_upstream_is_stream_for_provider_impl,
};
pub(crate) use crate::ai_serving::{
CLAUDE_CHAT_STREAM_PLAN_KIND, CLAUDE_CHAT_SYNC_PLAN_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, EXECUTION_RUNTIME_STREAM_ACTION,
CLAUDE_CLI_SYNC_PLAN_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND, EXECUTION_RUNTIME_STREAM_ACTION,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
@@ -20,9 +21,10 @@ pub(crate) use crate::ai_serving::{
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND, OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_SEARCH_SYNC_PLAN_KIND,
OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_VIDEO_CONTENT_PLAN_KIND,
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_serving::AiRequestedModelFamily as RequestedModelFamily;
@@ -54,10 +56,10 @@ pub(crate) fn resolve_upstream_is_stream_for_provider(
client_is_stream: bool,
hard_requires_streaming: bool,
) -> bool {
let hard_requires_streaming = hard_requires_streaming
|| force_upstream_streaming_for_provider(provider_type, provider_api_format);
resolve_upstream_is_stream_from_endpoint_config_impl(
resolve_upstream_is_stream_for_provider_impl(
endpoint_config,
provider_type,
provider_api_format,
client_is_stream,
hard_requires_streaming,
)
@@ -178,6 +180,27 @@ mod tests {
true,
false,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"codex",
"openai:image",
true,
true,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"codex",
"openai:responses:compact",
true,
true,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"custom",
"openai:responses:compact",
true,
true,
));
}
#[test]
@@ -1,14 +1,15 @@
use crate::ai_serving::planner::common::{
CLAUDE_CHAT_STREAM_PLAN_KIND, CLAUDE_CHAT_SYNC_PLAN_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, GEMINI_CHAT_STREAM_PLAN_KIND, GEMINI_CHAT_SYNC_PLAN_KIND,
GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND, GEMINI_EMBEDDING_SYNC_PLAN_KIND,
GEMINI_FILES_DELETE_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND,
GEMINI_FILES_LIST_PLAN_KIND, GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND,
GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_SYNC_PLAN_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DELETE_PLAN_KIND,
GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND, GEMINI_FILES_LIST_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_SYNC_PLAN_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_RERANK_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND, OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND,
OPENAI_RESPONSES_STREAM_PLAN_KIND, OPENAI_RESPONSES_SYNC_PLAN_KIND,
OPENAI_SEARCH_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND, OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
@@ -100,12 +101,15 @@ fn build_sync_plan_payload_from_decision(
OPENAI_RESPONSES_SYNC_PLAN_KIND => {
build_openai_responses_sync_plan_from_decision(parts, body_json, payload, false)?
}
OPENAI_IMAGE_SYNC_PLAN_KIND => build_passthrough_sync_plan_from_decision(parts, payload)?,
OPENAI_IMAGE_SYNC_PLAN_KIND | OPENAI_SEARCH_SYNC_PLAN_KIND => {
build_passthrough_sync_plan_from_decision(parts, payload)?
}
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND => {
build_openai_responses_sync_plan_from_decision(parts, body_json, payload, true)?
}
CLAUDE_CHAT_SYNC_PLAN_KIND
| CLAUDE_CLI_SYNC_PLAN_KIND
| CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND
| OPENAI_EMBEDDING_SYNC_PLAN_KIND
| OPENAI_RERANK_SYNC_PLAN_KIND => {
build_standard_sync_plan_from_decision(parts, body_json, payload)?
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,146 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use serde_json::Value;
use crate::ai_serving::transport::{
build_gemini_cli_v1internal_request, build_standard_provider_request_headers,
GatewayProviderTransportSnapshot, GeminiCliRequestAuth, GeminiCliRequestAuthSupport,
GeminiCliRequestEnvelopeSupport, StandardProviderRequestHeaders,
StandardProviderRequestHeadersInput, GEMINI_CLI_USER_AGENT,
};
use crate::AppState;
pub(crate) enum GeminiCliV1InternalRequestError {
ProjectUnavailable,
EnvelopeUnsupported,
UpstreamUrlUnavailable,
HeaderRulesApplyFailed,
}
pub(crate) struct GeminiCliV1InternalRequestInput<'a> {
pub(crate) state: &'a AppState,
pub(crate) parts: &'a http::request::Parts,
pub(crate) transport: &'a Arc<GatewayProviderTransportSnapshot>,
pub(crate) trace_id: &'a str,
pub(crate) mapped_model: &'a str,
pub(crate) provider_api_format: &'a str,
pub(crate) auth_header: &'a str,
pub(crate) auth_value: &'a str,
pub(crate) request_headers: &'a http::HeaderMap,
pub(crate) original_request_body: &'a Value,
pub(crate) gemini_request_body: &'a Value,
pub(crate) upstream_is_stream: bool,
}
pub(crate) struct GeminiCliV1InternalRequest {
pub(crate) transport: Arc<GatewayProviderTransportSnapshot>,
pub(crate) body: Value,
pub(crate) headers: StandardProviderRequestHeaders,
pub(crate) upstream_url: String,
}
pub(crate) async fn build_gemini_cli_v1internal_provider_request(
input: GeminiCliV1InternalRequestInput<'_>,
) -> Result<GeminiCliV1InternalRequest, GeminiCliV1InternalRequestError> {
let payload = build_gemini_cli_v1internal_payload(
input.state,
input.transport,
input.trace_id,
input.mapped_model,
input.gemini_request_body,
)
.await?;
let upstream_url = crate::ai_serving::build_provider_transport_request_url_for_request_body(
&payload.transport,
input.provider_api_format,
Some(input.mapped_model),
input.upstream_is_stream,
input.parts.uri.query(),
None,
None,
Some(&payload.body),
)
.ok_or(GeminiCliV1InternalRequestError::UpstreamUrlUnavailable)?;
let extra_headers =
BTreeMap::from([("user-agent".to_string(), GEMINI_CLI_USER_AGENT.to_string())]);
let headers = build_standard_provider_request_headers(StandardProviderRequestHeadersInput {
transport: &payload.transport,
provider_api_format: input.provider_api_format,
same_format: false,
headers: input.request_headers,
auth_header: input.auth_header,
auth_value: input.auth_value,
extra_headers: &extra_headers,
header_rules: payload.transport.endpoint.header_rules.as_ref(),
provider_request_body: &payload.body,
original_request_body: input.original_request_body,
upstream_is_stream: input.upstream_is_stream,
})
.ok_or(GeminiCliV1InternalRequestError::HeaderRulesApplyFailed)?;
Ok(GeminiCliV1InternalRequest {
transport: payload.transport,
body: payload.body,
headers,
upstream_url,
})
}
struct GeminiCliV1InternalPayload {
transport: Arc<GatewayProviderTransportSnapshot>,
body: Value,
}
async fn build_gemini_cli_v1internal_payload(
state: &AppState,
transport: &Arc<GatewayProviderTransportSnapshot>,
trace_id: &str,
mapped_model: &str,
gemini_request_body: &Value,
) -> Result<GeminiCliV1InternalPayload, GeminiCliV1InternalRequestError> {
let mut resolved_transport = Arc::clone(transport);
let mut auth = match crate::ai_serving::transport::resolve_local_gemini_cli_request_auth(
&resolved_transport,
) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => {
return Err(GeminiCliV1InternalRequestError::ProjectUnavailable);
}
};
if auth.project_id.is_none() {
auth = match state
.hydrate_gemini_cli_project_metadata_for_transport(&resolved_transport)
.await
{
Some(hydrated) => {
resolved_transport = Arc::new(hydrated);
match crate::ai_serving::transport::resolve_local_gemini_cli_request_auth(
&resolved_transport,
) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => GeminiCliRequestAuth::default(),
}
}
None => GeminiCliRequestAuth::default(),
};
}
let body = match build_gemini_cli_v1internal_request(
&auth,
trace_id,
mapped_model,
gemini_request_body,
) {
GeminiCliRequestEnvelopeSupport::Supported(envelope) => envelope,
GeminiCliRequestEnvelopeSupport::Unsupported(_) => {
return Err(GeminiCliV1InternalRequestError::EnvelopeUnsupported);
}
};
Ok(GeminiCliV1InternalPayload {
transport: resolved_transport,
body,
})
}
@@ -1,6 +1,7 @@
use crate::ai_serving::{AiExecutionDecision, AiExecutionPlanPayload, GatewayControlDecision};
use crate::{AppState, GatewayError};
mod antigravity;
mod candidate_affinity_cache;
mod candidate_materialization;
mod candidate_metadata;
@@ -12,12 +13,15 @@ mod candidate_transport_ranking_facts;
mod common;
mod decision;
mod decision_input;
mod gemini_cli;
mod materialization_policy;
mod passthrough;
mod plan_builders;
mod pool_scheduler;
pub(crate) mod pool_scores;
mod redaction;
mod report_context;
mod request_gzip;
mod route;
mod runtime_miss;
mod spec_metadata;
@@ -30,6 +34,7 @@ pub(crate) use self::candidate_resolution::{
candidate_auth_channel_skip_reason, read_candidate_transport_snapshot,
EligibleLocalExecutionCandidate, LocalExecutionCandidateKind, SkippedLocalExecutionCandidate,
};
pub(crate) use self::common::resolve_upstream_is_stream_for_provider;
pub(crate) use self::passthrough::{
build_local_same_format_stream_attempt_source, build_local_same_format_stream_plan_and_reports,
build_local_same_format_sync_attempt_source, build_local_same_format_sync_plan_and_reports,
@@ -44,6 +49,7 @@ pub(crate) use self::plan_builders::{
pub(crate) use self::pool_scores::{
build_provider_key_pool_score_upsert, provider_key_pool_score_id, provider_key_pool_score_scope,
};
pub(crate) use self::request_gzip::resolve_transport_request_encoding_policy;
pub(crate) use self::route::is_matching_stream_request as planner_is_matching_stream_request;
pub(crate) use self::runtime_miss::{
apply_local_runtime_candidate_terminal_reason, record_local_runtime_candidate_skip_reason,
@@ -74,7 +80,8 @@ pub(crate) use self::standard::{
build_local_stream_plan_and_reports as build_standard_family_stream_plan_and_reports,
build_local_sync_attempt_source as build_standard_family_sync_attempt_source,
build_local_sync_plan_and_reports as build_standard_family_sync_plan_and_reports,
set_local_openai_chat_execution_exhausted_diagnostic,
codex_model_capabilities_for_transport, set_local_openai_chat_execution_exhausted_diagnostic,
validate_final_openai_provider_request,
};
pub(crate) use self::state::{
GatewayAuthApiKeySnapshot, GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
@@ -86,6 +93,71 @@ pub(crate) use aether_ai_serving::{
CandidateFailureDiagnostic, CandidateFailureDiagnosticKind,
};
pub(crate) struct ResolvedTunnelSchedulerAffinityContext {
pub(crate) requested_model: String,
pub(crate) client_session_affinity: Option<aether_scheduler_core::ClientSessionAffinity>,
pub(crate) policy_context: Option<crate::scheduler::affinity::SchedulerAffinityPolicyContext>,
pub(crate) routing_overlay: Option<aether_routing_core::RankingOverlay>,
}
pub(crate) async fn resolve_tunnel_scheduler_affinity_context(
state: &AppState,
parts: &http::request::Parts,
decision: &GatewayControlDecision,
requested_model: String,
body_json: &serde_json::Value,
client_api_format: &str,
) -> Result<Option<ResolvedTunnelSchedulerAffinityContext>, GatewayError> {
let Some(auth_context) = decision.auth_context.as_ref() else {
return Ok(None);
};
let execution_auth_context =
crate::ai_serving::build_execution_runtime_auth_context(auth_context);
let Some(auth_snapshot) = state
.read_cached_auth_api_key_snapshot(
&execution_auth_context.user_id,
&execution_auth_context.api_key_id,
crate::clock::current_unix_secs(),
)
.await?
else {
return Ok(None);
};
let resolved_auth_input = decision_input::ResolvedLocalDecisionAuthInput {
auth_context: execution_auth_context,
auth_snapshot,
required_capabilities: None,
model_directive_policy: decision.model_directive_policy.clone(),
};
let mut input = decision_input::build_local_requested_model_decision_input(
resolved_auth_input,
requested_model,
);
decision_input::attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
client_api_format,
)
.await?;
let policy_context = input
.routing_policy
.as_ref()
.map(crate::scheduler::affinity::SchedulerAffinityPolicyContext::from_routing_policy);
let routing_overlay = input
.routing_policy
.as_ref()
.map(|policy| policy.ranking_overlay.clone());
Ok(Some(ResolvedTunnelSchedulerAffinityContext {
requested_model: input.requested_model,
client_session_affinity: input.client_session_affinity,
policy_context,
routing_overlay,
}))
}
pub(crate) async fn maybe_build_sync_decision_payload(
state: &AppState,
parts: &http::request::Parts,
@@ -71,6 +71,7 @@ pub(crate) fn build_passthrough_stream_plan_from_decision(
.content_type
.take()
.or_else(|| provider_request_headers.get("content-type").cloned());
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -84,7 +85,7 @@ pub(crate) fn build_passthrough_stream_plan_from_decision(
body_bytes_b64: None,
body_ref: None,
},
stream: true,
stream,
},
);
@@ -66,7 +66,7 @@ pub(crate) async fn maybe_build_sync_local_same_format_provider_decision_payload
candidate_count,
);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) =
maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
@@ -134,7 +134,7 @@ pub(crate) async fn maybe_build_stream_local_same_format_provider_decision_paylo
candidate_count,
);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) =
maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
@@ -24,7 +24,7 @@ use crate::ai_serving::{
ai_local_execution_contract_for_formats, extract_pool_sticky_session_token,
resolve_local_decision_execution_runtime_auth_context, GatewayControlDecision, PlannerAppState,
};
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::client_session_affinity::client_session_affinity_from_api_request;
use crate::clock::current_unix_secs;
use crate::{AppState, GatewayError};
@@ -60,7 +60,9 @@ pub(crate) async fn resolve_local_same_format_provider_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -79,7 +81,13 @@ pub(crate) async fn resolve_local_same_format_provider_decision_input(
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
input.client_surface = decision.client_surface;
input.gateway_credential_carrier = decision.gateway_credential_carrier;
input.client_session_affinity = client_session_affinity_from_api_request(
spec_metadata.api_format,
&parts.headers,
Some(body_json),
);
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
@@ -115,15 +123,23 @@ pub(crate) async fn materialize_local_same_format_provider_candidate_attempts(
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::SameFormatProviderDecision,
);
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec_metadata.api_format, Some(&input.requested_model));
let routing_model = model_directive_resolution
.base_model()
.unwrap_or(&input.requested_model);
let (candidates, preselection_skipped) = planner_state
.list_selectable_candidates_with_skip_reasons(
.list_selectable_candidates_with_skip_reasons_for_request_operation(
spec_metadata.api_format,
&input.requested_model,
routing_model,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
spec.operation.map(|operation| operation.as_str()),
)
.await?;
let outcome = materialize_local_execution_candidates_with_serving(
@@ -212,15 +228,23 @@ pub(crate) async fn build_local_same_format_provider_candidate_attempt_source<'a
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::SameFormatProviderDecision,
);
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec_metadata.api_format, Some(&input.requested_model));
let routing_model = model_directive_resolution
.base_model()
.unwrap_or(&input.requested_model);
let (candidates, preselection_skipped) = planner_state
.list_selectable_candidates_with_skip_reasons(
.list_selectable_candidates_with_skip_reasons_for_request_operation(
spec_metadata.api_format,
&input.requested_model,
routing_model,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
spec.operation.map(|operation| operation.as_str()),
)
.await?;
@@ -1,5 +1,8 @@
use serde_json::json;
use aether_ai_serving::{AdaptationMode, AiRequestGzipPolicy, OriginalRequestPayload};
use aether_contracts::{ExecutionResponseBodyMode, EXECUTION_RESPONSE_BODY_MODE_HEADER};
use crate::ai_serving::ai_local_execution_contract_for_formats;
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::candidate_materialization::{
@@ -11,12 +14,14 @@ use crate::ai_serving::planner::materialization_policy::{
build_local_candidate_persistence_policy, LocalCandidatePersistencePolicyKind,
};
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_same_format_provider_spec_metadata;
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
@@ -55,10 +60,17 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
let Some(resolved) = resolve_local_same_format_provider_candidate_payload_parts(
state, parts, trace_id, body_json, input, &attempt, spec,
)
.await
.await?
else {
return Ok(None);
};
let request_redacted = resolved.request_redacted;
let compatibility_edits_empty = resolved.compatibility_edits.is_empty();
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
Some(body_json)
};
let prompt_cache_key = resolved
.provider_request_body
@@ -75,6 +87,51 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
.clone()
.or_else(|| resolve_transport_profile(&resolved.transport));
let mut extra_fields = serde_json::Map::new();
extra_fields.insert(
"provider_type".to_string(),
json!(resolved.transport.provider.provider_type.as_str()),
);
if let Some(operation) = spec.operation {
extra_fields.insert("api_operation".to_string(), json!(operation.as_str()));
}
if let Some(client_surface) = input.client_surface {
extra_fields.insert("client_surface".to_string(), json!(client_surface.as_str()));
}
if let Some(carrier) = input.gateway_credential_carrier {
extra_fields.insert(
"gateway_credential_carrier".to_string(),
json!(carrier.as_str()),
);
}
extra_fields.insert(
"upstream_credential_mode".to_string(),
json!(resolved.transport.key.auth_type.trim().to_ascii_lowercase()),
);
let mut adaptation_mode = if resolved.compatibility_edits.is_empty() {
AdaptationMode::NativeTransparent
} else {
AdaptationMode::SameFormatCompat
};
if crate::ai_serving::normalize_api_format_alias(&resolved.provider_api_format)
== "claude:messages"
{
let compatibility_profile =
crate::ai_serving::transport::resolve_anthropic_compatibility_profile(
&resolved.transport,
&resolved.provider_api_format,
);
extra_fields.insert(
"anthropic_compatibility_profile".to_string(),
json!(compatibility_profile.as_str()),
);
if compatibility_profile.uses_claude_code_compatibility() {
adaptation_mode = AdaptationMode::SameFormatCompat;
}
}
extra_fields.insert(
"adaptation_mode".to_string(),
json!(adaptation_mode.as_str()),
);
if let Some(proxy_value) =
build_request_trace_proxy_value(Some(&resolved.transport), proxy.as_ref())
{
@@ -90,6 +147,21 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
"envelope_name".to_string(),
json!(super::super::ANTIGRAVITY_ENVELOPE_NAME),
);
insert_native_client_envelope_name(
&mut extra_fields,
super::super::ANTIGRAVITY_ENVELOPE_NAME,
parts.uri.path(),
);
} else if resolved.is_gemini_cli {
extra_fields.insert(
"envelope_name".to_string(),
json!(crate::ai_serving::transport::GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME),
);
}
if !resolved.compatibility_edits.is_empty() {
if let Ok(value) = serde_json::to_value(&resolved.compatibility_edits) {
extra_fields.insert("request_body_compatibility_edits".to_string(), value);
}
}
let provider_api_format = resolved.provider_api_format.clone();
let effective_headers = input.effective_headers(&parts.headers);
@@ -124,16 +196,17 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
.and_then(serde_json::Value::as_bool)
.unwrap_or(false),
upstream_is_stream: resolved.upstream_is_stream,
has_envelope: resolved.is_kiro || resolved.is_antigravity,
has_envelope: resolved.is_kiro || resolved.is_antigravity || resolved.is_gemini_cli,
needs_conversion: false,
extra_fields,
}),
@@ -147,6 +220,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
let super::request::LocalSameFormatProviderCandidatePayloadParts {
transport,
is_antigravity: _,
is_gemini_cli: _,
is_kiro: _,
auth_header,
auth_value,
@@ -158,7 +232,10 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
provider_request_headers,
provider_request_body,
transport_profile: _,
compatibility_edits: _,
request_redacted: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -168,6 +245,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -185,6 +263,8 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -193,10 +273,84 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
enforce_provider_api_operation_invariants(
spec.operation,
decision.provider_request_body.as_mut(),
&mut decision.provider_request_headers,
);
decision.provider_request_body_base64 = original_request_body_base64(
parts,
decision.provider_request_body.as_ref(),
adaptation_mode,
request_redacted,
compatibility_edits_empty,
decision.content_encoding.as_deref(),
decision.request_gzip.as_ref(),
);
decision
.provider_request_headers
.retain(|name, _| !name.eq_ignore_ascii_case(EXECUTION_RESPONSE_BODY_MODE_HEADER));
if !spec_metadata.require_streaming && decision.provider_request_body_base64.is_some() {
decision.provider_request_headers.insert(
EXECUTION_RESPONSE_BODY_MODE_HEADER.to_string(),
ExecutionResponseBodyMode::PreserveBytes
.as_str()
.to_string(),
);
}
Ok(Some(decision))
}
fn enforce_provider_api_operation_invariants(
operation: Option<crate::ai_serving::ApiOperation>,
provider_request_body: Option<&mut serde_json::Value>,
provider_request_headers: &mut std::collections::BTreeMap<String, String>,
) {
if operation != Some(crate::ai_serving::ApiOperation::ClaudeCountTokens) {
return;
}
if let Some(provider_request_body) = provider_request_body {
crate::ai_serving::transport::enforce_same_format_provider_api_operation_body_policy(
provider_request_body,
operation,
);
}
for header_name in ["accept", "content-type"] {
provider_request_headers.retain(|name, _| !name.eq_ignore_ascii_case(header_name));
provider_request_headers.insert(header_name.to_string(), "application/json".to_string());
}
}
fn original_request_body_base64(
parts: &http::request::Parts,
provider_request_body: Option<&serde_json::Value>,
adaptation_mode: AdaptationMode,
request_redacted: bool,
compatibility_edits_empty: bool,
content_encoding: Option<&str>,
request_gzip: Option<&AiRequestGzipPolicy>,
) -> Option<String> {
if adaptation_mode != AdaptationMode::NativeTransparent
|| request_redacted
|| !compatibility_edits_empty
|| content_encoding.is_some_and(|value| !value.trim().is_empty())
|| request_gzip.is_some_and(|policy| policy.enabled != Some(false))
{
return None;
}
parts
.extensions
.get::<OriginalRequestPayload>()?
.body_bytes_base64_if_unchanged(provider_request_body?)
}
pub(super) async fn mark_skipped_local_same_format_provider_candidate(
state: &AppState,
input: &LocalSameFormatProviderDecisionInput,
@@ -280,3 +434,177 @@ pub(super) async fn mark_skipped_local_same_format_provider_candidate_with_failu
)
.await;
}
#[cfg(test)]
mod tests {
use std::collections::BTreeMap;
use base64::Engine as _;
use super::{
enforce_provider_api_operation_invariants, original_request_body_base64, AdaptationMode,
AiRequestGzipPolicy, OriginalRequestPayload,
};
use crate::ai_serving::ApiOperation;
fn request_parts_with_original_payload(
body_json: serde_json::Value,
body_bytes: &[u8],
) -> http::request::Parts {
let (mut parts, ()) = http::Request::new(()).into_parts();
parts
.extensions
.insert(OriginalRequestPayload::from_parsed_json(
body_json, body_bytes,
));
parts
}
#[test]
fn count_tokens_invariants_win_after_provider_routing_mutations() {
let mut body = serde_json::json!({
"model": "claude-sonnet-4",
"messages": [],
"stream": true
});
let mut headers = BTreeMap::from([
("Accept".to_string(), "text/event-stream".to_string()),
("Content-Type".to_string(), "text/plain".to_string()),
("x-provider-route".to_string(), "kept".to_string()),
]);
enforce_provider_api_operation_invariants(
Some(ApiOperation::ClaudeCountTokens),
Some(&mut body),
&mut headers,
);
assert!(body.get("stream").is_none());
assert_eq!(
headers.get("accept").map(String::as_str),
Some("application/json")
);
assert_eq!(
headers.get("content-type").map(String::as_str),
Some("application/json")
);
assert_eq!(
headers.get("x-provider-route").map(String::as_str),
Some("kept")
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("accept"))
.count(),
1
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("content-type"))
.count(),
1
);
}
#[test]
fn unchanged_same_format_body_preserves_original_json_bytes() {
let raw = br#"{ "unknown": {"enabled":true}, "messages": [], "model": "claude-sonnet-4" }"#;
let body_json: serde_json::Value = serde_json::from_slice(raw).expect("body should parse");
let parts = request_parts_with_original_payload(body_json.clone(), raw);
let encoded = original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
None,
None,
)
.expect("unchanged request should retain exact bytes");
assert_eq!(
base64::engine::general_purpose::STANDARD
.decode(encoded)
.expect("body should decode"),
raw
);
}
#[test]
fn request_edits_or_encoding_disable_original_json_bytes() {
let raw = br#"{"model":"claude-sonnet-4","messages":[]}"#;
let body_json: serde_json::Value = serde_json::from_slice(raw).expect("body should parse");
let parts = request_parts_with_original_payload(body_json.clone(), raw);
let changed_body = serde_json::json!({
"model": "claude-sonnet-4-5",
"messages": []
});
assert!(original_request_body_base64(
&parts,
Some(&changed_body),
AdaptationMode::NativeTransparent,
false,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
true,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
false,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::SameFormatCompat,
false,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
Some("gzip"),
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
None,
Some(&AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1),
}),
)
.is_none());
}
}
@@ -7,17 +7,25 @@ use serde_json::Value;
use crate::ai_serving::planner::common::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
};
use crate::ai_serving::planner::redaction::{
request_identity_response_encoding_when_redacted, resolve_provider_chat_pii_redaction,
};
use crate::ai_serving::transport::antigravity::{
build_antigravity_safe_v1internal_request, build_antigravity_static_identity_headers,
classify_local_antigravity_request_support, AntigravityEnvelopeRequestType,
AntigravityRequestEnvelopeSupport, AntigravityRequestSideSupport,
AntigravityRequestAuthUnsupportedReason, AntigravityRequestEnvelopeSupport,
AntigravityRequestSideSupport, AntigravityRequestSideUnsupportedReason,
};
use crate::ai_serving::transport::{
build_grok_browser_headers, build_grok_upstream_url, build_same_format_provider_headers,
GrokHeaderInput, SameFormatProviderHeadersInput, GROK_CHAT_PATH,
build_gemini_cli_v1internal_request, build_grok_browser_headers, build_grok_upstream_url,
build_same_format_provider_headers, resolve_local_gemini_cli_request_auth,
GeminiCliRequestAuth, GeminiCliRequestAuthSupport, GeminiCliRequestEnvelopeSupport,
GrokHeaderInput, SameFormatProviderCompatibilityEdit,
SameFormatProviderCompatibilityEditAction, SameFormatProviderHeadersInput,
GEMINI_CLI_USER_AGENT, GROK_CHAT_PATH,
};
use crate::ai_serving::{CandidateFailureDiagnostic, GatewayProviderTransportSnapshot};
use crate::AppState;
use crate::{AppState, GatewayError};
mod policy;
mod prepare;
@@ -32,7 +40,10 @@ use super::{
LocalSameFormatProviderCandidateAttempt, LocalSameFormatProviderDecisionInput,
LocalSameFormatProviderSpec,
};
use crate::ai_serving::planner::standard::same_format_provider_request_body_failure_extra_data;
use crate::ai_serving::planner::standard::{
codex_model_capabilities_for_transport, openai_provider_request_contract_failure_extra_data,
same_format_provider_request_body_failure_extra_data,
};
pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trace(
transport: &GatewayProviderTransportSnapshot,
@@ -43,6 +54,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
"openai:chat" => "openai:chat",
"openai:responses" => "openai:responses",
"openai:responses:compact" => "openai:responses:compact",
"openai:search" => "openai:search",
"openai:embedding" => "openai:embedding",
"openai:rerank" => "openai:rerank",
"claude:messages" => "claude:messages",
@@ -51,6 +63,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
"jina:embedding" => "jina:embedding",
"jina:rerank" => "jina:rerank",
"doubao:embedding" => "doubao:embedding",
"aliyun:multimodal_embedding" => "aliyun:multimodal_embedding",
_ => return Some("transport_api_format_unsupported"),
};
let behavior = policy::classify_same_format_provider_request_behavior(
@@ -63,9 +76,11 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
decision_kind: "trace_candidate_metadata",
report_kind: Some("trace_candidate_metadata"),
},
None,
);
if !behavior.is_antigravity
&& !behavior.is_claude_code
&& !behavior.is_claude_code_transport
&& !behavior.is_gemini_cli
&& !behavior.is_vertex
&& !behavior.is_kiro
{
@@ -88,6 +103,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
pub(crate) struct LocalSameFormatProviderCandidatePayloadParts {
pub(super) transport: Arc<GatewayProviderTransportSnapshot>,
pub(super) is_antigravity: bool,
pub(super) is_gemini_cli: bool,
pub(super) is_kiro: bool,
pub(super) auth_header: Option<String>,
pub(super) auth_value: Option<String>,
@@ -99,6 +115,8 @@ pub(crate) struct LocalSameFormatProviderCandidatePayloadParts {
pub(super) provider_request_headers: BTreeMap<String, String>,
pub(super) provider_request_body: Value,
pub(super) transport_profile: Option<ResolvedTransportProfile>,
pub(super) compatibility_edits: Vec<SameFormatProviderCompatibilityEdit>,
pub(super) request_redacted: bool,
}
pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
@@ -109,9 +127,26 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
input: &LocalSameFormatProviderDecisionInput,
attempt: &LocalSameFormatProviderCandidateAttempt,
spec: LocalSameFormatProviderSpec,
) -> Option<LocalSameFormatProviderCandidatePayloadParts> {
) -> Result<Option<LocalSameFormatProviderCandidatePayloadParts>, GatewayError> {
let candidate = &attempt.eligible.candidate;
let prepared = prepare_local_same_format_provider_candidate(
if let Some(skip_reason) = same_format_provider_operation_skip_reason(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
spec.operation,
) {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
let Some(prepared) = prepare_local_same_format_provider_candidate(
state,
trace_id,
input,
@@ -120,18 +155,45 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
&attempt.candidate_id,
spec,
)
.await?;
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
state,
spec.api_format,
Some(&input.requested_model),
)
.await;
.await
else {
return Ok(None);
};
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec.api_format, Some(&input.requested_model));
let model_directive_mapping =
match model_directive_resolution.mapping_patch_for_mapped_model(&prepared.mapped_model) {
Ok(mapping) => mapping,
Err(skip_reason) => {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
};
let effective_headers = input.effective_headers(&parts.headers);
let redaction = resolve_provider_chat_pii_redaction(
state,
parts,
body_json,
&input.auth_context,
spec.api_format,
&attempt.candidate_id,
)
.await?;
let body_json = redaction.body_json.as_ref();
let mut transport = Arc::clone(&prepared.transport);
let Some(mut base_provider_request_body) =
super::super::request::build_same_format_provider_request_body(
let Some(base_provider_request) =
super::super::request::build_same_format_provider_request_body_with_compatibility_report(
body_json,
prepared.provider_api_format.as_str(),
&prepared.mapped_model,
@@ -142,7 +204,7 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
prepared.force_body_stream_field,
prepared.kiro_auth.as_ref(),
prepared.is_claude_code,
enable_model_directives,
false,
)
else {
mark_skipped_local_same_format_provider_candidate_with_extra_data(
@@ -165,20 +227,23 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
if let Some(mapping) =
crate::system_features::reasoning_model_directive_mapping_for_api_format_and_model(
state,
spec.api_format,
Some(&input.requested_model),
)
.await
{
let mut base_provider_request_body = base_provider_request.body;
let mut compatibility_edits = base_provider_request.compatibility_edits;
if let Some(mapping) = model_directive_mapping.as_ref() {
let before_mapping = base_provider_request_body.clone();
crate::ai_serving::apply_model_directive_mapping_patch(
&mut base_provider_request_body,
&mapping,
mapping,
);
if before_mapping != base_provider_request_body {
compatibility_edits.push(SameFormatProviderCompatibilityEdit {
field: "model_directive_mapping".to_string(),
action: SameFormatProviderCompatibilityEditAction::RuntimeRewrite,
detail: "applied configured model directive mapping patch".to_string(),
});
}
// Directive mapping is a deep-merge patch and may overwrite/add `stream`;
// re-enforce stream-field policy afterward.
// Kiro behavior classification already hard-requires upstream streaming,
@@ -193,12 +258,81 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
}
}
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(input.requested_model.as_str());
let codex_model_capabilities = codex_model_capabilities_for_transport(
&transport,
prepared.provider_api_format.as_str(),
prepared.mapped_model.as_str(),
source_model,
);
if let Err(violation) =
crate::ai_serving::finalize_openai_provider_request_with_codex_model_capabilities(
&mut base_provider_request_body,
crate::ai_serving::OpenAiProviderRequestFinalization {
source_api_format: spec.api_format,
provider_api_format: prepared.provider_api_format.as_str(),
provider_type: transport.provider.provider_type.as_str(),
provider_model: prepared.mapped_model.as_str(),
source_model,
body_rules: transport.endpoint.body_rules.as_ref(),
upstream_is_stream: prepared.upstream_is_stream,
require_body_stream_field: request_requires_body_stream_field(
body_json,
prepared.force_body_stream_field,
),
},
codex_model_capabilities.as_ref(),
)
{
mark_skipped_local_same_format_provider_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
Some(openai_provider_request_contract_failure_extra_data(
&violation,
spec.api_format,
prepared.provider_api_format.as_str(),
"same_format_provider_request_finalization",
)),
)
.await;
return Ok(None);
}
let antigravity_auth = if prepared.is_antigravity {
match classify_local_antigravity_request_support(
&prepared.transport,
let mut antigravity_support = classify_local_antigravity_request_support(
&transport,
&base_provider_request_body,
AntigravityEnvelopeRequestType::Agent,
);
if matches!(
antigravity_support,
AntigravityRequestSideSupport::Unsupported(
AntigravityRequestSideUnsupportedReason::UnsupportedAuth(
AntigravityRequestAuthUnsupportedReason::MissingProjectId
)
)
) {
if let Some(hydrated) = state
.hydrate_antigravity_project_metadata_for_transport(&transport)
.await
{
transport = Arc::new(hydrated);
antigravity_support = classify_local_antigravity_request_support(
&transport,
&base_provider_request_body,
AntigravityEnvelopeRequestType::Agent,
);
}
}
match antigravity_support {
AntigravityRequestSideSupport::Supported(spec) => Some(spec.auth),
AntigravityRequestSideSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate(
@@ -211,13 +345,51 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
"transport_unsupported",
)
.await;
return None;
return Ok(None);
}
}
} else {
None
};
let provider_request_body = if let Some(antigravity_auth) = antigravity_auth.as_ref() {
let gemini_cli_auth = if prepared.behavior.is_gemini_cli {
let mut auth = match resolve_local_gemini_cli_request_auth(&transport) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"transport_auth_unavailable",
)
.await;
return Ok(None);
}
};
if auth.project_id.is_none() {
auth = match state
.hydrate_gemini_cli_project_metadata_for_transport(&transport)
.await
{
Some(hydrated) => {
transport = Arc::new(hydrated);
match resolve_local_gemini_cli_request_auth(&transport) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => {
GeminiCliRequestAuth::default()
}
}
}
None => GeminiCliRequestAuth::default(),
};
}
Some(auth)
} else {
None
};
let mut provider_request_body = if let Some(antigravity_auth) = antigravity_auth.as_ref() {
match build_antigravity_safe_v1internal_request(
antigravity_auth,
trace_id,
@@ -243,12 +415,50 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
}
}
} else if let Some(gemini_cli_auth) = gemini_cli_auth.as_ref() {
match build_gemini_cli_v1internal_request(
gemini_cli_auth,
trace_id,
&prepared.mapped_model,
&base_provider_request_body,
) {
GeminiCliRequestEnvelopeSupport::Supported(envelope) => envelope,
GeminiCliRequestEnvelopeSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_missing",
same_format_provider_request_body_failure_extra_data(
body_json,
attempt.eligible.provider_api_format.as_str(),
prepared.transport.endpoint.body_rules.as_ref(),
"gemini_cli_v1internal_envelope",
),
)
.await;
return Ok(None);
}
}
} else {
base_provider_request_body
};
if crate::ai_serving::transport::enforce_same_format_provider_api_operation_body_policy(
&mut provider_request_body,
spec.operation,
) {
compatibility_edits.push(SameFormatProviderCompatibilityEdit {
field: "stream".to_string(),
action: SameFormatProviderCompatibilityEditAction::RuntimeRewrite,
detail: "removed stream field for non-streaming API operation".to_string(),
});
}
let is_grok = prepared
.transport
@@ -256,14 +466,13 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
.provider_type
.trim()
.eq_ignore_ascii_case("grok");
let transport_profile =
crate::ai_serving::transport::resolve_transport_profile(&prepared.transport);
let transport_profile = crate::ai_serving::transport::resolve_transport_profile(&transport);
let upstream_url = if is_grok {
Some(build_grok_upstream_url(&prepared.transport, GROK_CHAT_PATH))
Some(build_grok_upstream_url(&transport, GROK_CHAT_PATH))
} else {
super::super::request::build_same_format_upstream_url(
parts,
&prepared.transport,
&transport,
&prepared.mapped_model,
prepared.provider_api_format.as_str(),
spec,
@@ -288,21 +497,24 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
let extra_headers = antigravity_auth
let mut extra_headers = antigravity_auth
.as_ref()
.map(build_antigravity_static_identity_headers)
.unwrap_or_default();
let Some(provider_request_headers) = (if is_grok {
if prepared.behavior.is_gemini_cli {
extra_headers.insert("user-agent".to_string(), GEMINI_CLI_USER_AGENT.to_string());
}
let Some(mut provider_request_headers) = (if is_grok {
build_grok_browser_headers(GrokHeaderInput {
transport: &prepared.transport,
transport: &transport,
transport_profile: transport_profile.as_ref(),
request_headers: Some(effective_headers),
content_type: "application/json",
accept: "text/event-stream",
header_rules: prepared.transport.endpoint.header_rules.as_ref(),
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &provider_request_body,
original_request_body: body_json,
})
@@ -311,12 +523,12 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
headers: effective_headers,
provider_request_body: &provider_request_body,
original_request_body: body_json,
header_rules: prepared.transport.endpoint.header_rules.as_ref(),
header_rules: transport.endpoint.header_rules.as_ref(),
behavior: prepared.behavior,
api_operation: spec.operation,
auth_header: prepared.auth_header.as_deref(),
auth_value: prepared.auth_value.as_deref(),
extra_headers: &extra_headers,
key_fingerprint: prepared.transport.key.fingerprint.as_ref(),
kiro_auth_config: prepared.kiro_auth.as_ref().map(|auth| &auth.auth_config),
kiro_machine_id: prepared
.kiro_auth
@@ -339,12 +551,39 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
crate::ai_serving::apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
transport.provider.provider_type.as_str(),
prepared.provider_api_format.as_str(),
Some(trace_id),
transport.key.decrypted_auth_config.as_deref(),
);
let provider_model = provider_request_body
.get("model")
.and_then(Value::as_str)
.unwrap_or(prepared.mapped_model.as_str());
crate::ai_serving::apply_codex_openai_responses_lite_header_for_request_body_with_capabilities(
&mut provider_request_headers,
Some(&provider_request_body),
transport.provider.provider_type.as_str(),
prepared.provider_api_format.as_str(),
provider_model,
source_model,
codex_model_capabilities.as_ref(),
);
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
redaction.redacted,
);
Some(LocalSameFormatProviderCandidatePayloadParts {
transport: prepared.transport,
Ok(Some(LocalSameFormatProviderCandidatePayloadParts {
transport,
is_antigravity: prepared.is_antigravity,
is_gemini_cli: prepared.behavior.is_gemini_cli,
is_kiro: prepared.is_kiro,
auth_header: prepared.auth_header,
auth_value: prepared.auth_value,
@@ -356,5 +595,102 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
provider_request_headers,
provider_request_body,
transport_profile,
})
compatibility_edits,
request_redacted: redaction.redacted,
}))
}
fn same_format_provider_operation_skip_reason(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
operation: Option<crate::ai_serving::ApiOperation>,
) -> Option<&'static str> {
(!crate::ai_serving::transport::transport_supports_api_operation(
transport,
provider_api_format,
operation,
))
.then_some("transport_operation_unsupported")
}
#[cfg(test)]
mod tests {
use super::same_format_provider_operation_skip_reason;
use crate::ai_serving::transport::snapshot::{
GatewayProviderTransportEndpoint, GatewayProviderTransportKey,
GatewayProviderTransportProvider,
};
use crate::ai_serving::{ApiOperation, GatewayProviderTransportSnapshot};
fn private_adapter_transport(provider_type: &str) -> GatewayProviderTransportSnapshot {
GatewayProviderTransportSnapshot {
provider: GatewayProviderTransportProvider {
id: "provider-1".to_string(),
name: provider_type.to_string(),
provider_type: provider_type.to_string(),
website: None,
is_active: true,
keep_priority_on_conversion: false,
enable_format_conversion: true,
concurrent_limit: None,
max_retries: None,
proxy: None,
request_timeout_secs: None,
stream_first_byte_timeout_secs: None,
config: None,
},
endpoint: GatewayProviderTransportEndpoint {
id: "endpoint-1".to_string(),
provider_id: "provider-1".to_string(),
api_format: "claude:messages".to_string(),
api_family: Some("claude".to_string()),
endpoint_kind: Some("chat".to_string()),
is_active: true,
base_url: "https://private.example".to_string(),
header_rules: None,
body_rules: None,
max_retries: None,
custom_path: None,
config: None,
format_acceptance_config: None,
proxy: None,
},
key: GatewayProviderTransportKey {
id: "key-1".to_string(),
provider_id: "provider-1".to_string(),
name: "key".to_string(),
auth_type: "oauth".to_string(),
is_active: true,
api_formats: None,
auth_type_by_format: None,
allow_auth_channel_mismatch_formats: None,
allowed_models: None,
capabilities: None,
rate_multipliers: None,
global_priority_by_format: None,
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: String::new(),
decrypted_auth_config: None,
},
}
}
#[test]
fn private_adapter_count_tokens_is_rejected_by_pre_auth_operation_gate() {
for provider_type in ["kiro", "grok"] {
let transport = private_adapter_transport(provider_type);
assert_eq!(
same_format_provider_operation_skip_reason(
&transport,
"claude:messages",
Some(ApiOperation::ClaudeCountTokens),
),
Some("transport_operation_unsupported"),
"provider_type={provider_type}"
);
}
}
}
@@ -1,6 +1,6 @@
use crate::ai_serving::planner::spec_metadata::LocalExecutionSurfaceSpecMetadata;
use crate::ai_serving::transport::{
classify_same_format_provider_request_behavior as classify_same_format_provider_request_behavior_impl,
classify_same_format_provider_request_behavior_for_operation as classify_same_format_provider_request_behavior_impl,
resolve_same_format_provider_direct_auth as resolve_same_format_provider_direct_auth_impl,
same_format_provider_transport_supported as same_format_provider_transport_supported_impl,
same_format_provider_transport_unsupported_reason as same_format_provider_transport_unsupported_reason_impl,
@@ -15,6 +15,7 @@ pub(super) fn classify_same_format_provider_request_behavior(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
spec_metadata: LocalExecutionSurfaceSpecMetadata,
api_operation: Option<crate::ai_serving::ApiOperation>,
) -> SameFormatProviderRequestBehavior {
classify_same_format_provider_request_behavior_impl(
transport,
@@ -25,6 +26,7 @@ pub(super) fn classify_same_format_provider_request_behavior(
.report_kind
.expect("same-format provider specs should declare report kind"),
},
api_operation,
)
}
@@ -60,11 +62,13 @@ pub(super) fn should_try_same_format_provider_oauth_auth(
behavior: &SameFormatProviderRequestBehavior,
transport: &GatewayProviderTransportSnapshot,
family: LocalSameFormatProviderFamily,
provider_api_format: &str,
) -> bool {
should_try_same_format_provider_oauth_auth_impl(
behavior,
transport,
same_format_provider_family(family),
provider_api_format,
)
}
@@ -72,11 +76,13 @@ pub(super) fn resolve_same_format_provider_direct_auth(
behavior: &SameFormatProviderRequestBehavior,
transport: &GatewayProviderTransportSnapshot,
family: LocalSameFormatProviderFamily,
provider_api_format: &str,
) -> Option<(String, String)> {
resolve_same_format_provider_direct_auth_impl(
behavior,
transport,
same_format_provider_family(family),
provider_api_format,
)
}
@@ -28,6 +28,7 @@ pub(super) struct PreparedSameFormatProviderCandidate {
pub(super) behavior: SameFormatProviderRequestBehavior,
pub(super) is_antigravity: bool,
pub(super) is_claude_code: bool,
pub(super) is_gemini_cli: bool,
pub(super) is_vertex: bool,
pub(super) is_kiro: bool,
pub(super) kiro_auth: Option<KiroRequestAuth>,
@@ -58,6 +59,7 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
&transport,
provider_api_format,
spec_metadata,
spec.operation,
);
if !same_format_provider_transport_supported(
@@ -91,8 +93,12 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
} else {
None
};
let should_try_oauth_auth =
should_try_same_format_provider_oauth_auth(&behavior, &transport, spec.family);
let should_try_oauth_auth = should_try_same_format_provider_oauth_auth(
&behavior,
&transport,
spec.family,
provider_api_format,
);
let oauth_auth = if should_try_oauth_auth {
resolve_candidate_oauth_auth(
planner_state,
@@ -117,7 +123,12 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
{
Some((name.clone(), value.clone()))
} else {
resolve_same_format_provider_direct_auth(&behavior, &transport, spec.family)
resolve_same_format_provider_direct_auth(
&behavior,
&transport,
spec.family,
provider_api_format,
)
};
let (auth_header, auth_value) = match auth {
Some((name, value)) => (Some(name), Some(value)),
@@ -175,6 +186,7 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
behavior,
is_antigravity: behavior.is_antigravity,
is_claude_code: behavior.is_claude_code,
is_gemini_cli: behavior.is_gemini_cli,
is_vertex: behavior.is_vertex,
is_kiro: behavior.is_kiro,
kiro_auth,
@@ -190,7 +190,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalSameFormatProviderSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -213,6 +213,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalSameFormatProviderSyncA
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
@@ -220,7 +235,7 @@ impl LocalExecutionAttemptSource<AiStreamAttempt>
for LocalSameFormatProviderStreamAttemptSource<'_>
{
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -243,6 +258,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt>
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalSameFormatProviderSyncAttemptSource<'_> {
@@ -372,7 +402,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -458,7 +488,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -2,4 +2,5 @@ mod body;
mod url;
pub(super) use self::body::build_same_format_provider_request_body;
pub(super) use self::body::build_same_format_provider_request_body_with_compatibility_report;
pub(super) use self::url::build_same_format_upstream_url;
@@ -3,7 +3,9 @@ use serde_json::Value;
use super::super::LocalSameFormatProviderSpec;
use crate::ai_serving::transport::{
build_same_format_provider_request_body as build_same_format_provider_request_body_impl,
build_same_format_provider_request_body_with_compatibility_report as build_same_format_provider_request_body_with_compatibility_report_impl,
SameFormatProviderFamily, SameFormatProviderRequestBodyInput,
SameFormatProviderRequestBodyOutput,
};
pub(crate) fn build_same_format_provider_request_body(
@@ -36,6 +38,38 @@ pub(crate) fn build_same_format_provider_request_body(
})
}
pub(crate) fn build_same_format_provider_request_body_with_compatibility_report(
body_json: &Value,
provider_api_format: &str,
mapped_model: &str,
spec: LocalSameFormatProviderSpec,
body_rules: Option<&Value>,
request_headers: Option<&http::HeaderMap>,
upstream_is_stream: bool,
force_body_stream_field: bool,
kiro_auth: Option<&crate::ai_serving::transport::kiro::KiroRequestAuth>,
is_claude_code: bool,
enable_model_directives: bool,
) -> Option<SameFormatProviderRequestBodyOutput> {
build_same_format_provider_request_body_with_compatibility_report_impl(
SameFormatProviderRequestBodyInput {
body_json,
mapped_model,
client_api_format: spec.api_format,
provider_api_format,
source_model: body_json.get("model").and_then(Value::as_str),
family: same_format_provider_family(spec.family),
body_rules,
request_headers,
upstream_is_stream,
force_body_stream_field,
kiro_auth_config: kiro_auth.map(|auth| &auth.auth_config),
is_claude_code,
enable_model_directives,
},
)
}
fn same_format_provider_family(
family: super::super::LocalSameFormatProviderFamily,
) -> SameFormatProviderFamily {
@@ -24,6 +24,7 @@ pub(crate) fn build_same_format_upstream_url(
upstream_is_stream,
request_query: parts.uri.query(),
kiro_api_region: kiro_auth.map(|auth| auth.auth_config.effective_api_region()),
api_operation: spec.operation,
provider_request_body,
},
)
@@ -6,7 +6,6 @@ use aether_data_contracts::repository::provider_catalog::StoredProviderCatalogKe
use aether_pool_core::{
score_pool_member_with_rules, PoolMemberScoreInput, PoolMemberScoreRules, POOL_SCORE_VERSION,
};
use aether_scheduler_core::any_provider_key_circuit_open_at;
use serde_json::Value;
use crate::handlers::shared::{provider_key_health_summary, provider_key_status_snapshot_payload};
@@ -100,7 +99,6 @@ fn provider_key_score_input(
.and_then(|snapshot| snapshot.get("account"))
.and_then(Value::as_object);
let (health_score, _, _, _, _) = provider_key_health_summary(key);
let active_circuit_open = any_provider_key_circuit_open_at(key, now_unix_secs);
let health_score = key
.health_by_format
.as_ref()
@@ -127,10 +125,9 @@ fn provider_key_score_input(
.and_then(Value::as_bool)
.unwrap_or(false),
oauth_invalid_reason: key.oauth_invalid_reason.clone(),
circuit_open: active_circuit_open,
success_count: key.success_count.unwrap_or(0).into(),
error_count: key.error_count.unwrap_or(0).into(),
total_response_time_ms: key.total_response_time_ms.unwrap_or(0).into(),
total_response_time_ms: key.total_response_time_ms.unwrap_or(0),
total_tokens: key.total_tokens,
total_cost_usd: key.total_cost_usd,
last_used_at: key.last_used_at_unix_secs,
@@ -213,7 +210,7 @@ mod tests {
}
#[test]
fn future_circuit_probe_deadline_keeps_pool_score_in_cooldown() {
fn future_key_circuit_probe_deadline_does_not_drive_pool_score_cooldown() {
let now_unix_secs = 1_000;
let key = sample_key_with_circuit_next_probe(1_100);
@@ -225,6 +222,6 @@ mod tests {
PoolMemberScoreRules::default(),
);
assert_eq!(score.hard_state, PoolMemberHardState::Cooldown);
assert_eq!(score.hard_state, PoolMemberHardState::Available);
}
}
@@ -0,0 +1,239 @@
use std::borrow::Cow;
use std::time::{Instant, SystemTime, UNIX_EPOCH};
use serde_json::Value;
use tracing::warn;
use crate::ai_serving::ExecutionRuntimeAuthContext;
use crate::privacy::{
build_redaction_session_config, read_chat_pii_redaction_runtime_config,
try_mask_chat_pii_request_value_with_cache_options, CachedRequestRedaction,
ChatPiiRedactionRequestFormat, MaskChatRequestOptions, RedactionMaskError,
RedactionSessionSlot, RedisRedactionMappingCache,
};
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{AppState, GatewayError};
pub(crate) struct ProviderRequestRedaction<'a> {
pub(crate) body_json: Cow<'a, Value>,
pub(crate) redacted: bool,
}
impl<'a> ProviderRequestRedaction<'a> {
fn disabled(body_json: &'a Value) -> Self {
Self {
body_json: Cow::Borrowed(body_json),
redacted: false,
}
}
}
#[derive(Clone, Copy, Debug, Default)]
struct ChatPiiRedactionFeatureSettings {
enabled: Option<bool>,
}
impl ChatPiiRedactionFeatureSettings {
fn merge_from_value(&mut self, value: Option<&Value>) {
let Some(settings) = value
.and_then(Value::as_object)
.and_then(|features| features.get("chat_pii_redaction"))
.and_then(Value::as_object)
else {
return;
};
if let Some(enabled) = settings.get("enabled").and_then(Value::as_bool) {
self.enabled = Some(enabled);
}
}
fn effective_enabled(self) -> bool {
self.enabled.unwrap_or(false)
}
}
pub(crate) fn request_identity_response_encoding_when_redacted(
headers: &mut std::collections::BTreeMap<String, String>,
redacted: bool,
) {
if redacted {
headers.insert("accept-encoding".to_string(), "identity".to_string());
}
}
pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
state: &AppState,
parts: &http::request::Parts,
body_json: &'a Value,
auth_context: &ExecutionRuntimeAuthContext,
client_api_format: &str,
candidate_id: &str,
) -> Result<ProviderRequestRedaction<'a>, GatewayError> {
let Some(format) = ChatPiiRedactionRequestFormat::from_api_format(client_api_format) else {
return Ok(ProviderRequestRedaction::disabled(body_json));
};
let Some(slot) = parts.extensions.get::<RedactionSessionSlot>() else {
return Ok(ProviderRequestRedaction::disabled(body_json));
};
let request_cache_key = request_redaction_cache_key(format, body_json);
if let Some(cached) = slot.cached_request_redaction(&request_cache_key) {
crate::stage_metrics::record_chat_pii_redaction_request_cache_hit();
observe_gateway_stage_ms("chat_pii_redaction_request_cache_hit", 0);
return Ok(provider_redaction_from_cached(
slot,
candidate_id,
body_json,
cached,
));
}
crate::stage_metrics::record_chat_pii_redaction_request_cache_miss();
let runtime_config_started_at = Instant::now();
let runtime_config = read_chat_pii_redaction_runtime_config(state)
.await
.map_err(|err| {
warn!(
error = ?err,
"gateway failed to read chat pii redaction runtime config"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
observe_gateway_stage_ms(
"chat_pii_redaction_runtime_config",
runtime_config_started_at.elapsed().as_millis() as u64,
);
if !runtime_config.enabled {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction::disabled(body_json));
}
let feature_settings_started_at = Instant::now();
let feature_settings = resolve_chat_pii_redaction_feature_settings(state, auth_context).await?;
observe_gateway_stage_ms(
"chat_pii_redaction_feature_settings",
feature_settings_started_at.elapsed().as_millis() as u64,
);
if !feature_settings.effective_enabled() {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction::disabled(body_json));
}
let Some(hmac_key) = state.encryption_key().map(str::as_bytes).map(Vec::from) else {
warn!("gateway chat pii redaction is enabled but encryption key is unavailable");
return Err(GatewayError::Internal(
"chat pii redaction setup failed".to_string(),
));
};
let now_unix_secs = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_secs();
let cache = RedisRedactionMappingCache::new(state.runtime_state.as_ref());
let mask_started_at = Instant::now();
let masked = try_mask_chat_pii_request_value_with_cache_options(
body_json,
format,
build_redaction_session_config(hmac_key, &runtime_config, now_unix_secs),
MaskChatRequestOptions::runtime(),
Some(&cache),
)
.await
.map_err(redaction_mask_error_to_gateway_error)?;
observe_gateway_stage_ms(
"chat_pii_redaction_mask_body",
mask_started_at.elapsed().as_millis() as u64,
);
if !masked.redacted {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction {
body_json: Cow::Borrowed(body_json),
redacted: false,
});
}
let Some(masked_body_json) = masked.body_json else {
warn!("gateway pii redaction reported redacted without masked body");
return Err(GatewayError::Internal(
"chat pii redaction setup failed".to_string(),
));
};
slot.put_cached_request_redaction(
request_cache_key,
CachedRequestRedaction::redacted(masked_body_json.clone(), masked.session.clone()),
);
slot.put_for_candidate(candidate_id, masked.session);
Ok(ProviderRequestRedaction {
body_json: Cow::Owned(masked_body_json),
redacted: true,
})
}
fn request_redaction_cache_key(format: ChatPiiRedactionRequestFormat, body_json: &Value) -> String {
format!("{format:?}:{:p}", body_json)
}
fn provider_redaction_from_cached<'a>(
slot: &RedactionSessionSlot,
candidate_id: &str,
body_json: &'a Value,
cached: CachedRequestRedaction,
) -> ProviderRequestRedaction<'a> {
if !cached.redacted {
return ProviderRequestRedaction::disabled(body_json);
}
let Some(masked_body_json) = cached.body_json else {
return ProviderRequestRedaction::disabled(body_json);
};
if let Some(session) = cached.session {
slot.put_for_candidate(candidate_id, session);
}
ProviderRequestRedaction {
body_json: Cow::Owned(masked_body_json),
redacted: true,
}
}
async fn resolve_chat_pii_redaction_feature_settings(
state: &AppState,
auth_context: &ExecutionRuntimeAuthContext,
) -> Result<ChatPiiRedactionFeatureSettings, GatewayError> {
let user_settings_fut = state.read_user_feature_settings(&auth_context.user_id);
let key_settings_fut = state.read_auth_api_key_feature_settings(
&auth_context.user_id,
&auth_context.api_key_id,
auth_context.api_key_is_standalone,
);
let (user_settings, key_settings) = tokio::try_join!(user_settings_fut, key_settings_fut)
.map_err(|err| {
warn!(
error = ?err,
"gateway failed to read chat pii redaction feature settings"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
let mut settings = ChatPiiRedactionFeatureSettings::default();
settings.merge_from_value(user_settings.as_ref());
settings.merge_from_value(key_settings.as_ref());
Ok(settings)
}
fn redaction_mask_error_to_gateway_error(error: RedactionMaskError) -> GatewayError {
match error {}
}
#[cfg(test)]
mod tests {
use serde_json::json;
use super::ChatPiiRedactionFeatureSettings;
#[test]
fn chat_pii_redaction_feature_settings_only_control_enablement() {
let mut settings = ChatPiiRedactionFeatureSettings::default();
settings.merge_from_value(Some(&json!({
"chat_pii_redaction": {
"enabled": true
}
})));
assert!(settings.effective_enabled());
}
}
@@ -6,6 +6,7 @@ use aether_ai_serving::{
provider_stream_event_api_format_for_provider_type as ai_provider_stream_event_api_format_for_provider_type,
AiExecutionReportContextParts, AiRequestOrigin,
};
use aether_routing_core::ResolvedRoutingPolicy;
use aether_runtime_state::RuntimeLockLease;
use aether_scheduler_core::{ClientSessionAffinity, SchedulerRankingOutcome};
use serde_json::{Map, Value};
@@ -20,8 +21,9 @@ use crate::client_session_affinity::{
};
use crate::orchestration::{
insert_pool_key_lease_report_context_fields, ExecutionAttemptIdentity,
SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD,
ROUTING_POOL_POLICY_OVERRIDE_REPORT_FIELD, SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD,
};
use crate::scheduler::affinity::insert_scheduler_affinity_policy_report_context_field;
pub(crate) struct LocalExecutionReportContextParts<'a> {
pub(crate) auth_context: &'a ExecutionRuntimeAuthContext,
@@ -55,6 +57,7 @@ pub(crate) struct LocalExecutionReportContextParts<'a> {
pub(crate) original_request_body_json: Option<&'a Value>,
pub(crate) original_request_body_base64: Option<&'a str>,
pub(crate) client_session_affinity: Option<&'a ClientSessionAffinity>,
pub(crate) routing_policy: Option<&'a ResolvedRoutingPolicy>,
pub(crate) scheduler_affinity_epoch: Option<u64>,
pub(crate) client_requested_stream: bool,
pub(crate) upstream_is_stream: bool,
@@ -105,6 +108,16 @@ pub(crate) fn build_local_execution_report_context(
merge_incoming_tls_fingerprint(&mut extra_fields, incoming_tls);
}
insert_pool_key_lease_report_context_fields(&mut extra_fields, parts.pool_key_lease);
insert_scheduler_affinity_policy_report_context_field(&mut extra_fields, parts.routing_policy);
if let Some(override_policy) = parts
.routing_policy
.and_then(|policy| policy.pool_policy_overrides.get(parts.provider_id))
.filter(|override_policy| !override_policy.scheduling_presets.is_empty())
{
if let Ok(value) = serde_json::to_value(override_policy) {
extra_fields.insert(ROUTING_POOL_POLICY_OVERRIDE_REPORT_FIELD.to_string(), value);
}
}
if let Some(epoch) = parts.scheduler_affinity_epoch {
extra_fields.insert(
SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD.to_string(),
@@ -198,6 +211,21 @@ pub(crate) fn insert_provider_stream_event_api_format(
insert_ai_provider_stream_event_api_format(extra_fields, provider_type);
}
pub(crate) fn insert_native_client_envelope_name(
extra_fields: &mut Map<String, Value>,
envelope_name: &str,
request_path: &str,
) {
if envelope_name.eq_ignore_ascii_case("antigravity:v1internal")
&& request_path == "/v1internal:streamGenerateContent"
{
extra_fields.insert(
"client_envelope_name".to_string(),
Value::String(envelope_name.to_string()),
);
}
}
fn merge_incoming_tls_fingerprint(extra_fields: &mut Map<String, Value>, incoming_tls: Value) {
let entry = extra_fields
.entry("tls_fingerprint".to_string())
@@ -300,6 +328,7 @@ mod tests {
original_request_body_json: Some(&json!({"model": "gpt-5"})),
original_request_body_base64: None,
client_session_affinity: Some(&client_session_affinity),
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: false,
@@ -382,6 +411,7 @@ mod tests {
})),
original_request_body_base64: None,
client_session_affinity: None,
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: true,
@@ -448,6 +478,7 @@ mod tests {
original_request_body_json: Some(&json!({"model": "gpt-5"})),
original_request_body_base64: None,
client_session_affinity: None,
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: false,
@@ -0,0 +1,406 @@
use aether_ai_serving::AiRequestGzipPolicy;
use serde_json::Value;
use crate::ai_serving::{normalize_api_format_alias, parse_codex_auth_identity};
use super::state::GatewayProviderTransportSnapshot;
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub(crate) struct TransportRequestEncodingPolicy {
pub content_encoding: Option<String>,
pub request_gzip: Option<AiRequestGzipPolicy>,
}
pub(crate) fn resolve_transport_request_encoding_policy(
transport: &GatewayProviderTransportSnapshot,
) -> TransportRequestEncodingPolicy {
if transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex")
&& normalize_api_format_alias(transport.endpoint.api_format.as_str())
== "openai:responses:compact"
{
return TransportRequestEncodingPolicy::default();
}
let request_gzip = transport_request_gzip_policy_from_config(
transport.endpoint.config.as_ref(),
)
.or_else(|| transport_request_gzip_policy_from_config(transport.provider.config.as_ref()));
if request_gzip.is_some() {
return TransportRequestEncodingPolicy {
content_encoding: None,
request_gzip,
};
}
TransportRequestEncodingPolicy {
content_encoding: default_transport_request_content_encoding(transport),
request_gzip: None,
}
}
fn default_transport_request_content_encoding(
transport: &GatewayProviderTransportSnapshot,
) -> Option<String> {
if !transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex")
{
return None;
}
if !is_codex_request_compression_api_format(transport.endpoint.api_format.as_str()) {
return None;
}
let auth_type =
crate::ai_serving::transport::auth::resolve_local_auth_type_for_transport_format(transport);
let uses_codex_backend = auth_type == "oauth"
|| (auth_type == "bearer"
&& parse_codex_auth_identity(transport.key.decrypted_auth_config.as_deref())
.uses_codex_backend);
if !uses_codex_backend {
return None;
}
Some("zstd".to_string())
}
fn is_codex_request_compression_api_format(api_format: &str) -> bool {
normalize_api_format_alias(api_format) == "openai:responses"
}
fn transport_request_gzip_policy_from_config(
config: Option<&Value>,
) -> Option<AiRequestGzipPolicy> {
let object = config?.as_object()?;
for key in ["request_gzip", "request_body_gzip"] {
if let Some(policy) = object
.get(key)
.and_then(transport_request_gzip_policy_from_value)
{
return Some(policy);
}
}
let enabled = first_config_bool(
object,
&["request_gzip_enabled", "request_body_gzip_enabled"],
);
let min_bytes = first_config_usize(
object,
&["request_gzip_min_bytes", "request_body_gzip_min_bytes"],
);
match (enabled, min_bytes) {
(Some(false), _) => Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
}),
(Some(true), min_bytes) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes,
}),
(None, Some(min_bytes)) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(min_bytes),
}),
(None, None) => None,
}
}
fn transport_request_gzip_policy_from_value(value: &Value) -> Option<AiRequestGzipPolicy> {
if let Some(enabled) = value.as_bool() {
return Some(AiRequestGzipPolicy {
enabled: Some(enabled),
min_bytes: None,
});
}
let object = value.as_object()?;
let enabled = first_config_bool(object, &["enabled"]);
let min_bytes = first_config_usize(object, &["min_bytes"]);
match (enabled, min_bytes) {
(Some(false), _) => Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
}),
(Some(true), min_bytes) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes,
}),
(None, Some(min_bytes)) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(min_bytes),
}),
(None, None) => None,
}
}
fn first_config_bool(object: &serde_json::Map<String, Value>, keys: &[&str]) -> Option<bool> {
keys.iter()
.find_map(|key| object.get(*key).and_then(config_bool))
}
fn config_bool(value: &Value) -> Option<bool> {
value.as_bool().or_else(|| {
value.as_str().and_then(|text| {
let normalized = text.trim();
if normalized.eq_ignore_ascii_case("true") {
Some(true)
} else if normalized.eq_ignore_ascii_case("false") {
Some(false)
} else {
None
}
})
})
}
fn first_config_usize(object: &serde_json::Map<String, Value>, keys: &[&str]) -> Option<usize> {
keys.iter()
.find_map(|key| object.get(*key).and_then(config_usize))
}
fn config_usize(value: &Value) -> Option<usize> {
value
.as_u64()
.and_then(|number| usize::try_from(number).ok())
.or_else(|| {
value
.as_str()
.and_then(|text| text.trim().parse::<usize>().ok())
})
}
#[cfg(test)]
mod tests {
use super::*;
use aether_provider_transport::snapshot::{
GatewayProviderTransportEndpoint, GatewayProviderTransportKey,
GatewayProviderTransportProvider, GatewayProviderTransportSnapshot,
};
use serde_json::{json, Value};
fn sample_transport(
provider_type: &str,
endpoint_api_format: &str,
provider_config: Option<Value>,
endpoint_config: Option<Value>,
) -> GatewayProviderTransportSnapshot {
GatewayProviderTransportSnapshot {
provider: GatewayProviderTransportProvider {
id: "provider-1".to_string(),
name: "Provider".to_string(),
provider_type: provider_type.to_string(),
website: None,
is_active: true,
keep_priority_on_conversion: false,
enable_format_conversion: true,
concurrent_limit: None,
max_retries: None,
proxy: None,
request_timeout_secs: None,
stream_first_byte_timeout_secs: None,
config: provider_config,
},
endpoint: GatewayProviderTransportEndpoint {
id: "endpoint-1".to_string(),
provider_id: "provider-1".to_string(),
api_format: endpoint_api_format.to_string(),
api_family: None,
endpoint_kind: None,
is_active: true,
base_url: "https://api.example.test".to_string(),
header_rules: None,
body_rules: None,
max_retries: None,
custom_path: None,
config: endpoint_config,
format_acceptance_config: None,
proxy: None,
},
key: GatewayProviderTransportKey {
id: "key-1".to_string(),
provider_id: "provider-1".to_string(),
name: "key".to_string(),
auth_type: "api_key".to_string(),
is_active: true,
api_formats: None,
auth_type_by_format: None,
allow_auth_channel_mismatch_formats: None,
allowed_models: None,
capabilities: None,
rate_multipliers: None,
global_priority_by_format: None,
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "secret".to_string(),
decrypted_auth_config: None,
},
}
}
fn resolved_gzip_policy(
transport: &GatewayProviderTransportSnapshot,
) -> Option<AiRequestGzipPolicy> {
resolve_transport_request_encoding_policy(transport).request_gzip
}
fn resolved_content_encoding(transport: &GatewayProviderTransportSnapshot) -> Option<String> {
resolve_transport_request_encoding_policy(transport).content_encoding
}
#[test]
fn endpoint_request_gzip_policy_overrides_provider_policy() {
let transport = sample_transport(
"openai",
"openai:responses",
Some(json!({"request_gzip": false})),
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 1024}})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1024),
})
);
}
#[test]
fn endpoint_request_gzip_false_disables_provider_and_codex_defaults() {
let transport = sample_transport(
"codex",
"openai:responses",
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 1024}})),
Some(json!({"request_gzip": false})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
})
);
}
#[test]
fn request_gzip_policy_supports_top_level_aliases() {
let transport = sample_transport(
"openai",
"openai:responses",
None,
Some(json!({
"request_body_gzip_enabled": true,
"request_body_gzip_min_bytes": "4096"
})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(4096),
})
);
}
#[test]
fn request_gzip_policy_treats_min_bytes_only_as_enabled() {
let transport = sample_transport(
"openai",
"openai:responses",
None,
Some(json!({"request_gzip_min_bytes": 1})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1),
})
);
}
#[test]
fn codex_responses_endpoint_uses_zstd_without_a_size_threshold() {
let mut transport = sample_transport("codex", "openai:responses", None, None);
transport.key.auth_type = "oauth".to_string();
assert_eq!(
resolved_content_encoding(&transport).as_deref(),
Some("zstd")
);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_responses_api_key_auth_does_not_enable_default_compression() {
let transport = sample_transport("codex", "openai:responses", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_responses_bearer_auth_uses_identity_metadata_for_backend_compression() {
let mut transport = sample_transport("codex", "openai:responses", None, None);
transport.key.auth_type = "bearer".to_string();
transport.key.decrypted_auth_config =
Some(r#"{"provider_type":"codex","account_id":"account-1"}"#.to_string());
assert_eq!(
resolved_content_encoding(&transport).as_deref(),
Some("zstd")
);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_image_endpoint_does_not_get_responses_request_gzip_policy() {
let transport = sample_transport("codex", "openai:image", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_compact_endpoint_does_not_get_default_request_gzip_policy() {
let transport = sample_transport("codex", "openai:responses:compact", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_compact_endpoint_rejects_an_explicit_request_gzip_policy() {
let transport = sample_transport(
"codex",
"openai:responses:compact",
None,
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 2048}})),
);
assert_eq!(resolved_gzip_policy(&transport), None);
assert_eq!(resolved_content_encoding(&transport), None);
}
#[test]
fn non_codex_endpoint_does_not_get_default_request_gzip_policy() {
let transport = sample_transport("openai", "openai:responses", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
}
@@ -1,8 +1,8 @@
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{
is_matching_stream_http_request as is_matching_stream_http_request_impl,
resolve_execution_runtime_stream_plan_kind as resolve_execution_runtime_stream_plan_kind_impl,
resolve_execution_runtime_sync_plan_kind as resolve_execution_runtime_sync_plan_kind_impl,
resolve_execution_runtime_stream_plan_kind_with_client_surface as resolve_execution_runtime_stream_plan_kind_impl,
resolve_execution_runtime_sync_plan_kind_with_client_surface as resolve_execution_runtime_sync_plan_kind_impl,
supports_stream_execution_decision_kind as supports_stream_execution_decision_kind_impl,
supports_sync_execution_decision_kind as supports_sync_execution_decision_kind_impl,
};
@@ -11,28 +11,34 @@ pub(crate) fn resolve_execution_runtime_stream_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
resolve_execution_runtime_stream_plan_kind_impl(
let plan_kind = resolve_execution_runtime_stream_plan_kind_impl(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, true, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn resolve_execution_runtime_sync_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
resolve_execution_runtime_sync_plan_kind_impl(
let plan_kind = resolve_execution_runtime_sync_plan_kind_impl(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, false, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn is_matching_stream_request(
@@ -62,7 +68,7 @@ mod tests {
resolve_execution_runtime_sync_plan_kind, supports_stream_execution_decision_kind,
supports_sync_execution_decision_kind,
};
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{ApiOperation, ClientSurface, GatewayControlDecision};
fn sample_decision(route_family: &str, route_kind: &str) -> GatewayControlDecision {
GatewayControlDecision {
@@ -71,12 +77,16 @@ mod tests {
route_class: Some("ai_public".to_string()),
route_family: Some(route_family.to_string()),
route_kind: Some(route_kind.to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_context: None,
admin_principal: None,
auth_endpoint_signature: None,
execution_runtime_candidate: true,
local_auth_rejection: None,
model_directive_policy: Default::default(),
}
}
@@ -120,7 +130,9 @@ mod tests {
let (claude_parts, _) = claude_request.into_parts();
let claude_api_key = sample_decision_with_auth_channel("claude", "messages", "api_key");
let claude_bearer = sample_decision_with_auth_channel("claude", "messages", "bearer_like");
let mut claude_bearer =
sample_decision_with_auth_channel("claude", "messages", "bearer_like");
claude_bearer.client_surface = Some(ClientSurface::ClaudeCode);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&claude_parts, &claude_api_key),
Some("claude_chat_sync")
@@ -130,6 +142,13 @@ mod tests {
Some("claude_cli_stream")
);
let claude_sdk_bearer =
sample_decision_with_auth_channel("claude", "messages", "bearer_like");
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&claude_parts, &claude_sdk_bearer),
Some("claude_chat_sync")
);
let gemini_request = Request::builder()
.method(Method::POST)
.uri("/v1beta/models/gemini-2.5-pro:generateContent")
@@ -151,6 +170,36 @@ mod tests {
);
}
#[test]
fn resolves_claude_count_tokens_as_native_sync_operation() {
let request = Request::builder()
.method(Method::POST)
.uri("/v1/messages/count_tokens")
.body(())
.expect("request should build");
let (parts, _) = request.into_parts();
let mut decision = sample_decision("claude", "count_tokens");
decision.api_operation = Some(ApiOperation::ClaudeCountTokens);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&parts, &decision),
Some("claude_count_tokens_sync")
);
assert!(supports_sync_execution_decision_kind(
"claude_count_tokens_sync"
));
decision.api_operation = Some(ApiOperation::ClaudeMessagesCreate);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&parts, &decision),
None
);
assert_eq!(
resolve_execution_runtime_stream_plan_kind(&parts, &decision),
None
);
}
#[test]
fn stream_matching_uses_surface_route_logic() {
let request = Request::builder()
@@ -175,7 +175,7 @@ pub(crate) async fn build_local_gemini_files_stream_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalGeminiFilesSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -193,12 +193,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalGeminiFilesSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalGeminiFilesStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -216,6 +231,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalGeminiFilesStreamAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalGeminiFilesSyncAttemptSource<'_> {
@@ -323,7 +353,7 @@ pub(crate) async fn maybe_build_sync_local_gemini_files_decision_payload(
let (mut source, _) =
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -365,7 +395,7 @@ pub(crate) async fn maybe_build_stream_local_gemini_files_decision_payload(
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
let empty_body_json = serde_json::Value::Null;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -414,7 +444,7 @@ async fn build_local_sync_plan_and_reports(
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -467,7 +497,7 @@ async fn build_local_stream_plan_and_reports(
let mut plans = Vec::new();
let empty_body_json = serde_json::Value::Null;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -7,13 +7,14 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_gemini_files_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{AiExecutionDecision, AppState, GatewayError};
use crate::{append_local_failover_policy_to_value, AiExecutionDecision, AppState, GatewayError};
use super::request::resolve_local_gemini_files_candidate_payload_parts;
use super::support::{
@@ -106,6 +107,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
original_request_body_json: Some(body_json),
original_request_body_base64: resolved.provider_request_body_base64.as_deref(),
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: spec_metadata.require_streaming,
upstream_is_stream: spec_metadata.require_streaming,
@@ -113,6 +115,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
needs_conversion: false,
extra_fields,
});
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let super::request::LocalGeminiFilesCandidatePayloadParts {
transport: _,
auth_header,
@@ -123,6 +126,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
upstream_url,
file_name: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -132,6 +136,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -154,6 +159,8 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -162,6 +169,10 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -53,7 +53,9 @@ pub(super) async fn resolve_local_gemini_files_decision_input(
state,
auth_context,
None,
decision.auth_endpoint_signature.as_deref(),
Some(&explicit_required_capabilities),
&decision.model_directive_policy,
)
.await
{
@@ -253,7 +253,7 @@ pub(crate) async fn build_local_image_stream_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiImageSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -271,12 +271,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiImageSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiImageStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -294,6 +309,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiImageStreamAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiImageSyncAttemptSource<'_> {
@@ -421,7 +451,7 @@ pub(crate) async fn maybe_build_sync_local_image_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -482,7 +512,7 @@ pub(crate) async fn maybe_build_stream_local_image_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -540,7 +570,7 @@ async fn build_local_sync_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -617,7 +647,7 @@ async fn build_local_stream_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -5,14 +5,16 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_openai_image_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{
append_execution_contract_fields_to_value, AiExecutionDecision, AppState, GatewayError,
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_image_candidate_payload_parts;
@@ -82,21 +84,8 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
"chatgpt_web_image".to_string(),
serde_json::Value::Bool(true),
);
extra_fields.insert(
"local_failover_policy".to_string(),
serde_json::json!({
"stop_status_codes": [400, 401, 403, 429, 500, 502, 503, 504],
"error_stop_patterns": [
{ "pattern": ".*" }
]
}),
);
}
let upstream_is_stream = resolved
.provider_request_body
.get("stream")
.and_then(serde_json::Value::as_bool)
.unwrap_or(spec_metadata.require_streaming);
let upstream_is_stream = resolved.upstream_is_stream;
let effective_headers = input.effective_headers(&parts.headers);
let report_context = append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -131,6 +120,7 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
original_request_body_json: Some(body_json),
original_request_body_base64: body_base64,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: spec_metadata.require_streaming,
upstream_is_stream,
@@ -143,6 +133,8 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
spec_metadata.api_format,
provider_api_format.as_str(),
);
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -152,6 +144,7 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -169,6 +162,8 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
provider_request_body: Some(resolved.provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -177,6 +172,10 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -16,8 +16,8 @@ use crate::ai_serving::transport::{
ProviderOpenAiImageHeadersInput, StandardProviderRequestHeadersInput, GROK_CHAT_PATH,
};
use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
build_chatgpt_web_image_request_body,
apply_codex_openai_special_headers, build_chatgpt_web_image_request_body,
build_codex_openai_image_api_provider_request_body,
build_gemini_image_request_body_from_openai_image_request,
build_openai_image_api_provider_request_body, build_openai_image_provider_request_body,
default_model_for_openai_image_operation, normalize_openai_image_request,
@@ -48,6 +48,7 @@ pub(super) struct LocalOpenAiImageCandidatePayloadParts {
pub(super) upstream_url: String,
pub(super) input_summary: Value,
pub(super) transport_profile: Option<ResolvedTransportProfile>,
pub(super) upstream_is_stream: bool,
}
pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
@@ -130,7 +131,10 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
parts,
body_json,
body_base64,
openai_image_normalize_options_for_provider(&transport.provider.provider_type),
openai_image_normalize_options_for_provider(
&transport.provider.provider_type,
Some(prepared_candidate.mapped_model.as_str()),
),
);
let Some(normalized_request) = normalized_request else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
@@ -174,29 +178,56 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
} else {
build_openai_image_upstream_url(transport, Some(parts.uri.path()), parts.uri.query())
};
let mut provider_request_body = if is_chatgpt_web {
match build_chatgpt_web_image_request_body(parts, body_json, body_base64) {
Ok(body) => body,
Err(err) => err.to_error_json(),
}
} else if is_codex || is_grok {
build_openai_image_provider_request_body(&normalized_request)
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
spec_metadata.api_format,
spec_metadata.require_streaming && candidate.supports_streaming,
false,
);
let provider_request_body = if is_chatgpt_web {
Some(
match build_chatgpt_web_image_request_body(parts, body_json, body_base64) {
Ok(body) => body,
Err(err) => err.to_error_json(),
},
)
} else if is_codex {
build_codex_openai_image_api_provider_request_body(
&normalized_request,
Some(prepared_candidate.mapped_model.as_str()),
upstream_is_stream,
)
} else if is_grok {
Some(build_openai_image_provider_request_body(
&normalized_request,
))
} else {
build_openai_image_api_provider_request_body(
&normalized_request,
Some(prepared_candidate.mapped_model.as_str()),
upstream_is_stream,
)
};
if !is_chatgpt_web {
apply_codex_openai_responses_special_body_edits(
&mut provider_request_body,
transport.provider.provider_type.as_str(),
spec_metadata.api_format,
transport.endpoint.body_rules.as_ref(),
Some(candidate.key_id.as_str()),
);
}
let Some(provider_request_body) = provider_request_body else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_missing",
CandidateFailureDiagnostic::provider_request_body_missing(
spec_metadata.api_format,
spec_metadata.api_format,
"codex_openai_images_request_contract",
),
)
.await;
return None;
};
let Some(mut provider_request_headers) = (if is_grok {
build_grok_browser_headers(GrokHeaderInput {
transport,
@@ -210,9 +241,17 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
})
} else {
build_openai_image_headers(ProviderOpenAiImageHeadersInput {
transport,
headers: effective_headers,
auth_header: &auth_header,
auth_value: &auth_value,
accept: if is_codex {
None
} else if upstream_is_stream {
Some("text/event-stream")
} else {
Some("application/json")
},
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &provider_request_body,
original_request_body: body_json,
@@ -239,7 +278,7 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
provider_request_headers.insert("x-aether-chatgpt-web-image".to_string(), "1".to_string());
} else if is_grok {
} else {
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
@@ -281,6 +320,7 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
upstream_url,
input_summary,
transport_profile,
upstream_is_stream,
})
}
@@ -397,7 +437,14 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
return None;
}
};
let upstream_is_stream = spec_metadata.require_streaming;
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
provider_api_format,
spec_metadata.require_streaming && candidate.supports_streaming,
false,
);
let Some(upstream_url) = crate::ai_serving::planner::standard::build_standard_upstream_url(
parts,
transport,
@@ -468,6 +515,7 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
upstream_url,
input_summary: converted.summary_json,
transport_profile: None,
upstream_is_stream,
})
}
@@ -58,7 +58,9 @@ pub(super) async fn resolve_local_openai_image_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -124,6 +126,7 @@ pub(super) async fn list_local_openai_image_candidate_attempts(
matches_client_format.then_some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -144,8 +147,8 @@ pub(super) async fn list_local_openai_image_candidate_attempts(
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
candidate,
false,
)
});
}
@@ -197,6 +200,7 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
matches_client_format.then_some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -206,16 +210,16 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
candidate,
false,
)
});
format_skipped.retain(|candidate| {
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
&candidate.candidate,
false,
)
});
}
@@ -105,7 +105,7 @@ pub(crate) async fn build_local_video_sync_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalVideoCreateSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -123,6 +123,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalVideoCreateSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalVideoCreateSyncAttemptSource<'_> {
@@ -195,7 +210,7 @@ pub(crate) async fn maybe_build_sync_local_video_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
state, parts, body_json, trace_id, &input, attempt, spec,
)
@@ -240,7 +255,7 @@ async fn build_local_sync_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
state, parts, body_json, trace_id, &input, attempt, spec,
)
@@ -5,13 +5,14 @@ use crate::ai_serving::planner::report_context::{
};
use crate::ai_serving::planner::spec_metadata::local_video_create_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{AiExecutionDecision, AppState, GatewayError};
use crate::{append_local_failover_policy_to_value, AiExecutionDecision, AppState, GatewayError};
use super::request::resolve_local_video_create_candidate_payload_parts;
use super::support::{LocalVideoCreateCandidateAttempt, LocalVideoCreateDecisionInput};
@@ -87,6 +88,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
original_request_body_json: Some(body_json),
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: false,
upstream_is_stream: false,
@@ -94,6 +96,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
needs_conversion: false,
extra_fields,
});
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let super::request::LocalVideoCreateCandidatePayloadParts {
transport: _,
auth_header,
@@ -103,6 +106,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
provider_request_body,
upstream_url,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: false,
@@ -112,6 +116,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -135,6 +140,8 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
@@ -143,6 +150,10 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -62,7 +62,9 @@ pub(super) async fn resolve_local_video_create_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -130,6 +132,7 @@ pub(super) async fn list_local_video_create_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -186,6 +189,7 @@ pub(super) async fn build_local_video_create_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -3,5 +3,43 @@
mod tests;
pub(crate) use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_identity_headers, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_special_headers,
};
pub(crate) fn codex_model_capabilities_for_transport(
transport: &crate::ai_serving::GatewayProviderTransportSnapshot,
provider_api_format: &str,
provider_model: &str,
source_model: &str,
) -> Option<crate::ai_serving::CodexResponsesModelCapabilities> {
codex_model_capabilities(
&transport.provider.provider_type,
provider_api_format,
provider_model,
source_model,
transport.key.upstream_metadata.as_ref(),
)
}
fn codex_model_capabilities(
provider_type: &str,
provider_api_format: &str,
provider_model: &str,
source_model: &str,
upstream_metadata: Option<&serde_json::Value>,
) -> Option<crate::ai_serving::CodexResponsesModelCapabilities> {
let uses_codex_model_catalog =
crate::ai_serving::is_openai_responses_family_format(provider_api_format)
|| crate::ai_serving::api_format_alias_matches(provider_api_format, "openai:search");
if !provider_type.trim().eq_ignore_ascii_case("codex") || !uses_codex_model_catalog {
return None;
}
Some(
crate::ai_serving::resolve_codex_responses_model_capabilities(
provider_model,
source_model,
upstream_metadata,
),
)
}
@@ -1,15 +1,58 @@
use std::collections::BTreeMap;
use super::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_identity_headers, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_special_headers, codex_model_capabilities,
};
use crate::ai_serving::planner::standard::{
build_cross_format_openai_responses_request_body, build_local_openai_responses_request_body,
};
use http::{HeaderMap, HeaderValue};
use serde_json::json;
#[test]
fn search_uses_live_codex_model_catalog_capabilities() {
let metadata = crate::ai_serving::build_codex_model_catalog_metadata(&[json!({
"slug": "gpt-search-custom",
"default_reasoning_level": "low",
"supported_reasoning_levels": [
{"effort": "low"},
{"effort": "max"}
],
"supports_parallel_tool_calls": true
})]);
let capabilities = codex_model_capabilities(
"codex",
"openai:search",
"gpt-search-custom",
"gpt-search-custom",
Some(&metadata),
)
.expect("Search should resolve capabilities from the Codex model catalog");
assert_eq!(
capabilities.default_reasoning_effort.as_deref(),
Some("low")
);
assert_eq!(
capabilities.supported_reasoning_efforts,
vec!["low".to_string(), "max".to_string()]
);
assert!(codex_model_capabilities(
"codex",
"openai:chat",
"gpt-search-custom",
"gpt-search-custom",
Some(&metadata),
)
.is_none());
}
#[test]
fn applies_codex_defaults_when_body_rules_do_not_handle_fields() {
let mut body = json!({
"model": "gpt-5",
"model": "gpt-5.4",
"max_output_tokens": 128,
"temperature": 0.3,
"top_p": 0.9,
@@ -30,10 +73,45 @@ fn applies_codex_defaults_when_body_rules_do_not_handle_fields() {
assert!(body.get("top_p").is_none());
assert!(body.get("metadata").is_none());
assert_eq!(body["store"], false);
assert_eq!(body["instructions"], "");
assert!(body.get("instructions").is_none());
assert_eq!(body["include"], json!(["reasoning.encrypted_content"]));
assert_eq!(body["parallel_tool_calls"], true);
assert!(body.get("reasoning").is_none());
assert_eq!(body["reasoning"]["effort"], "medium");
assert!(body["reasoning"].get("summary").is_none());
}
#[test]
fn local_openai_responses_codex_body_wraps_string_input_for_backend() {
let body = json!({
"model": "gpt-5",
"input": "hello"
});
let provider_request_body = build_local_openai_responses_request_body(
&body,
"gpt-5-upstream",
false,
false,
"codex",
"openai:responses",
None,
Some("key-123"),
&HeaderMap::new(),
false,
)
.expect("codex local openai responses body should build");
assert_eq!(
provider_request_body["input"],
json!([{
"type": "message",
"role": "user",
"content": [{
"type": "input_text",
"text": "hello"
}]
}])
);
}
#[test]
@@ -45,7 +123,7 @@ fn strips_store_for_compact_even_when_body_rules_handle_it() {
{"action":"set","path":"top_p","value":0.5}
]);
let mut body = json!({
"model": "gpt-5",
"model": "gpt-5.4",
"max_output_tokens": 128,
"metadata": {"client": "desktop", "mode": "custom"},
"store": true,
@@ -64,12 +142,29 @@ fn strips_store_for_compact_even_when_body_rules_handle_it() {
assert!(body.get("max_output_tokens").is_none());
assert!(body.get("store").is_none());
assert_eq!(body["instructions"], "Keep custom");
assert_eq!(body["metadata"]["mode"], "custom");
assert_eq!(body["top_p"], 0.5);
assert!(body.get("metadata").is_none());
assert!(body.get("top_p").is_none());
assert_eq!(body["parallel_tool_calls"], true);
assert!(body.as_object().is_some_and(|object| {
object.keys().all(|field| {
matches!(
field.as_str(),
"model"
| "input"
| "instructions"
| "tools"
| "parallel_tool_calls"
| "reasoning"
| "service_tier"
| "prompt_cache_key"
| "text"
)
})
}));
}
#[test]
fn injects_stable_prompt_cache_key_for_codex_requests() {
fn does_not_synthesize_prompt_cache_key_from_api_key_identity() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
@@ -83,83 +178,478 @@ fn injects_stable_prompt_cache_key_for_codex_requests() {
Some("key-123"),
);
assert!(body.get("prompt_cache_key").is_none());
}
#[test]
fn adapts_generic_prompt_cache_key_to_codex_native_identity() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
"prompt_cache_key": "ltm-pc-v2-5557e02f5c9b447a97673ba330dbe77a",
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
let expected_identity = "d9c5d122-7c1c-5fb1-ba9d-656062eda44e";
assert_eq!(body["prompt_cache_key"], expected_identity);
assert_eq!(body["client_metadata"]["session_id"], expected_identity);
assert_eq!(body["client_metadata"]["thread_id"], expected_identity);
}
#[test]
fn preserves_native_codex_cache_identity_and_metadata() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
"prompt_cache_key": "guardian:parent-thread",
"client_metadata": {
"session_id": "native-session",
"thread_id": "native-thread",
"turn_id": "native-turn"
}
});
let expected = body.clone();
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
assert_eq!(body["prompt_cache_key"], expected["prompt_cache_key"]);
assert_eq!(body["client_metadata"], expected["client_metadata"]);
}
#[test]
fn preserves_uuid_prompt_cache_key_while_completing_codex_identity() {
let identity = "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3";
let mut body = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": identity
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
None,
);
assert_eq!(body["prompt_cache_key"], identity);
assert_eq!(body["client_metadata"]["session_id"], identity);
assert_eq!(body["client_metadata"]["thread_id"], identity);
}
#[test]
fn keeps_codex_prompt_cache_domains_distinct() {
let mut first = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "tenant-a"
});
let mut second = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "tenant-b"
});
for body in [&mut first, &mut second] {
apply_codex_openai_responses_special_body_edits(
body,
"codex",
"openai:responses",
None,
None,
);
}
assert_ne!(first["prompt_cache_key"], second["prompt_cache_key"]);
assert_eq!(
body["prompt_cache_key"],
"53363264-dbb0-5f9d-b9c7-3e92c45c5bdf"
first["prompt_cache_key"],
first["client_metadata"]["session_id"]
);
assert_eq!(
second["prompt_cache_key"],
second["client_metadata"]["session_id"]
);
}
#[test]
fn keeps_existing_prompt_cache_key_for_codex_requests() {
let mut body = json!({
"model": "gpt-5",
fn completes_partial_and_null_codex_client_metadata() {
let mut partial = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "existing-key",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"thread_id": "native-thread",
"caller": "sdk"
}
});
let mut null_metadata = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": null
});
let mut null_session = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"session_id": null,
"thread_id": null,
"caller": "sdk"
}
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
for body in [&mut partial, &mut null_metadata, &mut null_session] {
apply_codex_openai_responses_special_body_edits(
body,
"codex",
"openai:responses",
None,
None,
);
}
assert_eq!(body["prompt_cache_key"], "existing-key");
assert_eq!(partial["client_metadata"]["thread_id"], "native-thread");
assert_eq!(partial["client_metadata"]["caller"], "sdk");
assert_eq!(
partial["client_metadata"]["session_id"],
partial["prompt_cache_key"]
);
assert_eq!(
null_metadata["client_metadata"]["session_id"],
null_metadata["prompt_cache_key"]
);
assert_eq!(
null_metadata["client_metadata"]["thread_id"],
null_metadata["prompt_cache_key"]
);
assert_eq!(
null_session["client_metadata"]["session_id"],
null_session["prompt_cache_key"]
);
assert_eq!(
null_session["client_metadata"]["thread_id"],
null_session["prompt_cache_key"]
);
assert_eq!(null_session["client_metadata"]["caller"], "sdk");
}
#[test]
fn injects_chatgpt_account_id_and_session_headers_for_codex_requests() {
fn leaves_malformed_codex_client_metadata_unchanged() {
let mut body = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": "invalid"
});
let mut malformed_fields = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"session_id": 42,
"thread_id": ""
}
});
let expected_malformed_metadata = malformed_fields["client_metadata"].clone();
for candidate in [&mut body, &mut malformed_fields] {
apply_codex_openai_responses_special_body_edits(
candidate,
"codex",
"openai:responses",
None,
None,
);
}
assert_eq!(body["prompt_cache_key"], "generic-affinity");
assert_eq!(body["client_metadata"], "invalid");
assert_eq!(malformed_fields["prompt_cache_key"], "generic-affinity");
assert_eq!(
malformed_fields["client_metadata"],
expected_malformed_metadata
);
}
#[test]
fn limits_prompt_cache_identity_adaptation_to_codex_responses_family() {
let original = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity"
});
let mut standard_openai = original.clone();
let mut codex_compact = original.clone();
apply_codex_openai_responses_special_body_edits(
&mut standard_openai,
"openai",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_special_body_edits(
&mut codex_compact,
"codex",
"openai:responses:compact",
None,
None,
);
assert_eq!(standard_openai, original);
assert_ne!(codex_compact["prompt_cache_key"], "generic-affinity");
assert!(codex_compact.get("client_metadata").is_none());
}
#[test]
fn chat_to_codex_responses_adapts_prompt_cache_identity_end_to_end() {
let body = json!({
"model": "gpt-5.6-luna",
"messages": [{"role": "user", "content": "hello"}],
"prompt_cache_key": "ltm-pc-v2-5557e02f5c9b447a97673ba330dbe77a"
});
let provider_request_body = build_cross_format_openai_responses_request_body(
&body,
"gpt-5.6-luna",
"openai:chat",
"openai:responses",
true,
false,
"codex",
None,
None,
&HeaderMap::new(),
false,
)
.expect("chat to Codex Responses request should build");
let expected_identity = "d9c5d122-7c1c-5fb1-ba9d-656062eda44e";
assert_eq!(provider_request_body["prompt_cache_key"], expected_identity);
assert_eq!(
provider_request_body["client_metadata"]["session_id"],
expected_identity
);
assert_eq!(
provider_request_body["client_metadata"]["thread_id"],
expected_identity
);
let mut provider_request_headers = BTreeMap::new();
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
&HeaderMap::new(),
"codex",
"openai:responses",
Some("trace-codex-cache-identity"),
None,
);
apply_codex_openai_responses_identity_headers(
&mut provider_request_headers,
&provider_request_body,
"codex",
"openai:responses",
);
assert_eq!(
provider_request_headers
.get("session-id")
.map(String::as_str),
Some(expected_identity)
);
assert_eq!(
provider_request_headers
.get("thread-id")
.map(String::as_str),
Some(expected_identity)
);
}
#[test]
fn projects_uuid_prompt_cache_identity_into_missing_session_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
});
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
Some("trace-codex-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(headers.get("x-client-request-id"), None);
assert_eq!(
headers.get("user-agent"),
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
}
#[test]
fn projects_native_codex_metadata_for_non_uuid_cache_overrides() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "guardian:parent-thread",
"client_metadata": {
"session_id": "019f687b-8e92-7842-9631-d5bf0dba0a3b",
"thread_id": "019f6d20-1111-7222-8333-444455556666"
}
});
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("session-id").map(String::as_str),
Some("019f687b-8e92-7842-9631-d5bf0dba0a3b")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("019f6d20-1111-7222-8333-444455556666")
);
}
#[test]
fn leaves_non_native_cache_keys_out_of_identity_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "generic-cache-key"
});
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert!(!headers.contains_key("session-id"));
assert!(!headers.contains_key("thread-id"));
}
#[test]
fn leaves_malformed_native_metadata_out_of_identity_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
"client_metadata": {
"session_id": 42,
"thread_id": ""
}
});
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert!(!headers.contains_key("session-id"));
assert!(!headers.contains_key("thread-id"));
}
#[test]
fn injects_only_codex_client_headers_for_images_requests() {
let mut headers = BTreeMap::new();
apply_codex_openai_special_headers(
&mut headers,
&json!({
"model": "gpt-image-2",
"prompt": "draw a city"
}),
&HeaderMap::new(),
"codex",
"openai:image",
Some("trace-codex-image-123"),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(
headers.get("x-client-request-id"),
Some(&"trace-codex-123".to_string())
);
assert_eq!(
headers.get("user-agent"),
Some(
&"codex-tui/0.122.0 (Mac OS 15.2.0; arm64) vscode/2.6.11 (codex-tui; 0.122.0)"
.to_string()
)
);
assert_eq!(headers.get("originator"), Some(&"codex-tui".to_string()));
assert_eq!(
headers.get("session_id"),
Some(&"ab5ecce4f0d110fe".to_string())
);
assert_eq!(
headers.get("conversation_id"),
Some(&"ab5ecce4f0d110fe".to_string())
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
for name in ["x-client-request-id", "session-id", "thread-id"] {
assert!(
!headers.contains_key(name),
"unexpected Images header: {name}"
);
}
}
#[test]
fn respects_existing_codex_request_and_session_headers() {
fn preserves_client_context_headers_and_enforces_codex_provider_identity() {
let mut headers = BTreeMap::new();
headers.insert(
"x-client-request-id".to_string(),
"kept-by-rule-request".to_string(),
);
headers.insert("session_id".to_string(), "kept-by-rule".to_string());
headers.insert("session-id".to_string(), "kept-by-rule-session".to_string());
headers.insert("thread-id".to_string(), "kept-by-rule-thread".to_string());
headers.insert(
"chatgpt-account-id".to_string(),
"configured-spoof".to_string(),
);
headers.insert(
"x-openai-fedramp".to_string(),
"configured-false".to_string(),
);
headers.insert(
"User-Agent".to_string(),
"AsyncOpenAI/Python 2.44.0".to_string(),
);
headers.insert("ORIGINATOR".to_string(), "sdk-client".to_string());
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
@@ -170,12 +660,12 @@ fn respects_existing_codex_request_and_session_headers() {
HeaderValue::from_static("user-specified-request"),
);
original_headers.insert(
"session_id",
"session-id",
HeaderValue::from_static("user-specified-session"),
);
original_headers.insert(
"conversation_id",
HeaderValue::from_static("user-specified-conversation"),
"thread-id",
HeaderValue::from_static("user-specified-thread"),
);
original_headers.insert(
"user-agent",
@@ -185,64 +675,105 @@ fn respects_existing_codex_request_and_session_headers() {
"originator",
HeaderValue::from_static("user-specified-originator"),
);
original_headers.insert("version", HeaderValue::from_static("user-version"));
original_headers.insert("x-openai-fedramp", HeaderValue::from_static("user-fedramp"));
original_headers.insert(
"chatgpt-account-id",
HeaderValue::from_static("user-account"),
);
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&original_headers,
"codex",
"openai:responses",
Some("trace-codex-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("x-client-request-id"),
Some(&"kept-by-rule-request".to_string())
);
assert!(!headers.contains_key("user-agent"));
assert!(!headers.contains_key("originator"));
assert_eq!(headers.get("session_id"), Some(&"kept-by-rule".to_string()));
assert!(!headers.contains_key("conversation_id"));
assert_eq!(
headers.get("user-agent"),
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("user-agent"))
.count(),
1
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("originator"))
.count(),
1
);
assert!(!headers.contains_key("version"));
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session-id"),
Some(&"kept-by-rule-session".to_string())
);
assert_eq!(
headers.get("thread-id"),
Some(&"kept-by-rule-thread".to_string())
);
}
#[test]
fn skips_conversation_id_for_compact_codex_requests() {
fn compact_projects_uuid_prompt_cache_identity_into_session_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
});
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses:compact",
Some("trace-codex-compact-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(
&mut headers,
&body,
"codex",
"openai:responses:compact",
);
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(
headers.get("x-client-request-id"),
Some(&"trace-codex-compact-123".to_string())
);
assert_eq!(headers.get("x-client-request-id"), None);
assert_eq!(
headers.get("user-agent"),
Some(
&"codex-tui/0.122.0 (Mac OS 15.2.0; arm64) vscode/2.6.11 (codex-tui; 0.122.0)"
.to_string()
)
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex-tui".to_string()));
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session_id"),
Some(&"ab5ecce4f0d110fe".to_string())
headers.get("session-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert!(!headers.contains_key("conversation_id"));
}
@@ -0,0 +1,386 @@
use serde_json::{json, Value};
pub(crate) fn is_deepseek_provider(provider_type: &str, base_url: &str) -> bool {
let provider_type = provider_type.trim().to_ascii_lowercase();
if matches!(
provider_type.as_str(),
"deepseek" | "deepseek_openai" | "deepseek_anthropic" | "deepseek_compatible"
) {
return true;
}
let host = base_url_host(base_url);
host == "deepseek.com" || host.ends_with(".deepseek.com")
}
pub(crate) fn apply_deepseek_tool_call_thinking_compat(
provider_request_body: &mut Value,
provider_type: &str,
base_url: &str,
provider_api_format: &str,
original_request_body: Option<&Value>,
) {
if !is_deepseek_provider(provider_type, base_url) {
return;
}
match crate::ai_serving::normalize_api_format_alias(provider_api_format).as_str() {
"openai:chat" => {
apply_deepseek_openai_chat_thinking_compat(provider_request_body, original_request_body)
}
"claude:messages" => apply_deepseek_claude_messages_thinking_compat(
provider_request_body,
original_request_body,
),
_ => {}
}
}
fn base_url_host(base_url: &str) -> String {
let lower = base_url.trim().to_ascii_lowercase();
let without_scheme = lower
.split_once("://")
.map(|(_, rest)| rest)
.unwrap_or(lower.as_str());
let without_userinfo = without_scheme
.rsplit_once('@')
.map(|(_, host)| host)
.unwrap_or(without_scheme);
without_userinfo
.split(['/', '?', '#'])
.next()
.unwrap_or_default()
.split(':')
.next()
.unwrap_or_default()
.to_string()
}
fn source_disables_thinking(
original_request_body: Option<&Value>,
provider_request_body: &Value,
) -> bool {
request_explicitly_disables_thinking(provider_request_body)
|| original_request_body.is_some_and(request_explicitly_disables_thinking)
}
fn request_explicitly_disables_thinking(body: &Value) -> bool {
thinking_type(body).is_some_and(|value| value.eq_ignore_ascii_case("disabled"))
|| reasoning_effort(body).is_some_and(|value| value.eq_ignore_ascii_case("none"))
}
fn thinking_type(body: &Value) -> Option<&str> {
body.get("thinking")
.and_then(Value::as_object)
.and_then(|thinking| thinking.get("type"))
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
}
fn reasoning_effort(body: &Value) -> Option<&str> {
body.get("reasoning_effort")
.and_then(Value::as_str)
.or_else(|| {
body.get("reasoning")
.and_then(Value::as_object)
.and_then(|reasoning| reasoning.get("effort"))
.and_then(Value::as_str)
})
.map(str::trim)
.filter(|value| !value.is_empty())
}
fn set_deepseek_thinking_type(body: &mut Value, thinking_type: &str) {
let Some(object) = body.as_object_mut() else {
return;
};
match object.get_mut("thinking") {
Some(Value::Object(thinking)) => {
thinking.insert("type".to_string(), Value::String(thinking_type.to_string()));
}
_ => {
object.insert(
"thinking".to_string(),
json!({
"type": thinking_type,
}),
);
}
}
}
fn apply_deepseek_openai_chat_thinking_compat(
provider_request_body: &mut Value,
original_request_body: Option<&Value>,
) {
let disabled = source_disables_thinking(original_request_body, provider_request_body);
set_deepseek_thinking_type(
provider_request_body,
if disabled { "disabled" } else { "enabled" },
);
let Some(object) = provider_request_body.as_object_mut() else {
return;
};
if disabled {
if reasoning_effort(&Value::Object(object.clone()))
.is_some_and(|value| value.eq_ignore_ascii_case("none"))
{
object.remove("reasoning_effort");
}
return;
}
let Some(messages) = object.get_mut("messages").and_then(Value::as_array_mut) else {
return;
};
for message in messages {
let Some(message_object) = message.as_object_mut() else {
continue;
};
let is_assistant = message_object
.get("role")
.and_then(Value::as_str)
.is_some_and(|role| role.trim().eq_ignore_ascii_case("assistant"));
if !is_assistant {
continue;
}
if message_object
.get("reasoning_content")
.is_some_and(|value| !value.is_null())
{
continue;
}
message_object.insert(
"reasoning_content".to_string(),
Value::String(String::new()),
);
}
}
fn apply_deepseek_claude_messages_thinking_compat(
provider_request_body: &mut Value,
original_request_body: Option<&Value>,
) {
if source_disables_thinking(original_request_body, provider_request_body) {
set_deepseek_thinking_type(provider_request_body, "disabled");
return;
}
let Some(messages) = provider_request_body
.get_mut("messages")
.and_then(Value::as_array_mut)
else {
return;
};
for message in messages {
let Some(message_object) = message.as_object_mut() else {
continue;
};
let is_assistant = message_object
.get("role")
.and_then(Value::as_str)
.is_some_and(|role| role.trim().eq_ignore_ascii_case("assistant"));
if !is_assistant {
continue;
}
ensure_claude_assistant_message_has_thinking_block(message_object);
}
}
fn ensure_claude_assistant_message_has_thinking_block(
message: &mut serde_json::Map<String, Value>,
) {
let thinking_block = json!({
"type": "thinking",
"thinking": "",
});
match message.get_mut("content") {
Some(Value::Array(blocks)) => {
if blocks.iter().any(is_claude_thinking_block) {
return;
}
blocks.insert(0, thinking_block);
}
Some(Value::String(text)) => {
let text = std::mem::take(text);
message.insert(
"content".to_string(),
Value::Array(vec![
thinking_block,
json!({
"type": "text",
"text": text,
}),
]),
);
}
Some(Value::Null) | None => {
message.insert("content".to_string(), Value::Array(vec![thinking_block]));
}
Some(other) => {
let existing = std::mem::take(other);
message.insert(
"content".to_string(),
Value::Array(vec![thinking_block, existing]),
);
}
}
}
fn is_claude_thinking_block(block: &Value) -> bool {
block
.get("type")
.and_then(Value::as_str)
.is_some_and(|block_type| block_type.trim().eq_ignore_ascii_case("thinking"))
}
#[cfg(test)]
mod tests {
use serde_json::json;
use super::{apply_deepseek_tool_call_thinking_compat, is_deepseek_provider};
#[test]
fn detects_deepseek_provider_by_type_or_host() {
assert!(is_deepseek_provider(
"deepseek",
"https://relay.example.com"
));
assert!(is_deepseek_provider(
"custom",
"https://api.deepseek.com/v1"
));
assert!(!is_deepseek_provider(
"custom",
"https://example.com/deepseek"
));
}
#[test]
fn openai_chat_deepseek_adds_thinking_and_empty_reasoning_content() {
let mut body = json!({
"model": "deepseek-chat",
"messages": [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": null, "tool_calls": [{
"id": "call_1",
"type": "function",
"function": {"name": "lookup", "arguments": "{}"}
}]},
{"role": "tool", "tool_call_id": "call_1", "content": "{}"}
]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com/v1",
"openai:chat",
None,
);
assert_eq!(body["thinking"]["type"], "enabled");
assert_eq!(body["messages"][1]["reasoning_content"], "");
}
#[test]
fn openai_chat_deepseek_honors_disabled_thinking() {
let original = json!({"reasoning_effort": "none"});
let mut body = json!({
"model": "deepseek-chat",
"reasoning_effort": "none",
"messages": [
{"role": "assistant", "content": "hi"}
]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com/v1",
"openai:chat",
Some(&original),
);
assert_eq!(body["thinking"]["type"], "disabled");
assert!(body.get("reasoning_effort").is_none());
assert!(body["messages"][0].get("reasoning_content").is_none());
}
#[test]
fn claude_messages_deepseek_prepends_empty_thinking_block() {
let mut body = json!({
"model": "deepseek-3.2",
"messages": [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": [
{"type": "tool_use", "id": "call_1", "name": "lookup", "input": {}}
]}
]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com",
"claude:messages",
None,
);
assert_eq!(body["messages"][1]["content"][0]["type"], "thinking");
assert_eq!(body["messages"][1]["content"][0]["thinking"], "");
assert_eq!(body["messages"][1]["content"][1]["type"], "tool_use");
}
#[test]
fn claude_messages_deepseek_converts_string_assistant_content_to_blocks() {
let mut body = json!({
"model": "deepseek-3.2",
"messages": [{
"role": "assistant",
"content": "done"
}]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com",
"claude:messages",
None,
);
assert_eq!(body["messages"][0]["content"][0]["type"], "thinking");
assert_eq!(body["messages"][0]["content"][1]["type"], "text");
assert_eq!(body["messages"][0]["content"][1]["text"], "done");
}
#[test]
fn claude_messages_deepseek_preserves_existing_thinking_block() {
let mut body = json!({
"model": "deepseek-3.2",
"messages": [{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "plan", "signature": "sig"},
{"type": "text", "text": "answer"}
]
}]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com",
"claude:messages",
None,
);
assert_eq!(body["messages"][0]["content"].as_array().unwrap().len(), 2);
assert_eq!(body["messages"][0]["content"][0]["thinking"], "plan");
assert_eq!(body["messages"][0]["content"][0]["signature"], "sig");
}
}
@@ -178,7 +178,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalStandardSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -201,12 +201,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalStandardSyncAttemptSour
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalStandardStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -229,6 +244,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalStandardStreamAttempt
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalStandardSyncAttemptSource<'_> {
@@ -340,7 +370,7 @@ pub(crate) async fn maybe_build_sync_via_standard_family_payload(
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -390,7 +420,7 @@ pub(crate) async fn maybe_build_stream_via_standard_family_payload(
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -449,7 +479,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
return Ok(Vec::new());
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -524,7 +554,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
return Ok(Vec::new());
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -60,7 +60,9 @@ pub(super) async fn resolve_local_standard_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -119,8 +121,10 @@ pub(super) async fn materialize_local_standard_candidate_attempts(
);
let preselection = preselect_local_execution_candidates_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
false,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -243,9 +247,11 @@ pub(super) async fn build_local_standard_candidate_attempt_source<'a>(
let (source, candidate_count) =
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
spec_metadata.api_format,
&input.requested_model,
None,
spec_metadata.require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
@@ -338,8 +344,10 @@ async fn maybe_append_gemini_image_openai_image_preselection(
let image_preselection = preselect_local_execution_candidates_for_api_formats_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -9,12 +9,14 @@ use crate::ai_serving::planner::materialization_policy::{
};
use crate::ai_serving::planner::passthrough::maybe_build_local_same_format_provider_decision_payload_for_candidate;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_standard_spec_metadata;
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
@@ -74,10 +76,15 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
let Some(resolved) = resolve_local_standard_candidate_payload_parts(
state, parts, trace_id, body_json, input, &attempt, spec,
)
.await
.await?
else {
return Ok(None);
};
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
Some(body_json)
};
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
@@ -92,6 +99,7 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
"envelope_name".to_string(),
serde_json::Value::String(envelope_name.to_string()),
);
insert_native_client_envelope_name(&mut extra_fields, envelope_name, parts.uri.path());
}
let (execution_strategy, conversion_mode) = ai_local_execution_contract_for_formats(
spec_metadata.api_format,
@@ -129,9 +137,10 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -166,7 +175,9 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
envelope_name: _,
transport,
transport_profile: _,
request_redacted: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -176,6 +187,7 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: candidate.provider_name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -193,6 +205,8 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
@@ -201,7 +215,11 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -356,10 +374,13 @@ mod tests {
auth_snapshot: sample_auth_snapshot(),
required_capabilities: None,
request_auth_channel: None,
client_surface: None,
gateway_credential_carrier: None,
client_session_affinity: None,
routing_policy: None,
routing_trace_seed: None,
routing_context: None,
model_directive_policy: Default::default(),
}
}
@@ -431,6 +452,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "sk-upstream".to_string(),
decrypted_auth_config: None,
},
@@ -462,6 +484,7 @@ mod tests {
} else {
"gpt-4o-upstream".to_string()
},
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -486,6 +509,46 @@ mod tests {
}
}
fn sample_gemini_cli_attempt(candidate_index: u32) -> LocalExecutionCandidateAttempt {
let mut transport = sample_transport("gemini:generate_content", "endpoint-gemini-cli");
transport.provider.provider_type = "gemini_cli".to_string();
transport.provider.name = "gemini".to_string();
transport.endpoint.base_url = "https://cloudcode-pa.googleapis.com".to_string();
transport.endpoint.custom_path = Some("/v1internal:{action}".to_string());
transport.endpoint.endpoint_kind = Some("generate_content".to_string());
transport.key.auth_type = "bearer".to_string();
transport.key.api_formats = Some(vec!["gemini:generate_content".to_string()]);
transport.key.global_priority_by_format = Some(json!({
"gemini:generate_content": 1,
}));
transport.key.upstream_metadata = Some(json!({
"gemini_cli": {
"project_id": "test-project"
}
}));
let mut candidate = sample_candidate("gemini:generate_content", "endpoint-gemini-cli");
candidate.provider_name = "gemini".to_string();
candidate.provider_type = "gemini_cli".to_string();
candidate.key_auth_type = "bearer".to_string();
candidate.selected_provider_model_name = "gemini-2.5-pro".to_string();
candidate.global_model_name = "gemini-2.5-pro".to_string();
LocalExecutionCandidateAttempt {
eligible: EligibleLocalExecutionCandidate {
kind: LocalExecutionCandidateKind::SingleKey,
candidate,
transport: Arc::new(transport),
provider_api_format: "gemini:generate_content".to_string(),
orchestration: LocalExecutionCandidateMetadata::default(),
ranking: None,
},
candidate_index,
retry_index: 0,
candidate_id: format!("candidate-{candidate_index}"),
}
}
fn claude_stream_spec() -> LocalStandardSpec {
LocalStandardSpec {
api_format: "claude:messages",
@@ -577,4 +640,90 @@ mod tests {
Some("bidirectional")
);
}
#[tokio::test]
async fn standard_family_wraps_gemini_cli_cross_format_body_in_v1internal_envelope() {
let state = crate::AppState::new().expect("state should build");
let request = http::Request::builder()
.method("POST")
.uri("/v1/chat/completions")
.header(http::header::CONTENT_TYPE, "application/json")
.body(())
.expect("request should build");
let (parts, _) = request.into_parts();
let body_json = json!({
"model": "gemini-2.5-pro",
"messages": [{"role": "user", "content": "hello"}],
"temperature": 0.2,
"stream": true
});
let mut input = sample_input();
input.requested_model = "gemini-2.5-pro".to_string();
let spec = LocalStandardSpec {
api_format: "openai:chat",
decision_kind: "openai_chat_stream",
report_kind: "openai_chat_stream_success",
family: LocalStandardSourceFamily::Standard,
mode: LocalStandardSourceMode::Chat,
require_streaming: true,
};
let payload = maybe_build_local_standard_decision_payload_for_candidate(
&state,
&parts,
"trace-gemini-cli-cross-format",
&body_json,
&input,
sample_gemini_cli_attempt(0),
spec,
)
.await
.expect("cross-format candidate should not fail routing mutation")
.expect("gemini_cli candidate should build a payload");
assert_eq!(
payload.upstream_url.as_deref(),
Some("https://cloudcode-pa.googleapis.com/v1internal:streamGenerateContent?alt=sse")
);
assert_eq!(
payload.execution_strategy.as_deref(),
Some("local_cross_format")
);
assert_eq!(
payload.provider_api_format.as_deref(),
Some("gemini:generate_content")
);
assert_eq!(
payload
.provider_request_headers
.get("user-agent")
.map(String::as_str),
Some(crate::ai_serving::transport::GEMINI_CLI_USER_AGENT)
);
let provider_body = payload
.provider_request_body
.as_ref()
.expect("provider request body should be present");
assert_eq!(provider_body["model"], "gemini-2.5-pro");
assert_eq!(provider_body["project"], "test-project");
assert_eq!(
provider_body["user_prompt_id"],
"trace-gemini-cli-cross-format"
);
assert!(provider_body.get("contents").is_none());
assert!(provider_body.get("generationConfig").is_none());
assert!(provider_body["request"].get("contents").is_some());
let report_context = payload
.report_context
.as_ref()
.expect("report context should be present");
assert_eq!(
report_context
.get("envelope_name")
.and_then(|value| value.as_str()),
Some(crate::ai_serving::transport::GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME)
);
}
}
@@ -12,9 +12,19 @@ use crate::ai_serving::planner::common::{
endpoint_config_forces_body_stream_field, enforce_provider_body_stream_policy,
request_requires_body_stream_field, resolve_upstream_is_stream_for_provider,
};
use crate::ai_serving::planner::gemini_cli::{
build_gemini_cli_v1internal_provider_request, GeminiCliV1InternalRequestError,
GeminiCliV1InternalRequestInput,
};
use crate::ai_serving::planner::redaction::{
request_identity_response_encoding_when_redacted, resolve_provider_chat_pii_redaction,
};
use crate::ai_serving::planner::spec_metadata::local_standard_spec_metadata;
use crate::ai_serving::planner::standard::{
apply_codex_openai_responses_special_headers, request_body_build_failure_extra_data,
apply_codex_openai_special_headers, apply_deepseek_tool_call_thinking_compat,
codex_model_capabilities_for_transport, is_deepseek_provider,
openai_provider_request_contract_failure_extra_data, request_body_build_failure_extra_data,
request_conversion_failure_extra_data,
};
use crate::ai_serving::transport::kiro::{
build_kiro_provider_headers, build_kiro_provider_request_body,
@@ -26,17 +36,20 @@ use crate::ai_serving::transport::{
build_openai_image_headers, build_openai_image_upstream_url,
build_standard_provider_request_headers, build_windsurf_cascade_headers,
build_windsurf_cascade_request_body, build_windsurf_cascade_upstream_url,
is_windsurf_provider_transport,
is_gemini_cli_provider_transport, is_windsurf_provider_transport,
local_windsurf_request_transport_unsupported_reason_with_network,
openai_image_transport_unsupported_reason, resolve_grok_session_auth,
resolve_openai_image_auth, GrokHeaderInput, ProviderOpenAiImageHeadersInput,
StandardProviderRequestHeadersInput, GROK_CHAT_PATH, WINDSURF_ENVELOPE_NAME,
StandardProviderRequestHeadersInput, GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME, GROK_CHAT_PATH,
WINDSURF_ENVELOPE_NAME,
};
use crate::ai_serving::{
build_openai_image_request_body_from_gemini_image_request, gemini_request_is_image_generation,
project_codex_openai_image_api_request_body, project_openai_image_api_request_body,
CandidateFailureDiagnostic, GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
OpenAiImageOperation,
};
use crate::AppState;
use crate::{AppState, GatewayError};
use super::payload::{
mark_skipped_local_standard_candidate, mark_skipped_local_standard_candidate_with_extra_data,
@@ -44,6 +57,8 @@ use super::payload::{
};
use super::{LocalStandardCandidateAttempt, LocalStandardDecisionInput, LocalStandardSpec};
const OMITTED_THINKING_TEXT: &str = "Previous thinking omitted.";
pub(crate) struct LocalStandardCandidatePayloadParts {
pub(super) auth_header: String,
pub(super) auth_value: String,
@@ -56,6 +71,7 @@ pub(crate) struct LocalStandardCandidatePayloadParts {
pub(super) envelope_name: Option<&'static str>,
pub(super) transport: Arc<GatewayProviderTransportSnapshot>,
pub(super) transport_profile: Option<ResolvedTransportProfile>,
pub(super) request_redacted: bool,
}
fn is_grok_text_provider_api_format(provider_api_format: &str) -> bool {
@@ -114,8 +130,6 @@ fn sanitize_claude_thinking_block(block: Value) -> (Option<Value>, bool) {
}
fn sanitize_claude_message_content_for_non_native_thinking(content: &mut Value) -> bool {
const OMITTED_THINKING_TEXT: &str = "Previous thinking omitted.";
if content.is_object() {
let original = std::mem::take(content);
let (sanitized, changed) = sanitize_claude_thinking_block(original);
@@ -175,6 +189,81 @@ fn sanitize_claude_request_thinking_signatures_for_non_native(body_json: &mut Va
.unwrap_or(false)
}
fn remove_claude_redacted_thinking_block(block: Value) -> (Option<Value>, bool) {
let Some(object) = block.as_object() else {
return (Some(block), false);
};
let block_type = object
.get("type")
.and_then(Value::as_str)
.map(str::trim)
.unwrap_or_default();
if block_type == "redacted_thinking" {
return (None, true);
}
(Some(block), false)
}
fn sanitize_claude_message_content_for_deepseek_thinking(content: &mut Value) -> bool {
if content.is_object() {
let original = std::mem::take(content);
let (sanitized, changed) = remove_claude_redacted_thinking_block(original);
if changed {
*content = sanitized.unwrap_or_else(|| {
serde_json::json!({
"type": "text",
"text": OMITTED_THINKING_TEXT,
})
});
}
return changed;
}
let Some(blocks) = content.as_array_mut() else {
return false;
};
let original_blocks = std::mem::take(blocks);
let mut changed = false;
let mut sanitized_blocks = Vec::with_capacity(original_blocks.len());
for block in original_blocks {
let (sanitized, block_changed) = remove_claude_redacted_thinking_block(block);
changed |= block_changed;
if let Some(sanitized) = sanitized {
sanitized_blocks.push(sanitized);
}
}
if changed && sanitized_blocks.is_empty() {
sanitized_blocks.push(serde_json::json!({
"type": "text",
"text": OMITTED_THINKING_TEXT,
}));
}
*blocks = sanitized_blocks;
changed
}
fn sanitize_claude_request_redacted_thinking_for_deepseek(body_json: &mut Value) -> bool {
body_json
.get_mut("messages")
.and_then(Value::as_array_mut)
.map(|messages| {
messages.iter_mut().fold(false, |changed, message| {
let is_assistant = message
.get("role")
.and_then(Value::as_str)
.is_some_and(|role| role.trim().eq_ignore_ascii_case("assistant"));
if !is_assistant {
return changed;
}
let content_changed = message
.get_mut("content")
.is_some_and(sanitize_claude_message_content_for_deepseek_thinking);
changed || content_changed
})
})
.unwrap_or(false)
}
fn apply_non_native_claude_thinking_signature_compat(
provider_request_body: &mut Value,
provider_api_format: &str,
@@ -183,6 +272,13 @@ fn apply_non_native_claude_thinking_signature_compat(
if crate::ai_serving::normalize_api_format_alias(provider_api_format) != "claude:messages" {
return;
}
if is_deepseek_provider(
transport.provider.provider_type.as_str(),
transport.endpoint.base_url.as_str(),
) {
let _ = sanitize_claude_request_redacted_thinking_for_deepseek(provider_request_body);
return;
}
if provider_preserves_claude_thinking_signatures(
transport.provider.provider_type.as_str(),
transport.endpoint.base_url.as_str(),
@@ -201,7 +297,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
input: &LocalStandardDecisionInput,
attempt: &LocalStandardCandidateAttempt,
spec: LocalStandardSpec,
) -> Option<LocalStandardCandidatePayloadParts> {
) -> Result<Option<LocalStandardCandidatePayloadParts>, GatewayError> {
let spec_metadata = local_standard_spec_metadata(spec);
let planner_state = crate::ai_serving::PlannerAppState::new(state);
let candidate = &attempt.eligible.candidate;
@@ -218,10 +314,18 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
&& provider_api_format == "openai:image"
&& gemini_request_is_image_generation(body_json)
{
return resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
state, parts, trace_id, body_json, input, attempt,
)
.await;
return Ok(
resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
state,
parts,
trace_id,
body_json,
input,
attempt,
spec_metadata.require_streaming,
)
.await,
);
}
let is_kiro_claude_cli = is_kiro_claude_messages_transport(transport, provider_api_format);
if is_grok && is_grok_text_provider_api_format(provider_api_format) {
@@ -250,10 +354,21 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
skip_reason,
)
.await;
return None;
return Ok(None);
}
};
let redaction = resolve_provider_chat_pii_redaction(
state,
parts,
body_json,
&input.auth_context,
spec_metadata.api_format,
&attempt.candidate_id,
)
.await?;
let body_json = redaction.body_json.as_ref();
let mut provider_request_body = body_json.clone();
if let Some(object) = provider_request_body.as_object_mut() {
object.insert(
@@ -279,7 +394,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
);
let upstream_url = build_grok_upstream_url(transport, GROK_CHAT_PATH);
let Some(provider_request_headers) = build_grok_browser_headers(GrokHeaderInput {
let Some(mut provider_request_headers) = build_grok_browser_headers(GrokHeaderInput {
transport,
transport_profile: transport_profile.as_ref(),
request_headers: Some(effective_headers),
@@ -304,10 +419,14 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
redaction.redacted,
);
return Some(LocalStandardCandidatePayloadParts {
return Ok(Some(LocalStandardCandidatePayloadParts {
auth_header: prepared_candidate.auth_header,
auth_value: prepared_candidate.auth_value,
mapped_model: prepared_candidate.mapped_model,
@@ -319,7 +438,8 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
envelope_name: None,
transport: Arc::clone(transport),
transport_profile,
});
request_redacted: redaction.redacted,
}));
}
if !crate::ai_serving::request_pair_allowed_for_transport(
@@ -327,7 +447,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
spec_metadata.api_format,
provider_api_format,
) {
return None;
return Ok(None);
}
let is_windsurf_cascade =
@@ -352,7 +472,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
skip_reason,
)
.await;
return None;
return Ok(None);
}
let oauth_context = OauthPreparationContext {
@@ -380,7 +500,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
"transport_auth_unavailable",
)
.await;
return None;
return Ok(None);
}
}
} else {
@@ -405,7 +525,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
skip_reason,
)
.await;
return None;
return Ok(None);
}
}
} else {
@@ -430,7 +550,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
skip_reason,
)
.await;
return None;
return Ok(None);
}
}
};
@@ -444,13 +564,45 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
);
let force_body_stream_field =
endpoint_config_forces_body_stream_field(transport.endpoint.config.as_ref());
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
state,
provider_api_format,
Some(&input.requested_model),
)
.await;
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(provider_api_format, Some(&input.requested_model));
let model_directive_mapping = match model_directive_resolution
.mapping_patch_for_mapped_model(&prepared_candidate.mapped_model)
{
Ok(mapping) => mapping,
Err(skip_reason) => {
mark_skipped_local_standard_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
};
crate::ai_serving::hydrate_openai_response_history(
state.runtime_state(),
body_json,
spec_metadata.api_format,
provider_api_format,
input.auth_context.api_key_id.as_str(),
)
.await?;
let redaction = resolve_provider_chat_pii_redaction(
state,
parts,
body_json,
&input.auth_context,
spec_metadata.api_format,
&attempt.candidate_id,
)
.await?;
let body_json = redaction.body_json.as_ref();
let mut provider_request_body =
match crate::ai_serving::planner::standard::build_standard_request_body_with_model_directives_and_request_headers(
body_json,
@@ -467,7 +619,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
},
Some(input.auth_context.api_key_id.as_str()),
Some(effective_headers),
enable_model_directives,
false,
) {
Some(body) => body,
None => {
@@ -479,14 +631,18 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
request_body_build_failure_extra_data(
request_conversion_failure_extra_data(
body_json,
spec_metadata.api_format,
provider_api_format,
Some(prepared_candidate.mapped_model.as_str()),
Some(parts.uri.path()),
upstream_is_stream,
"standard_family_request_conversion",
),
)
.await;
return None;
return Ok(None);
}
};
enforce_provider_body_stream_policy(
@@ -516,25 +672,22 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
}
apply_non_native_claude_thinking_signature_compat(
&mut provider_request_body,
provider_api_format,
transport,
);
if let Some(mapping) =
crate::system_features::reasoning_model_directive_mapping_for_api_format_and_model(
state,
provider_api_format,
Some(&input.requested_model),
)
.await
{
crate::ai_serving::apply_model_directive_mapping_patch(
&mut provider_request_body,
&mapping,
);
apply_deepseek_tool_call_thinking_compat(
&mut provider_request_body,
transport.provider.provider_type.as_str(),
transport.endpoint.base_url.as_str(),
provider_api_format,
Some(body_json),
);
if let Some(mapping) = model_directive_mapping.as_ref() {
crate::ai_serving::apply_model_directive_mapping_patch(&mut provider_request_body, mapping);
// Directive mapping is a deep-merge patch and may overwrite/add `stream`;
// re-enforce stream-field policy afterward.
enforce_provider_body_stream_policy(
@@ -564,17 +717,79 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
}
apply_non_native_claude_thinking_signature_compat(
&mut provider_request_body,
provider_api_format,
transport,
);
apply_deepseek_tool_call_thinking_compat(
&mut provider_request_body,
transport.provider.provider_type.as_str(),
transport.endpoint.base_url.as_str(),
provider_api_format,
Some(body_json),
);
}
let normalized_provider_api_format =
crate::ai_serving::normalize_api_format_alias(provider_api_format);
if matches!(
normalized_provider_api_format.as_str(),
"openai:chat" | "openai:responses" | "openai:responses:compact"
) {
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(input.requested_model.as_str());
let codex_model_capabilities = codex_model_capabilities_for_transport(
transport,
provider_api_format,
prepared_candidate.mapped_model.as_str(),
source_model,
);
if let Err(violation) =
crate::ai_serving::finalize_openai_provider_request_with_codex_model_capabilities(
&mut provider_request_body,
crate::ai_serving::OpenAiProviderRequestFinalization {
source_api_format: spec_metadata.api_format,
provider_api_format,
provider_type: transport.provider.provider_type.as_str(),
provider_model: prepared_candidate.mapped_model.as_str(),
source_model,
body_rules: transport.endpoint.body_rules.as_ref(),
upstream_is_stream,
require_body_stream_field: request_requires_body_stream_field(
body_json,
force_body_stream_field,
),
},
codex_model_capabilities.as_ref(),
)
{
mark_skipped_local_standard_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
Some(openai_provider_request_contract_failure_extra_data(
&violation,
spec_metadata.api_format,
provider_api_format,
"standard_family_request_finalization",
)),
)
.await;
return Ok(None);
}
}
if let Some(kiro_auth) = kiro_auth.as_ref() {
return build_kiro_cross_format_payload_parts(
return Ok(build_kiro_cross_format_payload_parts(
state,
parts,
trace_id,
@@ -589,11 +804,12 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
provider_request_body,
upstream_is_stream,
kiro_auth,
redaction.redacted,
)
.await;
.await);
}
if is_windsurf_cascade {
return build_windsurf_cross_format_payload_parts(
return Ok(build_windsurf_cross_format_payload_parts(
state,
parts,
trace_id,
@@ -607,8 +823,32 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
prepared_candidate.auth_value,
provider_request_body,
upstream_is_stream,
redaction.redacted,
)
.await;
.await);
}
if normalized_provider_api_format == "gemini:generate_content"
&& is_gemini_cli_provider_transport(transport)
{
return Ok(build_gemini_cli_cross_format_payload_parts(
state,
parts,
trace_id,
body_json,
input,
attempt,
transport,
spec_metadata.api_format,
provider_api_format,
prepared_candidate.mapped_model,
prepared_candidate.auth_header,
prepared_candidate.auth_value,
provider_request_body,
upstream_is_stream,
redaction.redacted,
)
.await);
}
let upstream_url = match crate::ai_serving::planner::standard::build_standard_upstream_url(
@@ -636,7 +876,7 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
}
};
let Some(resolved_headers) =
@@ -669,10 +909,10 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
let mut provider_request_headers = resolved_headers.headers;
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
@@ -681,8 +921,12 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
Some(trace_id),
transport.key.decrypted_auth_config.as_deref(),
);
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
redaction.redacted,
);
Some(LocalStandardCandidatePayloadParts {
Ok(Some(LocalStandardCandidatePayloadParts {
auth_header: resolved_headers.auth_header,
auth_value: resolved_headers.auth_value,
mapped_model: prepared_candidate.mapped_model,
@@ -694,7 +938,8 @@ pub(crate) async fn resolve_local_standard_candidate_payload_parts(
envelope_name: None,
transport: Arc::clone(transport),
transport_profile: None,
})
request_redacted: redaction.redacted,
}))
}
fn apply_transport_request_body_semantics(
@@ -709,6 +954,144 @@ fn apply_transport_request_body_semantics(
)
}
#[allow(clippy::too_many_arguments)]
async fn build_gemini_cli_cross_format_payload_parts(
state: &AppState,
parts: &http::request::Parts,
trace_id: &str,
original_body_json: &serde_json::Value,
input: &LocalStandardDecisionInput,
attempt: &LocalStandardCandidateAttempt,
transport: &Arc<GatewayProviderTransportSnapshot>,
client_api_format: &str,
provider_api_format: &str,
mapped_model: String,
auth_header: String,
auth_value: String,
gemini_request_body: Value,
upstream_is_stream: bool,
request_redacted: bool,
) -> Option<LocalStandardCandidatePayloadParts> {
let candidate = &attempt.eligible.candidate;
let effective_headers = input.effective_headers(&parts.headers);
let resolved =
match build_gemini_cli_v1internal_provider_request(GeminiCliV1InternalRequestInput {
state,
parts,
transport,
trace_id,
mapped_model: &mapped_model,
provider_api_format,
auth_header: &auth_header,
auth_value: &auth_value,
request_headers: effective_headers,
original_request_body: original_body_json,
gemini_request_body: &gemini_request_body,
upstream_is_stream,
})
.await
{
Ok(resolved) => resolved,
Err(GeminiCliV1InternalRequestError::ProjectUnavailable) => {
mark_skipped_local_standard_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"transport_auth_unavailable",
)
.await;
return None;
}
Err(GeminiCliV1InternalRequestError::EnvelopeUnsupported) => {
mark_skipped_local_standard_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
request_body_build_failure_extra_data(
original_body_json,
client_api_format,
provider_api_format,
),
)
.await;
return None;
}
Err(GeminiCliV1InternalRequestError::UpstreamUrlUnavailable) => {
mark_skipped_local_standard_candidate_with_failure_diagnostic(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"upstream_url_missing",
CandidateFailureDiagnostic::upstream_url_missing(
client_api_format,
provider_api_format,
"standard_family_gemini_cli_url",
),
)
.await;
return None;
}
Err(GeminiCliV1InternalRequestError::HeaderRulesApplyFailed) => {
mark_skipped_local_standard_candidate_with_failure_diagnostic(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"transport_header_rules_apply_failed",
CandidateFailureDiagnostic::header_rules_apply_failed(
client_api_format,
provider_api_format,
"standard_family_gemini_cli_headers",
),
)
.await;
return None;
}
};
let mut provider_request_headers = resolved.headers.headers;
apply_codex_openai_special_headers(
&mut provider_request_headers,
&resolved.body,
effective_headers,
resolved.transport.provider.provider_type.as_str(),
provider_api_format,
Some(trace_id),
resolved.transport.key.decrypted_auth_config.as_deref(),
);
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
request_redacted,
);
Some(LocalStandardCandidatePayloadParts {
auth_header: resolved.headers.auth_header,
auth_value: resolved.headers.auth_value,
mapped_model,
provider_api_format: provider_api_format.to_string(),
provider_request_body: resolved.body,
provider_request_headers,
upstream_url: resolved.upstream_url,
upstream_is_stream,
envelope_name: Some(GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME),
transport: resolved.transport,
transport_profile: None,
request_redacted,
})
}
#[allow(clippy::too_many_arguments)]
async fn build_windsurf_cross_format_payload_parts(
state: &AppState,
@@ -724,6 +1107,7 @@ async fn build_windsurf_cross_format_payload_parts(
auth_value: String,
openai_chat_request_body: Value,
upstream_is_stream: bool,
request_redacted: bool,
) -> Option<LocalStandardCandidatePayloadParts> {
let candidate = &attempt.eligible.candidate;
let effective_headers = input.effective_headers(&parts.headers);
@@ -779,7 +1163,7 @@ async fn build_windsurf_cross_format_payload_parts(
return None;
}
};
let provider_request_headers = match build_windsurf_cascade_headers(
let mut provider_request_headers = match build_windsurf_cascade_headers(
effective_headers,
&provider_request_body,
original_body_json,
@@ -808,6 +1192,10 @@ async fn build_windsurf_cross_format_payload_parts(
return None;
}
};
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
request_redacted,
);
Some(LocalStandardCandidatePayloadParts {
auth_header,
@@ -821,6 +1209,7 @@ async fn build_windsurf_cross_format_payload_parts(
envelope_name: Some(WINDSURF_ENVELOPE_NAME),
transport: Arc::clone(transport),
transport_profile: None,
request_redacted,
})
}
@@ -831,6 +1220,7 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
body_json: &serde_json::Value,
input: &LocalStandardDecisionInput,
attempt: &LocalStandardCandidateAttempt,
client_requires_streaming: bool,
) -> Option<LocalStandardCandidatePayloadParts> {
let client_api_format = "gemini:generate_content";
let provider_api_format = "openai:image";
@@ -906,17 +1296,60 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
return None;
};
let upstream_is_stream = true;
let upstream_url =
build_openai_image_upstream_url(transport, Some("/v1/images/generations"), None);
let upstream_is_stream = resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
provider_api_format,
client_requires_streaming && candidate.supports_streaming,
false,
);
let is_codex = transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex");
let mut provider_request_body = converted.body_json;
if upstream_is_stream {
provider_request_body
.as_object_mut()?
.insert("stream".to_string(), Value::Bool(true));
}
provider_request_body = project_openai_image_api_request_body(
&provider_request_body,
&prepared_candidate.mapped_model,
converted.operation,
crate::image_capabilities::openai_image_provider_max_generation_count_for_model(
transport.provider.provider_type.as_str(),
Some(prepared_candidate.mapped_model.as_str()),
),
)?;
if is_codex {
provider_request_body = project_codex_openai_image_api_request_body(
&provider_request_body,
converted.operation,
)?;
}
let request_path = match converted.operation {
OpenAiImageOperation::Generate => "/v1/images/generations",
OpenAiImageOperation::Edit => "/v1/images/edits",
};
let upstream_url = build_openai_image_upstream_url(transport, Some(request_path), None);
let effective_headers = input.effective_headers(&parts.headers);
let Some(mut provider_request_headers) =
build_openai_image_headers(ProviderOpenAiImageHeadersInput {
transport,
headers: effective_headers,
auth_header: &prepared_candidate.auth_header,
auth_value: &prepared_candidate.auth_value,
accept: if is_codex {
None
} else if upstream_is_stream {
Some("text/event-stream")
} else {
Some("application/json")
},
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &converted.body_json,
provider_request_body: &provider_request_body,
original_request_body: body_json,
})
else {
@@ -937,9 +1370,9 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
.await;
return None;
};
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&converted.body_json,
&provider_request_body,
effective_headers,
transport.provider.provider_type.as_str(),
provider_api_format,
@@ -952,13 +1385,14 @@ async fn resolve_local_gemini_image_to_openai_image_candidate_payload_parts(
auth_value: prepared_candidate.auth_value,
mapped_model: converted.mapped_model,
provider_api_format: provider_api_format.to_string(),
provider_request_body: converted.body_json,
provider_request_body,
provider_request_headers,
upstream_url,
upstream_is_stream,
envelope_name: None,
transport: Arc::clone(transport),
transport_profile: None,
request_redacted: false,
})
}
@@ -978,6 +1412,7 @@ async fn build_kiro_cross_format_payload_parts(
claude_request_body: Value,
upstream_is_stream: bool,
kiro_auth: &KiroRequestAuth,
request_redacted: bool,
) -> Option<LocalStandardCandidatePayloadParts> {
let candidate = &attempt.eligible.candidate;
let effective_headers = input.effective_headers(&parts.headers);
@@ -1036,7 +1471,7 @@ async fn build_kiro_cross_format_payload_parts(
return None;
}
};
let provider_request_headers = match build_kiro_provider_headers(KiroProviderHeadersInput {
let mut provider_request_headers = match build_kiro_provider_headers(KiroProviderHeadersInput {
headers: effective_headers,
provider_request_body: &provider_request_body,
original_request_body: original_body_json,
@@ -1066,6 +1501,10 @@ async fn build_kiro_cross_format_payload_parts(
return None;
}
};
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
request_redacted,
);
Some(LocalStandardCandidatePayloadParts {
auth_header,
@@ -1079,6 +1518,7 @@ async fn build_kiro_cross_format_payload_parts(
envelope_name: Some(KIRO_ENVELOPE_NAME),
transport: Arc::clone(transport),
transport_profile: None,
request_redacted,
})
}
@@ -1086,6 +1526,7 @@ async fn build_kiro_cross_format_payload_parts(
mod tests {
use super::{
provider_preserves_claude_thinking_signatures,
sanitize_claude_request_redacted_thinking_for_deepseek,
sanitize_claude_request_thinking_signatures_for_non_native,
};
use serde_json::json;
@@ -1149,6 +1590,46 @@ mod tests {
);
}
#[test]
fn deepseek_sanitizer_preserves_plain_thinking_but_removes_redacted() {
let mut body = json!({
"model": "claude-opus-4-1",
"messages": [{
"role": "assistant",
"content": [
{
"type": "thinking",
"thinking": "I should keep this short.",
"signature": "sig_123"
},
{
"type": "redacted_thinking",
"data": "opaque"
},
{
"type": "text",
"text": "Done."
}
]
}]
});
assert!(sanitize_claude_request_redacted_thinking_for_deepseek(
&mut body
));
assert_eq!(body["messages"][0]["content"].as_array().unwrap().len(), 2);
assert_eq!(body["messages"][0]["content"][0]["type"], json!("thinking"));
assert_eq!(
body["messages"][0]["content"][0]["thinking"],
json!("I should keep this short.")
);
assert_eq!(
body["messages"][0]["content"][0]["signature"],
json!("sig_123")
);
assert_eq!(body["messages"][0]["content"][1]["text"], json!("Done."));
}
#[test]
fn official_claude_providers_preserve_thinking_signatures() {
assert!(provider_preserves_claude_thinking_signatures(
@@ -1167,6 +1648,14 @@ mod tests {
"amazon_bedrock",
"https://relay.example.com"
));
assert!(!provider_preserves_claude_thinking_signatures(
"deepseek",
"https://relay.example.com"
));
assert!(!provider_preserves_claude_thinking_signatures(
"custom",
"https://api.deepseek.com"
));
assert!(!provider_preserves_claude_thinking_signatures(
"openai",
"https://relay.example.com"
@@ -1,12 +1,10 @@
use std::collections::BTreeMap;
use aether_contracts::RequestBody;
use super::{
augment_sync_report_context, build_ai_execution_plan_from_decision,
generic_decision_missing_exact_provider_request, take_ai_decision_plan_core,
take_ai_upstream_auth_pair, take_non_empty_string, AiExecutionPlanFromDecisionParts,
AiStreamAttempt, AiSyncAttempt,
generic_decision_missing_exact_provider_request, resolve_ai_passthrough_sync_request_body,
take_ai_decision_plan_core, take_ai_upstream_auth_pair, take_non_empty_string,
AiExecutionPlanFromDecisionParts, AiStreamAttempt, AiSyncAttempt,
};
use crate::ai_serving::transport::{
build_standard_plan_fallback_headers, StandardPlanFallbackAcceptPolicy,
@@ -61,6 +59,10 @@ pub(crate) fn build_gemini_sync_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
@@ -70,7 +72,7 @@ pub(crate) fn build_gemini_sync_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream,
},
);
@@ -129,6 +131,10 @@ pub(crate) fn build_gemini_stream_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -137,7 +143,7 @@ pub(crate) fn build_gemini_stream_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream: true,
},
);
@@ -8,14 +8,17 @@ use crate::{AiExecutionDecision, AppState, GatewayError};
mod claude;
mod codex;
mod deepseek;
mod family;
mod gemini;
mod normalize;
mod openai;
pub(crate) use self::codex::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_special_body_edits, apply_codex_openai_special_headers,
codex_model_capabilities_for_transport,
};
pub(crate) use self::deepseek::{apply_deepseek_tool_call_thinking_compat, is_deepseek_provider};
pub(crate) use self::family::{
build_local_stream_attempt_source, build_local_stream_plan_and_reports,
build_local_sync_attempt_source, build_local_sync_plan_and_reports,
@@ -23,9 +26,11 @@ pub(crate) use self::family::{
pub(crate) use self::normalize::{
build_cross_format_openai_chat_request_body, build_cross_format_openai_chat_upstream_url,
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_codex_model_capabilities,
build_cross_format_openai_responses_upstream_url, build_local_openai_chat_request_body,
build_local_openai_chat_upstream_url, build_local_openai_responses_request_body,
build_local_openai_responses_upstream_url,
build_local_openai_responses_request_body_with_codex_model_capabilities,
build_local_openai_responses_upstream_url, validate_final_openai_provider_request,
};
pub(crate) use self::openai::{
build_local_openai_chat_stream_attempt_source_for_kind,
@@ -60,7 +65,8 @@ pub(crate) use crate::ai_serving::{
normalize_openai_responses_request_to_openai_chat_request, parse_openai_tool_result_content,
};
pub(crate) use aether_ai_serving::{
request_body_build_failure_extra_data, same_format_provider_request_body_failure_extra_data,
openai_provider_request_contract_failure_extra_data, request_body_build_failure_extra_data,
request_conversion_failure_extra_data, same_format_provider_request_body_failure_extra_data,
};
pub(crate) fn build_standard_upstream_url(
@@ -78,6 +84,7 @@ pub(crate) fn build_standard_upstream_url(
upstream_is_stream,
parts.uri.query(),
None,
None,
provider_request_body,
)
}
@@ -294,7 +301,7 @@ mod tests {
let converted = build_standard_request_body(
&request,
"claude:messages",
"gpt-5",
"gpt-5.4",
"codex",
"openai:responses",
"/v1/messages",
@@ -306,7 +313,7 @@ mod tests {
assert!(converted.get("metadata").is_none());
assert_eq!(converted["store"], false);
assert_eq!(converted["instructions"], "");
assert!(converted.get("instructions").is_none());
assert_eq!(converted["include"], json!(["reasoning.encrypted_content"]));
assert_eq!(converted["parallel_tool_calls"], true);
assert_eq!(converted["reasoning"]["effort"], "medium");
@@ -12,9 +12,38 @@ pub(crate) use self::chat::{
};
pub(crate) use self::responses::{
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_codex_model_capabilities,
build_cross_format_openai_responses_upstream_url, build_local_openai_responses_request_body,
build_local_openai_responses_request_body_with_codex_model_capabilities,
build_local_openai_responses_upstream_url,
};
pub(super) use crate::ai_serving::planner::common::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
};
pub(crate) fn validate_final_openai_provider_request(
provider_api_format: &str,
mapped_model: &str,
source_request_body: &serde_json::Value,
provider_request_body: &serde_json::Value,
) -> Option<()> {
let provider_model = provider_request_body
.get("model")
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.unwrap_or(mapped_model);
let source_model = source_request_body
.get("model")
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.unwrap_or(mapped_model);
crate::ai_serving::validate_openai_provider_request_contract(
provider_api_format,
provider_model,
source_model,
provider_request_body,
)
.ok()
}
@@ -9,7 +9,10 @@ use crate::ai_serving::{
GatewayProviderTransportSnapshot,
};
use super::{enforce_provider_body_stream_policy, request_requires_body_stream_field};
use super::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
validate_final_openai_provider_request,
};
pub(crate) fn build_local_openai_chat_request_body(
body_json: &Value,
@@ -39,6 +42,12 @@ pub(crate) fn build_local_openai_chat_request_body(
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
"openai:chat",
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -92,6 +101,12 @@ pub(crate) fn build_cross_format_openai_chat_request_body(
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -2,14 +2,16 @@ use serde_json::Value;
use crate::ai_serving::transport::apply_standard_provider_request_body_rules_with_request_headers;
use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits,
apply_openai_responses_compact_special_body_edits,
build_cross_format_openai_responses_request_body_with_model_directives as surface_build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_model_directives_and_history_scope as surface_build_cross_format_openai_responses_request_body,
build_local_openai_responses_request_body_with_model_directives as surface_build_local_openai_responses_request_body,
GatewayProviderTransportSnapshot,
};
use super::{enforce_provider_body_stream_policy, request_requires_body_stream_field};
use super::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
validate_final_openai_provider_request,
};
pub(crate) fn build_local_openai_responses_request_body(
body_json: &Value,
@@ -19,9 +21,35 @@ pub(crate) fn build_local_openai_responses_request_body(
provider_type: &str,
provider_api_format: &str,
body_rules: Option<&Value>,
user_api_key_id: Option<&str>,
_user_api_key_id: Option<&str>,
request_headers: &http::HeaderMap,
enable_model_directives: bool,
) -> Option<Value> {
build_local_openai_responses_request_body_with_codex_model_capabilities(
body_json,
mapped_model,
require_streaming,
force_body_stream_field,
provider_type,
provider_api_format,
body_rules,
request_headers,
None,
enable_model_directives,
)
}
pub(crate) fn build_local_openai_responses_request_body_with_codex_model_capabilities(
body_json: &Value,
mapped_model: &str,
require_streaming: bool,
force_body_stream_field: bool,
provider_type: &str,
provider_api_format: &str,
body_rules: Option<&Value>,
request_headers: &http::HeaderMap,
model_capabilities: Option<&crate::ai_serving::CodexResponsesModelCapabilities>,
enable_model_directives: bool,
) -> Option<Value> {
let provider_request_body = surface_build_local_openai_responses_request_body(
body_json,
@@ -36,23 +64,39 @@ pub(crate) fn build_local_openai_responses_request_body(
body_json,
request_headers,
)?;
apply_codex_openai_responses_special_body_edits(
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(mapped_model);
crate::ai_serving::apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities(
&mut provider_request_body,
provider_type,
provider_api_format,
mapped_model,
source_model,
model_capabilities,
body_rules,
user_api_key_id,
);
apply_openai_responses_compact_special_body_edits(
&mut provider_request_body,
provider_api_format,
);
crate::ai_serving::strip_incompatible_openai_responses_reasoning_items(
&mut provider_request_body,
provider_api_format,
);
enforce_provider_body_stream_policy(
&mut provider_request_body,
provider_api_format,
require_streaming,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -68,6 +112,36 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
user_api_key_id: Option<&str>,
request_headers: &http::HeaderMap,
enable_model_directives: bool,
) -> Option<Value> {
build_cross_format_openai_responses_request_body_with_codex_model_capabilities(
body_json,
mapped_model,
client_api_format,
provider_api_format,
upstream_is_stream,
force_body_stream_field,
provider_type,
body_rules,
request_headers,
user_api_key_id,
None,
enable_model_directives,
)
}
pub(crate) fn build_cross_format_openai_responses_request_body_with_codex_model_capabilities(
body_json: &Value,
mapped_model: &str,
client_api_format: &str,
provider_api_format: &str,
upstream_is_stream: bool,
force_body_stream_field: bool,
provider_type: &str,
body_rules: Option<&Value>,
request_headers: &http::HeaderMap,
history_scope: Option<&str>,
model_capabilities: Option<&crate::ai_serving::CodexResponsesModelCapabilities>,
enable_model_directives: bool,
) -> Option<Value> {
let provider_request_body = surface_build_cross_format_openai_responses_request_body(
body_json,
@@ -76,6 +150,7 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
provider_api_format,
upstream_is_stream,
enable_model_directives,
history_scope,
)?;
let mut provider_request_body =
apply_standard_provider_request_body_rules_with_request_headers(
@@ -84,23 +159,39 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
body_json,
request_headers,
)?;
apply_codex_openai_responses_special_body_edits(
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(mapped_model);
crate::ai_serving::apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities(
&mut provider_request_body,
provider_type,
provider_api_format,
mapped_model,
source_model,
model_capabilities,
body_rules,
user_api_key_id,
);
apply_openai_responses_compact_special_body_edits(
&mut provider_request_body,
provider_api_format,
);
crate::ai_serving::strip_incompatible_openai_responses_reasoning_items(
&mut provider_request_body,
provider_api_format,
);
enforce_provider_body_stream_policy(
&mut provider_request_body,
provider_api_format,
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -6,8 +6,8 @@ use http::Request;
use serde_json::{json, Value};
use super::{
build_cross_format_openai_responses_request_body, build_local_openai_responses_request_body,
build_local_openai_responses_upstream_url,
build_cross_format_openai_responses_request_body, build_local_openai_chat_request_body,
build_local_openai_responses_request_body, build_local_openai_responses_upstream_url,
};
fn object_keys(value: &Value) -> Vec<&str> {
@@ -68,6 +68,7 @@ fn sample_transport(base_url: &str, api_format: &str) -> GatewayProviderTranspor
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "__placeholder__".to_string(),
decrypted_auth_config: None,
},
@@ -145,12 +146,47 @@ fn local_openai_responses_wrapper_preserves_body_order_after_edits() {
"reasoning",
"tool_choice",
"parallel_tool_calls",
"instructions",
"prompt_cache_key",
]
);
assert_eq!(provider_request_body["parallel_tool_calls"], json!(true));
assert_eq!(provider_request_body["instructions"], json!(""));
assert!(provider_request_body.get("instructions").is_none());
}
#[test]
fn local_openai_responses_wrapper_strips_foreign_reasoning_item_ids() {
let body_json = json!({
"model": "gpt-5.4",
"input": [
{"type": "reasoning", "id": "rs_provider_123", "summary": []},
{
"type": "reasoning",
"id": "item_72d3bd8d367d01977ace23f1",
"summary": []
},
{"type": "message", "role": "user", "content": "continue"}
]
});
let provider_request_body = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
false,
false,
"codex",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.expect("local OpenAI Responses body should build");
let input = provider_request_body["input"]
.as_array()
.expect("input array");
assert_eq!(input.len(), 2);
assert_eq!(input[0]["id"], "rs_provider_123");
assert_eq!(input[1]["type"], "message");
}
#[test]
@@ -180,18 +216,49 @@ fn local_openai_responses_compact_wrapper_strips_store_for_same_format_requests(
}
#[test]
fn local_openai_responses_compact_wrapper_strips_include_for_codex_requests() {
fn local_codex_compact_wrapper_applies_the_complete_request_projection() {
let body_json = json!({
"model": "gpt-5.4",
"input": [],
"model": "gpt-5.6-sol",
"input": [{
"type": "message",
"role": "user",
"content": [{"type": "input_text", "text": "hello"}]
}],
"instructions": "Work carefully",
"client_metadata": {"origin": "codex"},
"include": ["reasoning.encrypted_content"],
"store": true,
"stream": true
"stream": true,
"stream_options": {"reasoning_summary_delivery": "sequential_cutoff"},
"tool_choice": "auto",
"parallel_tool_calls": true,
"reasoning": {"effort": "max", "summary": "auto", "context": "all_turns"},
"text": {"verbosity": "medium"},
"tools": [{
"type": "function",
"name": "lookup",
"parameters": {"type": "object", "properties": {}}
}],
"service_tier": "priority",
"prompt_cache_key": "thread-compact"
});
let provider_request_body = build_local_openai_responses_request_body(
let regular = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
"gpt-5.6-sol",
true,
false,
"codex",
"openai:responses",
None,
Some("key-123"),
&http::HeaderMap::new(),
false,
)
.expect("local Codex Responses body should build");
let compact = build_local_openai_responses_request_body(
&body_json,
"gpt-5.6-sol",
false,
false,
"codex",
@@ -201,22 +268,44 @@ fn local_openai_responses_compact_wrapper_strips_include_for_codex_requests() {
&http::HeaderMap::new(),
false,
)
.expect("local codex compact body should build");
.expect("local Codex Compact body should build");
assert!(provider_request_body.get("include").is_none());
assert!(provider_request_body.get("store").is_none());
assert!(provider_request_body.get("stream").is_none());
assert_eq!(provider_request_body["instructions"], "");
assert_eq!(
provider_request_body["prompt_cache_key"],
"3d2e2842-74cb-55dd-803a-b8940b3500c2"
);
for field in [
"client_metadata",
"include",
"store",
"stream",
"stream_options",
"tool_choice",
] {
assert!(
regular.get(field).is_some(),
"Responses should contain {field}"
);
assert!(compact.get(field).is_none(), "Compact should omit {field}");
}
for field in [
"model",
"input",
"instructions",
"parallel_tool_calls",
"reasoning",
"text",
"tools",
"service_tier",
"prompt_cache_key",
] {
assert_eq!(
compact[field], regular[field],
"Compact should preserve {field}"
);
}
}
#[test]
fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
let body_json = json!({
"model": "gpt-5.4-max",
"model": "gpt-5.6-sol-max",
"input": "hello",
"reasoning": {"effort": "low", "summary": "auto"}
});
@@ -226,7 +315,7 @@ fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
let provider_request_body = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
"gpt-5.6-sol",
false,
false,
"openai",
@@ -238,11 +327,137 @@ fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
)
.expect("local openai responses body should build");
assert_eq!(provider_request_body["reasoning"]["effort"], "xhigh");
assert_eq!(provider_request_body["reasoning"]["effort"], "max");
assert_eq!(provider_request_body["reasoning"]["summary"], "auto");
assert_eq!(provider_request_body["metadata"]["override_seen"], true);
}
#[test]
fn final_openai_provider_contract_uses_the_mapped_model_for_reasoning() {
let alias = json!({
"model": "deployment-alias",
"input": "hello",
"reasoning": {"effort": "max"}
});
assert!(build_local_openai_responses_request_body(
&alias,
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_some());
assert!(build_local_openai_responses_request_body(
&alias,
"gpt-5.4",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let minimal = json!({
"model": "deployment-alias",
"messages": [{"role": "user", "content": "hello"}],
"reasoning_effort": "minimal"
});
assert!(build_local_openai_chat_request_body(
&minimal,
"gpt-5.6-terra",
false,
false,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let opaque_mapping = json!({
"model": "gpt-5.6-sol-max",
"input": "hello",
"reasoning": {"effort": "max", "mode": "pro"},
"prompt_cache_options": {"mode": "explicit", "ttl": "30m"}
});
assert!(build_local_openai_responses_request_body(
&opaque_mapping,
"azure-production",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_some());
assert!(build_local_openai_responses_request_body(
&opaque_mapping,
"gpt-5.4",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
}
#[test]
fn final_openai_provider_contract_validates_body_rule_output() {
let body = json!({
"model": "gpt-5.6-sol",
"input": "hello",
"reasoning": {"effort": "max"}
});
let model_override = json!([
{"action":"set","path":"model","value":"gpt-5.4"}
]);
assert!(build_local_openai_responses_request_body(
&body,
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
Some(&model_override),
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let cache_override = json!([
{"action":"set","path":"prompt_cache_options.ttl","value":"1h"}
]);
assert!(build_local_openai_responses_request_body(
&json!({"model":"gpt-5.6-sol","input":"hello"}),
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
Some(&cache_override),
None,
&http::HeaderMap::new(),
false,
)
.is_none());
}
#[test]
fn local_openai_responses_upstream_url_preserves_codex_base_path() {
let request = Request::builder()
@@ -291,6 +506,47 @@ fn strips_metadata_for_codex_openai_responses_requests() {
assert!(provider_request_body.get("metadata").is_none());
}
#[test]
fn openai_chat_to_codex_responses_preserves_json_mode_chat_messages() {
let body_json = json!({
"model": "gpt-5.5",
"messages": [
{"role": "system", "content": "Return a JSON object."},
{"role": "user", "content": "Why did this JSON request fail?"}
],
"response_format": {"type": "json_object"}
});
let provider_request_body = build_cross_format_openai_responses_request_body(
&body_json,
"gpt-5.5-upstream",
"openai:chat",
"openai:responses",
false,
false,
"codex",
None,
None,
&http::HeaderMap::new(),
false,
)
.expect("openai chat to codex responses request should build");
assert_eq!(
provider_request_body["text"]["format"]["type"],
"json_object"
);
assert_eq!(provider_request_body["input"][0]["role"], "user");
assert_eq!(
provider_request_body["input"][0]["content"][0]["text"],
"Why did this JSON request fail?"
);
assert_eq!(
provider_request_body["instructions"],
"Return a JSON object."
);
}
#[test]
fn applies_codex_defaults_unless_body_rules_handle_the_field() {
let body_json = json!({
@@ -329,7 +585,7 @@ fn applies_codex_defaults_unless_body_rules_handle_the_field() {
}
#[test]
fn injects_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
fn omits_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
let body_json = json!({
"model": "claude-sonnet-4-5",
"messages": [{
@@ -353,14 +609,11 @@ fn injects_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
)
.expect("claude cli to codex request should build");
assert_eq!(
provider_request_body["prompt_cache_key"],
"b4dfeb75-b105-544c-a706-39b92f0bddb0"
);
assert!(provider_request_body.get("prompt_cache_key").is_none());
}
#[test]
fn injects_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
fn omits_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
let body_json = json!({
"model": "gpt-5",
"messages": [{
@@ -383,8 +636,5 @@ fn injects_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
)
.expect("openai chat to codex request should build");
assert_eq!(
provider_request_body["prompt_cache_key"],
"4ee6ea6e-3ac6-5a18-8cb8-1f8b956419e5"
);
assert!(provider_request_body.get("prompt_cache_key").is_none());
}
@@ -6,6 +6,7 @@ mod request;
mod support;
pub(super) use self::payload::maybe_build_local_openai_chat_decision_payload_for_candidate;
pub(super) use self::request::LocalOpenAiChatRequestPreparation;
pub(super) use self::support::{
build_lazy_local_openai_chat_candidate_attempt_source,
build_local_openai_chat_candidate_attempt_source,
@@ -2,21 +2,25 @@ use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::common::OPENAI_CHAT_STREAM_PLAN_KIND;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, insert_provider_stream_event_api_format,
LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
insert_provider_stream_event_api_format, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_chat_candidate_payload_parts;
use super::request::{
resolve_local_openai_chat_candidate_payload_parts, LocalOpenAiChatRequestPreparation,
};
use super::support::{LocalOpenAiChatCandidateAttempt, LocalOpenAiChatDecisionInput};
#[allow(clippy::too_many_arguments)]
@@ -26,6 +30,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
trace_id: &str,
body_json: &serde_json::Value,
input: &LocalOpenAiChatDecisionInput,
preparation: Option<&mut LocalOpenAiChatRequestPreparation>,
attempt: LocalOpenAiChatCandidateAttempt,
decision_kind: &str,
report_kind: &str,
@@ -39,12 +44,15 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
candidate_id,
..
} = attempt;
let upstream_is_stream = upstream_is_stream && eligible.candidate.supports_streaming;
let payload_started_at = std::time::Instant::now();
let Some(resolved) = resolve_local_openai_chat_candidate_payload_parts(
state,
parts,
trace_id,
body_json,
input,
preparation,
&eligible,
candidate_index,
&candidate_id,
@@ -54,9 +62,25 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
)
.await?
else {
observe_gateway_stage_ms(
"stream_candidate_payload_parts",
payload_started_at.elapsed().as_millis() as u64,
);
return Ok(None);
};
observe_gateway_stage_ms(
"stream_candidate_payload_parts",
payload_started_at.elapsed().as_millis() as u64,
);
let candidate = &eligible.candidate;
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
resolved.transport.endpoint.config.as_ref(),
resolved.transport.provider.provider_type.as_str(),
resolved.provider_api_format.as_str(),
upstream_is_stream,
false,
);
let prompt_cache_key = resolved
.provider_request_body
@@ -65,9 +89,14 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned);
let proxy_started_at = std::time::Instant::now();
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
observe_gateway_stage_ms(
"stream_candidate_proxy",
proxy_started_at.elapsed().as_millis() as u64,
);
let transport_profile = resolved
.transport_profile
.clone()
@@ -84,6 +113,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
"envelope_name".to_string(),
serde_json::Value::String(envelope_name.to_string()),
);
insert_native_client_envelope_name(&mut extra_fields, envelope_name, parts.uri.path());
}
insert_provider_stream_event_api_format(
&mut extra_fields,
@@ -103,15 +133,6 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
.eq_ignore_ascii_case("chatgpt_web")
{
extra_fields.insert("chatgpt_web_image".to_string(), serde_json::json!(true));
extra_fields.insert(
"local_failover_policy".to_string(),
serde_json::json!({
"stop_status_codes": [400, 401, 403, 429, 500, 502, 503, 504],
"error_stop_patterns": [
{ "pattern": ".*" }
]
}),
);
}
let super::request::LocalOpenAiChatCandidatePayloadParts {
client_api_format,
@@ -137,6 +158,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
Some(body_json)
};
let effective_headers = input.effective_headers(&parts.headers);
let report_context_started_at = std::time::Instant::now();
let report_context = append_local_failover_policy_to_value(
append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -171,6 +193,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -191,7 +214,13 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
),
&transport,
);
observe_gateway_stage_ms(
"stream_candidate_report_context",
report_context_started_at.elapsed().as_millis() as u64,
);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let decision_started_at = std::time::Instant::now();
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream,
decision_kind: decision_kind.to_string(),
@@ -200,6 +229,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -217,6 +247,8 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
@@ -225,6 +257,14 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
observe_gateway_stage_ms(
"stream_candidate_decision_build",
decision_started_at.elapsed().as_millis() as u64,
);
Ok(Some(decision))
}
File diff suppressed because it is too large Load Diff
@@ -293,9 +293,11 @@ pub(crate) async fn build_lazy_local_openai_chat_candidate_attempt_source<'a>(
);
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
"openai:chat",
&input.requested_model,
None,
require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
@@ -13,6 +13,7 @@ use self::decision::{
build_lazy_local_openai_chat_candidate_attempt_source,
maybe_build_local_openai_chat_decision_payload_for_candidate, LocalOpenAiChatCandidateAttempt,
LocalOpenAiChatCandidateAttemptSource, LocalOpenAiChatDecisionInput,
LocalOpenAiChatRequestPreparation,
};
use self::plans::{
build_local_openai_chat_stream_attempt_source, build_local_openai_chat_stream_plan_and_reports,
@@ -147,7 +148,7 @@ pub(crate) async fn maybe_build_sync_local_decision_payload(
)
.await;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let upstream_is_stream = self::plans::openai_chat_upstream_is_stream_for_candidate(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
@@ -159,6 +160,7 @@ pub(crate) async fn maybe_build_sync_local_decision_payload(
trace_id,
body_json,
&input,
None,
attempt,
OPENAI_CHAT_SYNC_PLAN_KIND,
"openai_chat_sync_success",
@@ -199,7 +201,7 @@ pub(crate) async fn maybe_build_stream_local_decision_payload(
)
.await;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let upstream_is_stream = self::plans::openai_chat_upstream_is_stream_for_candidate(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
@@ -211,6 +213,7 @@ pub(crate) async fn maybe_build_stream_local_decision_payload(
trace_id,
body_json,
&input,
None,
attempt,
OPENAI_CHAT_STREAM_PLAN_KIND,
"openai_chat_stream_success",
@@ -31,7 +31,12 @@ pub(super) fn openai_chat_upstream_is_stream_for_candidate(
crate::ai_serving::transport::kiro::is_kiro_claude_messages_transport(
transport,
provider_api_format,
);
) || openai_chat_antigravity_requires_upstream_streaming(transport, provider_api_format)
|| openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport,
provider_api_format,
client_is_stream,
);
resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
@@ -41,6 +46,27 @@ pub(super) fn openai_chat_upstream_is_stream_for_candidate(
)
}
fn openai_chat_antigravity_requires_upstream_streaming(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
) -> bool {
crate::ai_serving::transport::antigravity::is_antigravity_provider_transport(transport)
&& crate::ai_serving::normalize_api_format_alias(provider_api_format)
== "gemini:generate_content"
}
fn openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
client_is_stream: bool,
) -> bool {
crate::ai_serving::transport::gemini_cli::is_gemini_cli_provider_transport(transport)
&& crate::ai_serving::transport::gemini_cli::gemini_cli_v1internal_requires_upstream_streaming(
provider_api_format,
client_is_stream,
)
}
#[cfg(test)]
mod tests {
use super::openai_chat_upstream_is_stream_for_candidate;
@@ -103,6 +129,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "secret".to_string(),
decrypted_auth_config: None,
},
@@ -164,4 +191,44 @@ mod tests {
false,
));
}
#[test]
fn openai_chat_policy_resolver_preserves_gemini_cli_streaming_requests() {
let gemini_cli = sample_transport(
"gemini_cli",
"gemini:generate_content",
Some(json!({"upstream_stream_policy": "force_non_stream"})),
);
assert!(openai_chat_upstream_is_stream_for_candidate(
&gemini_cli,
"gemini:generate_content",
true,
));
assert!(!openai_chat_upstream_is_stream_for_candidate(
&gemini_cli,
"gemini:generate_content",
false,
));
}
#[test]
fn openai_chat_policy_resolver_preserves_antigravity_streaming_envelope() {
let antigravity = sample_transport(
"antigravity",
"gemini:generate_content",
Some(json!({"upstream_stream_policy": "force_non_stream"})),
);
assert!(openai_chat_upstream_is_stream_for_candidate(
&antigravity,
"gemini:generate_content",
false,
));
assert!(openai_chat_upstream_is_stream_for_candidate(
&antigravity,
"gemini:generate_content",
true,
));
}
}
@@ -21,8 +21,10 @@ pub(crate) async fn list_local_openai_chat_candidates(
> {
let outcome = preselect_local_execution_candidates_with_serving(
PlannerAppState::new(state),
&input.model_directive_policy,
"openai:chat",
&input.requested_model,
None,
require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -9,6 +9,7 @@ use crate::ai_serving::planner::decision_input::{
};
use crate::ai_serving::resolve_local_decision_execution_runtime_auth_context;
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{AppState, GatewayError};
pub(crate) async fn resolve_local_openai_chat_decision_input(
@@ -59,11 +60,14 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
return Ok(None);
};
let auth_started_at = std::time::Instant::now();
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context.clone(),
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -106,10 +110,20 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
return Err(err);
}
};
observe_gateway_stage_ms(
"openai_chat_decision_input_auth",
auth_started_at.elapsed().as_millis() as u64,
);
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
let affinity_started_at = std::time::Instant::now();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
observe_gateway_stage_ms(
"openai_chat_decision_input_affinity",
affinity_started_at.elapsed().as_millis() as u64,
);
let routing_started_at = std::time::Instant::now();
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
@@ -126,5 +140,9 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
);
return Err(err);
}
observe_gateway_stage_ms(
"openai_chat_decision_input_routing",
routing_started_at.elapsed().as_millis() as u64,
);
Ok(Some(input))
}
@@ -1,11 +1,12 @@
use async_trait::async_trait;
use std::collections::VecDeque;
use tracing::warn;
use super::super::{
build_lazy_local_openai_chat_candidate_attempt_source,
maybe_build_local_openai_chat_decision_payload_for_candidate, AppState, GatewayControlDecision,
GatewayError, LocalOpenAiChatCandidateAttempt, LocalOpenAiChatCandidateAttemptSource,
LocalOpenAiChatDecisionInput,
LocalOpenAiChatDecisionInput, LocalOpenAiChatRequestPreparation,
};
use super::diagnostic::{
set_local_openai_chat_candidate_evaluation_diagnostic, set_local_openai_chat_miss_diagnostic,
@@ -18,6 +19,23 @@ use crate::ai_serving::planner::plan_builders::{
build_openai_chat_stream_plan_from_decision, AiStreamAttempt,
};
use crate::ai_serving::planner::runtime_miss::apply_local_runtime_candidate_terminal_reason;
use crate::ai_serving::planner::standard::build_local_openai_chat_upstream_url;
use crate::ai_serving::transport::{
is_windsurf_provider_transport, local_openai_chat_transport_unsupported_reason,
};
use crate::clock::request_distribution_seed;
use crate::stage_metrics::{
observe_gateway_stage_ms, record_openai_chat_stream_payload_build_prefetch_avoided,
record_openai_chat_stream_payload_build_selected,
record_openai_chat_stream_raw_candidates_scanned,
record_openai_chat_stream_target_select_selected_rank,
};
use crate::upstream_admission::upstream_target_key_from_url;
const OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW_ENV: &str =
"AETHER_GATEWAY_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW";
const DEFAULT_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW: usize = 2;
const MAX_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW: usize = 8;
pub(crate) struct LocalOpenAiChatStreamAttemptSource<'a> {
state: &'a AppState,
@@ -26,6 +44,8 @@ pub(crate) struct LocalOpenAiChatStreamAttemptSource<'a> {
body_json: serde_json::Value,
input: LocalOpenAiChatDecisionInput,
candidates: LocalOpenAiChatCandidateAttemptSource<'a>,
prefetched_attempts: VecDeque<LocalOpenAiChatCandidateAttempt>,
request_preparation: LocalOpenAiChatRequestPreparation,
}
pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
@@ -40,6 +60,7 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
return Ok(None);
}
let attempt_source_started_at = std::time::Instant::now();
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, true,
)
@@ -76,6 +97,10 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
Some(input.requested_model.as_str()),
candidate_count,
);
observe_gateway_stage_ms(
"openai_chat_attempt_source_build",
attempt_source_started_at.elapsed().as_millis() as u64,
);
Ok(Some((
LocalOpenAiChatStreamAttemptSource {
@@ -85,6 +110,8 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
body_json: effective_body_json,
input,
candidates,
prefetched_attempts: VecDeque::new(),
request_preparation: LocalOpenAiChatRequestPreparation,
},
candidate_count,
)))
@@ -93,18 +120,13 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiChatStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
}
}
apply_local_runtime_candidate_terminal_reason(
self.state,
self.trace_id,
"no_local_stream_plans",
let select_started_at = std::time::Instant::now();
let selected = self.next_execution_attempt_with_target_select().await?;
observe_gateway_stage_ms(
"openai_chat_stream_target_select",
select_started_at.elapsed().as_millis() as u64,
);
Ok(None)
Ok(selected)
}
async fn drain_execution_attempts(&mut self) -> Result<Vec<AiStreamAttempt>, GatewayError> {
@@ -116,11 +138,178 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiChatStreamAttem
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.key_id != key_id);
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.endpoint_id != endpoint_id);
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.provider_id != provider_id);
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiChatStreamAttemptSource<'_> {
async fn build_stream_attempt(
async fn next_execution_attempt_with_target_select(
&mut self,
) -> Result<Option<AiStreamAttempt>, GatewayError> {
loop {
let Some(attempt) = self.next_raw_attempt_with_target_select().await? else {
apply_local_runtime_candidate_terminal_reason(
self.state,
self.trace_id,
"no_local_stream_plans",
);
return Ok(None);
};
let plan_started_at = std::time::Instant::now();
record_openai_chat_stream_payload_build_selected();
match self.build_stream_attempt(attempt).await? {
Some(attempt) => {
observe_gateway_stage_ms(
"stream_candidate_plan_build",
plan_started_at.elapsed().as_millis() as u64,
);
return Ok(Some(attempt));
}
None => {
observe_gateway_stage_ms(
"stream_candidate_plan_build",
plan_started_at.elapsed().as_millis() as u64,
);
continue;
}
}
}
}
async fn next_raw_attempt_with_target_select(
&mut self,
) -> Result<Option<LocalOpenAiChatCandidateAttempt>, GatewayError> {
let select_window = openai_chat_stream_target_select_window();
if select_window <= 1 {
return self.next_raw_attempt_linear().await;
}
let mut attempts = Vec::with_capacity(select_window);
for _ in 0..select_window {
match self.next_raw_attempt_linear().await? {
Some(attempt) => attempts.push(attempt),
None => break,
}
}
if attempts.is_empty() {
return Ok(None);
}
record_openai_chat_stream_raw_candidates_scanned(attempts.len());
let seed = request_distribution_seed();
let target_keys = attempts
.iter()
.map(|attempt| self.lightweight_target_key_for_attempt(attempt))
.collect::<Vec<_>>();
for target_key in target_keys.iter().flatten() {
self.state
.upstream_target_admission
.record_raw_seen_for_target_key(target_key);
}
let selected_index = if target_keys.iter().all(Option::is_some) {
let choices = attempts
.iter()
.zip(target_keys.iter())
.map(|(attempt, target_key)| {
let target_key = target_key.as_deref().unwrap_or("-");
let snapshot = self
.state
.upstream_target_admission
.snapshot_for_target_key(target_key);
TargetSelectChoice {
target_key,
identity: target_select_candidate_identity(attempt),
in_flight: snapshot
.as_ref()
.map(|snapshot| snapshot.in_flight)
.unwrap_or(0),
selection_pressure_total: snapshot
.as_ref()
.map(|snapshot| snapshot.selection_pressure_total)
.unwrap_or(0),
}
})
.collect::<Vec<_>>();
select_target_index(seed, &choices)
} else {
0
};
record_openai_chat_stream_target_select_selected_rank(selected_index);
record_openai_chat_stream_payload_build_prefetch_avoided(attempts.len().saturating_sub(1));
if let Some(Some(target_key)) = target_keys.get(selected_index) {
self.state
.upstream_target_admission
.record_preselect_for_target_key(target_key);
}
let selected = attempts.remove(selected_index);
self.prefetched_attempts.extend(attempts);
Ok(Some(selected))
}
async fn next_raw_attempt_linear(
&mut self,
) -> Result<Option<LocalOpenAiChatCandidateAttempt>, GatewayError> {
if let Some(attempt) = self.prefetched_attempts.pop_front() {
return Ok(Some(attempt));
}
let source_started_at = std::time::Instant::now();
let attempt = self.candidates.next_attempt().await?;
observe_gateway_stage_ms(
"stream_candidate_source_next",
source_started_at.elapsed().as_millis() as u64,
);
Ok(attempt)
}
fn lightweight_target_key_for_attempt(
&self,
attempt: &LocalOpenAiChatCandidateAttempt,
) -> Option<String> {
let provider_api_format = attempt.eligible.provider_api_format.trim();
if !provider_api_format.eq_ignore_ascii_case("openai:chat") {
return None;
}
let transport = &attempt.eligible.transport;
if transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("grok")
|| is_windsurf_provider_transport(transport)
|| local_openai_chat_transport_unsupported_reason(transport).is_some()
{
return None;
}
if transport.provider.proxy.is_some()
|| transport.endpoint.proxy.is_some()
|| transport.key.proxy.is_some()
{
return None;
}
let upstream_url = build_local_openai_chat_upstream_url(self.parts, transport)?;
upstream_target_key_from_url(upstream_url.as_str(), None)
}
async fn build_stream_attempt(
&mut self,
attempt: LocalOpenAiChatCandidateAttempt,
) -> Result<Option<AiStreamAttempt>, GatewayError> {
let upstream_is_stream = openai_chat_upstream_is_stream_for_candidate(
@@ -134,6 +323,7 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
self.trace_id,
&self.body_json,
&self.input,
Some(&mut self.request_preparation),
attempt,
OPENAI_CHAT_STREAM_PLAN_KIND,
"openai_chat_stream_success",
@@ -158,6 +348,97 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
}
}
fn openai_chat_stream_target_select_window() -> usize {
std::env::var(OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW_ENV)
.ok()
.and_then(|value| value.trim().parse::<usize>().ok())
.filter(|value| *value > 0)
.unwrap_or(DEFAULT_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW)
.clamp(1, MAX_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW)
}
#[derive(Clone, Copy)]
struct TargetSelectCandidateIdentity<'a> {
provider_id: &'a str,
endpoint_id: &'a str,
key_id: &'a str,
candidate_id: &'a str,
}
#[derive(Clone, Copy)]
struct TargetSelectChoice<'a> {
target_key: &'a str,
identity: TargetSelectCandidateIdentity<'a>,
in_flight: usize,
selection_pressure_total: u64,
}
fn select_target_index(seed: u64, choices: &[TargetSelectChoice<'_>]) -> usize {
choices
.iter()
.enumerate()
.min_by_key(|(index, choice)| {
target_select_score(
seed,
choice.target_key,
&choice.identity,
*index,
choice.in_flight,
choice.selection_pressure_total,
)
})
.map(|(index, _)| index)
.unwrap_or(0)
}
fn target_select_candidate_identity(
attempt: &LocalOpenAiChatCandidateAttempt,
) -> TargetSelectCandidateIdentity<'_> {
TargetSelectCandidateIdentity {
provider_id: &attempt.eligible.candidate.provider_id,
endpoint_id: &attempt.eligible.candidate.endpoint_id,
key_id: &attempt.eligible.candidate.key_id,
candidate_id: &attempt.candidate_id,
}
}
fn target_select_tie_break(
seed: u64,
target_key: &str,
identity: &TargetSelectCandidateIdentity<'_>,
index: usize,
) -> u64 {
let mut hash = seed ^ ((index as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15));
hash = hash_string(hash, target_key);
hash = hash_string(hash, identity.provider_id);
hash = hash_string(hash, identity.endpoint_id);
hash = hash_string(hash, identity.key_id);
hash_string(hash, identity.candidate_id)
}
fn hash_string(mut hash: u64, value: &str) -> u64 {
for byte in value.as_bytes() {
hash ^= u64::from(*byte);
hash = hash.wrapping_mul(0x100_0000_01B3);
}
hash
}
fn target_select_score(
seed: u64,
target_key: &str,
identity: &TargetSelectCandidateIdentity<'_>,
index: usize,
in_flight: usize,
selected_total: u64,
) -> (usize, u64, u64) {
(
in_flight,
selected_total,
target_select_tie_break(seed, target_key, identity, index),
)
}
pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
state: &AppState,
parts: &http::request::Parts,
@@ -207,3 +488,82 @@ pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
Ok(plans)
}
#[cfg(test)]
mod tests {
use super::*;
fn identity<'a>(
endpoint_id: &'a str,
candidate_id: &'a str,
) -> TargetSelectCandidateIdentity<'a> {
TargetSelectCandidateIdentity {
provider_id: "provider",
endpoint_id,
key_id: "key",
candidate_id,
}
}
#[test]
fn target_select_score_prefers_lower_in_flight() {
let busy = identity("endpoint-a", "candidate-a");
let idle = identity("endpoint-b", "candidate-b");
assert!(
target_select_score(7, "http://127.0.0.1:18182|proxy=-", &idle, 1, 0, 10)
< target_select_score(7, "http://127.0.0.1:18181|proxy=-", &busy, 0, 5, 0)
);
}
#[test]
fn target_select_tie_break_distinguishes_equivalent_targets() {
let left = identity("endpoint-a", "candidate-a");
let right = identity("endpoint-b", "candidate-b");
assert_ne!(
target_select_tie_break(11, "http://127.0.0.1:18181|proxy=-", &left, 0),
target_select_tie_break(11, "http://127.0.0.1:18182|proxy=-", &right, 1)
);
}
#[test]
fn select_target_index_prefers_lower_in_flight_target() {
let choices = [
TargetSelectChoice {
target_key: "http://127.0.0.1:18181|proxy=-",
identity: identity("endpoint-a", "candidate-a"),
in_flight: 8,
selection_pressure_total: 0,
},
TargetSelectChoice {
target_key: "http://127.0.0.1:18182|proxy=-",
identity: identity("endpoint-b", "candidate-b"),
in_flight: 1,
selection_pressure_total: 100,
},
];
assert_eq!(select_target_index(17, &choices), 1);
}
#[test]
fn select_target_index_uses_selection_pressure_before_tie_break() {
let choices = [
TargetSelectChoice {
target_key: "http://127.0.0.1:18181|proxy=-",
identity: identity("endpoint-a", "candidate-a"),
in_flight: 0,
selection_pressure_total: 20,
},
TargetSelectChoice {
target_key: "http://127.0.0.1:18182|proxy=-",
identity: identity("endpoint-b", "candidate-b"),
in_flight: 0,
selection_pressure_total: 1,
},
];
assert_eq!(select_target_index(19, &choices), 1);
}
}
@@ -93,7 +93,7 @@ pub(crate) async fn build_local_openai_chat_sync_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiChatSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -116,6 +116,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiChatSyncAttemptSo
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiChatSyncAttemptSource<'_> {
@@ -134,6 +149,7 @@ impl LocalOpenAiChatSyncAttemptSource<'_> {
self.trace_id,
&self.body_json,
&self.input,
None,
attempt,
OPENAI_CHAT_SYNC_PLAN_KIND,
"openai_chat_sync_success",
@@ -129,7 +129,7 @@ pub(crate) fn build_openai_chat_stream_plan_from_decision(
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
stream: effective_upstream_is_stream,
},
);
@@ -229,7 +229,7 @@ pub(crate) fn build_openai_responses_stream_plan_from_decision(
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
stream: effective_upstream_is_stream,
},
);
@@ -291,6 +291,7 @@ mod tests {
request_id: Some("req_123".to_string()),
candidate_id: Some("cand_123".to_string()),
provider_name: Some("Codex".to_string()),
provider_type: Some("codex".to_string()),
provider_id: Some("prov_123".to_string()),
endpoint_id: Some("ep_123".to_string()),
key_id: Some("key_123".to_string()),
@@ -326,6 +327,8 @@ mod tests {
})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -383,6 +386,46 @@ mod tests {
);
}
#[test]
fn build_compact_stream_plan_preserves_non_stream_upstream_mode() {
let parts = http::Request::builder()
.uri("http://localhost/v1/responses/compact")
.body(())
.expect("request should build")
.into_parts()
.0;
let mut payload = sample_responses_payload();
payload.decision_kind = Some("openai_responses_compact_stream".to_string());
payload.upstream_url = Some("https://example.com/v1/responses/compact".to_string());
payload.provider_api_format = Some("openai:responses:compact".to_string());
payload.client_api_format = Some("openai:responses:compact".to_string());
payload.upstream_is_stream = false;
payload.provider_request_body = Some(json!({
"model": "gpt-5.6-sol",
"input": [],
"instructions": "You are Codex.",
"tools": [],
"parallel_tool_calls": true,
"reasoning": {"effort": "high"},
"prompt_cache_key": "cache-key",
"text": {"verbosity": "low"}
}));
let built =
build_openai_responses_stream_plan_from_decision(&parts, &json!({}), payload, true)
.expect("plan build should succeed")
.expect("plan should be produced");
assert!(!built.plan.stream);
assert!(built
.plan
.body
.json_body
.as_ref()
.is_some_and(|body| body.get("stream").is_none()));
assert!(!built.plan.headers.contains_key("accept"));
}
#[test]
fn build_openai_chat_stream_plan_fallback_preserves_complete_same_format_headers() {
let parts = http::Request::builder()
@@ -402,6 +445,7 @@ mod tests {
request_id: Some("req_stream_456".to_string()),
candidate_id: Some("cand_stream_456".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_stream_456".to_string()),
endpoint_id: Some("ep_stream_456".to_string()),
key_id: Some("key_stream_456".to_string()),
@@ -422,6 +466,8 @@ mod tests {
provider_request_body: Some(json!({"model":"gpt-5.4","messages":[],"stream":true})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -458,7 +504,7 @@ mod tests {
}
#[test]
fn build_openai_chat_stream_plan_keeps_downstream_stream_for_force_non_stream_upstream() {
fn build_openai_chat_stream_plan_preserves_force_non_stream_upstream_mode() {
fn force_non_stream_payload(provider_request_body: Option<Value>) -> AiExecutionDecision {
AiExecutionDecision {
action: "stream".to_string(),
@@ -468,6 +514,7 @@ mod tests {
request_id: Some("req_force_non_stream".to_string()),
candidate_id: Some("cand_force_non_stream".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_force_non_stream".to_string()),
endpoint_id: Some("ep_force_non_stream".to_string()),
key_id: Some("key_force_non_stream".to_string()),
@@ -488,6 +535,8 @@ mod tests {
provider_request_body,
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -517,7 +566,7 @@ mod tests {
.expect("plan build should succeed")
.expect("plan should be produced");
assert!(built.plan.stream);
assert!(!built.plan.stream);
assert_eq!(
built
.plan
@@ -542,7 +591,7 @@ mod tests {
.expect("fallback plan build should succeed")
.expect("fallback plan should be produced");
assert!(built.plan.stream);
assert!(!built.plan.stream);
assert_eq!(
built
.plan
@@ -573,6 +622,7 @@ mod tests {
request_id: Some("req_stream_789".to_string()),
candidate_id: Some("cand_stream_789".to_string()),
provider_name: Some("Claude".to_string()),
provider_type: Some("anthropic".to_string()),
provider_id: Some("prov_stream_789".to_string()),
endpoint_id: Some("ep_stream_789".to_string()),
key_id: Some("key_stream_789".to_string()),
@@ -595,6 +645,8 @@ mod tests {
),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -257,6 +257,7 @@ mod tests {
request_id: Some("req_123".to_string()),
candidate_id: Some("cand_123".to_string()),
provider_name: Some("Codex".to_string()),
provider_type: Some("codex".to_string()),
provider_id: Some("prov_123".to_string()),
endpoint_id: Some("ep_123".to_string()),
key_id: Some("key_123".to_string()),
@@ -292,6 +293,8 @@ mod tests {
})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -367,6 +370,7 @@ mod tests {
request_id: Some("req_456".to_string()),
candidate_id: Some("cand_456".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_456".to_string()),
endpoint_id: Some("ep_456".to_string()),
key_id: Some("key_456".to_string()),
@@ -387,6 +391,8 @@ mod tests {
provider_request_body: Some(json!({"model":"gpt-5.4","messages":[],"stream":false})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -436,6 +442,7 @@ mod tests {
request_id: Some("req_789".to_string()),
candidate_id: Some("cand_789".to_string()),
provider_name: Some("Claude".to_string()),
provider_type: Some("anthropic".to_string()),
provider_id: Some("prov_789".to_string()),
endpoint_id: Some("ep_789".to_string()),
key_id: Some("key_789".to_string()),
@@ -458,6 +465,8 @@ mod tests {
),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -4,12 +4,13 @@ use tracing::debug;
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, insert_provider_stream_event_api_format,
LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
insert_provider_stream_event_api_format, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_openai_responses_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
@@ -51,11 +52,16 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
&candidate_id,
spec,
)
.await
.await?
else {
return Ok(None);
};
let candidate = &eligible.candidate;
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
Some(body_json)
};
let prompt_cache_key = resolved
.provider_request_body
@@ -80,6 +86,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
}
if let Some(envelope_name) = resolved.envelope_name {
extra_fields.insert("envelope_name".to_string(), json!(envelope_name));
insert_native_client_envelope_name(&mut extra_fields, envelope_name, parts.uri.path());
}
if let Some(image_request_summary) = resolved.image_request_summary.as_ref() {
extra_fields.insert("image_request".to_string(), image_request_summary.clone());
@@ -95,15 +102,6 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
.eq_ignore_ascii_case("chatgpt_web")
{
extra_fields.insert("chatgpt_web_image".to_string(), json!(true));
extra_fields.insert(
"local_failover_policy".to_string(),
json!({
"stop_status_codes": [400, 401, 403, 429, 500, 502, 503, 504],
"error_stop_patterns": [
{ "pattern": ".*" }
]
}),
);
}
insert_provider_stream_event_api_format(
&mut extra_fields,
@@ -141,9 +139,10 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -204,7 +203,9 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
transport,
transport_profile: _,
image_request_summary: _,
request_redacted: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
@@ -214,6 +215,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
@@ -231,6 +233,8 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
@@ -239,6 +243,10 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(input, &mut decision)?;
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -31,8 +31,8 @@ use crate::ai_serving::planner::spec_metadata::local_openai_responses_spec_metad
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::{
ai_local_execution_contract_for_formats, extract_pool_sticky_session_token,
resolve_local_decision_execution_runtime_auth_context, ExecutionRuntimeAuthContext,
GatewayControlDecision, PlannerAppState,
openai_responses_request_operation, resolve_local_decision_execution_runtime_auth_context,
ExecutionRuntimeAuthContext, GatewayControlDecision, PlannerAppState,
};
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::{AppState, GatewayError};
@@ -91,7 +91,9 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
state,
auth_context.clone(),
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -161,6 +163,7 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
spec: LocalOpenAiResponsesSpec,
) -> Result<(Vec<LocalOpenAiResponsesCandidateAttempt>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let request_operation = openai_responses_request_operation(spec_metadata.api_format, body_json);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
@@ -171,8 +174,10 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
);
let preselection = preselect_local_execution_candidates_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
request_operation,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -259,6 +264,7 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
spec: LocalOpenAiResponsesSpec,
) -> Result<(LocalOpenAiResponsesCandidateAttemptSource<'a>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let request_operation = openai_responses_request_operation(spec_metadata.api_format, body_json);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
@@ -280,9 +286,11 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
Ok(
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
spec_metadata.api_format,
&input.requested_model,
request_operation,
spec_metadata.require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
@@ -365,8 +373,10 @@ pub(crate) async fn build_local_openai_responses_image_candidate_attempt_source<
);
let preselection = preselect_local_execution_candidates_for_api_formats_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
false,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
@@ -114,7 +114,7 @@ pub(crate) async fn maybe_build_sync_local_openai_responses_decision_payload(
)
.await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -153,7 +153,7 @@ pub(crate) async fn maybe_build_stream_local_openai_responses_decision_payload(
)
.await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -162,7 +162,7 @@ pub(super) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiResponsesSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -185,12 +185,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiResponsesSyncAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiResponsesStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -213,6 +228,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiResponsesStream
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiResponsesSyncAttemptSource<'_> {
@@ -331,7 +361,7 @@ pub(super) async fn build_local_sync_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -403,7 +433,7 @@ pub(super) async fn build_local_stream_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
@@ -1,9 +1,8 @@
use std::collections::BTreeMap;
use aether_contracts::RequestBody;
use super::{
augment_sync_report_context, build_ai_execution_plan_from_decision, take_ai_decision_plan_core,
augment_sync_report_context, build_ai_execution_plan_from_decision,
resolve_ai_passthrough_sync_request_body, take_ai_decision_plan_core,
take_ai_upstream_auth_pair, take_non_empty_string, AiExecutionPlanFromDecisionParts,
AiStreamAttempt, AiSyncAttempt,
};
@@ -63,6 +62,10 @@ pub(crate) fn build_standard_sync_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
@@ -72,7 +75,7 @@ pub(crate) fn build_standard_sync_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream,
},
);
@@ -146,6 +149,11 @@ pub(crate) fn build_standard_stream_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -154,8 +162,8 @@ pub(crate) fn build_standard_stream_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
body: request_body,
stream,
},
);
@@ -165,3 +173,88 @@ pub(crate) fn build_standard_stream_plan_from_decision(
report_context,
}))
}
#[cfg(test)]
mod tests {
use aether_contracts::{ExecutionResponseBodyMode, EXECUTION_RESPONSE_BODY_MODE_HEADER};
use serde_json::json;
use super::{
build_standard_stream_plan_from_decision, build_standard_sync_plan_from_decision,
AiExecutionDecision,
};
fn decision_with_raw_body(upstream_is_stream: bool) -> AiExecutionDecision {
serde_json::from_value(json!({
"action": if upstream_is_stream { "stream" } else { "sync" },
"request_id": "req-raw",
"provider_id": "provider-raw",
"endpoint_id": "endpoint-raw",
"key_id": "key-raw",
"upstream_url": "https://api.anthropic.test/v1/messages",
"provider_api_format": "claude:messages",
"client_api_format": "claude:messages",
"provider_request_headers": {
"content-type": "application/json",
(EXECUTION_RESPONSE_BODY_MODE_HEADER): ExecutionResponseBodyMode::PreserveBytes.as_str()
},
"provider_request_body": {
"model": "claude-sonnet-4",
"messages": []
},
"provider_request_body_base64": "eyAibW9kZWwiOiAiY2xhdWRlLXNvbm5ldC00IiwgIm1lc3NhZ2VzIjogW10gfQ==",
"content_type": "application/json",
"upstream_is_stream": upstream_is_stream
}))
.expect("decision should deserialize")
}
fn request_parts() -> http::request::Parts {
http::Request::builder()
.uri("http://localhost/v1/messages")
.body(())
.expect("request should build")
.into_parts()
.0
}
#[test]
fn standard_sync_plan_prefers_exact_request_body_bytes() {
let built = build_standard_sync_plan_from_decision(
&request_parts(),
&json!({}),
decision_with_raw_body(false),
)
.expect("plan should build")
.expect("plan should exist");
assert!(built.plan.body.json_body.is_none());
assert_eq!(
built.plan.body.body_bytes_b64.as_deref(),
Some("eyAibW9kZWwiOiAiY2xhdWRlLXNvbm5ldC00IiwgIm1lc3NhZ2VzIjogW10gfQ==")
);
assert_eq!(
built
.plan
.headers
.get(EXECUTION_RESPONSE_BODY_MODE_HEADER)
.map(String::as_str),
Some(ExecutionResponseBodyMode::PreserveBytes.as_str())
);
}
#[test]
fn standard_stream_plan_prefers_exact_request_body_bytes() {
let built = build_standard_stream_plan_from_decision(
&request_parts(),
&json!({}),
decision_with_raw_body(true),
false,
)
.expect("plan should build")
.expect("plan should exist");
assert!(built.plan.body.json_body.is_none());
assert!(built.plan.body.body_bytes_b64.is_some());
}
}
@@ -10,9 +10,7 @@ impl<'a> PlannerAppState<'a> {
now_unix_secs: u64,
) -> Result<Option<GatewayAuthApiKeySnapshot>, GatewayError> {
self.app()
.data
.read_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
.read_cached_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
.await
.map_err(|err| GatewayError::Internal(err.to_string()))
}
}
@@ -10,16 +10,15 @@ impl<'a> PlannerAppState<'a> {
api_key_id: &str,
requested_model: Option<&str>,
explicit_required_capabilities: Option<&Value>,
model_directive_base_model: Option<&str>,
) -> Option<Value> {
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled(self.app()).await;
crate::request_candidate_runtime::resolve_request_candidate_required_capabilities(
self.app(),
user_id,
api_key_id,
requested_model,
explicit_required_capabilities,
enable_model_directives,
model_directive_base_model,
)
.await
}
@@ -20,14 +20,8 @@ impl<'a> PlannerAppState<'a> {
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
) -> Result<Vec<SchedulerMinimalCandidateSelectionCandidate>, GatewayError> {
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
self.app(),
api_format,
Some(global_model_name),
)
.await;
crate::scheduler::candidate::list_selectable_candidates(
self.app().data.as_ref(),
self.app(),
@@ -52,6 +46,39 @@ impl<'a> PlannerAppState<'a> {
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
Vec<SchedulerSkippedCandidate>,
),
GatewayError,
> {
self.list_selectable_candidates_with_skip_reasons_for_request_operation(
api_format,
global_model_name,
require_streaming,
required_capabilities,
auth_snapshot,
client_session_affinity,
now_unix_secs,
enable_model_directives,
None,
)
.await
}
pub(crate) async fn list_selectable_candidates_with_skip_reasons_for_request_operation(
self,
api_format: &str,
global_model_name: &str,
require_streaming: bool,
required_capabilities: Option<&serde_json::Value>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
now_unix_secs: u64,
enable_model_directives: bool,
request_operation: Option<&str>,
) -> Result<
(
Vec<SchedulerMinimalCandidateSelectionCandidate>,
@@ -63,16 +90,8 @@ impl<'a> PlannerAppState<'a> {
let wait_interval = Duration::from_millis(API_KEY_CONCURRENCY_WAIT_POLL_INTERVAL_MS.max(1));
let wait_deadline = Instant::now() + wait_timeout;
let mut attempt_now_unix_secs = now_unix_secs;
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
self.app(),
api_format,
Some(global_model_name),
)
.await;
loop {
let result = crate::scheduler::candidate::list_selectable_candidates_with_skip_reasons(
let result = crate::scheduler::candidate::list_selectable_candidates_with_skip_reasons_for_request_operation(
self.app().data.as_ref(),
self.app(),
api_format,
@@ -83,6 +102,7 @@ impl<'a> PlannerAppState<'a> {
client_session_affinity,
attempt_now_unix_secs,
enable_model_directives,
request_operation,
)
.await?;
@@ -3,8 +3,20 @@ pub(crate) use crate::ai_serving::transport::{
GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
};
use crate::GatewayError;
use std::sync::Arc;
impl<'a> PlannerAppState<'a> {
pub(crate) async fn read_provider_transport_snapshot_arc(
self,
provider_id: &str,
endpoint_id: &str,
key_id: &str,
) -> Result<Option<Arc<GatewayProviderTransportSnapshot>>, GatewayError> {
self.app()
.read_provider_transport_snapshot_arc(provider_id, endpoint_id, key_id)
.await
}
pub(crate) async fn read_provider_transport_snapshot(
self,
provider_id: &str,
+111 -56
View File
@@ -3,14 +3,21 @@ pub(crate) use aether_ai_formats::api::{
aggregate_openai_chat_stream_sync_response, aggregate_openai_responses_stream_sync_response,
aggregate_standard_chat_stream_sync_response, aggregate_standard_cli_stream_sync_response,
api_format_alias_matches, api_format_storage_aliases,
apply_codex_openai_responses_chat_body_edits, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_responses_special_headers, apply_model_directive_mapping_patch,
apply_codex_openai_compact_terminal_headers, apply_codex_openai_responses_chat_body_edits,
apply_codex_openai_responses_identity_headers,
apply_codex_openai_responses_lite_header_for_request_body_with_capabilities,
apply_codex_openai_responses_lite_header_with_capabilities,
apply_codex_openai_responses_special_body_edits,
apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities,
apply_codex_openai_special_headers, apply_model_directive_mapping_patch,
apply_model_directive_overrides_from_model, apply_model_directive_overrides_from_request,
apply_openai_responses_compact_special_body_edits, build_chatgpt_web_image_request_body,
build_codex_model_catalog_metadata, build_codex_openai_image_api_provider_request_body,
build_core_error_body_for_client_format, build_cross_format_openai_chat_request_body,
build_cross_format_openai_chat_request_body_with_model_directives,
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_model_directives,
build_cross_format_openai_responses_request_body_with_model_directives_and_history_scope,
build_gemini_image_request_body_from_openai_image_request,
build_gemini_image_response_from_openai_image_response,
build_gemini_image_response_from_openai_responses_image_response, build_generated_tool_call_id,
@@ -41,17 +48,21 @@ pub(crate) use aether_ai_formats::api::{
convert_standard_chat_response, convert_standard_cli_response, copy_request_number_field,
copy_request_number_field_as, core_error_background_report_kind,
core_error_default_client_api_format, core_success_background_report_kind,
default_model_directive_mapping_patch, default_model_directive_suffixes,
default_model_for_openai_image_operation, encode_done_sse, encode_json_sse,
encode_kiro_sse_events, endpoint_config_forces_upstream_stream_policy,
enforce_request_body_stream_field, estimate_kiro_tokens, extract_openai_text_content,
finalize_openai_provider_request,
finalize_openai_provider_request_with_codex_model_capabilities,
find_kiro_real_thinking_end_tag, find_kiro_real_thinking_end_tag_at_buffer_end,
find_kiro_real_thinking_start_tag, force_upstream_streaming_for_provider,
gemini_request_is_image_generation, implicit_sync_finalize_report_kind,
is_core_error_finalize_kind, is_matching_stream_http_request, is_matching_stream_request,
is_openai_image_stream_request, is_openai_responses_family_format, is_openai_responses_format,
kiro_crc32, map_claude_stop_reason, map_openai_reasoning_effort_to_claude_output,
map_openai_reasoning_effort_to_gemini_budget, maybe_bridge_standard_sync_json_to_stream,
maybe_build_ai_surface_stream_rewriter,
find_kiro_real_thinking_start_tag, forbid_upstream_streaming_for_provider,
force_upstream_streaming_for_provider, gemini_request_is_image_generation,
hydrate_response_history, implicit_sync_finalize_report_kind, is_core_error_finalize_kind,
is_matching_stream_http_request, is_matching_stream_request, is_openai_image_stream_request,
is_openai_responses_compact_format, is_openai_responses_family_format,
is_openai_responses_format, kiro_crc32, map_claude_stop_reason,
map_openai_reasoning_effort_to_claude_output, map_openai_reasoning_effort_to_gemini_budget,
maybe_bridge_standard_sync_json_to_stream, maybe_build_ai_surface_stream_rewriter,
maybe_build_openai_chat_cross_format_sync_product_from_normalized_payload,
maybe_build_openai_image_sync_finalize_product,
maybe_build_openai_responses_cross_format_sync_product_from_normalized_payload,
@@ -60,23 +71,31 @@ pub(crate) use aether_ai_formats::api::{
maybe_build_standard_cross_format_sync_product_from_normalized_payload,
maybe_build_standard_same_format_sync_body_from_normalized_payload,
maybe_build_standard_sync_finalize_product_from_normalized_payload, model_directive_base_model,
normalize_api_format_alias, normalize_claude_request_to_openai_chat_request,
normalize_gemini_request_to_openai_chat_request, normalize_openai_image_request,
normalize_openai_image_request_with_options,
model_directive_builtin_suffix_supported_for_source_model,
model_directive_suffix_has_builtin_mapping, normalize_api_format_alias,
normalize_claude_request_to_openai_chat_request,
normalize_gemini_request_to_openai_chat_request, normalize_openai_image_quality,
normalize_openai_image_request, normalize_openai_image_request_with_options,
normalize_openai_responses_request_to_openai_chat_request,
normalize_provider_private_report_context, normalize_provider_private_response_value,
normalize_standard_request_to_openai_chat_request, openai_image_operation_from_path,
parse_direct_request_body, parse_openai_stop_sequences, parse_openai_tool_result_content,
prepare_local_success_response_parts, prepare_local_success_response_parts_owned,
provider_adaptation_allows_sync_finalize_envelope, provider_adaptation_anchor_api_format,
provider_adaptation_descriptor_for_envelope, provider_adaptation_descriptor_for_provider_type,
parse_codex_auth_identity, parse_direct_request_body, parse_model_directive,
parse_model_directive_with_suffixes, parse_openai_stop_sequences,
parse_openai_tool_result_content, prepare_local_success_response_parts,
prepare_local_success_response_parts_owned, project_codex_openai_image_api_request_body,
project_openai_image_api_request_body, provider_adaptation_allows_sync_finalize_envelope,
provider_adaptation_anchor_api_format, provider_adaptation_descriptor_for_envelope,
provider_adaptation_descriptor_for_provider_type,
provider_adaptation_requires_eventstream_accept,
provider_adaptation_should_unwrap_stream_envelope,
provider_private_response_allows_sync_finalize, request_candidate_api_format_preference,
request_candidate_api_formats, request_conversion_kind,
request_conversion_requires_enable_flag, request_path_implies_stream_request,
resolve_claude_stream_spec, resolve_claude_sync_spec,
resolve_execution_runtime_stream_plan_kind, resolve_execution_runtime_sync_plan_kind,
provider_private_response_allows_sync_finalize, record_converted_response_history,
request_candidate_api_format_preference, request_candidate_api_formats,
request_conversion_kind, request_conversion_requires_enable_flag,
request_path_implies_stream_request, resolve_claude_stream_spec, resolve_claude_sync_spec,
resolve_codex_responses_model_capabilities, resolve_execution_runtime_stream_plan_kind,
resolve_execution_runtime_stream_plan_kind_with_client_surface,
resolve_execution_runtime_sync_plan_kind,
resolve_execution_runtime_sync_plan_kind_with_client_surface,
resolve_finalize_stream_rewrite_mode, resolve_gemini_files_stream_spec,
resolve_gemini_files_sync_spec, resolve_gemini_stream_spec, resolve_gemini_sync_spec,
resolve_local_image_stream_spec, resolve_local_image_sync_spec,
@@ -85,24 +104,28 @@ pub(crate) use aether_ai_formats::api::{
resolve_openai_embedding_sync_spec, resolve_openai_responses_stream_spec,
resolve_openai_responses_sync_spec, resolve_requested_gemini_image_model_for_request,
resolve_requested_openai_image_model_for_request,
resolve_upstream_is_stream_from_endpoint_config, sanitize_request_path,
sanitize_request_path_and_query, sanitize_request_query_string,
stream_body_contains_error_event, supports_stream_execution_decision_kind,
supports_sync_execution_decision_kind, sync_chat_response_conversion_kind,
sync_cli_response_conversion_kind, transform_provider_private_stream_line, value_as_u64,
AiControlPlanRequest, AiSurfaceFinalizeError, AiSurfaceStreamRewriter, CanonicalStreamFrame,
ChatGptWebImageRequestError, ClaudeClientEmitter, ClaudeProviderState,
ExecutionRuntimeAuthContext, FinalizeStreamRewriteMode, FormatContext, GeminiClientEmitter,
GeminiImageRequestForOpenAi, GeminiProviderState, KiroToClaudeCliStreamState,
LocalCoreSyncErrorKind, LocalGeminiFilesSpec, LocalOpenAiImageSpec, LocalOpenAiResponsesSpec,
LocalSameFormatProviderFamily, LocalSameFormatProviderSpec, LocalStandardSourceFamily,
LocalStandardSourceMode, LocalStandardSpec, LocalSyncReportParts, LocalVideoCreateFamily,
LocalVideoCreateSpec, NormalizedOpenAiImageRequest, OpenAIChatClientEmitter,
OpenAIChatProviderState, OpenAIResponsesClientEmitter, OpenAIResponsesProviderState,
OpenAiImageNormalizeOptions, OpenAiImageOperation, OpenAiImageRequestForGemini,
OpenAiImageResponseFormat, OpenAiImageStreamState, OpenAiImageSyncFinalizeProduct,
resolve_upstream_is_stream_for_provider as resolve_format_upstream_is_stream_for_provider,
resolve_upstream_is_stream_from_endpoint_config, response_history_is_loaded,
response_history_storage_key, sanitize_request_path, sanitize_request_path_and_query,
sanitize_request_query_string, stream_body_contains_error_event,
supports_stream_execution_decision_kind, supports_sync_execution_decision_kind,
sync_chat_response_conversion_kind, sync_cli_response_conversion_kind,
transform_provider_private_stream_line, validate_openai_provider_request_contract,
value_as_u64, AiControlPlanRequest, AiSurfaceFinalizeError, AiSurfaceStreamRewriter,
CanonicalStreamFrame, ChatGptWebImageRequestError, ClaudeClientEmitter, ClaudeProviderState,
CodexResponsesModelCapabilities, ExecutionRuntimeAuthContext, FinalizeStreamRewriteMode,
FormatContext, GeminiClientEmitter, GeminiImageRequestForOpenAi, GeminiProviderState,
KiroToClaudeCliStreamState, LocalCoreSyncErrorKind, LocalGeminiFilesSpec, LocalOpenAiImageSpec,
LocalOpenAiResponsesSpec, LocalSameFormatProviderFamily, LocalSameFormatProviderSpec,
LocalStandardSourceFamily, LocalStandardSourceMode, LocalStandardSpec, LocalSyncReportParts,
LocalVideoCreateFamily, LocalVideoCreateSpec, NormalizedOpenAiImageRequest,
OpenAIChatClientEmitter, OpenAIChatProviderState, OpenAIResponsesClientEmitter,
OpenAIResponsesProviderState, OpenAiImageNormalizeOptions, OpenAiImageOperation,
OpenAiImageRequestForGemini, OpenAiImageResponseFormat, OpenAiImageStreamState,
OpenAiImageSyncFinalizeProduct, OpenAiProviderRequestFinalization,
ProviderAdaptationDescriptor, ProviderAdaptationSurface, ProviderPrivateStreamNormalizer,
RequestConversionKind, StandardCrossFormatSyncProduct, StandardSyncFinalizeNormalizedProduct,
ReasoningEffort, RequestConversionKind, ResponseHistoryRecord, ServiceTier,
StandardCrossFormatSyncProduct, StandardSyncFinalizeNormalizedProduct,
StreamingStandardFormatMatrix, SyncChatResponseConversionKind, SyncCliResponseConversionKind,
SyncToStreamBridgeOutcome, ANTIGRAVITY_V1INTERNAL_ENVELOPE_NAME, CLAUDE_CHAT_STREAM_PLAN_KIND,
CLAUDE_CHAT_STREAM_SUCCESS_REPORT_KIND, CLAUDE_CHAT_SYNC_ERROR_REPORT_KIND,
@@ -110,14 +133,15 @@ pub(crate) use aether_ai_formats::api::{
CLAUDE_CHAT_SYNC_SUCCESS_REPORT_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_STREAM_SUCCESS_REPORT_KIND, CLAUDE_CLI_SYNC_ERROR_REPORT_KIND,
CLAUDE_CLI_SYNC_FINALIZE_REPORT_KIND, CLAUDE_CLI_SYNC_PLAN_KIND,
CLAUDE_CLI_SYNC_SUCCESS_REPORT_KIND, CODEX_OPENAI_IMAGE_DEFAULT_MODEL,
CODEX_OPENAI_IMAGE_DEFAULT_OUTPUT_FORMAT, CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_MODEL,
CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_PROMPT, CODEX_OPENAI_IMAGE_INTERNAL_MODEL,
EXECUTION_RUNTIME_STREAM_ACTION, EXECUTION_RUNTIME_STREAM_DECISION_ACTION,
EXECUTION_RUNTIME_SYNC_ACTION, EXECUTION_RUNTIME_SYNC_DECISION_ACTION,
GEMINI_CHAT_STREAM_PLAN_KIND, GEMINI_CHAT_STREAM_SUCCESS_REPORT_KIND,
GEMINI_CHAT_SYNC_ERROR_REPORT_KIND, GEMINI_CHAT_SYNC_FINALIZE_REPORT_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CHAT_SYNC_SUCCESS_REPORT_KIND, GEMINI_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_SUCCESS_REPORT_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND,
CODEX_OPENAI_IMAGE_DEFAULT_MODEL, CODEX_OPENAI_IMAGE_DEFAULT_OUTPUT_FORMAT,
CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_MODEL, CODEX_OPENAI_IMAGE_DEFAULT_VARIATION_PROMPT,
CODEX_OPENAI_IMAGE_INTERNAL_MODEL, EXECUTION_RUNTIME_STREAM_ACTION,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_STREAM_SUCCESS_REPORT_KIND, GEMINI_CHAT_SYNC_ERROR_REPORT_KIND,
GEMINI_CHAT_SYNC_FINALIZE_REPORT_KIND, GEMINI_CHAT_SYNC_PLAN_KIND,
GEMINI_CHAT_SYNC_SUCCESS_REPORT_KIND, GEMINI_CLI_STREAM_PLAN_KIND,
GEMINI_CLI_STREAM_SUCCESS_REPORT_KIND, GEMINI_CLI_SYNC_ERROR_REPORT_KIND,
GEMINI_CLI_SYNC_FINALIZE_REPORT_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
GEMINI_CLI_SYNC_SUCCESS_REPORT_KIND, GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME,
@@ -125,22 +149,53 @@ pub(crate) use aether_ai_formats::api::{
GEMINI_FILES_DELETE_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND,
GEMINI_FILES_LIST_PLAN_KIND, GEMINI_FILES_UPLOAD_PLAN_KIND, GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND,
GEMINI_VIDEO_CREATE_SYNC_FINALIZE_REPORT_KIND, GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND,
KIRO_ENVELOPE_NAME, KIRO_MAX_THINKING_BUFFER, OPENAI_CHAT_STREAM_PLAN_KIND,
OPENAI_CHAT_STREAM_SUCCESS_REPORT_KIND, OPENAI_CHAT_SYNC_ERROR_REPORT_KIND,
OPENAI_CHAT_SYNC_FINALIZE_REPORT_KIND, OPENAI_CHAT_SYNC_PLAN_KIND,
OPENAI_CHAT_SYNC_SUCCESS_REPORT_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_STREAM_SUCCESS_REPORT_KIND,
OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_IMAGE_SYNC_SUCCESS_REPORT_KIND, OPENAI_RERANK_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_SUCCESS_REPORT_KIND,
KIRO_ENVELOPE_NAME, KIRO_MAX_THINKING_BUFFER, MODEL_DIRECTIVE_API_FORMATS,
OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_STREAM_SUCCESS_REPORT_KIND,
OPENAI_CHAT_SYNC_ERROR_REPORT_KIND, OPENAI_CHAT_SYNC_FINALIZE_REPORT_KIND,
OPENAI_CHAT_SYNC_PLAN_KIND, OPENAI_CHAT_SYNC_SUCCESS_REPORT_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND,
OPENAI_IMAGE_STREAM_SUCCESS_REPORT_KIND, OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_IMAGE_SYNC_SUCCESS_REPORT_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_SUCCESS_REPORT_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_ERROR_REPORT_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_FINALIZE_REPORT_KIND, OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_SUCCESS_REPORT_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_STREAM_SUCCESS_REPORT_KIND, OPENAI_RESPONSES_SYNC_ERROR_REPORT_KIND,
OPENAI_RESPONSES_SYNC_FINALIZE_REPORT_KIND, OPENAI_RESPONSES_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_SUCCESS_REPORT_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_SUCCESS_REPORT_KIND, OPENAI_SEARCH_SYNC_PLAN_KIND,
OPENAI_SEARCH_SYNC_SUCCESS_REPORT_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_formats::{is_embedding_api_format, is_rerank_api_format};
pub(crate) use aether_ai_formats::{
api_format_defaults_to_client_error_failover, api_format_defaults_to_non_stream,
api_format_permission_covers, intersect_api_format_allowed_lists, is_embedding_api_format,
is_rerank_api_format, openai_responses_request_operation,
openai_responses_synthetic_reasoning_item_id,
strip_incompatible_openai_responses_reasoning_items, ApiOperation, ClientSurface,
};
pub(crate) fn plan_kind_matches_api_operation(
plan_kind: &str,
require_streaming: bool,
expected_operation: Option<ApiOperation>,
) -> bool {
let Some(expected_operation) = expected_operation else {
return true;
};
if expected_operation == ApiOperation::OpenAiResponsesCompact {
return if require_streaming {
plan_kind == OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND
} else {
plan_kind == OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND
};
}
let resolved_operation = if require_streaming {
resolve_local_same_format_stream_spec(plan_kind).and_then(|spec| spec.operation)
} else {
resolve_local_same_format_sync_spec(plan_kind).and_then(|spec| spec.operation)
};
resolved_operation == Some(expected_operation)
}
@@ -0,0 +1,96 @@
use crate::ai_serving::{
hydrate_response_history, normalize_api_format_alias, record_converted_response_history,
response_history_is_loaded, response_history_storage_key, ResponseHistoryRecord,
};
use aether_runtime_state::RuntimeState;
use serde_json::Value;
use tracing::warn;
use crate::GatewayError;
pub(crate) async fn hydrate_openai_response_history(
runtime_state: &RuntimeState,
request: &Value,
client_api_format: &str,
provider_api_format: &str,
history_scope: &str,
) -> Result<(), GatewayError> {
if normalize_api_format_alias(client_api_format) != "openai:responses"
|| normalize_api_format_alias(provider_api_format) != "openai:chat"
{
return Ok(());
}
let Some(previous_response_id) = request
.get("previous_response_id")
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
else {
return Ok(());
};
if response_history_is_loaded(previous_response_id, Some(history_scope)) {
return Ok(());
}
let storage_key = response_history_storage_key(previous_response_id, Some(history_scope));
let payload = runtime_state.kv_get(&storage_key).await.map_err(|error| {
warn!(
event_name = "openai_response_history_read_failed",
log_type = "ops",
backend = runtime_state.backend_kind().as_str(),
error = ?error,
"gateway failed to read shared OpenAI response history"
);
GatewayError::Internal("OpenAI response history lookup failed".to_string())
})?;
let Some(payload) = payload else {
return Ok(());
};
if let Err(error) =
hydrate_response_history(previous_response_id, Some(history_scope), &payload)
{
let _ = runtime_state.kv_delete(&storage_key).await;
warn!(
event_name = "openai_response_history_invalid",
log_type = "ops",
backend = runtime_state.backend_kind().as_str(),
error = %error,
"gateway rejected invalid shared OpenAI response history"
);
return Err(GatewayError::Internal(
"OpenAI response history validation failed".to_string(),
));
}
Ok(())
}
pub(crate) async fn persist_response_history_record(
runtime_state: &RuntimeState,
record: ResponseHistoryRecord,
) {
if let Err(error) = runtime_state
.kv_set(&record.storage_key, record.payload, Some(record.ttl))
.await
{
warn!(
event_name = "openai_response_history_write_failed",
log_type = "ops",
backend = runtime_state.backend_kind().as_str(),
error = ?error,
"gateway failed to persist shared OpenAI response history"
);
}
}
pub(crate) async fn persist_converted_response_history(
runtime_state: &RuntimeState,
report_context: &Value,
response: Option<&Value>,
) {
let Some(response) = response else {
return;
};
if let Some(record) = record_converted_response_history(report_context, response) {
persist_response_history_record(runtime_state, record).await;
}
}
+33 -19
View File
@@ -18,6 +18,10 @@ pub(crate) mod grok {
pub(crate) use aether_provider_transport::grok::*;
}
pub(crate) mod gemini_cli {
pub(crate) use aether_provider_transport::gemini_cli::*;
}
pub(crate) mod oauth_refresh {
pub(crate) use aether_provider_transport::oauth_refresh::*;
}
@@ -55,19 +59,21 @@ pub(crate) mod windsurf {
}
pub(crate) use aether_provider_transport::{
append_transport_diagnostics_to_value, apply_local_body_rules,
apply_local_body_rules_with_request_headers, apply_local_header_rules,
append_transport_diagnostics_to_value, apply_local_auth_config_header_overrides,
apply_local_body_rules, apply_local_body_rules_with_request_headers, apply_local_header_rules,
apply_local_header_rules_with_request_headers, apply_standard_provider_request_body_rules,
apply_standard_provider_request_body_rules_with_request_headers,
apply_transport_request_body_semantics, body_rules_are_locally_supported,
body_rules_handle_path, body_rules_have_enabled_rules,
build_cross_format_openai_chat_upstream_url, build_cross_format_openai_responses_upstream_url,
build_gemini_files_headers, build_gemini_files_request_body, build_gemini_files_upstream_url,
build_grok_app_chat_body, build_grok_browser_headers, build_grok_upstream_url,
build_kiro_cross_format_upstream_url, build_local_openai_chat_upstream_url,
build_local_openai_responses_upstream_url, build_openai_image_headers,
build_openai_image_upstream_url, build_passthrough_headers, build_request_trace_proxy_value,
build_same_format_provider_headers, build_same_format_provider_request_body,
build_gemini_cli_v1internal_request, build_gemini_files_headers,
build_gemini_files_request_body, build_gemini_files_upstream_url, build_grok_app_chat_body,
build_grok_browser_headers, build_grok_upstream_url, build_kiro_cross_format_upstream_url,
build_local_openai_chat_upstream_url, build_local_openai_responses_upstream_url,
build_openai_image_headers, build_openai_image_upstream_url, build_passthrough_headers,
build_request_trace_proxy_value, build_same_format_provider_headers,
build_same_format_provider_request_body,
build_same_format_provider_request_body_with_compatibility_report,
build_same_format_provider_upstream_url, build_standard_plan_fallback_headers,
build_standard_plan_fallback_openai_chat_url,
build_standard_plan_fallback_openai_responses_url, build_standard_provider_request_headers,
@@ -76,8 +82,10 @@ pub(crate) use aether_provider_transport::{
build_windsurf_cascade_headers, build_windsurf_cascade_request_body,
build_windsurf_cascade_upstream_url, candidate_common_transport_skip_reason,
candidate_transport_pair_skip_reason, classify_same_format_provider_request_behavior,
ensure_upstream_auth_header, gemini_files_transport_unsupported_reason,
header_rules_are_locally_supported, header_rules_have_enabled_rules,
classify_same_format_provider_request_behavior_for_operation,
enforce_same_format_provider_api_operation_body_policy, ensure_upstream_auth_header,
gemini_files_transport_unsupported_reason, header_rules_are_locally_supported,
header_rules_have_enabled_rules, is_gemini_cli_provider_transport,
is_windsurf_provider_transport, local_gemini_transport_unsupported_reason_with_network,
local_openai_chat_transport_unsupported_reason,
local_standard_transport_unsupported_reason_with_network,
@@ -85,8 +93,10 @@ pub(crate) use aether_provider_transport::{
openai_image_transport_unsupported_reason, request_conversion_direct_auth,
request_conversion_enabled_for_transport, request_conversion_transport_supported,
request_conversion_transport_unsupported_reason, request_pair_allowed_for_transport,
request_pair_direct_auth, request_pair_transport_unsupported_reason, resolve_gemini_files_auth,
resolve_grok_session_auth, resolve_openai_image_auth, resolve_same_format_provider_direct_auth,
request_pair_direct_auth, request_pair_transport_unsupported_reason,
resolve_anthropic_compatibility_profile, resolve_gemini_cli_project_id,
resolve_gemini_files_auth, resolve_grok_session_auth, resolve_local_gemini_cli_request_auth,
resolve_openai_image_auth, resolve_same_format_provider_direct_auth,
resolve_transport_execution_timeouts, resolve_transport_profile,
resolve_transport_proxy_snapshot, resolve_transport_proxy_snapshot_with_tunnel_affinity,
resolve_video_create_auth, same_format_provider_transport_supported,
@@ -94,15 +104,19 @@ pub(crate) use aether_provider_transport::{
should_try_same_format_provider_oauth_auth, supports_local_gemini_transport_with_network,
supports_local_generic_oauth_request_auth_resolution,
supports_local_oauth_request_auth_resolution, transport_proxy_is_locally_supported,
video_create_transport_unsupported_reason, CandidateTransportPolicyFacts,
GatewayProviderTransportSnapshot, GeminiFilesHeadersInput, GeminiFilesRequestBodyError,
transport_supports_api_operation, video_create_transport_unsupported_reason,
AnthropicCompatibilityProfile, CandidateTransportPolicyFacts, GatewayProviderTransportSnapshot,
GeminiCliRequestAuth, GeminiCliRequestAuthSupport, GeminiCliRequestAuthUnsupportedReason,
GeminiCliRequestEnvelopeSupport, GeminiFilesHeadersInput, GeminiFilesRequestBodyError,
GeminiFilesRequestBodyParts, GrokHeaderInput, LocalResolvedOAuthRequestAuth,
ProviderOpenAiImageHeadersInput, ProviderVideoCreateFamily, ProviderVideoCreateHeadersInput,
SameFormatProviderCompatibilityEdit, SameFormatProviderCompatibilityEditAction,
SameFormatProviderFamily, SameFormatProviderHeadersInput, SameFormatProviderRequestBehavior,
SameFormatProviderRequestBehaviorParams, SameFormatProviderRequestBodyInput,
SameFormatProviderUpstreamUrlParams, StandardPlanFallbackAcceptPolicy,
StandardPlanFallbackHeadersInput, StandardProviderRequestHeaders,
StandardProviderRequestHeadersInput, TransportRequestBodySemanticsError,
TransportRequestUrlParams, GROK_CHAT_PATH, GROK_INTERNAL_HEADER, GROK_RATE_LIMITS_PATH,
WINDSURF_ENVELOPE_NAME,
SameFormatProviderRequestBodyOutput, SameFormatProviderUpstreamUrlParams,
StandardPlanFallbackAcceptPolicy, StandardPlanFallbackHeadersInput,
StandardProviderRequestHeaders, StandardProviderRequestHeadersInput,
TransportRequestBodySemanticsError, TransportRequestUrlParams, GEMINI_CLI_USER_AGENT,
GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME, GROK_CHAT_PATH, GROK_INTERNAL_HEADER,
GROK_RATE_LIMITS_PATH, WINDSURF_ENVELOPE_NAME,
};
@@ -0,0 +1,218 @@
use aether_runtime::{MetricKind, MetricSample};
pub(crate) fn gateway_allocator_metric_samples() -> Vec<MetricSample> {
match allocator_snapshot() {
Some(snapshot) => snapshot.to_metric_samples(),
None => unavailable_metric_samples(),
}
}
#[derive(Debug, Clone, Copy, Default)]
struct AllocatorSnapshot {
allocated_bytes: u64,
active_bytes: u64,
resident_bytes: u64,
mapped_bytes: u64,
retained_bytes: u64,
metadata_bytes: u64,
}
impl AllocatorSnapshot {
fn to_metric_samples(self) -> Vec<MetricSample> {
vec![
gauge(
"gateway_allocator_observability_available",
"Whether gateway allocator heap metrics were available for this scrape.",
1,
),
gauge(
"gateway_allocator_allocated_bytes",
"Bytes currently allocated by the gateway allocator.",
self.allocated_bytes,
),
gauge(
"gateway_allocator_active_bytes",
"Bytes in active pages managed by the gateway allocator.",
self.active_bytes,
),
gauge(
"gateway_allocator_resident_bytes",
"Bytes resident in physical memory for the gateway allocator.",
self.resident_bytes,
),
gauge(
"gateway_allocator_mapped_bytes",
"Bytes mapped by the gateway allocator.",
self.mapped_bytes,
),
gauge(
"gateway_allocator_retained_bytes",
"Bytes retained by the gateway allocator for future use.",
self.retained_bytes,
),
gauge(
"gateway_allocator_metadata_bytes",
"Bytes used for allocator metadata.",
self.metadata_bytes,
),
gauge(
"gateway_allocator_active_to_allocated_basis_points",
"Active allocator bytes divided by allocated bytes in basis points.",
ratio_basis_points(self.active_bytes, self.allocated_bytes),
),
gauge(
"gateway_allocator_resident_to_allocated_basis_points",
"Resident allocator bytes divided by allocated bytes in basis points.",
ratio_basis_points(self.resident_bytes, self.allocated_bytes),
),
]
}
}
fn unavailable_metric_samples() -> Vec<MetricSample> {
vec![
gauge(
"gateway_allocator_observability_available",
"Whether gateway allocator heap metrics were available for this scrape.",
0,
),
gauge(
"gateway_allocator_allocated_bytes",
"Bytes currently allocated by the gateway allocator.",
0,
),
gauge(
"gateway_allocator_active_bytes",
"Bytes in active pages managed by the gateway allocator.",
0,
),
gauge(
"gateway_allocator_resident_bytes",
"Bytes resident in physical memory for the gateway allocator.",
0,
),
gauge(
"gateway_allocator_mapped_bytes",
"Bytes mapped by the gateway allocator.",
0,
),
gauge(
"gateway_allocator_retained_bytes",
"Bytes retained by the gateway allocator for future use.",
0,
),
gauge(
"gateway_allocator_metadata_bytes",
"Bytes used for allocator metadata.",
0,
),
gauge(
"gateway_allocator_active_to_allocated_basis_points",
"Active allocator bytes divided by allocated bytes in basis points.",
0,
),
gauge(
"gateway_allocator_resident_to_allocated_basis_points",
"Resident allocator bytes divided by allocated bytes in basis points.",
0,
),
]
}
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
fn allocator_snapshot() -> Option<AllocatorSnapshot> {
refresh_jemalloc_epoch()?;
Some(AllocatorSnapshot {
allocated_bytes: read_jemalloc_stat("stats.allocated\0")?,
active_bytes: read_jemalloc_stat("stats.active\0")?,
resident_bytes: read_jemalloc_stat("stats.resident\0")?,
mapped_bytes: read_jemalloc_stat("stats.mapped\0")?,
retained_bytes: read_jemalloc_stat("stats.retained\0")?,
metadata_bytes: read_jemalloc_stat("stats.metadata\0")?,
})
}
#[cfg(not(all(feature = "jemalloc", not(target_env = "msvc"))))]
fn allocator_snapshot() -> Option<AllocatorSnapshot> {
None
}
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
fn refresh_jemalloc_epoch() -> Option<()> {
let mut epoch = 1_u64;
let result = unsafe {
tikv_jemalloc_sys::mallctl(
c"epoch".as_ptr(),
std::ptr::null_mut(),
std::ptr::null_mut(),
(&mut epoch as *mut u64).cast(),
std::mem::size_of::<u64>(),
)
};
if result == 0 {
Some(())
} else {
None
}
}
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
fn read_jemalloc_stat(name: &str) -> Option<u64> {
let mut value = 0_usize;
let mut size = std::mem::size_of::<usize>();
let result = unsafe {
tikv_jemalloc_sys::mallctl(
name.as_ptr().cast(),
(&mut value as *mut usize).cast(),
&mut size,
std::ptr::null_mut(),
0,
)
};
if result == 0 {
Some(u64_from_usize(value))
} else {
None
}
}
fn gauge(name: &'static str, help: &'static str, value: u64) -> MetricSample {
MetricSample::new(name, help, MetricKind::Gauge, value)
}
fn ratio_basis_points(numerator: u64, denominator: u64) -> u64 {
if denominator == 0 {
return 0;
}
numerator.saturating_mul(10_000) / denominator
}
fn u64_from_usize(value: usize) -> u64 {
u64::try_from(value).unwrap_or(u64::MAX)
}
#[cfg(test)]
mod tests {
use super::{gateway_allocator_metric_samples, ratio_basis_points};
#[test]
fn renders_allocator_metric_samples() {
let samples = gateway_allocator_metric_samples();
assert!(samples
.iter()
.any(|sample| sample.name == "gateway_allocator_observability_available"));
assert!(samples
.iter()
.any(|sample| sample.name == "gateway_allocator_allocated_bytes"));
assert!(samples
.iter()
.any(|sample| sample.name == "gateway_allocator_active_to_allocated_basis_points"));
}
#[test]
fn computes_ratio_basis_points() {
assert_eq!(ratio_basis_points(150, 100), 15_000);
assert_eq!(ratio_basis_points(1, 0), 0);
}
}
+15
View File
@@ -0,0 +1,15 @@
pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
match crate::ai_serving::normalize_api_format_alias(api_format).as_str() {
"aliyun:multimodal_embedding" => Some("aliyun:multimodal_embedding"),
_ => None,
}
}
pub(crate) fn local_path(api_format: &str) -> Option<&'static str> {
match crate::ai_serving::normalize_api_format_alias(api_format).as_str() {
"aliyun:multimodal_embedding" => {
Some("/api/v1/services/embeddings/multimodal-embedding/multimodal-embedding")
}
_ => None,
}
}
+2
View File
@@ -1,6 +1,7 @@
pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
match crate::ai_serving::normalize_api_format_alias(api_format).as_str() {
"gemini:generate_content" => Some("gemini:generate_content"),
"gemini:interactions" => Some("gemini:interactions"),
"gemini:embedding" => Some("gemini:embedding"),
"gemini:video" => Some("gemini:video"),
"gemini:files" => Some("gemini:files"),
@@ -11,6 +12,7 @@ pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
pub(crate) fn local_path(api_format: &str) -> Option<&'static str> {
match crate::ai_serving::normalize_api_format_alias(api_format).as_str() {
"gemini" | "gemini:generate_content" => Some("/v1beta/models/{model}:{action}"),
"gemini:interactions" => Some("/v1/interactions"),
"gemini:embedding" => Some("/v1beta/models/{model}:{action}"),
"gemini:video" => Some("/v1beta/models/{model}:predictLongRunning"),
"gemini:files" => Some("/v1beta/files"),
+1
View File
@@ -1,3 +1,4 @@
mod aliyun;
mod claude;
mod doubao;
mod gemini;
+2
View File
@@ -5,6 +5,7 @@ pub(crate) fn normalized_signature(api_format: &str) -> Option<&'static str> {
"openai:rerank" => Some("openai:rerank"),
"openai:responses" => Some("openai:responses"),
"openai:responses:compact" => Some("openai:responses:compact"),
"openai:search" => Some("openai:search"),
"openai:image" => Some("openai:image"),
"openai:video" => Some("openai:video"),
_ => None,
@@ -18,6 +19,7 @@ pub(crate) fn local_path(api_format: &str) -> Option<&'static str> {
"openai:rerank" => Some("/v1/rerank"),
"openai:responses" => Some("/v1/responses"),
"openai:responses:compact" => Some("/v1/responses/compact"),
"openai:search" => Some("/v1/alpha/search"),
"openai:image" => Some("/v1/images/generations"),
"openai:video" => Some("/v1/videos"),
_ => None,

Some files were not shown because too many files have changed in this diff Show More