Compare commits

..
911 Commits
Author SHA1 Message Date
elky 06f5d3c8c0 fix(gateway): complete worker registration cleanup 2026-07-31 11:32:07 +08:00
elky 082407fa51 Merge PR #697: prevent duplicate worker registrations 2026-07-31 11:11:14 +08:00
fawney19 6688ee26db Merge pull request #702 from MMEXA/fix/reconcile-auth-channel-mismatch-formats
fix(gateway): 修复批量更新 API 格式时的认证通道状态冲突
2026-07-31 10:28:37 +08:00
elky beb003b7ad feat(models): add external catalog proxy selection 2026-07-31 09:32:25 +08:00
MMEXA 6ecfe0f0a1 fix(gateway): reconcile auth mismatch formats on key update 2026-07-30 22:14:08 +08:00
ZheFox 12057db476 Merge pull request #701 from zhefox/main
Persist OpenAI Responses continuation history across instances
2026-07-30 21:08:34 +08:00
ZheFox ff47d8d48a fix(gateway): route response history through ai seam 2026-07-30 20:34:41 +08:00
ZheFox ef5f36cc2b fix(ai): satisfy response history clippy checks 2026-07-30 20:13:36 +08:00
ZheFox 84022c4d48 Merge upstream/main into main 2026-07-30 19:40:39 +08:00
ZheFox 118f441029 feat(gateway): persist OpenAI Responses continuation history 2026-07-30 19:26:52 +08:00
elky 20399b004d Merge PR #700: fix admin pool batch update body buffering
Preserve main's failover and usage metadata fixes, restore default tunnel regression coverage, and satisfy current Clippy.
2026-07-30 17:56:37 +08:00
elky 050eb77508 fix(ai): harden responses replay and failure diagnostics 2026-07-30 17:19:54 +08:00
elky 1ab4f079c9 fix(gateway): restore failover and usage diagnostics 2026-07-30 09:12:11 +08:00
MMEXA 6c733f7590 fix(usage): preserve request diagnostics in event seeds 2026-07-30 06:44:59 +08:00
MMEXA d7d8db45ba test(gateway): align tunnel error fixture with failover policy 2026-07-30 06:44:59 +08:00
MMEXA 8cf9af79da fix(ci): remove redundant usage policy update 2026-07-30 05:45:38 +08:00
MMEXA e55793c765 fix(ci): satisfy gateway clippy on upstream baseline 2026-07-30 05:14:34 +08:00
MMEXA d8902ea612 fix(gateway): buffer admin pool batch update bodies 2026-07-30 05:14:34 +08:00
elky a04673a90d feat(gateway): harden failover and payload handling
Retry pre-response transport failures across candidates with an explicit stop policy, and propagate end-to-end timing into usage records and UI diagnostics.

Remove legacy body, import, cookie, PII, and tunnel replay caps while preserving optional operator-configured gateway limits.
2026-07-30 01:03:27 +08:00
ZheFox a97acc07fc Merge pull request #698 from zhefox/main
fix(ci): stabilize cross-platform workflow checks
2026-07-29 22:16:17 +08:00
zhefox f8000012f7 fix(ci): stabilize cross-platform workflow checks 2026-07-29 21:55:43 +08:00
worker-2 6080f8cc88 fix(gateway): stabilize worker task records
Key worker boot records by task so process restarts update the existing
row instead of registering another row for each gateway instance.

Closes #693
Confidence: high
Scope-risk: narrow
2026-07-29 17:27:51 +08:00
ZheFox 37df5b93b1 Merge pull request #696 from zhefox/main
Fix client metadata handling across formats
2026-07-28 18:12:59 +08:00
ZheFox e53abdaec2 Merge branch 'fawney19:main' into main 2026-07-28 18:12:28 +08:00
ZheFox 2db32ea97e Merge branch 'main' of https://github.com/zhefox/Aether 2026-07-28 17:40:46 +08:00
ZheFox 581897ee74 fix(formats): ignore responses client metadata across targets 2026-07-28 17:40:41 +08:00
ZheFox 9a88f966d8 Merge pull request #695 from zhefox/main
Enhance provider capabilities and clean up OAuth keys
2026-07-28 13:58:33 +08:00
ZheFox 9d9316e434 Merge branch 'fawney19:main' into main 2026-07-28 13:56:34 +08:00
ZheFox 1b697b1111 feat(providers): support FedRAMP Codex agent identity registration 2026-07-28 13:29:30 +08:00
ZheFox 3043982486 fix(providers): derive Codex primary quota label from window 2026-07-28 12:57:45 +08:00
ZheFox 0bf92ffffc feat(providers): advertise responses API agent capability 2026-07-28 12:02:03 +08:00
ZheFox f0f87b56a3 feat(providers): add credential-fenced OAuth key cleanup 2026-07-28 11:32:11 +08:00
elky 4148ab1931 fix(routing): harden routed pool scheduling 2026-07-27 22:06:28 +08:00
elky 550cc36760 feat(providers): expand OAuth account management
Add Claude Code manual and cookie authorization, including redacted batch tasks. Harden OAuth imports, duplicate replacement, provider dialogs, and related account-management tests.
2026-07-27 15:53:28 +08:00
elky 531cf11025 feat(gateway): harden provider request execution
Preserve exact request payloads and model client surface and API operation explicitly.

Add Anthropic compatibility profiles, bounded stream commitment, and scoped OAuth retry behavior across provider transports.
2026-07-27 09:36:31 +08:00
elky 79b70f7b5c fix(frontend): align sidebar collapse button 2026-07-26 15:07:51 +08:00
elky 10d369f59c feat(providers): add provider transfer limits 2026-07-26 15:06:56 +08:00
elky 2ef7ac79bc feat(frontend): add collapsible navigation sidebar
Persist the desktop sidebar state, provide accessible compact navigation tooltips, and cover the collapsed navigation markup with a focused component test.
2026-07-25 21:28:51 +08:00
elky 778cfb1a5c feat(data): complete portable SQL backend parity
Align MySQL and SQLite schemas, migrations, usage, stats, export, and backfill behavior with the shared data contracts. Extend gateway startup and maintenance support across all SQL drivers.
2026-07-25 21:28:21 +08:00
elky 764e9fd131 feat(frontend): improve provider detail drawer and pool actions 2026-07-25 11:16:59 +08:00
elky 387134ca87 fix(models): correct fast pricing and online sync 2026-07-24 01:45:38 +08:00
elky a0767d957c fix(frontend): synchronize pool account state 2026-07-23 16:20:18 +08:00
ZheFox b94ef91d07 Merge pull request #692 from zhefox/main
Sync global model prices and track online pricing sources
2026-07-23 16:02:17 +08:00
ZheFox e7910751d9 Merge branch 'fawney19:main' into main 2026-07-23 15:19:48 +08:00
ZheFox 1d2655432d feat(models): track online pricing sources and unsupported fields 2026-07-23 15:18:08 +08:00
ZheFox 323273ff30 feat(models): sync global model prices from online catalog 2026-07-23 13:29:28 +08:00
ZheFox fb2009c65b Merge pull request #691 from zhefox/main
fix(formats): ignore Responses client transport metadata
2026-07-23 12:18:40 +08:00
ZheFox e186cc6848 Merge branch 'main' of https://github.com/zhefox/Aether 2026-07-23 12:17:46 +08:00
ZheFox 615ac99ad7 fix(formats): ignore Responses client transport metadata 2026-07-23 12:16:44 +08:00
ZheFox ec36cfbf75 Merge pull request #690 from zhefox/main
fix(provider): classify deleted Codex agent runtime as invalid
2026-07-23 11:22:30 +08:00
ZheFox 7bf228a33c fix(provider): classify deleted Codex agent runtime as invalid 2026-07-23 11:21:55 +08:00
elky 3606290ac8 fix(provider): harden Agent Identity OAuth lifecycle 2026-07-23 09:33:00 +08:00
elky e49024d33b fix(frontend): shorten Agent Identity tab label 2026-07-22 20:26:49 +08:00
elky fdbc2607ec feat(provider): add dedicated Codex Agent Identity flow 2026-07-22 20:19:29 +08:00
elky c7cc8fd7db test(provider): simplify agent identity assertions 2026-07-22 14:19:58 +08:00
elky 07efcb5146 fix(data): repair legacy active flag synchronization 2026-07-22 14:19:34 +08:00
elky 856605defa fix(model-directives): harden suffix configuration 2026-07-22 14:19:09 +08:00
elky 713010fa0a fix(gateway): restore auth role refresh and Rust checks
Refresh the resolved user role without bypassing owner group and key policies. Resolve Rust 1.95 Clippy failures and make the pending persistence bound test scheduler-independent.
2026-07-22 11:25:24 +08:00
ZheFox cd2fbeeead Merge pull request #689 from AAEE86/feat/agent-identity-support
feat(codex): enroll agent identity from session token
2026-07-22 10:32:06 +08:00
AAEE86 a4350a482a feat(codex): enroll agent identity from session token 2026-07-22 10:21:11 +08:00
ZheFox c825375367 Merge pull request #688 from AAEE86/feat/agent-identity-support
feat(codex): support agent identity accounts
2026-07-22 09:20:11 +08:00
elky fc92c4f431 perf(gateway): scale request hot paths for 20k streams
Shard and singleflight hot-path caches, batch and prioritize candidate and usage lifecycle persistence, and extend database and pressure-test instrumentation for 20k concurrent streams.
2026-07-22 02:11:08 +08:00
AAEE86 b61c590bdb feat(codex): support agent identity accounts 2026-07-21 20:58:49 +08:00
ZheFox 7756c0913f Merge pull request #685 from zhefox/main
fix(gateway): apply group policy to admin-owned keys
2026-07-20 15:52:06 +08:00
ZheFox c34ec7c1ee fix(gateway): apply group policy to admin-owned keys 2026-07-20 15:51:01 +08:00
elky f8778c4a23 feat(gateway): configure cyber policy failover 2026-07-19 23:27:19 +08:00
elky e0dbb233f7 fix(frontend): avoid misleading cache TTL fallback label 2026-07-19 22:21:46 +08:00
elky 9725f9abae fix(frontend): clarify processing tier pricing 2026-07-19 22:00:12 +08:00
elky 5d575f1590 test(stats): treat bulk API key snapshots as authoritative 2026-07-19 19:07:04 +08:00
elky d562c594c3 fix(frontend): preserve compact scope and detail badge 2026-07-19 16:42:32 +08:00
MMEXA ce226a3010 Merge 0c3f51bcec into 644ae9c1bf 2026-07-19 16:12:37 +08:00
elky 644ae9c1bf feat(pool): add table-driven account batch actions 2026-07-19 16:09:20 +08:00
elky 95053f9502 Merge PR #672: 支持账号批量配置与可用模型管理 2026-07-18 22:06:32 +08:00
elky 8fbda84acb fix(data): preserve API key history end to end 2026-07-18 21:58:21 +08:00
elky 03b7d573e0 Merge PR #683: decouple API key historical identity 2026-07-18 21:20:35 +08:00
MMEXA 0c3f51bcec merge(main): 解决 usage 模型展示契约冲突 2026-07-18 19:26:14 +08:00
elky e3d97b573b fix(usage): align fast-tier pricing and model metadata 2026-07-18 16:57:04 +08:00
MMEXA ac3796af84 fix(gateway): 恢复响应边界并统一格式入口 2026-07-18 06:45:37 +08:00
MMEXA f9c343eb07 fix(gateway): 适配 Rust 1.95 整除检查 2026-07-18 05:55:06 +08:00
MMEXA e31df5989a merge(main): 解决 usage 展示与生命周期同步冲突 2026-07-18 05:38:38 +08:00
MMEXA 98fbf029fc fix(data): 解耦 API Key 历史统计身份 2026-07-18 04:42:45 +08:00
MMEXA 4d9a648202 test(gateway): 统一流错误测试的格式层入口 2026-07-18 03:37:31 +08:00
MMEXA 405ca3e66a fix(ci): 恢复非流式错误体边界并适配新版 Clippy 2026-07-18 03:26:53 +08:00
MMEXA 0355c28683 fix(data): 解耦候选记录的 API Key 历史身份 2026-07-18 02:40:34 +08:00
fawney19 6c33b8d8fb Merge pull request #682 from MMEXA/codex/codex-prompt-cache-identity-20260717
fix(codex): 统一通用缓存键与原生会话身份
2026-07-18 00:33:20 +08:00
elky a6c6f14b09 style(frontend): align pool cycle stats values 2026-07-18 00:13:35 +08:00
elky e558f55cd9 style(frontend): refine badges and cycle stats 2026-07-18 00:05:22 +08:00
elky 88a057b8d9 fix(usage): force fast badge background transparent 2026-07-17 22:56:54 +08:00
elky ed27d404ac style(usage): make fast badge background transparent 2026-07-17 22:52:47 +08:00
elky 5dda34c66e style(usage): give fast tier an amber accent 2026-07-17 22:36:55 +08:00
elky 373ebf26d6 fix(pricing): default zero tier ratios to one 2026-07-17 21:33:21 +08:00
elky f65ed2795c fix(codex): support dynamic quota windows 2026-07-17 20:18:04 +08:00
elky 664c063a06 feat(usage): enrich audit metadata and detail views 2026-07-17 19:20:16 +08:00
MMEXA 75795c6fbc test(codex): 对齐 Compact 确定性缓存身份 2026-07-17 08:49:35 +08:00
MMEXA 5b332da7d7 fix(codex): 补齐缓存身份终态请求头 2026-07-17 08:11:16 +08:00
MMEXA d9796d502b fix(codex): 统一通用缓存键与原生会话身份 2026-07-17 06:13:09 +08:00
MMEXA 3b0d87b0fd Merge remote-tracking branch 'origin/main' into codex/pool-key-bulk-management-20260714 2026-07-17 00:17:02 +08:00
MMEXA 3c348dff3a Merge remote-tracking branch 'origin/main' into codex/usage-pending-reasoning-reset-expiry-20260712
# Conflicts:
#	frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts
2026-07-17 00:16:58 +08:00
MMEXA ec1783a35c Merge remote-tracking branch 'origin/main' into codex/pool-key-bulk-management-20260714
# Conflicts:
#	apps/aether-gateway/src/handlers/admin/request/provider/tasks.rs
#	frontend/src/api/endpoints/pool.ts
2026-07-16 23:43:04 +08:00
MMEXA 427030c5de Merge remote-tracking branch 'origin/main' into codex/usage-pending-reasoning-reset-expiry-20260712
# Conflicts:
#	crates/aether-ai-formats/src/formats/openai/responses/mod.rs
#	crates/aether-usage/runtime/src/runtime.rs
#	frontend/src/features/usage/components/UsageRecordsTable.vue
#	frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts
2026-07-16 23:41:58 +08:00
elky 0be380243b feat(pricing): support processing tier multipliers 2026-07-16 23:30:42 +08:00
fawney19 312583f055 Merge pull request #680 from Kayphoon/codex/s3-backup-user-agent
feat(admin): configure S3 backup User-Agent
2026-07-16 23:30:30 +08:00
fawney19 33f49ea9b0 Merge pull request #678 from AAEE86/fix
fix: map Developer role to "system" in OpenAI Chat Completions output
2026-07-16 23:29:55 +08:00
fawney19 470cef17cf Merge pull request #676 from MMEXA/codex/sync-capture-envelope-finalize-20260716
修复同步 finalize 的 Responses 流聚合与转换
2026-07-16 23:29:38 +08:00
ZheFox 3f5f65eb9a Merge pull request #681 from zhefox/main
Codex 重置功能和显示缓存修复以及批量key的导入和管理功能
2026-07-16 19:48:33 +08:00
ZheFox 6664c2dbb8 feat(pool): 支持批量导入 Key 和选择性更新设置 2026-07-16 19:31:07 +08:00
ZheFox 0099167a6d fix(codex): 避免重置机会缺失触发配额刷新 2026-07-16 19:13:09 +08:00
ZheFox f009fb73c3 缓存问题修复 2026-07-16 18:55:28 +08:00
ZheFox 715a5ed626 修复重置次数缓存问题 2026-07-16 18:09:10 +08:00
ZheFox 5cf38d1b35 Codex 重置功能和显示修复 2026-07-16 17:23:41 +08:00
Kayphoon 6b707f29a2 feat(admin): configure S3 backup User-Agent 2026-07-16 08:56:56 +00:00
elky 9ea84f9748 fix(frontend): show service tier transitions 2026-07-16 16:38:30 +08:00
elky c32d043afb fix(frontend): preserve fetched model preset pricing 2026-07-16 16:38:30 +08:00
elky 8fe4d24408 fix(usage): canonicalize cached token totals 2026-07-16 16:38:30 +08:00
elky e369e4aab1 fix(formats): preserve chat-backed Responses metadata 2026-07-16 16:38:30 +08:00
ZheFox 7dc919e8e3 Merge pull request #679 from zhefox/main
fix(frontend): 优化移动端弹窗并完善提供商配额刷新
2026-07-16 15:48:49 +08:00
ZheFox 1333efdad5 fix(frontend): 优化移动端弹窗并完善提供商配额刷新 2026-07-16 15:21:11 +08:00
AAEE86 cd8de1aa13 fix: map Developer role to "system" in OpenAI Chat Completions output 2026-07-16 14:29:08 +08:00
elky d6215d9dec ci(tunnel): reduce artifact retention 2026-07-16 13:12:30 +08:00
elky 9a47267545 fix(usage): bound terminal event persistence
Add end-to-end terminal admission, bounded database fallback, and observable overload handling. Preserve first-byte lifecycle state across asynchronous runtime and frontend updates.
2026-07-16 13:12:30 +08:00
MMEXA 7851503fbc fix(finalize): 严格聚合并投影同步 Responses 流 2026-07-16 12:50:00 +08:00
ZheFox c6d373e6aa Merge pull request #677 from zhefox/main
fix(codex): 移除 Responses Lite 请求中的 context_management
2026-07-16 12:28:34 +08:00
ZheFox 71fcb9c168 fix(codex): 服务端压缩使用标准 Responses 合约 2026-07-16 12:02:33 +08:00
ZheFox 3976652942 fix(codex): 移除 Responses Lite 请求中的 context_management 2026-07-16 11:25:11 +08:00
MMEXA 7b56546e21 fix(finalize): 聚合同步流捕获包装 2026-07-16 10:24:32 +08:00
fawney19 85854e4476 Merge pull request #675 from fawney19/fix/pr-669-tail
feat(codex): complete PR #669 protocol follow-up
2026-07-16 08:54:29 +08:00
elky b50242ab9f fix(test): handle absent empty testkit bin directory 2026-07-16 01:29:51 +08:00
MMEXA 20b27a13b2 feat(codex): 按操作语义路由 Responses V2 压缩
(cherry picked from commit 2fc604e047)
2026-07-16 00:34:42 +08:00
MMEXA 598b2fb374 fix(auth): 授权 Responses Compact 伴随端点
(cherry picked from commit e8afa03e45)
2026-07-16 00:32:14 +08:00
MMEXA ff7988430d fix(openai): encode tool errors in Responses output
(cherry picked from commit f127b67e73)
2026-07-16 00:31:23 +08:00
MMEXA 25da99fac2 fix(codex): preserve reset consume request body
(cherry picked from commit fc2dfb82d2)
2026-07-16 00:27:28 +08:00
elky 8616fe6ee2 refactor(workspace): enforce layered crate boundaries 2026-07-15 23:47:19 +08:00
MMEXA 9b8724453b test(pool): 使用正式 Gemini API 格式 2026-07-14 08:39:41 +08:00
MMEXA 01e104d86a fix(pool): 对齐账号批量配置语义 2026-07-14 08:07:28 +08:00
MMEXA 0acd1de29c fix(gateway): 保持密钥更新模块显式所有权 2026-07-14 05:19:04 +08:00
MMEXA a25fab371a feat(pool): add bulk key configuration management 2026-07-14 04:57:05 +08:00
MMEXA 93e2f95c47 fix(frontend): 按端点能力约束会话压缩映射 2026-07-14 02:05:10 +08:00
MMEXA f10d631a9c feat(frontend): 澄清模型映射适用范围 2026-07-14 00:30:39 +08:00
MMEXA cfc4894dab fix(usage): 保留最新进行态生命周期事件 2026-07-14 00:30:24 +08:00
MMEXA 3f86fdd6bc feat(usage): 展示压缩操作与进行态请求语义 2026-07-13 22:03:44 +08:00
MMEXA b09d1f1c33 fix(usage): expose pending reasoning and exact reset expiry 2026-07-13 22:03:44 +08:00
MMEXA 2fc604e047 feat(codex): 按操作语义路由 Responses V2 压缩 2026-07-13 22:03:34 +08:00
MMEXA e8afa03e45 fix(auth): 授权 Responses Compact 伴随端点 2026-07-13 06:07:54 +08:00
MMEXA fc2dfb82d2 fix(codex): preserve reset consume request body 2026-07-12 23:04:33 +08:00
elky a728c090a9 fix(gateway): scope concurrency helper to tests 2026-07-12 22:36:48 +08:00
elky e58621a735 Merge PR #669: align GPT-5.6 and Codex request protocols 2026-07-12 21:50:20 +08:00
MMEXA b1be370b2e fix(gateway): scope concurrency test helper to tests 2026-07-12 21:08:30 +08:00
MMEXA cf0d957ac7 Merge f127b67e73 into 7f61bb43c7 2026-07-12 20:34:39 +08:00
MMEXA f127b67e73 fix(openai): encode tool errors in Responses output 2026-07-12 20:34:31 +08:00
elky 7f61bb43c7 feat(security): harden gateway request and runtime controls 2026-07-12 14:10:54 +08:00
MMEXA 25c49dd804 fix(data): keep terminal usage state monotonic 2026-07-12 05:45:37 +08:00
MMEXA 72222d935c test(gateway): use valid tunnel relay envelopes 2026-07-12 05:45:32 +08:00
MMEXA 063d517306 test(gateway): compare timeout response numerically 2026-07-12 04:43:13 +08:00
MMEXA 02495ce28e fix(admin): preserve inactive endpoint key counts 2026-07-12 04:19:06 +08:00
MMEXA 63936aa110 fix(gateway): route format rules through serving facade 2026-07-12 03:56:53 +08:00
MMEXA 8d4d42a887 fix(auth): resolve group policy before key intersection 2026-07-12 03:35:42 +08:00
MMEXA 2316df5c9a feat(codex): align Search and execution protocol 2026-07-12 03:04:15 +08:00
MMEXA 59d37ae1dd fix(frontend): import structured models.dev pricing 2026-07-11 18:12:55 +08:00
MMEXA 3014fd50c6 fix(billing): preserve effective cache and tier facts 2026-07-11 18:12:55 +08:00
MMEXA 14c4e3a04e fix(codex): enforce provider request identity 2026-07-11 18:09:14 +08:00
MMEXA 8f1070a451 feat(frontend): expose processing tier pricing 2026-07-11 12:27:09 +08:00
MMEXA 0b30cc6b0f feat(openai): unify tier authorization and settlement 2026-07-11 12:27:05 +08:00
MMEXA b2f596b8f0 fix(gateway): route Codex header through serving facade 2026-07-11 09:43:12 +08:00
MMEXA 01a96fed74 fix(codex): simplify summary normalization 2026-07-11 09:20:07 +08:00
MMEXA 46a903aada fix(codex): align current reasoning request semantics 2026-07-11 09:15:08 +08:00
MMEXA dfa121dd5b feat(openai): align GPT-5.6 and Codex request contracts 2026-07-11 07:40:12 +08:00
elky bc1da3bf3f feat(security): harden client IP and admin controls 2026-07-10 15:13:12 +08:00
elky 6e0dc3b59e feat(frontend): refine global model pricing dialog 2026-07-10 15:13:12 +08:00
elky 4bf5d4c044 Fix cache token accounting and tiered pricing 2026-07-10 15:13:12 +08:00
fawney19 736fc76345 Merge pull request #668 from MMEXA/codex/antigravity-empty-output-retry-20260709
修复 Gemini 空输出按候选重试处理
2026-07-10 09:16:56 +08:00
MMEXA b6b2ca38f4 触发 CI 重跑 2026-07-10 00:41:22 +08:00
MMEXA f07eb25cfc 修复 usage 详情 body 引用解包 2026-07-10 00:23:31 +08:00
MMEXA d2ea437c1c 修复 Gemini 空输出按候选重试处理 2026-07-09 23:10:30 +08:00
fawney19 7bc7d0f8d8 Merge pull request #666 from xixiknow/main
Fix provider key response time counter overflow
2026-07-09 18:04:11 +08:00
fawney19 14cf639aba Merge pull request #667 from MMEXA/codex/antigravity-v1internal-query-20260709
修复 Antigravity v1internal 查询参数透传
2026-07-09 17:50:38 +08:00
yangrs 55cdab592c Remove redundant response time conversion 2026-07-09 16:26:09 +08:00
MMEXA ee0ec18283 修复 Antigravity v1internal 查询参数透传 2026-07-09 16:03:29 +08:00
Start f31c9e03e2 Merge branch 'fawney19:main' into main 2026-07-09 15:11:44 +08:00
fawney19 e50db10439 Merge pull request #663 from MMEXA/codex/gemini-interactions-antigravity-20260705
完善 Gemini Interactions 与 Antigravity 全链路兼容
2026-07-09 14:55:51 +08:00
yangrs 192dc6c20d Fix provider key response time overflow 2026-07-09 14:40:48 +08:00
elky 5e1d14f19b Fix timeline duration display from latency 2026-07-09 11:46:45 +08:00
MMEXA b8b89d21b7 fix: 同步提交本地 sync 错误上报 2026-07-08 23:49:18 +08:00
MMEXA 5eddf4f9ee 细化 Antigravity 测试模型项目元数据补全 2026-07-08 22:45:30 +08:00
MMEXA c7186e1720 完善 Antigravity 配额展示与 CI 断言 2026-07-08 22:34:29 +08:00
MMEXA 4866509938 移除 Antigravity 未知重置时间噪音 2026-07-08 22:34:29 +08:00
MMEXA 2122660a5c 对齐原生 Antigravity 控制面与显示模型 2026-07-08 22:34:29 +08:00
MMEXA 5c68ab896a 恢复历史 backfill 兼容 live 账本 2026-07-08 22:34:29 +08:00
MMEXA f9c8ec41f4 完善 Antigravity 与 Gemini 跨格式兼容 2026-07-08 22:34:29 +08:00
MMEXA b1ed6b24b0 触发 CI 复跑 2026-07-08 22:34:29 +08:00
MMEXA c17c78ad4b 修正 Antigravity Gemini 3.5 Flash 档位展示 2026-07-08 22:34:29 +08:00
MMEXA 9ec48ab6b9 优化 Antigravity 配额展示顺序 2026-07-08 22:34:29 +08:00
MMEXA accd250226 修正 Antigravity 配额模型标签 2026-07-08 22:34:29 +08:00
MMEXA 80a6579766 支持 Gemini Interactions 与 Antigravity 配额精细化 2026-07-08 22:34:29 +08:00
fawney19 1ca83ca3fb Merge pull request #664 from MMEXA/codex/wallet-auth-cache-delay-20260706
修复钱包余额变更后的鉴权缓存延迟
2026-07-07 01:59:44 +08:00
fawney19 a931da0764 Merge pull request #662 from MMEXA/codex/reset-credit-20260704
增加 Codex 重置次数功能
2026-07-07 01:58:44 +08:00
fawney19 a61374c595 Merge pull request #661 from MMEXA/codex/frontend-debug-20260704
修复前端调试与基础交互问题
2026-07-07 01:57:41 +08:00
MMEXA c3136126e5 修复钱包余额变更后的鉴权缓存延迟 2026-07-06 06:15:31 +08:00
MMEXA b23d299533 重跑 Codex 重置次数 CI 2026-07-05 01:37:31 +08:00
MMEXA b03aae18c3 修复 Codex 重置次数 CI 检查 2026-07-04 15:47:01 +08:00
MMEXA 99b6fe468f 简化 Codex 重置机会展示标签 2026-07-04 15:16:38 +08:00
MMEXA ef77ec04ca 增加 Codex 重置次数功能 2026-07-04 06:10:53 +08:00
MMEXA 242081433e 修复前端调试与基础交互问题 2026-07-04 05:24:40 +08:00
ZheFox b86d4e1f0c Merge pull request #660 from zhefox/main
refactor(frontend): unify mobile menu background styles
2026-07-03 13:17:59 +08:00
ZheFox a151f37d63 refactor(frontend): unify mobile menu background styles 2026-07-03 13:17:18 +08:00
ZheFox 42f7907740 Merge pull request #659 from zhefox/main
refactor(frontend): improve mobile overflow handling
2026-07-03 12:54:01 +08:00
ZheFox e72e25c59c refactor(frontend): improve mobile overflow handling 2026-07-03 12:53:22 +08:00
ZheFox 1b0440481b Merge pull request #658 from zhefox/main
修复管理端额度显示、节点表格显示与移动端滚动问题
2026-07-03 12:49:15 +08:00
ZheFox 26d85681f0 refactor(frontend): improve mobile overflow and proxy node table 2026-07-03 12:25:53 +08:00
ZheFox 1dcee77055 refactor(frontend): improve dialog and mobile overflow handling 2026-07-03 11:26:40 +08:00
elky 1ac16005f9 Stabilize usage worker autoscale tests 2026-07-02 17:28:25 +08:00
elky 2f1cdb6a0b Record exhausted usage failures synchronously 2026-07-02 16:08:04 +08:00
elky ac93851b2a Stabilize Gateway h2c transport test 2026-07-02 14:02:26 +08:00
elky 400b3125a4 Preserve terminal request candidate state 2026-07-02 01:40:57 +08:00
elky 2e5ff32e1a perf(frontend): 收敛导航预取并去重首屏请求
- 导航预取仅保留 pointerdown 触发,移除 mouseenter/focus,避免鼠标划过误触发
- 后台预取只做组件懒加载,不再预取各页业务数据,减少首屏资源争抢
- 版本状态检查增加 sessionStorage 缓存(正常 20 分钟 / 错误 5 分钟 TTL)
- fetchModules、必读公告拉取增加请求去重,避免并发重复请求
- 更新检查改用可清理的定时器,组件卸载时清理
- UsageRecordsTable 搜索防抖改为自定义实现,卸载时取消挂起 emit 并补充测试
2026-07-01 20:42:52 +08:00
elky a0f7074e59 chore: disable Redis persistence by default, document triage and policy 2026-07-01 14:15:16 +08:00
elky 7c32be46ca Mark sync usage active earlier 2026-07-01 02:21:20 +08:00
Entropy.Xu 6ed2f9bd0a fix: apply actual billing cost to wallet settlement 2026-07-01 01:12:40 +08:00
elky 778b106023 test: stabilize gateway nextest timing 2026-06-30 18:42:57 +08:00
elky f179ee72f9 chore: update gateway pressure observability 2026-06-30 17:01:39 +08:00
elky 974def5fef refactor(frontend): extract provider key identity block 2026-06-30 17:01:39 +08:00
elky e5351b7d9d refactor(frontend): extract provider key actions 2026-06-30 17:01:39 +08:00
elky ed83184d55 refactor(frontend): extract provider quota display components 2026-06-30 17:01:39 +08:00
elky 15b6606c82 refactor(frontend): extract pool key display panels 2026-06-30 17:01:39 +08:00
elky d7411a3104 refactor(frontend): extract pool header and theme toggle 2026-06-30 17:01:39 +08:00
elky 9f138d09e6 refactor(frontend): modularize i18n architecture 2026-06-30 17:01:39 +08:00
ZheFox bf29129a4b Merge pull request #654 from zhefox/main
Cancel upstream streams on client disconnect and void cancelled usage billing
2026-06-29 01:14:11 +08:00
zhefox f6293b6812 fix(usage): void cancelled usage and cancel dropped streams 2026-06-29 00:30:41 +08:00
elky 7e9424008f Add usage queue worker autoscaling 2026-06-26 14:02:57 +08:00
elky 6c5e70ccb1 fix monitoring error totals and counter health 2026-06-26 10:48:45 +08:00
elky 063834e95b Split admin operations dashboard route 2026-06-26 01:48:37 +08:00
elky c76d6b6396 Add admin operations dashboard and usage state fixes 2026-06-26 01:32:45 +08:00
elky 6f00e9fc67 Improve gateway transport and usage runtime 2026-06-25 22:36:27 +08:00
elky d336d1a7fa Improve gateway scheduling and runtime admission 2026-06-24 01:53:45 +08:00
ZheFox cf0af8fa1e Merge pull request #652 from zhefox/main
fix(usage): preserve token counts in body redaction
2026-06-23 14:40:59 +08:00
zhefox fd220b6c42 fix(usage): preserve token counts in body redaction 2026-06-23 14:38:39 +08:00
zhefox 3472bb75e7 ci: combine gateway clippy and nextest jobs 2026-06-23 14:06:43 +08:00
zhefox ba65c96c74 Merge branch 'main' of https://github.com/zhefox/Aether 2026-06-23 13:36:00 +08:00
zhefox c54b214657 ci: shard gateway tests and disable debug info in rust ci 2026-06-23 13:35:56 +08:00
ZheFox 4fcc17114f Merge pull request #651 from zhefox/main
fix(ai-formats): accept Claude context_management in responses conversion
2026-06-23 10:43:26 +08:00
zhefox deb5f55786 fix(ai-formats): clean up cross-format safety rules for Gemini requests 2026-06-23 10:30:13 +08:00
zhefox 1836c2b652 fix(ai-formats): accept Claude context_management in responses conversion 2026-06-23 10:03:55 +08:00
elky 5b7805181b perf: queue request candidate persistence 2026-06-22 02:49:17 +08:00
elky f75894acbb perf: reduce gateway db pressure under load 2026-06-22 00:08:48 +08:00
elky 541cc197c4 fix: preserve in-memory user export fallback 2026-06-22 00:08:48 +08:00
fawney19 363d1aba9a Merge pull request #615 from AAEE86/main
feat: 健康监控仪表盘与关联下钻优化,完善使用记录展示
2026-06-21 12:48:13 +08:00
fawney19 eb2cf662b7 Merge pull request #650 from stabey/pr/claude-system-responses-20260620
fix(ai-formats): preserve Claude in-message system guidance in Responses
2026-06-21 12:47:26 +08:00
elky 900f8a7163 fix(pool): allow zero cooldown settings 2026-06-21 12:20:08 +08:00
elky 61bdd304b7 Handle inactive PAT owner as invalid OAuth token 2026-06-21 11:39:20 +08:00
elky 279735ae7f Auto-size SQL pool defaults 2026-06-21 11:15:40 +08:00
elky cc2830f6ec Merge branch 'review/pr-639' 2026-06-21 10:48:49 +08:00
elky 8dbd730568 fix: respect imported oauth authorization headers 2026-06-21 02:27:06 +08:00
stabey bb6aa03485 fix(ai-formats): strip Claude billing headers from preserved guidance 2026-06-21 00:22:57 +08:00
stabey 6a22488698 fix(ai-formats): preserve Claude in-message system guidance in responses 2026-06-21 00:07:57 +08:00
elky f1c30439ff fix: preserve provider auth metadata 2026-06-20 22:11:42 +08:00
fawney19 1123095bb7 Merge pull request #624 from MMEXA/codex/fix-antigravity-oauth-quota
修复 Antigravity OAuth 导入后配额复检缺 project
2026-06-19 23:11:38 +08:00
MMEXA 938f11981d fix(ai-serving): route Antigravity auth enum through facade 2026-06-19 22:25:52 +08:00
MMEXA 6c4e730e60 修复 Antigravity OAuth 配额复检缺 project 2026-06-19 22:21:22 +08:00
elky 16584067d7 Add route-backed routing profile views 2026-06-18 02:06:34 +08:00
fawney19 6de0fe75a4 Merge pull request #641 from Kayphoon/codex/usage-cleanup-break-condition
fix(usage): align cleanup loop break conditions with candidate row count
2026-06-17 11:04:55 +08:00
fawney19 34f0913ed0 Merge pull request #645 from zhefox/main
修复 OpenAI Chat/Responses/Messages 转换兼容性并透传 Codex cyber_policy 错误
2026-06-17 11:03:29 +08:00
zhefox 5b305c64e1 fix(ai-formats): omit request tool call ids in OpenAI Responses input 2026-06-17 09:15:38 +08:00
zhefox 8ad97761e8 fix(ai-formats): preserve OpenAI Responses tool call item ids 2026-06-17 08:29:25 +08:00
zhefox 0f92ef664d fix(ai-formats): support OpenAI Responses custom tool/raw passthrough 2026-06-17 04:24:56 +08:00
zhefox 16a4fd3687 Merge branch 'main' of https://github.com/zhefox/Aether 2026-06-17 04:07:53 +08:00
zhefox 3a3fcbe46a fix(ai-formats): preserve OpenAI tool call item ids 2026-06-17 04:05:09 +08:00
zhefox 18d8ea2052 fix(ai-formats): preserve OpenAI tool call item ids 2026-06-17 04:03:40 +08:00
zhefox 6ab08f4014 fix(ai-formats): preserve Claude raw blocks, reasoning tokens, and test stack safety 2026-06-17 03:45:23 +08:00
zhefox 628a3a0d8d fix(ai-formats): support cyber policy failover and custom tool/audio passthrough 2026-06-17 02:33:14 +08:00
elky f52628e00b Handle OpenAI Responses keepalive stream events 2026-06-16 22:52:06 +08:00
zhefox f9d97ececb fix(ai-formats): ignore OpenAI Responses metadata events 2026-06-16 22:34:38 +08:00
fawney19 803e555022 Merge pull request #635 from zhefox/main
fix(gateway): 支持 OpenAI 图片编辑端点请求
2026-06-16 22:31:52 +08:00
zhefox b1bd727978 将 JSON 提示注入为 developer 输入 2026-06-16 21:40:15 +08:00
zhefox c2748dc868 忽略 OpenAI Responses keepalive 事件 2026-06-16 20:00:46 +08:00
ZheFox 302620cb94 Merge branch 'fawney19:main' into main 2026-06-16 12:17:28 +08:00
AAEE86 c255f29e98 Merge remote-tracking branch 'upstream/main' 2026-06-16 10:53:47 +08:00
Kayphoon 6d1b818414 fix(usage): align cleanup loop break conditions with candidate row count
The cleanup loop break condition used rows_affected() from the UPDATE
statement, but for rows that only had blob/audit refs (no inline
compressed body data), the UPDATE reported 0 affected rows. This caused
the loop to exit after the first batch, skipping the majority of
candidates.

Change the break condition in all 4 cleanup functions from:
  if cleaned == 0 || cleaned < batch_size
to:
  if rows.len() < batch_size

This ensures the loop continues as long as SELECT returns a full batch,
regardless of how many rows the UPDATE actually modified.

Affected functions:
- cleanup_usage_raw_body_fields
- cleanup_usage_compressed_body_fields
- cleanup_usage_header_fields
- cleanup_usage_stale_body_fields
2026-06-16 04:19:58 +08:00
elky 669636d3e4 Harden PII redaction format conversion 2026-06-14 20:36:57 +08:00
elky 68038c182b Distinguish expired OAuth token status 2026-06-12 19:43:59 +08:00
elky 308cc88ef7 Fix provider deletion cleanup 2026-06-12 16:25:11 +08:00
elky 30b545785f feat: improve failover rules and request timeline 2026-06-11 00:49:29 +08:00
elky 31fade82f6 Preserve OpenAI encrypted reasoning blocks 2026-06-10 20:02:03 +08:00
elky 0246ba93dd fix(transport): preserve safe accept encoding 2026-06-10 19:58:57 +08:00
elky ff7ec8575c fix(ai-formats): ignore null stream errors 2026-06-10 18:44:46 +08:00
elky aa58cb4a05 build: speed up release image linking 2026-06-10 18:37:23 +08:00
elky e9b4efc2d4 fix(ai-serving): preserve explicit request encoding 2026-06-10 18:16:55 +08:00
elky ea76f7bb0b Support OpenAI Responses builtin tool stream items 2026-06-10 14:49:13 +08:00
elky 8edcbdcb29 feat(usage): expose request timing details 2026-06-10 09:16:15 +08:00
ndllz 5249660e07 fix: respect oauth module disabled state 2026-06-09 18:13:07 +08:00
ndllz 84b99a641a fix: speed up usage activity heatmap render 2026-06-09 16:52:58 +08:00
AAEE86 4824e4a487 fix(frontend): 移除账号导入重复处理中提示 2026-06-09 16:40:45 +08:00
zhefox ba723ebe48 fix(usage): always use truncated body placeholder when limit exceeded 2026-06-09 09:59:10 +08:00
zhefox 04ba8cbe9e fix(gateway): support openai image accept negotiation 2026-06-08 20:40:30 +08:00
elky 84f41dae77 feat(format): audit same-format compatibility rewrites 2026-06-08 16:12:37 +08:00
zhefox 82040bfc21 fix(gateway): support OpenAI image edit requests 2026-06-08 13:22:08 +08:00
elky 6155ffefcc fix(format): avoid false cache-control conversion blocks 2026-06-08 00:52:48 +08:00
elky bf4279a590 Merge remote-tracking branch 'origin/main' into dev
# Conflicts:
#	crates/aether-ai-formats/src/formats/openai/chat/stream.rs
#	crates/aether-ai-formats/src/formats/openai/responses/response.rs
#	crates/aether-ai-formats/src/formats/shared/sync_products.rs
2026-06-08 00:17:30 +08:00
elky 77759fac54 feat: 新增提供商批量处理功能 2026-06-07 22:57:02 +08:00
elky 63a2fd4dcf Merge origin/main into dev 2026-06-06 03:11:38 +08:00
elky 7a19891c60 fix: distinguish unaudited conversion fields 2026-06-06 00:35:27 +08:00
AAEE86 85573d7980 Merge remote-tracking branch 'upstream/main' 2026-06-05 08:28:43 +08:00
zhefox ebd59246a8 fix(test): assert responses timestamps and output text in finalize tests 2026-06-04 17:35:10 +08:00
zhefox fd27f55fe5 fix(provider): normalize OpenAI Responses modern fields and stream events 2026-06-04 15:45:37 +08:00
fawney19 69b8b96fb8 Merge pull request #625 from stabey/pr/responses-call-items-cache-control-20260604
fix: 剥离 Codex cache_control 并完善 Responses 工具调用展示
2026-06-04 13:59:56 +08:00
fawney19 19d1d36043 Merge pull request #620 from zhefox/main
fix(provider): 修复 Chat reasoning_effort 值域与 Responses 扩展透传
2026-06-04 13:59:42 +08:00
stabey 9f19ca5754 fix(usage): keep streamed call args in responses completion
The response.completed fallback rebuilt every call item with responsesCallInput(), which returns '{}' for a function_call lacking arguments. Since '{}' is truthy, ensureToolCall overwrote arguments already collected from streamed delta events. Guard the completed branch with responsesCallHasInput (matching the output_item.done branch) so empty/default inputs no longer clobber streamed args, and align its dedupe key with the streaming phase to avoid duplicate tool-call rendering when an item has no id. Drop the now-dead '工具调用' fallbacks since responsesCallName never returns empty.
2026-06-04 13:30:20 +08:00
stabey 2de2a792f6 fix(codex): strip cache_control before responses upstream 2026-06-04 12:01:36 +08:00
stabey ada690624b fix(usage): render responses call items in conversation view 2026-06-04 11:06:36 +08:00
elky 465476985b fix: preserve provider schema drift safely 2026-06-03 22:29:24 +08:00
elky da5624c98e chore: add format field coverage generator 2026-06-03 21:44:33 +08:00
elky b2f68bbaf7 feat: enforce full format field coverage audit 2026-06-03 21:18:49 +08:00
elky 7507af5829 feat: audit strict format conversion contracts 2026-06-03 20:27:15 +08:00
zhefox 5e39801bba fix(provider): clamp reasoning effort and filter chat extensions 2026-06-03 10:52:26 +08:00
elky 5ac153a0bb Fix gateway nextest stack limit 2026-06-03 01:32:25 +08:00
elky c7a5155ce4 Fix sync CLI test stack overflow 2026-06-03 01:12:40 +08:00
elky 869c3d3037 Fix finalize local test stack overflow 2026-06-03 00:48:36 +08:00
elky ef6a11c146 fix(gateway): preserve heartbeat no-path fallback 2026-06-03 00:25:01 +08:00
elky 21432911de Merge remote-tracking branch 'origin/pr/605' 2026-06-03 00:16:22 +08:00
elky eb98340924 Merge remote-tracking branch 'origin/pr/604' 2026-06-02 23:34:59 +08:00
elky 746af0d93e Fix Kiro cache usage reporting 2026-06-02 23:06:29 +08:00
elky 08ac9c5c58 Merge remote-tracking branch 'origin/pr/614' 2026-06-02 22:10:41 +08:00
elky bce3bf2b6e Merge remote-tracking branch 'origin/pr/593' 2026-06-02 21:40:59 +08:00
elky 4ec9ca61cf Merge remote-tracking branch 'origin/pr/613'
# Conflicts:
#	apps/aether-gateway/src/tests/usage/direct.rs
2026-06-02 19:27:50 +08:00
elky 657e6aa672 Merge remote-tracking branch 'origin/pr/619' 2026-06-02 19:23:44 +08:00
AAEE86 7835840ebd feat(dashboard): 增加全站实时指标和自动刷新
- 管理员仪表盘新增全站 RPM/TPM 与在线/启用用户指标
- 合并今日请求/费用、全站 RPM/TPM、在线/启用用户卡片展示
- 在线用户按最近 5 分钟活跃请求去重统计
- 全站 RPM/TPM 按最近 60 秒请求与 Token 统计
- 新增仪表盘自动刷新按钮,开启后每 10 秒静默刷新数据
- 同步前端类型、空态占位和仪表盘测试
2026-06-02 18:16:24 +08:00
zhefox 6cabcd85aa fix(provider): preserve Claude messages defaults in responses conversion 2026-06-02 17:14:26 +08:00
elky 03e436707d Fix gateway usage nextest stack overflow 2026-06-02 16:59:13 +08:00
elky 781bc5ac58 Merge branch 'pr-617' 2026-06-02 10:40:34 +08:00
AAEE86 86f72da3d9 feat(health): 增加历史状态条指标 Tooltip
- 为健康监控时间轴返回 timeline_details 分段指标
- Hover 历史状态柱时展示总请求/成功/失败/可用率/状态
- 展示平均耗时/TTFB/速度和完整时间范围
- 修复历史状态柱 Tooltip 触发区域不可用的问题
- 补齐前端类型、详情抽屉透传和 mock 数据
2026-06-02 10:36:54 +08:00
elky 0a2c674ad8 Ignore tunnel release tags for app build version 2026-06-02 09:43:12 +08:00
zhefox 0daa8c196b fix(provider): preserve reasoning and Claude tool results in responses conversion 2026-06-02 09:04:55 +08:00
zhefox 98dc5925a5 fix(provider): preserve openai responses tool history in chat conversion 2026-06-02 00:35:19 +08:00
AAEE86 d5d3f09846 refactor(health): add dashboard overview and related drill-down
- Replace health monitor tabs with a dashboard layout
- Add related health drill-down for endpoint, model, and provider cards
- Render provider health as cards and hide empty monitors
2026-06-02 00:10:59 +08:00
AAEE86 0e6fc96eb1 test(gateway): run wallet usage settlement test on larger stack
Wrap the wallet settlement usage test with the large-stack async test helper to
avoid stack overflow in the default test thread.
2026-06-01 22:40:22 +08:00
AAEE86 b052f40ffb test(gateway): run base usage body capture test on larger stack
Wrap the request_record_level=base local gateway usage test with the existing
large-stack async test helper to avoid stack overflow in the default test thread.
2026-06-01 22:23:54 +08:00
AAEE86 2aef9d2478 Refine mobile usage record metadata layout 2026-06-01 21:59:45 +08:00
AAEE86 8627a18f2e test(gateway): run local usage report test on large stack
Wrap the local OpenAI chat sync usage-reporting test in the existing
large-stack harness to avoid stack overflows under nextest suite load.
2026-06-01 21:45:44 +08:00
AAEE86 21c478be22 fix(health): hide empty endpoint monitors
- Remove raw API format label from endpoint health cards
- Hide endpoint health cards with no requests
2026-06-01 21:24:20 +08:00
AAEE86 9d8f7d158b Refine mobile usage record details
- Move mobile usage actions into the card header
- Add compact user/provider metadata line on mobile
- Preserve hidden unknown toggle and auto refresh controls
2026-06-01 21:13:10 +08:00
AAEE86 c1649fe837 refactor(health): consolidate monitor components 2026-06-01 20:49:38 +08:00
AAEE86 d3c8317939 fix(health): align model health card layout 2026-06-01 18:54:56 +08:00
AAEE86 7ffe33f867 feat(health): refine health monitor metrics
- add TPS to model and provider health payloads

- exclude user-cancelled 499 requests from health statistics

- update model/provider health cards with average latency, average TTFB, TPS, and availability
2026-06-01 18:34:51 +08:00
elky 6c2a57f237 fix rust ci failures 2026-06-01 02:42:10 +08:00
github-actions[bot] 0f4141ef3f chore(tunnel): update download links for tunnel-v0.3.16 2026-05-31 17:46:42 +00:00
elky 37413c0211 Refactor tunnel stability protocol 2026-06-01 01:36:49 +08:00
Entropy.Xu d1b64b6748 修复:完善 Kiro 模拟缓存共享回收 2026-05-31 22:43:27 +08:00
Entropy.Xu c2bcfab7d4 修复:Kiro 模拟缓存接入共享运行时 2026-05-31 22:13:45 +08:00
elky 392353ffff Merge remote-tracking branch 'entropy-xu/codex/ccswitch-import' 2026-05-31 20:52:26 +08:00
Entropy.Xu 9734be31cf 修复:收敛 Kiro 模拟缓存断点语义 2026-05-31 20:45:16 +08:00
elky 905453d62b revert: remove usage elapsed clock calibration 2026-05-31 20:30:51 +08:00
Entropy.Xu a3b8a99709 修复:补齐 Kiro 模拟缓存 TTL 和消息级断点 2026-05-31 20:22:18 +08:00
Entropy.Xu 2e24e5f358 修复:扩大 Kiro 模拟缓存前缀读取范围 2026-05-31 19:52:18 +08:00
github-actions[bot] 40eb3cf6e1 chore(tunnel): update download links for tunnel-v0.3.15 2026-05-31 11:17:51 +00:00
elky 549463088c chore: bump aether-tunnel version to 0.3.15 2026-05-31 19:09:56 +08:00
elky f8b5651883 Support encoded tunnel node names 2026-05-31 16:33:06 +08:00
stabey de0a880ca6 test(gateway): 修复 usage wallet 测试栈溢出 2026-05-31 03:35:01 +08:00
stabey ba4e194cb5 test(gateway): 修复 usage base 记录测试栈溢出 2026-05-31 03:21:29 +08:00
stabey 1c05a722c1 test(gateway): 修复 usage local 同步测试栈溢出 2026-05-31 03:09:11 +08:00
stabey eda94913cf test(gateway): 修复 usage 同步测试栈溢出
CI 中 gateway_records_pending_usage_before_execution_runtime_sync_result_arrives 仍会在默认测试栈上溢出。

复用 large-stack tokio runtime 包装该测试,避免 gateway 全量测试在无业务失败时被 SIGABRT 中断。
2026-05-31 02:54:11 +08:00
stabey 3dfafbc379 fix(ai): 按 Responses 文本分片去重快照
upstream 已有 8abedecb 处理单个 OpenAI Responses 文本流中 delta 与 done/completed 快照重复输出的问题。

本提交保留该方向,并把去重状态从全局文本扩展为按 output_index/item_id 与 content_index 分片记录,避免多个 message item 或多个 text content part 共用同一段快照状态。
2026-05-31 02:54:11 +08:00
stabey 8d1e54eba6 fix(stream): 中途失败时不合成正常收尾
上游流式读取失败后,已经缓冲的局部转换状态可能是不完整的工具调用。

在 terminal failure 存在时跳过 normalizer 和 rewriter 的 finish 路径,避免把半截 tool_use 补成正常的 Claude message_stop。
2026-05-31 02:54:11 +08:00
stabey 6bfd56b54f fix(usage): 避免上游流式错误误记为成功
当上游流式响应中途失败时,sync error payload 可能同时包含合成错误体和部分上游流 body。

优先使用合成错误体生成 usage 终态,避免只因为上游先返回过 200 和部分 SSE 内容就把失败请求记录为 completed/settled。
2026-05-31 02:54:11 +08:00
elky 49f952692b Fix remaining sync chat stack overflows 2026-05-31 02:12:22 +08:00
elky fde15c9b60 Fix sync chat test stack overflow 2026-05-31 01:13:09 +08:00
elky 06f26cfacf Merge remote-tracking branch 'origin/pr/597' 2026-05-31 00:09:42 +08:00
elky 5360665432 test(gateway): avoid stack overflow in cors proxy test 2026-05-30 22:48:19 +08:00
elky a20ac1d31f fix(pool): align oauth status filter with visible state 2026-05-30 21:37:21 +08:00
elky 1bdd300606 Merge branch 'review-pr-612' 2026-05-30 21:26:18 +08:00
elky c56f0198ff Merge branch 'review-pr-611' 2026-05-30 21:26:12 +08:00
elky 02fc6bd4ef Merge branch 'review-pr-610' 2026-05-30 21:26:07 +08:00
elky c3a8352d76 Merge branch 'review-pr-603' 2026-05-30 21:25:58 +08:00
elky 463576915f Merge branch 'review-pr-602' 2026-05-30 21:25:52 +08:00
elky 1db6b9d307 Merge branch 'review-pr-596' 2026-05-30 21:25:46 +08:00
elky b5a02a118f Merge branch 'review-pr-595' 2026-05-30 21:25:39 +08:00
elky ae96d5d61b fix usage trace active key selection 2026-05-30 19:48:18 +08:00
cym ce1d532e3c fix(pool): align status filters with visible key state 2026-05-30 18:49:26 +08:00
Entropy.Xu f27485ec05 fix(kiro): 忽略图片 base64 token 估算 2026-05-30 02:59:04 +08:00
Entropy.Xu 9616f458de fix(kiro): 模拟缓存读取移动断点前缀 2026-05-30 00:42:14 +08:00
MMEXA 3455faf7da 修复格式转换优先级保持的首轮候选排序
让开启格式转换优先级保持的跨格式候选进入首轮候选页。

普通跨格式候选仍延后到后续页,保持原有兜底语义。
2026-05-29 22:01:11 +08:00
Entropy.Xu 7ed4b84654 feat(ccswitch): 添加一键导入和用量查询 2026-05-29 21:39:45 +08:00
github-actions[bot] 0d76a8e478 chore(tunnel): update download links for tunnel-v0.3.14 2026-05-29 13:32:35 +00:00
elky b9612fef9b chore: bump aether-tunnel version to 0.3.14 2026-05-29 21:21:30 +08:00
fawney19 92ae88f1be fix: avoid postgres migration version collision 2026-05-29 16:18:59 +08:00
ZheFox 91a5e58cec Merge branch 'fawney19:main' into main 2026-05-29 15:27:46 +08:00
fawney19 1658925f52 Disable key circuit breaker for pool providers 2026-05-29 15:16:06 +08:00
Entropy.Xu bb5a4454a5 feat(gateway): 添加标准文本非流式心跳 2026-05-29 14:35:16 +08:00
fawney19 9fb600df1b Fix PR 599 check regressions 2026-05-29 12:30:56 +08:00
ZheFox fff4fe4e20 Merge branch 'fawney19:main' into main 2026-05-29 12:27:38 +08:00
fawney19 3e4dfd2bac Merge branch 'pr-599' 2026-05-29 02:46:07 +08:00
fawney19 8bd82c8c95 Update endpoint base URL placeholders 2026-05-29 02:44:03 +08:00
fawney19 b59c724455 Normalize endpoint API root handling 2026-05-29 02:29:33 +08:00
ZheFox 3ee272fd53 Merge branch 'fawney19:main' into main 2026-05-29 01:48:50 +08:00
AAEE86 ab5d1f266f fix(usage): Optimize the billing layout of the request details page for mobile devices 2026-05-29 00:01:51 +08:00
Entropy.Xu 906742e3c4 fix(billing): 复用待支付套餐订单 2026-05-28 23:09:36 +08:00
AAEE86 0ee45f41e1 feat(mobile): Refine mobile usage record layout 2026-05-28 22:43:20 +08:00
RWDai 6d285410c2 Preserve dashboard daily breakdown rows 2026-05-28 22:02:42 +08:00
Entropy.Xu eaabfb83ed fix(tunnel): bound upstream clients and heartbeat deltas 2026-05-28 20:34:08 +08:00
fawney19 ef2953038e Fix usage records filtering and pool trace display 2026-05-28 20:24:48 +08:00
zhefox cc1a63bf01 fix: repair missing routing profiles snapshot 2026-05-28 18:59:04 +08:00
Novick Yuan 4b2d8cef3c fix(usage): calibrate active elapsed clock efficiently 2026-05-28 18:45:16 +08:00
fawney19 df518ad668 fix contracts usage server time header 2026-05-28 17:54:56 +08:00
fawney19 37b0c00701 Merge remote-tracking branch 'origin/main' 2026-05-28 17:19:04 +08:00
fawney19 88f03aaef2 Keep key circuit breaker out of pool scoring 2026-05-28 17:18:42 +08:00
fawney19 ffd8d273c4 Revert "Merge remote-tracking branch 'origin/pr/592'"
This reverts commit 3504875922, reversing
changes made to 5c3a1aecbe.
2026-05-28 17:10:27 +08:00
fawney19 ef2a96bcc4 Remove default hot pool size cap 2026-05-28 17:01:03 +08:00
Novick Yuan 734717899b Invalidate model routing cache after admin model writes 2026-05-28 16:57:28 +08:00
RWDai d2d28c30d9 Use bearer auth for OpenAI embedding passthrough 2026-05-28 16:53:00 +08:00
fawney19 10532e1a55 Merge pull request #594 from AAEE86/main
feat(usage): support output_config effort badge source
2026-05-28 16:48:02 +08:00
fawney19 47886abd2b Preserve streaming usage timing on refresh 2026-05-28 16:33:10 +08:00
fawney19 d076f64db3 Harden usage server timing header passthrough 2026-05-28 16:28:19 +08:00
AAEE86 60e3ffc402 feat(usage): support output_config effort badge source
- extract reasoning effort from provider request body output_config.effort
- include output_config.effort in usage list fallback SQL
- cover the new request body shape in usage metadata tests
2026-05-28 16:14:37 +08:00
fawney19 b21be24faa Merge remote-tracking branch 'origin/pr/591' 2026-05-28 16:07:08 +08:00
fawney19 3504875922 Merge remote-tracking branch 'origin/pr/592' 2026-05-28 16:07:07 +08:00
Entropy.Xu 0f6d4b9146 feat(embedding): 接入阿里云多模态向量端点 2026-05-28 16:05:36 +08:00
Mas0nShi 6ebd39ed0b Stabilize stream first-byte usage test 2026-05-28 15:33:28 +08:00
fawney19 5c3a1aecbe Merge remote-tracking branch 'origin/pr/591' 2026-05-28 15:23:26 +08:00
Mas0nShi 1a45ec9386 Fix Gemini CLI streaming policy for OpenAI chat 2026-05-28 15:05:05 +08:00
fawney19 97133f657f style: 突出路由策略选中标签样式 2026-05-28 15:03:14 +08:00
Novick Yuan b108dc5ea6 Assert usage server timing over HTTP 2026-05-28 15:01:59 +08:00
Novick Yuan 35cf44b38e Extract active usage elapsed clock 2026-05-28 14:30:15 +08:00
Novick Yuan 6412294262 Use header-only usage server timing 2026-05-28 14:30:15 +08:00
Novick Yuan 01c8592ca6 Align admin user usage timing samples 2026-05-28 14:30:15 +08:00
Novick Yuan 9b5c3ecd23 Use shared clock for active usage timers 2026-05-28 14:30:15 +08:00
Novick Yuan 6de684df59 Track server clock offset for usage data 2026-05-28 14:30:15 +08:00
Novick Yuan 8aca1f8b93 Add server time to usage responses 2026-05-28 14:30:15 +08:00
fawney19 bb2fc2ec00 Optimize health monitor database reads 2026-05-28 14:24:37 +08:00
fawney19 18566b5837 Merge remote-tracking branch 'origin/pr/587' 2026-05-28 13:56:04 +08:00
fawney19 069e1c1e60 Fix merged PR check regressions 2026-05-28 13:53:04 +08:00
fawney19 2c28d9979c Merge commit 'refs/pr/585'
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/request.rs
#	apps/aether-gateway/src/tests/ai_execute/stream_provider_gemini/local_cli.rs
#	apps/aether-gateway/src/tests/ai_execute/sync/gemini/cli.rs
#	apps/aether-gateway/src/tests/control/admin/provider_query.rs
#	crates/aether-provider-transport/src/gemini_cli/mod.rs
#	crates/aether-provider-transport/src/gemini_cli/request.rs
#	crates/aether-provider-transport/src/gemini_cli/url.rs
#	crates/aether-provider-transport/src/lib.rs
2026-05-28 13:19:42 +08:00
fawney19 0efb3d340d Merge commit 'refs/pr/530' 2026-05-28 12:53:45 +08:00
fawney19 535039c29e Merge remote-tracking branch 'origin/pr/584' 2026-05-28 12:16:51 +08:00
fawney19 93d3de1644 feat: improve routing policy diagnostics 2026-05-28 12:11:43 +08:00
AAEE86 4ce056fe45 feat(health): add model and provider health monitoring
- Rename the original health monitor to endpoint health monitor
- Add tab navigation for endpoint, model, and provider health views
- Add model health monitor cards with availability, latency, first-byte latency, and 60-point history
- Add admin-only provider health monitor with collapsible active-provider sections
- Show per-provider model health cards after expanding a provider
- Add backend model health and provider health monitor payload builders
- Add admin endpoint for provider health monitoring
- Add provider-scoped usage breakdown filtering for per-provider model statistics
- Add frontend API types and request helpers for model/provider health data
- Add demo mock data for model and provider health monitoring
- Fix model health timeline time-unit handling so request history segments render correctly

Verification:
- cargo fmt
- npm run type-check
- npm run build
- cargo test -p aether-gateway health_models
- cargo test -p aether-gateway health_providers
- cargo test -p aether-gateway gateway_exposes_frontdoor_manifest_without_proxying_upstream
2026-05-28 12:00:20 +08:00
Mas0nShi 9ad9858ac2 Merge origin/main into fix/gemini-cli-v1internal 2026-05-28 11:58:00 +08:00
MMEXA adca142d1e fix: adapt gemini cli to v1internal endpoint 2026-05-28 00:24:56 +08:00
stabey 739e39e1ca fix: 修复缓存 token usage 转换语义
统一 OpenAI、Gemini、Claude 之间缓存 token 的 usage 语义,避免 Claude 侧重复统计缓存输入 token。

同时补充 stream 合并逻辑、字段注释和覆盖转换链路的测试。
2026-05-27 23:49:07 +08:00
fawney19 14ad6e9b75 Merge remote-tracking branch 'origin/pr/583' 2026-05-27 18:42:57 +08:00
fawney19 d46d225a90 暗色模式下交换流式徽章填充与描边样式 2026-05-27 18:38:59 +08:00
ZheFox c05d227df2 Merge branch 'fawney19:main' into main 2026-05-27 17:06:31 +08:00
fawney19 42e723ff7c Clarify API key concurrency skip reasons 2026-05-27 17:00:09 +08:00
ZheFox b02d62642a Merge branch 'fawney19:main' into main 2026-05-27 16:18:21 +08:00
zhefox 8abedecb16 fix(ai): dedupe OpenAI responses text snapshot deltas 2026-05-27 16:11:02 +08:00
fawney19 d77a572dc7 fix: cast usage provider body before jsonb type checks 2026-05-27 15:48:12 +08:00
fawney19 8606455355 Merge pull request #581 from zhefox/main
fix(gateway): preserve JSON mode chat hints in responses normalization
2026-05-27 15:40:17 +08:00
fawney19 21e52722e6 Prefer provider request body for usage badges 2026-05-27 15:39:39 +08:00
fawney19 6673ab6d4a Add fast model directive service tier 2026-05-27 15:08:23 +08:00
fawney19 d488b1a680 fix auth refresh request body 2026-05-27 15:06:51 +08:00
fawney19 b9ac97ebc3 Simplify request detail cost overview 2026-05-27 14:24:57 +08:00
zhefox e09d3199c1 fix(gateway): update codex prompt cache key test 2026-05-27 13:57:48 +08:00
fawney19 ccfc4cbddc Cache provider catalog lookups 2026-05-27 13:56:39 +08:00
zhefox 41ad422002 fix(gateway): preserve JSON mode chat hints in responses normalization 2026-05-27 13:35:16 +08:00
fawney19 674cc85005 Stabilize stream runtime nextest timing 2026-05-27 11:00:04 +08:00
fawney19 dd2da69361 Merge pull request #580 from AAEE86/main
fix(mobile): improve usage and pool management layouts
2026-05-27 10:23:17 +08:00
fawney19 0ee6e393ce Record stream first byte on upstream event 2026-05-27 10:19:49 +08:00
fawney19 433a4d3c7d test(gateway): run claude pii redaction cases on large stack 2026-05-27 09:21:57 +08:00
AAEE86 049f26c03b fix(mobile): improve usage and pool management layouts
- Fix pool account batch dialog scrolling on mobile
- Rework usage records mobile filters into clearer rows
- Align user filter styling with other select filters
- Improve request detail drawer metric layout on mobile
2026-05-27 09:20:10 +08:00
fawney19 cf8372c8cb fix(usage): record visible stream first byte timing 2026-05-27 02:48:01 +08:00
fawney19 f03550415b style(usage): make fast badge white 2026-05-27 02:11:59 +08:00
fawney19 5a710c4f5e Merge remote-tracking branch 'origin/pr/578' 2026-05-27 02:07:52 +08:00
fawney19 56901f91ce Merge remote-tracking branch 'origin/pr/576' 2026-05-27 02:06:22 +08:00
fawney19 1109c3547c Merge branch 'pr-577' 2026-05-27 01:35:25 +08:00
fawney19 d24ead234d Merge branch 'pr-575'
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/family/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/family/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/chat/decision/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/request.rs
2026-05-27 01:34:19 +08:00
fawney19 d816ae5c88 Merge remote-tracking branch 'origin/pr/573' 2026-05-27 01:06:57 +08:00
fawney19 8c6e586063 Merge remote-tracking branch 'zhefox/main' 2026-05-27 01:01:34 +08:00
fawney19 c632ec616d Merge remote-tracking branch 'origin/pr/564' 2026-05-27 00:52:15 +08:00
fawney19 bd71a46c25 Merge pull request #561 from Kayphoon/codex/s3-integrated-backup 2026-05-27 00:48:59 +08:00
fawney19 e2b5c3acc8 docs: remove simple query inventory 2026-05-27 00:45:03 +08:00
AAEE86 e27ca671fd feat(usage): show reasoning and fast badges in usage records
- extract provider reasoning effort from request body metadata
- extract priority service tier and expose it as service_tier
- show reasoning level and fast badges after model names
- include badges in active request updates and usage list payloads
- add targeted backend and frontend coverage
2026-05-27 00:36:52 +08:00
fawney19 614c999871 feat(admin): expose s3 backup as module 2026-05-27 00:30:37 +08:00
fawney19 42693c2c52 chore: remove s3 backup docs 2026-05-27 00:07:37 +08:00
fawney19 e21cd72181 fix: preserve pool scan budget for exhausted accounts 2026-05-26 23:49:26 +08:00
MMEXA a9e6a7d644 fix(frontend): recover login redirect navigation 2026-05-26 23:22:30 +08:00
yangrs ba72770cab fix: wire windsurf oauth runtime scheduling 2026-05-26 22:40:46 +08:00
Kayphoon 7530bec7de test(gateway): cover chat pii redaction formats 2026-05-26 22:29:45 +08:00
zhefox 1173a4d9d5 fix(provider): split partial model fetch warnings from errors 2026-05-26 17:45:06 +08:00
zhefox aa409a8a9c fix(provider): refine endpoint default paths for openai and claude roots 2026-05-26 16:50:36 +08:00
AAEE86 949e251b2e fix(provider): 模型测试按 Key 模型权限过滤
测试模型前检查 provider key 的 allowed_models:
- 空权限视为允许所有模型
- 非空权限需匹配请求模型或映射后的实际模型
- 不匹配的 key 标记为跳过,避免发起测试请求

同时补充相关单测和前端跳过原因文案。
2026-05-26 16:46:21 +08:00
fawney19 4933ae9014 Merge pull request #572 from AAEE86/main
fix(admin): add User-Agent for Done-hub provider ops
2026-05-26 16:19:02 +08:00
AAEE86 1793443b09 fix(admin): add User-Agent for Done-hub provider ops
- Done-hub Cookie 请求增加浏览器 User-Agent
- 覆盖认证验证和余额查询的共享请求头
- 补充请求头测试,确认 Cookie 与 User-Agent 同时发送
2026-05-26 16:07:02 +08:00
ZheFox 23a36e37bb Merge branch 'fawney19:main' into main 2026-05-26 15:56:17 +08:00
zhefox c9cf1d458a fix(provider): factor model fetch route test type alias 2026-05-26 15:56:03 +08:00
zhefox d28a389a93 fix(provider): support unversioned API roots in model fetch 2026-05-26 15:39:56 +08:00
fawney19 7e76c9763d Clarify stream first byte timeout message 2026-05-26 15:02:50 +08:00
Kayphoon 9e029462aa test(gateway): stabilize stream timeout regression 2026-05-26 14:41:56 +08:00
Kayphoon 4523a2c67b fix(data): cast MySQL usage aggregates 2026-05-26 14:41:56 +08:00
Kayphoon 84c8bc960e feat(admin): add configurable S3 backups 2026-05-26 14:41:56 +08:00
zhefox c4927162b7 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-26 13:57:27 +08:00
zhefox 1ebe0aeadf fix(users): allow clearing explicit admin group memberships 2026-05-26 13:57:22 +08:00
ZheFox 992c58f2bd Merge branch 'fawney19:main' into main 2026-05-26 13:08:59 +08:00
zhefox 0bf63cc80e fix(provider): support multi-key selection in model tests 2026-05-26 13:08:35 +08:00
zhefox 5fc6dc8019 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-26 12:31:03 +08:00
zhefox 96184caa48 fix(codex): strip unsupported OpenAI responses body fields 2026-05-26 12:30:58 +08:00
fawney19 12ff87949d fix(ai): preserve combined Gemini builtin tools 2026-05-26 11:14:24 +08:00
fawney19 b75953bf4c Merge remote-tracking branch 'origin/pr/569' 2026-05-26 11:11:13 +08:00
MMEXA c733139091 fix(ai): normalize Gemini search grounding tools 2026-05-26 05:33:34 +08:00
fawney19 331d37be26 test: run sub2api balance provider ops on larger stack 2026-05-26 02:30:39 +08:00
fawney19 57ccd44b89 fix: fill provider quota execution timeout defaults 2026-05-26 02:11:10 +08:00
fawney19 d60b6e7454 test: run gemini image bridge case on larger stack 2026-05-26 01:52:01 +08:00
fawney19 50e4f27276 Merge remote-tracking branch 'origin/pr/567' 2026-05-26 01:21:03 +08:00
fawney19 a0f22ae659 fix: tighten pr 566 claude and deepseek handling 2026-05-26 00:57:28 +08:00
fawney19 235f32e10e Merge remote-tracking branch 'origin/pr/566' into review/pr-566-fix 2026-05-26 00:43:45 +08:00
fawney19 e7b3acdec3 fix(provider): format oauth import tests 2026-05-25 23:48:04 +08:00
fawney19 f3a367b02d Merge commit 'refs/pull/563/head' of github-fawney19:fawney19/Aether into review/pr-562 2026-05-25 23:44:02 +08:00
fawney19 c03aebba3f Merge branch 'pr-562' into review/pr-562 2026-05-25 23:10:29 +08:00
fawney19 4fb8955bc2 fix(provider): move key model auto-match into dialog 2026-05-25 23:03:52 +08:00
Novick Yuan 5dfccdec3e Fix stream candidate watchdog timeout semantics 2026-05-25 21:43:13 +08:00
fawney19 8b386b0aac Merge remote-tracking branch 'origin/pr/555' 2026-05-25 21:14:45 +08:00
hemo94931 9010f0806a fix(ai): sanitize Claude Read pages passthrough 2026-05-25 21:01:10 +08:00
hemo94931 495795327c fix(ai): sanitize empty Read pages for Claude tools 2026-05-25 21:01:10 +08:00
root 686311eabf test: align OpenAI image stream keepalive expectation 2026-05-25 21:01:10 +08:00
root e68b843875 Add DeepSeek thinking compatibility 2026-05-25 21:00:08 +08:00
root b46028cb85 refactor: clarify OpenAI SSE control policy 2026-05-25 21:00:08 +08:00
root fa172ecb95 fix: avoid synthetic keepalive for OpenAI streams 2026-05-25 21:00:08 +08:00
root 431311979a fix: handle split streaming terminal events 2026-05-25 20:58:17 +08:00
fawney19 d3249485fa Fix stream timeout semantics 2026-05-25 20:09:37 +08:00
zhefox 4c22a819f9 fix(provider): always show batch assign models action 2026-05-25 20:05:13 +08:00
zhefox f4d66021e4 Merge remote-tracking branch 'upstream/main'
# Conflicts:
#	crates/aether-data/src/repository/usage/mysql.rs
2026-05-25 17:29:12 +08:00
fawney19 54c5d5803b fix: cast mysql usage aggregate counters 2026-05-25 15:36:13 +08:00
fawney19 ae138ddb56 feat: update admin config import and runtime handling 2026-05-25 15:23:02 +08:00
zhefox b72abec2fc fix(gateway): spawn oauth account refresh asynchronously 2026-05-25 15:09:41 +08:00
zhefox db4f3fd210 fix(usage): cast mysql usage aggregates to numeric types 2026-05-25 13:56:12 +08:00
zhefox 230ce5df5f Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-25 13:38:38 +08:00
zhefox ec681335e8 fix(usage): treat empty body_state as missing terminal event 2026-05-25 13:38:18 +08:00
Entropy.Xu aaad113190 feat(admin-users): 支持按创建时间排序 2026-05-25 12:21:58 +08:00
calida-tec 63681b4be3 fix(provider): accept common OAuth token JSON aliases 2026-05-25 10:02:47 +08:00
MMEXA b347f1816d Fix native Antigravity stream envelope handling 2026-05-25 09:39:55 +08:00
ZheFox e9efc5c42a Merge branch 'fawney19:main' into main 2026-05-25 09:15:13 +08:00
MMEXA 28c3a5dbe4 Add Antigravity v1internal gateway adapter 2026-05-25 06:58:15 +08:00
fawney19 505d9fd8bc Merge remote-tracking branch 'origin/main' 2026-05-25 01:56:34 +08:00
fawney19 932397d1b3 fix(gateway): defer ChatGPT web image quota decrement 2026-05-25 01:56:25 +08:00
fawney19 1be445423c Merge remote-tracking branch 'origin/pr/558' 2026-05-25 01:22:03 +08:00
fawney19 2df9615fb9 Merge pull request #560 from Kayphoon/codex/remove-install-migration
fix(install): remove pg single-node migration entrypoint
2026-05-25 01:21:30 +08:00
fawney19 fe7fb17ff5 fix(gateway): align admin key health circuit summary 2026-05-25 01:18:07 +08:00
Kayphoon 72d43878ef fix(install): remove pg single-node migration entrypoint 2026-05-25 01:14:37 +08:00
fawney19 cc3ce8b5d7 Merge remote-tracking branch 'origin/pr/557' 2026-05-25 01:08:22 +08:00
fawney19 d4ae6e0e64 fix: read imported usage aggregates in dashboards 2026-05-25 00:51:46 +08:00
MMEXA 480579a0d5 fix(gateway): expire provider key circuit cooldowns 2026-05-25 00:40:38 +08:00
ZheFox ba188aea92 Merge branch 'fawney19:main' into main 2026-05-24 23:40:40 +08:00
fawney19 40b4e52508 Filter format-scoped key fields on import 2026-05-24 22:44:12 +08:00
fawney19 e2d5fc9dfb Preserve usage data in system imports 2026-05-24 21:40:48 +08:00
zhefox c92bdfba16 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-24 18:51:28 +08:00
ZheFox 83a2609344 Merge branch 'fawney19:main' into main 2026-05-24 18:51:00 +08:00
fawney19 18d9004f22 fix docker app logging permissions 2026-05-24 18:50:39 +08:00
Codex 74c8bfc59f 调整 ChatGPT Web 生图 token 估算口径 2026-05-24 18:50:14 +08:00
Codex 3bf7469d30 修复 ChatGPT Web 生图 usage 估算 2026-05-24 18:50:14 +08:00
Codex 66837b7d7f 修复 ChatGPT Web 生图发起即扣额度 2026-05-24 18:50:14 +08:00
Codex 14b182d09b 修复 ChatGPT Web 生图调用起始预扣额度 2026-05-24 18:50:14 +08:00
Codex 9dba6ec1d9 修复 ChatGPT Web 生图预扣与 Free 限额继承 2026-05-24 18:50:14 +08:00
Codex a52b513533 调整 ChatGPT Web 生图请求预扣额度 2026-05-24 18:50:14 +08:00
Codex 218ca8e6eb 修复 ChatGPT Web 生图额度递减显示 2026-05-24 18:50:14 +08:00
Codex 2b2754b779 修复 ChatGPT Web 生图成功后额度同步 2026-05-24 18:50:13 +08:00
Codex 952d1c840d 修复 ChatGPT Web 额度刷新 403 误判 2026-05-24 18:50:13 +08:00
zhefox 2207b60834 fix(usage): preserve failed status for active request refreshes 2026-05-24 18:48:50 +08:00
ZheFox c09bb28d16 Merge branch 'fawney19:main' into main 2026-05-24 18:07:09 +08:00
github-actions[bot] 13e0759d5a chore(tunnel): update download links for tunnel-v0.3.13 2026-05-24 08:43:41 +00:00
fawney19 80054276eb chore: bump aether-tunnel version to 0.3.13 2026-05-24 16:36:57 +08:00
fawney19 517c9e5108 Merge pull request #552 from RWDai/issue-549-fix
Bound models route data reads
2026-05-24 16:17:41 +08:00
fawney19 576918daa5 Optimize provider scheduler database hotspots 2026-05-24 15:47:56 +08:00
zhefox bbd4338e8e Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-24 15:21:27 +08:00
zhefox e6423a91aa feat(provider): auto-match batch assign models from key 2026-05-24 15:19:37 +08:00
fawney19 78523f122d Limit pool score interest feedback writes 2026-05-24 12:35:55 +08:00
fawney19 e1df06f06c revert compose data layout to legacy paths 2026-05-24 00:37:03 +08:00
fawney19 ecfe04f48a Merge pull request #554 from zhefox/main 2026-05-24 00:15:47 +08:00
ZheFox ff7f27d4f8 Merge branch 'fawney19:main' into main 2026-05-23 23:46:18 +08:00
fawney19 dd07425d21 Merge remote-tracking branch 'origin/pr/550'
# Conflicts:
#	frontend/src/features/usage/components/UsageRecordsTable.vue
2026-05-23 23:40:10 +08:00
ZheFox ba9aa7c1bd Merge branch 'fawney19:main' into main 2026-05-23 23:32:40 +08:00
zhefox 92fd253cae Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-23 22:24:09 +08:00
zhefox 9f6fac418e feat(usage): normalize provider stats across usage and cost views 2026-05-23 22:22:37 +08:00
fawney19 e53a2757eb Merge branch 'pr-548' 2026-05-23 22:19:51 +08:00
fawney19 db32b1d982 fix(usage): ignore truncated stream captures for terminal inference 2026-05-23 22:16:02 +08:00
fawney19 6b1c1e4f50 Merge remote-tracking branch 'origin/pr/547' 2026-05-23 22:03:17 +08:00
fawney19 3eb9614e68 Merge pull request #546 from novcky/fix/pool-scheduler-skip-invalid-oauth-accounts
修复 Provider Pool 热池反复调度已失效 OAuth 账号的问题
2026-05-23 21:38:18 +08:00
fawney19 8087a98c9d Merge remote-tracking branch 'origin/pr/475'
# Conflicts:
#	apps/aether-gateway/src/handlers/admin/provider/oauth/dispatch/refresh/execution.rs
#	apps/aether-gateway/src/handlers/admin/provider/oauth/dispatch/refresh/response.rs
#	apps/aether-gateway/src/handlers/admin/provider/oauth/errors.rs
#	apps/aether-gateway/src/handlers/admin/provider/oauth/quota/shared.rs
#	apps/aether-gateway/src/state/oauth.rs
#	apps/aether-gateway/src/tests/control/admin/oauth.rs
#	crates/aether-admin/src/provider/quota.rs
2026-05-23 21:26:22 +08:00
fawney19 5643b2c901 feat: add compose data layout migration helper 2026-05-23 20:51:02 +08:00
fawney19 18eac2dd7a feat: clarify deployment update strategies 2026-05-23 20:14:26 +08:00
ZheFox 0f2a96554f Merge branch 'fawney19:main' into main 2026-05-23 19:54:24 +08:00
RWDai 0655e868a2 Bound models route data reads 2026-05-23 19:36:20 +08:00
fawney19 4b66cadf15 Merge remote-tracking branch 'origin/pr/544' 2026-05-23 18:41:52 +08:00
Kayphoon 4ee64339ba fix(usage): show retry marker with fallback 2026-05-23 17:54:48 +08:00
Kayphoon babe328565 fix(usage): include embedding formats in filters 2026-05-23 16:19:23 +08:00
fawney19 6447fda852 fix: harden tunnel security and timeout handling 2026-05-23 14:17:04 +08:00
fawney19 74f7348529 Merge remote-tracking branch 'origin/pr/535' 2026-05-23 13:00:18 +08:00
wzw bf456450d7 feat: 代理池均衡分发&批量添加代理节点 2026-05-23 10:40:25 +08:00
zhefox 91cb2bbbcc fix(usage): refine stream terminal capture gating for OpenAI responses 2026-05-23 03:31:15 +08:00
Novick Yuan c0252387b4 修复热池调度已失效 OAuth 账号 2026-05-23 03:17:09 +08:00
zhefox bd1e155332 fix(provider): normalize OpenAI chat tool history for Claude messages 2026-05-23 02:45:04 +08:00
fawney19 ab048b8a03 fix monitoring trace lookup fallback 2026-05-23 01:22:24 +08:00
zhefox e5ce2ac7a4 fix(usage): detect missing terminal events in stream reporting 2026-05-23 00:47:36 +08:00
fawney19 c7641dad0a fix: propagate build version through local deploy 2026-05-23 00:26:07 +08:00
fawney19 b5e942ca9d fix: increase postgres shared memory for dashboard queries 2026-05-23 00:21:34 +08:00
fawney19 74c82d9948 Merge remote-tracking branch 'origin/main' 2026-05-22 23:59:05 +08:00
fawney19 d18b13a91a fix: harden frontdoor and usage ingestion 2026-05-22 23:57:38 +08:00
Mas0nShi 0d80db8a8d Refactor Gemini CLI v1internal planner request builder 2026-05-22 18:57:03 +08:00
Mas0nShi 8cc6888c5b Fix Gemini CLI OpenAI conversion envelope 2026-05-22 18:40:11 +08:00
fawney19 9af1507238 Merge pull request #545 from RWDai/feat/expand-client-types
feat: expand client type recognition
2026-05-22 18:36:53 +08:00
Mas0nShi c67818ee86 Fix Gemini CLI standard conversion envelope 2026-05-22 18:24:40 +08:00
fawney19 6ef6cbade2 Fallback dashboard aggregate reads on schema mismatch 2026-05-22 17:52:36 +08:00
Mas0nShi 8df0e1790d Fix Gemini CLI batch import parse error entry 2026-05-22 17:25:13 +08:00
Mas0nShi 8b7643e150 Merge remote-tracking branch 'origin/main' into fix/gemini-cli-v1internal
# Conflicts:
#	apps/aether-gateway/src/ai_serving/transport.rs
#	apps/aether-gateway/src/handlers/admin/provider/oauth/dispatch/batch/parse.rs
#	apps/aether-gateway/src/handlers/shared/catalog.rs
#	crates/aether-admin/src/provider/quota.rs
#	crates/aether-model-fetch/src/strategy.rs
#	crates/aether-provider-pool/src/lib.rs
#	crates/aether-provider-pool/src/service.rs
#	crates/aether-provider-transport/src/provider_types.rs
#	frontend/src/features/providers/components/ProviderDetailDrawer.vue
#	frontend/src/utils/__tests__/providerKeyQuota.spec.ts
#	frontend/src/utils/providerKeyQuota.ts
#	frontend/src/views/admin/PoolManagement.vue
2026-05-22 17:13:57 +08:00
fawney19 ef04f4b0fb Add automatic B/T compact unit formatting 2026-05-22 17:13:16 +08:00
Mas0nShi ce02f1ae8c Add Gemini CLI v1internal quota support 2026-05-22 16:58:04 +08:00
RWDai c2d0f60784 fix(gateway): accept header sessions for unknown clients 2026-05-22 16:10:27 +08:00
zhiqicloud 781830a202 fix: resolve gateway clippy regressions 2026-05-22 15:54:51 +08:00
RWDai 9dd545353c Preserve omitted tunnel security in CLI mode 2026-05-22 15:54:10 +08:00
zhiqicloud 4c7ebf8b8d style: format admin update settings 2026-05-22 15:32:49 +08:00
zhiqicloud a4a5f70a10 Merge upstream/main into feat/one-click-update 2026-05-22 15:28:27 +08:00
RWDai 9633bce2e1 feat(frontend): centralize client family labels 2026-05-22 15:28:19 +08:00
RWDai ca341703bc feat(usage): infer additional client families 2026-05-22 15:27:39 +08:00
RWDai cb2ff61bbc feat(gateway): expand client session family detection 2026-05-22 15:26:56 +08:00
RWDai 08b27806a7 Respect explicit tunnel security off 2026-05-22 15:20:36 +08:00
zhiqicloud b59c3a9e3b feat: support admin online update and deploy flow 2026-05-22 15:15:17 +08:00
fawney19 ecf6019ccb Merge pull request #543 from zhefox/main
Accept Claude message bodies in OpenAI chat endpoints
2026-05-22 14:53:25 +08:00
zhefox 2a298de971 fix(provider): serialize Claude tool results as JSON strings 2026-05-22 14:41:27 +08:00
RWDai 33633637e5 Fix secure tunnel session handling 2026-05-22 14:35:43 +08:00
fawney19 8ede01ad4e Merge remote-tracking branch 'origin/main' 2026-05-22 14:16:07 +08:00
fawney19 8966fd6aac Protect background workers under DB pool pressure 2026-05-22 14:11:47 +08:00
ZheFox 2b8ff8a743 Merge branch 'fawney19:main' into main 2026-05-22 14:11:00 +08:00
zhefox 97de4ff8a3 fix(gateway): accept Claude Messages bodies on OpenAI chat endpoints 2026-05-22 14:09:51 +08:00
fawney19 f6c3ebf7d3 Merge pull request #542 from zhefox/main
Handle OpenAI chat body responses and SSE passthrough
2026-05-22 12:05:36 +08:00
ZheFox 56abfdf39d Merge branch 'fawney19:main' into main 2026-05-22 11:51:53 +08:00
zhefox 9b95fa4d95 fix(gateway): handle responses-shaped OpenAI chat bodies and SSE passthrough 2026-05-22 11:39:38 +08:00
fawney19 f341c573eb Merge pull request #541 from AAEE86/main
feat(notification): add Bark push support
2026-05-22 11:24:11 +08:00
fawney19 b227669985 Merge pull request #540 from zhefox/main
fix(usage): treat stream terminal failures as failures on HTTP 200
2026-05-22 11:23:53 +08:00
AAEE86 ce44d35eb6 feat(notification): add Bark push support
Add Bark as a notification-service delivery channel, including encrypted Device Key configuration, server URL/template settings, module status integration, and admin UI support.
2026-05-22 11:08:00 +08:00
RWDai b05f2a270a Fix gateway secure tunnel Clippy warning 2026-05-22 10:13:06 +08:00
RWDai 2e701a90c9 Complete secure tunnel encryption support 2026-05-22 09:47:28 +08:00
zhefox 507f2f8250 fix(usage): treat stream terminal failures as failures on HTTP 200 2026-05-22 09:34:22 +08:00
Mas0nShi 9533bd7043 fix: route Gemini CLI generateContent through stream 2026-05-22 09:30:55 +08:00
fawney19 2b32b9a445 Fix system data import export flows 2026-05-22 02:46:49 +08:00
fawney19 3714c211dc Merge pull request #534 from RWDai/fix/issue-528-cache-affinity-health
fix(gateway): preserve cache affinity during health updates
2026-05-22 02:10:45 +08:00
fawney19 504c1ccb37 feat: restructure notification services 2026-05-22 01:45:46 +08:00
fawney19 d5e64d6ad9 Merge branch 'pr-503'
# Conflicts:
#	apps/aether-gateway/src/handlers/admin/provider/summary/value.rs
#	apps/aether-gateway/src/lib.rs
#	apps/aether-gateway/src/maintenance/mod.rs
#	apps/aether-gateway/src/maintenance/runtime/workers.rs
#	frontend/src/api/endpoints/types/provider.ts
2026-05-22 00:19:23 +08:00
RWDai 9859aec16c fix(gateway): evict pooled affinity after sibling key failures 2026-05-21 23:50:34 +08:00
fawney19 0e6d7539ad Merge remote-tracking branch 'origin/pr/538' 2026-05-21 23:12:42 +08:00
fawney19 d6eb41aa78 Merge remote-tracking branch 'origin/pr/536'
# Conflicts:
#	apps/aether-gateway/src/execution_runtime/stream/execution.rs
2026-05-21 23:03:14 +08:00
fawney19 97997685b5 Merge remote-tracking branch 'origin/pr/498' 2026-05-21 22:56:43 +08:00
fawney19 ab0a90de97 fix(runtime-state): govern redis connections 2026-05-21 22:53:37 +08:00
ZheFox eeb7995214 Merge branch 'fawney19:main' into main 2026-05-21 22:27:07 +08:00
zhefox 68d8f86dc6 fix(gateway): stop stream polling on downstream disconnect and preserve Codex cache keys 2026-05-21 22:26:14 +08:00
RWDai 18fe5a4f11 fix(gateway): preserve affinity during adaptive retries 2026-05-21 21:56:26 +08:00
RWDai 26ee1a9958 fix(gateway): preserve affinity during quota telemetry 2026-05-21 21:56:26 +08:00
zhefox 77c2d91eb0 fix(gateway): drain downstream-disconnected streams and stop inferring cancelled usage 2026-05-21 20:30:44 +08:00
fawney19 b8a65cbdec Merge pull request #532 from RWDai/opencode/cosmic-nebula
fix: raise group rate limits by access tier
2026-05-21 19:39:50 +08:00
zhefox 71f9afc526 fix(gateway): move openai responses helper import to ai_serving module 2026-05-21 18:09:57 +08:00
zhefox 8ca4a10f24 fix(gateway): require terminal events for OpenAI responses streams 2026-05-21 17:46:21 +08:00
Mas0nShi 66f21de50e fix: unwrap Gemini CLI model test envelopes 2026-05-21 17:39:10 +08:00
Mas0nShi 3e6ce6cf4a fix: hydrate Gemini CLI project metadata 2026-05-21 17:10:17 +08:00
stabeyandClaude Opus 4.7 cadc45c5b8 fix(data): cleanup uses failed candidate status instead of 504
The stale-pending cleanup task previously hardcoded status_code=504 and a
generic timeout message for every usage row it finalized. When a request
had already been observed as failing — e.g. upstream Connection reset by
peer, watchdog 504, or an authenticated 4xx — the cleanup overwrote that
context with a misleading "服务器超时" outcome and 504 status, hiding the
real cause from the dashboards and customer.

Pull the most recent failed/cancelled candidate per stale request_id and,
if present, finalize the usage row with the candidate's status_code
(defaulting to 502 when none was recorded) and error_message. Requests
that have no terminal candidate (truly stuck pending/streaming) keep the
existing 504 + timeout-message behavior, since they really are timeouts
from the cleanup's perspective. Applied to all three SQL backends with
parameterized UPDATE statements.

The Postgres failed-candidate lookup orders by
COALESCE(finished_at, started_at, created_at) DESC, matching the MySQL
and SQLite ORDER BY clauses so the three backends pick the same
"most recent terminal candidate" under every NULL combination of timing
columns.

Co-Authored-By: Claude Opus 4.7 <[email protected]>
2026-05-21 16:56:56 +08:00
zhefox b7b7b4f718 fix(gateway): report terminal stream failure errors consistently 2026-05-21 16:15:03 +08:00
RWDai 4f49dd5943 Document tunnel security MVP config 2026-05-21 16:09:38 +08:00
RWDai 888414c41b Forward proxy tunnel security aliases 2026-05-21 16:09:00 +08:00
RWDai 8f41bc1558 Generate secure tunnel install sessions 2026-05-21 16:08:37 +08:00
RWDai a543ca9e07 Add tunnel installer security envs 2026-05-21 16:08:14 +08:00
RWDai 40b9db3545 Add tunnel security setup fields 2026-05-21 16:07:56 +08:00
RWDai bd4f6b9206 Add tunnel security config fields 2026-05-21 16:07:27 +08:00
RWDai 776f95b1ab chore: format auth rate limit tests for rustfmt 1.95 2026-05-21 15:53:40 +08:00
zhefox 12abde2aeb fix(usage): handle terminal stream failures and preserve usage updates 2026-05-21 15:52:51 +08:00
RWDai 965a8c79af fix(gateway): preserve cache affinity during health updates 2026-05-21 15:51:56 +08:00
RWDai d33043288d fix(frontend): clarify user group policy help 2026-05-21 15:38:53 +08:00
RWDai a2ad556f2b fix(gateway): raise group rate limits by tier 2026-05-21 15:38:42 +08:00
Mas0nShi e53d5f07e8 feat: support Gemini CLI v1internal quota 2026-05-21 15:38:10 +08:00
stabeyandClaude Opus 4.7 40434005c0 fix(gateway): use total_ms for non-stream upstream watchdog
When the endpoint forces upstream_stream_policy=force_non_stream while
the client streams, the local stream candidate watchdog still preferred
timeouts.first_byte_ms — a non-stream upstream produces no early first
byte, so the watchdog fired before the HTTP request_timeout and aborted
otherwise-healthy attempts at ~300s.

Read upstream_is_stream from report_context and invert the priority:
non-stream upstreams use total_ms first, falling back to first_byte_ms
and then the default; streaming upstreams keep the previous order.

Co-Authored-By: Claude Opus 4.7 <[email protected]>
2026-05-21 15:26:42 +08:00
stabeyandClaude Opus 4.7 3330b2ac4c refactor(report-context): extract UPSTREAM_IS_STREAM_KEY constant
The "upstream_is_stream" JSON key flows from the AI execution report
context producer (aether-ai-serving::report_context) through several
consumers — usage runtime metadata copy/move, gateway watchdog, sync
execution decision, observability handlers, and the per-driver usage
repositories. Each site spelled the key as a bare string literal, so a
producer-side rename would silently degrade every consumer to its
fallback (typically assuming streaming) with no compile-time signal.

Introduce a single pub const UPSTREAM_IS_STREAM_KEY in
aether-ai-formats (the lowest crate every consumer already depends on),
re-export from the crate root, and route producer + all map-style
consumers through it. The change is purely a string-literal → constant
swap; behaviour is identical.

Sites left as literals (intentional):
- `json!({"upstream_is_stream": ...})` macro keys, which must be string
  literals at the macro layer; these are also API-response payload
  field names (an external contract that should not silently track
  internal report-context renames).
- SQL column accessors (`try_get::<...>("upstream_is_stream")`), which
  refer to the database schema column, not the JSON key.
- Test fixtures and assertions, which validate the on-the-wire contract
  and should keep verifying the actual string.

Co-Authored-By: Claude Opus 4.7 <[email protected]>
2026-05-21 15:26:42 +08:00
zhefox be6e49b9c2 Merge branch 'main' of https://github.com/zhefox/Aether 2026-05-21 12:30:48 +08:00
zhefox e59e6c3797 fix(gateway): sanitize Claude thinking and handle missing stream finish 2026-05-21 12:30:42 +08:00
Entropy.Xu 6b04a0a3a6 fix(windsurf): 修复 native 工具流式回程 2026-05-21 02:55:05 +08:00
Entropy.Xu 4112a8b2ea fix(provider): 修复 Windsurf PR CI 失败 2026-05-21 02:21:35 +08:00
fawney19 b84e4a96e2 chore: tune default postgres settings for 2c4g 2026-05-21 01:36:05 +08:00
fawney19 e7f8b259ac Revert "Merge pull request #517 from zhiqicloud/feat/provider-balance-query"
This reverts commit 7e95e769d5, reversing
changes made to 490306c242.
2026-05-21 01:24:17 +08:00
Entropy.Xu 65c361115a fix(provider): 修复 Windsurf PR 冲突残留 2026-05-21 01:02:02 +08:00
Entropy.Xu 129c7c90c0 fix(provider): 修复 Windsurf 原生工具桥接 2026-05-21 01:02:02 +08:00
Entropy.Xu 82637ad882 fix(provider): 修复 Windsurf Connect 请求与端点计数 2026-05-21 01:02:02 +08:00
Entropy.Xu 931c577345 fix(provider): 接入 Windsurf 模型测试链路 2026-05-21 01:02:02 +08:00
Entropy.Xu 9466d92a7a fix(provider): 接入 Windsurf 模型拉取和格式转换 2026-05-21 01:02:02 +08:00
Entropy.Xu 0a0a8b31c7 fix(provider): 隐藏 Windsurf refresh token 刷新入口 2026-05-21 01:02:02 +08:00
Entropy.Xu 208c77a062 fix(provider): 对齐 Windsurf PostAuth 登录链路 2026-05-21 01:02:02 +08:00
Entropy.Xu 02d1343436 feat(provider): 补充 Windsurf 邮箱密码导入表单 2026-05-21 01:02:02 +08:00
Entropy.Xu 0226e14251 feat(provider): 原生接入 Windsurf provider 2026-05-21 01:02:02 +08:00
fawney19 923515ab28 fix: align image generation checks 2026-05-21 00:45:02 +08:00
fawney19 4d0c654822 Merge remote-tracking branch 'origin/pr/524' 2026-05-20 23:17:31 +08:00
ZheFox 779877acd0 Merge branch 'fawney19:main' into main 2026-05-20 22:49:49 +08:00
fawney19 d49b0a8a45 Make extension modules reorderable 2026-05-20 22:45:32 +08:00
ZheFox 5a9f19cbf2 fix(gateway): filter upstream SSE control-only blocks 2026-05-20 22:33:41 +08:00
zhiqicloud 9562295d8b feat: 添加在线更新功能 2026-05-20 22:29:11 +08:00
ZheFox 3c6924238f feat(usage): include cache token details in stream usage payloads 2026-05-20 21:20:10 +08:00
ZheFox 64ad0f694b feat(usage): include cache token details in stream usage payloads 2026-05-20 21:06:51 +08:00
fawney19 754f672ee2 Merge pull request #526 from MMEXA/fix/codex-responses-pending-recovery-20260520
修复 Codex Responses 工具字段和成功请求回收标记
2026-05-20 21:01:54 +08:00
fawney19 e50ceeba5a Merge pull request #525 from final0920/fix/issue-505-priority-order
fix: 修复优先级管理重新打开顺序回退
2026-05-20 21:00:06 +08:00
fawney19 57910f906d Preserve usage provider identity 2026-05-20 20:56:36 +08:00
ZheFox 8838e9289b fix(provider): remove stale image preview block from model test dialog 2026-05-20 20:53:13 +08:00
ZheFox d3355a8a09 fix(gateway): decode stream-encoded provider response JSON 2026-05-20 20:40:48 +08:00
fawney19 c972bbd397 Fix provider pool exhaustion scheduling 2026-05-20 20:16:57 +08:00
MMEXA fed9bdf01a Fix Codex Responses tool schema and pending recovery 2026-05-20 12:02:00 +00:00
ZheFox ab2287202d feat(provider): show image previews in model test dialog 2026-05-20 19:59:20 +08:00
流云 b47282fe4c fix: 修复优先级管理重新打开顺序回退
优先级管理弹窗依赖 grouped-by-format 接口回显格式优先级,但该接口此前读取 summary key 行。summary 查询会清空 global_priority_by_format 和 internal_priority 等路由字段,导致保存后的数据库顺序存在,重新打开页面却回退为前端占位顺序。

改为使用完整 key 查询,并增加 summary 字段被清空时仍能回显真实优先级的回归测试。

Fixes #505

Constraint: grouped-by-format 是优先级管理弹窗的数据源,必须返回真实 per-format priority 字段。
Rejected: 修改前端继续猜测顺序 | 无法区分真实数据库优先级与占位回退。
Confidence: high
Scope-risk: narrow
Tested: cargo fmt --check; git diff --check
Not-tested: cargo test on local Windows blocked by missing NASM for boring-sys2
2026-05-20 19:44:05 +08:00
ZheFox a31e237cc3 Merge upstream main 2026-05-20 19:28:49 +08:00
ZheFox cfa32a2b4d fix(gateway): preserve openai image 200 responses and sync success reporting 2026-05-20 19:17:42 +08:00
ZheFox 2cacf66a37 fix(gateway): route openai image streams with images surface 2026-05-20 18:13:34 +08:00
fawney19 d0981c2fd5 Merge pull request #522 from RWDai/fix/admin-users-server-pagination
Fix admin users server-side pagination
2026-05-20 18:03:11 +08:00
RWDai 3e12c06627 Fix user group options cache busting 2026-05-20 17:51:29 +08:00
RWDai 74a3df3f1e Fix user group options cache invalidation 2026-05-20 17:46:16 +08:00
fawney19 cb894208a1 Merge pull request #521 from Avilianb/fix/provider-query-responses-compact-body
Fix provider model test compact request bodies
2026-05-20 17:42:53 +08:00
fawney19 76752beca6 chore(postgres): 支持通过环境变量配置 PG 性能参数 2026-05-20 17:41:52 +08:00
RWDai e80e7b0cfa Fix admin users pagination review issues 2026-05-20 17:41:06 +08:00
RWDai 77051e6245 Fix admin users pagination follow-ups 2026-05-20 17:32:47 +08:00
ZheFox de7be4f15b fix(gateway): route openai image api requests with mapped models 2026-05-20 17:16:38 +08:00
RWDai 44fb1af287 Fix admin user count clippy lint 2026-05-20 17:06:33 +08:00
fawney19 e7a76b0510 chore(docker): postgres 启用 pg_stat_statements 扩展 2026-05-20 17:00:44 +08:00
fawney19 2881ff097a feat(usage): 管理员用量统计支持筛选刷新与失败保留旧数据
- 用量统计/聚合接口新增 skipCache 选项与 120s 超时,便于强制绕过缓存
- loadStats 增加 force/preserveOnFailure 选项,背景刷新失败时保留旧数据
- 管理员页面在用户/模型/Provider 筛选变化时强制刷新统计,并将筛选条件传入统计接口
- 手动刷新与自动刷新分离,自动刷新不再重载长期热力图相关聚合
2026-05-20 16:51:52 +08:00
RWDai 462c3dde79 Load admin users with server-side pagination 2026-05-20 16:51:46 +08:00
RWDai 1be703b56e Track admin users pagination in frontend data layer 2026-05-20 16:51:46 +08:00
RWDai 5130da9710 Return paginated admin users metadata 2026-05-20 16:51:46 +08:00
RWDai 8440846bae Expose admin user export counts through gateway state 2026-05-20 16:51:46 +08:00
RWDai 831554f11d Add admin user export count queries 2026-05-20 16:51:46 +08:00
Avilianb 97fb588a4a Apply rustfmt to compact provider test 2026-05-20 16:47:30 +08:00
Avilianb cd994d57d2 Fix provider model test compact request bodies 2026-05-20 16:29:46 +08:00
fawney19 70f2882a43 refactor(api-keys): 将独立余额 Key 表单的额度、IP 限制、敏感信息保护移至左侧列 2026-05-20 16:28:58 +08:00
fawney19 fa5d26ce38 Merge remote-tracking branch 'origin/main' 2026-05-20 16:14:09 +08:00
fawney19 f76bbaab52 feat: support api key ip restriction rules 2026-05-20 16:11:49 +08:00
ZheFox cde2062618 chore(gateway): make openai image intent import local 2026-05-20 16:09:24 +08:00
ZheFox 9673fc4c01 fix(codex): trigger image override only on explicit tool_choice 2026-05-20 15:34:31 +08:00
fawney19 19ae7c8902 Merge pull request #519 from RWDai/feat/aether-tunnel-ip-family-options
feat(tunnel): add tunnel IP family controls
2026-05-20 15:30:17 +08:00
RWDai eb63838f6a fix(proxy): keep legacy installer URLs working 2026-05-20 14:43:24 +08:00
RWDai 232006f71d fix(tunnel): restore IP family flag test builds 2026-05-20 14:42:52 +08:00
fawney19 b6bdc08267 Merge branch 'pr-501' 2026-05-20 14:05:00 +08:00
fawney19 7e95e769d5 Merge pull request #517 from zhiqicloud/feat/provider-balance-query
支持 Provider Key 余额查询与自动刷新
2026-05-20 13:59:24 +08:00
RWDai f4c79c80ac fix(tunnel): allow explicit false IP family flags 2026-05-20 13:50:58 +08:00
RWDai edded777e7 docs(tunnel): document IP family controls 2026-05-20 13:43:22 +08:00
RWDai 7284165f39 feat(tunnel): add tunnel IP family controls 2026-05-20 13:42:55 +08:00
RWDai 1604a6d87d fix(gateway): allow clearing API key IP whitelists 2026-05-20 13:41:14 +08:00
ZheFox 7d5ad1e70e chore(gateway): silence dead code warnings in openai image bridge 2026-05-20 13:32:14 +08:00
ZheFox 89860bec97 fix(codex): restrict openai image routing to codex responses 2026-05-20 13:17:24 +08:00
zhiqicloud ebc1774300 修正 new_api 验证测试断言 2026-05-20 13:04:22 +08:00
zhiqicloud 122daf0f87 支持 Provider Key 余额查询与自动刷新 2026-05-20 12:33:31 +08:00
RWDai 149651f831 test(gateway): expect image bridge tools 2026-05-20 11:22:45 +08:00
fawney19 490306c242 Merge commit 'refs/pull/511/head' of github-fawney19:fawney19/Aether 2026-05-20 11:19:58 +08:00
RWDai 316b1e3207 Merge remote-tracking branch 'upstream/main' into feat/500-api-key-ip-whitelist 2026-05-20 11:14:43 +08:00
RWDai 84c4c2f9c2 fix(gateway): preserve image generation tools 2026-05-20 11:02:41 +08:00
fawney19 4d856f3deb Merge remote-tracking branch 'origin/pr/516' 2026-05-20 10:54:32 +08:00
fawney19 61bcbe826a fix: tighten OAuth auto cleanup signals 2026-05-20 10:34:22 +08:00
RWDai bdc848b19e Merge upstream main into feat/500-api-key-ip-whitelist 2026-05-20 10:26:56 +08:00
mayrain 65e3b6f3da fix(codex): trigger image override only on explicit tool_choice
The codex `apply_codex_openai_responses_special_body_edits` override
previously triggered whenever the `tools` array contained an
`image_generation` entry, regardless of whether the caller actually
asked to use it. Codex CLI advertises `image_generation` alongside
~20 other tools under `tool_choice: "auto"`, so every routine codex
conversation was being rewritten into image-generation-only form:

  - `model` forced to `gpt-5.4-mini` (CODEX_OPENAI_IMAGE_INTERNAL_MODEL)
  - `stream` forced to `true`
  - `tools` truncated to a single `image_generation` entry
  - `tool_choice` overwritten to `{"type":"image_generation"}`

The upstream ChatGPT codex backend then rejected the request with
`400 Tool choice 'image_generation' not found in 'tools' parameter`,
which the gateway surfaced as a retryable 503 to clients. The bug
reproduced on every codex CLI session that included the image tool
in its tool catalogue, even though the user never requested image
generation.

Narrow the trigger to the caller's actual selection. The new helper
`codex_openai_responses_tool_choice_references_image_generation`
matches only the explicit string `"image_generation"` or the object
form `{"type":"image_generation"}`. The pre-existing
`is_openai_image_request(provider_api_format)` branch still handles
genuine `openai:image` traffic, so true image-generation flows are
unaffected.

Tests:
  - lock the regression: `tool_choice: "auto"` with image_generation
    in tools must not trigger the override (model/tools preserved)
  - lock variants: string `"image_generation"` and object form both
    still trigger; other tool_choice values and an absent
    `tool_choice` do not
2026-05-20 09:52:39 +08:00
zhiqicloud 97b05e744f 支持官方直连支付、退款配置与套餐联动 2026-05-20 08:16:52 +08:00
fawney19 fbda210b84 Merge pull request #511 2026-05-20 01:22:55 +08:00
fawney19 ed75ae6d56 Merge pull request #509 from beilo/feat/key-ranking-usage
Add paginated API key usage leaderboard
2026-05-20 01:06:58 +08:00
fawney19 d1ad1815f7 Merge pull request #513 from mayrainnn/fix/request-body-content-encoding
feat: normalize compressed request bodies
2026-05-20 01:05:59 +08:00
fawney19 b1a3a26815 Merge pull request #512 from Entropy-Xu/codex/fix-wallet-overdraft-settlement
[codex] 修复钱包余额不足后重复消费
2026-05-20 01:05:26 +08:00
fawney19 94760dbc14 refactor(tunnel): rename aether-proxy to aether-tunnel 2026-05-20 01:02:01 +08:00
zhiqicloud 3a318a86b2 支持官方直连支付、退款配置与套餐联动 2026-05-20 00:21:56 +08:00
fawney19 f4d0d5904a Remove image_generation tools from OpenAI image bridges 2026-05-20 00:20:56 +08:00
fawney19 25c7bb935e chore(frontend): simplify concurrent limit hint text 2026-05-19 23:52:59 +08:00
fawney19 f5deed8709 refactor(proxy): improve tunnel throughput and observability 2026-05-19 23:49:36 +08:00
beilo fe3a848eb5 feat(admin): add paginated API key usage leaderboard 2026-05-19 23:17:10 +08:00
mayrain 8f4f4d2d82 refactor(gateway): route decoded body access through ai_serving 2026-05-19 22:58:03 +08:00
mayrain 66a54cc39e feat(gateway): normalize compressed request bodies 2026-05-19 22:57:45 +08:00
Entropy.Xu 7ace958710 fix(wallet): 修复余额不足后重复消费
- 有限钱包结算允许扣成负数,先扣充值余额再扣赠送余额,缺口回写到充值余额
- 日额度钱包补扣路径在存在钱包时不再把余额不足标成 insufficient_quota
- 补充内存和 SQLite 结算回归覆盖,三种数据库实现保持一致
验证:
- cargo fmt --all --check
- cargo clippy -p aether-data --all-targets -- -D warnings
- cargo clippy -p aether-gateway --all-targets -- -D warnings
- cargo clippy --workspace --exclude aether-gateway --exclude aether-data --all-targets -- -D warnings
- cargo nextest run -p aether-gateway
- cargo nextest run -p aether-data
- cargo nextest run --workspace --exclude aether-gateway --exclude aether-data
- cargo test -p aether-data sqlite --lib
- Postgres/MySQL data_db_smoke commands from rust-ci.yml
2026-05-19 19:29:22 +08:00
fawney19 57655bdb25 Hide capability tags from UI 2026-05-19 19:10:30 +08:00
fawney19 124077a0a1 Merge branch 'pr-506' 2026-05-19 18:36:36 +08:00
fawney19 1b570daf72 Revert "Merge PR #504"
This reverts commit d216a9e219, reversing
changes made to 21e82abd54.
2026-05-19 17:23:57 +08:00
fawney19 8bcd5b8189 Revert "Merge remote-tracking branch 'origin/pr-473'"
This reverts commit f2cdb74ed8, reversing
changes made to 3f0fd15395.
2026-05-19 16:31:58 +08:00
fawney19 63202a63ef Merge branch 'codex/pool-hot-trace-fix'
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/candidate_materialization.rs
2026-05-19 14:47:37 +08:00
fawney19 7e9ca88e00 fix: restore provider key circuit breaker backoff 2026-05-19 13:57:36 +08:00
beilo 75aa3dc0cc fix: refine oauth auto-removal behavior 2026-05-19 13:41:18 +08:00
fawney19 6a1da6a5ff fix: speed up admin pool loading 2026-05-19 12:05:36 +08:00
mayrain d88f092dd1 feat(kiro): add simulated cache provider toggle 2026-05-19 10:47:17 +08:00
mayrain b4d17a392a feat(kiro): simulate prompt cache usage accounting 2026-05-19 10:47:17 +08:00
fawney19 c6a408d4e8 Fix local gateway Docker native deps 2026-05-19 10:43:04 +08:00
RWDai d19bf71343 Refresh migration cutoff expectations 2026-05-19 10:25:06 +08:00
RWDai 02f09c2056 Regenerate API key schema baselines 2026-05-19 10:24:47 +08:00
RWDai 6a104d5736 Add allowed IPs to data schema sources 2026-05-19 10:24:27 +08:00
RWDai 95af482ffe Update auth snapshot observability fixture 2026-05-19 10:23:54 +08:00
RWDai b05264ff74 Stabilize wallet today usage test timing 2026-05-19 10:23:32 +08:00
RWDai 2aab1ea97b Clean up gateway API key whitelist handlers 2026-05-19 10:23:15 +08:00
RWDai f1687017e6 Fix provider endpoint format normalization path 2026-05-19 10:22:50 +08:00
fawney19 052de6b96e test: stabilize merged PR checks 2026-05-19 10:18:25 +08:00
fawney19 d2b42e91d2 test: align cancelled sync billing status 2026-05-19 08:25:39 +08:00
fawney19 d8a9d7eb5e chore: format merged PR changes 2026-05-19 08:13:56 +08:00
fawney19 d216a9e219 Merge PR #504 2026-05-19 08:10:57 +08:00
fawney19 21e82abd54 Merge PR #502 2026-05-19 08:10:48 +08:00
fawney19 18ed9a57c2 Merge PR #499 2026-05-19 08:10:37 +08:00
fawney19 702dc3ceb4 Merge PR #489 (ours: keep current main) 2026-05-19 03:38:26 +08:00
fawney19 3180ca2bf9 Merge PR #488 (ours: keep current main) 2026-05-19 03:38:22 +08:00
fawney19 69af74b1e0 Merge PR #488 and #489 2026-05-19 03:16:35 +08:00
MMEXA ef9c0ebbd4 Merge latest origin/main into codex/gemini-embedding-batch
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/mod.rs
#	apps/aether-gateway/src/execution_runtime/fallback.rs
2026-05-18 19:11:42 +00:00
MMEXA ca4d0dc819 Merge origin/main into codex/gemini-embedding-batch
# Conflicts:
#	apps/aether-gateway/src/ai_serving/api.rs
#	apps/aether-gateway/src/ai_serving/planner/passthrough/provider/family/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/family/request.rs
#	apps/aether-gateway/src/ai_serving/transport.rs
#	apps/aether-gateway/src/handlers/admin/provider/query/models/model_test/summary.rs
#	apps/aether-gateway/src/handlers/admin/provider/query/models/model_test/tests.rs
#	crates/aether-data/src/repository/candidate_selection/postgres.rs
#	crates/aether-model-fetch/src/strategy.rs
2026-05-18 19:02:19 +00:00
fawney19 cd1aa92931 Merge branch 'merge-pr-483' 2026-05-19 02:28:43 +08:00
fawney19 9fc9c334c9 Merge remote-tracking branch 'origin/pr/487'
# Conflicts:
#	crates/aether-data/src/lifecycle/bootstrap/postgres.rs
#	crates/aether-data/src/lifecycle/migrate/tests.rs
#	crates/aether-data/src/repository/oauth_providers/postgres.rs
#	crates/aether-data/src/repository/oauth_providers/sqlite.rs
#	frontend/src/views/admin/OAuthSettings.vue
2026-05-19 02:27:39 +08:00
fawney19 c563c192b7 Merge remote-tracking branch 'origin/pr-483' into merge-pr-483
# Conflicts:
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/chat/decision/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/chat/decision/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/payload.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/request.rs
#	apps/aether-gateway/src/execution_runtime/chatgpt_web_image.rs
2026-05-19 02:23:14 +08:00
fawney19 afedd90c80 Merge commit 'refs/pull/481/head' of github-fawney19:fawney19/Aether
# Conflicts:
#	apps/aether-gateway/src/ai_serving/api.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/family/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/chat/decision/request.rs
#	apps/aether-gateway/src/ai_serving/planner/standard/openai/responses/decision/request.rs
2026-05-19 01:46:41 +08:00
MMEXA be2e8e594c fix(frontend): align Vertex Gemini endpoint controls 2026-05-18 17:36:04 +00:00
fawney19 d392681c58 Merge branch 'pr-478'
# Conflicts:
#	apps/aether-gateway/src/data/state/mod.rs
#	apps/aether-gateway/src/handlers/admin/mod.rs
#	apps/aether-gateway/src/handlers/admin/routes.rs
#	crates/aether-data/src/lifecycle/bootstrap/postgres.rs
#	crates/aether-data/src/lifecycle/migrate/tests.rs
#	crates/aether-data/src/repository/announcements/postgres.rs
#	frontend/src/features/auth/components/RegisterDialog.vue
2026-05-19 01:27:42 +08:00
MMEXA 5ed8325592 fix(gateway): use Vertex model garden catalog endpoint 2026-05-18 17:06:42 +00:00
fawney19 ed1d9fdb57 Fix postgres bigint counter decoding 2026-05-19 00:58:09 +08:00
ZheFox c58ce63fc3 Merge branch 'fawney19:main' into main 2026-05-19 00:50:52 +08:00
fawney19 5eed329916 Rootfix usage counter outbox 2026-05-19 00:44:37 +08:00
fawney19 19c8688eb1 Merge pull request #477 from RWDai/feat/356-usage-record-columns
Add configurable usage record columns
2026-05-19 00:35:30 +08:00
MMEXA 4317ff78b1 Expose local scheduling failures in usage UI 2026-05-18 16:28:37 +00:00
yangrsandClaude Opus 4.7 6c16f399d4 feat: 重要通知模块、Server 酱独立配置与额度提醒
- 新增重要通知统一模块(邮件 + Server 酱)作为后台任务通知出口
- 拆出独立的 Server 酱 配置页(SendKey + Markdown 模板,支持 {title}/{body} 变量替换),通过仪表盘内置工具入口进入
- 新增提供商额度提醒后台 worker:余额低于阈值时通过重要通知推送,提供商配置页加入额度提醒开关与阈值
- 重要通知页加入配置可用性守卫:未配置任一通道时禁用总开关,未配置邮件/SendKey 时禁用对应通道开关
- 测试通知端点支持 channel 过滤(all/email/server_chan),并绕过总开关与通道开关,便于配置阶段先验证通道
- 修复:测试通知路由未在 buffered-body 白名单导致 channel 参数丢失、测试时邮件分支被误触发
- 修复:sub2api 验证响应中 username 为 null 时正确回退到 email,避免误报"验证响应缺少: 用户信息"

Co-Authored-By: Claude Opus 4.7 <[email protected]>
2026-05-18 23:51:14 +08:00
MMEXA b480f3aaff fix(gateway): normalize Gemini Vertex embedding transport 2026-05-18 15:39:29 +00:00
MMEXA 84f312fa4c fix(gateway): close dropped sync attempts 2026-05-18 14:15:22 +00:00
ZheFox 61d5fdb0ec fix(frontend): preserve fixed provider model test key inheritance 2026-05-18 22:06:27 +08:00
ZheFox 995ab302be Merge remote-tracking branch 'upstream/main'
# Conflicts:
#	apps/aether-gateway/src/handlers/admin/provider/endpoints_admin/payloads.rs
#	apps/aether-gateway/src/handlers/admin/provider/endpoints_admin/reads.rs
#	apps/aether-gateway/src/handlers/admin/provider/endpoints_admin/update.rs
#	apps/aether-gateway/src/tests/control/admin/endpoints/routes.rs
#	frontend/src/features/models/components/GlobalModelFormDialog.vue
#	frontend/src/features/providers/components/ProviderModelFormDialog.vue
#	frontend/src/features/providers/components/provider-tabs/__tests__/model-test-request.spec.ts
#	frontend/src/features/providers/components/provider-tabs/model-test-request.ts
2026-05-18 22:02:31 +08:00
RWDai c6ee558180 Add admin user key IP whitelist UI 2026-05-18 20:48:28 +08:00
RWDai 460cd63d3a Add user API key IP whitelist UI 2026-05-18 20:48:21 +08:00
RWDai 2a3593d9c5 Update planner auth snapshot fixtures 2026-05-18 20:48:14 +08:00
RWDai bcca8d295a Update auth context test fixtures 2026-05-18 20:48:05 +08:00
RWDai d8f68c1d9a Preserve allowed IP defaults in admin key imports 2026-05-18 20:47:57 +08:00
RWDai ab53326865 Support allowed IPs in admin user key endpoints 2026-05-18 20:47:46 +08:00
RWDai 96b857f642 Support allowed IPs in user API key endpoints 2026-05-18 20:47:37 +08:00
RWDai fc12cc8a36 Enforce API key IP restrictions in proxy auth 2026-05-18 20:47:28 +08:00
RWDai 276d19b63c Persist allowed IPs in auth repositories 2026-05-18 20:47:17 +08:00
RWDai 437024cdb1 Add API key allowed IP data types 2026-05-18 20:47:03 +08:00
RWDai 64b41434ce Add API key allowed IP migrations 2026-05-18 20:46:53 +08:00
RWDai 24129d7f12 Merge upstream/main into feat/356-usage-record-columns 2026-05-18 20:33:05 +08:00
fawney19 d51b44d642 Merge remote-tracking branch 'origin/pr-485'
# Conflicts:
#	crates/aether-data/src/repository/provider_catalog/sqlite.rs
2026-05-18 19:44:18 +08:00
RWDai e8c55e8f1f Label OpenAI JS SDK in cache monitoring 2026-05-18 19:30:39 +08:00
RWDai 9179516b19 Label OpenAI JS SDK in usage records 2026-05-18 19:30:19 +08:00
RWDai 37abfe66f0 Label OpenAI JS SDK in user usage 2026-05-18 19:29:59 +08:00
RWDai ae9d4038b1 Prefer typed admin usage client family 2026-05-18 19:29:36 +08:00
RWDai b6f558d10b Project usage client family in Postgres reads 2026-05-18 19:29:10 +08:00
RWDai 6d994917a0 Add usage client family read model field 2026-05-18 19:28:36 +08:00
fawney19 b99f43783c Merge remote-tracking branch 'origin/pr-482'
# Conflicts:
#	.github/workflows/release.yml
2026-05-18 18:55:11 +08:00
fawney19 7f76eff827 Merge remote-tracking branch 'origin/pr-484' 2026-05-18 18:01:57 +08:00
fawney19 be939f7e63 Merge remote-tracking branch 'origin/pr-493' 2026-05-18 18:01:43 +08:00
fawney19 a8a87d2b41 Merge remote-tracking branch 'origin/pr-494' 2026-05-18 18:00:47 +08:00
RWDai 0a26accca4 Merge remote-tracking branch 'upstream/main' into feat/356-usage-record-columns 2026-05-18 17:49:35 +08:00
fawney19 40e925680c Merge remote-tracking branch 'origin/pr-480' 2026-05-18 16:46:35 +08:00
fawney19 f2cdb74ed8 Merge remote-tracking branch 'origin/pr-473'
# Conflicts:
#	frontend/src/features/providers/components/ProviderModelFormDialog.vue
2026-05-18 16:46:22 +08:00
Codex 7ac8159728 fix: use Vertex Model Garden models endpoint 2026-05-18 08:17:09 +00:00
fawney19 3f0fd15395 Merge remote-tracking branch 'origin/pr-491' 2026-05-18 16:14:07 +08:00
fawney19 90dbc279fd Merge remote-tracking branch 'origin/pr-476' 2026-05-18 16:13:45 +08:00
fawney19 3723f165bc Merge remote-tracking branch 'origin/pr-496' 2026-05-18 16:13:03 +08:00
fawney19 3bffecddf2 fix: route access log sanitizer through gateway api 2026-05-18 15:47:02 +08:00
fawney19 a818af7833 Merge remote-tracking branch 'origin/pr-492' 2026-05-18 14:52:29 +08:00
fawney19 187b3a08fb Merge remote-tracking branch 'origin/pr-495' 2026-05-18 14:51:50 +08:00
fawney19 2405ccc0f2 Merge remote-tracking branch 'origin/pr-490' 2026-05-18 14:51:14 +08:00
fawney19 0fcfcae9c8 Merge remote-tracking branch 'origin/pr-486' 2026-05-18 14:50:48 +08:00
ZheFox 2de0cbbafc Merge branch 'fawney19:main' into main 2026-05-18 13:13:37 +08:00
ZheFox a4012ad353 Merge upstream/main 2026-05-18 13:11:11 +08:00
fawney19 e4315fbbf0 fix: refine frontend admin and auth UI 2026-05-18 12:41:57 +08:00
ZheFox a10c02ef63 feat(billing): add image output range pricing support 2026-05-18 11:52:01 +08:00
MMEXA 66f154a251 fix(gateway): avoid low OpenAI-compatible test token cap 2026-05-18 03:06:46 +00:00
fawney19 92813e6122 feat: add routing profile scheduling policies 2026-05-18 11:03:49 +08:00
MMEXA f50f26e599 fix(gateway): cover Google OpenAI-compatible roots 2026-05-18 02:15:23 +00:00
ZheFox a3094fda53 fix(gateway): ignore OpenAI tools for image intent routing 2026-05-18 09:54:10 +08:00
MMEXA b004a02e4a fix(gateway): harden Gemini endpoint routing 2026-05-18 00:53:34 +00:00
MMEXA 9586f5158e Fix fixed-provider endpoint key counts 2026-05-17 20:55:36 +00:00
ZheFox f5ace4fd6d fix(frontend): stabilize image generation overrides in model forms 2026-05-18 03:39:55 +08:00
ZheFox a05ae94cea fix(frontend): update checkbox bindings to checked events 2026-05-18 03:27:31 +08:00
ZheFox dff17b6cb1 feat(billing): support image output pricing in model forms and details 2026-05-18 03:19:58 +08:00
ZheFox 0b2a8fafce feat(billing): add image output pricing and usage tracking 2026-05-18 02:49:56 +08:00
ZheFox 691ccaaa04 fix(usage): track OpenAI image SSE completion and usage estimates 2026-05-18 01:32:45 +08:00
RWDai 26900c8c9e Show only client family in usage client column 2026-05-18 00:46:41 +08:00
MMEXA 226ce45d0d fix: pass explicit local build version 2026-05-17 16:38:22 +00:00
Kayphoon ae472d6744 fix(data): require provider id for provider usage aggregation 2026-05-18 00:36:29 +08:00
MMEXA 89d08d9953 fix: align provider model fetch state 2026-05-17 16:17:47 +00:00
ZheFox c024c782e4 feat(billing): add image quality pricing and usage tracking 2026-05-18 00:11:30 +08:00
MMEXA 84fc1e35e2 fix(frontend): support login autofill and reliable redirect 2026-05-17 16:03:57 +00:00
MMEXA 995be3781a fix(frontend): align usage APIs with Rust routes 2026-05-17 15:37:52 +00:00
MMEXA ed570155a1 fix(gateway): redact credential query values in access logs 2026-05-17 15:26:56 +00:00
ZheFox 680b617b00 feat(gateway): route OpenAI image streams through chat bridge 2026-05-17 23:09:50 +08:00
RWDai 0a62e4bc77 Remove usage request path column 2026-05-17 23:05:02 +08:00
RWDai 96d40dd21d Infer usage client family from User-Agent 2026-05-17 23:04:31 +08:00
MMEXA ff83b54c3b fix(gateway): reject empty Gemini success responses 2026-05-17 14:55:33 +00:00
MMEXA 81ff375bfd fix(gateway): route Gemini embedding batches correctly 2026-05-17 14:24:32 +00:00
dalamudx b0fc6e68ef fix: add icon_url to mysql generated baseline 2026-05-17 22:00:28 +08:00
dalamudx e1a73be2e1 fix: update schema baselines, logical schema, and snapshot cutoff for icon_url 2026-05-17 21:52:36 +08:00
dalamudx cf2c74e8e8 fix: add new migration version to test whitelist 2026-05-17 21:42:56 +08:00
dalamudx 2cb01f7a69 fix: add missing icon_url arg in test helper 2026-05-17 21:40:54 +08:00
dalamudx 48deff15c4 fix(oauth): fix login failures and add provider icon_url config
- Fix FIND_OAUTH_LINKED_USER_SQL missing allowed_providers_mode columns
- Fix TOUCH_OAUTH_LINK_SQL json/jsonb type mismatch in COALESCE
- Add icon_url field to OAuth provider config (DB, API, frontend)
- Fix admin OAuth test: accept 404 as reachable, use system proxy
2026-05-17 21:33:02 +08:00
ZheFox d6c8c14de7 feat(gateway): route OpenAI image intents through image bridge 2026-05-17 21:19:17 +08:00
RWDai b2266b588e Preserve usage origin metadata in Postgres lists 2026-05-17 21:09:59 +08:00
RWDai fcaafb3939 chore(ci): retrigger flaky sqlite smoke 2026-05-17 20:38:55 +08:00
fawney19 7d569127ae Fix active probe pool fallback tracing 2026-05-17 20:34:06 +08:00
RWDai 981020a5ab fix(ci): apply rustfmt to usage metadata handlers 2026-05-17 20:28:08 +08:00
ZheFox d9c8119bda fix(routing): match provider model names in admin routing counts 2026-05-17 18:59:41 +08:00
ZheFox 485d166912 fix(provider): count inherited endpoint formats for model tests 2026-05-17 18:04:16 +08:00
Kayphoon 5060532c51 fix: package release sqlite compose template 2026-05-17 16:33:35 +08:00
Kayphoon f29cca72ba refactor(data): limit query abstraction to postgres and sqlite 2026-05-17 15:40:44 +08:00
Kayphoon 77640d51a6 refactor(data): add select query abstraction 2026-05-17 15:04:35 +08:00
HsungKayphoon f0a6fffa87 refactor(data): introduce simple query helper 2026-05-17 14:07:23 +08:00
Entropy.Xu 1b24e1c22a fix(wallet): 修复额度耗尽后仍可消费 2026-05-17 11:49:08 +08:00
ZheFox ab0d766f47 fix(usage): stop inferring cache reads from prompt_cache_key 2026-05-17 03:04:33 +08:00
ZheFox 41ffb18604 fix(billing): avoid double counting cache read in OpenAI cache hit context 2026-05-17 02:40:01 +08:00
ZheFox b903f8ff7d fix(usage): estimate cache read tokens for cancelled requests 2026-05-17 02:13:13 +08:00
ZheFox 0483d001b4 fix(usage): bill cancelled terminal usage and preserve total token estimates 2026-05-17 01:11:42 +08:00
HsungKayphoon d290a1fdb9 fix: satisfy rust 1.95 clippy 2026-05-17 00:00:46 +08:00
ZheFox 2803e9317d fix(usage): bill cancelled terminal usage and preserve terminal telemetry 2026-05-16 23:43:40 +08:00
HsungKayphoon c5e26a1ed6 fix: align sqlite repositories with postgres behavior 2026-05-16 23:43:40 +08:00
fawney19 a2f91b4108 Cancel upstream stream on downstream disconnect 2026-05-16 22:04:11 +08:00
fawney19 664bd98056 Merge pull request #479 from Entropy-Xu/codex/fix-balance-cost-estimate
fix(gateway): 修复额度预检误判余额不足
2026-05-16 21:23:23 +08:00
mayrain 5bf236957e feat(grok): add runtime image surfaces 2026-05-16 21:15:38 +08:00
mayrain 936e1ae37b feat(grok): add admin oauth and quota support 2026-05-16 21:15:38 +08:00
mayrain cbfe1d378f feat(grok): add provider pool and transport support 2026-05-16 21:15:38 +08:00
mayrain e5f1f52759 feat(model-test): wire image previews into provider tests 2026-05-16 21:15:27 +08:00
mayrain edacc5a7d0 feat(model-test): add image-aware request helpers 2026-05-16 21:15:27 +08:00
fawney19 ed9267562b Adjust provider detail key pagination 2026-05-16 21:08:25 +08:00
Entropy.Xu bddae47454 fix(gateway): 修复额度预检误判余额不足 2026-05-16 20:45:40 +08:00
Entropy.Xu 973eb1a614 feat(referrals): 添加邀请返利和注册确认功能 2026-05-16 17:41:52 +08:00
HsungKayphoon f9f1fa928a refactor: drive pg sqlite copy from target schema 2026-05-16 17:29:14 +08:00
HsungKayphoon 527feb69db fix: cover portable migration tables in logical schema 2026-05-16 15:58:41 +08:00
RWDai ba661f1b3c Add usage record column controls 2026-05-16 15:55:38 +08:00
RWDai 4f584a71df Add usage metadata frontend plumbing 2026-05-16 15:55:38 +08:00
RWDai 97cd92a1a2 Expose usage record metadata fields 2026-05-16 15:55:38 +08:00
RWDai df7b2824a6 Persist client family usage metadata 2026-05-16 15:55:38 +08:00
RWDai 03ba1f94d7 Show execution failure reasons in request details 2026-05-16 15:42:13 +08:00
RWDai 6dd5d2fe14 Add failure notice resolver for usage records 2026-05-16 15:42:13 +08:00
RWDai a2649718ea Expose scheduling failure details in usage payload 2026-05-16 15:42:13 +08:00
RWDai 9325c2ad9d fix(providers): remove unreachable manual model add flow 2026-05-16 15:35:26 +08:00
RWDai 590151f40b feat(models): allow manual global model creation 2026-05-16 15:27:34 +08:00
HsungKayphoon 4b12ec8913 feat: add postgres to single-node migration 2026-05-16 15:26:43 +08:00
beilo 2ca4b486ec fix: auto-remove invalid oauth pool keys 2026-05-16 15:08:16 +08:00
2229 changed files with 506953 additions and 88016 deletions
+32 -5
View File
@@ -4,6 +4,10 @@
# 应用端口(默认 8084)
APP_PORT=8084
# 对外访问地址,用于一键安装、CC Switch 导入、支付回调等需要生成公网 URL 的场景。
# 生产环境建议显式配置为不带内部端口的公网域名,例如 https://aether.example.com
# AETHER_PUBLIC_BASE_URL=https://aether.example.com
# Docker Compose 镜像(默认正式版 latest;提前测试可改 rc/beta;也可固定具体版本)
# 示例:
# APP_IMAGE=ghcr.io/fawney19/aether:latest
@@ -49,15 +53,38 @@ ENCRYPTION_KEY=change-this-to-another-secure-random-string
# 启动自举管理员(仅在当前库里还没有活动管理员时生效)
# 手动部署时取消注释并设置;install.sh 首次生成配置时会提示输入。
ADMIN_EMAIL=[email protected]
ADMIN_USERNAME=admin
ADMIN_USERNAME=admin123456
# ADMIN_PASSWORD=
# ==================== 可选配置(有默认值) ====================
# 可信反向代理 IP/CIDR,只有这些来源发送的 X-Real-IP / X-Forwarded-For 会被采用。
# 默认仅信任本机回环代理:127.0.0.0/8,::1/128。
# Docker/Nginx 位于独立容器时,请按实际容器网络设置,例如:172.16.0.0/12。
# AETHER_TRUSTED_PROXY_CIDRS=127.0.0.0/8,::1/128,172.16.0.0/12
# docker compose 下 app 启动前自动执行 pending migration/backfill(默认 true)
# AETHER_GATEWAY_AUTO_PREPARE_DATABASE=true
# PostgreSQL 连接池配置(默认适合单实例/小型部署;高并发可按需调大)
# AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS=1
# AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS=20
# AETHER_GATEWAY_DATA_POSTGRES_IDLE_TIMEOUT_MS=30000
# PostgreSQL 连接池配置(默认按 CPU 自动计算;正式高并发环境可显式预算)
# AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS=12
# AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS=80
# AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS=2048
# AETHER_GATEWAY_REQUEST_BODY_BUFFER_BUDGET_MB=256
# AETHER_GATEWAY_REQUEST_BODY_READ_TIMEOUT_MS=120000
# 可选的 Payload 上限(MiB);默认及 0 均表示不限制。
# AETHER_MAX_REQUEST_BODY_MB=0
# AETHER_GATEWAY_SECURITY_CACHE_TTL_MS=1000
# AETHER_MAX_REDACTED_SYNC_RESPONSE_BODY_MB=0
# AETHER_MAX_INTERNAL_BUFFERED_BODY_MB=0
# AETHER_TUNNEL_NODE_STATUS_QUEUE_CAPACITY=1024
# PostgreSQL 容器调优:docker-compose.yml 已内置通用默认值,通常不用配置。
# 只有在 Postgres 独占大内存、或压测显示 DB 缓存/排序/维护任务成为瓶颈时再覆盖。
# 内置默认:shared_buffers=1GB, effective_cache_size=3GB, shm_size=512mb,
# work_mem=16MB, maintenance_work_mem=256MB。
# POSTGRES_SHARED_BUFFERS=8GB
# POSTGRES_EFFECTIVE_CACHE_SIZE=24GB
# POSTGRES_SHM_SIZE=2gb
# POSTGRES_WORK_MEM=16MB
# POSTGRES_MAINTENANCE_WORK_MEM=1GB
@@ -1,15 +1,15 @@
name: Build aether-proxy
name: Build aether-tunnel
on:
push:
tags: ['proxy-v*']
tags: ['tunnel-v*']
workflow_dispatch:
permissions:
contents: write
concurrency:
group: build-proxy-${{ github.ref }}
group: build-tunnel-${{ github.ref }}
cancel-in-progress: false
jobs:
@@ -19,23 +19,23 @@ jobs:
steps:
- uses: actions/checkout@v5
- name: Ensure proxy tag matches Cargo version
- name: Ensure tunnel tag matches Cargo version
shell: bash
run: |
TAG="${GITHUB_REF_NAME}"
EXPECTED="${TAG#proxy-v}"
ACTUAL="$(cargo metadata --manifest-path apps/aether-proxy/Cargo.toml --locked --no-deps --format-version 1 | jq -r '.packages[] | select(.name == "aether-proxy") | .version')"
EXPECTED="${TAG#tunnel-v}"
ACTUAL="$(cargo metadata --manifest-path apps/aether-tunnel/Cargo.toml --locked --no-deps --format-version 1 | jq -r '.packages[] | select(.name == "aether-tunnel") | .version')"
echo "tag version: ${EXPECTED}"
echo "cargo version: ${ACTUAL}"
if [ -z "${ACTUAL}" ]; then
echo "Could not resolve aether-proxy package version" >&2
echo "Could not resolve aether-tunnel package version" >&2
exit 1
fi
if [ "${EXPECTED}" != "${ACTUAL}" ]; then
echo "proxy tag ${TAG} does not match apps/aether-proxy/Cargo.toml version ${ACTUAL}" >&2
echo "tunnel tag ${TAG} does not match apps/aether-tunnel/Cargo.toml version ${ACTUAL}" >&2
exit 1
fi
@@ -91,7 +91,7 @@ jobs:
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
workspaces: apps/aether-proxy -> target
workspaces: apps/aether-tunnel -> target
key: ${{ matrix.target }}
- name: Install cross
@@ -99,7 +99,7 @@ jobs:
uses: taiki-e/install-action@cross
- name: Build
working-directory: apps/aether-proxy
working-directory: apps/aether-tunnel
shell: bash
run: |
if [ "${{ matrix.use_cross }}" = "true" ]; then
@@ -113,24 +113,25 @@ jobs:
shell: bash
run: |
cd target/${{ matrix.target }}/release
chmod +x aether-proxy
tar czf ../../../aether-proxy-${{ matrix.name }}.tar.gz aether-proxy
chmod +x aether-tunnel
tar czf ../../../aether-tunnel-${{ matrix.name }}.tar.gz aether-tunnel
- name: Package (Windows)
if: runner.os == 'Windows'
shell: bash
run: |
cd target/${{ matrix.target }}/release
7z a ../../../aether-proxy-${{ matrix.name }}.zip aether-proxy.exe
7z a ../../../aether-tunnel-${{ matrix.name }}.zip aether-tunnel.exe
- name: Upload artifact
uses: actions/upload-artifact@v5
with:
name: aether-proxy-${{ matrix.name }}
name: aether-tunnel-${{ matrix.name }}
path: |
aether-proxy-*.tar.gz
aether-proxy-*.zip
aether-tunnel-*.tar.gz
aether-tunnel-*.zip
if-no-files-found: error
retention-days: 1
release:
needs: build
@@ -145,7 +146,7 @@ jobs:
- name: Generate checksums
working-directory: artifacts
run: sha256sum aether-proxy-* > SHA256SUMS.txt
run: sha256sum aether-tunnel-* > SHA256SUMS.txt
- name: Delete stale draft releases for tag
env:
@@ -174,7 +175,7 @@ jobs:
name: "${{ github.ref_name }}"
generate_release_notes: true
files: |
artifacts/aether-proxy-*
artifacts/aether-tunnel-*
artifacts/SHA256SUMS.txt
fail_on_unmatched_files: true
@@ -191,25 +192,25 @@ jobs:
env:
TAG: ${{ github.ref_name }}
run: |
VERSION="${TAG#proxy-v}"
VERSION="${TAG#tunnel-v}"
BASE="https://github.com/fawney19/Aether/releases/download/${TAG}"
if [ -d apps/aether-proxy ]; then
PROXY_DIR="apps/aether-proxy"
if [ -d apps/aether-tunnel ]; then
TUNNEL_DIR="apps/aether-tunnel"
else
PROXY_DIR="aether-proxy"
TUNNEL_DIR="aether-tunnel"
fi
cd "$PROXY_DIR"
cd "$TUNNEL_DIR"
TABLE="| Platform | Download |\n|----------|----------|\n"
TABLE+="| Linux x86_64 (GNU) | [aether-proxy-linux-amd64.tar.gz](${BASE}/aether-proxy-linux-amd64.tar.gz) |\n"
TABLE+="| Linux ARM64 (GNU) | [aether-proxy-linux-arm64.tar.gz](${BASE}/aether-proxy-linux-arm64.tar.gz) |\n"
TABLE+="| Linux x86_64 (musl) | [aether-proxy-linux-musl-amd64.tar.gz](${BASE}/aether-proxy-linux-musl-amd64.tar.gz) |\n"
TABLE+="| Linux ARM64 (musl) | [aether-proxy-linux-musl-arm64.tar.gz](${BASE}/aether-proxy-linux-musl-arm64.tar.gz) |\n"
TABLE+="| macOS x86_64 | [aether-proxy-macos-amd64.tar.gz](${BASE}/aether-proxy-macos-amd64.tar.gz) |\n"
TABLE+="| macOS ARM64 | [aether-proxy-macos-arm64.tar.gz](${BASE}/aether-proxy-macos-arm64.tar.gz) |\n"
TABLE+="| Windows x86_64 | [aether-proxy-windows-amd64.zip](${BASE}/aether-proxy-windows-amd64.zip) |"
TABLE+="| Linux x86_64 (GNU) | [aether-tunnel-linux-amd64.tar.gz](${BASE}/aether-tunnel-linux-amd64.tar.gz) |\n"
TABLE+="| Linux ARM64 (GNU) | [aether-tunnel-linux-arm64.tar.gz](${BASE}/aether-tunnel-linux-arm64.tar.gz) |\n"
TABLE+="| Linux x86_64 (musl) | [aether-tunnel-linux-musl-amd64.tar.gz](${BASE}/aether-tunnel-linux-musl-amd64.tar.gz) |\n"
TABLE+="| Linux ARM64 (musl) | [aether-tunnel-linux-musl-arm64.tar.gz](${BASE}/aether-tunnel-linux-musl-arm64.tar.gz) |\n"
TABLE+="| macOS x86_64 | [aether-tunnel-macos-amd64.tar.gz](${BASE}/aether-tunnel-macos-amd64.tar.gz) |\n"
TABLE+="| macOS ARM64 | [aether-tunnel-macos-arm64.tar.gz](${BASE}/aether-tunnel-macos-arm64.tar.gz) |\n"
TABLE+="| Windows x86_64 | [aether-tunnel-windows-amd64.zip](${BASE}/aether-tunnel-windows-amd64.zip) |"
# Replace content between markers
if grep -q '<!-- DOWNLOAD_TABLE_START -->' README.md; then
@@ -222,17 +223,17 @@ jobs:
- name: Commit and push
run: |
if [ -d apps/aether-proxy ]; then
PROXY_DIR="apps/aether-proxy"
if [ -d apps/aether-tunnel ]; then
TUNNEL_DIR="apps/aether-tunnel"
else
PROXY_DIR="aether-proxy"
TUNNEL_DIR="aether-tunnel"
fi
cd "$PROXY_DIR"
cd "$TUNNEL_DIR"
git config user.name "github-actions[bot]"
git config user.email "github-actions[bot]@users.noreply.github.com"
git add README.md
git diff --cached --quiet && exit 0
TAG="${GITHUB_REF#refs/tags/}"
git commit -m "chore(proxy): update download links for ${TAG}"
git commit -m "chore(tunnel): update download links for ${TAG}"
git push
+13 -2
View File
@@ -146,6 +146,7 @@ jobs:
- name: Build
env:
AETHER_VERSION: ${{ needs.preflight.outputs.version_tag }}
AETHER_BUILD_TYPE: release
CARGO_TERM_COLOR: always
shell: bash
run: |
@@ -247,8 +248,10 @@ jobs:
set -euo pipefail
if [[ "${GITHUB_REF_TYPE}" == "tag" ]]; then
VERSION="${GITHUB_REF_NAME}"
SOURCE_REF="${GITHUB_REF_NAME}"
else
VERSION="snapshot-${GITHUB_SHA::7}"
SOURCE_REF="${GITHUB_SHA}"
fi
mkdir -p package release-assets
@@ -262,9 +265,14 @@ jobs:
install -m 0755 "artifacts/aether-gateway-${platform}-${arch}/aether-gateway" "${root}/bin/aether-gateway"
cp -R artifacts/frontend-dist/. "${root}/frontend/"
sed "s/^VERSION=\"\${AETHER_VERSION:-}\"/VERSION=\"\${AETHER_VERSION:-${VERSION}}\"/" install.sh > "${root}/install.sh"
sed \
-e "s/^SOURCE_REF=\"\${AETHER_SOURCE_REF:-main}\"/SOURCE_REF=\"\${AETHER_SOURCE_REF:-${SOURCE_REF}}\"/" \
-e "s/^VERSION=\"\${AETHER_VERSION:-}\"/VERSION=\"\${AETHER_VERSION:-${VERSION}}\"/" \
install.sh > "${root}/install.sh"
chmod 0755 "${root}/install.sh"
install -m 0755 update.sh "${root}/update.sh"
install -m 0644 docker-compose.yml "${root}/docker-compose.yml"
install -m 0644 docker-compose.single-node.yml "${root}/docker-compose.single-node.yml"
install -m 0644 .env.example "${root}/.env.example"
install -m 0755 generate_keys.sh "${root}/generate_keys.sh"
install -m 0644 README.md "${root}/README.md"
@@ -274,7 +282,10 @@ jobs:
done
done
sed "s/^VERSION=\"\${AETHER_VERSION:-}\"/VERSION=\"\${AETHER_VERSION:-${VERSION}}\"/" install.sh > release-assets/install.sh
sed \
-e "s/^SOURCE_REF=\"\${AETHER_SOURCE_REF:-main}\"/SOURCE_REF=\"\${AETHER_SOURCE_REF:-${SOURCE_REF}}\"/" \
-e "s/^VERSION=\"\${AETHER_VERSION:-}\"/VERSION=\"\${AETHER_VERSION:-${VERSION}}\"/" \
install.sh > release-assets/install.sh
chmod +x release-assets/install.sh
(cd release-assets && sha256sum *.tar.gz > SHA256SUMS)
+172 -18
View File
@@ -25,6 +25,8 @@ concurrency:
env:
CARGO_INCREMENTAL: 0
CARGO_PROFILE_DEV_DEBUG: 0
CARGO_PROFILE_TEST_DEBUG: 0
CARGO_TERM_COLOR: always
jobs:
@@ -68,7 +70,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo clippy -p aether-gateway --all-targets -- -D warnings
run: cargo clippy -p aether-gateway --lib --bins --examples -- -D warnings
- name: Show sccache stats
if: always()
@@ -136,7 +138,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo clippy --workspace --exclude aether-gateway --exclude aether-data --all-targets -- -D warnings
run: cargo clippy --workspace --exclude aether-gateway --exclude aether-data --exclude aether-integration-tests --all-targets -- -D warnings
- name: Show sccache stats
if: always()
@@ -184,14 +186,27 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Setup mold
uses: rui314/setup-mold@v1
- name: Install nextest
uses: taiki-e/install-action@nextest
- name: Test
- name: Test lib
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run -p aether-gateway
RUST_MIN_STACK: "16777216"
RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
run: cargo nextest run -p aether-gateway --lib
- name: Test bin
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
RUST_MIN_STACK: "16777216"
RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
run: cargo nextest run -p aether-gateway --bin aether-gateway
- name: Show sccache stats
if: always()
@@ -237,6 +252,45 @@ jobs:
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
check_data_features:
name: Check (Data Feature - ${{ matrix.feature }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
feature:
- postgres
- mysql
- sqlite
- all-drivers
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Check selected data driver
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo check -p aether-data --no-default-features --features ${{ matrix.feature }}
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
test_rest:
name: Test (Workspace Rest)
runs-on: ubuntu-latest
@@ -265,7 +319,79 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run --workspace --exclude aether-gateway --exclude aether-data
run: cargo nextest run --workspace --exclude aether-gateway --exclude aether-data --exclude aether-integration-tests
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
test_data_adapters:
name: Test (Data Adapter - ${{ matrix.package }})
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
package:
- aether-data-postgres
- aether-data-mysql
- aether-data-sqlite
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Install nextest
uses: taiki-e/install-action@nextest
- name: Test adapter
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo nextest run -p ${{ matrix.package }}
- name: Show sccache stats
if: always()
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: sccache --show-stats
check_integration_scenarios:
name: Test (Integration Scenarios)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
- name: Rust cache
uses: Swatinem/rust-cache@v2
with:
shared-key: rust-ci-${{ runner.os }}
workspaces: . -> target
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Test scenario binaries
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo test -p aether-integration-tests --bins
- name: Show sccache stats
if: always()
@@ -280,14 +406,20 @@ jobs:
needs:
- test_gateway
- test_data
- check_data_features
- test_rest
- test_data_adapters
- check_integration_scenarios
if: ${{ always() }}
steps:
- name: Verify test jobs
run: |
if [ "${{ needs.test_gateway.result }}" != "success" ] || \
[ "${{ needs.test_data.result }}" != "success" ] || \
[ "${{ needs.test_rest.result }}" != "success" ]; then
[ "${{ needs.check_data_features.result }}" != "success" ] || \
[ "${{ needs.test_rest.result }}" != "success" ] || \
[ "${{ needs.test_data_adapters.result }}" != "success" ] || \
[ "${{ needs.check_integration_scenarios.result }}" != "success" ]; then
echo "Tests failed"
exit 1
fi
@@ -317,7 +449,7 @@ jobs:
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
run: cargo test -p aether-data sqlite --lib
run: cargo test -p aether-data --all-features sqlite --lib
- name: Show sccache stats
if: always()
@@ -361,26 +493,48 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/[email protected]
- name: Add PostgreSQL server binaries to PATH
run: echo "$(pg_config --bindir)" >> "$GITHUB_PATH"
- name: Run Postgres migration smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data postgres_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features postgres_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
- name: Run Postgres provider metadata migration smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data --all-features postgres_provider_upstream_metadata_migration_preserves_json_when_url_is_set --lib -- --nocapture
- name: Run Postgres API key lifecycle tests
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_REQUIRE_LOCAL_POSTGRES_TESTS: "true"
run: |
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_request_candidates_preserve_deleted_api_key_identity --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_request_candidate_migration_decouples_legacy_api_key_foreign_key --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_stats_daily_api_key_migration_decouples_legacy_foreign_key --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_expired_api_key_cleanup_preserves_historical_identity --lib -- --exact --nocapture
cargo test -p aether-data --all-features lifecycle::migrate::tests::postgres_api_key_leaderboard_user_filter_preserves_aggregate_history --lib -- --exact --nocapture
- name: Run Postgres core export smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data postgres_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features postgres_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
- name: Run SQLite-to-Postgres import smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_POSTGRES_URL: postgres://aether:[email protected]:5432/aether_test
run: cargo test -p aether-data sqlite_core_export_reads_migrated_database_rows --lib -- --nocapture
run: cargo test -p aether-data --all-features sqlite_core_export_reads_migrated_database_rows --lib -- --nocapture
- name: Show sccache stats
if: always()
@@ -430,56 +584,56 @@ jobs:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_migrations_create_core_config_tables_when_url_is_set --lib -- --nocapture
- name: Run MySQL usage write smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_usage_write_repository_upserts_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_usage_write_repository_upserts_and_flushes_counters_when_url_is_set --lib -- --nocapture
- name: Run MySQL usage read smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_usage_read_repository_reads_usage_contract_views_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_usage_read_repository_reads_usage_contract_views_when_url_is_set --lib -- --nocapture
- name: Run MySQL provider catalog smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_provider_catalog_repository_round_trips_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_provider_catalog_repository_round_trips_when_url_is_set --lib -- --nocapture
- name: Run MySQL core export smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_core_export_reads_migrated_database_rows_when_url_is_set --lib -- --nocapture
- name: Run MySQL wallet read smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_wallet_read_repository_reads_wallet_contract_views --lib -- --nocapture
run: cargo test -p aether-data-mysql mysql_wallet_read_repository_reads_wallet_contract_views --lib -- --nocapture
- name: Run MySQL wallet daily usage aggregation smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_wallet_daily_usage_aggregation_uses_settlement_wallets_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_wallet_daily_usage_aggregation_uses_settlement_wallets_when_url_is_set --lib -- --nocapture
- name: Run MySQL stats aggregation smoke test
env:
RUSTC_WRAPPER: sccache
SCCACHE_GHA_ENABLED: "true"
AETHER_TEST_MYSQL_URL: mysql://aether:[email protected]:3306/aether_test
run: cargo test -p aether-data mysql_stats_aggregation_runs_after_mysql_migrations_when_url_is_set --lib -- --nocapture
run: cargo test -p aether-data --all-features mysql_stats_aggregation_runs_after_mysql_migrations_when_url_is_set --lib -- --nocapture
- name: Show sccache stats
if: always()
+4 -1
View File
@@ -1,6 +1,9 @@
# Created by https://www.toptal.com/developers/gitignore/api/python
# Edit at https://www.toptal.com/developers/gitignore?templates=python
*.rsa
*_rsa
# AI Assistant Configuration
.codex/
.claude/
@@ -244,4 +247,4 @@ src/_version.py
# Analysis folder (third-party code for reference)
analysis/
new-api/
apps/aether-proxy/aether-proxy.toml
apps/aether-tunnel/aether-tunnel.toml
+2
View File
@@ -0,0 +1,2 @@
[tools]
rust = "latest"
Generated
+963 -49
View File
File diff suppressed because it is too large Load Diff
+73 -27
View File
@@ -1,32 +1,52 @@
[workspace]
members = [
"apps/aether-proxy",
"crates/aether-ai-formats",
"apps/aether-tunnel",
"crates/aether-ai/formats",
"crates/aether-admin",
"crates/aether-ai-serving",
"crates/aether-admission-core",
"crates/aether-ai/serving",
"crates/aether-pool-core",
"crates/aether-provider-pool",
"crates/aether-data-contracts",
"crates/aether-data-schema",
"crates/aether-provider/core",
"crates/aether-provider/pool",
"crates/aether-routing-core",
"crates/aether-data/contracts",
"crates/aether-data/adapters/postgres",
"crates/aether-data/adapters/mysql",
"crates/aether-data/adapters/sqlite",
"crates/aether-data/query",
"crates/aether-data/schema",
"crates/aether-dispatch-core",
"crates/aether-cache",
"crates/aether-billing",
"crates/aether-wallet",
"crates/aether-crypto",
"crates/aether-contracts",
"crates/aether-data",
"crates/aether-data/runtime",
"crates/aether-model-fetch",
"crates/aether-oauth",
"crates/aether-provider-transport",
"crates/aether-provider/transport",
"crates/aether-scheduler-core",
"crates/aether-runtime-state",
"crates/aether-task-runtime",
"crates/aether-usage-runtime",
"crates/aether-runtime/state",
"crates/aether-task/runtime",
"crates/aether-task/core",
"crates/aether-gateway/frontdoor",
"crates/aether-gateway/control",
"crates/aether-gateway/execution",
"crates/aether-gateway/workers",
"crates/aether-gateway/tunnel",
"crates/aether-testing/loadtools",
"crates/aether-testing/integration",
"crates/aether-usage/core",
"crates/aether-testing/support",
"crates/aether-usage/runtime",
"crates/aether-video-tasks-core",
"apps/aether-gateway",
"crates/aether-http",
"crates/aether-runtime",
"crates/aether-testkit",
"crates/aether-runtime/base",
"crates/aether-testing/testkit",
]
default-members = [
"apps/aether-gateway",
]
resolver = "2"
@@ -37,32 +57,50 @@ repository = "https://github.com/fawney19/Aether.git"
[workspace.dependencies]
aether-admin = { path = "crates/aether-admin" }
aether-ai-formats = { path = "crates/aether-ai-formats" }
aether-ai-serving = { path = "crates/aether-ai-serving" }
aether-admission-core = { path = "crates/aether-admission-core" }
aether-ai-formats = { path = "crates/aether-ai/formats" }
aether-ai-serving = { path = "crates/aether-ai/serving" }
aether-pool-core = { path = "crates/aether-pool-core" }
aether-provider-pool = { path = "crates/aether-provider-pool" }
aether-data-contracts = { path = "crates/aether-data-contracts" }
aether-data-schema = { path = "crates/aether-data-schema" }
aether-provider-core = { path = "crates/aether-provider/core" }
aether-provider-pool = { path = "crates/aether-provider/pool" }
aether-routing-core = { path = "crates/aether-routing-core" }
aether-data-contracts = { path = "crates/aether-data/contracts" }
aether-data-postgres = { path = "crates/aether-data/adapters/postgres" }
aether-data-mysql = { path = "crates/aether-data/adapters/mysql" }
aether-data-sqlite = { path = "crates/aether-data/adapters/sqlite" }
aether-data-query = { path = "crates/aether-data/query" }
aether-data-schema = { path = "crates/aether-data/schema" }
aether-dispatch-core = { path = "crates/aether-dispatch-core" }
aether-cache = { path = "crates/aether-cache" }
aether-billing = { path = "crates/aether-billing" }
aether-wallet = { path = "crates/aether-wallet" }
aether-crypto = { path = "crates/aether-crypto" }
aether-contracts = { path = "crates/aether-contracts" }
aether-data = { path = "crates/aether-data" }
aether-data = { path = "crates/aether-data/runtime" }
aether-model-fetch = { path = "crates/aether-model-fetch" }
aether-oauth = { path = "crates/aether-oauth" }
aether-provider-transport = { path = "crates/aether-provider-transport" }
aether-provider-transport = { path = "crates/aether-provider/transport" }
aether-scheduler-core = { path = "crates/aether-scheduler-core" }
aether-runtime-state = { path = "crates/aether-runtime-state" }
aether-task-runtime = { path = "crates/aether-task-runtime" }
aether-usage-runtime = { path = "crates/aether-usage-runtime" }
aether-runtime-state = { path = "crates/aether-runtime/state" }
aether-task-runtime = { path = "crates/aether-task/runtime" }
aether-task-core = { path = "crates/aether-task/core" }
aether-gateway-frontdoor = { path = "crates/aether-gateway/frontdoor" }
aether-gateway-control = { path = "crates/aether-gateway/control" }
aether-gateway-execution = { path = "crates/aether-gateway/execution" }
aether-gateway-workers = { path = "crates/aether-gateway/workers" }
aether-gateway-tunnel = { path = "crates/aether-gateway/tunnel" }
aether-loadtools = { path = "crates/aether-testing/loadtools" }
aether-integration-tests = { path = "crates/aether-testing/integration" }
aether-test-support = { path = "crates/aether-testing/support" }
aether-usage-core = { path = "crates/aether-usage/core" }
aether-usage-runtime = { path = "crates/aether-usage/runtime" }
aether-video-tasks-core = { path = "crates/aether-video-tasks-core" }
aether-gateway = { path = "apps/aether-gateway" }
aether-http = { path = "crates/aether-http" }
aether-runtime = { path = "crates/aether-runtime" }
aether-testkit = { path = "crates/aether-testkit" }
aether-runtime = { path = "crates/aether-runtime/base" }
aether-testkit = { path = "crates/aether-testing/testkit" }
aes = "0.8"
aes-gcm = "0.10"
async-stream = "0.3"
async-trait = "0.1"
axum = "0.8"
@@ -72,13 +110,16 @@ bytes = "1"
cbc = "0.1"
chrono = { version = "0.4", features = ["serde"] }
chrono-tz = "0.10"
crypto_box = { version = "0.9", features = ["seal"] }
ed25519-dalek = { version = "2.2", features = ["pkcs8"] }
flate2 = "1"
futures-util = "0.3"
hmac = "0.12"
http = "1"
object_store = { version = "0.12", default-features = false, features = ["aws"] }
pbkdf2 = { version = "0.12", default-features = false, features = ["hmac"] }
reqwest = { version = "0.12", default-features = false, features = ["json", "stream", "rustls-tls", "http2", "socks"] }
redis = { version = "0.28", default-features = false, features = ["tokio-comp", "script", "streams"] }
redis = { version = "0.28", default-features = false, features = ["tokio-comp", "script", "streams", "connection-manager"] }
regex = "1"
rustls = { version = "0.23", features = ["ring"] }
semver = "1"
@@ -86,7 +127,9 @@ serde = { version = "1", features = ["derive"] }
serde_json = { version = "1", features = ["preserve_order"] }
serde_path_to_error = "0.1"
sha2 = "0.10"
sqlx = { version = "0.8", default-features = false, features = ["postgres", "mysql", "sqlite", "runtime-tokio-rustls", "chrono"] }
socket2 = "0.6"
tar = "0.4"
sqlx = { version = "0.8", default-features = false, features = ["runtime-tokio-rustls", "chrono"] }
thiserror = "2"
tokio = { version = "1", features = ["macros", "net", "rt-multi-thread", "signal", "sync", "time"] }
tokio-util = { version = "0.7", features = ["codec", "io-util"] }
@@ -94,7 +137,10 @@ tracing = "0.1"
tracing-subscriber = { version = "0.3", features = ["env-filter", "json"] }
uuid = { version = "1", features = ["serde", "v4", "v5"] }
webpki-roots = "0.26"
wreq = { version = "6.0.0-rc.28", default-features = false, features = ["json", "stream", "socks", "webpki-roots", "ws"] }
wreq-util = "3.0.0-rc.10"
url = "2"
zstd = "0.13"
[profile.dev]
# Keep file/line information for backtraces while avoiding full debug info
+27 -15
View File
@@ -1,31 +1,43 @@
# syntax=docker/dockerfile:1
# Aether Gateway 运行时镜像(交叉编译方案)
# 二进制和前端产物均由 CI 预先构建,此 Dockerfile 仅做打包
# 用法: docker buildx build --platform linux/amd64,linux/arm64 -f Dockerfile.app .
# Aether Gateway runtime image (cross-compilation)
# Binary and frontend assets are pre-built by CI; this Dockerfile only packages them.
# Usage: docker buildx build --platform linux/amd64,linux/arm64 -f Dockerfile.app .
#
# 构建上下文中须包含:
# dist/aether-gateway-amd64 (x86_64-unknown-linux-musl 交叉编译产物)
# dist/aether-gateway-arm64 (aarch64-unknown-linux-musl 交叉编译产物)
# dist/frontend/ (npm run build 产物)
# Build context must contain:
# dist/aether-gateway-amd64 (x86_64-unknown-linux-musl cross-compiled binary)
# dist/aether-gateway-arm64 (aarch64-unknown-linux-musl cross-compiled binary)
# dist/frontend/ (npm run build output)
FROM gcr.io/distroless/static-debian12
# --- layout stage: create /opt/aether directory structure with symlink ---
# distroless has no shell, so we use busybox to set up the symlink.
FROM busybox:1.37-musl AS layout
# TARGETARCH 由 buildx 自动注入: amd64 或 arm64
ARG TARGETARCH
COPY dist/aether-gateway-${TARGETARCH} /usr/local/bin/aether-gateway
COPY dist/frontend/ /srv/frontend
RUN mkdir -p /opt/aether/releases/image/bin /opt/aether/releases/image/frontend /opt/aether/logs
WORKDIR /app
COPY dist/aether-gateway-${TARGETARCH} /opt/aether/releases/image/bin/aether-gateway
RUN chmod 0755 /opt/aether/releases/image/bin/aether-gateway
COPY dist/frontend/ /opt/aether/releases/image/frontend/
RUN ln -s /opt/aether/releases/image /opt/aether/current
# --- final stage: distroless runtime ---
FROM gcr.io/distroless/static-debian12
COPY --from=layout /opt/aether /opt/aether
WORKDIR /opt/aether
ENV RUST_LOG=aether_gateway=info \
APP_PORT=8084 \
AETHER_GATEWAY_STATIC_DIR=/srv/frontend
AETHER_UPDATE_STRATEGY=docker \
AETHER_GATEWAY_STATIC_DIR=/opt/aether/current/frontend
EXPOSE 8084
HEALTHCHECK --interval=30s --timeout=10s --start-period=5s --retries=3 \
CMD ["/usr/local/bin/aether-gateway", "--healthcheck"]
CMD ["/opt/aether/current/bin/aether-gateway", "--healthcheck"]
USER root
ENTRYPOINT ["/usr/local/bin/aether-gateway"]
ENTRYPOINT ["/opt/aether/current/bin/aether-gateway"]
+24 -8
View File
@@ -1,11 +1,16 @@
# syntax=docker/dockerfile:1
# syntax=docker.m.daocloud.io/docker/dockerfile:1
# Aether 运行镜像:Rust gateway 直接服务 API + 前端静态文件(国内镜像源版本)
# 构建命令: docker build -f Dockerfile.app.local -t aether-app:latest .
# 构建命令: docker build --build-arg AETHER_BUILD_VERSION=v0.7.2 -f Dockerfile.app.local -t aether-app:latest .
ARG RUST_VERSION=1.95.0
ARG NODE_BASE_IMAGE=docker.m.daocloud.io/library/node:22-slim
ARG RUST_BASE_IMAGE=docker.m.daocloud.io/library/rust:${RUST_VERSION}-slim
# ==================== 前端构建 ====================
FROM node:22-slim AS frontend-builder
FROM ${NODE_BASE_IMAGE} AS frontend-builder
ARG AETHER_BUILD_VERSION
ENV AETHER_BUILD_VERSION=${AETHER_BUILD_VERSION} \
AETHER_VERSION=${AETHER_BUILD_VERSION}
WORKDIR /app/frontend
COPY frontend/package*.json ./
RUN --mount=type=cache,id=aether-npm-cache,target=/root/.npm,sharing=locked \
@@ -15,13 +20,14 @@ COPY frontend/ ./
RUN npm run build
# ==================== Rust gateway 构建 ====================
FROM rust:${RUST_VERSION}-slim AS gateway-base
FROM ${RUST_BASE_IMAGE} AS gateway-base
WORKDIR /build
# 本地镜像优先缩短构建时间,保留 release 语义,但改用更快的 thin LTO。
# 生产级 release 构建:保留 thin LTO,同时用 lld 缩短最终链接阶段。
ENV CARGO_REGISTRIES_CRATES_IO_PROTOCOL=sparse \
CARGO_PROFILE_RELEASE_LTO=thin \
CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16 \
RUSTFLAGS="-C linker=clang -C link-arg=-fuse-ld=lld"
RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
--mount=type=cache,target=/var/lib/apt,sharing=locked \
@@ -29,7 +35,12 @@ RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
apt-get update && apt-get install -y --no-install-recommends \
build-essential \
ca-certificates \
clang \
cmake \
git \
libclang-dev \
libssl-dev \
lld \
pkg-config \
perl
@@ -44,11 +55,14 @@ COPY crates/ ./crates/
RUN cargo chef prepare --recipe-path recipe.json
FROM gateway-base AS gateway-builder
ARG AETHER_BUILD_VERSION
ENV AETHER_BUILD_VERSION=${AETHER_BUILD_VERSION} \
AETHER_VERSION=${AETHER_BUILD_VERSION}
COPY --from=gateway-planner /build/recipe.json ./recipe.json
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-local,target=/build/target,sharing=locked \
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --recipe-path recipe.json
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --features jemalloc --recipe-path recipe.json
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
@@ -56,7 +70,8 @@ COPY crates/ ./crates/
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-local,target=/build/target,sharing=locked \
cargo build --release --locked -p aether-gateway && \
set -eux; \
cargo build --release --locked -p aether-gateway --bin aether-gateway --features jemalloc; \
cp target/release/aether-gateway /tmp/aether-gateway
# ==================== 最小运行时打包 ====================
@@ -125,6 +140,7 @@ ENV LANG=C.UTF-8 \
LC_ALL=C.UTF-8 \
RUST_LOG=aether_gateway=info \
APP_PORT=8084 \
AETHER_UPDATE_STRATEGY=manual \
AETHER_GATEWAY_STATIC_DIR=/srv/frontend
EXPOSE 8084
+150
View File
@@ -0,0 +1,150 @@
# syntax=docker.m.daocloud.io/docker/dockerfile:1
# Aether 本地发布版联调镜像
# 作用:用当前源码构建一个 release-layout 容器,专门测试管理后台在线更新流程。
ARG RUST_VERSION=1.95.0
ARG NODE_BASE_IMAGE=docker.m.daocloud.io/library/node:22-slim
ARG RUST_BASE_IMAGE=docker.m.daocloud.io/library/rust:${RUST_VERSION}-slim
# ==================== 前端构建 ====================
FROM ${NODE_BASE_IMAGE} AS frontend-builder
ARG AETHER_BUILD_VERSION
ENV AETHER_BUILD_VERSION=${AETHER_BUILD_VERSION} \
AETHER_VERSION=${AETHER_BUILD_VERSION}
WORKDIR /app/frontend
COPY frontend/package*.json ./
RUN --mount=type=cache,id=aether-npm-cache,target=/root/.npm,sharing=locked \
npm config set registry https://registry.npmmirror.com && \
npm ci --no-audit --no-fund
COPY frontend/ ./
RUN npm run build
# ==================== Rust gateway 构建 ====================
FROM ${RUST_BASE_IMAGE} AS gateway-base
WORKDIR /build
ENV CARGO_REGISTRIES_CRATES_IO_PROTOCOL=sparse \
CARGO_PROFILE_RELEASE_LTO=thin \
CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16
RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
--mount=type=cache,target=/var/lib/apt,sharing=locked \
sed -i 's/deb.debian.org/mirrors.tuna.tsinghua.edu.cn/g' /etc/apt/sources.list.d/debian.sources && \
apt-get update && apt-get install -y --no-install-recommends \
build-essential \
ca-certificates \
cmake \
git \
libclang-dev \
libssl-dev \
pkg-config \
perl
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
cargo install cargo-chef --locked
FROM gateway-base AS gateway-planner
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
COPY crates/ ./crates/
RUN cargo chef prepare --recipe-path recipe.json
FROM gateway-base AS gateway-builder
ARG AETHER_BUILD_VERSION
ARG AETHER_BUILD_TYPE=release
ENV AETHER_BUILD_VERSION=${AETHER_BUILD_VERSION} \
AETHER_VERSION=${AETHER_BUILD_VERSION} \
AETHER_BUILD_TYPE=${AETHER_BUILD_TYPE}
COPY --from=gateway-planner /build/recipe.json ./recipe.json
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-release-local,target=/build/target,sharing=locked \
cargo chef cook --release --locked --package aether-gateway --bin aether-gateway --features jemalloc --recipe-path recipe.json
COPY Cargo.toml Cargo.lock ./
COPY apps/ ./apps/
COPY crates/ ./crates/
RUN --mount=type=cache,id=aether-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=aether-cargo-git,target=/usr/local/cargo/git,sharing=locked \
--mount=type=cache,id=aether-cargo-target-release-local,target=/build/target,sharing=locked \
cargo build --release --locked -p aether-gateway --features jemalloc && \
cp target/release/aether-gateway /tmp/aether-gateway
# ==================== 最小运行时打包 ====================
FROM gateway-builder AS runtime-prep
RUN set -eux; \
mkdir -p \
/runtime-root/app/data \
/runtime-root/etc \
/runtime-root/etc/ssl \
/runtime-root/lib \
/runtime-root/lib64 \
/runtime-root/usr/lib \
/runtime-root/opt/aether/logs \
/runtime-root/opt/aether/releases/image/bin \
/runtime-root/opt/aether/releases/image/frontend; \
cp /tmp/aether-gateway /runtime-root/opt/aether/releases/image/bin/aether-gateway; \
ln -s /opt/aether/releases/image /runtime-root/opt/aether/current; \
: > /tmp/runtime-libs.txt; \
: > /tmp/runtime-scan-queue.txt; \
printf '%s\n' /tmp/aether-gateway >> /tmp/runtime-scan-queue.txt; \
while [ -s /tmp/runtime-scan-queue.txt ]; do \
current="$(head -n1 /tmp/runtime-scan-queue.txt)"; \
sed -i '1d' /tmp/runtime-scan-queue.txt; \
ldd "$current" | awk '/=>/ { print $3 } $1 ~ /^\// { print $1 }' | while read -r lib; do \
[ -n "$lib" ]; \
if ! grep -Fxq "$lib" /tmp/runtime-libs.txt; then \
printf '%s\n' "$lib" >> /tmp/runtime-libs.txt; \
printf '%s\n' "$lib" >> /tmp/runtime-scan-queue.txt; \
fi; \
done; \
done; \
sort -u /tmp/runtime-libs.txt -o /tmp/runtime-libs.txt; \
while read -r lib; do \
[ -n "$lib" ]; \
dest="/runtime-root$(dirname "$lib")"; \
mkdir -p "$dest"; \
cp -L "$lib" "$dest/"; \
done < /tmp/runtime-libs.txt; \
for lib in \
/lib/x86_64-linux-gnu/libnss_dns.so.2 \
/lib/x86_64-linux-gnu/libnss_files.so.2 \
/lib/x86_64-linux-gnu/libresolv.so.2; do \
if [ -f "$lib" ]; then \
dest="/runtime-root$(dirname "$lib")"; \
mkdir -p "$dest"; \
cp -L "$lib" "$dest/"; \
fi; \
done; \
cp -a /usr/lib/ssl /runtime-root/usr/lib/; \
cp -a /etc/ssl/certs /runtime-root/etc/ssl/; \
if [ -f /etc/ssl/openssl.cnf ]; then \
cp /etc/ssl/openssl.cnf /runtime-root/etc/ssl/openssl.cnf; \
fi; \
if [ -f /etc/nsswitch.conf ]; then \
cp /etc/nsswitch.conf /runtime-root/etc/nsswitch.conf; \
fi
COPY --from=frontend-builder /app/frontend/dist /runtime-root/opt/aether/releases/image/frontend
# ==================== 运行时镜像 ====================
FROM scratch
COPY --from=runtime-prep /runtime-root/ /
WORKDIR /app
ENV LANG=C.UTF-8 \
LC_ALL=C.UTF-8 \
RUST_LOG=aether_gateway=info \
APP_PORT=8084 \
AETHER_BASE_DIR=/opt/aether \
AETHER_UPDATE_STRATEGY=self \
AETHER_GATEWAY_STATIC_DIR=/opt/aether/current/frontend
EXPOSE 8084
HEALTHCHECK --interval=30s --timeout=10s --start-period=5s --retries=3 \
CMD ["/opt/aether/current/bin/aether-gateway", "--healthcheck"]
ENTRYPOINT ["/opt/aether/current/bin/aether-gateway"]
+5
View File
@@ -335,6 +335,11 @@ if ! command -v curl >/dev/null 2>&1; then
exit 1
fi
if [ -z "$${RUSTC_WRAPPER:-}" ] && command -v sccache >/dev/null 2>&1; then
export RUSTC_WRAPPER="$$(command -v sccache)"
echo "=> 启用 Rust 编译缓存: $${RUSTC_WRAPPER}"
fi
if ! ensure_dev_infra; then
exit 1
fi
+72 -12
View File
@@ -48,17 +48,68 @@ cp .env.example .env
./generate_keys.sh
# 编辑 .env 设置 ADMIN_PASSWORD
# 3. 首次部署 / 更新 (从以下数据库、内存策略任选其一)
# 3. 首次部署 / 更新 (从以下部署形态任选其一)
# Postgres + Redis (适用于企业或多人使用)
docker compose pull && docker compose up -d
# 仅SQLite (适用于个人用户或朋友分享)
docker compose -f docker-compose.sqlite.yml pull && docker compose -f docker-compose.sqlite.yml up -d
# Single Node (适用于个人用户或朋友分享)
docker compose -f docker-compose.single-node.yml pull && docker compose -f docker-compose.single-node.yml up -d
```
### 一键安装(可选部署方式 Linux: systemd; Mac: launchd)
### 一键更新
Docker Compose 部署后,可在部署目录直接执行:
```bash
cd Aether && cd Aether
./update.sh
```
`update.sh` 会拉取最新 `app` 镜像并重建 `app` 容器,Docker named volumes、`./data` 和 `./logs` 不会被删除。Single Node 部署也可显式指定:
```bash
./update.sh --mode single-node
```
仓库自带的 Docker Compose 默认把应用日志输出到容器 `stdout/stderr`,直接用 `docker compose logs -f app` 查看,并由 Docker 轮转日志,避免正式发布镜像切换到非 root 用户后再被宿主机挂载日志目录的权限问题拖垮启动。如果你确实需要文件日志,需要在 compose 里把 `AETHER_LOG_DESTINATION` 改成 `file|both`,并额外挂载一个容器用户可写的目录到 `/opt/aether/logs`。
管理后台右上角“版本信息”会检测新版本。Docker Compose 部署只提示版本,实际更新继续执行 `./update.sh`;systemd / launchd / 二进制部署才使用后台自更新,流程是下载对应平台的 GitHub Release 包、强制校验 `SHA256SUMS`、解压到 `/opt/aether/releases/<version>`,再切换 `/opt/aether/current` 并退出进程,交给 systemd / launchd 拉起新版本。
源码或本地构建版本不会启用后台在线更新,请继续使用源码更新流程。Docker Compose 用户如果希望“容器重建后也保持镜像层面的新版本”,仍建议定期运行 `./update.sh` 拉取并重建 app 镜像。服务器访问 GitHub 需要代理时,可设置 `AETHER_UPDATE_PROXY_URL`,也兼容 `UPDATE_PROXY_URL`、`HTTPS_PROXY`、`ALL_PROXY`、`HTTP_PROXY` 以及 `NO_PROXY`。共享出口触发 GitHub API 限流时,可设置只读 `AETHER_UPDATE_GITHUB_TOKEN`,也兼容 `GITHUB_TOKEN` / `GH_TOKEN`。下载总超时默认 600 秒,连续无响应/无数据默认 30 秒,可通过 `AETHER_UPDATE_DOWNLOAD_TIMEOUT_SECS` 和 `AETHER_UPDATE_DOWNLOAD_IDLE_TIMEOUT_SECS` 调整。
标准 Docker Compose 使用 Docker named volumes 存放 Postgres/Redis/MySQL 数据;Single Node 使用部署目录下的 `./data` 存放 SQLite 数据。
如果是本地源码构建镜像的部署,继续使用:
```bash
./deploy.sh
```
如果要在本机联调“管理后台在线更新”本身,可启动仓库内置的 release-layout 测试环境:
```bash
docker compose -f docker-compose.release-local.yml up -d --build
```
这套环境会用当前源码构建一个本地测试镜像,但编译为 `release` 类型,并默认伪装成 `v0.7.0`,这样后台会按正式发布版逻辑开放“立即更新”。默认监听 `http://127.0.0.1:18085`,数据目录使用 `./data-release-local`;日志默认走 `docker logs`,不会影响你正在跑的源码构建容器。
如果这套容器在 `prepare-update` 时访问 GitHub 失败,而你本机是通过代理出网,请在 `.env` 里把 `AETHER_UPDATE_PROXY_URL` 写成宿主机地址,例如 `http://host.docker.internal:7890`;容器内的 `127.0.0.1` 指向容器自身,不是宿主机。
如果想重置这套联调环境(包括 `/opt/aether/current` 和已下载的历史版本),执行:
```bash
docker compose -f docker-compose.release-local.yml down -v
```
可选变量:
- `AETHER_RELEASE_LOCAL_VERSION`:本地联调镜像对外声明的当前版本,默认 `v0.7.0`
- `AETHER_RELEASE_LOCAL_PORT`:本地联调端口,默认 `18085`
- `LOCAL_RELEASE_APP_IMAGE`:本地联调镜像名,默认 `aether-app:release-local`
### 一键安装(默认 Single Node:Linux systemd / macOS launchd + SQLite)
```bash
git clone https://github.com/fawney19/Aether.git
cd Aether
curl -fsSL https://raw.githubusercontent.com/fawney19/Aether/main/install.sh | sudo bash
```
@@ -73,14 +124,14 @@ make dev
`make dev` 会同时启动后端 `aether-gateway` 和前端 `frontend` 的 Vite dev server。需要单独启动时可使用 `make dev-backend` 或 `make dev-frontend`。
Postgres / Redis 本地依赖未就绪时,`make dev` 会自动执行 `docker compose up -d postgres redis`。
## Aether Proxy (可选)
## Aether Tunnel (可选)
Aether Proxy 是配套的正向代理节点,部署在海外 VPS 上,为墙内的 Aether 实例中转 API 流量。
Aether Tunnel 是配套的正向代理节点,部署在海外 VPS 上,为墙内的 Aether 实例中转 API 流量。
- Docker Compose 部署或下载预编译二进制直接运行
- 提供 macOS/Linux 与 Windows 一键脚本,自动下载最新 `proxy-v*` 制品并向现有 `aether-proxy.toml` 追加 `[[servers]]`
- 通过 `aether-proxy setup` 完成交互式配置,自动注册为系统服务
- 详细文档见 [apps/aether-proxy/README.md](apps/aether-proxy/README.md)
- 提供 macOS/Linux 与 Windows 一键脚本,自动下载最新 `tunnel-v*` 制品并向现有 `aether-tunnel.toml` 追加 `[[servers]]`
- 通过 `aether-tunnel setup` 完成交互式配置,自动注册为系统服务
- 详细文档见 [apps/aether-tunnel/README.md](apps/aether-tunnel/README.md)
## API 文档
@@ -91,8 +142,17 @@ Aether Proxy 是配套的正向代理节点,部署在海外 VPS 上,为墙
- `APP_PORT`:`aether-gateway` 唯一监听端口,固定绑定 `0.0.0.0:${APP_PORT}`
- `DATABASE_URL`:数据库连接串;SQLite 例如 `sqlite:///opt/aether/data/aether.db`,Postgres 例如 `postgresql://postgres:aether@postgres:5432/aether`
- `AETHER_GATEWAY_DATA_POSTGRES_MIN_CONNECTIONS` / `AETHER_GATEWAY_DATA_POSTGRES_MAX_CONNECTIONS`:数据库连接池手动覆盖值;未配置时会自动推导,SQLite 固定 `1/1`,Postgres/MySQL 按 CPU 核心数计算并默认封顶 `100`
- `AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS`:单实例请求并发上限;未配置时按 CPU 自动推导(基础范围 `512-65536`),低文件描述符预算时会进一步下调
- `AETHER_GATEWAY_REQUEST_BODY_BUFFER_BUDGET_MB`:单实例同时读取和解压请求体的加权内存预算,默认 `256MB`
- `AETHER_GATEWAY_REQUEST_BODY_READ_TIMEOUT_MS`:请求体完整读取超时,默认 `120000ms`
- `AETHER_MAX_REQUEST_BODY_MB`:可选的单请求解压后请求体上限;未配置或设为 `0` 时不限制
- `AETHER_MAX_INTERNAL_BUFFERED_BODY_MB`:可选的 heartbeat、管理探测等内部整包响应体上限;未配置或设为 `0` 时不限制
- `AETHER_TUNNEL_NODE_STATUS_QUEUE_CAPACITY`:隧道节点状态上报队列容量,默认 `1024`;满载时拒绝新事件,避免控制面故障导致无界内存增长
- `AETHER_GATEWAY_SECURITY_CACHE_TTL_MS`:IP 黑白名单本地缓存时间,默认 `1000ms`,写操作会主动失效相关缓存
- `AETHER_MAX_REDACTED_SYNC_RESPONSE_BODY_MB`:可选的 PII 恢复同步响应缓冲上限;未配置或设为 `0` 时不限制
- `REDIS_URL`:Redis 连接串;仅 Postgres + Redis 的 Docker Compose 部署需要配置
- `AETHER_RUNTIME_BACKEND=memory|redis`:运行时缓存/协调后端。SQLite 默认用 `memory`,不会连接 Redis
- `AETHER_RUNTIME_BACKEND=memory|redis`:运行时缓存/协调后端。SQLite 默认用 `memory`,不会连接 Redis;多节点部署和需要跨 gateway 重启恢复 OpenAI Responses continuation history 的部署必须使用共享 Redis
- `AETHER_GATEWAY_AUTO_PREPARE_DATABASE`:常规启动前自动执行挂起的 schema migration 和 backfill;仓库自带的 `docker-compose.yml` 默认开启
- `JWT_SECRET_KEY` / `ENCRYPTION_KEY`:认证和敏感数据加密所需密钥
- `API_KEY_PREFIX`:用户和管理员新建 API Key 时使用的前缀,默认 `sk`
@@ -117,4 +177,4 @@ Aether Proxy 是配套的正向代理节点,部署在海外 VPS 上,为墙
## Star History
[![Star History Chart](https://api.star-history.com/svg?repos=fawney19/Aether&type=Date)](https://star-history.com/#fawney19/Aether&Date)
[![Star History Chart](https://api.star-history.com/svg?repos=fawney19/Aether&type=date&legend=top-left)](https://www.star-history.com/?repos=fawney19%2FAether&type=date&legend=top-left)
+35 -5
View File
@@ -6,6 +6,16 @@ license.workspace = true
repository.workspace = true
description = "Rust ingress gateway for Aether phase 3a transparent proxy"
[features]
default = []
jemalloc = [
"dep:tikv-jemallocator",
"dep:tikv-jemalloc-sys",
"tikv-jemallocator/stats",
"tikv-jemalloc-sys/stats",
]
testkit = []
[dependencies]
aether-admin.workspace = true
aether-ai-formats.workspace = true
@@ -14,15 +24,21 @@ aether-billing.workspace = true
aether-cache.workspace = true
aether-contracts.workspace = true
aether-crypto.workspace = true
aether-data.workspace = true
aether-data = { workspace = true, features = ["all-drivers"] }
aether-data-contracts.workspace = true
aether-dispatch-core.workspace = true
aether-gateway-frontdoor.workspace = true
aether-gateway-control.workspace = true
aether-gateway-execution.workspace = true
aether-http.workspace = true
aether-gateway-workers.workspace = true
aether-gateway-tunnel.workspace = true
aether-model-fetch.workspace = true
aether-oauth.workspace = true
aether-pool-core.workspace = true
aether-provider-pool.workspace = true
aether-provider-transport.workspace = true
aether-routing-core.workspace = true
aether-scheduler-core.workspace = true
aether-runtime.workspace = true
aether-runtime-state.workspace = true
@@ -30,6 +46,7 @@ aether-task-runtime.workspace = true
aether-usage-runtime.workspace = true
aether-video-tasks-core.workspace = true
aether-wallet.workspace = true
aes-gcm.workspace = true
async-stream.workspace = true
async-trait.workspace = true
axum = { version = "0.8", features = ["ws"] }
@@ -44,17 +61,26 @@ flate2.workspace = true
futures-util.workspace = true
hmac.workspace = true
http.workspace = true
http-body-util = "0.1"
hyper = { version = "1", features = ["client", "server", "http1", "http2"] }
hyper-util = { version = "0.1", features = ["client-legacy", "client-pool", "server-auto", "service", "tokio"] }
ldap3 = { version = "0.11", default-features = false, features = ["sync", "tls-rustls"] }
libc = "0.2"
md-5 = "0.10"
object_store.workspace = true
parking_lot = "0.12"
regex.workspace = true
reqwest.workspace = true
rsa = "0.9.10"
rustls.workspace = true
serde.workspace = true
serde_json.workspace = true
sha1 = "0.10"
sha2.workspace = true
sqlx.workspace = true
sha2 = { workspace = true, features = ["oid"] }
socket2.workspace = true
tar.workspace = true
sqlx = { workspace = true, features = ["postgres", "mysql", "sqlite", "migrate"] }
sysinfo = "0.32"
thiserror.workspace = true
tokio.workspace = true
tokio-util.workspace = true
@@ -64,10 +90,14 @@ tracing.workspace = true
url.workspace = true
uuid.workspace = true
webpki-roots.workspace = true
wreq.workspace = true
wreq-util.workspace = true
zstd.workspace = true
[target.'cfg(not(target_env = "msvc"))'.dependencies]
tikv-jemallocator = "0.6"
tikv-jemallocator = { version = "0.6", optional = true }
tikv-jemalloc-sys = { version = "0.6", optional = true }
[dev-dependencies]
aether-testkit.workspace = true
aether-test-support.workspace = true
tracing-subscriber.workspace = true
+26 -16
View File
@@ -2,30 +2,44 @@ use std::env;
use std::process::Command;
fn main() {
println!("cargo:rerun-if-env-changed=AETHER_BUILD_VERSION");
println!("cargo:rerun-if-env-changed=AETHER_BUILD_TYPE");
println!("cargo:rerun-if-env-changed=AETHER_VERSION");
println!("cargo:rerun-if-env-changed=GITHUB_REF_NAME");
println!("cargo:rerun-if-changed=../../.git/HEAD");
let package_version = env::var("CARGO_PKG_VERSION").unwrap_or_else(|_| "unknown".to_string());
let version = env::var("AETHER_VERSION")
let version = env::var("AETHER_BUILD_VERSION")
.ok()
.filter(|value| !value.trim().is_empty())
.and_then(|value| normalize_gateway_version_source(&value))
.or_else(|| {
env::var("AETHER_VERSION")
.ok()
.and_then(|value| normalize_gateway_version_source(&value))
})
.or_else(|| {
env::var("GITHUB_REF_NAME")
.ok()
.filter(|value| value.trim().starts_with('v'))
.and_then(|value| normalize_gateway_version_source(&value))
})
.or_else(git_describe_version)
.map(|value| normalize_version(&value))
.filter(|value| !value.is_empty())
.unwrap_or(package_version);
println!("cargo:rustc-env=AETHER_BUILD_VERSION={version}");
let build_type = env::var("AETHER_BUILD_TYPE")
.ok()
.filter(|value| !value.trim().is_empty())
.unwrap_or_else(|| "source".to_string());
println!("cargo:rustc-env=AETHER_BUILD_TYPE={build_type}");
}
fn git_describe_version() -> Option<String> {
let output = Command::new("git")
.args(["describe", "--tags", "--always", "--dirty"])
.args([
"describe", "--tags", "--match", "v[0-9]*", "--always", "--dirty",
])
.output()
.ok()?;
if !output.status.success() {
@@ -33,17 +47,13 @@ fn git_describe_version() -> Option<String> {
}
let version = String::from_utf8(output.stdout).ok()?;
let version = version.trim();
if version.is_empty() {
None
} else {
Some(version.to_string())
}
normalize_gateway_version_source(version)
}
fn normalize_version(value: &str) -> String {
value
.trim()
.strip_prefix('v')
.unwrap_or(value.trim())
.to_string()
fn normalize_gateway_version_source(value: &str) -> Option<String> {
let trimmed = value.trim();
if trimmed.is_empty() || trimmed.starts_with("tunnel-v") {
return None;
}
Some(trimmed.strip_prefix('v').unwrap_or(trimmed).to_string())
}
+5 -4
View File
@@ -7,10 +7,11 @@ pub(crate) use crate::handlers::admin::{
provider_quota_refresh_endpoint_for_provider, provider_type_supports_quota_refresh,
reconcile_admin_fixed_provider_template_endpoints,
refresh_provider_oauth_account_state_after_update, refresh_provider_pool_quota_locally,
update_existing_provider_oauth_catalog_key, AdminAppState,
AdminGatewayProviderTransportSnapshot, AdminLocalOAuthRefreshError, AdminRequestContext,
AdminRouteRequest, AdminRouteResponse, AdminRouteResult, AdminStatsTimeRange,
AdminStatsUsageFilter, OAUTH_ACCOUNT_BLOCK_PREFIX, OAUTH_REQUEST_FAILED_PREFIX,
store_admin_provider_ops_balance_cache, update_existing_provider_oauth_catalog_key,
AdminAppState, AdminGatewayProviderTransportSnapshot, AdminLocalOAuthRefreshError,
AdminRequestContext, AdminRouteRequest, AdminRouteResponse, AdminRouteResult,
AdminStatsTimeRange, AdminStatsUsageFilter, OAUTH_ACCOUNT_BLOCK_PREFIX,
OAUTH_REQUEST_FAILED_PREFIX,
};
use crate::handlers::admin::{
@@ -28,6 +28,12 @@ pub(crate) fn maybe_normalize_provider_private_sync_report_payload(
let mut normalized = payload.clone();
normalized.report_context = normalize_provider_private_report_context(Some(report_context));
if let (Some(body_json), Some(context)) = (
payload.body_json.as_ref(),
normalized.report_context.as_mut(),
) {
maybe_attach_gemini_cli_v1internal_credits_context(report_context, body_json, context);
}
if let Some(body_json) = payload.body_json.clone() {
normalized.body_json = normalize_provider_private_response_value(body_json, report_context);
@@ -55,6 +61,44 @@ pub(crate) fn maybe_normalize_provider_private_sync_report_payload(
Ok(Some(normalized))
}
fn maybe_attach_gemini_cli_v1internal_credits_context(
original_report_context: &Value,
body_json: &Value,
normalized_report_context: &mut Value,
) {
if !original_report_context
.get("envelope_name")
.and_then(Value::as_str)
.is_some_and(|value| value.eq_ignore_ascii_case("gemini_cli:v1internal"))
{
return;
}
let mut credits = serde_json::Map::new();
for (source, target) in [
("remainingCredits", "remainingCredits"),
("consumedCredits", "consumedCredits"),
("traceId", "traceId"),
] {
if let Some(value) = body_json
.get(source)
.cloned()
.filter(|value| !value.is_null())
{
credits.insert(target.to_string(), value);
}
}
if credits.is_empty() {
return;
}
if let Some(object) = normalized_report_context.as_object_mut() {
object.insert(
"gemini_cli_v1internal_credits".to_string(),
Value::Object(credits),
);
}
}
fn normalize_provider_private_stream_bytes(
report_context: &Value,
body: &[u8],
+161 -35
View File
@@ -41,62 +41,82 @@ pub(crate) use crate::ai_serving::{
AiExecutionDecision, AiExecutionPlanPayload, AiStreamAttempt, AiSyncAttempt,
};
pub(crate) use aether_ai_formats::api::{
build_core_error_body_for_client_format, core_error_background_report_kind,
core_error_default_client_api_format, core_success_background_report_kind,
encode_kiro_sse_events, implicit_sync_finalize_report_kind, is_core_error_finalize_kind,
normalize_provider_private_report_context, normalize_provider_private_response_value,
provider_private_response_allows_sync_finalize, resolve_claude_stream_spec,
resolve_claude_sync_spec, resolve_gemini_stream_spec, resolve_gemini_sync_spec,
resolve_local_image_stream_spec, resolve_local_image_sync_spec,
build_core_error_body_for_client_format, convert_standard_chat_response,
core_error_background_report_kind, core_error_default_client_api_format,
core_success_background_report_kind, encode_kiro_sse_events,
extract_provider_private_stream_error_body, implicit_sync_finalize_report_kind,
is_core_error_finalize_kind, normalize_provider_private_report_context,
normalize_provider_private_response_value, provider_private_response_allows_sync_finalize,
resolve_claude_stream_spec, resolve_claude_sync_spec, resolve_gemini_stream_spec,
resolve_gemini_sync_spec, resolve_local_image_stream_spec, resolve_local_image_sync_spec,
resolve_local_same_format_stream_spec, resolve_local_same_format_sync_spec,
AiControlPlanRequest, ExecutionRuntimeAuthContext, LocalCoreSyncErrorKind,
LocalOpenAiImageSpec, LocalSameFormatProviderFamily, LocalSameFormatProviderSpec,
LocalStandardSourceFamily, LocalStandardSourceMode, LocalStandardSpec,
StreamingStandardTerminalObserver, EXECUTION_RUNTIME_STREAM_DECISION_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_FILES_DOWNLOAD_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND,
OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
resolve_openai_embedding_sync_spec, sanitize_request_path_and_query, AiControlPlanRequest,
CanonicalContentPart, CanonicalStreamEvent, CanonicalStreamFrame, ClaudeClientEmitter,
ExecutionRuntimeAuthContext, LocalCoreSyncErrorKind, LocalOpenAiImageSpec,
LocalSameFormatProviderFamily, LocalSameFormatProviderSpec, LocalStandardSourceFamily,
LocalStandardSourceMode, LocalStandardSpec, OpenAIChatClientEmitter,
OpenAIResponsesClientEmitter, StreamingStandardTerminalObserver, CLAUDE_CHAT_STREAM_PLAN_KIND,
CLAUDE_CLI_STREAM_PLAN_KIND, EXECUTION_RUNTIME_STREAM_DECISION_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_CHAT_STREAM_PLAN_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND,
OPENAI_IMAGE_SYNC_FINALIZE_REPORT_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_STREAM_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_formats::protocol::stream::CanonicalUsage as StreamingCanonicalUsage;
pub(crate) use aether_ai_formats::CODEX_RESPONSES_LITE_HEADER;
pub(crate) fn parse_direct_request_body(
parts: &http::request::Parts,
body_bytes: &axum::body::Bytes,
) -> Option<(serde_json::Value, Option<String>)> {
aether_ai_formats::api::parse_direct_request_body(
is_json_request(&parts.headers),
body_bytes.as_ref(),
)
let is_json_request = is_json_request(&parts.headers);
let body_bytes = if is_json_request {
crate::ai_serving::decoded_request_body_bytes(&parts.headers, body_bytes.as_ref()).ok()?
} else {
std::borrow::Cow::Borrowed(body_bytes.as_ref())
};
aether_ai_formats::api::parse_direct_request_body(is_json_request, body_bytes.as_ref())
}
pub(crate) fn resolve_execution_runtime_stream_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
aether_ai_formats::api::resolve_execution_runtime_stream_plan_kind(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
let plan_kind =
aether_ai_formats::api::resolve_execution_runtime_stream_plan_kind_with_client_surface(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, true, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn resolve_execution_runtime_sync_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
aether_ai_formats::api::resolve_execution_runtime_sync_plan_kind(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
let plan_kind =
aether_ai_formats::api::resolve_execution_runtime_sync_plan_kind_with_client_surface(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, false, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn is_matching_stream_request(
@@ -133,3 +153,109 @@ pub(crate) fn aggregate_claude_stream_sync_response(body: &[u8]) -> Option<serde
pub(crate) fn aggregate_gemini_stream_sync_response(body: &[u8]) -> Option<serde_json::Value> {
aether_ai_formats::api::aggregate_gemini_stream_sync_response(body)
}
pub(crate) fn gemini_generate_content_response_has_visible_output(
body: &serde_json::Value,
) -> bool {
if aether_ai_formats::formats::gemini::generate_content::response::from_raw(body).is_some() {
return true;
}
openai_chat_response_has_visible_output(body) || openai_responses_body_has_visible_output(body)
}
fn openai_chat_response_has_visible_output(body: &serde_json::Value) -> bool {
body.get("choices")
.and_then(serde_json::Value::as_array)
.is_some_and(|choices| {
choices.iter().any(|choice| {
choice
.get("message")
.or_else(|| choice.get("delta"))
.is_some_and(message_like_value_has_visible_output)
|| value_has_non_empty_text(choice.get("text"))
|| choice
.get("finish_reason")
.and_then(serde_json::Value::as_str)
.is_some_and(|value| !value.trim().is_empty() && value != "length")
})
})
}
fn openai_responses_body_has_visible_output(body: &serde_json::Value) -> bool {
body.get("output")
.and_then(serde_json::Value::as_array)
.is_some_and(|items| {
items.iter().any(|item| {
item.get("type")
.and_then(serde_json::Value::as_str)
.is_some_and(|kind| matches!(kind, "function_call" | "image_generation_call"))
|| item
.get("content")
.and_then(serde_json::Value::as_array)
.is_some_and(|content| {
content.iter().any(response_content_has_visible_output)
})
})
})
}
fn message_like_value_has_visible_output(value: &serde_json::Value) -> bool {
value_has_non_empty_text(value.get("content"))
|| value
.get("tool_calls")
.and_then(serde_json::Value::as_array)
.is_some_and(|items| !items.is_empty())
}
fn response_content_has_visible_output(value: &serde_json::Value) -> bool {
value_has_non_empty_text(value.get("text"))
|| value
.get("type")
.and_then(serde_json::Value::as_str)
.is_some_and(|kind| matches!(kind, "function_call" | "output_image"))
}
fn value_has_non_empty_text(value: Option<&serde_json::Value>) -> bool {
match value {
Some(serde_json::Value::String(text)) => !text.trim().is_empty(),
Some(serde_json::Value::Array(items)) => items.iter().any(|item| {
value_has_non_empty_text(item.get("text"))
|| value_has_non_empty_text(item.get("content"))
|| item
.get("type")
.and_then(serde_json::Value::as_str)
.is_some_and(|kind| matches!(kind, "image_url" | "input_image"))
}),
_ => false,
}
}
#[cfg(test)]
mod tests {
use super::parse_direct_request_body;
use axum::body::Bytes;
use axum::http::{header, Request};
#[test]
fn parse_direct_request_body_reads_zstd_encoded_json_body() {
let (parts, _) = Request::builder()
.method("POST")
.uri("/v1/responses")
.header(header::CONTENT_TYPE, "application/json")
.header(header::CONTENT_ENCODING, "zstd")
.body(())
.expect("request should build")
.into_parts();
let encoded =
zstd::stream::encode_all(br#"{"model":"gpt-5.4","stream":true}"#.as_slice(), 0)
.expect("zstd body should encode");
let (body_json, body_base64) =
parse_direct_request_body(&parts, &Bytes::from(encoded)).expect("body should parse");
assert_eq!(body_json["model"].as_str(), Some("gpt-5.4"));
assert_eq!(body_json["stream"].as_bool(), Some(true));
assert!(body_base64.is_none());
}
}
@@ -2,6 +2,7 @@ use serde_json::Value;
use crate::ai_serving::{
maybe_build_ai_surface_stream_rewriter, AiSurfaceFinalizeError, AiSurfaceStreamRewriter,
ResponseHistoryRecord,
};
use crate::GatewayError;
@@ -24,6 +25,10 @@ impl LocalStreamRewriter<'_> {
pub(crate) fn finish(&mut self) -> Result<Vec<u8>, GatewayError> {
self.inner.finish().map_err(map_surface_error)
}
pub(crate) fn take_response_history_record(&mut self) -> Option<ResponseHistoryRecord> {
self.inner.take_response_history_record()
}
}
fn map_surface_error(error: AiSurfaceFinalizeError) -> GatewayError {
@@ -8,6 +8,45 @@ fn utf8(bytes: Vec<u8>) -> String {
String::from_utf8(bytes).expect("utf8 should decode")
}
#[test]
fn same_format_claude_local_stream_rewriter_sanitizes_read_input_json_delta() {
let report_context = json!({
"provider_api_format": "claude:messages",
"client_api_format": "claude:messages",
"anthropic_compatibility_profile": "claude_code_legacy",
"needs_conversion": false,
});
let mut rewriter =
maybe_build_local_stream_rewriter(Some(&report_context)).expect("rewriter should exist");
let mut output = rewriter
.push_chunk(
b"event: content_block_start\n\
data: {\"type\":\"content_block_start\",\"index\":0,\"content_block\":{\"type\":\"tool_use\",\"id\":\"call_read_1\",\"name\":\"Read\",\"input\":{}}}\n\n",
)
.expect("start should be accepted");
output.extend(
rewriter
.push_chunk(
b"event: content_block_delta\n\
data: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"input_json_delta\",\"partial_json\":\"{\\\"file_path\\\":\\\"/tmp/a.txt\\\",\\\"pages\\\":\\\"\\\"}\"}}\n\n",
)
.expect("delta should be accepted"),
);
output.extend(
rewriter
.push_chunk(
b"event: content_block_stop\n\
data: {\"type\":\"content_block_stop\",\"index\":0}\n\n",
)
.expect("stop should flush sanitized delta"),
);
let output_text = utf8(output);
assert!(output_text.contains("\"name\":\"Read\""));
assert!(output_text.contains("\\\"file_path\\\":\\\"/tmp/a.txt\\\""));
assert!(!output_text.contains("\\\"pages\\\":\\\"\\\""));
}
#[test]
fn standard_sync_bridge_converts_openai_chat_sync_json_to_openai_chat_sse() {
let outcome = maybe_bridge_standard_sync_json_to_stream(
@@ -25,12 +25,16 @@ fn test_decision() -> GatewayControlDecision {
route_class: Some("ai_public".to_string()),
route_family: Some("openai".to_string()),
route_kind: Some("compact".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("openai:responses:compact".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
}
}
@@ -172,6 +176,9 @@ fn aggregates_openai_responses_stream_completed_event_to_final_response() {
let result = aggregate_openai_responses_stream_sync_response(body.as_bytes())
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -180,6 +187,9 @@ fn aggregates_openai_responses_stream_completed_event_to_final_response() {
"object": "response",
"model": "gpt-5",
"status": "completed",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Hello",
"output": [{
"type": "message",
"id": "resp_123_msg",
@@ -215,6 +225,9 @@ fn aggregates_openai_responses_stream_tool_call_events_to_final_response() {
let result = aggregate_openai_responses_stream_sync_response(body.as_bytes())
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -223,6 +236,9 @@ fn aggregates_openai_responses_stream_tool_call_events_to_final_response() {
"object": "response",
"model": "gpt-5",
"status": "completed",
"created_at": created_at,
"completed_at": created_at,
"output_text": "",
"output": [{
"type": "function_call",
"id": "call_123",
@@ -811,6 +827,9 @@ fn converts_claude_cli_response_to_openai_responses_response() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -819,6 +838,9 @@ fn converts_claude_cli_response_to_openai_responses_response() {
"object": "response",
"status": "completed",
"model": "claude-code-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Hello Claude CLI",
"output": [{
"type": "message",
"id": "msg_cli_123_msg",
@@ -868,6 +890,9 @@ fn converts_claude_cli_tool_use_to_openai_responses_function_call() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -876,6 +901,9 @@ fn converts_claude_cli_tool_use_to_openai_responses_function_call() {
"object": "response",
"status": "completed",
"model": "claude-code-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Running tool.",
"output": [
{
"type": "message",
@@ -933,6 +961,9 @@ fn converts_gemini_cli_response_to_openai_responses_response() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -941,6 +972,9 @@ fn converts_gemini_cli_response_to_openai_responses_response() {
"object": "response",
"status": "completed",
"model": "gemini-cli-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Hello Gemini CLI",
"output": [{
"type": "message",
"id": "resp_cli_123_msg",
@@ -995,6 +1029,9 @@ fn converts_gemini_cli_function_call_to_openai_responses_function_call() {
}),
)
.expect("result should exist");
let created_at = result["created_at"]
.as_i64()
.expect("created_at should be a unix timestamp");
assert_eq!(
result,
@@ -1003,6 +1040,9 @@ fn converts_gemini_cli_function_call_to_openai_responses_function_call() {
"object": "response",
"status": "completed",
"model": "gemini-cli-upstream",
"created_at": created_at,
"completed_at": created_at,
"output_text": "Need a tool.",
"output": [
{
"type": "message",
@@ -1130,7 +1170,7 @@ fn local_finalize_handles_openai_responses_compact_cross_format_sync_response()
assert_eq!(report.report_kind, "openai_responses_compact_sync_success");
assert_eq!(
report.client_body_json.expect("client body should exist")["object"],
"response"
"response.compaction"
);
}
@@ -1186,7 +1226,7 @@ fn local_finalize_handles_openai_responses_compact_cross_format_function_call_re
.background_report
.expect("compact tool-call should downgrade to success report");
let client_body = report.client_body_json.expect("client body should exist");
assert_eq!(client_body["object"], "response");
assert_eq!(client_body["object"], "response.compaction");
assert_eq!(client_body["output"][1]["type"], "function_call");
}
@@ -1403,6 +1443,57 @@ fn local_finalize_handles_openai_responses_cross_format_stream_response_from_gem
);
}
#[test]
fn local_finalize_rejects_antigravity_usage_only_gemini_wrapper() {
let payload = GatewaySyncReportRequest {
trace_id: "trace-antigravity-empty-gemini-wrapper".to_string(),
report_kind: "gemini_chat_sync_finalize".to_string(),
report_context: Some(json!({
"client_api_format": "gemini:generate_content",
"provider_api_format": "gemini:generate_content",
"model": "gemini-3.5-flash",
"mapped_model": "gemini-3-flash-agent",
"needs_conversion": false,
"has_envelope": true,
"envelope_name": "antigravity:v1internal",
"upstream_is_stream": true,
})),
status_code: 200,
headers: BTreeMap::from([("content-type".to_string(), "application/json".to_string())]),
body_json: Some(json!({
"chunks": [{
"response": {
"responseId": "resp-usage-only",
"modelVersion": "gemini-3-flash-agent",
"usageMetadata": {
"promptTokenCount": 5528,
"totalTokenCount": 5528
}
},
"metadata": {},
"traceId": "trace-antigravity-empty-gemini-wrapper"
}],
"metadata": {
"stream": true,
"stored_chunks": 1,
"total_chunks": 1
}
})),
client_body_json: None,
body_base64: None,
telemetry: None,
};
let outcome = maybe_build_local_core_sync_finalize_response(
"trace-antigravity-empty-gemini-wrapper",
&test_decision(),
&payload,
)
.expect("local finalize should evaluate payload");
assert!(outcome.is_none());
}
#[test]
fn local_finalize_handles_openai_responses_compact_openai_family_stream_response_even_when_conversion_flagged(
) {
@@ -1702,6 +1793,93 @@ fn local_finalize_handles_openai_chat_cross_format_sync_response_from_openai_res
assert_eq!(client_body["usage"]["total_tokens"], 5);
}
#[test]
fn local_finalize_aggregates_openai_responses_capture_envelope_before_chat_conversion() {
let payload = GatewaySyncReportRequest {
trace_id: "trace-openai-chat-capture-envelope-sync-123".to_string(),
report_kind: "openai_chat_sync_finalize".to_string(),
report_context: Some(json!({
"client_api_format": "openai:chat",
"provider_api_format": "openai:responses",
"model": "gpt-5.6-luna",
"mapped_model": "gpt-5.6-luna",
"needs_conversion": true,
"has_envelope": false,
})),
status_code: 200,
headers: BTreeMap::from([("content-type".to_string(), "application/json".to_string())]),
body_json: Some(json!({
"chunks": [
{
"type": "response.output_text.delta",
"response_id": "resp_capture_gateway_123",
"output_index": 0,
"content_index": 0,
"delta": "Gateway "
},
{
"type": "response.output_text.done",
"response_id": "resp_capture_gateway_123",
"output_index": 0,
"content_index": 0,
"text": "Gateway capture"
},
{
"type": "response.completed",
"response": {
"id": "resp_capture_gateway_123",
"object": "response",
"status": "completed",
"model": "gpt-5.6-luna",
"output": [],
"usage": {
"input_tokens": 2,
"output_tokens": 3,
"total_tokens": 5
}
}
}
],
"metadata": {}
})),
client_body_json: None,
body_base64: None,
telemetry: None,
};
let outcome = maybe_build_local_core_sync_finalize_response(
"trace-openai-chat-capture-envelope-sync-123",
&test_decision(),
&payload,
)
.expect("capture envelope finalize should succeed")
.expect("capture envelope finalize should match");
let report = outcome
.background_report
.expect("capture envelope conversion should produce a success report");
assert_eq!(report.report_kind, "openai_chat_sync_success");
let provider_body = report
.body_json
.expect("aggregated provider body should exist");
assert_eq!(provider_body["id"], "resp_capture_gateway_123");
assert_eq!(
provider_body["output"][0]["content"][0]["text"],
"Gateway capture"
);
assert!(provider_body.get("chunks").is_none());
let client_body = report
.client_body_json
.expect("converted client body should exist");
assert_eq!(
client_body["choices"][0]["message"]["content"],
"Gateway capture"
);
assert_eq!(client_body["usage"]["prompt_tokens"], 2);
assert_eq!(client_body["usage"]["completion_tokens"], 3);
assert_eq!(client_body["usage"]["total_tokens"], 5);
}
#[test]
fn local_finalize_handles_claude_chat_cross_format_sync_response_from_openai_chat() {
let payload = GatewaySyncReportRequest {
@@ -1748,12 +1926,16 @@ fn local_finalize_handles_claude_chat_cross_format_sync_response_from_openai_cha
route_class: Some("ai_public".to_string()),
route_family: Some("claude".to_string()),
route_kind: Some("chat".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("claude:messages".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
},
&payload,
)
@@ -1815,12 +1997,16 @@ fn local_finalize_handles_gemini_cli_cross_format_sync_response_from_claude_cli(
route_class: Some("ai_public".to_string()),
route_family: Some("gemini".to_string()),
route_kind: Some("cli".to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_endpoint_signature: Some("gemini:generate_content".to_string()),
execution_runtime_candidate: true,
auth_context: None,
admin_principal: None,
local_auth_rejection: None,
model_directive_policy: Default::default(),
},
&payload,
)
+58 -12
View File
@@ -3,6 +3,7 @@ pub(crate) mod api;
mod finalize;
mod planner;
mod pure;
mod response_history;
pub(crate) mod transport;
use axum::body::Body;
@@ -13,6 +14,9 @@ use crate::{usage::GatewaySyncReportRequest, AppState, GatewayError};
pub(crate) use self::adaptation::{
maybe_build_provider_private_stream_normalizer, ProviderPrivateStreamNormalizer,
};
pub(crate) use self::api::{
gemini_generate_content_response_has_visible_output, CODEX_RESPONSES_LITE_HEADER,
};
pub(crate) use self::finalize::common::LocalCoreSyncFinalizeOutcome;
pub(crate) use self::finalize::internal::{
maybe_bridge_standard_sync_json_to_stream, maybe_build_stream_response_rewriter,
@@ -47,26 +51,33 @@ pub(crate) use self::planner::{
build_standard_family_stream_plan_and_reports, build_standard_family_sync_attempt_source,
build_standard_family_sync_plan_and_reports, build_standard_stream_plan_from_decision,
build_standard_sync_plan_from_decision, candidate_auth_channel_skip_reason,
extract_pool_sticky_session_token, maybe_build_stream_decision_payload,
maybe_build_stream_plan_payload, maybe_build_sync_decision_payload,
maybe_build_sync_plan_payload, planner_is_matching_stream_request, provider_key_pool_score_id,
provider_key_pool_score_scope, read_candidate_transport_snapshot,
record_local_runtime_candidate_skip_reason,
codex_model_capabilities_for_transport, extract_pool_sticky_session_token,
maybe_build_stream_decision_payload, maybe_build_stream_plan_payload,
maybe_build_sync_decision_payload, maybe_build_sync_plan_payload,
planner_is_matching_stream_request, provider_key_pool_score_id, provider_key_pool_score_scope,
read_candidate_transport_snapshot, record_local_runtime_candidate_skip_reason,
resolve_tunnel_scheduler_affinity_context, resolve_upstream_is_stream_for_provider,
set_local_openai_chat_execution_exhausted_diagnostic,
set_local_openai_image_execution_exhausted_diagnostic, CandidateFailureDiagnostic,
CandidateFailureDiagnosticKind, EligibleLocalExecutionCandidate, GatewayAuthApiKeySnapshot,
GatewayProviderTransportSnapshot, LocalExecutionAttemptSource, LocalExecutionCandidateKind,
LocalResolvedOAuthRequestAuth, PlannerAppState, SkippedLocalExecutionCandidate,
set_local_openai_image_execution_exhausted_diagnostic, validate_final_openai_provider_request,
CandidateFailureDiagnostic, CandidateFailureDiagnosticKind, EligibleLocalExecutionCandidate,
GatewayAuthApiKeySnapshot, GatewayProviderTransportSnapshot, LocalExecutionAttemptSource,
LocalExecutionCandidateKind, LocalResolvedOAuthRequestAuth, PlannerAppState,
SkippedLocalExecutionCandidate,
};
pub(crate) use self::pure::*;
pub(crate) use self::response_history::{
hydrate_openai_response_history, persist_converted_response_history,
persist_response_history_record,
};
pub(crate) use self::transport::{
append_transport_diagnostics_to_value, build_request_trace_proxy_value,
candidate_common_transport_skip_reason, candidate_transport_pair_skip_reason,
request_conversion_direct_auth, request_conversion_enabled_for_transport,
request_conversion_transport_supported, request_conversion_transport_unsupported_reason,
request_pair_allowed_for_transport, CandidateTransportPolicyFacts,
request_pair_allowed_for_transport, request_pair_direct_auth,
request_pair_transport_unsupported_reason, CandidateTransportPolicyFacts,
};
pub(crate) use crate::control::GatewayControlDecision;
pub(crate) use crate::control::{GatewayControlDecision, GatewayCredentialCarrier};
pub(crate) use crate::execution_runtime::{ConversionMode, ExecutionStrategy};
pub(crate) use crate::headers::RequestOrigin;
pub(crate) use aether_ai_serving::{
@@ -84,6 +95,7 @@ pub(crate) fn build_provider_transport_request_url(
upstream_is_stream: bool,
request_query: Option<&str>,
kiro_api_region: Option<&str>,
api_operation: Option<ApiOperation>,
) -> Option<String> {
self::transport::build_transport_request_url(
transport,
@@ -93,10 +105,35 @@ pub(crate) fn build_provider_transport_request_url(
upstream_is_stream,
request_query,
kiro_api_region,
api_operation,
},
)
}
pub(crate) fn build_provider_transport_request_url_for_request_body(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
mapped_model: Option<&str>,
upstream_is_stream: bool,
request_query: Option<&str>,
kiro_api_region: Option<&str>,
api_operation: Option<ApiOperation>,
provider_request_body: Option<&serde_json::Value>,
) -> Option<String> {
self::transport::build_transport_request_url_for_request_body(
transport,
self::transport::TransportRequestUrlParams {
provider_api_format,
mapped_model,
upstream_is_stream,
request_query,
kiro_api_region,
api_operation,
},
provider_request_body,
)
}
pub(crate) async fn resolve_execution_runtime_auth_context(
state: &AppState,
decision: &GatewayControlDecision,
@@ -126,6 +163,13 @@ pub(crate) fn is_json_request(headers: &http::HeaderMap) -> bool {
crate::headers::is_json_request(headers)
}
pub(crate) fn decoded_request_body_bytes<'a>(
headers: &http::HeaderMap,
body_bytes: &'a [u8],
) -> Result<std::borrow::Cow<'a, [u8]>, crate::headers::RequestBodyNormalizationError> {
crate::headers::decoded_request_body_bytes(headers, body_bytes)
}
pub(crate) fn tls_fingerprint_from_headers(headers: &http::HeaderMap) -> Option<serde_json::Value> {
crate::headers::tls_fingerprint_from_headers(headers)
}
@@ -157,7 +201,9 @@ pub(crate) fn resolve_local_decision_execution_runtime_auth_context(
decision: &GatewayControlDecision,
) -> Option<ExecutionRuntimeAuthContext> {
resolve_decision_execution_runtime_auth_context(decision).filter(|auth_context| {
!auth_context.user_id.trim().is_empty() && !auth_context.api_key_id.trim().is_empty()
auth_context.access_allowed
&& !auth_context.user_id.trim().is_empty()
&& !auth_context.api_key_id.trim().is_empty()
})
}
@@ -0,0 +1,170 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use serde_json::Value;
use crate::ai_serving::transport::antigravity::{
build_antigravity_safe_v1internal_request, build_antigravity_static_identity_headers,
classify_local_antigravity_request_support, AntigravityEnvelopeRequestType,
AntigravityRequestAuth, AntigravityRequestAuthUnsupportedReason,
AntigravityRequestEnvelopeSupport, AntigravityRequestSideSupport,
AntigravityRequestSideUnsupportedReason,
};
use crate::ai_serving::transport::{
build_standard_provider_request_headers, GatewayProviderTransportSnapshot,
StandardProviderRequestHeaders, StandardProviderRequestHeadersInput,
};
use crate::AppState;
pub(crate) const ANTIGRAVITY_V1INTERNAL_ENVELOPE_NAME: &str = "antigravity:v1internal";
pub(crate) enum AntigravityV1InternalRequestError {
TransportUnsupported,
EnvelopeUnsupported,
UpstreamUrlUnavailable,
HeaderRulesApplyFailed,
}
pub(crate) struct AntigravityV1InternalRequestInput<'a> {
pub(crate) state: &'a AppState,
pub(crate) parts: &'a http::request::Parts,
pub(crate) transport: &'a Arc<GatewayProviderTransportSnapshot>,
pub(crate) trace_id: &'a str,
pub(crate) mapped_model: &'a str,
pub(crate) provider_api_format: &'a str,
pub(crate) auth_header: &'a str,
pub(crate) auth_value: &'a str,
pub(crate) request_headers: &'a http::HeaderMap,
pub(crate) original_request_body: &'a Value,
pub(crate) gemini_request_body: &'a Value,
pub(crate) upstream_is_stream: bool,
pub(crate) same_format: bool,
}
pub(crate) struct AntigravityV1InternalRequest {
pub(crate) transport: Arc<GatewayProviderTransportSnapshot>,
pub(crate) body: Value,
pub(crate) headers: StandardProviderRequestHeaders,
pub(crate) upstream_url: String,
}
pub(crate) async fn build_antigravity_v1internal_provider_request(
input: AntigravityV1InternalRequestInput<'_>,
) -> Result<AntigravityV1InternalRequest, AntigravityV1InternalRequestError> {
let payload = build_antigravity_v1internal_payload(
input.state,
input.transport,
input.trace_id,
input.mapped_model,
input.gemini_request_body,
)
.await?;
let upstream_url = crate::ai_serving::build_provider_transport_request_url_for_request_body(
&payload.transport,
input.provider_api_format,
Some(input.mapped_model),
input.upstream_is_stream,
input.parts.uri.query(),
None,
None,
Some(&payload.body),
)
.ok_or(AntigravityV1InternalRequestError::UpstreamUrlUnavailable)?;
let extra_headers: BTreeMap<String, String> =
build_antigravity_static_identity_headers(&payload.auth);
let mut headers =
build_standard_provider_request_headers(StandardProviderRequestHeadersInput {
transport: &payload.transport,
provider_api_format: input.provider_api_format,
same_format: input.same_format,
headers: input.request_headers,
auth_header: input.auth_header,
auth_value: input.auth_value,
extra_headers: &extra_headers,
header_rules: payload.transport.endpoint.header_rules.as_ref(),
provider_request_body: &payload.body,
original_request_body: input.original_request_body,
upstream_is_stream: input.upstream_is_stream,
})
.ok_or(AntigravityV1InternalRequestError::HeaderRulesApplyFailed)?;
headers
.headers
.insert("accept".to_string(), "text/event-stream".to_string());
Ok(AntigravityV1InternalRequest {
transport: payload.transport,
body: payload.body,
headers,
upstream_url,
})
}
struct AntigravityV1InternalPayload {
transport: Arc<GatewayProviderTransportSnapshot>,
auth: AntigravityRequestAuth,
body: Value,
}
async fn build_antigravity_v1internal_payload(
state: &AppState,
transport: &Arc<GatewayProviderTransportSnapshot>,
trace_id: &str,
mapped_model: &str,
gemini_request_body: &Value,
) -> Result<AntigravityV1InternalPayload, AntigravityV1InternalRequestError> {
let mut resolved_transport = Arc::clone(transport);
let mut antigravity_support = classify_local_antigravity_request_support(
&resolved_transport,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
);
if matches!(
antigravity_support,
AntigravityRequestSideSupport::Unsupported(
AntigravityRequestSideUnsupportedReason::UnsupportedAuth(
AntigravityRequestAuthUnsupportedReason::MissingProjectId
)
)
) {
if let Some(hydrated) = state
.hydrate_antigravity_project_metadata_for_transport(&resolved_transport)
.await
{
resolved_transport = Arc::new(hydrated);
antigravity_support = classify_local_antigravity_request_support(
&resolved_transport,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
);
}
}
let auth = match antigravity_support {
AntigravityRequestSideSupport::Supported(spec) => spec.auth,
AntigravityRequestSideSupport::Unsupported(_) => {
return Err(AntigravityV1InternalRequestError::TransportUnsupported);
}
};
let body = match build_antigravity_safe_v1internal_request(
&auth,
trace_id,
mapped_model,
gemini_request_body,
AntigravityEnvelopeRequestType::Agent,
) {
AntigravityRequestEnvelopeSupport::Supported(envelope) => envelope,
AntigravityRequestEnvelopeSupport::Unsupported(_) => {
return Err(AntigravityV1InternalRequestError::EnvelopeUnsupported);
}
};
Ok(AntigravityV1InternalPayload {
transport: resolved_transport,
auth,
body,
})
}
@@ -1,6 +1,8 @@
use aether_routing_core::ResolvedRoutingPolicy;
use aether_scheduler_core::{
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session, ClientSessionAffinity,
SchedulerAffinityTarget, SchedulerMinimalCandidateSelectionCandidate,
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope,
ClientSessionAffinity, SchedulerAffinityScope, SchedulerAffinityTarget,
SchedulerMinimalCandidateSelectionCandidate,
};
use crate::ai_serving::{GatewayAuthApiKeySnapshot, PlannerAppState};
@@ -8,25 +10,38 @@ use crate::scheduler::affinity::SCHEDULER_AFFINITY_TTL;
const PLANNER_SCHEDULER_AFFINITY_MAX_ENTRIES: usize = 10_000;
pub(crate) fn has_explicit_session_affinity(
client_session_affinity: Option<&ClientSessionAffinity>,
) -> bool {
client_session_affinity.is_some_and(ClientSessionAffinity::has_session_key)
}
pub(crate) fn read_cached_scheduler_affinity_target(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: Option<&str>,
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Option<SchedulerAffinityTarget> {
if !has_explicit_session_affinity(client_session_affinity) {
return None;
}
let requested_model = requested_model
.map(str::trim)
.filter(|value| !value.is_empty())?;
let api_key_id = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())?;
let cache_key = build_scheduler_affinity_cache_key_for_api_key_id_with_client_session(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
)?;
let affinity_scope = scheduler_affinity_scope_for_routing_policy(routing_policy);
let cache_key =
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
affinity_scope.as_ref(),
)?;
state
.app()
@@ -41,6 +56,9 @@ pub(crate) fn remember_scheduler_affinity_for_candidate(
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) {
if !has_explicit_session_affinity(client_session_affinity) {
return;
}
remember_scheduler_affinity_for_candidate_at_epoch(
state,
auth_snapshot,
@@ -61,18 +79,71 @@ pub(crate) fn remember_scheduler_affinity_for_candidate_at_epoch(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
expected_epoch: Option<u64>,
) {
remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state,
auth_snapshot,
client_session_affinity,
client_api_format,
requested_model,
candidate,
None,
expected_epoch,
);
}
#[allow(clippy::too_many_arguments)]
pub(crate) fn remember_scheduler_affinity_for_candidate_with_routing_policy_at_epoch(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
routing_policy: Option<&ResolvedRoutingPolicy>,
expected_epoch: Option<u64>,
) {
let affinity_scope = scheduler_affinity_scope_for_routing_policy(routing_policy);
remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state,
auth_snapshot,
client_session_affinity,
client_api_format,
requested_model,
candidate,
affinity_scope.as_ref(),
expected_epoch,
);
}
#[allow(clippy::too_many_arguments)]
fn remember_scheduler_affinity_for_candidate_with_scope_at_epoch(
state: PlannerAppState<'_>,
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
client_api_format: &str,
requested_model: &str,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
affinity_scope: Option<&SchedulerAffinityScope>,
expected_epoch: Option<u64>,
) {
if !has_explicit_session_affinity(client_session_affinity) {
return;
}
let Some(api_key_id) = auth_snapshot
.map(|snapshot| snapshot.api_key_id.trim())
.filter(|value| !value.is_empty())
else {
return;
};
let Some(cache_key) = build_scheduler_affinity_cache_key_for_api_key_id_with_client_session(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
) else {
let Some(cache_key) =
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
client_api_format,
requested_model,
client_session_affinity,
affinity_scope,
)
else {
return;
};
@@ -88,3 +159,15 @@ pub(crate) fn remember_scheduler_affinity_for_candidate_at_epoch(
expected_epoch,
);
}
fn scheduler_affinity_scope_for_routing_policy(
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Option<SchedulerAffinityScope> {
let policy = routing_policy?;
let group_id = policy
.group_id
.as_deref()
.map(str::trim)
.filter(|group_id| !group_id.is_empty())?;
Some(SchedulerAffinityScope::new(group_id, policy.group_version))
}
File diff suppressed because it is too large Load Diff
@@ -134,6 +134,7 @@ mod tests {
global_model_id: "global-1".to_string(),
global_model_name: "gpt-5.4".to_string(),
selected_provider_model_name: "gpt-5.4".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -198,6 +199,7 @@ mod tests {
}
}
})),
upstream_metadata: None,
decrypted_api_key: "sk-test".to_string(),
decrypted_auth_config: None,
},
@@ -254,6 +256,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "__placeholder__".to_string(),
decrypted_auth_config: None,
},
@@ -144,6 +144,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: String::new(),
decrypted_auth_config: None,
},
@@ -168,6 +169,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-test".to_string(),
selected_provider_model_name: "gpt-test-upstream".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -1,11 +1,11 @@
use std::collections::BTreeMap;
use aether_ai_serving::{
ai_ranking_context, build_ai_rankable_candidate, run_ai_candidate_ranking,
AiCandidateRankingPort, AiRankableCandidateParts, AiRankingContextConfig,
AiRankingSchedulingMode,
};
use aether_routing_core::{ResolvedRoutingPolicy, RoutingSchedulingMode, RoutingSetPriorityMode};
use async_trait::async_trait;
use tokio::sync::Mutex;
use tracing::warn;
use crate::ai_serving::{GatewayAuthApiKeySnapshot, PlannerAppState};
@@ -16,14 +16,14 @@ use crate::scheduler::config::{
};
use aether_scheduler_core::{
matches_affinity_target, ClientSessionAffinity, SchedulerAffinityTarget,
SchedulerMinimalCandidateSelectionCandidate, SchedulerRankableCandidate,
SchedulerMinimalCandidateSelectionCandidate, SchedulerPriorityMode, SchedulerRankableCandidate,
SchedulerRankingContext, SchedulerRankingOutcome,
};
use super::candidate_affinity_cache::read_cached_scheduler_affinity_target;
use super::candidate_resolution::EligibleLocalExecutionCandidate;
use super::candidate_resolution::{EligibleLocalExecutionCandidate, LocalExecutionCandidateKind};
use super::candidate_transport_ranking_facts::{
resolve_cached_transport_ranking_facts, CandidateTransportRankingFacts,
resolve_cached_transport_ranking_facts, CandidateTransportRankingFactsCache,
};
struct GatewayLocalCandidateRankingPort<'a> {
@@ -33,6 +33,8 @@ struct GatewayLocalCandidateRankingPort<'a> {
client_session_affinity: Option<&'a ClientSessionAffinity>,
required_capabilities: Option<&'a serde_json::Value>,
ordering_config: SchedulerOrderingConfig,
routing_policy: Option<&'a ResolvedRoutingPolicy>,
transport_ranking_facts_cache: Mutex<CandidateTransportRankingFactsCache>,
}
#[async_trait]
@@ -64,6 +66,7 @@ impl AiCandidateRankingPort for GatewayLocalCandidateRankingPort<'_> {
self.client_session_affinity,
normalized_client_api_format,
affinity_requested_model,
self.routing_policy,
))
}
@@ -82,15 +85,21 @@ impl AiCandidateRankingPort for GatewayLocalCandidateRankingPort<'_> {
normalized_client_api_format: &str,
cached_affinity_match: bool,
) -> Result<SchedulerRankableCandidate, Self::Error> {
let ranking_facts = resolve_transport_ranking_facts_for_candidate(
self.state,
&candidate.candidate,
candidate.transport.as_ref(),
self.ordering_config,
)
.await;
let ranking_facts = {
let mut cache = self.transport_ranking_facts_cache.lock().await;
resolve_cached_transport_ranking_facts(
self.state,
&mut cache,
&candidate.candidate,
candidate.transport.as_ref(),
self.ordering_config,
)
.await
};
let routing_overlaid_candidate =
routing_overlaid_candidate(self.routing_policy, candidate.kind, &candidate.candidate);
Ok(build_ai_rankable_candidate(AiRankableCandidateParts {
candidate: &candidate.candidate,
candidate: &routing_overlaid_candidate,
original_index,
normalized_client_api_format,
provider_api_format: candidate.provider_api_format.as_str(),
@@ -122,8 +131,9 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
required_capabilities: Option<&serde_json::Value>,
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> Vec<EligibleLocalExecutionCandidate> {
let ordering_config = read_scheduler_ordering_config_or_default(state).await;
let ordering_config = scheduler_ordering_config_for_routing_policy(state, routing_policy).await;
let port = GatewayLocalCandidateRankingPort {
state,
requested_model,
@@ -131,6 +141,8 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
client_session_affinity,
required_capabilities,
ordering_config,
routing_policy,
transport_ranking_facts_cache: Mutex::new(CandidateTransportRankingFactsCache::default()),
};
match run_ai_candidate_ranking(&port, candidates, normalized_client_api_format).await {
@@ -139,23 +151,6 @@ pub(crate) async fn rank_eligible_local_execution_candidates(
}
}
async fn resolve_transport_ranking_facts_for_candidate(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &crate::ai_serving::GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let mut ordering_cache = BTreeMap::new();
resolve_cached_transport_ranking_facts(
state,
&mut ordering_cache,
candidate,
transport,
ordering_config,
)
.await
}
fn cached_affinity_matches_local_execution_scope(
eligible: &EligibleLocalExecutionCandidate,
target: &SchedulerAffinityTarget,
@@ -189,6 +184,62 @@ fn ai_ranking_scheduling_mode(mode: SchedulerSchedulingMode) -> AiRankingSchedul
}
}
pub(crate) async fn scheduler_ordering_config_for_routing_policy(
state: PlannerAppState<'_>,
routing_policy: Option<&ResolvedRoutingPolicy>,
) -> SchedulerOrderingConfig {
match routing_policy {
Some(policy) => scheduler_ordering_config_from_routing_policy(policy),
None => read_scheduler_ordering_config_or_default(state).await,
}
}
fn scheduler_ordering_config_from_routing_policy(
policy: &ResolvedRoutingPolicy,
) -> SchedulerOrderingConfig {
SchedulerOrderingConfig {
priority_mode: match policy.priority_mode {
RoutingSetPriorityMode::Provider => SchedulerPriorityMode::Provider,
RoutingSetPriorityMode::GlobalKey => SchedulerPriorityMode::GlobalKey,
},
scheduling_mode: match policy.scheduling_mode {
RoutingSchedulingMode::FixedOrder => SchedulerSchedulingMode::FixedOrder,
RoutingSchedulingMode::CacheAffinity => SchedulerSchedulingMode::CacheAffinity,
RoutingSchedulingMode::LoadBalance => SchedulerSchedulingMode::LoadBalance,
},
keep_priority_on_conversion: policy.keep_priority_on_conversion,
}
}
fn routing_overlaid_candidate(
routing_policy: Option<&ResolvedRoutingPolicy>,
kind: LocalExecutionCandidateKind,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> SchedulerMinimalCandidateSelectionCandidate {
let Some(policy) = routing_policy else {
return candidate.clone();
};
let mut overlaid = candidate.clone();
overlaid.provider_priority = policy
.ranking_overlay
.provider_priority(candidate.provider_id.as_str(), candidate.provider_priority);
let overlaid_key_priority = match kind {
LocalExecutionCandidateKind::SingleKey => policy
.ranking_overlay
.key_priority_overrides
.get(candidate.key_id.as_str()),
LocalExecutionCandidateKind::PoolGroup => policy
.ranking_overlay
.pool_priority_overrides
.get(candidate.provider_id.as_str()),
};
if let Some(overlaid_key_priority) = overlaid_key_priority.copied() {
overlaid.key_internal_priority = overlaid_key_priority;
overlaid.key_global_priority_for_format = Some(overlaid_key_priority);
}
overlaid
}
async fn read_scheduler_ordering_config_or_default(
state: PlannerAppState<'_>,
) -> SchedulerOrderingConfig {
@@ -226,7 +277,9 @@ mod tests {
use serde_json::json;
use super::super::candidate_affinity_cache::remember_scheduler_affinity_for_candidate;
use super::super::candidate_transport_ranking_facts::resolve_cached_candidate_transport_ranking_facts;
use super::super::candidate_transport_ranking_facts::{
resolve_cached_candidate_transport_ranking_facts, CandidateTransportRankingFactsCache,
};
use super::{PlannerAppState, SchedulerMinimalCandidateSelectionCandidate};
use crate::ai_serving::planner::candidate_resolution::{
resolve_and_rank_local_execution_candidates,
@@ -248,7 +301,7 @@ mod tests {
let ordering_config = super::read_scheduler_ordering_config_or_default(state).await;
let mut candidates = candidates;
let mut rankables = Vec::with_capacity(candidates.len());
let mut ordering_cache = BTreeMap::new();
let mut ordering_cache = CandidateTransportRankingFactsCache::default();
for (original_index, candidate) in candidates.iter().enumerate() {
let ranking_facts = resolve_cached_candidate_transport_ranking_facts(
@@ -300,10 +353,78 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-4.1".to_string(),
selected_provider_model_name: "gpt-4.1".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
#[test]
fn routing_policy_priorities_fall_back_to_candidate_priorities() {
let mut candidate = sample_candidate("endpoint-1", "key-1");
candidate.provider_priority = 7;
candidate.key_internal_priority = 3;
candidate.key_global_priority_for_format = Some(2);
let policy = aether_routing_core::ResolvedRoutingPolicy {
group_id: Some("group-1".to_string()),
group_version: Some(1),
selection_source: "system_default".to_string(),
requested_model: "gpt-5".to_string(),
resolved_model: "gpt-5".to_string(),
priority_mode: aether_routing_core::RoutingSetPriorityMode::Provider,
scheduling_mode: aether_routing_core::RoutingSchedulingMode::CacheAffinity,
keep_priority_on_conversion: false,
ranking_overlay: aether_routing_core::RankingOverlay::default(),
mutation_plan: Default::default(),
pool_policy_overrides: BTreeMap::new(),
matched_rules: Vec::new(),
};
let overlaid = super::routing_overlaid_candidate(
Some(&policy),
LocalExecutionCandidateKind::SingleKey,
&candidate,
);
assert_eq!(overlaid.provider_priority, 7);
assert_eq!(overlaid.key_internal_priority, 3);
assert_eq!(overlaid.key_global_priority_for_format, Some(2));
}
#[test]
fn routing_policy_uses_pool_priority_for_pool_group_global_key_slot() {
let mut candidate = sample_candidate("endpoint-1", "representative-key");
candidate.provider_priority = 7;
candidate.key_internal_priority = 3;
candidate.key_global_priority_for_format = Some(2);
let policy = aether_routing_core::ResolvedRoutingPolicy {
group_id: Some("group-1".to_string()),
group_version: Some(1),
selection_source: "system_default".to_string(),
requested_model: "gpt-5".to_string(),
resolved_model: "gpt-5".to_string(),
priority_mode: aether_routing_core::RoutingSetPriorityMode::GlobalKey,
scheduling_mode: aether_routing_core::RoutingSchedulingMode::CacheAffinity,
keep_priority_on_conversion: false,
ranking_overlay: aether_routing_core::RankingOverlay {
pool_priority_overrides: BTreeMap::from([("provider-1".to_string(), 4)]),
key_priority_overrides: BTreeMap::from([("representative-key".to_string(), 1)]),
..Default::default()
},
mutation_plan: Default::default(),
pool_policy_overrides: BTreeMap::new(),
matched_rules: Vec::new(),
};
let overlaid = super::routing_overlaid_candidate(
Some(&policy),
LocalExecutionCandidateKind::PoolGroup,
&candidate,
);
assert_eq!(overlaid.key_internal_priority, 4);
assert_eq!(overlaid.key_global_priority_for_format, Some(4));
}
fn sample_provider() -> StoredProviderCatalogProvider {
sample_provider_with_options("provider-1", false, 0)
}
@@ -500,6 +621,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "gpt-4.1".to_string(),
selected_provider_model_name: "gpt-4.1".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -1062,6 +1184,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1141,6 +1264,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1216,6 +1340,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1282,6 +1407,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1364,6 +1490,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1401,6 +1528,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-cached",
"endpoint-cached",
@@ -1412,7 +1540,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1434,6 +1562,7 @@ mod tests {
"openai:chat",
"gpt-4.1",
Some(&auth_snapshot),
Some(&client_session_affinity),
None,
None,
None,
@@ -1515,6 +1644,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1573,6 +1703,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_cross_format = sample_priority_candidate(
"provider-shared",
"endpoint-openai",
@@ -1584,7 +1715,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"claude:messages",
"gpt-4.1",
&cached_cross_format,
@@ -1606,6 +1737,7 @@ mod tests {
"claude:messages",
"gpt-4.1",
Some(&auth_snapshot),
Some(&client_session_affinity),
None,
None,
None,
@@ -1713,6 +1845,7 @@ mod tests {
None,
None,
None,
None,
)
.await;
@@ -1772,6 +1905,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-pool",
"endpoint-pool",
@@ -1783,7 +1917,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1805,6 +1939,7 @@ mod tests {
"openai:chat",
Some("gpt-4.1"),
Some(&auth_snapshot),
Some(&client_session_affinity),
None,
None,
None,
@@ -1864,6 +1999,7 @@ mod tests {
.expect("state should build")
.with_data_state_for_tests(data_state);
let auth_snapshot = sample_auth_snapshot();
let client_session_affinity = ClientSessionAffinity::from_session_key("session-1");
let cached_candidate = sample_priority_candidate(
"provider-pool",
"endpoint-pool",
@@ -1875,7 +2011,7 @@ mod tests {
remember_scheduler_affinity_for_candidate(
PlannerAppState::new(&state),
Some(&auth_snapshot),
None,
Some(&client_session_affinity),
"openai:chat",
"gpt-4.1",
&cached_candidate,
@@ -1897,6 +2033,7 @@ mod tests {
"openai:chat",
Some("gpt-4.1"),
Some(&auth_snapshot),
Some(&client_session_affinity),
None,
None,
None,
@@ -1919,7 +2056,7 @@ mod tests {
}
#[tokio::test]
async fn remembers_scheduler_affinity_for_candidate_using_requested_model_key() {
async fn ignores_scheduler_affinity_without_client_session_scope() {
let state = AppState::new().expect("state should build");
let auth_snapshot = sample_auth_snapshot();
let candidate = sample_candidate("endpoint-1", "key-1");
@@ -1933,15 +2070,12 @@ mod tests {
&candidate,
);
let remembered = state
assert!(state
.read_scheduler_affinity_target(
"scheduler_affinity:api-key-1:openai:chat:gpt-5",
SCHEDULER_AFFINITY_TTL,
)
.expect("affinity target should be cached");
assert_eq!(remembered.provider_id, "provider-1");
assert_eq!(remembered.endpoint_id, "endpoint-1");
assert_eq!(remembered.key_id, "key-1");
.is_none());
}
#[tokio::test]
@@ -4,8 +4,10 @@ use aether_ai_serving::{
run_ai_candidate_resolution, AiCandidateResolutionMode, AiCandidateResolutionPort,
AiCandidateResolutionRequest,
};
use aether_routing_core::ResolvedRoutingPolicy;
use async_trait::async_trait;
use std::convert::Infallible;
use std::time::Instant;
use tracing::warn;
use aether_scheduler_core::{
@@ -19,6 +21,7 @@ use crate::ai_serving::{
PlannerAppState,
};
use crate::orchestration::LocalExecutionCandidateMetadata;
use crate::stage_metrics::observe_gateway_stage_ms;
use super::candidate_ranking::rank_eligible_local_execution_candidates;
@@ -60,13 +63,14 @@ struct GatewayLocalCandidateResolutionPort<'a> {
auth_snapshot: Option<&'a GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&'a ClientSessionAffinity>,
required_capabilities: Option<&'a serde_json::Value>,
routing_policy: Option<&'a ResolvedRoutingPolicy>,
request_auth_channel: Option<&'a str>,
}
#[async_trait]
impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
type Candidate = SchedulerMinimalCandidateSelectionCandidate;
type Transport = GatewayProviderTransportSnapshot;
type Transport = Arc<GatewayProviderTransportSnapshot>;
type Eligible = EligibleLocalExecutionCandidate;
type Skipped = SkippedLocalExecutionCandidate;
type Error = Infallible;
@@ -75,7 +79,12 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
&self,
candidate: &Self::Candidate,
) -> Result<Option<Self::Transport>, Self::Error> {
Ok(read_candidate_transport_snapshot(self.state, candidate).await)
let started_at = Instant::now();
let transport = read_candidate_transport_snapshot_arc(self.state, candidate).await;
let elapsed_ms = started_at.elapsed().as_millis() as u64;
observe_gateway_stage_ms("candidate_transport_snapshot", elapsed_ms);
observe_gateway_stage_ms("candidate_resolution_transport_read", elapsed_ms);
Ok(transport)
}
fn build_missing_transport_skipped_candidate(
@@ -97,6 +106,11 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
transport: &Self::Transport,
requested_model: Option<&str>,
) -> Option<&'static str> {
if let Some(skip_reason) =
routing_policy_candidate_skip_reason(self.routing_policy, candidate, transport)
{
return Some(skip_reason);
}
if provider_transport_uses_pool(transport) {
return pool_group_common_transport_skip_reason(candidate, transport);
}
@@ -132,7 +146,7 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
SkippedLocalExecutionCandidate {
candidate,
skip_reason,
transport: Some(Arc::new(transport)),
transport: Some(transport),
ranking: None,
extra_data: None,
}
@@ -152,7 +166,7 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
EligibleLocalExecutionCandidate {
kind,
candidate,
transport: Arc::new(transport),
transport,
provider_api_format,
orchestration: LocalExecutionCandidateMetadata::default(),
ranking: None,
@@ -164,7 +178,8 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
candidates: Vec<Self::Eligible>,
normalized_client_api_format: &str,
) -> Result<Vec<Self::Eligible>, Self::Error> {
Ok(rank_eligible_local_execution_candidates(
let started_at = Instant::now();
let ranked = rank_eligible_local_execution_candidates(
self.state,
candidates,
normalized_client_api_format,
@@ -172,8 +187,14 @@ impl AiCandidateResolutionPort for GatewayLocalCandidateResolutionPort<'_> {
self.auth_snapshot,
self.client_session_affinity,
self.required_capabilities,
self.routing_policy,
)
.await)
.await;
observe_gateway_stage_ms(
"candidate_resolution_rank",
started_at.elapsed().as_millis() as u64,
);
Ok(ranked)
}
async fn apply_pool_scheduler(
@@ -192,6 +213,7 @@ pub(crate) async fn resolve_and_rank_local_execution_candidates(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
required_capabilities: Option<&serde_json::Value>,
routing_policy: Option<&ResolvedRoutingPolicy>,
_sticky_session_token: Option<&str>,
request_auth_channel: Option<&str>,
) -> (
@@ -207,6 +229,7 @@ pub(crate) async fn resolve_and_rank_local_execution_candidates(
auth_snapshot,
client_session_affinity,
required_capabilities,
routing_policy,
None,
request_auth_channel,
AiCandidateResolutionMode::Standard,
@@ -222,6 +245,7 @@ pub(crate) async fn resolve_and_rank_local_execution_candidates_without_transpor
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
required_capabilities: Option<&serde_json::Value>,
routing_policy: Option<&ResolvedRoutingPolicy>,
_sticky_session_token: Option<&str>,
request_auth_channel: Option<&str>,
) -> (
@@ -237,6 +261,7 @@ pub(crate) async fn resolve_and_rank_local_execution_candidates_without_transpor
auth_snapshot,
client_session_affinity,
required_capabilities,
routing_policy,
None,
request_auth_channel,
AiCandidateResolutionMode::WithoutTransportPairGate,
@@ -252,6 +277,7 @@ pub(crate) async fn resolve_and_rank_logical_local_execution_candidates(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
required_capabilities: Option<&serde_json::Value>,
routing_policy: Option<&ResolvedRoutingPolicy>,
_sticky_session_token: Option<&str>,
request_auth_channel: Option<&str>,
mode: AiCandidateResolutionMode,
@@ -267,6 +293,7 @@ pub(crate) async fn resolve_and_rank_logical_local_execution_candidates(
auth_snapshot,
client_session_affinity,
required_capabilities,
routing_policy,
None,
request_auth_channel,
mode,
@@ -283,6 +310,7 @@ async fn resolve_and_rank_local_execution_candidates_with_mode(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
required_capabilities: Option<&serde_json::Value>,
routing_policy: Option<&ResolvedRoutingPolicy>,
_sticky_session_token: Option<&str>,
request_auth_channel: Option<&str>,
mode: AiCandidateResolutionMode,
@@ -298,6 +326,7 @@ async fn resolve_and_rank_local_execution_candidates_with_mode(
auth_snapshot,
client_session_affinity,
required_capabilities,
routing_policy,
None,
request_auth_channel,
mode,
@@ -315,6 +344,7 @@ async fn resolve_and_rank_local_execution_candidates_with_pool_expansion(
auth_snapshot: Option<&GatewayAuthApiKeySnapshot>,
client_session_affinity: Option<&ClientSessionAffinity>,
required_capabilities: Option<&serde_json::Value>,
routing_policy: Option<&ResolvedRoutingPolicy>,
_sticky_session_token: Option<&str>,
request_auth_channel: Option<&str>,
mode: AiCandidateResolutionMode,
@@ -330,6 +360,7 @@ async fn resolve_and_rank_local_execution_candidates_with_pool_expansion(
auth_snapshot,
client_session_affinity,
required_capabilities,
routing_policy,
request_auth_channel,
};
@@ -340,8 +371,13 @@ async fn resolve_and_rank_local_execution_candidates_with_pool_expansion(
expand_pool_groups,
};
let started_at = Instant::now();
match run_ai_candidate_resolution(&port, candidates, request).await {
Ok(mut outcome) => {
observe_gateway_stage_ms(
"candidate_resolution_core",
started_at.elapsed().as_millis() as u64,
);
for candidate in &mut outcome.eligible_candidates {
candidate.orchestration.scheduler_affinity_epoch = Some(scheduler_affinity_epoch);
}
@@ -369,6 +405,28 @@ fn provider_transport_uses_pool(transport: &GatewayProviderTransportSnapshot) ->
.is_some()
}
fn routing_policy_candidate_skip_reason(
routing_policy: Option<&ResolvedRoutingPolicy>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
) -> Option<&'static str> {
let policy = routing_policy?;
if !policy
.ranking_overlay
.provider_allowed(candidate.provider_id.as_str())
{
return Some("routing_profile_disallowed_provider");
}
if !provider_transport_uses_pool(transport)
&& !policy
.ranking_overlay
.key_allowed(candidate.key_id.as_str())
{
return Some("routing_profile_disallowed_key");
}
None
}
fn pool_group_common_transport_skip_reason(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
@@ -466,8 +524,17 @@ pub(crate) async fn read_candidate_transport_snapshot(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> Option<GatewayProviderTransportSnapshot> {
read_candidate_transport_snapshot_arc(state, candidate)
.await
.map(|transport| (*transport).clone())
}
pub(crate) async fn read_candidate_transport_snapshot_arc(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> Option<Arc<GatewayProviderTransportSnapshot>> {
match state
.read_provider_transport_snapshot(
.read_provider_transport_snapshot_arc(
&candidate.provider_id,
&candidate.endpoint_id,
&candidate.key_id,
@@ -551,6 +618,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "secret".to_string(),
decrypted_auth_config: None,
},
@@ -575,6 +643,7 @@ mod tests {
global_model_id: "global-model-1".to_string(),
global_model_name: "claude-sonnet".to_string(),
selected_provider_model_name: "claude-sonnet".to_string(),
supports_streaming: true,
mapping_matched_model: None,
}
}
File diff suppressed because it is too large Load Diff
@@ -1,8 +1,10 @@
use std::collections::BTreeMap;
use aether_contracts::ProxySnapshot;
use aether_scheduler_core::{
SchedulerMinimalCandidateSelectionCandidate, SchedulerTunnelAffinityBucket,
};
use serde_json::Value;
use tracing::warn;
use crate::ai_serving::{GatewayProviderTransportSnapshot, PlannerAppState};
@@ -10,7 +12,9 @@ use crate::scheduler::config::SchedulerOrderingConfig;
use super::candidate_resolution::read_candidate_transport_snapshot;
pub(super) type CandidateTransportIdentity<'a> = (&'a str, &'a str, &'a str);
const TUNNEL_OWNER_INSTANCE_ID_EXTRA_KEY: &str = "tunnel_owner_instance_id";
pub(super) type CandidateTransportIdentity = (String, String, String);
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(super) struct CandidateTransportRankingFacts {
@@ -18,43 +22,68 @@ pub(super) struct CandidateTransportRankingFacts {
pub(super) keep_priority_on_conversion: bool,
}
pub(super) async fn resolve_cached_candidate_transport_ranking_facts<'a>(
state: PlannerAppState<'_>,
cache: &mut BTreeMap<CandidateTransportIdentity<'a>, CandidateTransportRankingFacts>,
candidate: &'a SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.get(&identity).copied() {
return facts;
}
let facts = resolve_candidate_transport_ranking_facts(state, candidate, ordering_config).await;
cache.insert(identity, facts);
facts
#[derive(Debug, Default)]
pub(super) struct CandidateTransportRankingFactsCache {
candidate_facts: BTreeMap<CandidateTransportIdentity, CandidateTransportRankingFacts>,
configured_proxy_snapshots: BTreeMap<String, Option<ProxySnapshot>>,
system_proxy_snapshot: Option<Option<ProxySnapshot>>,
tunnel_buckets_by_node_id: BTreeMap<String, SchedulerTunnelAffinityBucket>,
}
pub(super) async fn resolve_cached_transport_ranking_facts<'a>(
pub(super) async fn resolve_cached_candidate_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut BTreeMap<CandidateTransportIdentity<'a>, CandidateTransportRankingFacts>,
candidate: &'a SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.get(&identity).copied() {
if let Some(facts) = cache.candidate_facts.get(&identity).copied() {
return facts;
}
let facts =
resolve_candidate_transport_ranking_facts_from_transport(state, transport, ordering_config)
.await;
cache.insert(identity, facts);
resolve_candidate_transport_ranking_facts(state, cache, candidate, ordering_config).await;
cache.candidate_facts.insert(identity, facts);
facts
}
pub(super) async fn resolve_cached_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
transport: &GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
let identity = candidate_transport_identity(candidate);
if let Some(facts) = cache.candidate_facts.get(&identity).copied() {
return facts;
}
let facts = resolve_candidate_transport_ranking_facts_from_transport(
state,
cache,
transport,
ordering_config,
)
.await;
cache.candidate_facts.insert(identity, facts);
facts
}
pub(super) async fn candidate_keeps_priority_on_conversion(
state: PlannerAppState<'_>,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> bool {
let mut cache = CandidateTransportRankingFactsCache::default();
resolve_candidate_transport_ranking_facts(state, &mut cache, candidate, ordering_config)
.await
.keep_priority_on_conversion
}
async fn resolve_candidate_transport_ranking_facts(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
candidate: &SchedulerMinimalCandidateSelectionCandidate,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
@@ -65,17 +94,23 @@ async fn resolve_candidate_transport_ranking_facts(
};
};
resolve_candidate_transport_ranking_facts_from_transport(state, &transport, ordering_config)
.await
resolve_candidate_transport_ranking_facts_from_transport(
state,
cache,
&transport,
ordering_config,
)
.await
}
async fn resolve_candidate_transport_ranking_facts_from_transport(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
ordering_config: SchedulerOrderingConfig,
) -> CandidateTransportRankingFacts {
CandidateTransportRankingFacts {
tunnel_bucket: resolve_tunnel_owner_affinity_from_transport(state, transport).await,
tunnel_bucket: resolve_tunnel_owner_affinity_from_transport(state, cache, transport).await,
keep_priority_on_conversion: ordering_config.keep_priority_on_conversion
|| transport.provider.keep_priority_on_conversion,
}
@@ -83,12 +118,11 @@ async fn resolve_candidate_transport_ranking_facts_from_transport(
async fn resolve_tunnel_owner_affinity_from_transport(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
) -> SchedulerTunnelAffinityBucket {
let Some(proxy) = state
.app()
.resolve_transport_proxy_snapshot_with_tunnel_affinity(transport)
.await
let Some(proxy) =
resolve_transport_proxy_snapshot_with_tunnel_affinity_cached(state, cache, transport).await
else {
return SchedulerTunnelAffinityBucket::Neutral;
};
@@ -104,10 +138,74 @@ async fn resolve_tunnel_owner_affinity_from_transport(
return SchedulerTunnelAffinityBucket::Neutral;
};
if let Some(bucket) = cache.tunnel_buckets_by_node_id.get(node_id).copied() {
return bucket;
}
let bucket = resolve_tunnel_owner_affinity_from_proxy(state, &proxy, node_id).await;
cache
.tunnel_buckets_by_node_id
.insert(node_id.to_string(), bucket);
bucket
}
async fn resolve_transport_proxy_snapshot_with_tunnel_affinity_cached(
state: PlannerAppState<'_>,
cache: &mut CandidateTransportRankingFactsCache,
transport: &GatewayProviderTransportSnapshot,
) -> Option<ProxySnapshot> {
for raw in [
transport.key.proxy.as_ref(),
transport.endpoint.proxy.as_ref(),
transport.provider.proxy.as_ref(),
]
.into_iter()
.flatten()
{
let cache_key = proxy_config_cache_key(raw);
if let Some(snapshot) = cache.configured_proxy_snapshots.get(&cache_key) {
if snapshot.is_some() {
return snapshot.clone();
}
continue;
}
let snapshot = state
.app()
.resolve_configured_proxy_snapshot_with_tunnel_affinity(Some(raw))
.await;
cache
.configured_proxy_snapshots
.insert(cache_key, snapshot.clone());
if snapshot.is_some() {
return snapshot;
}
}
if let Some(snapshot) = cache.system_proxy_snapshot.as_ref() {
return snapshot.clone();
}
let snapshot = state.app().resolve_system_proxy_snapshot().await;
cache.system_proxy_snapshot = Some(snapshot.clone());
snapshot
}
async fn resolve_tunnel_owner_affinity_from_proxy(
state: PlannerAppState<'_>,
proxy: &ProxySnapshot,
node_id: &str,
) -> SchedulerTunnelAffinityBucket {
if state.app().tunnel.has_local_proxy(node_id) {
return SchedulerTunnelAffinityBucket::LocalTunnel;
}
if let Some(owner_instance_id) = proxy_tunnel_owner_instance_id(proxy) {
return if owner_instance_id == state.app().tunnel.local_instance_id() {
SchedulerTunnelAffinityBucket::LocalTunnel
} else {
SchedulerTunnelAffinityBucket::RemoteTunnel
};
}
match state
.app()
.tunnel
@@ -134,10 +232,25 @@ async fn resolve_tunnel_owner_affinity_from_transport(
fn candidate_transport_identity(
candidate: &SchedulerMinimalCandidateSelectionCandidate,
) -> CandidateTransportIdentity<'_> {
) -> CandidateTransportIdentity {
(
candidate.provider_id.as_str(),
candidate.endpoint_id.as_str(),
candidate.key_id.as_str(),
candidate.provider_id.clone(),
candidate.endpoint_id.clone(),
candidate.key_id.clone(),
)
}
fn proxy_config_cache_key(raw: &Value) -> String {
serde_json::to_string(raw).unwrap_or_else(|_| raw.to_string())
}
fn proxy_tunnel_owner_instance_id(proxy: &ProxySnapshot) -> Option<&str> {
proxy
.extra
.as_ref()
.and_then(Value::as_object)
.and_then(|extra| extra.get(TUNNEL_OWNER_INSTANCE_ID_EXTRA_KEY))
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
}
@@ -1,27 +1,30 @@
use axum::body::Bytes;
use crate::ai_serving::is_json_request;
use crate::ai_serving::{
endpoint_config_forces_upstream_stream_policy as endpoint_config_forces_upstream_stream_policy_impl,
enforce_request_body_stream_field as enforce_request_body_stream_field_impl,
force_upstream_streaming_for_provider as force_upstream_streaming_for_provider_impl,
is_json_request, parse_direct_request_body as parse_direct_request_body_impl,
resolve_upstream_is_stream_from_endpoint_config as resolve_upstream_is_stream_from_endpoint_config_impl,
parse_direct_request_body as parse_direct_request_body_impl,
resolve_format_upstream_is_stream_for_provider as resolve_upstream_is_stream_for_provider_impl,
};
pub(crate) use crate::ai_serving::{
CLAUDE_CHAT_STREAM_PLAN_KIND, CLAUDE_CHAT_SYNC_PLAN_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, EXECUTION_RUNTIME_STREAM_ACTION,
CLAUDE_CLI_SYNC_PLAN_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND, EXECUTION_RUNTIME_STREAM_ACTION,
EXECUTION_RUNTIME_STREAM_DECISION_ACTION, EXECUTION_RUNTIME_SYNC_ACTION,
EXECUTION_RUNTIME_SYNC_DECISION_ACTION, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
GEMINI_FILES_DELETE_PLAN_KIND, GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND,
GEMINI_FILES_LIST_PLAN_KIND, GEMINI_FILES_UPLOAD_PLAN_KIND, GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND,
GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DELETE_PLAN_KIND,
GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND, GEMINI_FILES_LIST_PLAN_KIND,
GEMINI_FILES_UPLOAD_PLAN_KIND, GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND,
GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_SYNC_PLAN_KIND,
OPENAI_EMBEDDING_SYNC_PLAN_KIND, OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND,
OPENAI_RERANK_SYNC_PLAN_KIND, OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND, OPENAI_RESPONSES_STREAM_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND, OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
OPENAI_RESPONSES_SYNC_PLAN_KIND, OPENAI_SEARCH_SYNC_PLAN_KIND,
OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_VIDEO_CONTENT_PLAN_KIND,
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
pub(crate) use aether_ai_serving::AiRequestedModelFamily as RequestedModelFamily;
@@ -30,7 +33,13 @@ pub(crate) fn parse_direct_request_body(
parts: &http::request::Parts,
body_bytes: &Bytes,
) -> Option<(serde_json::Value, Option<String>)> {
parse_direct_request_body_impl(is_json_request(&parts.headers), body_bytes.as_ref())
let is_json_request = is_json_request(&parts.headers);
let body_bytes = if is_json_request {
crate::ai_serving::decoded_request_body_bytes(&parts.headers, body_bytes.as_ref()).ok()?
} else {
std::borrow::Cow::Borrowed(body_bytes.as_ref())
};
parse_direct_request_body_impl(is_json_request, body_bytes.as_ref())
}
pub(crate) fn force_upstream_streaming_for_provider(
@@ -47,10 +56,10 @@ pub(crate) fn resolve_upstream_is_stream_for_provider(
client_is_stream: bool,
hard_requires_streaming: bool,
) -> bool {
let hard_requires_streaming = hard_requires_streaming
|| force_upstream_streaming_for_provider(provider_type, provider_api_format);
resolve_upstream_is_stream_from_endpoint_config_impl(
resolve_upstream_is_stream_for_provider_impl(
endpoint_config,
provider_type,
provider_api_format,
client_is_stream,
hard_requires_streaming,
)
@@ -171,6 +180,27 @@ mod tests {
true,
false,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"codex",
"openai:image",
true,
true,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"codex",
"openai:responses:compact",
true,
true,
));
assert!(!resolve_upstream_is_stream_for_provider(
Some(&json!({"upstream_stream_policy": "force_stream"})),
"custom",
"openai:responses:compact",
true,
true,
));
}
#[test]
@@ -1,16 +1,17 @@
use crate::ai_serving::planner::common::{
CLAUDE_CHAT_STREAM_PLAN_KIND, CLAUDE_CHAT_SYNC_PLAN_KIND, CLAUDE_CLI_STREAM_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, GEMINI_CHAT_STREAM_PLAN_KIND, GEMINI_CHAT_SYNC_PLAN_KIND,
GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND, GEMINI_FILES_DELETE_PLAN_KIND,
CLAUDE_CLI_SYNC_PLAN_KIND, CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND, GEMINI_CHAT_STREAM_PLAN_KIND,
GEMINI_CHAT_SYNC_PLAN_KIND, GEMINI_CLI_STREAM_PLAN_KIND, GEMINI_CLI_SYNC_PLAN_KIND,
GEMINI_EMBEDDING_SYNC_PLAN_KIND, GEMINI_FILES_DELETE_PLAN_KIND,
GEMINI_FILES_DOWNLOAD_PLAN_KIND, GEMINI_FILES_GET_PLAN_KIND, GEMINI_FILES_LIST_PLAN_KIND,
GEMINI_VIDEO_CANCEL_SYNC_PLAN_KIND, GEMINI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_CHAT_STREAM_PLAN_KIND, OPENAI_CHAT_SYNC_PLAN_KIND, OPENAI_EMBEDDING_SYNC_PLAN_KIND,
OPENAI_IMAGE_STREAM_PLAN_KIND, OPENAI_IMAGE_SYNC_PLAN_KIND, OPENAI_RERANK_SYNC_PLAN_KIND,
OPENAI_RESPONSES_COMPACT_STREAM_PLAN_KIND, OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND,
OPENAI_RESPONSES_STREAM_PLAN_KIND, OPENAI_RESPONSES_SYNC_PLAN_KIND,
OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND, OPENAI_VIDEO_CONTENT_PLAN_KIND,
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND, OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND,
OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
OPENAI_SEARCH_SYNC_PLAN_KIND, OPENAI_VIDEO_CANCEL_SYNC_PLAN_KIND,
OPENAI_VIDEO_CONTENT_PLAN_KIND, OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND,
OPENAI_VIDEO_DELETE_SYNC_PLAN_KIND, OPENAI_VIDEO_REMIX_SYNC_PLAN_KIND,
};
use crate::ai_serving::planner::plan_builders::{
build_gemini_stream_plan_from_decision, build_gemini_sync_plan_from_decision,
@@ -100,17 +101,22 @@ fn build_sync_plan_payload_from_decision(
OPENAI_RESPONSES_SYNC_PLAN_KIND => {
build_openai_responses_sync_plan_from_decision(parts, body_json, payload, false)?
}
OPENAI_IMAGE_SYNC_PLAN_KIND => build_passthrough_sync_plan_from_decision(parts, payload)?,
OPENAI_IMAGE_SYNC_PLAN_KIND | OPENAI_SEARCH_SYNC_PLAN_KIND => {
build_passthrough_sync_plan_from_decision(parts, payload)?
}
OPENAI_RESPONSES_COMPACT_SYNC_PLAN_KIND => {
build_openai_responses_sync_plan_from_decision(parts, body_json, payload, true)?
}
CLAUDE_CHAT_SYNC_PLAN_KIND
| CLAUDE_CLI_SYNC_PLAN_KIND
| CLAUDE_COUNT_TOKENS_SYNC_PLAN_KIND
| OPENAI_EMBEDDING_SYNC_PLAN_KIND
| OPENAI_RERANK_SYNC_PLAN_KIND => {
build_standard_sync_plan_from_decision(parts, body_json, payload)?
}
GEMINI_CHAT_SYNC_PLAN_KIND | GEMINI_CLI_SYNC_PLAN_KIND => {
GEMINI_CHAT_SYNC_PLAN_KIND
| GEMINI_CLI_SYNC_PLAN_KIND
| GEMINI_EMBEDDING_SYNC_PLAN_KIND => {
build_gemini_sync_plan_from_decision(parts, body_json, payload)?
}
OPENAI_VIDEO_CREATE_SYNC_PLAN_KIND
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,146 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use serde_json::Value;
use crate::ai_serving::transport::{
build_gemini_cli_v1internal_request, build_standard_provider_request_headers,
GatewayProviderTransportSnapshot, GeminiCliRequestAuth, GeminiCliRequestAuthSupport,
GeminiCliRequestEnvelopeSupport, StandardProviderRequestHeaders,
StandardProviderRequestHeadersInput, GEMINI_CLI_USER_AGENT,
};
use crate::AppState;
pub(crate) enum GeminiCliV1InternalRequestError {
ProjectUnavailable,
EnvelopeUnsupported,
UpstreamUrlUnavailable,
HeaderRulesApplyFailed,
}
pub(crate) struct GeminiCliV1InternalRequestInput<'a> {
pub(crate) state: &'a AppState,
pub(crate) parts: &'a http::request::Parts,
pub(crate) transport: &'a Arc<GatewayProviderTransportSnapshot>,
pub(crate) trace_id: &'a str,
pub(crate) mapped_model: &'a str,
pub(crate) provider_api_format: &'a str,
pub(crate) auth_header: &'a str,
pub(crate) auth_value: &'a str,
pub(crate) request_headers: &'a http::HeaderMap,
pub(crate) original_request_body: &'a Value,
pub(crate) gemini_request_body: &'a Value,
pub(crate) upstream_is_stream: bool,
}
pub(crate) struct GeminiCliV1InternalRequest {
pub(crate) transport: Arc<GatewayProviderTransportSnapshot>,
pub(crate) body: Value,
pub(crate) headers: StandardProviderRequestHeaders,
pub(crate) upstream_url: String,
}
pub(crate) async fn build_gemini_cli_v1internal_provider_request(
input: GeminiCliV1InternalRequestInput<'_>,
) -> Result<GeminiCliV1InternalRequest, GeminiCliV1InternalRequestError> {
let payload = build_gemini_cli_v1internal_payload(
input.state,
input.transport,
input.trace_id,
input.mapped_model,
input.gemini_request_body,
)
.await?;
let upstream_url = crate::ai_serving::build_provider_transport_request_url_for_request_body(
&payload.transport,
input.provider_api_format,
Some(input.mapped_model),
input.upstream_is_stream,
input.parts.uri.query(),
None,
None,
Some(&payload.body),
)
.ok_or(GeminiCliV1InternalRequestError::UpstreamUrlUnavailable)?;
let extra_headers =
BTreeMap::from([("user-agent".to_string(), GEMINI_CLI_USER_AGENT.to_string())]);
let headers = build_standard_provider_request_headers(StandardProviderRequestHeadersInput {
transport: &payload.transport,
provider_api_format: input.provider_api_format,
same_format: false,
headers: input.request_headers,
auth_header: input.auth_header,
auth_value: input.auth_value,
extra_headers: &extra_headers,
header_rules: payload.transport.endpoint.header_rules.as_ref(),
provider_request_body: &payload.body,
original_request_body: input.original_request_body,
upstream_is_stream: input.upstream_is_stream,
})
.ok_or(GeminiCliV1InternalRequestError::HeaderRulesApplyFailed)?;
Ok(GeminiCliV1InternalRequest {
transport: payload.transport,
body: payload.body,
headers,
upstream_url,
})
}
struct GeminiCliV1InternalPayload {
transport: Arc<GatewayProviderTransportSnapshot>,
body: Value,
}
async fn build_gemini_cli_v1internal_payload(
state: &AppState,
transport: &Arc<GatewayProviderTransportSnapshot>,
trace_id: &str,
mapped_model: &str,
gemini_request_body: &Value,
) -> Result<GeminiCliV1InternalPayload, GeminiCliV1InternalRequestError> {
let mut resolved_transport = Arc::clone(transport);
let mut auth = match crate::ai_serving::transport::resolve_local_gemini_cli_request_auth(
&resolved_transport,
) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => {
return Err(GeminiCliV1InternalRequestError::ProjectUnavailable);
}
};
if auth.project_id.is_none() {
auth = match state
.hydrate_gemini_cli_project_metadata_for_transport(&resolved_transport)
.await
{
Some(hydrated) => {
resolved_transport = Arc::new(hydrated);
match crate::ai_serving::transport::resolve_local_gemini_cli_request_auth(
&resolved_transport,
) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => GeminiCliRequestAuth::default(),
}
}
None => GeminiCliRequestAuth::default(),
};
}
let body = match build_gemini_cli_v1internal_request(
&auth,
trace_id,
mapped_model,
gemini_request_body,
) {
GeminiCliRequestEnvelopeSupport::Supported(envelope) => envelope,
GeminiCliRequestEnvelopeSupport::Unsupported(_) => {
return Err(GeminiCliV1InternalRequestError::EnvelopeUnsupported);
}
};
Ok(GeminiCliV1InternalPayload {
transport: resolved_transport,
body,
})
}
@@ -1,6 +1,7 @@
use crate::ai_serving::{AiExecutionDecision, AiExecutionPlanPayload, GatewayControlDecision};
use crate::{AppState, GatewayError};
mod antigravity;
mod candidate_affinity_cache;
mod candidate_materialization;
mod candidate_metadata;
@@ -12,12 +13,15 @@ mod candidate_transport_ranking_facts;
mod common;
mod decision;
mod decision_input;
mod gemini_cli;
mod materialization_policy;
mod passthrough;
mod plan_builders;
mod pool_scheduler;
pub(crate) mod pool_scores;
mod redaction;
mod report_context;
mod request_gzip;
mod route;
mod runtime_miss;
mod spec_metadata;
@@ -30,6 +34,7 @@ pub(crate) use self::candidate_resolution::{
candidate_auth_channel_skip_reason, read_candidate_transport_snapshot,
EligibleLocalExecutionCandidate, LocalExecutionCandidateKind, SkippedLocalExecutionCandidate,
};
pub(crate) use self::common::resolve_upstream_is_stream_for_provider;
pub(crate) use self::passthrough::{
build_local_same_format_stream_attempt_source, build_local_same_format_stream_plan_and_reports,
build_local_same_format_sync_attempt_source, build_local_same_format_sync_plan_and_reports,
@@ -44,6 +49,7 @@ pub(crate) use self::plan_builders::{
pub(crate) use self::pool_scores::{
build_provider_key_pool_score_upsert, provider_key_pool_score_id, provider_key_pool_score_scope,
};
pub(crate) use self::request_gzip::resolve_transport_request_encoding_policy;
pub(crate) use self::route::is_matching_stream_request as planner_is_matching_stream_request;
pub(crate) use self::runtime_miss::{
apply_local_runtime_candidate_terminal_reason, record_local_runtime_candidate_skip_reason,
@@ -74,7 +80,8 @@ pub(crate) use self::standard::{
build_local_stream_plan_and_reports as build_standard_family_stream_plan_and_reports,
build_local_sync_attempt_source as build_standard_family_sync_attempt_source,
build_local_sync_plan_and_reports as build_standard_family_sync_plan_and_reports,
set_local_openai_chat_execution_exhausted_diagnostic,
codex_model_capabilities_for_transport, set_local_openai_chat_execution_exhausted_diagnostic,
validate_final_openai_provider_request,
};
pub(crate) use self::state::{
GatewayAuthApiKeySnapshot, GatewayProviderTransportSnapshot, LocalResolvedOAuthRequestAuth,
@@ -86,6 +93,71 @@ pub(crate) use aether_ai_serving::{
CandidateFailureDiagnostic, CandidateFailureDiagnosticKind,
};
pub(crate) struct ResolvedTunnelSchedulerAffinityContext {
pub(crate) requested_model: String,
pub(crate) client_session_affinity: Option<aether_scheduler_core::ClientSessionAffinity>,
pub(crate) policy_context: Option<crate::scheduler::affinity::SchedulerAffinityPolicyContext>,
pub(crate) routing_overlay: Option<aether_routing_core::RankingOverlay>,
}
pub(crate) async fn resolve_tunnel_scheduler_affinity_context(
state: &AppState,
parts: &http::request::Parts,
decision: &GatewayControlDecision,
requested_model: String,
body_json: &serde_json::Value,
client_api_format: &str,
) -> Result<Option<ResolvedTunnelSchedulerAffinityContext>, GatewayError> {
let Some(auth_context) = decision.auth_context.as_ref() else {
return Ok(None);
};
let execution_auth_context =
crate::ai_serving::build_execution_runtime_auth_context(auth_context);
let Some(auth_snapshot) = state
.read_cached_auth_api_key_snapshot(
&execution_auth_context.user_id,
&execution_auth_context.api_key_id,
crate::clock::current_unix_secs(),
)
.await?
else {
return Ok(None);
};
let resolved_auth_input = decision_input::ResolvedLocalDecisionAuthInput {
auth_context: execution_auth_context,
auth_snapshot,
required_capabilities: None,
model_directive_policy: decision.model_directive_policy.clone(),
};
let mut input = decision_input::build_local_requested_model_decision_input(
resolved_auth_input,
requested_model,
);
decision_input::attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
client_api_format,
)
.await?;
let policy_context = input
.routing_policy
.as_ref()
.map(crate::scheduler::affinity::SchedulerAffinityPolicyContext::from_routing_policy);
let routing_overlay = input
.routing_policy
.as_ref()
.map(|policy| policy.ranking_overlay.clone());
Ok(Some(ResolvedTunnelSchedulerAffinityContext {
requested_model: input.requested_model,
client_session_affinity: input.client_session_affinity,
policy_context,
routing_overlay,
}))
}
pub(crate) async fn maybe_build_sync_decision_payload(
state: &AppState,
parts: &http::request::Parts,
@@ -71,6 +71,7 @@ pub(crate) fn build_passthrough_stream_plan_from_decision(
.content_type
.take()
.or_else(|| provider_request_headers.get("content-type").cloned());
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -84,7 +85,7 @@ pub(crate) fn build_passthrough_stream_plan_from_decision(
body_bytes_b64: None,
body_ref: None,
},
stream: true,
stream,
},
);
@@ -33,7 +33,7 @@ pub(crate) async fn maybe_build_sync_local_same_format_provider_decision_payload
let Some(input) = resolve_local_same_format_provider_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -55,6 +55,7 @@ pub(crate) async fn maybe_build_sync_local_same_format_provider_decision_payload
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) = build_local_same_format_provider_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
)
@@ -65,12 +66,12 @@ pub(crate) async fn maybe_build_sync_local_same_format_provider_decision_payload
candidate_count,
);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) =
maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -100,7 +101,7 @@ pub(crate) async fn maybe_build_stream_local_same_format_provider_decision_paylo
let Some(input) = resolve_local_same_format_provider_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -122,6 +123,7 @@ pub(crate) async fn maybe_build_stream_local_same_format_provider_decision_paylo
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) = build_local_same_format_provider_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
)
@@ -132,12 +134,12 @@ pub(crate) async fn maybe_build_stream_local_same_format_provider_decision_paylo
candidate_count,
);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) =
maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -13,6 +13,7 @@ use crate::ai_serving::planner::candidate_metadata::{
use crate::ai_serving::planner::candidate_resolution::SkippedLocalExecutionCandidate;
use crate::ai_serving::planner::common::extract_requested_model_from_request;
use crate::ai_serving::planner::decision_input::{
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::planner::materialization_policy::{
@@ -23,7 +24,7 @@ use crate::ai_serving::{
ai_local_execution_contract_for_formats, extract_pool_sticky_session_token,
resolve_local_decision_execution_runtime_auth_context, GatewayControlDecision, PlannerAppState,
};
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::client_session_affinity::client_session_affinity_from_api_request;
use crate::clock::current_unix_secs;
use crate::{AppState, GatewayError};
@@ -39,30 +40,34 @@ pub(crate) async fn resolve_local_same_format_provider_decision_input(
decision: &GatewayControlDecision,
body_json: &serde_json::Value,
spec: LocalSameFormatProviderSpec,
) -> Option<LocalSameFormatProviderDecisionInput> {
) -> Result<Option<LocalSameFormatProviderDecisionInput>, GatewayError> {
let spec_metadata = local_same_format_provider_spec_metadata(spec);
let Some(auth_context) = resolve_local_decision_execution_runtime_auth_context(decision) else {
return None;
return Ok(None);
};
let requested_model = extract_requested_model_from_request(
let Some(requested_model) = extract_requested_model_from_request(
parts,
body_json,
spec_metadata
.requested_model_family
.expect("same-format provider specs should declare requested-model family"),
)?;
) else {
return Ok(None);
};
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
Ok(Some(resolved_input)) => resolved_input,
Ok(None) => return None,
Ok(None) => return Ok(None),
Err(err) => {
warn!(
trace_id = %trace_id,
@@ -70,14 +75,37 @@ pub(crate) async fn resolve_local_same_format_provider_decision_input(
error = ?err,
"gateway local same-format decision auth snapshot read failed"
);
return None;
return Err(err);
}
};
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
Some(input)
input.client_surface = decision.client_surface;
input.gateway_credential_carrier = decision.gateway_credential_carrier;
input.client_session_affinity = client_session_affinity_from_api_request(
spec_metadata.api_format,
&parts.headers,
Some(body_json),
);
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
spec_metadata.api_format,
)
.await
{
warn!(
trace_id = %trace_id,
api_format = spec_metadata.api_format,
error = ?err,
"gateway local same-format decision routing profile resolution failed"
);
return Err(err);
}
Ok(Some(input))
}
pub(crate) async fn materialize_local_same_format_provider_candidate_attempts(
@@ -95,15 +123,23 @@ pub(crate) async fn materialize_local_same_format_provider_candidate_attempts(
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::SameFormatProviderDecision,
);
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec_metadata.api_format, Some(&input.requested_model));
let routing_model = model_directive_resolution
.base_model()
.unwrap_or(&input.requested_model);
let (candidates, preselection_skipped) = planner_state
.list_selectable_candidates_with_skip_reasons(
.list_selectable_candidates_with_skip_reasons_for_request_operation(
spec_metadata.api_format,
&input.requested_model,
routing_model,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
spec.operation.map(|operation| operation.as_str()),
)
.await?;
let outcome = materialize_local_execution_candidates_with_serving(
@@ -114,6 +150,7 @@ pub(crate) async fn materialize_local_same_format_provider_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -191,15 +228,23 @@ pub(crate) async fn build_local_same_format_provider_candidate_attempt_source<'a
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::SameFormatProviderDecision,
);
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec_metadata.api_format, Some(&input.requested_model));
let routing_model = model_directive_resolution
.base_model()
.unwrap_or(&input.requested_model);
let (candidates, preselection_skipped) = planner_state
.list_selectable_candidates_with_skip_reasons(
.list_selectable_candidates_with_skip_reasons_for_request_operation(
spec_metadata.api_format,
&input.requested_model,
routing_model,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
spec.operation.map(|operation| operation.as_str()),
)
.await?;
@@ -211,6 +256,7 @@ pub(crate) async fn build_local_same_format_provider_candidate_attempt_source<'a
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -1,28 +1,34 @@
use serde_json::json;
use aether_ai_serving::{AdaptationMode, AiRequestGzipPolicy, OriginalRequestPayload};
use aether_contracts::{ExecutionResponseBodyMode, EXECUTION_RESPONSE_BODY_MODE_HEADER};
use crate::ai_serving::ai_local_execution_contract_for_formats;
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::candidate_materialization::{
mark_skipped_local_execution_candidate, mark_skipped_local_execution_candidate_with_extra_data,
mark_skipped_local_execution_candidate_with_failure_diagnostic,
};
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::materialization_policy::{
build_local_candidate_persistence_policy, LocalCandidatePersistencePolicyKind,
};
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_same_format_provider_spec_metadata;
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState,
AiExecutionDecision, AppState, GatewayError,
};
use aether_scheduler_core::SchedulerMinimalCandidateSelectionCandidate;
@@ -40,7 +46,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
input: &LocalSameFormatProviderDecisionInput,
attempt: LocalSameFormatProviderCandidateAttempt,
spec: LocalSameFormatProviderSpec,
) -> Option<AiExecutionDecision> {
) -> Result<Option<AiExecutionDecision>, GatewayError> {
let spec_metadata = local_same_format_provider_spec_metadata(spec);
let LocalSameFormatProviderCandidateAttempt {
eligible,
@@ -51,10 +57,20 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
let candidate = &eligible.candidate;
let (execution_strategy, conversion_mode) =
ai_local_execution_contract_for_formats(spec_metadata.api_format, spec_metadata.api_format);
let resolved = resolve_local_same_format_provider_candidate_payload_parts(
let Some(resolved) = resolve_local_same_format_provider_candidate_payload_parts(
state, parts, trace_id, body_json, input, &attempt, spec,
)
.await?;
.await?
else {
return Ok(None);
};
let request_redacted = resolved.request_redacted;
let compatibility_edits_empty = resolved.compatibility_edits.is_empty();
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
Some(body_json)
};
let prompt_cache_key = resolved
.provider_request_body
@@ -66,8 +82,56 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
let transport_profile = resolve_transport_profile(&resolved.transport);
let transport_profile = resolved
.transport_profile
.clone()
.or_else(|| resolve_transport_profile(&resolved.transport));
let mut extra_fields = serde_json::Map::new();
extra_fields.insert(
"provider_type".to_string(),
json!(resolved.transport.provider.provider_type.as_str()),
);
if let Some(operation) = spec.operation {
extra_fields.insert("api_operation".to_string(), json!(operation.as_str()));
}
if let Some(client_surface) = input.client_surface {
extra_fields.insert("client_surface".to_string(), json!(client_surface.as_str()));
}
if let Some(carrier) = input.gateway_credential_carrier {
extra_fields.insert(
"gateway_credential_carrier".to_string(),
json!(carrier.as_str()),
);
}
extra_fields.insert(
"upstream_credential_mode".to_string(),
json!(resolved.transport.key.auth_type.trim().to_ascii_lowercase()),
);
let mut adaptation_mode = if resolved.compatibility_edits.is_empty() {
AdaptationMode::NativeTransparent
} else {
AdaptationMode::SameFormatCompat
};
if crate::ai_serving::normalize_api_format_alias(&resolved.provider_api_format)
== "claude:messages"
{
let compatibility_profile =
crate::ai_serving::transport::resolve_anthropic_compatibility_profile(
&resolved.transport,
&resolved.provider_api_format,
);
extra_fields.insert(
"anthropic_compatibility_profile".to_string(),
json!(compatibility_profile.as_str()),
);
if compatibility_profile.uses_claude_code_compatibility() {
adaptation_mode = AdaptationMode::SameFormatCompat;
}
}
extra_fields.insert(
"adaptation_mode".to_string(),
json!(adaptation_mode.as_str()),
);
if let Some(proxy_value) =
build_request_trace_proxy_value(Some(&resolved.transport), proxy.as_ref())
{
@@ -83,8 +147,24 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
"envelope_name".to_string(),
json!(super::super::ANTIGRAVITY_ENVELOPE_NAME),
);
insert_native_client_envelope_name(
&mut extra_fields,
super::super::ANTIGRAVITY_ENVELOPE_NAME,
parts.uri.path(),
);
} else if resolved.is_gemini_cli {
extra_fields.insert(
"envelope_name".to_string(),
json!(crate::ai_serving::transport::GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME),
);
}
if !resolved.compatibility_edits.is_empty() {
if let Ok(value) = serde_json::to_value(&resolved.compatibility_edits) {
extra_fields.insert("request_body_compatibility_edits".to_string(), value);
}
}
let provider_api_format = resolved.provider_api_format.clone();
let effective_headers = input.effective_headers(&parts.headers);
let report_context = append_local_failover_policy_to_value(
append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -112,20 +192,21 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
body_rules: resolved.transport.endpoint.body_rules.as_ref(),
provider_request_method: Some(serde_json::Value::Null),
provider_request_headers: Some(&resolved.provider_request_headers),
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
.and_then(serde_json::Value::as_bool)
.unwrap_or(false),
upstream_is_stream: resolved.upstream_is_stream,
has_envelope: resolved.is_kiro || resolved.is_antigravity,
has_envelope: resolved.is_kiro || resolved.is_antigravity || resolved.is_gemini_cli,
needs_conversion: false,
extra_fields,
}),
@@ -139,6 +220,7 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
let super::request::LocalSameFormatProviderCandidatePayloadParts {
transport,
is_antigravity: _,
is_gemini_cli: _,
is_kiro: _,
auth_header,
auth_value,
@@ -149,43 +231,124 @@ pub(crate) async fn maybe_build_local_same_format_provider_decision_payload_for_
upstream_url,
provider_request_headers,
provider_request_body,
transport_profile: _,
compatibility_edits: _,
request_redacted: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: transport.provider.name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header,
auth_value,
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream,
report_kind: Some(report_kind.to_string()),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
))
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header,
auth_value,
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream,
report_kind: Some(report_kind.to_string()),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
enforce_provider_api_operation_invariants(
spec.operation,
decision.provider_request_body.as_mut(),
&mut decision.provider_request_headers,
);
decision.provider_request_body_base64 = original_request_body_base64(
parts,
decision.provider_request_body.as_ref(),
adaptation_mode,
request_redacted,
compatibility_edits_empty,
decision.content_encoding.as_deref(),
decision.request_gzip.as_ref(),
);
decision
.provider_request_headers
.retain(|name, _| !name.eq_ignore_ascii_case(EXECUTION_RESPONSE_BODY_MODE_HEADER));
if !spec_metadata.require_streaming && decision.provider_request_body_base64.is_some() {
decision.provider_request_headers.insert(
EXECUTION_RESPONSE_BODY_MODE_HEADER.to_string(),
ExecutionResponseBodyMode::PreserveBytes
.as_str()
.to_string(),
);
}
Ok(Some(decision))
}
fn enforce_provider_api_operation_invariants(
operation: Option<crate::ai_serving::ApiOperation>,
provider_request_body: Option<&mut serde_json::Value>,
provider_request_headers: &mut std::collections::BTreeMap<String, String>,
) {
if operation != Some(crate::ai_serving::ApiOperation::ClaudeCountTokens) {
return;
}
if let Some(provider_request_body) = provider_request_body {
crate::ai_serving::transport::enforce_same_format_provider_api_operation_body_policy(
provider_request_body,
operation,
);
}
for header_name in ["accept", "content-type"] {
provider_request_headers.retain(|name, _| !name.eq_ignore_ascii_case(header_name));
provider_request_headers.insert(header_name.to_string(), "application/json".to_string());
}
}
fn original_request_body_base64(
parts: &http::request::Parts,
provider_request_body: Option<&serde_json::Value>,
adaptation_mode: AdaptationMode,
request_redacted: bool,
compatibility_edits_empty: bool,
content_encoding: Option<&str>,
request_gzip: Option<&AiRequestGzipPolicy>,
) -> Option<String> {
if adaptation_mode != AdaptationMode::NativeTransparent
|| request_redacted
|| !compatibility_edits_empty
|| content_encoding.is_some_and(|value| !value.trim().is_empty())
|| request_gzip.is_some_and(|policy| policy.enabled != Some(false))
{
return None;
}
parts
.extensions
.get::<OriginalRequestPayload>()?
.body_bytes_base64_if_unchanged(provider_request_body?)
}
pub(super) async fn mark_skipped_local_same_format_provider_candidate(
@@ -271,3 +434,177 @@ pub(super) async fn mark_skipped_local_same_format_provider_candidate_with_failu
)
.await;
}
#[cfg(test)]
mod tests {
use std::collections::BTreeMap;
use base64::Engine as _;
use super::{
enforce_provider_api_operation_invariants, original_request_body_base64, AdaptationMode,
AiRequestGzipPolicy, OriginalRequestPayload,
};
use crate::ai_serving::ApiOperation;
fn request_parts_with_original_payload(
body_json: serde_json::Value,
body_bytes: &[u8],
) -> http::request::Parts {
let (mut parts, ()) = http::Request::new(()).into_parts();
parts
.extensions
.insert(OriginalRequestPayload::from_parsed_json(
body_json, body_bytes,
));
parts
}
#[test]
fn count_tokens_invariants_win_after_provider_routing_mutations() {
let mut body = serde_json::json!({
"model": "claude-sonnet-4",
"messages": [],
"stream": true
});
let mut headers = BTreeMap::from([
("Accept".to_string(), "text/event-stream".to_string()),
("Content-Type".to_string(), "text/plain".to_string()),
("x-provider-route".to_string(), "kept".to_string()),
]);
enforce_provider_api_operation_invariants(
Some(ApiOperation::ClaudeCountTokens),
Some(&mut body),
&mut headers,
);
assert!(body.get("stream").is_none());
assert_eq!(
headers.get("accept").map(String::as_str),
Some("application/json")
);
assert_eq!(
headers.get("content-type").map(String::as_str),
Some("application/json")
);
assert_eq!(
headers.get("x-provider-route").map(String::as_str),
Some("kept")
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("accept"))
.count(),
1
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("content-type"))
.count(),
1
);
}
#[test]
fn unchanged_same_format_body_preserves_original_json_bytes() {
let raw = br#"{ "unknown": {"enabled":true}, "messages": [], "model": "claude-sonnet-4" }"#;
let body_json: serde_json::Value = serde_json::from_slice(raw).expect("body should parse");
let parts = request_parts_with_original_payload(body_json.clone(), raw);
let encoded = original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
None,
None,
)
.expect("unchanged request should retain exact bytes");
assert_eq!(
base64::engine::general_purpose::STANDARD
.decode(encoded)
.expect("body should decode"),
raw
);
}
#[test]
fn request_edits_or_encoding_disable_original_json_bytes() {
let raw = br#"{"model":"claude-sonnet-4","messages":[]}"#;
let body_json: serde_json::Value = serde_json::from_slice(raw).expect("body should parse");
let parts = request_parts_with_original_payload(body_json.clone(), raw);
let changed_body = serde_json::json!({
"model": "claude-sonnet-4-5",
"messages": []
});
assert!(original_request_body_base64(
&parts,
Some(&changed_body),
AdaptationMode::NativeTransparent,
false,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
true,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
false,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::SameFormatCompat,
false,
true,
None,
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
Some("gzip"),
None,
)
.is_none());
assert!(original_request_body_base64(
&parts,
Some(&body_json),
AdaptationMode::NativeTransparent,
false,
true,
None,
Some(&AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1),
}),
)
.is_none());
}
}
@@ -1,21 +1,31 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use aether_contracts::ResolvedTransportProfile;
use serde_json::Value;
use crate::ai_serving::planner::common::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
};
use crate::ai_serving::planner::redaction::{
request_identity_response_encoding_when_redacted, resolve_provider_chat_pii_redaction,
};
use crate::ai_serving::transport::antigravity::{
build_antigravity_safe_v1internal_request, build_antigravity_static_identity_headers,
classify_local_antigravity_request_support, AntigravityEnvelopeRequestType,
AntigravityRequestEnvelopeSupport, AntigravityRequestSideSupport,
AntigravityRequestAuthUnsupportedReason, AntigravityRequestEnvelopeSupport,
AntigravityRequestSideSupport, AntigravityRequestSideUnsupportedReason,
};
use crate::ai_serving::transport::{
build_same_format_provider_headers, SameFormatProviderHeadersInput,
build_gemini_cli_v1internal_request, build_grok_browser_headers, build_grok_upstream_url,
build_same_format_provider_headers, resolve_local_gemini_cli_request_auth,
GeminiCliRequestAuth, GeminiCliRequestAuthSupport, GeminiCliRequestEnvelopeSupport,
GrokHeaderInput, SameFormatProviderCompatibilityEdit,
SameFormatProviderCompatibilityEditAction, SameFormatProviderHeadersInput,
GEMINI_CLI_USER_AGENT, GROK_CHAT_PATH,
};
use crate::ai_serving::{CandidateFailureDiagnostic, GatewayProviderTransportSnapshot};
use crate::AppState;
use crate::{AppState, GatewayError};
mod policy;
mod prepare;
@@ -30,7 +40,10 @@ use super::{
LocalSameFormatProviderCandidateAttempt, LocalSameFormatProviderDecisionInput,
LocalSameFormatProviderSpec,
};
use crate::ai_serving::planner::standard::same_format_provider_request_body_failure_extra_data;
use crate::ai_serving::planner::standard::{
codex_model_capabilities_for_transport, openai_provider_request_contract_failure_extra_data,
same_format_provider_request_body_failure_extra_data,
};
pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trace(
transport: &GatewayProviderTransportSnapshot,
@@ -41,6 +54,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
"openai:chat" => "openai:chat",
"openai:responses" => "openai:responses",
"openai:responses:compact" => "openai:responses:compact",
"openai:search" => "openai:search",
"openai:embedding" => "openai:embedding",
"openai:rerank" => "openai:rerank",
"claude:messages" => "claude:messages",
@@ -49,6 +63,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
"jina:embedding" => "jina:embedding",
"jina:rerank" => "jina:rerank",
"doubao:embedding" => "doubao:embedding",
"aliyun:multimodal_embedding" => "aliyun:multimodal_embedding",
_ => return Some("transport_api_format_unsupported"),
};
let behavior = policy::classify_same_format_provider_request_behavior(
@@ -61,9 +76,11 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
decision_kind: "trace_candidate_metadata",
report_kind: Some("trace_candidate_metadata"),
},
None,
);
if !behavior.is_antigravity
&& !behavior.is_claude_code
&& !behavior.is_claude_code_transport
&& !behavior.is_gemini_cli
&& !behavior.is_vertex
&& !behavior.is_kiro
{
@@ -86,6 +103,7 @@ pub(crate) fn resolve_same_format_provider_transport_unsupported_reason_for_trac
pub(crate) struct LocalSameFormatProviderCandidatePayloadParts {
pub(super) transport: Arc<GatewayProviderTransportSnapshot>,
pub(super) is_antigravity: bool,
pub(super) is_gemini_cli: bool,
pub(super) is_kiro: bool,
pub(super) auth_header: Option<String>,
pub(super) auth_value: Option<String>,
@@ -96,6 +114,9 @@ pub(crate) struct LocalSameFormatProviderCandidatePayloadParts {
pub(super) upstream_url: String,
pub(super) provider_request_headers: BTreeMap<String, String>,
pub(super) provider_request_body: Value,
pub(super) transport_profile: Option<ResolvedTransportProfile>,
pub(super) compatibility_edits: Vec<SameFormatProviderCompatibilityEdit>,
pub(super) request_redacted: bool,
}
pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
@@ -106,9 +127,26 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
input: &LocalSameFormatProviderDecisionInput,
attempt: &LocalSameFormatProviderCandidateAttempt,
spec: LocalSameFormatProviderSpec,
) -> Option<LocalSameFormatProviderCandidatePayloadParts> {
) -> Result<Option<LocalSameFormatProviderCandidatePayloadParts>, GatewayError> {
let candidate = &attempt.eligible.candidate;
let prepared = prepare_local_same_format_provider_candidate(
if let Some(skip_reason) = same_format_provider_operation_skip_reason(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
spec.operation,
) {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
let Some(prepared) = prepare_local_same_format_provider_candidate(
state,
trace_id,
input,
@@ -117,28 +155,56 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
&attempt.candidate_id,
spec,
)
.await
else {
return Ok(None);
};
let model_directive_resolution = input
.model_directive_policy
.resolve_reasoning(spec.api_format, Some(&input.requested_model));
let model_directive_mapping =
match model_directive_resolution.mapping_patch_for_mapped_model(&prepared.mapped_model) {
Ok(mapping) => mapping,
Err(skip_reason) => {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
skip_reason,
)
.await;
return Ok(None);
}
};
let effective_headers = input.effective_headers(&parts.headers);
let redaction = resolve_provider_chat_pii_redaction(
state,
parts,
body_json,
&input.auth_context,
spec.api_format,
&attempt.candidate_id,
)
.await?;
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled_for_api_format_and_model(
state,
spec.api_format,
Some(&input.requested_model),
)
.await;
let body_json = redaction.body_json.as_ref();
let mut transport = Arc::clone(&prepared.transport);
let Some(mut base_provider_request_body) =
super::super::request::build_same_format_provider_request_body(
let Some(base_provider_request) =
super::super::request::build_same_format_provider_request_body_with_compatibility_report(
body_json,
prepared.provider_api_format.as_str(),
&prepared.mapped_model,
spec,
prepared.transport.endpoint.body_rules.as_ref(),
Some(&parts.headers),
Some(effective_headers),
prepared.upstream_is_stream,
prepared.force_body_stream_field,
prepared.kiro_auth.as_ref(),
prepared.is_claude_code,
enable_model_directives,
false,
)
else {
mark_skipped_local_same_format_provider_candidate_with_extra_data(
@@ -161,20 +227,23 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
if let Some(mapping) =
crate::system_features::reasoning_model_directive_mapping_for_api_format_and_model(
state,
spec.api_format,
Some(&input.requested_model),
)
.await
{
let mut base_provider_request_body = base_provider_request.body;
let mut compatibility_edits = base_provider_request.compatibility_edits;
if let Some(mapping) = model_directive_mapping.as_ref() {
let before_mapping = base_provider_request_body.clone();
crate::ai_serving::apply_model_directive_mapping_patch(
&mut base_provider_request_body,
&mapping,
mapping,
);
if before_mapping != base_provider_request_body {
compatibility_edits.push(SameFormatProviderCompatibilityEdit {
field: "model_directive_mapping".to_string(),
action: SameFormatProviderCompatibilityEditAction::RuntimeRewrite,
detail: "applied configured model directive mapping patch".to_string(),
});
}
// Directive mapping is a deep-merge patch and may overwrite/add `stream`;
// re-enforce stream-field policy afterward.
// Kiro behavior classification already hard-requires upstream streaming,
@@ -189,12 +258,81 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
}
}
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(input.requested_model.as_str());
let codex_model_capabilities = codex_model_capabilities_for_transport(
&transport,
prepared.provider_api_format.as_str(),
prepared.mapped_model.as_str(),
source_model,
);
if let Err(violation) =
crate::ai_serving::finalize_openai_provider_request_with_codex_model_capabilities(
&mut base_provider_request_body,
crate::ai_serving::OpenAiProviderRequestFinalization {
source_api_format: spec.api_format,
provider_api_format: prepared.provider_api_format.as_str(),
provider_type: transport.provider.provider_type.as_str(),
provider_model: prepared.mapped_model.as_str(),
source_model,
body_rules: transport.endpoint.body_rules.as_ref(),
upstream_is_stream: prepared.upstream_is_stream,
require_body_stream_field: request_requires_body_stream_field(
body_json,
prepared.force_body_stream_field,
),
},
codex_model_capabilities.as_ref(),
)
{
mark_skipped_local_same_format_provider_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_build_failed",
Some(openai_provider_request_contract_failure_extra_data(
&violation,
spec.api_format,
prepared.provider_api_format.as_str(),
"same_format_provider_request_finalization",
)),
)
.await;
return Ok(None);
}
let antigravity_auth = if prepared.is_antigravity {
match classify_local_antigravity_request_support(
&prepared.transport,
let mut antigravity_support = classify_local_antigravity_request_support(
&transport,
&base_provider_request_body,
AntigravityEnvelopeRequestType::Agent,
);
if matches!(
antigravity_support,
AntigravityRequestSideSupport::Unsupported(
AntigravityRequestSideUnsupportedReason::UnsupportedAuth(
AntigravityRequestAuthUnsupportedReason::MissingProjectId
)
)
) {
if let Some(hydrated) = state
.hydrate_antigravity_project_metadata_for_transport(&transport)
.await
{
transport = Arc::new(hydrated);
antigravity_support = classify_local_antigravity_request_support(
&transport,
&base_provider_request_body,
AntigravityEnvelopeRequestType::Agent,
);
}
}
match antigravity_support {
AntigravityRequestSideSupport::Supported(spec) => Some(spec.auth),
AntigravityRequestSideSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate(
@@ -207,13 +345,51 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
"transport_unsupported",
)
.await;
return None;
return Ok(None);
}
}
} else {
None
};
let provider_request_body = if let Some(antigravity_auth) = antigravity_auth.as_ref() {
let gemini_cli_auth = if prepared.behavior.is_gemini_cli {
let mut auth = match resolve_local_gemini_cli_request_auth(&transport) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"transport_auth_unavailable",
)
.await;
return Ok(None);
}
};
if auth.project_id.is_none() {
auth = match state
.hydrate_gemini_cli_project_metadata_for_transport(&transport)
.await
{
Some(hydrated) => {
transport = Arc::new(hydrated);
match resolve_local_gemini_cli_request_auth(&transport) {
GeminiCliRequestAuthSupport::Supported(auth) => auth,
GeminiCliRequestAuthSupport::Unsupported(_) => {
GeminiCliRequestAuth::default()
}
}
}
None => GeminiCliRequestAuth::default(),
};
}
Some(auth)
} else {
None
};
let mut provider_request_body = if let Some(antigravity_auth) = antigravity_auth.as_ref() {
match build_antigravity_safe_v1internal_request(
antigravity_auth,
trace_id,
@@ -239,22 +415,73 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
}
}
} else if let Some(gemini_cli_auth) = gemini_cli_auth.as_ref() {
match build_gemini_cli_v1internal_request(
gemini_cli_auth,
trace_id,
&prepared.mapped_model,
&base_provider_request_body,
) {
GeminiCliRequestEnvelopeSupport::Supported(envelope) => envelope,
GeminiCliRequestEnvelopeSupport::Unsupported(_) => {
mark_skipped_local_same_format_provider_candidate_with_extra_data(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_missing",
same_format_provider_request_body_failure_extra_data(
body_json,
attempt.eligible.provider_api_format.as_str(),
prepared.transport.endpoint.body_rules.as_ref(),
"gemini_cli_v1internal_envelope",
),
)
.await;
return Ok(None);
}
}
} else {
base_provider_request_body
};
if crate::ai_serving::transport::enforce_same_format_provider_api_operation_body_policy(
&mut provider_request_body,
spec.operation,
) {
compatibility_edits.push(SameFormatProviderCompatibilityEdit {
field: "stream".to_string(),
action: SameFormatProviderCompatibilityEditAction::RuntimeRewrite,
detail: "removed stream field for non-streaming API operation".to_string(),
});
}
let Some(upstream_url) = super::super::request::build_same_format_upstream_url(
parts,
&prepared.transport,
&prepared.mapped_model,
prepared.provider_api_format.as_str(),
spec,
prepared.upstream_is_stream,
prepared.kiro_auth.as_ref(),
) else {
let is_grok = prepared
.transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("grok");
let transport_profile = crate::ai_serving::transport::resolve_transport_profile(&transport);
let upstream_url = if is_grok {
Some(build_grok_upstream_url(&transport, GROK_CHAT_PATH))
} else {
super::super::request::build_same_format_upstream_url(
parts,
&transport,
&prepared.mapped_model,
prepared.provider_api_format.as_str(),
spec,
prepared.upstream_is_stream,
prepared.kiro_auth.as_ref(),
Some(&provider_request_body),
)
};
let Some(upstream_url) = upstream_url else {
mark_skipped_local_same_format_provider_candidate_with_failure_diagnostic(
state,
input,
@@ -270,31 +497,45 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
let extra_headers = antigravity_auth
let mut extra_headers = antigravity_auth
.as_ref()
.map(build_antigravity_static_identity_headers)
.unwrap_or_default();
let Some(provider_request_headers) =
build_same_format_provider_headers(SameFormatProviderHeadersInput {
headers: &parts.headers,
if prepared.behavior.is_gemini_cli {
extra_headers.insert("user-agent".to_string(), GEMINI_CLI_USER_AGENT.to_string());
}
let Some(mut provider_request_headers) = (if is_grok {
build_grok_browser_headers(GrokHeaderInput {
transport: &transport,
transport_profile: transport_profile.as_ref(),
request_headers: Some(effective_headers),
content_type: "application/json",
accept: "text/event-stream",
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &provider_request_body,
original_request_body: body_json,
header_rules: prepared.transport.endpoint.header_rules.as_ref(),
})
} else {
build_same_format_provider_headers(SameFormatProviderHeadersInput {
headers: effective_headers,
provider_request_body: &provider_request_body,
original_request_body: body_json,
header_rules: transport.endpoint.header_rules.as_ref(),
behavior: prepared.behavior,
api_operation: spec.operation,
auth_header: prepared.auth_header.as_deref(),
auth_value: prepared.auth_value.as_deref(),
extra_headers: &extra_headers,
key_fingerprint: prepared.transport.key.fingerprint.as_ref(),
kiro_auth_config: prepared.kiro_auth.as_ref().map(|auth| &auth.auth_config),
kiro_machine_id: prepared
.kiro_auth
.as_ref()
.map(|auth| auth.machine_id.as_str()),
})
else {
}) else {
mark_skipped_local_same_format_provider_candidate_with_failure_diagnostic(
state,
input,
@@ -310,12 +551,39 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
),
)
.await;
return None;
return Ok(None);
};
crate::ai_serving::apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
effective_headers,
transport.provider.provider_type.as_str(),
prepared.provider_api_format.as_str(),
Some(trace_id),
transport.key.decrypted_auth_config.as_deref(),
);
let provider_model = provider_request_body
.get("model")
.and_then(Value::as_str)
.unwrap_or(prepared.mapped_model.as_str());
crate::ai_serving::apply_codex_openai_responses_lite_header_for_request_body_with_capabilities(
&mut provider_request_headers,
Some(&provider_request_body),
transport.provider.provider_type.as_str(),
prepared.provider_api_format.as_str(),
provider_model,
source_model,
codex_model_capabilities.as_ref(),
);
request_identity_response_encoding_when_redacted(
&mut provider_request_headers,
redaction.redacted,
);
Some(LocalSameFormatProviderCandidatePayloadParts {
transport: prepared.transport,
Ok(Some(LocalSameFormatProviderCandidatePayloadParts {
transport,
is_antigravity: prepared.is_antigravity,
is_gemini_cli: prepared.behavior.is_gemini_cli,
is_kiro: prepared.is_kiro,
auth_header: prepared.auth_header,
auth_value: prepared.auth_value,
@@ -326,5 +594,103 @@ pub(crate) async fn resolve_local_same_format_provider_candidate_payload_parts(
upstream_url,
provider_request_headers,
provider_request_body,
})
transport_profile,
compatibility_edits,
request_redacted: redaction.redacted,
}))
}
fn same_format_provider_operation_skip_reason(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
operation: Option<crate::ai_serving::ApiOperation>,
) -> Option<&'static str> {
(!crate::ai_serving::transport::transport_supports_api_operation(
transport,
provider_api_format,
operation,
))
.then_some("transport_operation_unsupported")
}
#[cfg(test)]
mod tests {
use super::same_format_provider_operation_skip_reason;
use crate::ai_serving::transport::snapshot::{
GatewayProviderTransportEndpoint, GatewayProviderTransportKey,
GatewayProviderTransportProvider,
};
use crate::ai_serving::{ApiOperation, GatewayProviderTransportSnapshot};
fn private_adapter_transport(provider_type: &str) -> GatewayProviderTransportSnapshot {
GatewayProviderTransportSnapshot {
provider: GatewayProviderTransportProvider {
id: "provider-1".to_string(),
name: provider_type.to_string(),
provider_type: provider_type.to_string(),
website: None,
is_active: true,
keep_priority_on_conversion: false,
enable_format_conversion: true,
concurrent_limit: None,
max_retries: None,
proxy: None,
request_timeout_secs: None,
stream_first_byte_timeout_secs: None,
config: None,
},
endpoint: GatewayProviderTransportEndpoint {
id: "endpoint-1".to_string(),
provider_id: "provider-1".to_string(),
api_format: "claude:messages".to_string(),
api_family: Some("claude".to_string()),
endpoint_kind: Some("chat".to_string()),
is_active: true,
base_url: "https://private.example".to_string(),
header_rules: None,
body_rules: None,
max_retries: None,
custom_path: None,
config: None,
format_acceptance_config: None,
proxy: None,
},
key: GatewayProviderTransportKey {
id: "key-1".to_string(),
provider_id: "provider-1".to_string(),
name: "key".to_string(),
auth_type: "oauth".to_string(),
is_active: true,
api_formats: None,
auth_type_by_format: None,
allow_auth_channel_mismatch_formats: None,
allowed_models: None,
capabilities: None,
rate_multipliers: None,
global_priority_by_format: None,
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: String::new(),
decrypted_auth_config: None,
},
}
}
#[test]
fn private_adapter_count_tokens_is_rejected_by_pre_auth_operation_gate() {
for provider_type in ["kiro", "grok"] {
let transport = private_adapter_transport(provider_type);
assert_eq!(
same_format_provider_operation_skip_reason(
&transport,
"claude:messages",
Some(ApiOperation::ClaudeCountTokens),
),
Some("transport_operation_unsupported"),
"provider_type={provider_type}"
);
}
}
}
@@ -1,6 +1,6 @@
use crate::ai_serving::planner::spec_metadata::LocalExecutionSurfaceSpecMetadata;
use crate::ai_serving::transport::{
classify_same_format_provider_request_behavior as classify_same_format_provider_request_behavior_impl,
classify_same_format_provider_request_behavior_for_operation as classify_same_format_provider_request_behavior_impl,
resolve_same_format_provider_direct_auth as resolve_same_format_provider_direct_auth_impl,
same_format_provider_transport_supported as same_format_provider_transport_supported_impl,
same_format_provider_transport_unsupported_reason as same_format_provider_transport_unsupported_reason_impl,
@@ -15,6 +15,7 @@ pub(super) fn classify_same_format_provider_request_behavior(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
spec_metadata: LocalExecutionSurfaceSpecMetadata,
api_operation: Option<crate::ai_serving::ApiOperation>,
) -> SameFormatProviderRequestBehavior {
classify_same_format_provider_request_behavior_impl(
transport,
@@ -25,6 +26,7 @@ pub(super) fn classify_same_format_provider_request_behavior(
.report_kind
.expect("same-format provider specs should declare report kind"),
},
api_operation,
)
}
@@ -60,11 +62,13 @@ pub(super) fn should_try_same_format_provider_oauth_auth(
behavior: &SameFormatProviderRequestBehavior,
transport: &GatewayProviderTransportSnapshot,
family: LocalSameFormatProviderFamily,
provider_api_format: &str,
) -> bool {
should_try_same_format_provider_oauth_auth_impl(
behavior,
transport,
same_format_provider_family(family),
provider_api_format,
)
}
@@ -72,11 +76,13 @@ pub(super) fn resolve_same_format_provider_direct_auth(
behavior: &SameFormatProviderRequestBehavior,
transport: &GatewayProviderTransportSnapshot,
family: LocalSameFormatProviderFamily,
provider_api_format: &str,
) -> Option<(String, String)> {
resolve_same_format_provider_direct_auth_impl(
behavior,
transport,
same_format_provider_family(family),
provider_api_format,
)
}
@@ -28,6 +28,7 @@ pub(super) struct PreparedSameFormatProviderCandidate {
pub(super) behavior: SameFormatProviderRequestBehavior,
pub(super) is_antigravity: bool,
pub(super) is_claude_code: bool,
pub(super) is_gemini_cli: bool,
pub(super) is_vertex: bool,
pub(super) is_kiro: bool,
pub(super) kiro_auth: Option<KiroRequestAuth>,
@@ -58,6 +59,7 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
&transport,
provider_api_format,
spec_metadata,
spec.operation,
);
if !same_format_provider_transport_supported(
@@ -91,8 +93,12 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
} else {
None
};
let should_try_oauth_auth =
should_try_same_format_provider_oauth_auth(&behavior, &transport, spec.family);
let should_try_oauth_auth = should_try_same_format_provider_oauth_auth(
&behavior,
&transport,
spec.family,
provider_api_format,
);
let oauth_auth = if should_try_oauth_auth {
resolve_candidate_oauth_auth(
planner_state,
@@ -117,7 +123,12 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
{
Some((name.clone(), value.clone()))
} else {
resolve_same_format_provider_direct_auth(&behavior, &transport, spec.family)
resolve_same_format_provider_direct_auth(
&behavior,
&transport,
spec.family,
provider_api_format,
)
};
let (auth_header, auth_value) = match auth {
Some((name, value)) => (Some(name), Some(value)),
@@ -175,6 +186,7 @@ pub(super) async fn prepare_local_same_format_provider_candidate(
behavior,
is_antigravity: behavior.is_antigravity,
is_claude_code: behavior.is_claude_code,
is_gemini_cli: behavior.is_gemini_cli,
is_vertex: behavior.is_vertex,
is_kiro: behavior.is_kiro,
kiro_auth,
@@ -31,7 +31,7 @@ pub(crate) struct LocalSameFormatProviderSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalSameFormatProviderDecisionInput,
spec: LocalSameFormatProviderSpec,
requested_model_family: RequestedModelFamily,
@@ -42,7 +42,7 @@ pub(crate) struct LocalSameFormatProviderStreamAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalSameFormatProviderDecisionInput,
spec: LocalSameFormatProviderSpec,
requested_model_family: RequestedModelFamily,
@@ -64,7 +64,7 @@ pub(crate) async fn build_local_sync_attempt_source<'a>(
let Some(input) = resolve_local_same_format_provider_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -85,8 +85,13 @@ pub(crate) async fn build_local_sync_attempt_source<'a>(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_local_same_format_provider_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
state,
trace_id,
&input,
&effective_body_json,
spec,
)
.await?;
apply_local_runtime_candidate_evaluation_progress_preserving_candidate_signal(
@@ -103,7 +108,7 @@ pub(crate) async fn build_local_sync_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
spec,
requested_model_family,
@@ -128,7 +133,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
let Some(input) = resolve_local_same_format_provider_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -149,8 +154,13 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_local_same_format_provider_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
state,
trace_id,
&input,
&effective_body_json,
spec,
)
.await?;
apply_local_runtime_candidate_evaluation_progress_preserving_candidate_signal(
@@ -167,7 +177,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
spec,
requested_model_family,
@@ -180,7 +190,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalSameFormatProviderSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -203,6 +213,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalSameFormatProviderSyncA
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
@@ -210,7 +235,7 @@ impl LocalExecutionAttemptSource<AiStreamAttempt>
for LocalSameFormatProviderStreamAttemptSource<'_>
{
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -233,6 +258,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt>
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalSameFormatProviderSyncAttemptSource<'_> {
@@ -244,12 +284,12 @@ impl LocalSameFormatProviderSyncAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
@@ -257,7 +297,7 @@ impl LocalSameFormatProviderSyncAttemptSource<'_> {
match build_sync_plan_from_requested_model_family(
self.requested_model_family,
self.parts,
self.body_json,
&self.body_json,
payload,
) {
Ok(value) => Ok(value),
@@ -282,12 +322,12 @@ impl LocalSameFormatProviderStreamAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
@@ -295,7 +335,7 @@ impl LocalSameFormatProviderStreamAttemptSource<'_> {
match build_stream_plan_from_requested_model_family(
self.requested_model_family,
self.parts,
self.body_json,
&self.body_json,
payload,
) {
Ok(value) => Ok(value),
@@ -326,7 +366,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
let Some(input) = resolve_local_same_format_provider_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -347,6 +387,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) = build_local_same_format_provider_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
)
@@ -361,11 +402,11 @@ pub(crate) async fn build_local_sync_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -411,7 +452,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
let Some(input) = resolve_local_same_format_provider_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -432,6 +473,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) = build_local_same_format_provider_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
)
@@ -446,11 +488,11 @@ pub(crate) async fn build_local_stream_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_same_format_provider_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -2,4 +2,5 @@ mod body;
mod url;
pub(super) use self::body::build_same_format_provider_request_body;
pub(super) use self::body::build_same_format_provider_request_body_with_compatibility_report;
pub(super) use self::url::build_same_format_upstream_url;
@@ -3,7 +3,9 @@ use serde_json::Value;
use super::super::LocalSameFormatProviderSpec;
use crate::ai_serving::transport::{
build_same_format_provider_request_body as build_same_format_provider_request_body_impl,
build_same_format_provider_request_body_with_compatibility_report as build_same_format_provider_request_body_with_compatibility_report_impl,
SameFormatProviderFamily, SameFormatProviderRequestBodyInput,
SameFormatProviderRequestBodyOutput,
};
pub(crate) fn build_same_format_provider_request_body(
@@ -36,6 +38,38 @@ pub(crate) fn build_same_format_provider_request_body(
})
}
pub(crate) fn build_same_format_provider_request_body_with_compatibility_report(
body_json: &Value,
provider_api_format: &str,
mapped_model: &str,
spec: LocalSameFormatProviderSpec,
body_rules: Option<&Value>,
request_headers: Option<&http::HeaderMap>,
upstream_is_stream: bool,
force_body_stream_field: bool,
kiro_auth: Option<&crate::ai_serving::transport::kiro::KiroRequestAuth>,
is_claude_code: bool,
enable_model_directives: bool,
) -> Option<SameFormatProviderRequestBodyOutput> {
build_same_format_provider_request_body_with_compatibility_report_impl(
SameFormatProviderRequestBodyInput {
body_json,
mapped_model,
client_api_format: spec.api_format,
provider_api_format,
source_model: body_json.get("model").and_then(Value::as_str),
family: same_format_provider_family(spec.family),
body_rules,
request_headers,
upstream_is_stream,
force_body_stream_field,
kiro_auth_config: kiro_auth.map(|auth| &auth.auth_config),
is_claude_code,
enable_model_directives,
},
)
}
fn same_format_provider_family(
family: super::super::LocalSameFormatProviderFamily,
) -> SameFormatProviderFamily {
@@ -14,6 +14,7 @@ pub(crate) fn build_same_format_upstream_url(
spec: LocalSameFormatProviderSpec,
upstream_is_stream: bool,
kiro_auth: Option<&crate::ai_serving::transport::kiro::KiroRequestAuth>,
provider_request_body: Option<&serde_json::Value>,
) -> Option<String> {
build_same_format_provider_upstream_url_impl(
transport,
@@ -23,6 +24,8 @@ pub(crate) fn build_same_format_upstream_url(
upstream_is_stream,
request_query: parts.uri.query(),
kiro_api_region: kiro_auth.map(|auth| auth.auth_config.effective_api_region()),
api_operation: spec.operation,
provider_request_body,
},
)
}
@@ -98,7 +98,7 @@ fn provider_key_score_input(
.as_object()
.and_then(|snapshot| snapshot.get("account"))
.and_then(Value::as_object);
let (health_score, _, _, any_circuit_open, _) = provider_key_health_summary(key);
let (health_score, _, _, _, _) = provider_key_health_summary(key);
let health_score = key
.health_by_format
.as_ref()
@@ -125,10 +125,9 @@ fn provider_key_score_input(
.and_then(Value::as_bool)
.unwrap_or(false),
oauth_invalid_reason: key.oauth_invalid_reason.clone(),
circuit_open: any_circuit_open,
success_count: key.success_count.unwrap_or(0).into(),
error_count: key.error_count.unwrap_or(0).into(),
total_response_time_ms: key.total_response_time_ms.unwrap_or(0).into(),
total_response_time_ms: key.total_response_time_ms.unwrap_or(0),
total_tokens: key.total_tokens,
total_cost_usd: key.total_cost_usd,
last_used_at: key.last_used_at_unix_secs,
@@ -159,3 +158,70 @@ fn stable_hash(bytes: &[u8]) -> u64 {
}
hash
}
#[cfg(test)]
mod tests {
use super::*;
use aether_data_contracts::repository::pool_scores::PoolMemberHardState;
use serde_json::json;
fn sample_key_with_circuit_next_probe(
next_probe_at_unix_secs: u64,
) -> StoredProviderCatalogKey {
let mut key = StoredProviderCatalogKey::new(
"key-gemini-5".to_string(),
"provider-google-api".to_string(),
"5".to_string(),
"api_key".to_string(),
None,
true,
)
.expect("sample key should be valid");
key.health_by_format = Some(json!({
"gemini:generate_content": {
"health_score": 0.2,
"consecutive_failures": 8
}
}));
key.circuit_breaker_by_format = Some(json!({
"gemini:generate_content": {
"open": true,
"reason": "consecutive_failures_8",
"next_probe_at_unix_secs": next_probe_at_unix_secs
}
}));
key
}
#[test]
fn expired_circuit_probe_deadline_does_not_leave_pool_score_in_cooldown() {
let now_unix_secs = 1_000;
let key = sample_key_with_circuit_next_probe(900);
let score = build_provider_key_pool_score_upsert(
&key,
"custom",
None,
now_unix_secs,
PoolMemberScoreRules::default(),
);
assert_eq!(score.hard_state, PoolMemberHardState::Available);
}
#[test]
fn future_key_circuit_probe_deadline_does_not_drive_pool_score_cooldown() {
let now_unix_secs = 1_000;
let key = sample_key_with_circuit_next_probe(1_100);
let score = build_provider_key_pool_score_upsert(
&key,
"custom",
None,
now_unix_secs,
PoolMemberScoreRules::default(),
);
assert_eq!(score.hard_state, PoolMemberHardState::Available);
}
}
@@ -0,0 +1,239 @@
use std::borrow::Cow;
use std::time::{Instant, SystemTime, UNIX_EPOCH};
use serde_json::Value;
use tracing::warn;
use crate::ai_serving::ExecutionRuntimeAuthContext;
use crate::privacy::{
build_redaction_session_config, read_chat_pii_redaction_runtime_config,
try_mask_chat_pii_request_value_with_cache_options, CachedRequestRedaction,
ChatPiiRedactionRequestFormat, MaskChatRequestOptions, RedactionMaskError,
RedactionSessionSlot, RedisRedactionMappingCache,
};
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{AppState, GatewayError};
pub(crate) struct ProviderRequestRedaction<'a> {
pub(crate) body_json: Cow<'a, Value>,
pub(crate) redacted: bool,
}
impl<'a> ProviderRequestRedaction<'a> {
fn disabled(body_json: &'a Value) -> Self {
Self {
body_json: Cow::Borrowed(body_json),
redacted: false,
}
}
}
#[derive(Clone, Copy, Debug, Default)]
struct ChatPiiRedactionFeatureSettings {
enabled: Option<bool>,
}
impl ChatPiiRedactionFeatureSettings {
fn merge_from_value(&mut self, value: Option<&Value>) {
let Some(settings) = value
.and_then(Value::as_object)
.and_then(|features| features.get("chat_pii_redaction"))
.and_then(Value::as_object)
else {
return;
};
if let Some(enabled) = settings.get("enabled").and_then(Value::as_bool) {
self.enabled = Some(enabled);
}
}
fn effective_enabled(self) -> bool {
self.enabled.unwrap_or(false)
}
}
pub(crate) fn request_identity_response_encoding_when_redacted(
headers: &mut std::collections::BTreeMap<String, String>,
redacted: bool,
) {
if redacted {
headers.insert("accept-encoding".to_string(), "identity".to_string());
}
}
pub(crate) async fn resolve_provider_chat_pii_redaction<'a>(
state: &AppState,
parts: &http::request::Parts,
body_json: &'a Value,
auth_context: &ExecutionRuntimeAuthContext,
client_api_format: &str,
candidate_id: &str,
) -> Result<ProviderRequestRedaction<'a>, GatewayError> {
let Some(format) = ChatPiiRedactionRequestFormat::from_api_format(client_api_format) else {
return Ok(ProviderRequestRedaction::disabled(body_json));
};
let Some(slot) = parts.extensions.get::<RedactionSessionSlot>() else {
return Ok(ProviderRequestRedaction::disabled(body_json));
};
let request_cache_key = request_redaction_cache_key(format, body_json);
if let Some(cached) = slot.cached_request_redaction(&request_cache_key) {
crate::stage_metrics::record_chat_pii_redaction_request_cache_hit();
observe_gateway_stage_ms("chat_pii_redaction_request_cache_hit", 0);
return Ok(provider_redaction_from_cached(
slot,
candidate_id,
body_json,
cached,
));
}
crate::stage_metrics::record_chat_pii_redaction_request_cache_miss();
let runtime_config_started_at = Instant::now();
let runtime_config = read_chat_pii_redaction_runtime_config(state)
.await
.map_err(|err| {
warn!(
error = ?err,
"gateway failed to read chat pii redaction runtime config"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
observe_gateway_stage_ms(
"chat_pii_redaction_runtime_config",
runtime_config_started_at.elapsed().as_millis() as u64,
);
if !runtime_config.enabled {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction::disabled(body_json));
}
let feature_settings_started_at = Instant::now();
let feature_settings = resolve_chat_pii_redaction_feature_settings(state, auth_context).await?;
observe_gateway_stage_ms(
"chat_pii_redaction_feature_settings",
feature_settings_started_at.elapsed().as_millis() as u64,
);
if !feature_settings.effective_enabled() {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction::disabled(body_json));
}
let Some(hmac_key) = state.encryption_key().map(str::as_bytes).map(Vec::from) else {
warn!("gateway chat pii redaction is enabled but encryption key is unavailable");
return Err(GatewayError::Internal(
"chat pii redaction setup failed".to_string(),
));
};
let now_unix_secs = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_secs();
let cache = RedisRedactionMappingCache::new(state.runtime_state.as_ref());
let mask_started_at = Instant::now();
let masked = try_mask_chat_pii_request_value_with_cache_options(
body_json,
format,
build_redaction_session_config(hmac_key, &runtime_config, now_unix_secs),
MaskChatRequestOptions::runtime(),
Some(&cache),
)
.await
.map_err(redaction_mask_error_to_gateway_error)?;
observe_gateway_stage_ms(
"chat_pii_redaction_mask_body",
mask_started_at.elapsed().as_millis() as u64,
);
if !masked.redacted {
slot.put_cached_request_redaction(request_cache_key, CachedRequestRedaction::unredacted());
return Ok(ProviderRequestRedaction {
body_json: Cow::Borrowed(body_json),
redacted: false,
});
}
let Some(masked_body_json) = masked.body_json else {
warn!("gateway pii redaction reported redacted without masked body");
return Err(GatewayError::Internal(
"chat pii redaction setup failed".to_string(),
));
};
slot.put_cached_request_redaction(
request_cache_key,
CachedRequestRedaction::redacted(masked_body_json.clone(), masked.session.clone()),
);
slot.put_for_candidate(candidate_id, masked.session);
Ok(ProviderRequestRedaction {
body_json: Cow::Owned(masked_body_json),
redacted: true,
})
}
fn request_redaction_cache_key(format: ChatPiiRedactionRequestFormat, body_json: &Value) -> String {
format!("{format:?}:{:p}", body_json)
}
fn provider_redaction_from_cached<'a>(
slot: &RedactionSessionSlot,
candidate_id: &str,
body_json: &'a Value,
cached: CachedRequestRedaction,
) -> ProviderRequestRedaction<'a> {
if !cached.redacted {
return ProviderRequestRedaction::disabled(body_json);
}
let Some(masked_body_json) = cached.body_json else {
return ProviderRequestRedaction::disabled(body_json);
};
if let Some(session) = cached.session {
slot.put_for_candidate(candidate_id, session);
}
ProviderRequestRedaction {
body_json: Cow::Owned(masked_body_json),
redacted: true,
}
}
async fn resolve_chat_pii_redaction_feature_settings(
state: &AppState,
auth_context: &ExecutionRuntimeAuthContext,
) -> Result<ChatPiiRedactionFeatureSettings, GatewayError> {
let user_settings_fut = state.read_user_feature_settings(&auth_context.user_id);
let key_settings_fut = state.read_auth_api_key_feature_settings(
&auth_context.user_id,
&auth_context.api_key_id,
auth_context.api_key_is_standalone,
);
let (user_settings, key_settings) = tokio::try_join!(user_settings_fut, key_settings_fut)
.map_err(|err| {
warn!(
error = ?err,
"gateway failed to read chat pii redaction feature settings"
);
GatewayError::Internal("chat pii redaction setup failed".to_string())
})?;
let mut settings = ChatPiiRedactionFeatureSettings::default();
settings.merge_from_value(user_settings.as_ref());
settings.merge_from_value(key_settings.as_ref());
Ok(settings)
}
fn redaction_mask_error_to_gateway_error(error: RedactionMaskError) -> GatewayError {
match error {}
}
#[cfg(test)]
mod tests {
use serde_json::json;
use super::ChatPiiRedactionFeatureSettings;
#[test]
fn chat_pii_redaction_feature_settings_only_control_enablement() {
let mut settings = ChatPiiRedactionFeatureSettings::default();
settings.merge_from_value(Some(&json!({
"chat_pii_redaction": {
"enabled": true
}
})));
assert!(settings.effective_enabled());
}
}
@@ -6,6 +6,7 @@ use aether_ai_serving::{
provider_stream_event_api_format_for_provider_type as ai_provider_stream_event_api_format_for_provider_type,
AiExecutionReportContextParts, AiRequestOrigin,
};
use aether_routing_core::ResolvedRoutingPolicy;
use aether_runtime_state::RuntimeLockLease;
use aether_scheduler_core::{ClientSessionAffinity, SchedulerRankingOutcome};
use serde_json::{Map, Value};
@@ -20,8 +21,9 @@ use crate::client_session_affinity::{
};
use crate::orchestration::{
insert_pool_key_lease_report_context_fields, ExecutionAttemptIdentity,
SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD,
ROUTING_POOL_POLICY_OVERRIDE_REPORT_FIELD, SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD,
};
use crate::scheduler::affinity::insert_scheduler_affinity_policy_report_context_field;
pub(crate) struct LocalExecutionReportContextParts<'a> {
pub(crate) auth_context: &'a ExecutionRuntimeAuthContext,
@@ -55,6 +57,7 @@ pub(crate) struct LocalExecutionReportContextParts<'a> {
pub(crate) original_request_body_json: Option<&'a Value>,
pub(crate) original_request_body_base64: Option<&'a str>,
pub(crate) client_session_affinity: Option<&'a ClientSessionAffinity>,
pub(crate) routing_policy: Option<&'a ResolvedRoutingPolicy>,
pub(crate) scheduler_affinity_epoch: Option<u64>,
pub(crate) client_requested_stream: bool,
pub(crate) upstream_is_stream: bool,
@@ -82,6 +85,18 @@ pub(crate) fn build_local_execution_report_context(
.client_session_affinity
.and_then(client_session_affinity_report_context_value)
{
if let Some(client_family) = value
.as_object()
.and_then(|object| object.get("client_family"))
.and_then(Value::as_str)
.map(str::trim)
.filter(|client_family| !client_family.is_empty())
{
extra_fields.insert(
"client_family".to_string(),
Value::String(client_family.to_ascii_lowercase()),
);
}
extra_fields.insert(
CLIENT_SESSION_AFFINITY_REPORT_CONTEXT_FIELD.to_string(),
value,
@@ -93,6 +108,16 @@ pub(crate) fn build_local_execution_report_context(
merge_incoming_tls_fingerprint(&mut extra_fields, incoming_tls);
}
insert_pool_key_lease_report_context_fields(&mut extra_fields, parts.pool_key_lease);
insert_scheduler_affinity_policy_report_context_field(&mut extra_fields, parts.routing_policy);
if let Some(override_policy) = parts
.routing_policy
.and_then(|policy| policy.pool_policy_overrides.get(parts.provider_id))
.filter(|override_policy| !override_policy.scheduling_presets.is_empty())
{
if let Ok(value) = serde_json::to_value(override_policy) {
extra_fields.insert(ROUTING_POOL_POLICY_OVERRIDE_REPORT_FIELD.to_string(), value);
}
}
if let Some(epoch) = parts.scheduler_affinity_epoch {
extra_fields.insert(
SCHEDULER_AFFINITY_EPOCH_REPORT_FIELD.to_string(),
@@ -186,6 +211,21 @@ pub(crate) fn insert_provider_stream_event_api_format(
insert_ai_provider_stream_event_api_format(extra_fields, provider_type);
}
pub(crate) fn insert_native_client_envelope_name(
extra_fields: &mut Map<String, Value>,
envelope_name: &str,
request_path: &str,
) {
if envelope_name.eq_ignore_ascii_case("antigravity:v1internal")
&& request_path == "/v1internal:streamGenerateContent"
{
extra_fields.insert(
"client_envelope_name".to_string(),
Value::String(envelope_name.to_string()),
);
}
}
fn merge_incoming_tls_fingerprint(extra_fields: &mut Map<String, Value>, incoming_tls: Value) {
let entry = extra_fields
.entry("tls_fingerprint".to_string())
@@ -288,6 +328,7 @@ mod tests {
original_request_body_json: Some(&json!({"model": "gpt-5"})),
original_request_body_base64: None,
client_session_affinity: Some(&client_session_affinity),
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: false,
@@ -370,6 +411,7 @@ mod tests {
})),
original_request_body_base64: None,
client_session_affinity: None,
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: true,
@@ -436,6 +478,7 @@ mod tests {
original_request_body_json: Some(&json!({"model": "gpt-5"})),
original_request_body_base64: None,
client_session_affinity: None,
routing_policy: None,
scheduler_affinity_epoch: None,
client_requested_stream: false,
upstream_is_stream: false,
@@ -0,0 +1,406 @@
use aether_ai_serving::AiRequestGzipPolicy;
use serde_json::Value;
use crate::ai_serving::{normalize_api_format_alias, parse_codex_auth_identity};
use super::state::GatewayProviderTransportSnapshot;
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub(crate) struct TransportRequestEncodingPolicy {
pub content_encoding: Option<String>,
pub request_gzip: Option<AiRequestGzipPolicy>,
}
pub(crate) fn resolve_transport_request_encoding_policy(
transport: &GatewayProviderTransportSnapshot,
) -> TransportRequestEncodingPolicy {
if transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex")
&& normalize_api_format_alias(transport.endpoint.api_format.as_str())
== "openai:responses:compact"
{
return TransportRequestEncodingPolicy::default();
}
let request_gzip = transport_request_gzip_policy_from_config(
transport.endpoint.config.as_ref(),
)
.or_else(|| transport_request_gzip_policy_from_config(transport.provider.config.as_ref()));
if request_gzip.is_some() {
return TransportRequestEncodingPolicy {
content_encoding: None,
request_gzip,
};
}
TransportRequestEncodingPolicy {
content_encoding: default_transport_request_content_encoding(transport),
request_gzip: None,
}
}
fn default_transport_request_content_encoding(
transport: &GatewayProviderTransportSnapshot,
) -> Option<String> {
if !transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex")
{
return None;
}
if !is_codex_request_compression_api_format(transport.endpoint.api_format.as_str()) {
return None;
}
let auth_type =
crate::ai_serving::transport::auth::resolve_local_auth_type_for_transport_format(transport);
let uses_codex_backend = auth_type == "oauth"
|| (auth_type == "bearer"
&& parse_codex_auth_identity(transport.key.decrypted_auth_config.as_deref())
.uses_codex_backend);
if !uses_codex_backend {
return None;
}
Some("zstd".to_string())
}
fn is_codex_request_compression_api_format(api_format: &str) -> bool {
normalize_api_format_alias(api_format) == "openai:responses"
}
fn transport_request_gzip_policy_from_config(
config: Option<&Value>,
) -> Option<AiRequestGzipPolicy> {
let object = config?.as_object()?;
for key in ["request_gzip", "request_body_gzip"] {
if let Some(policy) = object
.get(key)
.and_then(transport_request_gzip_policy_from_value)
{
return Some(policy);
}
}
let enabled = first_config_bool(
object,
&["request_gzip_enabled", "request_body_gzip_enabled"],
);
let min_bytes = first_config_usize(
object,
&["request_gzip_min_bytes", "request_body_gzip_min_bytes"],
);
match (enabled, min_bytes) {
(Some(false), _) => Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
}),
(Some(true), min_bytes) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes,
}),
(None, Some(min_bytes)) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(min_bytes),
}),
(None, None) => None,
}
}
fn transport_request_gzip_policy_from_value(value: &Value) -> Option<AiRequestGzipPolicy> {
if let Some(enabled) = value.as_bool() {
return Some(AiRequestGzipPolicy {
enabled: Some(enabled),
min_bytes: None,
});
}
let object = value.as_object()?;
let enabled = first_config_bool(object, &["enabled"]);
let min_bytes = first_config_usize(object, &["min_bytes"]);
match (enabled, min_bytes) {
(Some(false), _) => Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
}),
(Some(true), min_bytes) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes,
}),
(None, Some(min_bytes)) => Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(min_bytes),
}),
(None, None) => None,
}
}
fn first_config_bool(object: &serde_json::Map<String, Value>, keys: &[&str]) -> Option<bool> {
keys.iter()
.find_map(|key| object.get(*key).and_then(config_bool))
}
fn config_bool(value: &Value) -> Option<bool> {
value.as_bool().or_else(|| {
value.as_str().and_then(|text| {
let normalized = text.trim();
if normalized.eq_ignore_ascii_case("true") {
Some(true)
} else if normalized.eq_ignore_ascii_case("false") {
Some(false)
} else {
None
}
})
})
}
fn first_config_usize(object: &serde_json::Map<String, Value>, keys: &[&str]) -> Option<usize> {
keys.iter()
.find_map(|key| object.get(*key).and_then(config_usize))
}
fn config_usize(value: &Value) -> Option<usize> {
value
.as_u64()
.and_then(|number| usize::try_from(number).ok())
.or_else(|| {
value
.as_str()
.and_then(|text| text.trim().parse::<usize>().ok())
})
}
#[cfg(test)]
mod tests {
use super::*;
use aether_provider_transport::snapshot::{
GatewayProviderTransportEndpoint, GatewayProviderTransportKey,
GatewayProviderTransportProvider, GatewayProviderTransportSnapshot,
};
use serde_json::{json, Value};
fn sample_transport(
provider_type: &str,
endpoint_api_format: &str,
provider_config: Option<Value>,
endpoint_config: Option<Value>,
) -> GatewayProviderTransportSnapshot {
GatewayProviderTransportSnapshot {
provider: GatewayProviderTransportProvider {
id: "provider-1".to_string(),
name: "Provider".to_string(),
provider_type: provider_type.to_string(),
website: None,
is_active: true,
keep_priority_on_conversion: false,
enable_format_conversion: true,
concurrent_limit: None,
max_retries: None,
proxy: None,
request_timeout_secs: None,
stream_first_byte_timeout_secs: None,
config: provider_config,
},
endpoint: GatewayProviderTransportEndpoint {
id: "endpoint-1".to_string(),
provider_id: "provider-1".to_string(),
api_format: endpoint_api_format.to_string(),
api_family: None,
endpoint_kind: None,
is_active: true,
base_url: "https://api.example.test".to_string(),
header_rules: None,
body_rules: None,
max_retries: None,
custom_path: None,
config: endpoint_config,
format_acceptance_config: None,
proxy: None,
},
key: GatewayProviderTransportKey {
id: "key-1".to_string(),
provider_id: "provider-1".to_string(),
name: "key".to_string(),
auth_type: "api_key".to_string(),
is_active: true,
api_formats: None,
auth_type_by_format: None,
allow_auth_channel_mismatch_formats: None,
allowed_models: None,
capabilities: None,
rate_multipliers: None,
global_priority_by_format: None,
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "secret".to_string(),
decrypted_auth_config: None,
},
}
}
fn resolved_gzip_policy(
transport: &GatewayProviderTransportSnapshot,
) -> Option<AiRequestGzipPolicy> {
resolve_transport_request_encoding_policy(transport).request_gzip
}
fn resolved_content_encoding(transport: &GatewayProviderTransportSnapshot) -> Option<String> {
resolve_transport_request_encoding_policy(transport).content_encoding
}
#[test]
fn endpoint_request_gzip_policy_overrides_provider_policy() {
let transport = sample_transport(
"openai",
"openai:responses",
Some(json!({"request_gzip": false})),
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 1024}})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1024),
})
);
}
#[test]
fn endpoint_request_gzip_false_disables_provider_and_codex_defaults() {
let transport = sample_transport(
"codex",
"openai:responses",
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 1024}})),
Some(json!({"request_gzip": false})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(false),
min_bytes: None,
})
);
}
#[test]
fn request_gzip_policy_supports_top_level_aliases() {
let transport = sample_transport(
"openai",
"openai:responses",
None,
Some(json!({
"request_body_gzip_enabled": true,
"request_body_gzip_min_bytes": "4096"
})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(4096),
})
);
}
#[test]
fn request_gzip_policy_treats_min_bytes_only_as_enabled() {
let transport = sample_transport(
"openai",
"openai:responses",
None,
Some(json!({"request_gzip_min_bytes": 1})),
);
assert_eq!(
resolved_gzip_policy(&transport),
Some(AiRequestGzipPolicy {
enabled: Some(true),
min_bytes: Some(1),
})
);
}
#[test]
fn codex_responses_endpoint_uses_zstd_without_a_size_threshold() {
let mut transport = sample_transport("codex", "openai:responses", None, None);
transport.key.auth_type = "oauth".to_string();
assert_eq!(
resolved_content_encoding(&transport).as_deref(),
Some("zstd")
);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_responses_api_key_auth_does_not_enable_default_compression() {
let transport = sample_transport("codex", "openai:responses", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_responses_bearer_auth_uses_identity_metadata_for_backend_compression() {
let mut transport = sample_transport("codex", "openai:responses", None, None);
transport.key.auth_type = "bearer".to_string();
transport.key.decrypted_auth_config =
Some(r#"{"provider_type":"codex","account_id":"account-1"}"#.to_string());
assert_eq!(
resolved_content_encoding(&transport).as_deref(),
Some("zstd")
);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_image_endpoint_does_not_get_responses_request_gzip_policy() {
let transport = sample_transport("codex", "openai:image", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_compact_endpoint_does_not_get_default_request_gzip_policy() {
let transport = sample_transport("codex", "openai:responses:compact", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
#[test]
fn codex_compact_endpoint_rejects_an_explicit_request_gzip_policy() {
let transport = sample_transport(
"codex",
"openai:responses:compact",
None,
Some(json!({"request_gzip": {"enabled": true, "min_bytes": 2048}})),
);
assert_eq!(resolved_gzip_policy(&transport), None);
assert_eq!(resolved_content_encoding(&transport), None);
}
#[test]
fn non_codex_endpoint_does_not_get_default_request_gzip_policy() {
let transport = sample_transport("openai", "openai:responses", None, None);
assert_eq!(resolved_content_encoding(&transport), None);
assert_eq!(resolved_gzip_policy(&transport), None);
}
}
@@ -1,8 +1,8 @@
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{
is_matching_stream_http_request as is_matching_stream_http_request_impl,
resolve_execution_runtime_stream_plan_kind as resolve_execution_runtime_stream_plan_kind_impl,
resolve_execution_runtime_sync_plan_kind as resolve_execution_runtime_sync_plan_kind_impl,
resolve_execution_runtime_stream_plan_kind_with_client_surface as resolve_execution_runtime_stream_plan_kind_impl,
resolve_execution_runtime_sync_plan_kind_with_client_surface as resolve_execution_runtime_sync_plan_kind_impl,
supports_stream_execution_decision_kind as supports_stream_execution_decision_kind_impl,
supports_sync_execution_decision_kind as supports_sync_execution_decision_kind_impl,
};
@@ -11,28 +11,34 @@ pub(crate) fn resolve_execution_runtime_stream_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
resolve_execution_runtime_stream_plan_kind_impl(
let plan_kind = resolve_execution_runtime_stream_plan_kind_impl(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, true, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn resolve_execution_runtime_sync_plan_kind(
parts: &http::request::Parts,
decision: &GatewayControlDecision,
) -> Option<&'static str> {
resolve_execution_runtime_sync_plan_kind_impl(
let plan_kind = resolve_execution_runtime_sync_plan_kind_impl(
decision.route_class.as_deref(),
decision.route_family.as_deref(),
decision.route_kind.as_deref(),
decision.client_surface,
decision.request_auth_channel.as_deref(),
&parts.method,
parts.uri.path(),
)
)?;
crate::ai_serving::plan_kind_matches_api_operation(plan_kind, false, decision.api_operation)
.then_some(plan_kind)
}
pub(crate) fn is_matching_stream_request(
@@ -62,7 +68,7 @@ mod tests {
resolve_execution_runtime_sync_plan_kind, supports_stream_execution_decision_kind,
supports_sync_execution_decision_kind,
};
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{ApiOperation, ClientSurface, GatewayControlDecision};
fn sample_decision(route_family: &str, route_kind: &str) -> GatewayControlDecision {
GatewayControlDecision {
@@ -71,12 +77,16 @@ mod tests {
route_class: Some("ai_public".to_string()),
route_family: Some(route_family.to_string()),
route_kind: Some(route_kind.to_string()),
client_surface: None,
api_operation: None,
gateway_credential_carrier: None,
request_auth_channel: None,
auth_context: None,
admin_principal: None,
auth_endpoint_signature: None,
execution_runtime_candidate: true,
local_auth_rejection: None,
model_directive_policy: Default::default(),
}
}
@@ -120,7 +130,9 @@ mod tests {
let (claude_parts, _) = claude_request.into_parts();
let claude_api_key = sample_decision_with_auth_channel("claude", "messages", "api_key");
let claude_bearer = sample_decision_with_auth_channel("claude", "messages", "bearer_like");
let mut claude_bearer =
sample_decision_with_auth_channel("claude", "messages", "bearer_like");
claude_bearer.client_surface = Some(ClientSurface::ClaudeCode);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&claude_parts, &claude_api_key),
Some("claude_chat_sync")
@@ -130,6 +142,13 @@ mod tests {
Some("claude_cli_stream")
);
let claude_sdk_bearer =
sample_decision_with_auth_channel("claude", "messages", "bearer_like");
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&claude_parts, &claude_sdk_bearer),
Some("claude_chat_sync")
);
let gemini_request = Request::builder()
.method(Method::POST)
.uri("/v1beta/models/gemini-2.5-pro:generateContent")
@@ -151,6 +170,36 @@ mod tests {
);
}
#[test]
fn resolves_claude_count_tokens_as_native_sync_operation() {
let request = Request::builder()
.method(Method::POST)
.uri("/v1/messages/count_tokens")
.body(())
.expect("request should build");
let (parts, _) = request.into_parts();
let mut decision = sample_decision("claude", "count_tokens");
decision.api_operation = Some(ApiOperation::ClaudeCountTokens);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&parts, &decision),
Some("claude_count_tokens_sync")
);
assert!(supports_sync_execution_decision_kind(
"claude_count_tokens_sync"
));
decision.api_operation = Some(ApiOperation::ClaudeMessagesCreate);
assert_eq!(
resolve_execution_runtime_sync_plan_kind(&parts, &decision),
None
);
assert_eq!(
resolve_execution_runtime_stream_plan_kind(&parts, &decision),
None
);
}
#[test]
fn stream_matching_uses_surface_route_logic() {
let request = Request::builder()
@@ -29,7 +29,7 @@ use self::support::{
pub(crate) struct LocalGeminiFilesSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
body_base64: Option<&'a str>,
body_is_empty: bool,
trace_id: &'a str,
@@ -110,10 +110,11 @@ pub(crate) async fn build_local_gemini_files_sync_attempt_source_for_kind<'a>(
trace_id,
decision,
)
.await
.await?
else {
return Ok(None);
};
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) =
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
if candidate_count == 0 {
@@ -124,7 +125,7 @@ pub(crate) async fn build_local_gemini_files_sync_attempt_source_for_kind<'a>(
LocalGeminiFilesSyncAttemptSource {
state,
parts,
body_json,
body_json: effective_body_json,
body_base64,
body_is_empty,
trace_id,
@@ -148,7 +149,7 @@ pub(crate) async fn build_local_gemini_files_stream_attempt_source_for_kind<'a>(
};
let Some(input) =
resolve_local_gemini_files_decision_input(state, parts, None, trace_id, decision).await
resolve_local_gemini_files_decision_input(state, parts, None, trace_id, decision).await?
else {
return Ok(None);
};
@@ -174,7 +175,7 @@ pub(crate) async fn build_local_gemini_files_stream_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalGeminiFilesSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -192,12 +193,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalGeminiFilesSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalGeminiFilesStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -215,6 +231,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalGeminiFilesStreamAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalGeminiFilesSyncAttemptSource<'_> {
@@ -226,7 +257,7 @@ impl LocalGeminiFilesSyncAttemptSource<'_> {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
self.state,
self.parts,
self.body_json,
&self.body_json,
self.body_base64,
self.body_is_empty,
self.trace_id,
@@ -234,7 +265,7 @@ impl LocalGeminiFilesSyncAttemptSource<'_> {
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
@@ -272,7 +303,7 @@ impl LocalGeminiFilesStreamAttemptSource<'_> {
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
@@ -313,15 +344,16 @@ pub(crate) async fn maybe_build_sync_local_gemini_files_decision_payload(
trace_id,
decision,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let (mut source, _) =
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -333,7 +365,7 @@ pub(crate) async fn maybe_build_sync_local_gemini_files_decision_payload(
attempt,
spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -354,7 +386,7 @@ pub(crate) async fn maybe_build_stream_local_gemini_files_decision_payload(
};
let Some(input) =
resolve_local_gemini_files_decision_input(state, parts, None, trace_id, decision).await
resolve_local_gemini_files_decision_input(state, parts, None, trace_id, decision).await?
else {
return Ok(None);
};
@@ -363,7 +395,7 @@ pub(crate) async fn maybe_build_stream_local_gemini_files_decision_payload(
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
let empty_body_json = serde_json::Value::Null;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -375,7 +407,7 @@ pub(crate) async fn maybe_build_stream_local_gemini_files_decision_payload(
attempt,
spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -402,16 +434,17 @@ async fn build_local_sync_plan_and_reports(
trace_id,
decision,
)
.await
.await?
else {
return Ok(Vec::new());
};
let body_json = input.effective_body_json(body_json);
let (mut source, _) =
build_local_gemini_files_candidate_attempt_source(state, trace_id, &input).await?;
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -423,7 +456,7 @@ async fn build_local_sync_plan_and_reports(
attempt,
spec,
)
.await
.await?
else {
continue;
};
@@ -454,7 +487,7 @@ async fn build_local_stream_plan_and_reports(
) -> Result<Vec<AiStreamAttempt>, GatewayError> {
let spec_metadata = local_gemini_files_spec_metadata(spec);
let Some(input) =
resolve_local_gemini_files_decision_input(state, parts, None, trace_id, decision).await
resolve_local_gemini_files_decision_input(state, parts, None, trace_id, decision).await?
else {
return Ok(Vec::new());
};
@@ -464,7 +497,7 @@ async fn build_local_stream_plan_and_reports(
let mut plans = Vec::new();
let empty_body_json = serde_json::Value::Null;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_gemini_files_decision_payload_for_candidate(
state,
parts,
@@ -476,7 +509,7 @@ async fn build_local_stream_plan_and_reports(
attempt,
spec,
)
.await
.await?
else {
continue;
};
@@ -1,18 +1,20 @@
use serde_json::json;
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_gemini_files_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{AiExecutionDecision, AppState};
use crate::{append_local_failover_policy_to_value, AiExecutionDecision, AppState, GatewayError};
use super::request::resolve_local_gemini_files_candidate_payload_parts;
use super::support::{
@@ -31,7 +33,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
input: &LocalGeminiFilesDecisionInput,
attempt: LocalGeminiFilesCandidateAttempt,
spec: LocalGeminiFilesSpec,
) -> Option<AiExecutionDecision> {
) -> Result<Option<AiExecutionDecision>, GatewayError> {
let spec_metadata = local_gemini_files_spec_metadata(spec);
let planner_state = PlannerAppState::new(state);
let attempt_identity = attempt.attempt_identity();
@@ -46,7 +48,10 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
&attempt,
spec,
)
.await?;
.await;
let Some(resolved) = resolved else {
return Ok(None);
};
let LocalGeminiFilesCandidateAttempt {
eligible,
candidate_id,
@@ -69,6 +74,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
}
extra_fields.insert("file_key_id".to_string(), json!(candidate.key_id));
extra_fields.insert("file_name".to_string(), json!(resolved.file_name));
let effective_headers = input.effective_headers(&parts.headers);
let report_context = build_local_execution_report_context(LocalExecutionReportContextParts {
auth_context: &input.auth_context,
request_id: trace_id,
@@ -94,13 +100,14 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
body_rules: transport.endpoint.body_rules.as_ref(),
provider_request_method: None,
provider_request_headers: None,
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_base64: resolved.provider_request_body_base64.as_deref(),
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: spec_metadata.require_streaming,
upstream_is_stream: spec_metadata.require_streaming,
@@ -108,6 +115,7 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
needs_conversion: false,
extra_fields,
});
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let super::request::LocalGeminiFilesCandidatePayloadParts {
transport: _,
auth_header,
@@ -118,46 +126,53 @@ pub(super) async fn maybe_build_local_gemini_files_decision_payload_for_candidat
upstream_url,
file_name: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: Some(parts.method.to_string()),
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format: GEMINI_FILES_CLIENT_API_FORMAT.to_string(),
client_api_format: GEMINI_FILES_CLIENT_API_FORMAT.to_string(),
model_name: "gemini-files".to_string(),
mapped_model: candidate.selected_provider_model_name.clone(),
prompt_cache_key: None,
provider_request_headers,
provider_request_body,
provider_request_body_base64,
content_type: parts
.headers
.get(http::header::CONTENT_TYPE)
.and_then(|value| value.to_str().ok())
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream: spec_metadata.require_streaming,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
))
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: Some(parts.method.to_string()),
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format: GEMINI_FILES_CLIENT_API_FORMAT.to_string(),
client_api_format: GEMINI_FILES_CLIENT_API_FORMAT.to_string(),
model_name: "gemini-files".to_string(),
mapped_model: candidate.selected_provider_model_name.clone(),
prompt_cache_key: None,
provider_request_headers,
provider_request_body,
provider_request_body_base64,
content_type: effective_headers
.get(http::header::CONTENT_TYPE)
.and_then(|value| value.to_str().ok())
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream: spec_metadata.require_streaming,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -45,6 +45,7 @@ pub(super) async fn resolve_local_gemini_files_candidate_payload_parts(
let spec_metadata = local_gemini_files_spec_metadata(spec);
let candidate = &attempt.eligible.candidate;
let transport = &attempt.eligible.transport;
let effective_headers = input.effective_headers(&parts.headers);
if let Some(skip_reason) =
gemini_files_transport_unsupported_reason(transport, GEMINI_FILES_CANDIDATE_API_FORMAT)
@@ -103,7 +104,7 @@ pub(super) async fn resolve_local_gemini_files_candidate_payload_parts(
body_is_empty,
spec_metadata.decision_kind == GEMINI_FILES_UPLOAD_PLAN_KIND,
transport.endpoint.body_rules.as_ref(),
Some(&parts.headers),
Some(effective_headers),
) {
Ok(parts) => parts,
Err(GeminiFilesRequestBodyError::BodyRulesUnsupportedForBinaryUpload) => {
@@ -145,7 +146,7 @@ pub(super) async fn resolve_local_gemini_files_candidate_payload_parts(
};
let Some(provider_request_headers) = build_gemini_files_headers(GeminiFilesHeadersInput {
headers: &parts.headers,
headers: effective_headers,
auth_header: &auth_header,
auth_value: &auth_value,
header_rules: transport.endpoint.header_rules.as_ref(),
@@ -14,7 +14,8 @@ use crate::ai_serving::planner::candidate_metadata::{
build_local_execution_candidate_metadata_for_candidate, LocalExecutionCandidateMetadataParts,
};
use crate::ai_serving::planner::decision_input::{
build_local_authenticated_decision_input, resolve_local_authenticated_decision_input,
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::planner::materialization_policy::{
build_local_candidate_persistence_policy, LocalCandidatePersistencePolicyKind,
@@ -29,11 +30,12 @@ use crate::{AppState, GatewayError};
pub(super) use crate::ai_serving::planner::candidate_materialization::LocalExecutionCandidateAttempt as LocalGeminiFilesCandidateAttempt;
pub(super) use crate::ai_serving::planner::candidate_materialization::LocalExecutionCandidateAttemptSource as LocalGeminiFilesCandidateAttemptSource;
pub(super) use crate::ai_serving::planner::decision_input::LocalAuthenticatedDecisionInput as LocalGeminiFilesDecisionInput;
pub(super) use crate::ai_serving::planner::decision_input::LocalRequestedModelDecisionInput as LocalGeminiFilesDecisionInput;
pub(super) const GEMINI_FILES_CANDIDATE_API_FORMAT: &str = "gemini:files";
pub(super) const GEMINI_FILES_CLIENT_API_FORMAT: &str = "gemini:files";
pub(super) const GEMINI_FILES_REQUIRED_CAPABILITY: &str = "gemini_files";
pub(super) const GEMINI_FILES_ROUTING_MODEL: &str = "gemini-files";
pub(super) async fn resolve_local_gemini_files_decision_input(
state: &AppState,
@@ -41,9 +43,9 @@ pub(super) async fn resolve_local_gemini_files_decision_input(
body_json: Option<&serde_json::Value>,
trace_id: &str,
decision: &GatewayControlDecision,
) -> Option<LocalGeminiFilesDecisionInput> {
) -> Result<Option<LocalGeminiFilesDecisionInput>, GatewayError> {
let Some(auth_context) = resolve_local_decision_execution_runtime_auth_context(decision) else {
return None;
return Ok(None);
};
let explicit_required_capabilities = json!({ "gemini_files": true });
@@ -51,25 +53,40 @@ pub(super) async fn resolve_local_gemini_files_decision_input(
state,
auth_context,
None,
decision.auth_endpoint_signature.as_deref(),
Some(&explicit_required_capabilities),
&decision.model_directive_policy,
)
.await
{
Ok(Some(resolved_input)) => resolved_input,
Ok(None) => return None,
Ok(None) => return Ok(None),
Err(err) => {
warn!(
trace_id = %trace_id,
error = ?err,
"gateway local gemini files decision auth snapshot read failed"
);
return None;
return Err(err);
}
};
let mut input = build_local_authenticated_decision_input(resolved_input);
let routing_body_json = body_json.cloned().unwrap_or(serde_json::Value::Null);
let mut input = build_local_requested_model_decision_input(
resolved_input,
GEMINI_FILES_ROUTING_MODEL.to_string(),
);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, body_json);
Some(input)
attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
&routing_body_json,
GEMINI_FILES_CLIENT_API_FORMAT,
)
.await?;
Ok(Some(input))
}
pub(super) async fn materialize_local_gemini_files_candidate_attempts(
@@ -101,8 +118,9 @@ pub(super) async fn materialize_local_gemini_files_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
None,
None,
input.request_auth_channel.as_deref(),
persistence_policy,
candidates,
Vec::new(),
@@ -173,8 +191,9 @@ pub(super) async fn build_local_gemini_files_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
None,
None,
input.request_auth_channel.as_deref(),
persistence_policy,
candidates,
Vec::new(),
@@ -32,7 +32,7 @@ pub(super) use crate::ai_serving::LocalOpenAiImageSpec;
pub(crate) struct LocalOpenAiImageSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
body_base64: Option<&'a str>,
trace_id: &'a str,
input: LocalOpenAiImageDecisionInput,
@@ -43,7 +43,7 @@ pub(crate) struct LocalOpenAiImageSyncAttemptSource<'a> {
pub(crate) struct LocalOpenAiImageStreamAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
body_base64: Option<&'a str>,
trace_id: &'a str,
input: LocalOpenAiImageDecisionInput,
@@ -152,16 +152,17 @@ pub(crate) async fn build_local_image_sync_attempt_source_for_kind<'a>(
trace_id,
decision,
)
.await
.await?
else {
return Ok(None);
};
let effective_body_json = input.effective_body_json(body_json).clone();
let Some((candidates, candidate_count)) = build_local_openai_image_candidate_attempt_source(
state,
trace_id,
&input,
body_json,
&effective_body_json,
spec_metadata.api_format,
spec_metadata.decision_kind,
)
@@ -178,7 +179,7 @@ pub(crate) async fn build_local_image_sync_attempt_source_for_kind<'a>(
LocalOpenAiImageSyncAttemptSource {
state,
parts,
body_json,
body_json: effective_body_json,
body_base64,
trace_id,
input,
@@ -211,16 +212,17 @@ pub(crate) async fn build_local_image_stream_attempt_source_for_kind<'a>(
trace_id,
decision,
)
.await
.await?
else {
return Ok(None);
};
let effective_body_json = input.effective_body_json(body_json).clone();
let Some((candidates, candidate_count)) = build_local_openai_image_candidate_attempt_source(
state,
trace_id,
&input,
body_json,
&effective_body_json,
spec_metadata.api_format,
spec_metadata.decision_kind,
)
@@ -237,7 +239,7 @@ pub(crate) async fn build_local_image_stream_attempt_source_for_kind<'a>(
LocalOpenAiImageStreamAttemptSource {
state,
parts,
body_json,
body_json: effective_body_json,
body_base64,
trace_id,
input,
@@ -251,7 +253,7 @@ pub(crate) async fn build_local_image_stream_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiImageSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -269,12 +271,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiImageSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiImageStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -292,6 +309,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiImageStreamAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiImageSyncAttemptSource<'_> {
@@ -303,21 +335,21 @@ impl LocalOpenAiImageSyncAttemptSource<'_> {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
self.state,
self.parts,
self.body_json,
&self.body_json,
self.body_base64,
self.trace_id,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
let provider_api_format = payload.provider_api_format.as_deref().unwrap_or_default();
let built = if provider_api_format == "gemini:generate_content" {
build_gemini_sync_plan_from_decision(self.parts, self.body_json, payload)
build_gemini_sync_plan_from_decision(self.parts, &self.body_json, payload)
} else {
build_passthrough_sync_plan_from_decision(self.parts, payload)
};
@@ -345,23 +377,23 @@ impl LocalOpenAiImageStreamAttemptSource<'_> {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
self.state,
self.parts,
self.body_json,
&self.body_json,
self.body_base64,
self.trace_id,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
let provider_api_format = payload.provider_api_format.as_deref().unwrap_or_default();
let built = if provider_api_format == "gemini:generate_content" {
build_gemini_stream_plan_from_decision(self.parts, self.body_json, payload)
build_gemini_stream_plan_from_decision(self.parts, &self.body_json, payload)
} else {
build_standard_stream_plan_from_decision(self.parts, self.body_json, payload, false)
build_standard_stream_plan_from_decision(self.parts, &self.body_json, payload, false)
};
match built {
Ok(value) => Ok(value),
@@ -400,10 +432,11 @@ pub(crate) async fn maybe_build_sync_local_image_decision_payload(
trace_id,
decision,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let Some((mut source, _)) = build_local_openai_image_candidate_attempt_source(
state,
@@ -418,7 +451,7 @@ pub(crate) async fn maybe_build_sync_local_image_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -429,7 +462,7 @@ pub(crate) async fn maybe_build_sync_local_image_decision_payload(
attempt,
spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -460,10 +493,11 @@ pub(crate) async fn maybe_build_stream_local_image_decision_payload(
trace_id,
decision,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let Some((mut source, _)) = build_local_openai_image_candidate_attempt_source(
state,
@@ -478,7 +512,7 @@ pub(crate) async fn maybe_build_stream_local_image_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -489,7 +523,7 @@ pub(crate) async fn maybe_build_stream_local_image_decision_payload(
attempt,
spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -516,10 +550,11 @@ async fn build_local_sync_plan_and_reports(
trace_id,
decision,
)
.await
.await?
else {
return Ok(Vec::new());
};
let body_json = input.effective_body_json(body_json);
let Some((mut source, _)) = build_local_openai_image_candidate_attempt_source(
state,
@@ -535,7 +570,7 @@ async fn build_local_sync_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -546,7 +581,7 @@ async fn build_local_sync_plan_and_reports(
attempt,
spec,
)
.await
.await?
else {
continue;
};
@@ -592,10 +627,11 @@ async fn build_local_stream_plan_and_reports(
trace_id,
decision,
)
.await
.await?
else {
return Ok(Vec::new());
};
let body_json = input.effective_body_json(body_json);
let Some((mut source, _)) = build_local_openai_image_candidate_attempt_source(
state,
@@ -611,7 +647,7 @@ async fn build_local_stream_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_image_decision_payload_for_candidate(
state,
parts,
@@ -622,7 +658,7 @@ async fn build_local_stream_plan_and_reports(
attempt,
spec,
)
.await
.await?
else {
continue;
};
@@ -1,16 +1,21 @@
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_openai_image_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{append_execution_contract_fields_to_value, AiExecutionDecision, AppState};
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_image_candidate_payload_parts;
use super::support::{LocalOpenAiImageCandidateAttempt, LocalOpenAiImageDecisionInput};
@@ -25,11 +30,11 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
input: &LocalOpenAiImageDecisionInput,
attempt: LocalOpenAiImageCandidateAttempt,
spec: LocalOpenAiImageSpec,
) -> Option<AiExecutionDecision> {
) -> Result<Option<AiExecutionDecision>, GatewayError> {
let spec_metadata = local_openai_image_spec_metadata(spec);
let planner_state = PlannerAppState::new(state);
let attempt_identity = attempt.attempt_identity();
let resolved = resolve_local_openai_image_candidate_payload_parts(
let Some(resolved) = resolve_local_openai_image_candidate_payload_parts(
state,
parts,
body_json,
@@ -39,7 +44,10 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
&attempt,
spec,
)
.await?;
.await
else {
return Ok(None);
};
let LocalOpenAiImageCandidateAttempt {
eligible,
candidate_id,
@@ -57,7 +65,10 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
.app()
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&transport)
.await;
let transport_profile = resolve_transport_profile(&transport);
let transport_profile = resolved
.transport_profile
.clone()
.or_else(|| resolve_transport_profile(&transport));
let mut extra_fields = serde_json::Map::new();
if let Some(proxy_value) = build_request_trace_proxy_value(Some(&transport), proxy.as_ref()) {
extra_fields.insert("proxy".to_string(), proxy_value);
@@ -73,21 +84,9 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
"chatgpt_web_image".to_string(),
serde_json::Value::Bool(true),
);
extra_fields.insert(
"local_failover_policy".to_string(),
serde_json::json!({
"stop_status_codes": [400, 401, 403, 429, 500, 502, 503, 504],
"error_stop_patterns": [
{ "pattern": ".*" }
]
}),
);
}
let upstream_is_stream = resolved
.provider_request_body
.get("stream")
.and_then(serde_json::Value::as_bool)
.unwrap_or(spec_metadata.require_streaming);
let upstream_is_stream = resolved.upstream_is_stream;
let effective_headers = input.effective_headers(&parts.headers);
let report_context = append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
auth_context: &input.auth_context,
@@ -114,13 +113,14 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
body_rules: transport.endpoint.body_rules.as_ref(),
provider_request_method: Some(serde_json::Value::String(parts.method.to_string())),
provider_request_headers: Some(&resolved.provider_request_headers),
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_base64: body_base64,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: spec_metadata.require_streaming,
upstream_is_stream,
@@ -133,40 +133,49 @@ pub(super) async fn maybe_build_local_openai_image_decision_payload_for_candidat
spec_metadata.api_format,
provider_api_format.as_str(),
);
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url: resolved.upstream_url,
provider_request_method: Some(parts.method.to_string()),
auth_header: Some(resolved.auth_header),
auth_value: Some(resolved.auth_value),
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: resolved.requested_model,
mapped_model: resolved.mapped_model,
prompt_cache_key: None,
provider_request_headers: resolved.provider_request_headers,
provider_request_body: Some(resolved.provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
))
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url: resolved.upstream_url,
provider_request_method: Some(parts.method.to_string()),
auth_header: Some(resolved.auth_header),
auth_value: Some(resolved.auth_value),
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: resolved.requested_model,
mapped_model: resolved.mapped_model,
prompt_cache_key: None,
provider_request_headers: resolved.provider_request_headers,
provider_request_body: Some(resolved.provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -1,26 +1,30 @@
use std::collections::BTreeMap;
use std::sync::Arc;
use aether_contracts::ResolvedTransportProfile;
use serde_json::Value;
use crate::ai_serving::planner::candidate_preparation::{
prepare_header_authenticated_candidate, OauthPreparationContext,
};
use crate::ai_serving::planner::spec_metadata::local_openai_image_spec_metadata;
use crate::ai_serving::pure::normalize_openai_image_request_with_options;
use crate::ai_serving::transport::{
build_openai_image_headers, build_openai_image_upstream_url,
build_standard_provider_request_headers, openai_image_transport_unsupported_reason,
resolve_openai_image_auth, ProviderOpenAiImageHeadersInput,
StandardProviderRequestHeadersInput,
build_grok_browser_headers, build_grok_upstream_url, build_openai_image_headers,
build_openai_image_upstream_url, build_standard_provider_request_headers,
openai_image_transport_unsupported_reason, resolve_openai_image_auth, GrokHeaderInput,
ProviderOpenAiImageHeadersInput, StandardProviderRequestHeadersInput, GROK_CHAT_PATH,
};
use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
build_chatgpt_web_image_request_body,
apply_codex_openai_special_headers, build_chatgpt_web_image_request_body,
build_codex_openai_image_api_provider_request_body,
build_gemini_image_request_body_from_openai_image_request,
build_openai_image_provider_request_body, default_model_for_openai_image_operation,
normalize_openai_image_request, request_conversion_direct_auth, CandidateFailureDiagnostic,
GatewayProviderTransportSnapshot, PlannerAppState, RequestConversionKind,
build_openai_image_api_provider_request_body, build_openai_image_provider_request_body,
default_model_for_openai_image_operation, normalize_openai_image_request,
request_conversion_direct_auth, CandidateFailureDiagnostic, GatewayProviderTransportSnapshot,
PlannerAppState, RequestConversionKind,
};
use crate::image_capabilities::openai_image_normalize_options_for_provider;
use crate::AppState;
use super::support::{
@@ -43,6 +47,8 @@ pub(super) struct LocalOpenAiImageCandidatePayloadParts {
pub(super) provider_request_body: Value,
pub(super) upstream_url: String,
pub(super) input_summary: Value,
pub(super) transport_profile: Option<ResolvedTransportProfile>,
pub(super) upstream_is_stream: bool,
}
pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
@@ -59,6 +65,7 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
let candidate = &attempt.eligible.candidate;
let transport = &attempt.eligible.transport;
let provider_api_format = attempt.eligible.provider_api_format.as_str();
let effective_headers = input.effective_headers(&parts.headers);
if provider_api_format == "gemini:generate_content" {
return resolve_local_openai_image_to_gemini_candidate_payload_parts(
@@ -120,8 +127,16 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
let auth_header = prepared_candidate.auth_header;
let auth_value = prepared_candidate.auth_value;
let Some(normalized_request) = normalize_openai_image_request(parts, body_json, body_base64)
else {
let normalized_request = normalize_openai_image_request_with_options(
parts,
body_json,
body_base64,
openai_image_normalize_options_for_provider(
&transport.provider.provider_type,
Some(prepared_candidate.mapped_model.as_str()),
),
);
let Some(normalized_request) = normalized_request else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
state,
input,
@@ -145,39 +160,103 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
.provider_type
.trim()
.eq_ignore_ascii_case("chatgpt_web");
let is_grok = transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("grok");
let is_codex = transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("codex");
let transport_profile = crate::ai_serving::transport::resolve_transport_profile(transport);
let upstream_url = if is_chatgpt_web {
chatgpt_web_image_internal_url(&transport.endpoint.base_url)
} else if is_grok {
build_grok_upstream_url(transport, GROK_CHAT_PATH)
} else {
build_openai_image_upstream_url(transport, parts.uri.query())
build_openai_image_upstream_url(transport, Some(parts.uri.path()), parts.uri.query())
};
let mut provider_request_body = if is_chatgpt_web {
match build_chatgpt_web_image_request_body(parts, body_json, body_base64) {
Ok(body) => body,
Err(err) => err.to_error_json(),
}
} else {
build_openai_image_provider_request_body(&normalized_request)
};
if !is_chatgpt_web {
apply_codex_openai_responses_special_body_edits(
&mut provider_request_body,
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
spec_metadata.api_format,
transport.endpoint.body_rules.as_ref(),
Some(candidate.key_id.as_str()),
spec_metadata.require_streaming && candidate.supports_streaming,
false,
);
}
let Some(mut provider_request_headers) =
build_openai_image_headers(ProviderOpenAiImageHeadersInput {
headers: &parts.headers,
auth_header: &auth_header,
auth_value: &auth_value,
let provider_request_body = if is_chatgpt_web {
Some(
match build_chatgpt_web_image_request_body(parts, body_json, body_base64) {
Ok(body) => body,
Err(err) => err.to_error_json(),
},
)
} else if is_codex {
build_codex_openai_image_api_provider_request_body(
&normalized_request,
Some(prepared_candidate.mapped_model.as_str()),
upstream_is_stream,
)
} else if is_grok {
Some(build_openai_image_provider_request_body(
&normalized_request,
))
} else {
build_openai_image_api_provider_request_body(
&normalized_request,
Some(prepared_candidate.mapped_model.as_str()),
upstream_is_stream,
)
};
let Some(provider_request_body) = provider_request_body else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
state,
input,
trace_id,
candidate,
attempt.candidate_index,
&attempt.candidate_id,
"provider_request_body_missing",
CandidateFailureDiagnostic::provider_request_body_missing(
spec_metadata.api_format,
spec_metadata.api_format,
"codex_openai_images_request_contract",
),
)
.await;
return None;
};
let Some(mut provider_request_headers) = (if is_grok {
build_grok_browser_headers(GrokHeaderInput {
transport,
transport_profile: transport_profile.as_ref(),
request_headers: Some(effective_headers),
content_type: "application/json",
accept: "*/*",
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &provider_request_body,
original_request_body: body_json,
})
else {
} else {
build_openai_image_headers(ProviderOpenAiImageHeadersInput {
transport,
headers: effective_headers,
auth_header: &auth_header,
auth_value: &auth_value,
accept: if is_codex {
None
} else if upstream_is_stream {
Some("text/event-stream")
} else {
Some("application/json")
},
header_rules: transport.endpoint.header_rules.as_ref(),
provider_request_body: &provider_request_body,
original_request_body: body_json,
})
}) else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
state,
input,
@@ -197,11 +276,12 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
};
if is_chatgpt_web {
provider_request_headers.insert("x-aether-chatgpt-web-image".to_string(), "1".to_string());
} else if is_grok {
} else {
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
&parts.headers,
effective_headers,
transport.provider.provider_type.as_str(),
spec_metadata.api_format,
Some(trace_id),
@@ -222,7 +302,7 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
.unwrap_or_default()
.to_string();
let input_summary = if is_chatgpt_web {
let input_summary = if is_chatgpt_web || is_grok {
provider_request_body.clone()
} else {
normalized_request.summary_json
@@ -239,6 +319,8 @@ pub(super) async fn resolve_local_openai_image_candidate_payload_parts(
provider_request_body,
upstream_url,
input_summary,
transport_profile,
upstream_is_stream,
})
}
@@ -256,6 +338,7 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
let candidate = &attempt.eligible.candidate;
let transport = &attempt.eligible.transport;
let provider_api_format = "gemini:generate_content";
let effective_headers = input.effective_headers(&parts.headers);
let prepared_candidate = match prepare_header_authenticated_candidate(
PlannerAppState::new(state),
@@ -332,7 +415,7 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
converted.body_json,
transport.endpoint.body_rules.as_ref(),
body_json,
&parts.headers,
effective_headers,
) {
Some(body) => body,
None => {
@@ -354,13 +437,21 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
return None;
}
};
let upstream_is_stream = spec_metadata.require_streaming;
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
provider_api_format,
spec_metadata.require_streaming && candidate.supports_streaming,
false,
);
let Some(upstream_url) = crate::ai_serving::planner::standard::build_standard_upstream_url(
parts,
transport,
&converted.mapped_model,
provider_api_format,
upstream_is_stream,
Some(&converted.body_json),
) else {
mark_skipped_local_openai_image_candidate_with_failure_diagnostic(
state,
@@ -384,7 +475,7 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
transport,
provider_api_format,
same_format: false,
headers: &parts.headers,
headers: effective_headers,
auth_header: &prepared_candidate.auth_header,
auth_value: &prepared_candidate.auth_value,
extra_headers: &BTreeMap::new(),
@@ -423,6 +514,8 @@ async fn resolve_local_openai_image_to_gemini_candidate_payload_parts(
provider_request_body: converted.body_json,
upstream_url,
input_summary: converted.summary_json,
transport_profile: None,
upstream_is_stream,
})
}
@@ -13,6 +13,7 @@ use crate::ai_serving::planner::candidate_metadata::{
use crate::ai_serving::planner::candidate_resolution::SkippedLocalExecutionCandidate;
use crate::ai_serving::planner::candidate_source::auth_snapshot_allows_cross_format_candidate;
use crate::ai_serving::planner::decision_input::{
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::planner::materialization_policy::{
@@ -42,37 +43,59 @@ pub(super) async fn resolve_local_openai_image_decision_input(
body_base64: Option<&str>,
trace_id: &str,
decision: &GatewayControlDecision,
) -> Option<LocalOpenAiImageDecisionInput> {
) -> Result<Option<LocalOpenAiImageDecisionInput>, GatewayError> {
let Some(auth_context) = resolve_local_openai_image_auth_context(decision) else {
return None;
return Ok(None);
};
let requested_model = resolve_requested_image_model_for_request(parts, body_json, body_base64)?;
let Some(requested_model) =
resolve_requested_image_model_for_request(parts, body_json, body_base64)
else {
return Ok(None);
};
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
Ok(Some(resolved_input)) => resolved_input,
Ok(None) => return None,
Ok(None) => return Ok(None),
Err(err) => {
warn!(
trace_id = %trace_id,
error = ?err,
"gateway local openai image decision auth snapshot read failed"
);
return None;
return Err(err);
}
};
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
Some(input)
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
"openai:image",
)
.await
{
warn!(
trace_id = %trace_id,
error = ?err,
"gateway local openai image decision routing profile resolution failed"
);
return Err(err);
}
Ok(Some(input))
}
fn resolve_local_openai_image_auth_context(
@@ -103,6 +126,7 @@ pub(super) async fn list_local_openai_image_candidate_attempts(
matches_client_format.then_some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -123,8 +147,8 @@ pub(super) async fn list_local_openai_image_candidate_attempts(
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
candidate,
false,
)
});
}
@@ -176,6 +200,7 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
matches_client_format.then_some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -185,16 +210,16 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
candidate,
false,
)
});
format_skipped.retain(|candidate| {
auth_snapshot_allows_cross_format_candidate(
&input.auth_snapshot,
&input.requested_model,
None,
&candidate.candidate,
false,
)
});
}
@@ -229,6 +254,7 @@ pub(super) async fn build_local_openai_image_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -305,6 +331,7 @@ async fn materialize_local_openai_image_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -27,7 +27,7 @@ use self::support::{
pub(crate) struct LocalVideoCreateSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
trace_id: &'a str,
input: LocalVideoCreateDecisionInput,
spec: LocalVideoCreateSpec,
@@ -65,16 +65,17 @@ pub(crate) async fn build_local_video_sync_attempt_source_for_kind<'a>(
let Some(input) = resolve_local_video_create_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
return Ok(None);
};
let effective_body_json = input.effective_body_json(body_json).clone();
let Some((candidates, candidate_count)) = build_local_video_create_candidate_attempt_source(
state,
trace_id,
&input,
body_json,
&effective_body_json,
spec_metadata.api_format,
spec_metadata.decision_kind,
)
@@ -91,7 +92,7 @@ pub(crate) async fn build_local_video_sync_attempt_source_for_kind<'a>(
LocalVideoCreateSyncAttemptSource {
state,
parts,
body_json,
body_json: effective_body_json,
trace_id,
input,
spec,
@@ -104,7 +105,7 @@ pub(crate) async fn build_local_video_sync_attempt_source_for_kind<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalVideoCreateSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -122,6 +123,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalVideoCreateSyncAttemptS
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalVideoCreateSyncAttemptSource<'_> {
@@ -133,13 +149,13 @@ impl LocalVideoCreateSyncAttemptSource<'_> {
let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
self.state,
self.parts,
self.body_json,
&self.body_json,
self.trace_id,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
@@ -175,10 +191,11 @@ pub(crate) async fn maybe_build_sync_local_video_decision_payload(
let Some(input) = resolve_local_video_create_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let Some((mut source, _)) = build_local_video_create_candidate_attempt_source(
state,
@@ -193,11 +210,11 @@ pub(crate) async fn maybe_build_sync_local_video_decision_payload(
return Ok(None);
};
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
state, parts, body_json, trace_id, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -218,10 +235,11 @@ async fn build_local_sync_plan_and_reports(
let Some(input) = resolve_local_video_create_decision_input(
state, parts, trace_id, decision, body_json, spec,
)
.await
.await?
else {
return Ok(Vec::new());
};
let body_json = input.effective_body_json(body_json);
let Some((mut source, _)) = build_local_video_create_candidate_attempt_source(
state,
@@ -237,11 +255,11 @@ async fn build_local_sync_plan_and_reports(
};
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_video_create_decision_payload_for_candidate(
state, parts, body_json, trace_id, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -1,16 +1,18 @@
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_video_create_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::ai_serving::{ai_local_execution_contract_for_formats, PlannerAppState};
use crate::{AiExecutionDecision, AppState};
use crate::{append_local_failover_policy_to_value, AiExecutionDecision, AppState, GatewayError};
use super::request::resolve_local_video_create_candidate_payload_parts;
use super::support::{LocalVideoCreateCandidateAttempt, LocalVideoCreateDecisionInput};
@@ -24,14 +26,17 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
input: &LocalVideoCreateDecisionInput,
attempt: LocalVideoCreateCandidateAttempt,
spec: LocalVideoCreateSpec,
) -> Option<AiExecutionDecision> {
) -> Result<Option<AiExecutionDecision>, GatewayError> {
let spec_metadata = local_video_create_spec_metadata(spec);
let planner_state = PlannerAppState::new(state);
let attempt_identity = attempt.attempt_identity();
let resolved = resolve_local_video_create_candidate_payload_parts(
let Some(resolved) = resolve_local_video_create_candidate_payload_parts(
state, parts, body_json, trace_id, input, &attempt, spec,
)
.await?;
.await
else {
return Ok(None);
};
let LocalVideoCreateCandidateAttempt {
eligible,
candidate_id,
@@ -50,6 +55,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
if let Some(proxy_value) = build_request_trace_proxy_value(Some(&transport), proxy.as_ref()) {
extra_fields.insert("proxy".to_string(), proxy_value);
}
let effective_headers = input.effective_headers(&parts.headers);
let report_context = build_local_execution_report_context(LocalExecutionReportContextParts {
auth_context: &input.auth_context,
request_id: trace_id,
@@ -75,13 +81,14 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
body_rules: transport.endpoint.body_rules.as_ref(),
provider_request_method: None,
provider_request_headers: None,
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: false,
upstream_is_stream: false,
@@ -89,6 +96,7 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
needs_conversion: false,
extra_fields,
});
let report_context = append_local_failover_policy_to_value(report_context, &transport);
let super::request::LocalVideoCreateCandidatePayloadParts {
transport: _,
auth_header,
@@ -98,46 +106,54 @@ pub(super) async fn maybe_build_local_video_create_decision_payload_for_candidat
provider_request_body,
upstream_url,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream: false,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: Some(parts.method.to_string()),
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format: spec_metadata.api_format.to_string(),
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key: None,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: parts
.headers
.get(http::header::CONTENT_TYPE)
.and_then(|value| value.to_str().ok())
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream: false,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
))
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: false,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: Some(parts.method.to_string()),
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format: spec_metadata.api_format.to_string(),
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key: None,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: parts
.headers
.get(http::header::CONTENT_TYPE)
.and_then(|value| value.to_str().ok())
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts: resolve_transport_execution_timeouts(&transport),
upstream_is_stream: false,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
@@ -41,6 +41,7 @@ pub(super) async fn resolve_local_video_create_candidate_payload_parts(
let spec_metadata = local_video_create_spec_metadata(spec);
let candidate = &attempt.eligible.candidate;
let transport = &attempt.eligible.transport;
let effective_headers = input.effective_headers(&parts.headers);
let provider_family = provider_video_create_family(spec.family);
let transport_unsupported_reason = video_create_transport_unsupported_reason(
@@ -124,7 +125,7 @@ pub(super) async fn resolve_local_video_create_candidate_payload_parts(
provider_family,
&mapped_model,
transport.endpoint.body_rules.as_ref(),
Some(&parts.headers),
Some(effective_headers),
) else {
mark_skipped_local_video_candidate_with_failure_diagnostic(
state,
@@ -146,7 +147,7 @@ pub(super) async fn resolve_local_video_create_candidate_payload_parts(
let Some(provider_request_headers) =
build_video_create_headers(ProviderVideoCreateHeadersInput {
headers: &parts.headers,
headers: effective_headers,
auth_header: &auth_header,
auth_value: &auth_value,
header_rules: transport.endpoint.header_rules.as_ref(),
@@ -15,6 +15,7 @@ use crate::ai_serving::planner::candidate_metadata::{
use crate::ai_serving::planner::candidate_resolution::SkippedLocalExecutionCandidate;
use crate::ai_serving::planner::common::extract_requested_model_from_request;
use crate::ai_serving::planner::decision_input::{
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::planner::materialization_policy::{
@@ -41,30 +42,34 @@ pub(super) async fn resolve_local_video_create_decision_input(
decision: &GatewayControlDecision,
body_json: &serde_json::Value,
spec: LocalVideoCreateSpec,
) -> Option<LocalVideoCreateDecisionInput> {
) -> Result<Option<LocalVideoCreateDecisionInput>, GatewayError> {
let spec_metadata = local_video_create_spec_metadata(spec);
let Some(auth_context) = resolve_local_video_create_auth_context(decision, spec.family) else {
return None;
return Ok(None);
};
let requested_model = extract_requested_model_from_request(
let Some(requested_model) = extract_requested_model_from_request(
parts,
body_json,
spec_metadata
.requested_model_family
.expect("video specs should declare requested-model family"),
)?;
) else {
return Ok(None);
};
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
Ok(Some(resolved_input)) => resolved_input,
Ok(None) => return None,
Ok(None) => return Ok(None),
Err(err) => {
warn!(
trace_id = %trace_id,
@@ -72,14 +77,31 @@ pub(super) async fn resolve_local_video_create_decision_input(
error = ?err,
"gateway local video decision auth snapshot read failed"
);
return None;
return Err(err);
}
};
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
Some(input)
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
spec_metadata.api_format,
)
.await
{
warn!(
trace_id = %trace_id,
decision_kind = spec_metadata.decision_kind,
error = ?err,
"gateway local video decision routing profile resolution failed"
);
return Err(err);
}
Ok(Some(input))
}
fn resolve_local_video_create_auth_context(
@@ -110,6 +132,7 @@ pub(super) async fn list_local_video_create_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -166,6 +189,7 @@ pub(super) async fn build_local_video_create_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
current_unix_secs(),
false,
)
.await
{
@@ -196,6 +220,7 @@ pub(super) async fn build_local_video_create_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -261,6 +286,7 @@ async fn materialize_local_video_create_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -3,5 +3,43 @@
mod tests;
pub(crate) use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_identity_headers, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_special_headers,
};
pub(crate) fn codex_model_capabilities_for_transport(
transport: &crate::ai_serving::GatewayProviderTransportSnapshot,
provider_api_format: &str,
provider_model: &str,
source_model: &str,
) -> Option<crate::ai_serving::CodexResponsesModelCapabilities> {
codex_model_capabilities(
&transport.provider.provider_type,
provider_api_format,
provider_model,
source_model,
transport.key.upstream_metadata.as_ref(),
)
}
fn codex_model_capabilities(
provider_type: &str,
provider_api_format: &str,
provider_model: &str,
source_model: &str,
upstream_metadata: Option<&serde_json::Value>,
) -> Option<crate::ai_serving::CodexResponsesModelCapabilities> {
let uses_codex_model_catalog =
crate::ai_serving::is_openai_responses_family_format(provider_api_format)
|| crate::ai_serving::api_format_alias_matches(provider_api_format, "openai:search");
if !provider_type.trim().eq_ignore_ascii_case("codex") || !uses_codex_model_catalog {
return None;
}
Some(
crate::ai_serving::resolve_codex_responses_model_capabilities(
provider_model,
source_model,
upstream_metadata,
),
)
}
@@ -1,15 +1,58 @@
use std::collections::BTreeMap;
use super::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_responses_special_headers,
apply_codex_openai_responses_identity_headers, apply_codex_openai_responses_special_body_edits,
apply_codex_openai_special_headers, codex_model_capabilities,
};
use crate::ai_serving::planner::standard::{
build_cross_format_openai_responses_request_body, build_local_openai_responses_request_body,
};
use http::{HeaderMap, HeaderValue};
use serde_json::json;
#[test]
fn search_uses_live_codex_model_catalog_capabilities() {
let metadata = crate::ai_serving::build_codex_model_catalog_metadata(&[json!({
"slug": "gpt-search-custom",
"default_reasoning_level": "low",
"supported_reasoning_levels": [
{"effort": "low"},
{"effort": "max"}
],
"supports_parallel_tool_calls": true
})]);
let capabilities = codex_model_capabilities(
"codex",
"openai:search",
"gpt-search-custom",
"gpt-search-custom",
Some(&metadata),
)
.expect("Search should resolve capabilities from the Codex model catalog");
assert_eq!(
capabilities.default_reasoning_effort.as_deref(),
Some("low")
);
assert_eq!(
capabilities.supported_reasoning_efforts,
vec!["low".to_string(), "max".to_string()]
);
assert!(codex_model_capabilities(
"codex",
"openai:chat",
"gpt-search-custom",
"gpt-search-custom",
Some(&metadata),
)
.is_none());
}
#[test]
fn applies_codex_defaults_when_body_rules_do_not_handle_fields() {
let mut body = json!({
"model": "gpt-5",
"model": "gpt-5.4",
"max_output_tokens": 128,
"temperature": 0.3,
"top_p": 0.9,
@@ -30,10 +73,45 @@ fn applies_codex_defaults_when_body_rules_do_not_handle_fields() {
assert!(body.get("top_p").is_none());
assert!(body.get("metadata").is_none());
assert_eq!(body["store"], false);
assert_eq!(body["instructions"], "");
assert!(body.get("instructions").is_none());
assert_eq!(body["include"], json!(["reasoning.encrypted_content"]));
assert_eq!(body["parallel_tool_calls"], true);
assert!(body.get("reasoning").is_none());
assert_eq!(body["reasoning"]["effort"], "medium");
assert!(body["reasoning"].get("summary").is_none());
}
#[test]
fn local_openai_responses_codex_body_wraps_string_input_for_backend() {
let body = json!({
"model": "gpt-5",
"input": "hello"
});
let provider_request_body = build_local_openai_responses_request_body(
&body,
"gpt-5-upstream",
false,
false,
"codex",
"openai:responses",
None,
Some("key-123"),
&HeaderMap::new(),
false,
)
.expect("codex local openai responses body should build");
assert_eq!(
provider_request_body["input"],
json!([{
"type": "message",
"role": "user",
"content": [{
"type": "input_text",
"text": "hello"
}]
}])
);
}
#[test]
@@ -45,7 +123,7 @@ fn strips_store_for_compact_even_when_body_rules_handle_it() {
{"action":"set","path":"top_p","value":0.5}
]);
let mut body = json!({
"model": "gpt-5",
"model": "gpt-5.4",
"max_output_tokens": 128,
"metadata": {"client": "desktop", "mode": "custom"},
"store": true,
@@ -64,12 +142,29 @@ fn strips_store_for_compact_even_when_body_rules_handle_it() {
assert!(body.get("max_output_tokens").is_none());
assert!(body.get("store").is_none());
assert_eq!(body["instructions"], "Keep custom");
assert_eq!(body["metadata"]["mode"], "custom");
assert_eq!(body["top_p"], 0.5);
assert!(body.get("metadata").is_none());
assert!(body.get("top_p").is_none());
assert_eq!(body["parallel_tool_calls"], true);
assert!(body.as_object().is_some_and(|object| {
object.keys().all(|field| {
matches!(
field.as_str(),
"model"
| "input"
| "instructions"
| "tools"
| "parallel_tool_calls"
| "reasoning"
| "service_tier"
| "prompt_cache_key"
| "text"
)
})
}));
}
#[test]
fn injects_stable_prompt_cache_key_for_codex_requests() {
fn does_not_synthesize_prompt_cache_key_from_api_key_identity() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
@@ -83,83 +178,478 @@ fn injects_stable_prompt_cache_key_for_codex_requests() {
Some("key-123"),
);
assert!(body.get("prompt_cache_key").is_none());
}
#[test]
fn adapts_generic_prompt_cache_key_to_codex_native_identity() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
"prompt_cache_key": "ltm-pc-v2-5557e02f5c9b447a97673ba330dbe77a",
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
let expected_identity = "d9c5d122-7c1c-5fb1-ba9d-656062eda44e";
assert_eq!(body["prompt_cache_key"], expected_identity);
assert_eq!(body["client_metadata"]["session_id"], expected_identity);
assert_eq!(body["client_metadata"]["thread_id"], expected_identity);
}
#[test]
fn preserves_native_codex_cache_identity_and_metadata() {
let mut body = json!({
"model": "gpt-5",
"input": "hello",
"prompt_cache_key": "guardian:parent-thread",
"client_metadata": {
"session_id": "native-session",
"thread_id": "native-thread",
"turn_id": "native-turn"
}
});
let expected = body.clone();
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
assert_eq!(body["prompt_cache_key"], expected["prompt_cache_key"]);
assert_eq!(body["client_metadata"], expected["client_metadata"]);
}
#[test]
fn preserves_uuid_prompt_cache_key_while_completing_codex_identity() {
let identity = "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3";
let mut body = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": identity
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
None,
);
assert_eq!(body["prompt_cache_key"], identity);
assert_eq!(body["client_metadata"]["session_id"], identity);
assert_eq!(body["client_metadata"]["thread_id"], identity);
}
#[test]
fn keeps_codex_prompt_cache_domains_distinct() {
let mut first = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "tenant-a"
});
let mut second = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "tenant-b"
});
for body in [&mut first, &mut second] {
apply_codex_openai_responses_special_body_edits(
body,
"codex",
"openai:responses",
None,
None,
);
}
assert_ne!(first["prompt_cache_key"], second["prompt_cache_key"]);
assert_eq!(
body["prompt_cache_key"],
"172c39e6-c0a0-5a70-8b63-e0f8e0d185a3"
first["prompt_cache_key"],
first["client_metadata"]["session_id"]
);
assert_eq!(
second["prompt_cache_key"],
second["client_metadata"]["session_id"]
);
}
#[test]
fn keeps_existing_prompt_cache_key_for_codex_requests() {
let mut body = json!({
"model": "gpt-5",
fn completes_partial_and_null_codex_client_metadata() {
let mut partial = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "existing-key",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"thread_id": "native-thread",
"caller": "sdk"
}
});
let mut null_metadata = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": null
});
let mut null_session = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"session_id": null,
"thread_id": null,
"caller": "sdk"
}
});
apply_codex_openai_responses_special_body_edits(
&mut body,
"codex",
"openai:responses",
None,
Some("key-123"),
);
for body in [&mut partial, &mut null_metadata, &mut null_session] {
apply_codex_openai_responses_special_body_edits(
body,
"codex",
"openai:responses",
None,
None,
);
}
assert_eq!(body["prompt_cache_key"], "existing-key");
assert_eq!(partial["client_metadata"]["thread_id"], "native-thread");
assert_eq!(partial["client_metadata"]["caller"], "sdk");
assert_eq!(
partial["client_metadata"]["session_id"],
partial["prompt_cache_key"]
);
assert_eq!(
null_metadata["client_metadata"]["session_id"],
null_metadata["prompt_cache_key"]
);
assert_eq!(
null_metadata["client_metadata"]["thread_id"],
null_metadata["prompt_cache_key"]
);
assert_eq!(
null_session["client_metadata"]["session_id"],
null_session["prompt_cache_key"]
);
assert_eq!(
null_session["client_metadata"]["thread_id"],
null_session["prompt_cache_key"]
);
assert_eq!(null_session["client_metadata"]["caller"], "sdk");
}
#[test]
fn injects_chatgpt_account_id_and_session_headers_for_codex_requests() {
fn leaves_malformed_codex_client_metadata_unchanged() {
let mut body = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": "invalid"
});
let mut malformed_fields = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity",
"client_metadata": {
"session_id": 42,
"thread_id": ""
}
});
let expected_malformed_metadata = malformed_fields["client_metadata"].clone();
for candidate in [&mut body, &mut malformed_fields] {
apply_codex_openai_responses_special_body_edits(
candidate,
"codex",
"openai:responses",
None,
None,
);
}
assert_eq!(body["prompt_cache_key"], "generic-affinity");
assert_eq!(body["client_metadata"], "invalid");
assert_eq!(malformed_fields["prompt_cache_key"], "generic-affinity");
assert_eq!(
malformed_fields["client_metadata"],
expected_malformed_metadata
);
}
#[test]
fn limits_prompt_cache_identity_adaptation_to_codex_responses_family() {
let original = json!({
"model": "gpt-5.6-luna",
"input": "hello",
"prompt_cache_key": "generic-affinity"
});
let mut standard_openai = original.clone();
let mut codex_compact = original.clone();
apply_codex_openai_responses_special_body_edits(
&mut standard_openai,
"openai",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_special_body_edits(
&mut codex_compact,
"codex",
"openai:responses:compact",
None,
None,
);
assert_eq!(standard_openai, original);
assert_ne!(codex_compact["prompt_cache_key"], "generic-affinity");
assert!(codex_compact.get("client_metadata").is_none());
}
#[test]
fn chat_to_codex_responses_adapts_prompt_cache_identity_end_to_end() {
let body = json!({
"model": "gpt-5.6-luna",
"messages": [{"role": "user", "content": "hello"}],
"prompt_cache_key": "ltm-pc-v2-5557e02f5c9b447a97673ba330dbe77a"
});
let provider_request_body = build_cross_format_openai_responses_request_body(
&body,
"gpt-5.6-luna",
"openai:chat",
"openai:responses",
true,
false,
"codex",
None,
None,
&HeaderMap::new(),
false,
)
.expect("chat to Codex Responses request should build");
let expected_identity = "d9c5d122-7c1c-5fb1-ba9d-656062eda44e";
assert_eq!(provider_request_body["prompt_cache_key"], expected_identity);
assert_eq!(
provider_request_body["client_metadata"]["session_id"],
expected_identity
);
assert_eq!(
provider_request_body["client_metadata"]["thread_id"],
expected_identity
);
let mut provider_request_headers = BTreeMap::new();
apply_codex_openai_special_headers(
&mut provider_request_headers,
&provider_request_body,
&HeaderMap::new(),
"codex",
"openai:responses",
Some("trace-codex-cache-identity"),
None,
);
apply_codex_openai_responses_identity_headers(
&mut provider_request_headers,
&provider_request_body,
"codex",
"openai:responses",
);
assert_eq!(
provider_request_headers
.get("session-id")
.map(String::as_str),
Some(expected_identity)
);
assert_eq!(
provider_request_headers
.get("thread-id")
.map(String::as_str),
Some(expected_identity)
);
}
#[test]
fn projects_uuid_prompt_cache_identity_into_missing_session_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
});
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
Some("trace-codex-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(headers.get("x-client-request-id"), None);
assert_eq!(
headers.get("user-agent"),
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
}
#[test]
fn projects_native_codex_metadata_for_non_uuid_cache_overrides() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "guardian:parent-thread",
"client_metadata": {
"session_id": "019f687b-8e92-7842-9631-d5bf0dba0a3b",
"thread_id": "019f6d20-1111-7222-8333-444455556666"
}
});
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("session-id").map(String::as_str),
Some("019f687b-8e92-7842-9631-d5bf0dba0a3b")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("019f6d20-1111-7222-8333-444455556666")
);
}
#[test]
fn leaves_non_native_cache_keys_out_of_identity_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "generic-cache-key"
});
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses",
None,
None,
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert!(!headers.contains_key("session-id"));
assert!(!headers.contains_key("thread-id"));
}
#[test]
fn leaves_malformed_native_metadata_out_of_identity_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5.6-luna",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
"client_metadata": {
"session_id": 42,
"thread_id": ""
}
});
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert!(!headers.contains_key("session-id"));
assert!(!headers.contains_key("thread-id"));
}
#[test]
fn injects_only_codex_client_headers_for_images_requests() {
let mut headers = BTreeMap::new();
apply_codex_openai_special_headers(
&mut headers,
&json!({
"model": "gpt-image-2",
"prompt": "draw a city"
}),
&HeaderMap::new(),
"codex",
"openai:image",
Some("trace-codex-image-123"),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(
headers.get("x-client-request-id"),
Some(&"trace-codex-123".to_string())
);
assert_eq!(
headers.get("user-agent"),
Some(
&"codex-tui/0.122.0 (Mac OS 15.2.0; arm64) vscode/2.6.11 (codex-tui; 0.122.0)"
.to_string()
)
);
assert_eq!(headers.get("originator"), Some(&"codex-tui".to_string()));
assert_eq!(
headers.get("session_id"),
Some(&"ab5ecce4f0d110fe".to_string())
);
assert_eq!(
headers.get("conversation_id"),
Some(&"ab5ecce4f0d110fe".to_string())
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
for name in ["x-client-request-id", "session-id", "thread-id"] {
assert!(
!headers.contains_key(name),
"unexpected Images header: {name}"
);
}
}
#[test]
fn respects_existing_codex_request_and_session_headers() {
fn preserves_client_context_headers_and_enforces_codex_provider_identity() {
let mut headers = BTreeMap::new();
headers.insert(
"x-client-request-id".to_string(),
"kept-by-rule-request".to_string(),
);
headers.insert("session_id".to_string(), "kept-by-rule".to_string());
headers.insert("session-id".to_string(), "kept-by-rule-session".to_string());
headers.insert("thread-id".to_string(), "kept-by-rule-thread".to_string());
headers.insert(
"chatgpt-account-id".to_string(),
"configured-spoof".to_string(),
);
headers.insert(
"x-openai-fedramp".to_string(),
"configured-false".to_string(),
);
headers.insert(
"User-Agent".to_string(),
"AsyncOpenAI/Python 2.44.0".to_string(),
);
headers.insert("ORIGINATOR".to_string(), "sdk-client".to_string());
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
@@ -170,12 +660,12 @@ fn respects_existing_codex_request_and_session_headers() {
HeaderValue::from_static("user-specified-request"),
);
original_headers.insert(
"session_id",
"session-id",
HeaderValue::from_static("user-specified-session"),
);
original_headers.insert(
"conversation_id",
HeaderValue::from_static("user-specified-conversation"),
"thread-id",
HeaderValue::from_static("user-specified-thread"),
);
original_headers.insert(
"user-agent",
@@ -185,64 +675,105 @@ fn respects_existing_codex_request_and_session_headers() {
"originator",
HeaderValue::from_static("user-specified-originator"),
);
original_headers.insert("version", HeaderValue::from_static("user-version"));
original_headers.insert("x-openai-fedramp", HeaderValue::from_static("user-fedramp"));
original_headers.insert(
"chatgpt-account-id",
HeaderValue::from_static("user-account"),
);
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&original_headers,
"codex",
"openai:responses",
Some("trace-codex-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(&mut headers, &body, "codex", "openai:responses");
assert_eq!(
headers.get("x-client-request-id"),
Some(&"kept-by-rule-request".to_string())
);
assert!(!headers.contains_key("user-agent"));
assert!(!headers.contains_key("originator"));
assert_eq!(headers.get("session_id"), Some(&"kept-by-rule".to_string()));
assert!(!headers.contains_key("conversation_id"));
assert_eq!(
headers.get("user-agent"),
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("user-agent"))
.count(),
1
);
assert_eq!(
headers
.keys()
.filter(|name| name.eq_ignore_ascii_case("originator"))
.count(),
1
);
assert!(!headers.contains_key("version"));
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session-id"),
Some(&"kept-by-rule-session".to_string())
);
assert_eq!(
headers.get("thread-id"),
Some(&"kept-by-rule-thread".to_string())
);
}
#[test]
fn skips_conversation_id_for_compact_codex_requests() {
fn compact_projects_uuid_prompt_cache_identity_into_session_headers() {
let mut headers = BTreeMap::new();
let body = json!({
"model": "gpt-5",
"prompt_cache_key": "172c39e6-c0a0-5a70-8b63-e0f8e0d185a3",
});
apply_codex_openai_responses_special_headers(
apply_codex_openai_special_headers(
&mut headers,
&body,
&HeaderMap::new(),
"codex",
"openai:responses:compact",
Some("trace-codex-compact-123"),
Some(r#"{"account_id":"acc-123"}"#),
Some(r#"{"account_id":"acc-123","is_fedramp":true}"#),
);
apply_codex_openai_responses_identity_headers(
&mut headers,
&body,
"codex",
"openai:responses:compact",
);
assert_eq!(
headers.get("chatgpt-account-id"),
Some(&"acc-123".to_string())
);
assert_eq!(
headers.get("x-client-request-id"),
Some(&"trace-codex-compact-123".to_string())
);
assert_eq!(headers.get("x-client-request-id"), None);
assert_eq!(
headers.get("user-agent"),
Some(
&"codex-tui/0.122.0 (Mac OS 15.2.0; arm64) vscode/2.6.11 (codex-tui; 0.122.0)"
.to_string()
)
Some(&"codex_cli_rs/0.144.1".to_string())
);
assert_eq!(headers.get("originator"), Some(&"codex-tui".to_string()));
assert_eq!(headers.get("originator"), Some(&"codex_cli_rs".to_string()));
assert!(!headers.contains_key("version"));
assert_eq!(headers.get("x-openai-fedramp"), Some(&"true".to_string()));
assert_eq!(
headers.get("session_id"),
Some(&"ab5ecce4f0d110fe".to_string())
headers.get("session-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert_eq!(
headers.get("thread-id").map(String::as_str),
Some("172c39e6-c0a0-5a70-8b63-e0f8e0d185a3")
);
assert!(!headers.contains_key("conversation_id"));
}
@@ -0,0 +1,386 @@
use serde_json::{json, Value};
pub(crate) fn is_deepseek_provider(provider_type: &str, base_url: &str) -> bool {
let provider_type = provider_type.trim().to_ascii_lowercase();
if matches!(
provider_type.as_str(),
"deepseek" | "deepseek_openai" | "deepseek_anthropic" | "deepseek_compatible"
) {
return true;
}
let host = base_url_host(base_url);
host == "deepseek.com" || host.ends_with(".deepseek.com")
}
pub(crate) fn apply_deepseek_tool_call_thinking_compat(
provider_request_body: &mut Value,
provider_type: &str,
base_url: &str,
provider_api_format: &str,
original_request_body: Option<&Value>,
) {
if !is_deepseek_provider(provider_type, base_url) {
return;
}
match crate::ai_serving::normalize_api_format_alias(provider_api_format).as_str() {
"openai:chat" => {
apply_deepseek_openai_chat_thinking_compat(provider_request_body, original_request_body)
}
"claude:messages" => apply_deepseek_claude_messages_thinking_compat(
provider_request_body,
original_request_body,
),
_ => {}
}
}
fn base_url_host(base_url: &str) -> String {
let lower = base_url.trim().to_ascii_lowercase();
let without_scheme = lower
.split_once("://")
.map(|(_, rest)| rest)
.unwrap_or(lower.as_str());
let without_userinfo = without_scheme
.rsplit_once('@')
.map(|(_, host)| host)
.unwrap_or(without_scheme);
without_userinfo
.split(['/', '?', '#'])
.next()
.unwrap_or_default()
.split(':')
.next()
.unwrap_or_default()
.to_string()
}
fn source_disables_thinking(
original_request_body: Option<&Value>,
provider_request_body: &Value,
) -> bool {
request_explicitly_disables_thinking(provider_request_body)
|| original_request_body.is_some_and(request_explicitly_disables_thinking)
}
fn request_explicitly_disables_thinking(body: &Value) -> bool {
thinking_type(body).is_some_and(|value| value.eq_ignore_ascii_case("disabled"))
|| reasoning_effort(body).is_some_and(|value| value.eq_ignore_ascii_case("none"))
}
fn thinking_type(body: &Value) -> Option<&str> {
body.get("thinking")
.and_then(Value::as_object)
.and_then(|thinking| thinking.get("type"))
.and_then(Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
}
fn reasoning_effort(body: &Value) -> Option<&str> {
body.get("reasoning_effort")
.and_then(Value::as_str)
.or_else(|| {
body.get("reasoning")
.and_then(Value::as_object)
.and_then(|reasoning| reasoning.get("effort"))
.and_then(Value::as_str)
})
.map(str::trim)
.filter(|value| !value.is_empty())
}
fn set_deepseek_thinking_type(body: &mut Value, thinking_type: &str) {
let Some(object) = body.as_object_mut() else {
return;
};
match object.get_mut("thinking") {
Some(Value::Object(thinking)) => {
thinking.insert("type".to_string(), Value::String(thinking_type.to_string()));
}
_ => {
object.insert(
"thinking".to_string(),
json!({
"type": thinking_type,
}),
);
}
}
}
fn apply_deepseek_openai_chat_thinking_compat(
provider_request_body: &mut Value,
original_request_body: Option<&Value>,
) {
let disabled = source_disables_thinking(original_request_body, provider_request_body);
set_deepseek_thinking_type(
provider_request_body,
if disabled { "disabled" } else { "enabled" },
);
let Some(object) = provider_request_body.as_object_mut() else {
return;
};
if disabled {
if reasoning_effort(&Value::Object(object.clone()))
.is_some_and(|value| value.eq_ignore_ascii_case("none"))
{
object.remove("reasoning_effort");
}
return;
}
let Some(messages) = object.get_mut("messages").and_then(Value::as_array_mut) else {
return;
};
for message in messages {
let Some(message_object) = message.as_object_mut() else {
continue;
};
let is_assistant = message_object
.get("role")
.and_then(Value::as_str)
.is_some_and(|role| role.trim().eq_ignore_ascii_case("assistant"));
if !is_assistant {
continue;
}
if message_object
.get("reasoning_content")
.is_some_and(|value| !value.is_null())
{
continue;
}
message_object.insert(
"reasoning_content".to_string(),
Value::String(String::new()),
);
}
}
fn apply_deepseek_claude_messages_thinking_compat(
provider_request_body: &mut Value,
original_request_body: Option<&Value>,
) {
if source_disables_thinking(original_request_body, provider_request_body) {
set_deepseek_thinking_type(provider_request_body, "disabled");
return;
}
let Some(messages) = provider_request_body
.get_mut("messages")
.and_then(Value::as_array_mut)
else {
return;
};
for message in messages {
let Some(message_object) = message.as_object_mut() else {
continue;
};
let is_assistant = message_object
.get("role")
.and_then(Value::as_str)
.is_some_and(|role| role.trim().eq_ignore_ascii_case("assistant"));
if !is_assistant {
continue;
}
ensure_claude_assistant_message_has_thinking_block(message_object);
}
}
fn ensure_claude_assistant_message_has_thinking_block(
message: &mut serde_json::Map<String, Value>,
) {
let thinking_block = json!({
"type": "thinking",
"thinking": "",
});
match message.get_mut("content") {
Some(Value::Array(blocks)) => {
if blocks.iter().any(is_claude_thinking_block) {
return;
}
blocks.insert(0, thinking_block);
}
Some(Value::String(text)) => {
let text = std::mem::take(text);
message.insert(
"content".to_string(),
Value::Array(vec![
thinking_block,
json!({
"type": "text",
"text": text,
}),
]),
);
}
Some(Value::Null) | None => {
message.insert("content".to_string(), Value::Array(vec![thinking_block]));
}
Some(other) => {
let existing = std::mem::take(other);
message.insert(
"content".to_string(),
Value::Array(vec![thinking_block, existing]),
);
}
}
}
fn is_claude_thinking_block(block: &Value) -> bool {
block
.get("type")
.and_then(Value::as_str)
.is_some_and(|block_type| block_type.trim().eq_ignore_ascii_case("thinking"))
}
#[cfg(test)]
mod tests {
use serde_json::json;
use super::{apply_deepseek_tool_call_thinking_compat, is_deepseek_provider};
#[test]
fn detects_deepseek_provider_by_type_or_host() {
assert!(is_deepseek_provider(
"deepseek",
"https://relay.example.com"
));
assert!(is_deepseek_provider(
"custom",
"https://api.deepseek.com/v1"
));
assert!(!is_deepseek_provider(
"custom",
"https://example.com/deepseek"
));
}
#[test]
fn openai_chat_deepseek_adds_thinking_and_empty_reasoning_content() {
let mut body = json!({
"model": "deepseek-chat",
"messages": [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": null, "tool_calls": [{
"id": "call_1",
"type": "function",
"function": {"name": "lookup", "arguments": "{}"}
}]},
{"role": "tool", "tool_call_id": "call_1", "content": "{}"}
]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com/v1",
"openai:chat",
None,
);
assert_eq!(body["thinking"]["type"], "enabled");
assert_eq!(body["messages"][1]["reasoning_content"], "");
}
#[test]
fn openai_chat_deepseek_honors_disabled_thinking() {
let original = json!({"reasoning_effort": "none"});
let mut body = json!({
"model": "deepseek-chat",
"reasoning_effort": "none",
"messages": [
{"role": "assistant", "content": "hi"}
]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com/v1",
"openai:chat",
Some(&original),
);
assert_eq!(body["thinking"]["type"], "disabled");
assert!(body.get("reasoning_effort").is_none());
assert!(body["messages"][0].get("reasoning_content").is_none());
}
#[test]
fn claude_messages_deepseek_prepends_empty_thinking_block() {
let mut body = json!({
"model": "deepseek-3.2",
"messages": [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": [
{"type": "tool_use", "id": "call_1", "name": "lookup", "input": {}}
]}
]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com",
"claude:messages",
None,
);
assert_eq!(body["messages"][1]["content"][0]["type"], "thinking");
assert_eq!(body["messages"][1]["content"][0]["thinking"], "");
assert_eq!(body["messages"][1]["content"][1]["type"], "tool_use");
}
#[test]
fn claude_messages_deepseek_converts_string_assistant_content_to_blocks() {
let mut body = json!({
"model": "deepseek-3.2",
"messages": [{
"role": "assistant",
"content": "done"
}]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com",
"claude:messages",
None,
);
assert_eq!(body["messages"][0]["content"][0]["type"], "thinking");
assert_eq!(body["messages"][0]["content"][1]["type"], "text");
assert_eq!(body["messages"][0]["content"][1]["text"], "done");
}
#[test]
fn claude_messages_deepseek_preserves_existing_thinking_block() {
let mut body = json!({
"model": "deepseek-3.2",
"messages": [{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "plan", "signature": "sig"},
{"type": "text", "text": "answer"}
]
}]
});
apply_deepseek_tool_call_thinking_compat(
&mut body,
"deepseek",
"https://api.deepseek.com",
"claude:messages",
None,
);
assert_eq!(body["messages"][0]["content"].as_array().unwrap().len(), 2);
assert_eq!(body["messages"][0]["content"][0]["thinking"], "plan");
assert_eq!(body["messages"][0]["content"][0]["signature"], "sig");
}
}
@@ -29,7 +29,7 @@ pub(crate) struct LocalStandardSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalStandardDecisionInput,
spec: LocalStandardSpec,
requested_model_family: RequestedModelFamily,
@@ -40,7 +40,7 @@ pub(crate) struct LocalStandardStreamAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalStandardDecisionInput,
spec: LocalStandardSpec,
requested_model_family: RequestedModelFamily,
@@ -61,7 +61,7 @@ pub(crate) async fn build_local_sync_attempt_source<'a>(
.expect("standard spec metadata should include requested-model family");
let Some(input) =
resolve_local_standard_decision_input(state, parts, trace_id, decision, body_json, spec)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -82,9 +82,15 @@ pub(crate) async fn build_local_sync_attempt_source<'a>(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let (candidates, candidate_count) =
build_local_standard_candidate_attempt_source(state, trace_id, &input, body_json, spec)
.await?;
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_local_standard_candidate_attempt_source(
state,
trace_id,
&input,
&effective_body_json,
spec,
)
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
if candidate_count == 0 {
return Ok(None);
@@ -95,7 +101,7 @@ pub(crate) async fn build_local_sync_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
spec,
requested_model_family,
@@ -119,7 +125,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
.expect("standard spec metadata should include requested-model family");
let Some(input) =
resolve_local_standard_decision_input(state, parts, trace_id, decision, body_json, spec)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -140,9 +146,15 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let (candidates, candidate_count) =
build_local_standard_candidate_attempt_source(state, trace_id, &input, body_json, spec)
.await?;
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_local_standard_candidate_attempt_source(
state,
trace_id,
&input,
&effective_body_json,
spec,
)
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
if candidate_count == 0 {
return Ok(None);
@@ -153,7 +165,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
spec,
requested_model_family,
@@ -166,7 +178,7 @@ pub(crate) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalStandardSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -189,12 +201,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalStandardSyncAttemptSour
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalStandardStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -217,6 +244,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalStandardStreamAttempt
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalStandardSyncAttemptSource<'_> {
@@ -228,19 +270,19 @@ impl LocalStandardSyncAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
match build_sync_plan_from_requested_model_family(
self.requested_model_family,
self.parts,
self.body_json,
&self.body_json,
payload,
) {
Ok(value) => Ok(value),
@@ -265,19 +307,19 @@ impl LocalStandardStreamAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
match build_stream_plan_from_requested_model_family(
self.requested_model_family,
self.parts,
self.body_json,
&self.body_json,
payload,
) {
Ok(value) => Ok(value),
@@ -309,7 +351,7 @@ pub(crate) async fn maybe_build_sync_via_standard_family_payload(
let Some(input) =
resolve_local_standard_decision_input(state, parts, trace_id, decision, body_json, spec)
.await
.await?
else {
return Ok(None);
};
@@ -322,16 +364,17 @@ pub(crate) async fn maybe_build_sync_via_standard_family_payload(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) =
build_local_standard_candidate_attempt_source(state, trace_id, &input, body_json, spec)
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -358,7 +401,7 @@ pub(crate) async fn maybe_build_stream_via_standard_family_payload(
let Some(input) =
resolve_local_standard_decision_input(state, parts, trace_id, decision, body_json, spec)
.await
.await?
else {
return Ok(None);
};
@@ -371,16 +414,17 @@ pub(crate) async fn maybe_build_stream_via_standard_family_payload(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) =
build_local_standard_candidate_attempt_source(state, trace_id, &input, body_json, spec)
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -405,7 +449,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
.expect("standard spec metadata should include requested-model family");
let Some(input) =
resolve_local_standard_decision_input(state, parts, trace_id, decision, body_json, spec)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -426,6 +470,7 @@ pub(crate) async fn build_local_sync_plan_and_reports(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) =
build_local_standard_candidate_attempt_source(state, trace_id, &input, body_json, spec)
.await?;
@@ -434,11 +479,11 @@ pub(crate) async fn build_local_sync_plan_and_reports(
return Ok(Vec::new());
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -479,7 +524,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
.expect("standard spec metadata should include requested-model family");
let Some(input) =
resolve_local_standard_decision_input(state, parts, trace_id, decision, body_json, spec)
.await
.await?
else {
set_local_runtime_miss_diagnostic_reason(
state,
@@ -500,6 +545,7 @@ pub(crate) async fn build_local_stream_plan_and_reports(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let body_json = input.effective_body_json(body_json);
let (mut source, candidate_count) =
build_local_standard_candidate_attempt_source(state, trace_id, &input, body_json, spec)
.await?;
@@ -508,11 +554,11 @@ pub(crate) async fn build_local_stream_plan_and_reports(
return Ok(Vec::new());
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_standard_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -16,6 +16,7 @@ use crate::ai_serving::planner::candidate_source::{
};
use crate::ai_serving::planner::common::extract_requested_model_from_request;
use crate::ai_serving::planner::decision_input::{
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::planner::materialization_policy::{
@@ -39,30 +40,34 @@ pub(super) async fn resolve_local_standard_decision_input(
decision: &GatewayControlDecision,
body_json: &serde_json::Value,
spec: LocalStandardSpec,
) -> Option<LocalStandardDecisionInput> {
) -> Result<Option<LocalStandardDecisionInput>, GatewayError> {
let spec_metadata = local_standard_spec_metadata(spec);
let Some(auth_context) = resolve_local_decision_execution_runtime_auth_context(decision) else {
return None;
return Ok(None);
};
let requested_model = extract_requested_model_from_request(
let Some(requested_model) = extract_requested_model_from_request(
parts,
body_json,
spec_metadata
.requested_model_family
.expect("standard specs should declare requested-model family"),
)?;
) else {
return Ok(None);
};
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context,
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
Ok(Some(resolved_input)) => resolved_input,
Ok(None) => return None,
Ok(None) => return Ok(None),
Err(err) => {
warn!(
trace_id = %trace_id,
@@ -70,14 +75,31 @@ pub(super) async fn resolve_local_standard_decision_input(
error = ?err,
"gateway local standard decision auth snapshot read failed"
);
return None;
return Err(err);
}
};
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
Some(input)
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
spec_metadata.api_format,
)
.await
{
warn!(
trace_id = %trace_id,
api_format = spec_metadata.api_format,
error = ?err,
"gateway local standard decision routing profile resolution failed"
);
return Err(err);
}
Ok(Some(input))
}
pub(super) async fn materialize_local_standard_candidate_attempts(
@@ -99,11 +121,14 @@ pub(super) async fn materialize_local_standard_candidate_attempts(
);
let preselection = preselect_local_execution_candidates_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
spec_metadata.require_streaming,
None,
false,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
input.routing_policy.as_ref(),
input.client_session_affinity.as_ref(),
false,
LocalCandidatePreselectionKeyMode::ProviderEndpointKeyModelAndApiFormat,
@@ -128,6 +153,7 @@ pub(super) async fn materialize_local_standard_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -221,13 +247,16 @@ pub(super) async fn build_local_standard_candidate_attempt_source<'a>(
let (source, candidate_count) =
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
spec_metadata.api_format,
&input.requested_model,
None,
spec_metadata.require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -315,11 +344,14 @@ async fn maybe_append_gemini_image_openai_image_preselection(
let image_preselection = preselect_local_execution_candidates_for_api_formats_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
input.routing_policy.as_ref(),
input.client_session_affinity.as_ref(),
false,
LocalCandidatePreselectionKeyMode::ProviderEndpointKeyModelAndApiFormat,
@@ -3,17 +3,20 @@ use crate::ai_serving::planner::candidate_materialization::{
mark_skipped_local_execution_candidate, mark_skipped_local_execution_candidate_with_extra_data,
mark_skipped_local_execution_candidate_with_failure_diagnostic,
};
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::materialization_policy::{
build_local_candidate_persistence_policy, LocalCandidatePersistencePolicyKind,
};
use crate::ai_serving::planner::passthrough::maybe_build_local_same_format_provider_decision_payload_for_candidate;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_standard_spec_metadata;
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
@@ -24,7 +27,7 @@ use crate::ai_serving::{
};
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_standard_candidate_payload_parts;
@@ -38,7 +41,7 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
input: &LocalStandardDecisionInput,
attempt: LocalStandardCandidateAttempt,
spec: LocalStandardSpec,
) -> Option<AiExecutionDecision> {
) -> Result<Option<AiExecutionDecision>, GatewayError> {
let spec_metadata = local_standard_spec_metadata(spec);
if api_format_alias_matches(
&attempt.eligible.provider_api_format,
@@ -70,10 +73,18 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
..
} = &attempt;
let candidate = &eligible.candidate;
let resolved = resolve_local_standard_candidate_payload_parts(
let Some(resolved) = resolve_local_standard_candidate_payload_parts(
state, parts, trace_id, body_json, input, &attempt, spec,
)
.await?;
.await?
else {
return Ok(None);
};
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
Some(body_json)
};
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
@@ -88,11 +99,13 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
"envelope_name".to_string(),
serde_json::Value::String(envelope_name.to_string()),
);
insert_native_client_envelope_name(&mut extra_fields, envelope_name, parts.uri.path());
}
let (execution_strategy, conversion_mode) = ai_local_execution_contract_for_formats(
spec_metadata.api_format,
resolved.provider_api_format.as_str(),
);
let effective_headers = input.effective_headers(&parts.headers);
let report_context = append_local_failover_policy_to_value(
append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -120,13 +133,14 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
body_rules: resolved.transport.endpoint.body_rules.as_ref(),
provider_request_method: Some(serde_json::Value::Null),
provider_request_headers: Some(&resolved.provider_request_headers),
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -144,7 +158,10 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
),
&resolved.transport,
);
let transport_profile = resolve_transport_profile(&resolved.transport);
let transport_profile = resolved
.transport_profile
.clone()
.or_else(|| resolve_transport_profile(&resolved.transport));
let timeouts = resolve_transport_execution_timeouts(&resolved.transport);
let super::request::LocalStandardCandidatePayloadParts {
auth_header,
@@ -157,43 +174,53 @@ pub(super) async fn maybe_build_local_standard_decision_payload_for_candidate(
upstream_is_stream,
envelope_name: _,
transport,
transport_profile: _,
request_redacted: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: candidate.provider_name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key: None,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
proxy,
transport_profile,
timeouts,
upstream_is_stream,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
))
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.to_string(),
provider_name: candidate.provider_name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key: None,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
upstream_is_stream,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
pub(super) async fn mark_skipped_local_standard_candidate(
@@ -327,6 +354,7 @@ mod tests {
api_key_allowed_providers: None,
api_key_allowed_api_formats: None,
api_key_allowed_models: None,
api_key_ip_rules: None,
currently_usable: true,
}
}
@@ -346,7 +374,13 @@ mod tests {
auth_snapshot: sample_auth_snapshot(),
required_capabilities: None,
request_auth_channel: None,
client_surface: None,
gateway_credential_carrier: None,
client_session_affinity: None,
routing_policy: None,
routing_trace_seed: None,
routing_context: None,
model_directive_policy: Default::default(),
}
}
@@ -418,6 +452,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "sk-upstream".to_string(),
decrypted_auth_config: None,
},
@@ -449,6 +484,7 @@ mod tests {
} else {
"gpt-4o-upstream".to_string()
},
supports_streaming: true,
mapping_matched_model: None,
}
}
@@ -473,6 +509,46 @@ mod tests {
}
}
fn sample_gemini_cli_attempt(candidate_index: u32) -> LocalExecutionCandidateAttempt {
let mut transport = sample_transport("gemini:generate_content", "endpoint-gemini-cli");
transport.provider.provider_type = "gemini_cli".to_string();
transport.provider.name = "gemini".to_string();
transport.endpoint.base_url = "https://cloudcode-pa.googleapis.com".to_string();
transport.endpoint.custom_path = Some("/v1internal:{action}".to_string());
transport.endpoint.endpoint_kind = Some("generate_content".to_string());
transport.key.auth_type = "bearer".to_string();
transport.key.api_formats = Some(vec!["gemini:generate_content".to_string()]);
transport.key.global_priority_by_format = Some(json!({
"gemini:generate_content": 1,
}));
transport.key.upstream_metadata = Some(json!({
"gemini_cli": {
"project_id": "test-project"
}
}));
let mut candidate = sample_candidate("gemini:generate_content", "endpoint-gemini-cli");
candidate.provider_name = "gemini".to_string();
candidate.provider_type = "gemini_cli".to_string();
candidate.key_auth_type = "bearer".to_string();
candidate.selected_provider_model_name = "gemini-2.5-pro".to_string();
candidate.global_model_name = "gemini-2.5-pro".to_string();
LocalExecutionCandidateAttempt {
eligible: EligibleLocalExecutionCandidate {
kind: LocalExecutionCandidateKind::SingleKey,
candidate,
transport: Arc::new(transport),
provider_api_format: "gemini:generate_content".to_string(),
orchestration: LocalExecutionCandidateMetadata::default(),
ranking: None,
},
candidate_index,
retry_index: 0,
candidate_id: format!("candidate-{candidate_index}"),
}
}
fn claude_stream_spec() -> LocalStandardSpec {
LocalStandardSpec {
api_format: "claude:messages",
@@ -512,6 +588,7 @@ mod tests {
claude_stream_spec(),
)
.await
.expect("same-format candidate should not fail routing mutation")
.expect("same-format candidate should build a standard-family payload");
assert_eq!(payload.endpoint_id.as_deref(), Some("endpoint-claude"));
@@ -547,6 +624,7 @@ mod tests {
claude_stream_spec(),
)
.await
.expect("cross-format candidate should not fail routing mutation")
.expect("cross-format candidate should still build after the same-format candidate");
assert_eq!(
@@ -562,4 +640,90 @@ mod tests {
Some("bidirectional")
);
}
#[tokio::test]
async fn standard_family_wraps_gemini_cli_cross_format_body_in_v1internal_envelope() {
let state = crate::AppState::new().expect("state should build");
let request = http::Request::builder()
.method("POST")
.uri("/v1/chat/completions")
.header(http::header::CONTENT_TYPE, "application/json")
.body(())
.expect("request should build");
let (parts, _) = request.into_parts();
let body_json = json!({
"model": "gemini-2.5-pro",
"messages": [{"role": "user", "content": "hello"}],
"temperature": 0.2,
"stream": true
});
let mut input = sample_input();
input.requested_model = "gemini-2.5-pro".to_string();
let spec = LocalStandardSpec {
api_format: "openai:chat",
decision_kind: "openai_chat_stream",
report_kind: "openai_chat_stream_success",
family: LocalStandardSourceFamily::Standard,
mode: LocalStandardSourceMode::Chat,
require_streaming: true,
};
let payload = maybe_build_local_standard_decision_payload_for_candidate(
&state,
&parts,
"trace-gemini-cli-cross-format",
&body_json,
&input,
sample_gemini_cli_attempt(0),
spec,
)
.await
.expect("cross-format candidate should not fail routing mutation")
.expect("gemini_cli candidate should build a payload");
assert_eq!(
payload.upstream_url.as_deref(),
Some("https://cloudcode-pa.googleapis.com/v1internal:streamGenerateContent?alt=sse")
);
assert_eq!(
payload.execution_strategy.as_deref(),
Some("local_cross_format")
);
assert_eq!(
payload.provider_api_format.as_deref(),
Some("gemini:generate_content")
);
assert_eq!(
payload
.provider_request_headers
.get("user-agent")
.map(String::as_str),
Some(crate::ai_serving::transport::GEMINI_CLI_USER_AGENT)
);
let provider_body = payload
.provider_request_body
.as_ref()
.expect("provider request body should be present");
assert_eq!(provider_body["model"], "gemini-2.5-pro");
assert_eq!(provider_body["project"], "test-project");
assert_eq!(
provider_body["user_prompt_id"],
"trace-gemini-cli-cross-format"
);
assert!(provider_body.get("contents").is_none());
assert!(provider_body.get("generationConfig").is_none());
assert!(provider_body["request"].get("contents").is_some());
let report_context = payload
.report_context
.as_ref()
.expect("report context should be present");
assert_eq!(
report_context
.get("envelope_name")
.and_then(|value| value.as_str()),
Some(crate::ai_serving::transport::GEMINI_CLI_V1INTERNAL_ENVELOPE_NAME)
);
}
}
File diff suppressed because it is too large Load Diff
@@ -1,12 +1,10 @@
use std::collections::BTreeMap;
use aether_contracts::RequestBody;
use super::{
augment_sync_report_context, build_ai_execution_plan_from_decision,
generic_decision_missing_exact_provider_request, take_ai_decision_plan_core,
take_ai_upstream_auth_pair, take_non_empty_string, AiExecutionPlanFromDecisionParts,
AiStreamAttempt, AiSyncAttempt,
generic_decision_missing_exact_provider_request, resolve_ai_passthrough_sync_request_body,
take_ai_decision_plan_core, take_ai_upstream_auth_pair, take_non_empty_string,
AiExecutionPlanFromDecisionParts, AiStreamAttempt, AiSyncAttempt,
};
use crate::ai_serving::transport::{
build_standard_plan_fallback_headers, StandardPlanFallbackAcceptPolicy,
@@ -61,6 +59,10 @@ pub(crate) fn build_gemini_sync_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
@@ -70,7 +72,7 @@ pub(crate) fn build_gemini_sync_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream,
},
);
@@ -129,6 +131,10 @@ pub(crate) fn build_gemini_stream_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -137,7 +143,7 @@ pub(crate) fn build_gemini_stream_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream: true,
},
);
@@ -8,12 +8,17 @@ use crate::{AiExecutionDecision, AppState, GatewayError};
mod claude;
mod codex;
mod deepseek;
mod family;
mod gemini;
mod normalize;
mod openai;
pub(crate) use self::codex::apply_codex_openai_responses_special_headers;
pub(crate) use self::codex::{
apply_codex_openai_responses_special_body_edits, apply_codex_openai_special_headers,
codex_model_capabilities_for_transport,
};
pub(crate) use self::deepseek::{apply_deepseek_tool_call_thinking_compat, is_deepseek_provider};
pub(crate) use self::family::{
build_local_stream_attempt_source, build_local_stream_plan_and_reports,
build_local_sync_attempt_source, build_local_sync_plan_and_reports,
@@ -21,9 +26,11 @@ pub(crate) use self::family::{
pub(crate) use self::normalize::{
build_cross_format_openai_chat_request_body, build_cross_format_openai_chat_upstream_url,
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_codex_model_capabilities,
build_cross_format_openai_responses_upstream_url, build_local_openai_chat_request_body,
build_local_openai_chat_upstream_url, build_local_openai_responses_request_body,
build_local_openai_responses_upstream_url,
build_local_openai_responses_request_body_with_codex_model_capabilities,
build_local_openai_responses_upstream_url, validate_final_openai_provider_request,
};
pub(crate) use self::openai::{
build_local_openai_chat_stream_attempt_source_for_kind,
@@ -38,6 +45,7 @@ pub(crate) use self::openai::{
map_openai_reasoning_effort_to_gemini_budget, maybe_build_stream_local_decision_payload,
maybe_build_stream_local_openai_responses_decision_payload,
maybe_build_sync_local_decision_payload,
maybe_build_sync_local_openai_embedding_decision_payload,
maybe_build_sync_local_openai_responses_decision_payload, parse_openai_stop_sequences,
resolve_openai_chat_max_tokens, set_local_openai_chat_execution_exhausted_diagnostic,
value_as_u64,
@@ -57,7 +65,8 @@ pub(crate) use crate::ai_serving::{
normalize_openai_responses_request_to_openai_chat_request, parse_openai_tool_result_content,
};
pub(crate) use aether_ai_serving::{
request_body_build_failure_extra_data, same_format_provider_request_body_failure_extra_data,
openai_provider_request_contract_failure_extra_data, request_body_build_failure_extra_data,
request_conversion_failure_extra_data, same_format_provider_request_body_failure_extra_data,
};
pub(crate) fn build_standard_upstream_url(
@@ -66,14 +75,17 @@ pub(crate) fn build_standard_upstream_url(
mapped_model: &str,
provider_api_format: &str,
upstream_is_stream: bool,
provider_request_body: Option<&serde_json::Value>,
) -> Option<String> {
crate::ai_serving::build_provider_transport_request_url(
crate::ai_serving::build_provider_transport_request_url_for_request_body(
transport,
provider_api_format,
Some(mapped_model),
upstream_is_stream,
parts.uri.query(),
None,
None,
provider_request_body,
)
}
@@ -85,6 +97,14 @@ pub(crate) async fn maybe_build_sync_local_standard_decision_payload(
body_json: &serde_json::Value,
plan_kind: &str,
) -> Result<Option<AiExecutionDecision>, GatewayError> {
if let Some(payload) = self::openai::maybe_build_sync_local_openai_embedding_decision_payload(
state, parts, trace_id, decision, body_json, plan_kind,
)
.await?
{
return Ok(Some(payload));
}
if let Some(payload) = self::claude::maybe_build_sync_local_claude_decision_payload(
state, parts, trace_id, decision, body_json, plan_kind,
)
@@ -281,7 +301,7 @@ mod tests {
let converted = build_standard_request_body(
&request,
"claude:messages",
"gpt-5",
"gpt-5.4",
"codex",
"openai:responses",
"/v1/messages",
@@ -293,7 +313,7 @@ mod tests {
assert!(converted.get("metadata").is_none());
assert_eq!(converted["store"], false);
assert_eq!(converted["instructions"], "");
assert!(converted.get("instructions").is_none());
assert_eq!(converted["include"], json!(["reasoning.encrypted_content"]));
assert_eq!(converted["parallel_tool_calls"], true);
assert_eq!(converted["reasoning"]["effort"], "medium");
@@ -12,9 +12,38 @@ pub(crate) use self::chat::{
};
pub(crate) use self::responses::{
build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_codex_model_capabilities,
build_cross_format_openai_responses_upstream_url, build_local_openai_responses_request_body,
build_local_openai_responses_request_body_with_codex_model_capabilities,
build_local_openai_responses_upstream_url,
};
pub(super) use crate::ai_serving::planner::common::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
};
pub(crate) fn validate_final_openai_provider_request(
provider_api_format: &str,
mapped_model: &str,
source_request_body: &serde_json::Value,
provider_request_body: &serde_json::Value,
) -> Option<()> {
let provider_model = provider_request_body
.get("model")
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.unwrap_or(mapped_model);
let source_model = source_request_body
.get("model")
.and_then(serde_json::Value::as_str)
.map(str::trim)
.filter(|value| !value.is_empty())
.unwrap_or(mapped_model);
crate::ai_serving::validate_openai_provider_request_contract(
provider_api_format,
provider_model,
source_model,
provider_request_body,
)
.ok()
}
@@ -9,7 +9,10 @@ use crate::ai_serving::{
GatewayProviderTransportSnapshot,
};
use super::{enforce_provider_body_stream_policy, request_requires_body_stream_field};
use super::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
validate_final_openai_provider_request,
};
pub(crate) fn build_local_openai_chat_request_body(
body_json: &Value,
@@ -39,6 +42,12 @@ pub(crate) fn build_local_openai_chat_request_body(
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
"openai:chat",
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -92,6 +101,12 @@ pub(crate) fn build_cross_format_openai_chat_request_body(
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -2,14 +2,16 @@ use serde_json::Value;
use crate::ai_serving::transport::apply_standard_provider_request_body_rules_with_request_headers;
use crate::ai_serving::{
apply_codex_openai_responses_special_body_edits,
apply_openai_responses_compact_special_body_edits,
build_cross_format_openai_responses_request_body_with_model_directives as surface_build_cross_format_openai_responses_request_body,
build_cross_format_openai_responses_request_body_with_model_directives_and_history_scope as surface_build_cross_format_openai_responses_request_body,
build_local_openai_responses_request_body_with_model_directives as surface_build_local_openai_responses_request_body,
GatewayProviderTransportSnapshot,
};
use super::{enforce_provider_body_stream_policy, request_requires_body_stream_field};
use super::{
enforce_provider_body_stream_policy, request_requires_body_stream_field,
validate_final_openai_provider_request,
};
pub(crate) fn build_local_openai_responses_request_body(
body_json: &Value,
@@ -19,9 +21,35 @@ pub(crate) fn build_local_openai_responses_request_body(
provider_type: &str,
provider_api_format: &str,
body_rules: Option<&Value>,
user_api_key_id: Option<&str>,
_user_api_key_id: Option<&str>,
request_headers: &http::HeaderMap,
enable_model_directives: bool,
) -> Option<Value> {
build_local_openai_responses_request_body_with_codex_model_capabilities(
body_json,
mapped_model,
require_streaming,
force_body_stream_field,
provider_type,
provider_api_format,
body_rules,
request_headers,
None,
enable_model_directives,
)
}
pub(crate) fn build_local_openai_responses_request_body_with_codex_model_capabilities(
body_json: &Value,
mapped_model: &str,
require_streaming: bool,
force_body_stream_field: bool,
provider_type: &str,
provider_api_format: &str,
body_rules: Option<&Value>,
request_headers: &http::HeaderMap,
model_capabilities: Option<&crate::ai_serving::CodexResponsesModelCapabilities>,
enable_model_directives: bool,
) -> Option<Value> {
let provider_request_body = surface_build_local_openai_responses_request_body(
body_json,
@@ -36,23 +64,39 @@ pub(crate) fn build_local_openai_responses_request_body(
body_json,
request_headers,
)?;
apply_codex_openai_responses_special_body_edits(
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(mapped_model);
crate::ai_serving::apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities(
&mut provider_request_body,
provider_type,
provider_api_format,
mapped_model,
source_model,
model_capabilities,
body_rules,
user_api_key_id,
);
apply_openai_responses_compact_special_body_edits(
&mut provider_request_body,
provider_api_format,
);
crate::ai_serving::strip_incompatible_openai_responses_reasoning_items(
&mut provider_request_body,
provider_api_format,
);
enforce_provider_body_stream_policy(
&mut provider_request_body,
provider_api_format,
require_streaming,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -68,6 +112,36 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
user_api_key_id: Option<&str>,
request_headers: &http::HeaderMap,
enable_model_directives: bool,
) -> Option<Value> {
build_cross_format_openai_responses_request_body_with_codex_model_capabilities(
body_json,
mapped_model,
client_api_format,
provider_api_format,
upstream_is_stream,
force_body_stream_field,
provider_type,
body_rules,
request_headers,
user_api_key_id,
None,
enable_model_directives,
)
}
pub(crate) fn build_cross_format_openai_responses_request_body_with_codex_model_capabilities(
body_json: &Value,
mapped_model: &str,
client_api_format: &str,
provider_api_format: &str,
upstream_is_stream: bool,
force_body_stream_field: bool,
provider_type: &str,
body_rules: Option<&Value>,
request_headers: &http::HeaderMap,
history_scope: Option<&str>,
model_capabilities: Option<&crate::ai_serving::CodexResponsesModelCapabilities>,
enable_model_directives: bool,
) -> Option<Value> {
let provider_request_body = surface_build_cross_format_openai_responses_request_body(
body_json,
@@ -76,6 +150,7 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
provider_api_format,
upstream_is_stream,
enable_model_directives,
history_scope,
)?;
let mut provider_request_body =
apply_standard_provider_request_body_rules_with_request_headers(
@@ -84,23 +159,39 @@ pub(crate) fn build_cross_format_openai_responses_request_body(
body_json,
request_headers,
)?;
apply_codex_openai_responses_special_body_edits(
let source_model = body_json
.get("model")
.and_then(Value::as_str)
.unwrap_or(mapped_model);
crate::ai_serving::apply_codex_openai_responses_special_body_edits_with_source_model_and_capabilities(
&mut provider_request_body,
provider_type,
provider_api_format,
mapped_model,
source_model,
model_capabilities,
body_rules,
user_api_key_id,
);
apply_openai_responses_compact_special_body_edits(
&mut provider_request_body,
provider_api_format,
);
crate::ai_serving::strip_incompatible_openai_responses_reasoning_items(
&mut provider_request_body,
provider_api_format,
);
enforce_provider_body_stream_policy(
&mut provider_request_body,
provider_api_format,
upstream_is_stream,
request_requires_body_stream_field(body_json, force_body_stream_field),
);
validate_final_openai_provider_request(
provider_api_format,
mapped_model,
body_json,
&provider_request_body,
)?;
Some(provider_request_body)
}
@@ -6,8 +6,8 @@ use http::Request;
use serde_json::{json, Value};
use super::{
build_cross_format_openai_responses_request_body, build_local_openai_responses_request_body,
build_local_openai_responses_upstream_url,
build_cross_format_openai_responses_request_body, build_local_openai_chat_request_body,
build_local_openai_responses_request_body, build_local_openai_responses_upstream_url,
};
fn object_keys(value: &Value) -> Vec<&str> {
@@ -68,6 +68,7 @@ fn sample_transport(base_url: &str, api_format: &str) -> GatewayProviderTranspor
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "__placeholder__".to_string(),
decrypted_auth_config: None,
},
@@ -145,12 +146,47 @@ fn local_openai_responses_wrapper_preserves_body_order_after_edits() {
"reasoning",
"tool_choice",
"parallel_tool_calls",
"instructions",
"prompt_cache_key",
]
);
assert_eq!(provider_request_body["parallel_tool_calls"], json!(true));
assert_eq!(provider_request_body["instructions"], json!(""));
assert!(provider_request_body.get("instructions").is_none());
}
#[test]
fn local_openai_responses_wrapper_strips_foreign_reasoning_item_ids() {
let body_json = json!({
"model": "gpt-5.4",
"input": [
{"type": "reasoning", "id": "rs_provider_123", "summary": []},
{
"type": "reasoning",
"id": "item_72d3bd8d367d01977ace23f1",
"summary": []
},
{"type": "message", "role": "user", "content": "continue"}
]
});
let provider_request_body = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
false,
false,
"codex",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.expect("local OpenAI Responses body should build");
let input = provider_request_body["input"]
.as_array()
.expect("input array");
assert_eq!(input.len(), 2);
assert_eq!(input[0]["id"], "rs_provider_123");
assert_eq!(input[1]["type"], "message");
}
#[test]
@@ -180,18 +216,49 @@ fn local_openai_responses_compact_wrapper_strips_store_for_same_format_requests(
}
#[test]
fn local_openai_responses_compact_wrapper_strips_include_for_codex_requests() {
fn local_codex_compact_wrapper_applies_the_complete_request_projection() {
let body_json = json!({
"model": "gpt-5.4",
"input": [],
"model": "gpt-5.6-sol",
"input": [{
"type": "message",
"role": "user",
"content": [{"type": "input_text", "text": "hello"}]
}],
"instructions": "Work carefully",
"client_metadata": {"origin": "codex"},
"include": ["reasoning.encrypted_content"],
"store": true,
"stream": true
"stream": true,
"stream_options": {"reasoning_summary_delivery": "sequential_cutoff"},
"tool_choice": "auto",
"parallel_tool_calls": true,
"reasoning": {"effort": "max", "summary": "auto", "context": "all_turns"},
"text": {"verbosity": "medium"},
"tools": [{
"type": "function",
"name": "lookup",
"parameters": {"type": "object", "properties": {}}
}],
"service_tier": "priority",
"prompt_cache_key": "thread-compact"
});
let provider_request_body = build_local_openai_responses_request_body(
let regular = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
"gpt-5.6-sol",
true,
false,
"codex",
"openai:responses",
None,
Some("key-123"),
&http::HeaderMap::new(),
false,
)
.expect("local Codex Responses body should build");
let compact = build_local_openai_responses_request_body(
&body_json,
"gpt-5.6-sol",
false,
false,
"codex",
@@ -201,22 +268,44 @@ fn local_openai_responses_compact_wrapper_strips_include_for_codex_requests() {
&http::HeaderMap::new(),
false,
)
.expect("local codex compact body should build");
.expect("local Codex Compact body should build");
assert!(provider_request_body.get("include").is_none());
assert!(provider_request_body.get("store").is_none());
assert!(provider_request_body.get("stream").is_none());
assert_eq!(provider_request_body["instructions"], "");
assert_eq!(
provider_request_body["prompt_cache_key"],
"172c39e6-c0a0-5a70-8b63-e0f8e0d185a3"
);
for field in [
"client_metadata",
"include",
"store",
"stream",
"stream_options",
"tool_choice",
] {
assert!(
regular.get(field).is_some(),
"Responses should contain {field}"
);
assert!(compact.get(field).is_none(), "Compact should omit {field}");
}
for field in [
"model",
"input",
"instructions",
"parallel_tool_calls",
"reasoning",
"text",
"tools",
"service_tier",
"prompt_cache_key",
] {
assert_eq!(
compact[field], regular[field],
"Compact should preserve {field}"
);
}
}
#[test]
fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
let body_json = json!({
"model": "gpt-5.4-max",
"model": "gpt-5.6-sol-max",
"input": "hello",
"reasoning": {"effort": "low", "summary": "auto"}
});
@@ -226,7 +315,7 @@ fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
let provider_request_body = build_local_openai_responses_request_body(
&body_json,
"gpt-5.4",
"gpt-5.6-sol",
false,
false,
"openai",
@@ -238,11 +327,137 @@ fn local_openai_responses_wrapper_applies_model_directive_before_body_rules() {
)
.expect("local openai responses body should build");
assert_eq!(provider_request_body["reasoning"]["effort"], "xhigh");
assert_eq!(provider_request_body["reasoning"]["effort"], "max");
assert_eq!(provider_request_body["reasoning"]["summary"], "auto");
assert_eq!(provider_request_body["metadata"]["override_seen"], true);
}
#[test]
fn final_openai_provider_contract_uses_the_mapped_model_for_reasoning() {
let alias = json!({
"model": "deployment-alias",
"input": "hello",
"reasoning": {"effort": "max"}
});
assert!(build_local_openai_responses_request_body(
&alias,
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_some());
assert!(build_local_openai_responses_request_body(
&alias,
"gpt-5.4",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let minimal = json!({
"model": "deployment-alias",
"messages": [{"role": "user", "content": "hello"}],
"reasoning_effort": "minimal"
});
assert!(build_local_openai_chat_request_body(
&minimal,
"gpt-5.6-terra",
false,
false,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let opaque_mapping = json!({
"model": "gpt-5.6-sol-max",
"input": "hello",
"reasoning": {"effort": "max", "mode": "pro"},
"prompt_cache_options": {"mode": "explicit", "ttl": "30m"}
});
assert!(build_local_openai_responses_request_body(
&opaque_mapping,
"azure-production",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_some());
assert!(build_local_openai_responses_request_body(
&opaque_mapping,
"gpt-5.4",
false,
false,
"openai",
"openai:responses",
None,
None,
&http::HeaderMap::new(),
false,
)
.is_none());
}
#[test]
fn final_openai_provider_contract_validates_body_rule_output() {
let body = json!({
"model": "gpt-5.6-sol",
"input": "hello",
"reasoning": {"effort": "max"}
});
let model_override = json!([
{"action":"set","path":"model","value":"gpt-5.4"}
]);
assert!(build_local_openai_responses_request_body(
&body,
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
Some(&model_override),
None,
&http::HeaderMap::new(),
false,
)
.is_none());
let cache_override = json!([
{"action":"set","path":"prompt_cache_options.ttl","value":"1h"}
]);
assert!(build_local_openai_responses_request_body(
&json!({"model":"gpt-5.6-sol","input":"hello"}),
"gpt-5.6-sol",
false,
false,
"openai",
"openai:responses",
Some(&cache_override),
None,
&http::HeaderMap::new(),
false,
)
.is_none());
}
#[test]
fn local_openai_responses_upstream_url_preserves_codex_base_path() {
let request = Request::builder()
@@ -291,6 +506,47 @@ fn strips_metadata_for_codex_openai_responses_requests() {
assert!(provider_request_body.get("metadata").is_none());
}
#[test]
fn openai_chat_to_codex_responses_preserves_json_mode_chat_messages() {
let body_json = json!({
"model": "gpt-5.5",
"messages": [
{"role": "system", "content": "Return a JSON object."},
{"role": "user", "content": "Why did this JSON request fail?"}
],
"response_format": {"type": "json_object"}
});
let provider_request_body = build_cross_format_openai_responses_request_body(
&body_json,
"gpt-5.5-upstream",
"openai:chat",
"openai:responses",
false,
false,
"codex",
None,
None,
&http::HeaderMap::new(),
false,
)
.expect("openai chat to codex responses request should build");
assert_eq!(
provider_request_body["text"]["format"]["type"],
"json_object"
);
assert_eq!(provider_request_body["input"][0]["role"], "user");
assert_eq!(
provider_request_body["input"][0]["content"][0]["text"],
"Why did this JSON request fail?"
);
assert_eq!(
provider_request_body["instructions"],
"Return a JSON object."
);
}
#[test]
fn applies_codex_defaults_unless_body_rules_handle_the_field() {
let body_json = json!({
@@ -329,7 +585,7 @@ fn applies_codex_defaults_unless_body_rules_handle_the_field() {
}
#[test]
fn injects_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
fn omits_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
let body_json = json!({
"model": "claude-sonnet-4-5",
"messages": [{
@@ -353,14 +609,11 @@ fn injects_codex_prompt_cache_key_for_openai_responses_cross_format_requests() {
)
.expect("claude cli to codex request should build");
assert_eq!(
provider_request_body["prompt_cache_key"],
"172c39e6-c0a0-5a70-8b63-e0f8e0d185a3"
);
assert!(provider_request_body.get("prompt_cache_key").is_none());
}
#[test]
fn injects_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
fn omits_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
let body_json = json!({
"model": "gpt-5",
"messages": [{
@@ -383,8 +636,5 @@ fn injects_codex_prompt_cache_key_for_openai_chat_cross_format_requests() {
)
.expect("openai chat to codex request should build");
assert_eq!(
provider_request_body["prompt_cache_key"],
"172c39e6-c0a0-5a70-8b63-e0f8e0d185a3"
);
assert!(provider_request_body.get("prompt_cache_key").is_none());
}
@@ -6,6 +6,7 @@ mod request;
mod support;
pub(super) use self::payload::maybe_build_local_openai_chat_decision_payload_for_candidate;
pub(super) use self::request::LocalOpenAiChatRequestPreparation;
pub(super) use self::support::{
build_lazy_local_openai_chat_candidate_attempt_source,
build_local_openai_chat_candidate_attempt_source,
@@ -1,21 +1,26 @@
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::common::OPENAI_CHAT_STREAM_PLAN_KIND;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, insert_provider_stream_event_api_format,
LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
insert_provider_stream_event_api_format, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_chat_candidate_payload_parts;
use super::request::{
resolve_local_openai_chat_candidate_payload_parts, LocalOpenAiChatRequestPreparation,
};
use super::support::{LocalOpenAiChatCandidateAttempt, LocalOpenAiChatDecisionInput};
#[allow(clippy::too_many_arguments)]
@@ -25,6 +30,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
trace_id: &str,
body_json: &serde_json::Value,
input: &LocalOpenAiChatDecisionInput,
preparation: Option<&mut LocalOpenAiChatRequestPreparation>,
attempt: LocalOpenAiChatCandidateAttempt,
decision_kind: &str,
report_kind: &str,
@@ -38,12 +44,15 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
candidate_id,
..
} = attempt;
let upstream_is_stream = upstream_is_stream && eligible.candidate.supports_streaming;
let payload_started_at = std::time::Instant::now();
let Some(resolved) = resolve_local_openai_chat_candidate_payload_parts(
state,
parts,
trace_id,
body_json,
input,
preparation,
&eligible,
candidate_index,
&candidate_id,
@@ -53,9 +62,25 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
)
.await?
else {
observe_gateway_stage_ms(
"stream_candidate_payload_parts",
payload_started_at.elapsed().as_millis() as u64,
);
return Ok(None);
};
observe_gateway_stage_ms(
"stream_candidate_payload_parts",
payload_started_at.elapsed().as_millis() as u64,
);
let candidate = &eligible.candidate;
let upstream_is_stream =
crate::ai_serving::planner::common::resolve_upstream_is_stream_for_provider(
resolved.transport.endpoint.config.as_ref(),
resolved.transport.provider.provider_type.as_str(),
resolved.provider_api_format.as_str(),
upstream_is_stream,
false,
);
let prompt_cache_key = resolved
.provider_request_body
@@ -64,10 +89,18 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
.map(str::trim)
.filter(|value| !value.is_empty())
.map(ToOwned::to_owned);
let proxy_started_at = std::time::Instant::now();
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
let transport_profile = resolve_transport_profile(&resolved.transport);
observe_gateway_stage_ms(
"stream_candidate_proxy",
proxy_started_at.elapsed().as_millis() as u64,
);
let transport_profile = resolved
.transport_profile
.clone()
.or_else(|| resolve_transport_profile(&resolved.transport));
let timeouts = resolve_transport_execution_timeouts(&resolved.transport);
let mut extra_fields = serde_json::Map::new();
if let Some(proxy_value) =
@@ -80,12 +113,29 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
"envelope_name".to_string(),
serde_json::Value::String(envelope_name.to_string()),
);
insert_native_client_envelope_name(&mut extra_fields, envelope_name, parts.uri.path());
}
insert_provider_stream_event_api_format(
&mut extra_fields,
resolved.transport.provider.provider_type.as_str(),
);
if let Some(image_request_summary) = resolved.image_request_summary.as_ref() {
extra_fields.insert("image_request".to_string(), image_request_summary.clone());
}
if resolved
.provider_api_format
.eq_ignore_ascii_case("openai:image")
&& resolved
.transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("chatgpt_web")
{
extra_fields.insert("chatgpt_web_image".to_string(), serde_json::json!(true));
}
let super::request::LocalOpenAiChatCandidatePayloadParts {
client_api_format,
auth_header,
auth_value,
mapped_model,
@@ -99,12 +149,16 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
envelope_name,
transport,
request_redacted,
transport_profile: _,
image_request_summary: _,
} = resolved;
let original_request_body_json = if request_redacted {
Some(&provider_request_body)
} else {
Some(body_json)
};
let effective_headers = input.effective_headers(&parts.headers);
let report_context_started_at = std::time::Instant::now();
let report_context = append_local_failover_policy_to_value(
append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -122,7 +176,7 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
global_model_id: Some(&candidate.global_model_id),
global_model_name: Some(&candidate.global_model_name),
provider_api_format: &provider_api_format,
client_api_format: "openai:chat",
client_api_format: &client_api_format,
mapped_model: Some(&mapped_model),
candidate_group_id: eligible.orchestration.candidate_group_id.as_deref(),
pool_key_lease: eligible.orchestration.pool_key_lease.as_ref(),
@@ -132,13 +186,14 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
body_rules: transport.endpoint.body_rules.as_ref(),
provider_request_method: Some(serde_json::Value::Null),
provider_request_headers: Some(&provider_request_headers),
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -154,45 +209,62 @@ pub(crate) async fn maybe_build_local_openai_chat_decision_payload_for_candidate
}),
execution_strategy,
conversion_mode,
"openai:chat",
client_api_format.as_str(),
candidate.endpoint_api_format.as_str(),
),
&transport,
);
observe_gateway_stage_ms(
"stream_candidate_report_context",
report_context_started_at.elapsed().as_millis() as u64,
);
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Ok(Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream,
decision_kind: decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format,
client_api_format: "openai:chat".to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
proxy,
transport_profile,
timeouts,
upstream_is_stream,
report_kind: Some(report_kind),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
)))
let decision_started_at = std::time::Instant::now();
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream,
decision_kind: decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format,
client_api_format: "openai:chat".to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
upstream_is_stream,
report_kind: Some(report_kind),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
observe_gateway_stage_ms(
"stream_candidate_decision_build",
decision_started_at.elapsed().as_millis() as u64,
);
Ok(Some(decision))
}
File diff suppressed because it is too large Load Diff
@@ -140,6 +140,7 @@ pub(crate) async fn materialize_local_openai_chat_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -220,6 +221,7 @@ pub(crate) async fn build_local_openai_chat_candidate_attempt_source<'a>(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -291,13 +293,16 @@ pub(crate) async fn build_lazy_local_openai_chat_candidate_attempt_source<'a>(
);
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
"openai:chat",
&input.requested_model,
None,
require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -13,6 +13,7 @@ use self::decision::{
build_lazy_local_openai_chat_candidate_attempt_source,
maybe_build_local_openai_chat_decision_payload_for_candidate, LocalOpenAiChatCandidateAttempt,
LocalOpenAiChatCandidateAttemptSource, LocalOpenAiChatDecisionInput,
LocalOpenAiChatRequestPreparation,
};
use self::plans::{
build_local_openai_chat_stream_attempt_source, build_local_openai_chat_stream_plan_and_reports,
@@ -136,17 +137,18 @@ pub(crate) async fn maybe_build_sync_local_decision_payload(
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, false,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let (mut source, _) = build_lazy_local_openai_chat_candidate_attempt_source(
state, trace_id, &input, body_json, false,
)
.await;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let upstream_is_stream = self::plans::openai_chat_upstream_is_stream_for_candidate(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
@@ -158,6 +160,7 @@ pub(crate) async fn maybe_build_sync_local_decision_payload(
trace_id,
body_json,
&input,
None,
attempt,
OPENAI_CHAT_SYNC_PLAN_KIND,
"openai_chat_sync_success",
@@ -187,17 +190,18 @@ pub(crate) async fn maybe_build_stream_local_decision_payload(
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, false,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let (mut source, _) = build_lazy_local_openai_chat_candidate_attempt_source(
state, trace_id, &input, body_json, true,
)
.await;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let upstream_is_stream = self::plans::openai_chat_upstream_is_stream_for_candidate(
&attempt.eligible.transport,
attempt.eligible.provider_api_format.as_str(),
@@ -209,6 +213,7 @@ pub(crate) async fn maybe_build_stream_local_decision_payload(
trace_id,
body_json,
&input,
None,
attempt,
OPENAI_CHAT_STREAM_PLAN_KIND,
"openai_chat_stream_success",
@@ -31,7 +31,12 @@ pub(super) fn openai_chat_upstream_is_stream_for_candidate(
crate::ai_serving::transport::kiro::is_kiro_claude_messages_transport(
transport,
provider_api_format,
);
) || openai_chat_antigravity_requires_upstream_streaming(transport, provider_api_format)
|| openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport,
provider_api_format,
client_is_stream,
);
resolve_upstream_is_stream_for_provider(
transport.endpoint.config.as_ref(),
transport.provider.provider_type.as_str(),
@@ -41,6 +46,27 @@ pub(super) fn openai_chat_upstream_is_stream_for_candidate(
)
}
fn openai_chat_antigravity_requires_upstream_streaming(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
) -> bool {
crate::ai_serving::transport::antigravity::is_antigravity_provider_transport(transport)
&& crate::ai_serving::normalize_api_format_alias(provider_api_format)
== "gemini:generate_content"
}
fn openai_chat_gemini_cli_client_stream_requires_upstream_streaming(
transport: &GatewayProviderTransportSnapshot,
provider_api_format: &str,
client_is_stream: bool,
) -> bool {
crate::ai_serving::transport::gemini_cli::is_gemini_cli_provider_transport(transport)
&& crate::ai_serving::transport::gemini_cli::gemini_cli_v1internal_requires_upstream_streaming(
provider_api_format,
client_is_stream,
)
}
#[cfg(test)]
mod tests {
use super::openai_chat_upstream_is_stream_for_candidate;
@@ -103,6 +129,7 @@ mod tests {
expires_at_unix_secs: None,
proxy: None,
fingerprint: None,
upstream_metadata: None,
decrypted_api_key: "secret".to_string(),
decrypted_auth_config: None,
},
@@ -164,4 +191,44 @@ mod tests {
false,
));
}
#[test]
fn openai_chat_policy_resolver_preserves_gemini_cli_streaming_requests() {
let gemini_cli = sample_transport(
"gemini_cli",
"gemini:generate_content",
Some(json!({"upstream_stream_policy": "force_non_stream"})),
);
assert!(openai_chat_upstream_is_stream_for_candidate(
&gemini_cli,
"gemini:generate_content",
true,
));
assert!(!openai_chat_upstream_is_stream_for_candidate(
&gemini_cli,
"gemini:generate_content",
false,
));
}
#[test]
fn openai_chat_policy_resolver_preserves_antigravity_streaming_envelope() {
let antigravity = sample_transport(
"antigravity",
"gemini:generate_content",
Some(json!({"upstream_stream_policy": "force_non_stream"})),
);
assert!(openai_chat_upstream_is_stream_for_candidate(
&antigravity,
"gemini:generate_content",
false,
));
assert!(openai_chat_upstream_is_stream_for_candidate(
&antigravity,
"gemini:generate_content",
true,
));
}
}
@@ -21,11 +21,14 @@ pub(crate) async fn list_local_openai_chat_candidates(
> {
let outcome = preselect_local_execution_candidates_with_serving(
PlannerAppState::new(state),
&input.model_directive_policy,
"openai:chat",
&input.requested_model,
None,
require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
input.routing_policy.as_ref(),
input.client_session_affinity.as_ref(),
false,
LocalCandidatePreselectionKeyMode::ProviderEndpointKeyModel,
@@ -4,11 +4,13 @@ use super::super::{GatewayControlDecision, LocalOpenAiChatDecisionInput};
use super::diagnostic::set_local_openai_chat_miss_diagnostic;
use crate::ai_serving::planner::common::extract_standard_requested_model;
use crate::ai_serving::planner::decision_input::{
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::resolve_local_decision_execution_runtime_auth_context;
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::AppState;
use crate::stage_metrics::observe_gateway_stage_ms;
use crate::{AppState, GatewayError};
pub(crate) async fn resolve_local_openai_chat_decision_input(
state: &AppState,
@@ -18,7 +20,7 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
body_json: &serde_json::Value,
plan_kind: &str,
record_miss_diagnostic: bool,
) -> Option<LocalOpenAiChatDecisionInput> {
) -> Result<Option<LocalOpenAiChatDecisionInput>, GatewayError> {
let Some(auth_context) = resolve_local_decision_execution_runtime_auth_context(decision) else {
warn!(
trace_id = %trace_id,
@@ -37,7 +39,7 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
"missing_auth_context",
);
}
return None;
return Ok(None);
};
let Some(requested_model) = extract_standard_requested_model(body_json) else {
@@ -55,14 +57,17 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
"missing_requested_model",
);
}
return None;
return Ok(None);
};
let auth_started_at = std::time::Instant::now();
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context.clone(),
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -84,7 +89,7 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
"auth_snapshot_missing",
);
}
return None;
return Ok(None);
}
Err(err) => {
warn!(
@@ -102,12 +107,42 @@ pub(crate) async fn resolve_local_openai_chat_decision_input(
"auth_snapshot_read_failed",
);
}
return None;
return Err(err);
}
};
observe_gateway_stage_ms(
"openai_chat_decision_input_auth",
auth_started_at.elapsed().as_millis() as u64,
);
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
let affinity_started_at = std::time::Instant::now();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
Some(input)
observe_gateway_stage_ms(
"openai_chat_decision_input_affinity",
affinity_started_at.elapsed().as_millis() as u64,
);
let routing_started_at = std::time::Instant::now();
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
"openai:chat",
)
.await
{
warn!(
trace_id = %trace_id,
error = ?err,
"gateway local openai chat decision routing profile resolution failed"
);
return Err(err);
}
observe_gateway_stage_ms(
"openai_chat_decision_input_routing",
routing_started_at.elapsed().as_millis() as u64,
);
Ok(Some(input))
}
@@ -1,11 +1,12 @@
use async_trait::async_trait;
use std::collections::VecDeque;
use tracing::warn;
use super::super::{
build_lazy_local_openai_chat_candidate_attempt_source,
maybe_build_local_openai_chat_decision_payload_for_candidate, AppState, GatewayControlDecision,
GatewayError, LocalOpenAiChatCandidateAttempt, LocalOpenAiChatCandidateAttemptSource,
LocalOpenAiChatDecisionInput,
LocalOpenAiChatDecisionInput, LocalOpenAiChatRequestPreparation,
};
use super::diagnostic::{
set_local_openai_chat_candidate_evaluation_diagnostic, set_local_openai_chat_miss_diagnostic,
@@ -18,14 +19,33 @@ use crate::ai_serving::planner::plan_builders::{
build_openai_chat_stream_plan_from_decision, AiStreamAttempt,
};
use crate::ai_serving::planner::runtime_miss::apply_local_runtime_candidate_terminal_reason;
use crate::ai_serving::planner::standard::build_local_openai_chat_upstream_url;
use crate::ai_serving::transport::{
is_windsurf_provider_transport, local_openai_chat_transport_unsupported_reason,
};
use crate::clock::request_distribution_seed;
use crate::stage_metrics::{
observe_gateway_stage_ms, record_openai_chat_stream_payload_build_prefetch_avoided,
record_openai_chat_stream_payload_build_selected,
record_openai_chat_stream_raw_candidates_scanned,
record_openai_chat_stream_target_select_selected_rank,
};
use crate::upstream_admission::upstream_target_key_from_url;
const OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW_ENV: &str =
"AETHER_GATEWAY_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW";
const DEFAULT_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW: usize = 2;
const MAX_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW: usize = 8;
pub(crate) struct LocalOpenAiChatStreamAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalOpenAiChatDecisionInput,
candidates: LocalOpenAiChatCandidateAttemptSource<'a>,
prefetched_attempts: VecDeque<LocalOpenAiChatCandidateAttempt>,
request_preparation: LocalOpenAiChatRequestPreparation,
}
pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
@@ -40,16 +60,22 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
return Ok(None);
}
let attempt_source_started_at = std::time::Instant::now();
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, true,
)
.await
.await?
else {
return Ok(None);
};
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_lazy_local_openai_chat_candidate_attempt_source(
state, trace_id, &input, body_json, true,
state,
trace_id,
&input,
&effective_body_json,
true,
)
.await;
if candidate_count == 0 {
@@ -71,15 +97,21 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
Some(input.requested_model.as_str()),
candidate_count,
);
observe_gateway_stage_ms(
"openai_chat_attempt_source_build",
attempt_source_started_at.elapsed().as_millis() as u64,
);
Ok(Some((
LocalOpenAiChatStreamAttemptSource {
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
candidates,
prefetched_attempts: VecDeque::new(),
request_preparation: LocalOpenAiChatRequestPreparation,
},
candidate_count,
)))
@@ -88,18 +120,13 @@ pub(crate) async fn build_local_openai_chat_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiChatStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
}
}
apply_local_runtime_candidate_terminal_reason(
self.state,
self.trace_id,
"no_local_stream_plans",
let select_started_at = std::time::Instant::now();
let selected = self.next_execution_attempt_with_target_select().await?;
observe_gateway_stage_ms(
"openai_chat_stream_target_select",
select_started_at.elapsed().as_millis() as u64,
);
Ok(None)
Ok(selected)
}
async fn drain_execution_attempts(&mut self) -> Result<Vec<AiStreamAttempt>, GatewayError> {
@@ -111,11 +138,178 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiChatStreamAttem
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.key_id != key_id);
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.endpoint_id != endpoint_id);
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.prefetched_attempts
.retain(|attempt| attempt.eligible.candidate.provider_id != provider_id);
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiChatStreamAttemptSource<'_> {
async fn build_stream_attempt(
async fn next_execution_attempt_with_target_select(
&mut self,
) -> Result<Option<AiStreamAttempt>, GatewayError> {
loop {
let Some(attempt) = self.next_raw_attempt_with_target_select().await? else {
apply_local_runtime_candidate_terminal_reason(
self.state,
self.trace_id,
"no_local_stream_plans",
);
return Ok(None);
};
let plan_started_at = std::time::Instant::now();
record_openai_chat_stream_payload_build_selected();
match self.build_stream_attempt(attempt).await? {
Some(attempt) => {
observe_gateway_stage_ms(
"stream_candidate_plan_build",
plan_started_at.elapsed().as_millis() as u64,
);
return Ok(Some(attempt));
}
None => {
observe_gateway_stage_ms(
"stream_candidate_plan_build",
plan_started_at.elapsed().as_millis() as u64,
);
continue;
}
}
}
}
async fn next_raw_attempt_with_target_select(
&mut self,
) -> Result<Option<LocalOpenAiChatCandidateAttempt>, GatewayError> {
let select_window = openai_chat_stream_target_select_window();
if select_window <= 1 {
return self.next_raw_attempt_linear().await;
}
let mut attempts = Vec::with_capacity(select_window);
for _ in 0..select_window {
match self.next_raw_attempt_linear().await? {
Some(attempt) => attempts.push(attempt),
None => break,
}
}
if attempts.is_empty() {
return Ok(None);
}
record_openai_chat_stream_raw_candidates_scanned(attempts.len());
let seed = request_distribution_seed();
let target_keys = attempts
.iter()
.map(|attempt| self.lightweight_target_key_for_attempt(attempt))
.collect::<Vec<_>>();
for target_key in target_keys.iter().flatten() {
self.state
.upstream_target_admission
.record_raw_seen_for_target_key(target_key);
}
let selected_index = if target_keys.iter().all(Option::is_some) {
let choices = attempts
.iter()
.zip(target_keys.iter())
.map(|(attempt, target_key)| {
let target_key = target_key.as_deref().unwrap_or("-");
let snapshot = self
.state
.upstream_target_admission
.snapshot_for_target_key(target_key);
TargetSelectChoice {
target_key,
identity: target_select_candidate_identity(attempt),
in_flight: snapshot
.as_ref()
.map(|snapshot| snapshot.in_flight)
.unwrap_or(0),
selection_pressure_total: snapshot
.as_ref()
.map(|snapshot| snapshot.selection_pressure_total)
.unwrap_or(0),
}
})
.collect::<Vec<_>>();
select_target_index(seed, &choices)
} else {
0
};
record_openai_chat_stream_target_select_selected_rank(selected_index);
record_openai_chat_stream_payload_build_prefetch_avoided(attempts.len().saturating_sub(1));
if let Some(Some(target_key)) = target_keys.get(selected_index) {
self.state
.upstream_target_admission
.record_preselect_for_target_key(target_key);
}
let selected = attempts.remove(selected_index);
self.prefetched_attempts.extend(attempts);
Ok(Some(selected))
}
async fn next_raw_attempt_linear(
&mut self,
) -> Result<Option<LocalOpenAiChatCandidateAttempt>, GatewayError> {
if let Some(attempt) = self.prefetched_attempts.pop_front() {
return Ok(Some(attempt));
}
let source_started_at = std::time::Instant::now();
let attempt = self.candidates.next_attempt().await?;
observe_gateway_stage_ms(
"stream_candidate_source_next",
source_started_at.elapsed().as_millis() as u64,
);
Ok(attempt)
}
fn lightweight_target_key_for_attempt(
&self,
attempt: &LocalOpenAiChatCandidateAttempt,
) -> Option<String> {
let provider_api_format = attempt.eligible.provider_api_format.trim();
if !provider_api_format.eq_ignore_ascii_case("openai:chat") {
return None;
}
let transport = &attempt.eligible.transport;
if transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("grok")
|| is_windsurf_provider_transport(transport)
|| local_openai_chat_transport_unsupported_reason(transport).is_some()
{
return None;
}
if transport.provider.proxy.is_some()
|| transport.endpoint.proxy.is_some()
|| transport.key.proxy.is_some()
{
return None;
}
let upstream_url = build_local_openai_chat_upstream_url(self.parts, transport)?;
upstream_target_key_from_url(upstream_url.as_str(), None)
}
async fn build_stream_attempt(
&mut self,
attempt: LocalOpenAiChatCandidateAttempt,
) -> Result<Option<AiStreamAttempt>, GatewayError> {
let upstream_is_stream = openai_chat_upstream_is_stream_for_candidate(
@@ -127,8 +321,9 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
Some(&mut self.request_preparation),
attempt,
OPENAI_CHAT_STREAM_PLAN_KIND,
"openai_chat_stream_success",
@@ -139,7 +334,7 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
return Ok(None);
};
match build_openai_chat_stream_plan_from_decision(self.parts, self.body_json, payload) {
match build_openai_chat_stream_plan_from_decision(self.parts, &self.body_json, payload) {
Ok(value) => Ok(value),
Err(err) => {
warn!(
@@ -153,6 +348,97 @@ impl LocalOpenAiChatStreamAttemptSource<'_> {
}
}
fn openai_chat_stream_target_select_window() -> usize {
std::env::var(OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW_ENV)
.ok()
.and_then(|value| value.trim().parse::<usize>().ok())
.filter(|value| *value > 0)
.unwrap_or(DEFAULT_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW)
.clamp(1, MAX_OPENAI_CHAT_STREAM_TARGET_SELECT_WINDOW)
}
#[derive(Clone, Copy)]
struct TargetSelectCandidateIdentity<'a> {
provider_id: &'a str,
endpoint_id: &'a str,
key_id: &'a str,
candidate_id: &'a str,
}
#[derive(Clone, Copy)]
struct TargetSelectChoice<'a> {
target_key: &'a str,
identity: TargetSelectCandidateIdentity<'a>,
in_flight: usize,
selection_pressure_total: u64,
}
fn select_target_index(seed: u64, choices: &[TargetSelectChoice<'_>]) -> usize {
choices
.iter()
.enumerate()
.min_by_key(|(index, choice)| {
target_select_score(
seed,
choice.target_key,
&choice.identity,
*index,
choice.in_flight,
choice.selection_pressure_total,
)
})
.map(|(index, _)| index)
.unwrap_or(0)
}
fn target_select_candidate_identity(
attempt: &LocalOpenAiChatCandidateAttempt,
) -> TargetSelectCandidateIdentity<'_> {
TargetSelectCandidateIdentity {
provider_id: &attempt.eligible.candidate.provider_id,
endpoint_id: &attempt.eligible.candidate.endpoint_id,
key_id: &attempt.eligible.candidate.key_id,
candidate_id: &attempt.candidate_id,
}
}
fn target_select_tie_break(
seed: u64,
target_key: &str,
identity: &TargetSelectCandidateIdentity<'_>,
index: usize,
) -> u64 {
let mut hash = seed ^ ((index as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15));
hash = hash_string(hash, target_key);
hash = hash_string(hash, identity.provider_id);
hash = hash_string(hash, identity.endpoint_id);
hash = hash_string(hash, identity.key_id);
hash_string(hash, identity.candidate_id)
}
fn hash_string(mut hash: u64, value: &str) -> u64 {
for byte in value.as_bytes() {
hash ^= u64::from(*byte);
hash = hash.wrapping_mul(0x100_0000_01B3);
}
hash
}
fn target_select_score(
seed: u64,
target_key: &str,
identity: &TargetSelectCandidateIdentity<'_>,
index: usize,
in_flight: usize,
selected_total: u64,
) -> (usize, u64, u64) {
(
in_flight,
selected_total,
target_select_tie_break(seed, target_key, identity, index),
)
}
pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
state: &AppState,
parts: &http::request::Parts,
@@ -168,7 +454,7 @@ pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, true,
)
.await
.await?
else {
return Ok(Vec::new());
};
@@ -202,3 +488,82 @@ pub(crate) async fn build_local_openai_chat_stream_plan_and_reports(
Ok(plans)
}
#[cfg(test)]
mod tests {
use super::*;
fn identity<'a>(
endpoint_id: &'a str,
candidate_id: &'a str,
) -> TargetSelectCandidateIdentity<'a> {
TargetSelectCandidateIdentity {
provider_id: "provider",
endpoint_id,
key_id: "key",
candidate_id,
}
}
#[test]
fn target_select_score_prefers_lower_in_flight() {
let busy = identity("endpoint-a", "candidate-a");
let idle = identity("endpoint-b", "candidate-b");
assert!(
target_select_score(7, "http://127.0.0.1:18182|proxy=-", &idle, 1, 0, 10)
< target_select_score(7, "http://127.0.0.1:18181|proxy=-", &busy, 0, 5, 0)
);
}
#[test]
fn target_select_tie_break_distinguishes_equivalent_targets() {
let left = identity("endpoint-a", "candidate-a");
let right = identity("endpoint-b", "candidate-b");
assert_ne!(
target_select_tie_break(11, "http://127.0.0.1:18181|proxy=-", &left, 0),
target_select_tie_break(11, "http://127.0.0.1:18182|proxy=-", &right, 1)
);
}
#[test]
fn select_target_index_prefers_lower_in_flight_target() {
let choices = [
TargetSelectChoice {
target_key: "http://127.0.0.1:18181|proxy=-",
identity: identity("endpoint-a", "candidate-a"),
in_flight: 8,
selection_pressure_total: 0,
},
TargetSelectChoice {
target_key: "http://127.0.0.1:18182|proxy=-",
identity: identity("endpoint-b", "candidate-b"),
in_flight: 1,
selection_pressure_total: 100,
},
];
assert_eq!(select_target_index(17, &choices), 1);
}
#[test]
fn select_target_index_uses_selection_pressure_before_tie_break() {
let choices = [
TargetSelectChoice {
target_key: "http://127.0.0.1:18181|proxy=-",
identity: identity("endpoint-a", "candidate-a"),
in_flight: 0,
selection_pressure_total: 20,
},
TargetSelectChoice {
target_key: "http://127.0.0.1:18182|proxy=-",
identity: identity("endpoint-b", "candidate-b"),
in_flight: 0,
selection_pressure_total: 1,
},
];
assert_eq!(select_target_index(19, &choices), 1);
}
}
@@ -23,7 +23,7 @@ pub(crate) struct LocalOpenAiChatSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalOpenAiChatDecisionInput,
candidates: LocalOpenAiChatCandidateAttemptSource<'a>,
}
@@ -43,13 +43,18 @@ pub(crate) async fn build_local_openai_chat_sync_attempt_source<'a>(
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, true,
)
.await
.await?
else {
return Ok(None);
};
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_lazy_local_openai_chat_candidate_attempt_source(
state, trace_id, &input, body_json, false,
state,
trace_id,
&input,
&effective_body_json,
false,
)
.await;
if candidate_count == 0 {
@@ -77,7 +82,7 @@ pub(crate) async fn build_local_openai_chat_sync_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
candidates,
},
@@ -88,7 +93,7 @@ pub(crate) async fn build_local_openai_chat_sync_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiChatSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -111,6 +116,21 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiChatSyncAttemptSo
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiChatSyncAttemptSource<'_> {
@@ -127,8 +147,9 @@ impl LocalOpenAiChatSyncAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
None,
attempt,
OPENAI_CHAT_SYNC_PLAN_KIND,
"openai_chat_sync_success",
@@ -139,7 +160,7 @@ impl LocalOpenAiChatSyncAttemptSource<'_> {
return Ok(None);
};
match build_openai_chat_sync_plan_from_decision(self.parts, self.body_json, payload) {
match build_openai_chat_sync_plan_from_decision(self.parts, &self.body_json, payload) {
Ok(value) => Ok(value),
Err(err) => {
warn!(
@@ -168,7 +189,7 @@ pub(crate) async fn build_local_openai_chat_sync_plan_and_reports(
let Some(input) = resolve_local_openai_chat_decision_input(
state, parts, trace_id, decision, body_json, plan_kind, true,
)
.await
.await?
else {
return Ok(Vec::new());
};
@@ -0,0 +1,31 @@
use crate::ai_serving::resolve_openai_embedding_sync_spec as resolve_surface_sync_spec;
use crate::ai_serving::GatewayControlDecision;
use crate::{AiExecutionDecision, AppState, GatewayError};
use super::super::family::maybe_build_sync_via_standard_family_payload;
pub(crate) fn resolve_sync_spec(
plan_kind: &str,
) -> Option<super::super::family::LocalStandardSpec> {
resolve_surface_sync_spec(plan_kind)
}
pub(crate) async fn maybe_build_sync_local_openai_embedding_decision_payload(
state: &AppState,
parts: &http::request::Parts,
trace_id: &str,
decision: &GatewayControlDecision,
body_json: &serde_json::Value,
plan_kind: &str,
) -> Result<Option<AiExecutionDecision>, GatewayError> {
maybe_build_sync_via_standard_family_payload(
state,
parts,
trace_id,
decision,
body_json,
plan_kind,
resolve_sync_spec,
)
.await
}
@@ -0,0 +1,55 @@
use serde_json::Value;
pub(super) fn openai_request_is_image_generation_intent(
_requested_model: &str,
body_json: &Value,
) -> bool {
request_forces_image_generation_tool(body_json)
}
fn request_forces_image_generation_tool(body_json: &Value) -> bool {
body_json
.get("tool_choice")
.is_some_and(value_is_image_generation_tool)
}
fn value_is_image_generation_tool(value: &Value) -> bool {
value
.get("type")
.and_then(Value::as_str)
.is_some_and(|tool_type| tool_type.trim().eq_ignore_ascii_case("image_generation"))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn tools_declaration_without_tool_choice_does_not_trigger_image_generation() {
let body_json = serde_json::json!({
"model": "gpt-image-2",
"input": "Draw a mountain observatory",
"tools": [{"type": "image_generation"}]
});
assert!(!openai_request_is_image_generation_intent(
"gpt-image-2",
&body_json
));
}
#[test]
fn explicit_image_generation_tool_choice_triggers_image_generation() {
let body_json = serde_json::json!({
"model": "gpt-image-2",
"input": "Draw a mountain observatory",
"tools": [{"type": "image_generation"}],
"tool_choice": {"type": "image_generation"}
});
assert!(openai_request_is_image_generation_intent(
"gpt-image-2",
&body_json
));
}
}
@@ -1,4 +1,6 @@
mod chat;
mod embedding;
mod image_intent;
mod responses;
pub(crate) use crate::ai_serving::{
@@ -14,6 +16,8 @@ pub(crate) use chat::{
maybe_build_stream_local_decision_payload, maybe_build_sync_local_decision_payload,
set_local_openai_chat_execution_exhausted_diagnostic,
};
pub(crate) use embedding::maybe_build_sync_local_openai_embedding_decision_payload;
use image_intent::openai_request_is_image_generation_intent;
pub(crate) use responses::{
build_local_openai_responses_stream_attempt_source_for_kind,
build_local_openai_responses_stream_plan_and_reports_for_kind,
@@ -129,7 +129,7 @@ pub(crate) fn build_openai_chat_stream_plan_from_decision(
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
stream: effective_upstream_is_stream,
},
);
@@ -229,7 +229,7 @@ pub(crate) fn build_openai_responses_stream_plan_from_decision(
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
stream: effective_upstream_is_stream,
},
);
@@ -291,6 +291,7 @@ mod tests {
request_id: Some("req_123".to_string()),
candidate_id: Some("cand_123".to_string()),
provider_name: Some("Codex".to_string()),
provider_type: Some("codex".to_string()),
provider_id: Some("prov_123".to_string()),
endpoint_id: Some("ep_123".to_string()),
key_id: Some("key_123".to_string()),
@@ -326,6 +327,8 @@ mod tests {
})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -383,6 +386,46 @@ mod tests {
);
}
#[test]
fn build_compact_stream_plan_preserves_non_stream_upstream_mode() {
let parts = http::Request::builder()
.uri("http://localhost/v1/responses/compact")
.body(())
.expect("request should build")
.into_parts()
.0;
let mut payload = sample_responses_payload();
payload.decision_kind = Some("openai_responses_compact_stream".to_string());
payload.upstream_url = Some("https://example.com/v1/responses/compact".to_string());
payload.provider_api_format = Some("openai:responses:compact".to_string());
payload.client_api_format = Some("openai:responses:compact".to_string());
payload.upstream_is_stream = false;
payload.provider_request_body = Some(json!({
"model": "gpt-5.6-sol",
"input": [],
"instructions": "You are Codex.",
"tools": [],
"parallel_tool_calls": true,
"reasoning": {"effort": "high"},
"prompt_cache_key": "cache-key",
"text": {"verbosity": "low"}
}));
let built =
build_openai_responses_stream_plan_from_decision(&parts, &json!({}), payload, true)
.expect("plan build should succeed")
.expect("plan should be produced");
assert!(!built.plan.stream);
assert!(built
.plan
.body
.json_body
.as_ref()
.is_some_and(|body| body.get("stream").is_none()));
assert!(!built.plan.headers.contains_key("accept"));
}
#[test]
fn build_openai_chat_stream_plan_fallback_preserves_complete_same_format_headers() {
let parts = http::Request::builder()
@@ -402,6 +445,7 @@ mod tests {
request_id: Some("req_stream_456".to_string()),
candidate_id: Some("cand_stream_456".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_stream_456".to_string()),
endpoint_id: Some("ep_stream_456".to_string()),
key_id: Some("key_stream_456".to_string()),
@@ -422,6 +466,8 @@ mod tests {
provider_request_body: Some(json!({"model":"gpt-5.4","messages":[],"stream":true})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -458,7 +504,7 @@ mod tests {
}
#[test]
fn build_openai_chat_stream_plan_keeps_downstream_stream_for_force_non_stream_upstream() {
fn build_openai_chat_stream_plan_preserves_force_non_stream_upstream_mode() {
fn force_non_stream_payload(provider_request_body: Option<Value>) -> AiExecutionDecision {
AiExecutionDecision {
action: "stream".to_string(),
@@ -468,6 +514,7 @@ mod tests {
request_id: Some("req_force_non_stream".to_string()),
candidate_id: Some("cand_force_non_stream".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_force_non_stream".to_string()),
endpoint_id: Some("ep_force_non_stream".to_string()),
key_id: Some("key_force_non_stream".to_string()),
@@ -488,6 +535,8 @@ mod tests {
provider_request_body,
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -517,7 +566,7 @@ mod tests {
.expect("plan build should succeed")
.expect("plan should be produced");
assert!(built.plan.stream);
assert!(!built.plan.stream);
assert_eq!(
built
.plan
@@ -542,7 +591,7 @@ mod tests {
.expect("fallback plan build should succeed")
.expect("fallback plan should be produced");
assert!(built.plan.stream);
assert!(!built.plan.stream);
assert_eq!(
built
.plan
@@ -573,6 +622,7 @@ mod tests {
request_id: Some("req_stream_789".to_string()),
candidate_id: Some("cand_stream_789".to_string()),
provider_name: Some("Claude".to_string()),
provider_type: Some("anthropic".to_string()),
provider_id: Some("prov_stream_789".to_string()),
endpoint_id: Some("ep_stream_789".to_string()),
key_id: Some("key_stream_789".to_string()),
@@ -595,6 +645,8 @@ mod tests {
),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -257,6 +257,7 @@ mod tests {
request_id: Some("req_123".to_string()),
candidate_id: Some("cand_123".to_string()),
provider_name: Some("Codex".to_string()),
provider_type: Some("codex".to_string()),
provider_id: Some("prov_123".to_string()),
endpoint_id: Some("ep_123".to_string()),
key_id: Some("key_123".to_string()),
@@ -292,6 +293,8 @@ mod tests {
})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -367,6 +370,7 @@ mod tests {
request_id: Some("req_456".to_string()),
candidate_id: Some("cand_456".to_string()),
provider_name: Some("OpenAI".to_string()),
provider_type: Some("openai".to_string()),
provider_id: Some("prov_456".to_string()),
endpoint_id: Some("ep_456".to_string()),
key_id: Some("key_456".to_string()),
@@ -387,6 +391,8 @@ mod tests {
provider_request_body: Some(json!({"model":"gpt-5.4","messages":[],"stream":false})),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -436,6 +442,7 @@ mod tests {
request_id: Some("req_789".to_string()),
candidate_id: Some("cand_789".to_string()),
provider_name: Some("Claude".to_string()),
provider_type: Some("anthropic".to_string()),
provider_id: Some("prov_789".to_string()),
endpoint_id: Some("ep_789".to_string()),
key_id: Some("key_789".to_string()),
@@ -458,6 +465,8 @@ mod tests {
),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: None,
request_gzip: None,
proxy: None,
transport_profile: None,
timeouts: None,
@@ -2,20 +2,22 @@ use serde_json::json;
use tracing::debug;
use crate::ai_serving::build_request_trace_proxy_value;
use crate::ai_serving::planner::decision_input::apply_provider_request_routing_policy_to_decision;
use crate::ai_serving::planner::report_context::{
build_local_execution_report_context, insert_provider_stream_event_api_format,
LocalExecutionReportContextParts,
build_local_execution_report_context, insert_native_client_envelope_name,
insert_provider_stream_event_api_format, LocalExecutionReportContextParts,
};
use crate::ai_serving::planner::spec_metadata::local_openai_responses_spec_metadata;
use crate::ai_serving::planner::{
build_ai_execution_decision_response, AiExecutionDecisionResponseParts,
build_ai_execution_decision_response, resolve_transport_request_encoding_policy,
AiExecutionDecisionResponseParts,
};
use crate::ai_serving::transport::{
resolve_transport_execution_timeouts, resolve_transport_profile,
};
use crate::{
append_execution_contract_fields_to_value, append_local_failover_policy_to_value,
AiExecutionDecision, AppState,
AiExecutionDecision, AppState, GatewayError,
};
use super::request::resolve_local_openai_responses_candidate_payload_parts;
@@ -30,7 +32,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
input: &LocalOpenAiResponsesDecisionInput,
attempt: LocalOpenAiResponsesCandidateAttempt,
spec: LocalOpenAiResponsesSpec,
) -> Option<AiExecutionDecision> {
) -> Result<Option<AiExecutionDecision>, GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let attempt_identity = attempt.attempt_identity();
let LocalOpenAiResponsesCandidateAttempt {
@@ -39,7 +41,7 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
candidate_id,
..
} = attempt;
let resolved = resolve_local_openai_responses_candidate_payload_parts(
let Some(resolved) = resolve_local_openai_responses_candidate_payload_parts(
state,
parts,
trace_id,
@@ -50,8 +52,16 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
&candidate_id,
spec,
)
.await?;
.await?
else {
return Ok(None);
};
let candidate = &eligible.candidate;
let original_request_body_json = if resolved.request_redacted {
Some(&resolved.provider_request_body)
} else {
Some(body_json)
};
let prompt_cache_key = resolved
.provider_request_body
@@ -63,7 +73,10 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
let proxy = state
.resolve_transport_proxy_snapshot_with_tunnel_affinity(&resolved.transport)
.await;
let transport_profile = resolve_transport_profile(&resolved.transport);
let transport_profile = resolved
.transport_profile
.clone()
.or_else(|| resolve_transport_profile(&resolved.transport));
let timeouts = resolve_transport_execution_timeouts(&resolved.transport);
let mut extra_fields = serde_json::Map::new();
if let Some(proxy_value) =
@@ -73,11 +86,28 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
}
if let Some(envelope_name) = resolved.envelope_name {
extra_fields.insert("envelope_name".to_string(), json!(envelope_name));
insert_native_client_envelope_name(&mut extra_fields, envelope_name, parts.uri.path());
}
if let Some(image_request_summary) = resolved.image_request_summary.as_ref() {
extra_fields.insert("image_request".to_string(), image_request_summary.clone());
}
if resolved
.provider_api_format
.eq_ignore_ascii_case("openai:image")
&& resolved
.transport
.provider
.provider_type
.trim()
.eq_ignore_ascii_case("chatgpt_web")
{
extra_fields.insert("chatgpt_web_image".to_string(), json!(true));
}
insert_provider_stream_event_api_format(
&mut extra_fields,
resolved.transport.provider.provider_type.as_str(),
);
let effective_headers = input.effective_headers(&parts.headers);
let report_context = append_local_failover_policy_to_value(
append_execution_contract_fields_to_value(
build_local_execution_report_context(LocalExecutionReportContextParts {
@@ -105,13 +135,14 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
body_rules: resolved.transport.endpoint.body_rules.as_ref(),
provider_request_method: Some(serde_json::Value::Null),
provider_request_headers: Some(&resolved.provider_request_headers),
original_headers: &parts.headers,
original_headers: effective_headers,
request_path: Some(parts.uri.path()),
request_query_string: parts.uri.query(),
request_origin: Some(crate::ai_serving::request_origin_from_parts(parts)),
original_request_body_json: Some(body_json),
original_request_body_json,
original_request_body_base64: None,
client_session_affinity: input.client_session_affinity.as_ref(),
routing_policy: input.routing_policy.as_ref(),
scheduler_affinity_epoch: eligible.orchestration.scheduler_affinity_epoch,
client_requested_stream: body_json
.get("stream")
@@ -170,41 +201,52 @@ pub(crate) async fn maybe_build_local_openai_responses_decision_payload_for_cand
envelope_name: _,
upstream_is_stream,
transport,
transport_profile: _,
image_request_summary: _,
request_redacted: _,
} = resolved;
let request_encoding = resolve_transport_request_encoding_policy(&transport);
Some(build_ai_execution_decision_response(
AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
proxy,
transport_profile,
timeouts,
upstream_is_stream,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
},
))
let mut decision = build_ai_execution_decision_response(AiExecutionDecisionResponseParts {
decision_is_stream: spec_metadata.require_streaming,
decision_kind: spec_metadata.decision_kind.to_string(),
execution_strategy,
conversion_mode,
request_id: trace_id.to_string(),
candidate_id: candidate_id.clone(),
provider_name: transport.provider.name.clone(),
provider_type: transport.provider.provider_type.clone(),
provider_id: candidate.provider_id.clone(),
endpoint_id: candidate.endpoint_id.clone(),
key_id: candidate.key_id.clone(),
upstream_base_url: transport.endpoint.base_url.clone(),
upstream_url,
provider_request_method: None,
auth_header: Some(auth_header),
auth_value: Some(auth_value),
provider_api_format,
client_api_format: spec_metadata.api_format.to_string(),
model_name: input.requested_model.clone(),
mapped_model,
prompt_cache_key,
provider_request_headers,
provider_request_body: Some(provider_request_body),
provider_request_body_base64: None,
content_type: Some("application/json".to_string()),
content_encoding: request_encoding.content_encoding,
request_gzip: request_encoding.request_gzip,
proxy,
transport_profile,
timeouts,
upstream_is_stream,
report_kind: spec_metadata.report_kind.map(ToOwned::to_owned),
report_context: Some(report_context),
auth_context: input.auth_context.clone(),
});
apply_provider_request_routing_policy_to_decision(
input,
&mut decision,
Some(transport.as_ref()),
)?;
Ok(Some(decision))
}
File diff suppressed because it is too large Load Diff
@@ -15,10 +15,12 @@ use crate::ai_serving::planner::candidate_metadata::{
LocalExecutionCandidateMetadataParts,
};
use crate::ai_serving::planner::candidate_source::{
preselect_local_execution_candidates_for_api_formats_with_serving,
preselect_local_execution_candidates_with_serving, LocalCandidatePreselectionKeyMode,
};
use crate::ai_serving::planner::common::extract_standard_requested_model;
use crate::ai_serving::planner::decision_input::{
attach_routing_policy_to_local_requested_model_input,
build_local_requested_model_decision_input, resolve_local_authenticated_decision_input,
};
use crate::ai_serving::planner::materialization_policy::{
@@ -29,12 +31,13 @@ use crate::ai_serving::planner::spec_metadata::local_openai_responses_spec_metad
use crate::ai_serving::planner::CandidateFailureDiagnostic;
use crate::ai_serving::{
ai_local_execution_contract_for_formats, extract_pool_sticky_session_token,
resolve_local_decision_execution_runtime_auth_context, ExecutionRuntimeAuthContext,
GatewayControlDecision, PlannerAppState,
openai_responses_request_operation, resolve_local_decision_execution_runtime_auth_context,
ExecutionRuntimeAuthContext, GatewayControlDecision, PlannerAppState,
};
use crate::client_session_affinity::client_session_affinity_from_parts;
use crate::{AppState, GatewayError};
use super::super::super::openai_request_is_image_generation_intent;
use super::LocalOpenAiResponsesSpec;
pub(crate) use crate::ai_serving::planner::candidate_materialization::LocalExecutionCandidateAttempt as LocalOpenAiResponsesCandidateAttempt;
@@ -48,7 +51,7 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
decision: &GatewayControlDecision,
body_json: &serde_json::Value,
plan_kind: &str,
) -> Option<LocalOpenAiResponsesDecisionInput> {
) -> Result<Option<LocalOpenAiResponsesDecisionInput>, GatewayError> {
let Some(auth_context) = resolve_local_decision_execution_runtime_auth_context(decision) else {
warn!(
trace_id = %trace_id,
@@ -65,7 +68,7 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
extract_standard_requested_model(body_json).as_deref(),
"missing_auth_context",
);
return None;
return Ok(None);
};
let Some(requested_model) = extract_standard_requested_model(body_json) else {
@@ -81,14 +84,16 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
None,
"missing_requested_model",
);
return None;
return Ok(None);
};
let resolved_input = match resolve_local_authenticated_decision_input(
state,
auth_context.clone(),
Some(requested_model.as_str()),
decision.auth_endpoint_signature.as_deref(),
None,
&decision.model_directive_policy,
)
.await
{
@@ -108,7 +113,7 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
Some(requested_model.as_str()),
"auth_snapshot_missing",
);
return None;
return Ok(None);
}
Err(err) => {
warn!(
@@ -124,14 +129,30 @@ pub(crate) async fn resolve_local_openai_responses_decision_input(
Some(requested_model.as_str()),
"auth_snapshot_read_failed",
);
return None;
return Err(err);
}
};
let mut input = build_local_requested_model_decision_input(resolved_input, requested_model);
input.request_auth_channel = decision.request_auth_channel.clone();
input.client_session_affinity = client_session_affinity_from_parts(parts, Some(body_json));
Some(input)
if let Err(err) = attach_routing_policy_to_local_requested_model_input(
state,
parts,
&mut input,
body_json,
"openai:responses",
)
.await
{
warn!(
trace_id = %trace_id,
error = ?err,
"gateway local openai responses decision routing profile resolution failed"
);
return Err(err);
}
Ok(Some(input))
}
pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
@@ -142,6 +163,7 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
spec: LocalOpenAiResponsesSpec,
) -> Result<(Vec<LocalOpenAiResponsesCandidateAttempt>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let request_operation = openai_responses_request_operation(spec_metadata.api_format, body_json);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
@@ -152,11 +174,14 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
);
let preselection = preselect_local_execution_candidates_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
request_operation,
spec_metadata.require_streaming,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
input.routing_policy.as_ref(),
input.client_session_affinity.as_ref(),
true,
LocalCandidatePreselectionKeyMode::ProviderEndpointKeyModelAndApiFormat,
@@ -170,6 +195,7 @@ pub(crate) async fn materialize_local_openai_responses_candidate_attempts(
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -238,6 +264,7 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
spec: LocalOpenAiResponsesSpec,
) -> Result<(LocalOpenAiResponsesCandidateAttemptSource<'a>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let request_operation = openai_responses_request_operation(spec_metadata.api_format, body_json);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
@@ -246,16 +273,29 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::OpenAiResponsesDecision,
);
if openai_request_is_image_generation_intent(&input.requested_model, body_json) {
let (image_candidates, image_candidate_count) =
build_local_openai_responses_image_candidate_attempt_source(
state, trace_id, input, body_json, spec,
)
.await?;
if image_candidate_count > 0 {
return Ok((image_candidates, image_candidate_count));
}
}
Ok(
build_lazy_requested_model_execution_candidate_attempt_source_with_serving(
planner_state,
&input.model_directive_policy,
trace_id,
spec_metadata.api_format,
&input.requested_model,
request_operation,
spec_metadata.require_streaming,
&input.auth_snapshot,
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
@@ -315,6 +355,106 @@ pub(crate) async fn build_local_openai_responses_candidate_attempt_source<'a>(
)
}
pub(crate) async fn build_local_openai_responses_image_candidate_attempt_source<'a>(
state: &'a AppState,
trace_id: &str,
input: &LocalOpenAiResponsesDecisionInput,
body_json: &serde_json::Value,
spec: LocalOpenAiResponsesSpec,
) -> Result<(LocalOpenAiResponsesCandidateAttemptSource<'a>, usize), GatewayError> {
let spec_metadata = local_openai_responses_spec_metadata(spec);
let planner_state = PlannerAppState::new(state);
let sticky_session_token = extract_pool_sticky_session_token(body_json);
let auth_context: &ExecutionRuntimeAuthContext = &input.auth_context;
let persistence_policy = build_local_candidate_persistence_policy(
auth_context,
input.required_capabilities.as_ref(),
LocalCandidatePersistencePolicyKind::OpenAiResponsesDecision,
);
let preselection = preselect_local_execution_candidates_for_api_formats_with_serving(
planner_state,
&input.model_directive_policy,
spec_metadata.api_format,
&input.requested_model,
None,
false,
input.required_capabilities.as_ref(),
&input.auth_snapshot,
input.routing_policy.as_ref(),
input.client_session_affinity.as_ref(),
true,
LocalCandidatePreselectionKeyMode::ProviderEndpointKeyModelAndApiFormat,
vec!["openai:image".to_string()],
)
.await?;
Ok(build_local_execution_candidate_attempt_source_with_serving(
planner_state,
trace_id,
spec_metadata.api_format,
Some(&input.requested_model),
Some(&input.auth_snapshot),
input.client_session_affinity.as_ref(),
input.required_capabilities.as_ref(),
input.routing_policy.as_ref(),
sticky_session_token.as_deref(),
input.request_auth_channel.as_deref(),
persistence_policy,
preselection.candidates,
preselection.skipped_candidates,
LocalCandidateResolutionMode::WithoutTransportPairGate,
move |eligible| {
let provider_api_format = eligible.provider_api_format.clone();
let (execution_strategy, conversion_mode) = ai_local_execution_contract_for_formats(
spec_metadata.api_format,
&provider_api_format,
);
Some(build_local_execution_candidate_contract_metadata(
LocalExecutionCandidateMetadataParts {
eligible,
provider_api_format: provider_api_format.as_str(),
client_api_format: spec_metadata.api_format,
extra_fields: serde_json::Map::new(),
},
execution_strategy,
conversion_mode,
eligible.candidate.endpoint_api_format.as_str(),
))
},
move |mut skipped_candidate| {
let provider_api_format = skipped_candidate
.transport
.as_ref()
.map(|transport| transport.endpoint.api_format.trim().to_ascii_lowercase())
.unwrap_or_else(|| {
skipped_candidate
.candidate
.endpoint_api_format
.trim()
.to_ascii_lowercase()
});
let (execution_strategy, conversion_mode) = ai_local_execution_contract_for_formats(
spec_metadata.api_format,
&provider_api_format,
);
skipped_candidate.extra_data = Some(
build_local_execution_candidate_contract_metadata_for_candidate(
&skipped_candidate.candidate,
skipped_candidate.transport_ref(),
provider_api_format.as_str(),
spec_metadata.api_format,
serde_json::Map::new(),
execution_strategy,
conversion_mode,
provider_api_format.as_str(),
),
);
skipped_candidate
},
)
.await)
}
pub(crate) async fn mark_skipped_local_openai_responses_candidate(
state: &AppState,
input: &LocalOpenAiResponsesDecisionInput,
@@ -103,21 +103,22 @@ pub(crate) async fn maybe_build_sync_local_openai_responses_decision_payload(
let Some(input) = resolve_local_openai_responses_decision_input(
state, parts, trace_id, decision, body_json, plan_kind,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let (mut source, _) = build_local_openai_responses_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
)
.await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -141,21 +142,22 @@ pub(crate) async fn maybe_build_stream_local_openai_responses_decision_payload(
let Some(input) = resolve_local_openai_responses_decision_input(
state, parts, trace_id, decision, body_json, plan_kind,
)
.await
.await?
else {
return Ok(None);
};
let body_json = input.effective_body_json(body_json);
let (mut source, _) = build_local_openai_responses_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
)
.await?;
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
if let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
{
return Ok(Some(payload));
}
@@ -29,7 +29,7 @@ pub(crate) struct LocalOpenAiResponsesSyncAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalOpenAiResponsesDecisionInput,
spec: LocalOpenAiResponsesSpec,
candidates: LocalOpenAiResponsesCandidateAttemptSource<'a>,
@@ -39,7 +39,7 @@ pub(crate) struct LocalOpenAiResponsesStreamAttemptSource<'a> {
state: &'a AppState,
parts: &'a http::request::Parts,
trace_id: &'a str,
body_json: &'a serde_json::Value,
body_json: serde_json::Value,
input: LocalOpenAiResponsesDecisionInput,
spec: LocalOpenAiResponsesSpec,
candidates: LocalOpenAiResponsesCandidateAttemptSource<'a>,
@@ -62,7 +62,7 @@ pub(super) async fn build_local_sync_attempt_source<'a>(
body_json,
spec_metadata.decision_kind,
)
.await
.await?
else {
return Ok(None);
};
@@ -74,8 +74,13 @@ pub(super) async fn build_local_sync_attempt_source<'a>(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_local_openai_responses_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
state,
trace_id,
&input,
&effective_body_json,
spec,
)
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
@@ -88,7 +93,7 @@ pub(super) async fn build_local_sync_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
spec,
candidates,
@@ -114,7 +119,7 @@ pub(super) async fn build_local_stream_attempt_source<'a>(
body_json,
spec_metadata.decision_kind,
)
.await
.await?
else {
return Ok(None);
};
@@ -126,8 +131,13 @@ pub(super) async fn build_local_stream_attempt_source<'a>(
Some(input.requested_model.as_str()),
"candidate_evaluation_incomplete",
);
let effective_body_json = input.effective_body_json(body_json).clone();
let (candidates, candidate_count) = build_local_openai_responses_candidate_attempt_source(
state, trace_id, &input, body_json, spec,
state,
trace_id,
&input,
&effective_body_json,
spec,
)
.await?;
apply_local_runtime_candidate_evaluation_progress(state, trace_id, candidate_count);
@@ -140,7 +150,7 @@ pub(super) async fn build_local_stream_attempt_source<'a>(
state,
parts,
trace_id,
body_json,
body_json: effective_body_json,
input,
spec,
candidates,
@@ -152,7 +162,7 @@ pub(super) async fn build_local_stream_attempt_source<'a>(
#[async_trait]
impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiResponsesSyncAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiSyncAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_sync_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -175,12 +185,27 @@ impl LocalExecutionAttemptSource<AiSyncAttempt> for LocalOpenAiResponsesSyncAtte
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
#[async_trait]
impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiResponsesStreamAttemptSource<'_> {
async fn next_execution_attempt(&mut self) -> Result<Option<AiStreamAttempt>, GatewayError> {
while let Some(attempt) = self.candidates.next_attempt().await {
while let Some(attempt) = self.candidates.next_attempt().await? {
match self.build_stream_attempt(attempt).await? {
Some(attempt) => return Ok(Some(attempt)),
None => continue,
@@ -203,6 +228,21 @@ impl LocalExecutionAttemptSource<AiStreamAttempt> for LocalOpenAiResponsesStream
}
Ok(drained)
}
async fn skip_credential(&mut self, key_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_credential(key_id);
Ok(())
}
async fn skip_endpoint(&mut self, endpoint_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_endpoint(endpoint_id);
Ok(())
}
async fn skip_provider(&mut self, provider_id: &str) -> Result<(), GatewayError> {
self.candidates.skip_provider(provider_id);
Ok(())
}
}
impl LocalOpenAiResponsesSyncAttemptSource<'_> {
@@ -214,19 +254,19 @@ impl LocalOpenAiResponsesSyncAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
match build_openai_responses_sync_plan_from_decision(
self.parts,
self.body_json,
&self.body_json,
payload,
self.spec.compact,
) {
@@ -252,19 +292,19 @@ impl LocalOpenAiResponsesStreamAttemptSource<'_> {
self.state,
self.parts,
self.trace_id,
self.body_json,
&self.body_json,
&self.input,
attempt,
self.spec,
)
.await
.await?
else {
return Ok(None);
};
match build_openai_responses_stream_plan_from_decision(
self.parts,
self.body_json,
&self.body_json,
payload,
self.spec.compact,
) {
@@ -298,7 +338,7 @@ pub(super) async fn build_local_sync_plan_and_reports(
body_json,
spec_metadata.decision_kind,
)
.await
.await?
else {
return Ok(Vec::new());
};
@@ -321,11 +361,11 @@ pub(super) async fn build_local_sync_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -370,7 +410,7 @@ pub(super) async fn build_local_stream_plan_and_reports(
body_json,
spec_metadata.decision_kind,
)
.await
.await?
else {
return Ok(Vec::new());
};
@@ -393,11 +433,11 @@ pub(super) async fn build_local_stream_plan_and_reports(
}
let mut plans = Vec::new();
while let Some(attempt) = source.next_attempt().await {
while let Some(attempt) = source.next_attempt().await? {
let Some(payload) = maybe_build_local_openai_responses_decision_payload_for_candidate(
state, parts, trace_id, body_json, &input, attempt, spec,
)
.await
.await?
else {
continue;
};
@@ -1,9 +1,8 @@
use std::collections::BTreeMap;
use aether_contracts::RequestBody;
use super::{
augment_sync_report_context, build_ai_execution_plan_from_decision, take_ai_decision_plan_core,
augment_sync_report_context, build_ai_execution_plan_from_decision,
resolve_ai_passthrough_sync_request_body, take_ai_decision_plan_core,
take_ai_upstream_auth_pair, take_non_empty_string, AiExecutionPlanFromDecisionParts,
AiStreamAttempt, AiSyncAttempt,
};
@@ -63,6 +62,10 @@ pub(crate) fn build_standard_sync_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
@@ -72,7 +75,7 @@ pub(crate) fn build_standard_sync_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
body: request_body,
stream,
},
);
@@ -146,6 +149,11 @@ pub(crate) fn build_standard_stream_plan_from_decision(
&provider_request_headers,
&provider_request_body_value,
)?;
let request_body = resolve_ai_passthrough_sync_request_body(
Some(provider_request_body_value),
payload.provider_request_body_base64.take(),
);
let stream = payload.upstream_is_stream;
let plan = build_ai_execution_plan_from_decision(
&mut payload,
AiExecutionPlanFromDecisionParts {
@@ -154,8 +162,8 @@ pub(crate) fn build_standard_stream_plan_from_decision(
url,
headers: std::mem::take(&mut provider_request_headers),
content_type,
body: RequestBody::from_json(provider_request_body_value),
stream: true,
body: request_body,
stream,
},
);
@@ -165,3 +173,88 @@ pub(crate) fn build_standard_stream_plan_from_decision(
report_context,
}))
}
#[cfg(test)]
mod tests {
use aether_contracts::{ExecutionResponseBodyMode, EXECUTION_RESPONSE_BODY_MODE_HEADER};
use serde_json::json;
use super::{
build_standard_stream_plan_from_decision, build_standard_sync_plan_from_decision,
AiExecutionDecision,
};
fn decision_with_raw_body(upstream_is_stream: bool) -> AiExecutionDecision {
serde_json::from_value(json!({
"action": if upstream_is_stream { "stream" } else { "sync" },
"request_id": "req-raw",
"provider_id": "provider-raw",
"endpoint_id": "endpoint-raw",
"key_id": "key-raw",
"upstream_url": "https://api.anthropic.test/v1/messages",
"provider_api_format": "claude:messages",
"client_api_format": "claude:messages",
"provider_request_headers": {
"content-type": "application/json",
(EXECUTION_RESPONSE_BODY_MODE_HEADER): ExecutionResponseBodyMode::PreserveBytes.as_str()
},
"provider_request_body": {
"model": "claude-sonnet-4",
"messages": []
},
"provider_request_body_base64": "eyAibW9kZWwiOiAiY2xhdWRlLXNvbm5ldC00IiwgIm1lc3NhZ2VzIjogW10gfQ==",
"content_type": "application/json",
"upstream_is_stream": upstream_is_stream
}))
.expect("decision should deserialize")
}
fn request_parts() -> http::request::Parts {
http::Request::builder()
.uri("http://localhost/v1/messages")
.body(())
.expect("request should build")
.into_parts()
.0
}
#[test]
fn standard_sync_plan_prefers_exact_request_body_bytes() {
let built = build_standard_sync_plan_from_decision(
&request_parts(),
&json!({}),
decision_with_raw_body(false),
)
.expect("plan should build")
.expect("plan should exist");
assert!(built.plan.body.json_body.is_none());
assert_eq!(
built.plan.body.body_bytes_b64.as_deref(),
Some("eyAibW9kZWwiOiAiY2xhdWRlLXNvbm5ldC00IiwgIm1lc3NhZ2VzIjogW10gfQ==")
);
assert_eq!(
built
.plan
.headers
.get(EXECUTION_RESPONSE_BODY_MODE_HEADER)
.map(String::as_str),
Some(ExecutionResponseBodyMode::PreserveBytes.as_str())
);
}
#[test]
fn standard_stream_plan_prefers_exact_request_body_bytes() {
let built = build_standard_stream_plan_from_decision(
&request_parts(),
&json!({}),
decision_with_raw_body(true),
false,
)
.expect("plan should build")
.expect("plan should exist");
assert!(built.plan.body.json_body.is_none());
assert!(built.plan.body.body_bytes_b64.is_some());
}
}
@@ -10,9 +10,7 @@ impl<'a> PlannerAppState<'a> {
now_unix_secs: u64,
) -> Result<Option<GatewayAuthApiKeySnapshot>, GatewayError> {
self.app()
.data
.read_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
.read_cached_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
.await
.map_err(|err| GatewayError::Internal(err.to_string()))
}
}
@@ -10,16 +10,15 @@ impl<'a> PlannerAppState<'a> {
api_key_id: &str,
requested_model: Option<&str>,
explicit_required_capabilities: Option<&Value>,
model_directive_base_model: Option<&str>,
) -> Option<Value> {
let enable_model_directives =
crate::system_features::reasoning_model_directive_enabled(self.app()).await;
crate::request_candidate_runtime::resolve_request_candidate_required_capabilities(
self.app(),
user_id,
api_key_id,
requested_model,
explicit_required_capabilities,
enable_model_directives,
model_directive_base_model,
)
.await
}

Some files were not shown because too many files have changed in this diff Show More