mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-10 11:19:50 +08:00
feat(gateway): Codex/OpenAI Responses WebSocket 代理模式
在 /v1/responses 上支持 WebSocket 升级,把客户端帧中继到上游 Codex / OpenAI Responses WebSocket 端点,同时保持既有的路由、鉴权、配额与用量 语义: - 路由与准入:control/route/ai.rs 识别 WebSocket 升级请求; websocket/ingress.rs 复用 API Key 鉴权、IP 规则与并发许可,并引入 独立的 WebSocket 连接许可 - 中继:websocket/responses/* 按 connection / session / turn 分层, 帧解析归一化、socket 写入有界、continuation 保持调度亲和性 - 配额:orchestration/codex_quota_breaker.rs 在账号配额耗尽时熔断并 自动恢复,不再直接断开客户端连接 - 用量:每个 turn 的终态用量落库,request_metadata 记录 websocket_mode / websocket_transport,管理端与 usage 视图暴露 is_websocket - 管理端:provider 可配置 Responses WebSocket 开关
This commit is contained in:
@@ -1330,9 +1330,21 @@ struct Args {
|
||||
#[arg(long, env = "AETHER_GATEWAY_MAX_IN_FLIGHT_REQUESTS")]
|
||||
max_in_flight_requests: Option<usize>,
|
||||
|
||||
/// Maximum number of long-lived public WebSocket connections. When unset,
|
||||
/// this follows `max_in_flight_requests` while remaining an independent
|
||||
/// gate. Set `AETHER_GATEWAY_MAX_WEBSOCKET_CONNECTIONS` to override it.
|
||||
#[arg(long, env = "AETHER_GATEWAY_MAX_WEBSOCKET_CONNECTIONS")]
|
||||
max_websocket_connections: Option<usize>,
|
||||
|
||||
#[arg(long, env = "AETHER_GATEWAY_DISTRIBUTED_REQUEST_LIMIT")]
|
||||
distributed_request_limit: Option<usize>,
|
||||
|
||||
/// Optional distributed limit for long-lived WebSocket connections. When
|
||||
/// omitted, the distributed request limit is reused; set it to 0 to keep
|
||||
/// WebSocket admission local-only.
|
||||
#[arg(long, env = "AETHER_GATEWAY_DISTRIBUTED_WEBSOCKET_CONNECTION_LIMIT")]
|
||||
distributed_websocket_connection_limit: Option<usize>,
|
||||
|
||||
#[arg(long, env = "AETHER_GATEWAY_DISTRIBUTED_REQUEST_REDIS_URL")]
|
||||
distributed_request_redis_url: Option<String>,
|
||||
|
||||
@@ -1820,6 +1832,15 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
.max_in_flight_requests
|
||||
.filter(|limit| *limit > 0)
|
||||
.unwrap_or_else(automatic_gateway_request_concurrency);
|
||||
let websocket_connection_limit = args
|
||||
.max_websocket_connections
|
||||
.filter(|limit| *limit > 0)
|
||||
.unwrap_or(request_concurrency_limit);
|
||||
let distributed_websocket_connection_limit = match args.distributed_websocket_connection_limit {
|
||||
Some(limit) if limit > 0 => Some(limit),
|
||||
Some(_) => None,
|
||||
None => args.distributed_request_limit.filter(|limit| *limit > 0),
|
||||
};
|
||||
let usage_queue_request_concurrency_hint = usage_queue_request_concurrency_hint(
|
||||
Some(request_concurrency_limit),
|
||||
args.distributed_request_limit,
|
||||
@@ -1941,6 +1962,14 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
"auto"
|
||||
},
|
||||
distributed_request_limit = args.distributed_request_limit.unwrap_or_default(),
|
||||
max_websocket_connections = websocket_connection_limit,
|
||||
max_websocket_connections_source = if args.max_websocket_connections.is_some() {
|
||||
"explicit"
|
||||
} else {
|
||||
"request_concurrency_fallback"
|
||||
},
|
||||
distributed_websocket_connection_limit =
|
||||
distributed_websocket_connection_limit.unwrap_or_default(),
|
||||
distributed_request_redis_configured = args
|
||||
.distributed_request_redis_url
|
||||
.as_deref()
|
||||
@@ -1995,7 +2024,9 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
{
|
||||
state = state.with_video_task_store_path(path)?;
|
||||
}
|
||||
state = state.with_request_concurrency_limit(request_concurrency_limit);
|
||||
state = state
|
||||
.with_request_concurrency_limit(request_concurrency_limit)
|
||||
.with_websocket_connection_limit(websocket_connection_limit);
|
||||
if let Some(limit) = args.distributed_request_limit.filter(|limit| *limit > 0) {
|
||||
let distributed_gate = state
|
||||
.runtime_state()
|
||||
@@ -2013,6 +2044,23 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
})?;
|
||||
state = state.with_distributed_request_concurrency_gate(distributed_gate);
|
||||
}
|
||||
if let Some(limit) = distributed_websocket_connection_limit {
|
||||
let distributed_gate = state
|
||||
.runtime_state()
|
||||
.semaphore(
|
||||
"gateway_websocket_connections_distributed",
|
||||
limit,
|
||||
RuntimeSemaphoreConfig {
|
||||
lease_ttl_ms: args.distributed_request_lease_ttl_ms.max(1),
|
||||
renew_interval_ms: args.distributed_request_renew_interval_ms.max(1),
|
||||
command_timeout_ms: Some(args.distributed_request_command_timeout_ms.max(1)),
|
||||
},
|
||||
)
|
||||
.map_err(|err| {
|
||||
std::io::Error::new(std::io::ErrorKind::InvalidInput, err.to_string())
|
||||
})?;
|
||||
state = state.with_distributed_websocket_connection_gate(distributed_gate);
|
||||
}
|
||||
if matches!(args.deployment_topology, DeploymentTopologyArg::MultiNode)
|
||||
&& !state.has_usage_data_writer()
|
||||
{
|
||||
@@ -2531,7 +2579,9 @@ mod tests {
|
||||
video_task_poller_batch_size: 32,
|
||||
video_task_store_path: None,
|
||||
max_in_flight_requests: None,
|
||||
max_websocket_connections: None,
|
||||
distributed_request_limit: None,
|
||||
distributed_websocket_connection_limit: None,
|
||||
distributed_request_redis_url: None,
|
||||
distributed_request_redis_key_prefix: None,
|
||||
distributed_request_lease_ttl_ms: 30_000,
|
||||
|
||||
Reference in New Issue
Block a user