merge(main): sync latest main into security branch

This commit is contained in:
elky
2026-09-05 00:30:16 +08:00
40 changed files with 2513 additions and 119 deletions
@@ -1,6 +1,5 @@
use std::collections::BTreeMap;
use aether_ai_formats::openai_responses_message_item_id;
use axum::body::to_bytes;
use base64::Engine as _;
use serde_json::json;
@@ -12,10 +11,10 @@ use super::{
convert_gemini_chat_response_to_openai_chat, convert_gemini_response_to_openai_responses,
maybe_build_local_core_sync_finalize_response,
};
use crate::ai_serving::GatewayControlDecision;
use crate::ai_serving::{
convert_openai_chat_response_to_openai_responses,
convert_openai_responses_response_to_openai_chat,
convert_openai_responses_response_to_openai_chat, openai_responses_message_item_id,
GatewayControlDecision,
};
use crate::usage::GatewaySyncReportRequest;
@@ -684,7 +684,7 @@ mod tests {
.unwrap_or_default();
// One Arc is retained by the map and every active request
// owns one through its leader guard or follower state.
if participant_count >= participants + 1 {
if participant_count > participants {
break;
}
tokio::task::yield_now().await;
@@ -0,0 +1,568 @@
//! Terminal settlement for a local stream attempt whose future is dropped
//! mid-flight.
//!
//! A local stream attempt writes its `usage` row and its `request_candidates`
//! slot as `pending` before it dispatches to the provider, then keeps running
//! inside the downstream request future. When the client disconnects, axum drops
//! that future: the remaining `.await`s never resume and nothing settles either
//! row. They stay `pending` until the maintenance sweeper rewrites them as a 504
//! timeout roughly ten minutes later, which loses the real outcome and the real
//! latency.
//!
//! The stream transport therefore keeps a guard alive across the window between
//! the `pending` write and terminal settlement, and settles the attempt from
//! `Drop` when that window is left by cancellation instead of by a terminal
//! state.
use std::sync::Arc;
use std::time::Instant;
use aether_contracts::ExecutionPlan;
use aether_data_contracts::repository::candidates::RequestCandidateStatus;
use aether_scheduler_core::SchedulerRequestCandidateStatusUpdate;
use aether_usage_runtime::{
build_usage_event_data_seed_describing_request_bodies, UsageEvent, UsageEventData,
UsageEventType,
};
use serde_json::{json, Value};
use tracing::warn;
use crate::clock::current_unix_ms as current_request_candidate_unix_ms;
use crate::execution_runtime::attempt_lifecycle::CLIENT_CANCELLED_STATUS_CODE;
use crate::execution_runtime::transport_failure::StreamCandidateWatchdogProgress;
use crate::log_ids::short_request_id;
use crate::request_candidate_runtime::{
record_local_request_candidate_status_snapshot, LocalRequestCandidateStatusSnapshot,
};
use crate::request_diagnostics::{
attach_request_diagnostics_to_report_context, current_request_diagnostics, RequestDiagnostics,
};
use crate::AppState;
fn elapsed_ms_since(started_at: Instant) -> u64 {
started_at.elapsed().as_millis().min(u128::from(u64::MAX)) as u64
}
/// The facts the guard needs to settle the attempt it is watching.
///
/// This is held for the whole attempt, so it is deliberately free of request
/// bodies. A request body can be megabytes, and holding one per in-flight
/// attempt would cost far more than the row it settles: the usage seed is built
/// with [`build_usage_event_data_seed_describing_request_bodies`], which derives
/// every capture state, body reference and derived request fact from the real
/// plan and report context but keeps neither body. The terminal write it
/// produces therefore preserves the capture the `pending` write recorded instead
/// of clearing it.
struct ArmedAttempt {
request_id: String,
candidate_id: Option<String>,
candidate: Option<LocalRequestCandidateStatusSnapshot>,
// Boxed: the guard lives inside the stream request future, which is already
// very large, and `UsageEventData` is a wide struct.
usage_seed: Option<Box<UsageEventData>>,
request_diagnostics: Option<Arc<RequestDiagnostics>>,
candidate_started_unix_ms: u64,
candidate_started_at: Instant,
}
/// Settles an attempt as cancelled when its future is dropped before the
/// transport reaches a terminal state.
///
/// The guard is created disarmed and stays inert until [`Self::arm`] is called,
/// so an attempt that is dropped before it owns any `pending` row does not grow
/// a settlement row it never had. The owner disarms it as soon as the attempt
/// completes, whichever way it completes: from that point terminal settlement
/// belongs to the transport (for streams, to the stream finalizer that lives in
/// the response body), and the guard must not write a second terminal state.
///
/// A stream candidate also runs under a first-byte watchdog that drops the
/// attempt future when it gives up. That drop is not a client disconnect and the
/// watchdog settles the attempt itself, so the guard stands down for it.
pub(crate) struct AttemptCancellationGuard {
state: AppState,
error_type: &'static str,
error_message: &'static str,
watchdog: Option<Arc<StreamCandidateWatchdogProgress>>,
armed: Option<ArmedAttempt>,
}
impl AttemptCancellationGuard {
pub(crate) fn disarmed(
state: &AppState,
error_type: &'static str,
error_message: &'static str,
) -> Self {
Self {
state: state.clone(),
error_type,
error_message,
watchdog: StreamCandidateWatchdogProgress::current(),
armed: None,
}
}
/// Takes ownership of the attempt's settlement until it is disarmed.
pub(crate) fn arm(
&mut self,
plan: &ExecutionPlan,
report_context: Option<&Value>,
candidate: Option<&LocalRequestCandidateStatusSnapshot>,
candidate_started_unix_ms: u64,
candidate_started_at: Instant,
) {
let usage_seed = self.state.usage_runtime.is_enabled().then(|| {
Box::new(build_usage_event_data_seed_describing_request_bodies(
plan,
report_context,
))
});
self.armed = Some(ArmedAttempt {
request_id: plan.request_id.clone(),
candidate_id: plan.candidate_id.clone(),
candidate: candidate.cloned(),
usage_seed,
request_diagnostics: current_request_diagnostics(),
candidate_started_unix_ms,
candidate_started_at,
});
}
pub(crate) fn disarm(&mut self) {
self.armed = None;
}
}
/// Writes the candidate terminal row and the terminal usage event for an attempt
/// that never reached its own terminal path.
async fn settle_cancelled_attempt(
state: AppState,
armed: ArmedAttempt,
error_type: &'static str,
error_message: &'static str,
) {
let ArmedAttempt {
request_id,
candidate_id: _,
candidate,
usage_seed,
request_diagnostics,
candidate_started_unix_ms,
candidate_started_at,
} = armed;
let terminal_unix_ms = current_request_candidate_unix_ms();
let latency_ms = elapsed_ms_since(candidate_started_at);
if let Some(candidate) = candidate.as_ref() {
record_local_request_candidate_status_snapshot(
&state,
candidate,
SchedulerRequestCandidateStatusUpdate {
status: RequestCandidateStatus::Cancelled,
status_code: Some(CLIENT_CANCELLED_STATUS_CODE),
error_type: Some(error_type.to_string()),
error_message: Some(error_message.to_string()),
latency_ms: Some(latency_ms),
started_at_unix_ms: Some(candidate_started_unix_ms),
finished_at_unix_ms: Some(terminal_unix_ms),
},
)
.await;
}
let Some(usage_data) = usage_seed else {
return;
};
let mut usage_data = *usage_data;
// The seed was built when the attempt was armed, so it predates the
// diagnostics it should carry. Attaching them to the seed's metadata is the
// same write the report context would have carried into a seed built here:
// both land the same keys in the same object.
usage_data.request_metadata = attach_request_diagnostics_to_report_context(
usage_data.request_metadata.take(),
request_diagnostics.as_ref(),
);
usage_data.status_code = Some(CLIENT_CANCELLED_STATUS_CODE);
usage_data.error_message = Some(error_message.to_string());
usage_data.error_category = Some("cancelled".to_string());
usage_data.response_time_ms = Some(latency_ms);
let error_body = json!({
"error": {
"type": error_type,
"message": error_message,
"code": CLIENT_CANCELLED_STATUS_CODE
}
});
usage_data.response_headers = Some(json!({"content-type": "application/json"}));
usage_data.response_body = Some(error_body.clone());
usage_data.client_response_headers = Some(json!({"content-type": "application/json"}));
usage_data.client_response_body = Some(error_body);
state
.usage_runtime
.record_terminal_event_direct(
state.usage_lifecycle_data_state().as_ref(),
UsageEvent::new(UsageEventType::Cancelled, request_id, usage_data),
)
.await;
}
impl Drop for AttemptCancellationGuard {
fn drop(&mut self) {
let Some(armed) = self.armed.take() else {
return;
};
if self
.watchdog
.as_ref()
.is_some_and(|watchdog| watchdog.abandoned())
{
return;
}
let state = self.state.clone();
let error_type = self.error_type;
let error_message = self.error_message;
// `Drop` cannot await, and the settlement writes touch the database.
// Hand them to the runtime so they survive the dropped request future.
let Ok(handle) = tokio::runtime::Handle::try_current() else {
warn!(
event_name = "local_attempt_cancellation_guard_no_runtime",
log_type = "ops",
request_id = %short_request_id(armed.request_id.as_str()),
candidate_id = ?armed.candidate_id,
error_type,
"gateway could not settle dropped local attempt because no Tokio runtime is available"
);
return;
};
handle.spawn(async move {
settle_cancelled_attempt(state, armed, error_type, error_message).await;
});
}
}
#[cfg(test)]
mod tests {
use super::*;
use aether_contracts::RequestBody;
use aether_data::repository::candidates::InMemoryRequestCandidateRepository;
use aether_data::repository::usage::InMemoryUsageReadRepository;
use aether_data_contracts::repository::candidates::RequestCandidateReadRepository;
use aether_data_contracts::repository::usage::{
StoredRequestUsageAudit, UsageBodyCaptureState, UsageReadRepository, UsageWriteRepository,
};
use aether_usage_runtime::{
build_lifecycle_usage_seed, build_pending_usage_record, UsageRuntimeConfig,
};
use std::collections::BTreeMap;
use std::time::Duration;
use crate::request_candidate_runtime::{
ensure_execution_request_candidate_slot, snapshot_local_request_candidate_status,
};
const TEST_ERROR_TYPE: &str = "local_stream_attempt_cancelled";
const TEST_ERROR_MESSAGE: &str =
"Local stream attempt was dropped before terminal finalization.";
fn test_stream_plan(request_id: &str) -> ExecutionPlan {
ExecutionPlan {
request_id: request_id.to_string(),
candidate_id: None,
provider_name: Some("Anthropic".to_string()),
provider_id: "provider-1".to_string(),
endpoint_id: "endpoint-1".to_string(),
key_id: "key-1".to_string(),
method: "POST".to_string(),
url: "https://example.test/v1/messages".to_string(),
headers: BTreeMap::new(),
content_type: Some("application/json".to_string()),
content_encoding: None,
body: RequestBody::from_json(json!({"stream": true, "service_tier": "priority"})),
stream: true,
client_api_format: "claude:messages".to_string(),
provider_api_format: "claude:messages".to_string(),
model_name: Some("claude-sonnet-4-5".to_string()),
proxy: None,
transport_profile: None,
timeouts: None,
}
}
fn test_report_context() -> Option<Value> {
Some(json!({
"candidate_index": 0,
"retry_index": 0,
"user_id": "user-cancel",
"api_key_id": "api-key-cancel",
"client_api_format": "claude:messages",
"provider_api_format": "claude:messages",
"request_path": "/v1/messages",
"request_path_and_query": "/v1/messages?beta=true",
"upstream_url": "https://example.test/v1/messages",
"mapped_model": "claude-sonnet-4-5",
"original_request_body": {"stream": true, "messages": []},
}))
}
fn test_state(
usage_repository: &Arc<InMemoryUsageReadRepository>,
request_candidate_repository: &Arc<InMemoryRequestCandidateRepository>,
) -> AppState {
AppState::new()
.expect("gateway state should build")
.with_data_state_for_tests(
crate::data::GatewayDataState::with_request_candidate_and_usage_repository_for_tests(
Arc::clone(request_candidate_repository),
Arc::clone(usage_repository),
),
)
.with_usage_runtime_for_tests(UsageRuntimeConfig {
enabled: true,
..UsageRuntimeConfig::default()
})
}
/// Writes the `pending` rows the same way a stream attempt does before it
/// dispatches to the provider, and returns the candidate slot snapshot the
/// attempt owns from that point on.
async fn record_pending_attempt(
state: &AppState,
plan: &mut ExecutionPlan,
report_context: &mut Option<Value>,
candidate_started_unix_ms: u64,
) -> LocalRequestCandidateStatusSnapshot {
ensure_execution_request_candidate_slot(state, plan, report_context).await;
state.usage_runtime.record_pending(
state.usage_lifecycle_data_state().as_ref(),
build_lifecycle_usage_seed(plan, report_context.as_ref()),
);
let snapshot = snapshot_local_request_candidate_status(plan, report_context.as_ref())
.expect("attempt should own a candidate slot");
record_local_request_candidate_status_snapshot(
state,
&snapshot,
SchedulerRequestCandidateStatusUpdate {
status: RequestCandidateStatus::Pending,
status_code: None,
error_type: None,
error_message: None,
latency_ms: None,
started_at_unix_ms: Some(candidate_started_unix_ms),
finished_at_unix_ms: None,
},
)
.await;
snapshot
}
async fn wait_for_usage_status(
usage_repository: &InMemoryUsageReadRepository,
request_id: &str,
status: &str,
) -> Option<StoredRequestUsageAudit> {
for _ in 0..50 {
if let Some(usage) = usage_repository
.find_by_request_id(request_id)
.await
.expect("usage should read")
{
if usage.status == status {
return Some(usage);
}
}
tokio::time::sleep(Duration::from_millis(10)).await;
}
None
}
#[tokio::test]
async fn armed_guard_settles_a_dropped_attempt_as_cancelled() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let state = test_state(&usage_repository, &request_candidate_repository);
let mut plan = test_stream_plan("stream-cancel-guard-request");
let mut report_context = test_report_context();
let candidate_started_unix_ms = current_request_candidate_unix_ms();
let snapshot = record_pending_attempt(
&state,
&mut plan,
&mut report_context,
candidate_started_unix_ms,
)
.await;
{
let mut guard =
AttemptCancellationGuard::disarmed(&state, TEST_ERROR_TYPE, TEST_ERROR_MESSAGE);
guard.arm(
&plan,
report_context.as_ref(),
Some(&snapshot),
candidate_started_unix_ms,
Instant::now(),
);
}
let usage = wait_for_usage_status(
usage_repository.as_ref(),
"stream-cancel-guard-request",
"cancelled",
)
.await
.expect("cancelled usage should be recorded");
assert_eq!(usage.billing_status, "void");
assert_eq!(usage.status_code, Some(CLIENT_CANCELLED_STATUS_CODE));
assert_eq!(usage.error_category.as_deref(), Some("cancelled"));
assert!(usage.response_time_ms.is_some());
let candidates = request_candidate_repository
.list_by_request_id("stream-cancel-guard-request")
.await
.expect("candidates should read");
let candidate = candidates.first().expect("candidate row should exist");
assert_eq!(candidate.status, RequestCandidateStatus::Cancelled);
assert_eq!(candidate.status_code, Some(CLIENT_CANCELLED_STATUS_CODE));
assert_eq!(candidate.error_type.as_deref(), Some(TEST_ERROR_TYPE));
assert!(candidate.finished_at_unix_ms.is_some());
}
/// The guard holds no request body, so its settlement write must describe the
/// capture rather than deny it: a typed `none` capture state would clear the
/// stored request body instead of leaving it alone.
#[tokio::test]
async fn settling_a_dropped_attempt_leaves_the_captured_request_body_alone() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let state = test_state(&usage_repository, &request_candidate_repository);
let mut plan = test_stream_plan("stream-cancel-guard-capture");
let mut report_context = test_report_context();
let candidate_started_unix_ms = current_request_candidate_unix_ms();
let snapshot = record_pending_attempt(
&state,
&mut plan,
&mut report_context,
candidate_started_unix_ms,
)
.await;
// Stand in for a write that already captured this request's body.
let captured_body = json!({"stream": true, "service_tier": "priority"});
let mut capture = build_pending_usage_record(
&plan,
report_context.as_ref(),
current_request_candidate_unix_ms() / 1_000,
)
.expect("pending usage record should build");
capture.provider_request_body = Some(captured_body.clone());
capture.provider_request_body_state = Some(UsageBodyCaptureState::Inline);
usage_repository
.upsert(capture)
.await
.expect("captured request body should upsert");
{
let mut guard =
AttemptCancellationGuard::disarmed(&state, TEST_ERROR_TYPE, TEST_ERROR_MESSAGE);
guard.arm(
&plan,
report_context.as_ref(),
Some(&snapshot),
candidate_started_unix_ms,
Instant::now(),
);
}
let usage = wait_for_usage_status(
usage_repository.as_ref(),
"stream-cancel-guard-capture",
"cancelled",
)
.await
.expect("cancelled usage should be recorded");
assert_eq!(usage.provider_request_body, Some(captured_body));
assert_ne!(
usage.provider_request_body_state,
Some(UsageBodyCaptureState::None)
);
}
#[tokio::test]
async fn guard_stands_down_when_the_watchdog_abandons_the_attempt() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let state = test_state(&usage_repository, &request_candidate_repository);
let mut plan = test_stream_plan("stream-watchdog-guard-request");
let mut report_context = test_report_context();
let candidate_started_unix_ms = current_request_candidate_unix_ms();
let snapshot = record_pending_attempt(
&state,
&mut plan,
&mut report_context,
candidate_started_unix_ms,
)
.await;
let watchdog = StreamCandidateWatchdogProgress::shared();
Arc::clone(&watchdog)
.scope(async {
let mut guard =
AttemptCancellationGuard::disarmed(&state, TEST_ERROR_TYPE, TEST_ERROR_MESSAGE);
guard.arm(
&plan,
report_context.as_ref(),
Some(&snapshot),
candidate_started_unix_ms,
Instant::now(),
);
// The watchdog gives up and takes over settlement before the
// abandoned attempt is dropped.
watchdog.mark_abandoned();
})
.await;
assert!(wait_for_usage_status(
usage_repository.as_ref(),
"stream-watchdog-guard-request",
"cancelled",
)
.await
.is_none());
}
#[tokio::test]
async fn disarmed_guard_leaves_the_attempt_pending() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let state = test_state(&usage_repository, &request_candidate_repository);
let mut plan = test_stream_plan("stream-disarmed-guard-request");
let mut report_context = test_report_context();
let candidate_started_unix_ms = current_request_candidate_unix_ms();
let snapshot = record_pending_attempt(
&state,
&mut plan,
&mut report_context,
candidate_started_unix_ms,
)
.await;
{
let mut guard =
AttemptCancellationGuard::disarmed(&state, TEST_ERROR_TYPE, TEST_ERROR_MESSAGE);
guard.arm(
&plan,
report_context.as_ref(),
Some(&snapshot),
candidate_started_unix_ms,
Instant::now(),
);
guard.disarm();
}
assert!(wait_for_usage_status(
usage_repository.as_ref(),
"stream-disarmed-guard-request",
"cancelled",
)
.await
.is_none());
}
}
@@ -156,7 +156,22 @@ pub(crate) fn should_fallback_to_control_sync(
return true;
};
body_json.get("error").is_some()
sync_body_has_embedded_error(Some(body_json))
}
/// Mirrors the error-like body markers used by the formats layer. Successful OpenAI Responses
/// bodies contain `"error": null`, which must not route them through error finalization.
fn sync_body_has_embedded_error(body_json: Option<&serde_json::Value>) -> bool {
let Some(object) = body_json.and_then(serde_json::Value::as_object) else {
return false;
};
object.get("error").is_some_and(|error| !error.is_null())
|| object.get("status").and_then(serde_json::Value::as_str) == Some("failed")
|| object
.get("type")
.and_then(serde_json::Value::as_str)
.is_some_and(|value| value == "error")
}
pub(crate) fn should_finalize_sync_response(report_kind: Option<&str>) -> bool {
@@ -168,7 +183,7 @@ pub(crate) fn resolve_core_sync_error_finalize_report_kind(
result: &ExecutionResult,
body_json: Option<&serde_json::Value>,
) -> Option<String> {
let has_embedded_error = body_json.is_some_and(|value| value.get("error").is_some());
let has_embedded_error = sync_body_has_embedded_error(body_json);
if result.status_code < 400 && !has_embedded_error {
return None;
}
@@ -510,6 +525,74 @@ mod tests {
);
}
#[test]
fn successful_responses_body_with_null_error_stays_on_success_path() {
let result = ExecutionResult {
request_id: "req-1".to_string(),
candidate_id: None,
status_code: 200,
headers: Default::default(),
response_observation: None,
body: None,
telemetry: None,
error: None,
};
let body_json = serde_json::json!({
"id": "resp_1",
"object": "response",
"status": "completed",
"error": null,
"output": [],
});
assert_eq!(
resolve_core_sync_error_finalize_report_kind(
"openai_responses_sync",
&result,
Some(&body_json)
),
None
);
assert!(!should_fallback_to_control_sync(
"openai_responses_sync",
&result,
Some(&body_json),
true,
false,
false,
));
}
#[test]
fn error_like_success_status_bodies_still_map_to_error_finalize() {
let result = ExecutionResult {
request_id: "req-1".to_string(),
candidate_id: None,
status_code: 200,
headers: Default::default(),
response_observation: None,
body: None,
telemetry: None,
error: None,
};
for body_json in [
serde_json::json!({"status": "failed", "error": null}),
serde_json::json!({"type": "error"}),
serde_json::json!({"error": {"message": "boom"}}),
] {
assert_eq!(
resolve_core_sync_error_finalize_report_kind(
"openai_responses_sync",
&result,
Some(&body_json)
),
Some("openai_responses_sync_finalize".to_string()),
"error-like body must not escape through the success path: {body_json}"
);
}
}
#[test]
fn stream_failover_marks_chat_errors() {
assert!(should_fallback_to_control_stream(
@@ -4,6 +4,7 @@ use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
pub(crate) mod admission;
pub(crate) mod attempt_cancellation;
pub(crate) mod attempt_lifecycle;
mod chatgpt_web_image;
mod constants;
@@ -455,7 +455,10 @@ fn gemini_part_is_client_semantic(part: &Value) -> bool {
return true;
}
if part.get("thought").and_then(Value::as_bool) == Some(true) {
return false;
return part
.get("text")
.and_then(Value::as_str)
.is_some_and(|text| !text.is_empty());
}
if part.keys().all(|key| key == "thoughtSignature") {
return false;
@@ -584,17 +587,12 @@ mod tests {
}
#[test]
fn gemini_gate_waits_through_thought_and_commits_on_text() {
fn gemini_gate_commits_on_first_nonempty_thought() {
let mut gate = StreamCommitGate::new(gemini_policy());
let thought = b"data: {\"response\":{\"candidates\":[{\"content\":{\"role\":\"model\",\"parts\":[{\"thought\":true,\"text\":\"checking\"}]}}]}}\n\n";
let text = b"data: {\"response\":{\"candidates\":[{\"content\":{\"role\":\"model\",\"parts\":[{\"text\":\"answer\"}]}}]}}\n\n";
assert_eq!(
gate.observe_provider_bytes(thought),
StreamPrecommitObservation::Pending
);
assert_eq!(
gate.observe_provider_bytes(text),
StreamPrecommitObservation::Commit
);
assert_eq!(gate.state(), StreamCommitState::Committed);
@@ -615,7 +613,7 @@ mod tests {
#[test]
fn gemini_gate_rejects_malformed_function_call_before_commit() {
let mut gate = StreamCommitGate::new(gemini_policy());
let thought = b"data: {\"response\":{\"candidates\":[{\"content\":{\"role\":\"model\",\"parts\":[{\"thought\":true,\"text\":\"calling\"}]}}]}}\n\n";
let thought = b"data: {\"response\":{\"candidates\":[{\"content\":{\"role\":\"model\",\"parts\":[{\"thoughtSignature\":\"signature\",\"text\":\"\"}]}}]}}\n\n";
let malformed = b"data: {\"response\":{\"candidates\":[{\"content\":{\"role\":\"model\",\"parts\":[{\"thoughtSignature\":\"signature\",\"text\":\"\"}]},\"finishReason\":\"MALFORMED_FUNCTION_CALL\",\"finishMessage\":\"Malformed function call: Function call is empty - no input to parse.\"}]}}\n\n";
assert_eq!(
@@ -79,6 +79,7 @@ use crate::api::response::{
use crate::clock::current_unix_ms as current_request_candidate_unix_ms;
use crate::constants::{CONTROL_CANDIDATE_ID_HEADER, CONTROL_REQUEST_ID_HEADER};
use crate::control::GatewayControlDecision;
use crate::execution_runtime::attempt_cancellation::AttemptCancellationGuard;
use crate::execution_runtime::build_direct_execution_frame_stream;
use crate::execution_runtime::chatgpt_web_image::maybe_execute_chatgpt_web_image_stream;
use crate::execution_runtime::grok::maybe_execute_grok_stream;
@@ -151,6 +152,11 @@ use crate::{
AppState, GatewayError, GEMINI_FILES_DOWNLOAD_PLAN_KIND, OPENAI_VIDEO_CONTENT_PLAN_KIND,
};
/// Settlement labels for a stream attempt whose future is dropped before the
/// transport reaches a terminal state.
const STREAM_ATTEMPT_CANCELLED_ERROR_TYPE: &str = "local_stream_attempt_cancelled";
const STREAM_ATTEMPT_CANCELLED_ERROR_MESSAGE: &str = "Local stream attempt was dropped before terminal finalization, usually because the client disconnected or the request task was cancelled.";
const SSE_KEEPALIVE_INTERVAL: Duration = Duration::from_secs(15);
const SSE_KEEPALIVE_BYTES: &[u8] = b": aether-keepalive\n\n";
const SSE_CONTROL_FILTER_MAX_BUFFER_BYTES: usize = 1024 * 1024;
@@ -3718,17 +3724,30 @@ pub(crate) fn execute_execution_runtime_stream<'a>(
report_kind: Option<String>,
report_context: Option<serde_json::Value>,
) -> Pin<Box<dyn Future<Output = Result<Option<Response<Body>>, GatewayError>> + Send + 'a>> {
Box::pin(execute_execution_runtime_stream_inner(
state,
plan,
trace_id,
decision,
plan_kind,
report_kind,
report_context,
None,
None,
))
Box::pin(async move {
let mut cancellation_guard = AttemptCancellationGuard::disarmed(
state,
STREAM_ATTEMPT_CANCELLED_ERROR_TYPE,
STREAM_ATTEMPT_CANCELLED_ERROR_MESSAGE,
);
let result = execute_execution_runtime_stream_inner(
state,
plan,
trace_id,
decision,
plan_kind,
report_kind,
report_context,
None,
None,
&mut cancellation_guard,
)
.await;
// The attempt reached its own terminal path, or handed settlement to the
// stream finalizer that now lives in the response body.
cancellation_guard.disarm();
result
})
}
#[allow(clippy::too_many_arguments)]
@@ -3750,7 +3769,12 @@ pub(crate) fn execute_execution_runtime_stream_with_retry_scope<'a>(
Box::pin(async move {
let mut retry_scope = AiAttemptRetryScope::Candidate;
let mut fallback_response = None;
let response = execute_execution_runtime_stream_inner(
let mut cancellation_guard = AttemptCancellationGuard::disarmed(
state,
STREAM_ATTEMPT_CANCELLED_ERROR_TYPE,
STREAM_ATTEMPT_CANCELLED_ERROR_MESSAGE,
);
let result = execute_execution_runtime_stream_inner(
state,
plan,
trace_id,
@@ -3760,8 +3784,13 @@ pub(crate) fn execute_execution_runtime_stream_with_retry_scope<'a>(
report_context,
Some(&mut retry_scope),
Some(&mut fallback_response),
&mut cancellation_guard,
)
.await?;
.await;
// The attempt reached its own terminal path, or handed settlement to the
// stream finalizer that now lives in the response body.
cancellation_guard.disarm();
let response = result?;
Ok(match response {
Some(response) => AiAttemptExecutionOutcome::Responded(response),
None => AiAttemptExecutionOutcome::Retry {
@@ -3807,6 +3836,7 @@ async fn maybe_build_stream_transport_error_stop_response(
.map(Some)
}
#[allow(clippy::too_many_arguments)] // internal function, grouping would add unnecessary indirection
async fn execute_execution_runtime_stream_inner(
state: &AppState,
mut plan: ExecutionPlan,
@@ -3817,6 +3847,7 @@ async fn execute_execution_runtime_stream_inner(
mut report_context: Option<serde_json::Value>,
mut retry_scope_out: Option<&mut AiAttemptRetryScope>,
mut retry_fallback_out: Option<&mut Option<Response<Body>>>,
cancellation_guard: &mut AttemptCancellationGuard,
) -> Result<Option<Response<Body>>, GatewayError> {
let stream_started_at = Instant::now();
let mut stage_trace = RequestStageTrace::from_env();
@@ -3900,6 +3931,16 @@ async fn execute_execution_runtime_stream_inner(
)
.await;
}
// From here the attempt owns non-terminal rows, and everything that could
// settle them runs inside the downstream request future. Arm the guard so a
// client disconnect before the stream finalizer exists still settles them.
cancellation_guard.arm(
&plan,
report_context.as_ref(),
request_candidate_status_snapshot.as_ref(),
candidate_started_unix_secs,
stream_started_at,
);
let plan_request_id_for_log = short_request_id(plan.request_id.as_str());
let provider_name = plan
.provider_name
@@ -10764,7 +10805,7 @@ mod tests {
}
#[tokio::test]
async fn malformed_antigravity_function_call_retries_before_stream_commit() {
async fn malformed_antigravity_function_call_streams_thought_then_fails_in_band() {
let request_id = "req-antigravity-malformed-function-call";
let plan = antigravity_gemini_stream_plan(request_id);
let provider_catalog = provider_catalog_for_plan(
@@ -10843,10 +10884,35 @@ mod tests {
None,
)
.await
.expect("malformed Antigravity stream should resolve through failover");
.expect("malformed Antigravity stream should return a client stream")
.expect("the first reasoning delta should commit the selected candidate");
assert!(response.is_none());
assert_eq!(retry_scope, AiAttemptRetryScope::Candidate);
assert_eq!(response.status(), StatusCode::OK);
let body = to_bytes(response.into_body(), usize::MAX)
.await
.expect("response body should read");
let body = String::from_utf8(body.to_vec()).expect("response body should be utf8");
assert!(
body.contains("event: response.reasoning_summary_text.delta\n"),
"{body}"
);
assert!(
body.contains("\"delta\":\"Validating the document.\""),
"{body}"
);
assert!(body.contains("event: response.failed\n"), "{body}");
assert!(
body.contains("\"code\":\"MALFORMED_FUNCTION_CALL\""),
"{body}"
);
assert!(
body.contains(
"\"message\":\"Malformed function call: Function call is empty - no input to parse.\""
),
"{body}"
);
assert!(!body.contains("unsupported_finish_reason"), "{body}");
assert_eq!(retry_scope, AiAttemptRetryScope::Provider);
}
fn tunnel_proxy_snapshot(base_url: String) -> aether_contracts::ProxySnapshot {
@@ -22,6 +22,7 @@ const TRANSPORT_ERROR_CLIENT_MESSAGE: &str =
#[derive(Debug, Default)]
pub(crate) struct StreamCandidateWatchdogProgress {
terminal_started: AtomicBool,
abandoned: AtomicBool,
}
tokio::task_local! {
@@ -37,6 +38,24 @@ impl StreamCandidateWatchdogProgress {
self.terminal_started.load(Ordering::Acquire)
}
/// The watchdog gave up waiting and settles this attempt itself.
///
/// The attempt future is dropped once the watchdog returns, so its own
/// cancellation guard must stay out of the way instead of racing the
/// watchdog's terminal rows with a cancellation.
pub(crate) fn mark_abandoned(&self) {
self.abandoned.store(true, Ordering::Release);
}
pub(crate) fn abandoned(&self) -> bool {
self.abandoned.load(Ordering::Acquire)
}
/// The watchdog watching the attempt on this task, if it runs under one.
pub(crate) fn current() -> Option<Arc<Self>> {
STREAM_CANDIDATE_WATCHDOG_PROGRESS.try_with(Arc::clone).ok()
}
pub(crate) async fn scope<F>(self: Arc<Self>, future: F) -> F::Output
where
F: Future,
@@ -522,6 +522,7 @@ where
decision,
plan_kind,
transfer_tracker,
request_first_byte_started_at: Instant::now(),
};
let loop_result = run_ai_attempt_loop(&port, plan_and_reports).await;
if loop_result.is_err() {
@@ -602,6 +603,7 @@ where
decision,
plan_kind,
transfer_tracker,
request_first_byte_started_at: Instant::now(),
};
let loop_result = run_dynamic_attempt_loop(
&port,
@@ -1119,6 +1121,10 @@ struct StreamAttemptLoopPort<'a> {
decision: &'a GatewayControlDecision,
plan_kind: &'a str,
transfer_tracker: &'a ProviderTransferTracker,
/// All candidates in one downstream stream request share this origin.
/// Without it every retry receives a fresh full first-byte timeout and a
/// 30-second provider timeout can accumulate into a 60-120 second stall.
request_first_byte_started_at: Instant,
}
#[async_trait]
@@ -1248,6 +1254,7 @@ where
self.plan_kind,
plan,
watchdog_report_context,
self.request_first_byte_started_at,
stop_on_transport_errors,
move || async move {
if let Some(response) = execution_plan_cost_capacity_response(
@@ -1301,7 +1308,7 @@ where
http::StatusCode::GATEWAY_TIMEOUT.as_u16(),
"local_stream_candidate_watchdog_timeout",
stream_candidate_watchdog_timeout_message(),
watchdog_started_at.elapsed().as_millis() as u64,
self.request_first_byte_started_at.elapsed().as_millis() as u64,
)
.await?,
)
@@ -1751,6 +1758,7 @@ async fn execute_stream_candidate_with_watchdog<Fut>(
plan_kind: &str,
plan: &aether_contracts::ExecutionPlan,
report_context: Option<&serde_json::Value>,
request_first_byte_started_at: Instant,
stop_on_transport_errors: bool,
execute: impl FnOnce() -> Fut,
) -> Result<StreamCandidateWatchdogOutcome, GatewayError>
@@ -1760,6 +1768,7 @@ where
> + Send,
{
let timeout_duration = resolve_stream_candidate_watchdog_timeout(plan, report_context);
let request_first_byte_deadline = request_first_byte_started_at + timeout_duration;
let candidate_started_at = std::time::Instant::now();
let candidate_started_unix_ms = current_unix_ms();
let permit = match acquire_upstream_execution_gate(state, trace_id).await {
@@ -1785,7 +1794,14 @@ where
let watchdog_progress = StreamCandidateWatchdogProgress::shared();
let execution = watchdog_progress.clone().scope(execute());
tokio::pin!(execution);
let deadline = tokio::time::sleep(timeout_duration);
// This is an absolute request-level deadline, not a new timeout for this
// candidate. Retries therefore consume only the budget left by earlier
// candidates instead of resetting the full provider timeout.
let candidate_budget_ms = request_first_byte_deadline
.saturating_duration_since(Instant::now())
.as_millis()
.min(u128::from(u64::MAX)) as u64;
let deadline = tokio::time::sleep_until(request_first_byte_deadline);
tokio::pin!(deadline);
let execution_result = tokio::select! {
biased;
@@ -1801,6 +1817,10 @@ where
let outcome = match execution_result {
Some(result) => result.map(StreamCandidateWatchdogOutcome::Executed),
None => {
// The abandoned attempt is dropped when this function returns.
// Claim its settlement before that so its cancellation guard does
// not race the watchdog rows written just below.
watchdog_progress.mark_abandoned();
let finished_at_unix_ms = current_unix_ms();
let request_id = short_request_id(plan.request_id.as_str());
let provider_name = plan.provider_name.as_deref().unwrap_or("-");
@@ -1810,6 +1830,10 @@ where
.map(|value| value.to_string())
.unwrap_or_else(|| "-".to_string());
let timeout_ms = u64::try_from(timeout_duration.as_millis()).unwrap_or(u64::MAX);
let request_elapsed_ms = request_first_byte_started_at
.elapsed()
.as_millis()
.min(u128::from(u64::MAX)) as u64;
record_local_request_candidate_status(
state,
plan,
@@ -1838,6 +1862,8 @@ where
model_name,
candidate_index = candidate_index.as_str(),
timeout_ms,
candidate_budget_ms,
request_elapsed_ms,
"gateway local stream candidate watchdog timed out"
);
if stop_on_transport_errors {
@@ -3124,6 +3150,7 @@ mod tests {
"claude_cli_stream",
&plan,
Some(&report_context),
Instant::now(),
false,
|| {
std::future::pending::<
@@ -3162,6 +3189,56 @@ mod tests {
assert_eq!(record.candidate_index, 2);
}
#[tokio::test]
async fn stream_candidate_retry_does_not_reset_an_expired_request_first_byte_budget() {
let writer = Arc::new(TestRequestCandidateWriter::default());
let plan = test_plan(Some(ExecutionTimeouts {
first_byte_ms: Some(250),
..ExecutionTimeouts::default()
}));
let report_context = test_report_context();
// Stand in for earlier candidates having already consumed the request's
// complete first-byte budget. A per-candidate watchdog would wait a new
// 250 ms here; the shared absolute deadline must settle immediately.
let request_first_byte_started_at = Instant::now() - Duration::from_millis(300);
let result = tokio::time::timeout(
Duration::from_millis(100),
execute_stream_candidate_with_watchdog(
writer.as_ref(),
"trace_watchdog_shared_budget",
"claude_cli_stream",
&plan,
Some(&report_context),
request_first_byte_started_at,
false,
|| {
std::future::pending::<
Result<AiAttemptExecutionOutcome<Response<Body>>, GatewayError>,
>()
},
),
)
.await
.expect("an expired request-level first-byte budget must not restart per candidate");
assert!(matches!(
result,
Ok(StreamCandidateWatchdogOutcome::Executed(
AiAttemptExecutionOutcome::Retry {
scope: AiAttemptRetryScope::Candidate,
fallback_response: None,
}
))
));
let records = writer.records.lock().await;
assert_eq!(records.len(), 1);
assert_eq!(
records[0].error_type.as_deref(),
Some("local_stream_candidate_watchdog_timeout")
);
}
#[tokio::test]
async fn stream_candidate_watchdog_can_stop_on_transport_error() {
let writer = Arc::new(TestRequestCandidateWriter::default());
@@ -3177,6 +3254,7 @@ mod tests {
"claude_cli_stream",
&plan,
Some(&report_context),
Instant::now(),
true,
|| {
std::future::pending::<
@@ -3214,6 +3292,7 @@ mod tests {
"claude_cli_stream",
&plan,
Some(&report_context),
Instant::now(),
true,
|| async {
mark_stream_candidate_watchdog_terminal_started();
@@ -3246,6 +3325,7 @@ mod tests {
"claude_cli_stream",
&plan,
Some(&report_context),
Instant::now(),
true,
|| async {
Err(GatewayError::UpstreamUnavailable {
@@ -3285,6 +3365,7 @@ mod tests {
"claude_cli_stream",
&plan,
Some(&report_context),
Instant::now(),
false,
|| async {
panic!("execute future should not run while upstream execution gate is saturated")
@@ -3332,6 +3413,7 @@ mod tests {
"claude_cli_stream",
&plan,
Some(&report_context),
Instant::now(),
false,
|| async {
Err(GatewayError::AdmissionTimeout {
@@ -5,6 +5,7 @@ use super::shared::{
quota_key_auto_removed, quota_refresh_success_invalid_state,
resolve_provider_quota_execution_timeouts, ProviderQuotaExecutionOutcome,
};
use crate::handlers::admin::provider::shared::payloads::AdminImportProviderModelsRequest;
use crate::handlers::admin::request::{AdminAppState, AdminGatewayProviderTransportSnapshot};
use crate::GatewayError;
use aether_admin::provider::quota::{
@@ -22,6 +23,63 @@ use std::collections::BTreeMap;
use std::time::{SystemTime, UNIX_EPOCH};
use tracing::warn;
fn antigravity_discovered_model_ids(metadata_update: Option<&serde_json::Value>) -> Vec<String> {
metadata_update
.and_then(|value| value.pointer("/antigravity/quota_by_model"))
.and_then(serde_json::Value::as_object)
.into_iter()
.flat_map(|models| models.keys())
.map(String::as_str)
.filter(|model_id| aether_model_fetch::antigravity_model_id_is_routable(model_id))
.map(ToOwned::to_owned)
.collect()
}
async fn sync_antigravity_discovered_models(
state: &AdminAppState<'_>,
provider_id: &str,
metadata_update: Option<&serde_json::Value>,
) {
if !state.has_global_model_data_reader() || !state.has_global_model_data_writer() {
return;
}
let model_ids = antigravity_discovered_model_ids(metadata_update);
if model_ids.is_empty() {
return;
}
let result = state
.build_admin_import_provider_models_payload(
provider_id,
AdminImportProviderModelsRequest {
model_ids,
tiered_pricing: None,
price_per_request: None,
},
)
.await;
match result {
Ok(payload) => {
let errors = payload
.get("errors")
.and_then(serde_json::Value::as_array)
.map(Vec::len)
.unwrap_or(0);
if errors > 0 {
warn!(
provider_id,
errors, "Antigravity discovered-model catalog sync completed with item errors"
);
}
}
Err(error) => warn!(
provider_id,
error = %error,
"Antigravity discovered-model catalog sync failed"
),
}
}
async fn execute_antigravity_quota_plan(
state: &AdminAppState<'_>,
transport: &AdminGatewayProviderTransportSnapshot,
@@ -321,6 +379,10 @@ pub(crate) async fn refresh_antigravity_provider_quota_locally(
continue;
}
if status == "success" {
sync_antigravity_discovered_models(state, &provider.id, metadata_update.as_ref()).await;
}
if status == "success" {
success_count += 1;
} else {
@@ -606,6 +606,50 @@ pub(crate) async fn reserve_codex_account_reset(
Ok(None)
}
fn record_locally_consumed_codex_reset_credit(
codex: &mut serde_json::Map<String, serde_json::Value>,
observed_at_unix_secs: u64,
) {
let Some(reset_credits) = codex
.get_mut("reset_credits")
.and_then(serde_json::Value::as_object_mut)
else {
return;
};
let Some(available_count) = reset_credits
.get("available_count")
.and_then(admin_provider_quota_pure::coerce_json_u64)
else {
return;
};
reset_credits.insert(
"available_count".to_string(),
serde_json::json!(available_count.saturating_sub(1)),
);
reset_credits.insert(
"updated_at".to_string(),
serde_json::json!(observed_at_unix_secs),
);
reset_credits.insert(
"detail_source".to_string(),
serde_json::json!("local_consume"),
);
reset_credits.insert(
"detail_status".to_string(),
serde_json::json!("pending_refresh"),
);
reset_credits.remove("detail_error");
if let Some(credits) = reset_credits
.get_mut("credits")
.and_then(serde_json::Value::as_array_mut)
{
if !credits.is_empty() {
credits.remove(0);
}
}
}
pub(crate) async fn complete_codex_account_reset(
state: &AdminAppState<'_>,
key_id: &str,
@@ -671,6 +715,9 @@ pub(crate) async fn complete_codex_account_reset(
generation: reservation.generation,
outcome: outcome.to_string(),
};
if outcome == "reset" {
record_locally_consumed_codex_reset_credit(&mut codex, fence_unix_ms / 1_000);
}
codex_reset_write_bounded_history(&mut codex, &terminal);
if codex_reset_reservation_from_object(&codex).as_ref() == Some(reservation) {
codex.remove(admin_provider_quota_pure::CODEX_QUOTA_ACCOUNT_RESET_RESERVATION_KEY);
@@ -1972,7 +2019,19 @@ mod tests {
.expect("key should build");
key.encrypted_auth_config = Some(encrypted_auth_config.clone());
key.upstream_metadata = Some(json!({
"codex": {"credential_generation": "credential-v1"}
"codex": {
"credential_generation": "credential-v1",
"reset_credits": {
"available_count": 2,
"updated_at": 100u64,
"detail_source": "wham_readonly",
"detail_status": "available",
"credits": [
{"id": "credit-1", "expires_at": 20_000u64},
{"id": "credit-2", "expires_at": 30_000u64}
]
}
}
}));
let credential_fence = ProviderTransportCredentialFence {
encrypted_auth_config,
@@ -2070,6 +2129,13 @@ mod tests {
let key_id = "key-codex-reset-credential-generation";
let (app, repository, credential) = codex_reset_state_machine_test_state(key_id);
let admin_state = AdminAppState::new(&app);
let original_metadata = repository
.list_keys_by_ids(&[key_id.to_string()])
.await
.expect("key should load before reservation")
.pop()
.expect("key should exist before reservation")
.upstream_metadata;
let result = reserve_codex_account_reset(
&admin_state,
@@ -2093,10 +2159,7 @@ mod tests {
.expect("key should reload")
.pop()
.expect("key should exist");
assert_eq!(
stored.upstream_metadata.unwrap()["codex"],
json!({"credential_generation":"credential-v1"})
);
assert_eq!(stored.upstream_metadata, original_metadata);
}
#[tokio::test]
@@ -2231,6 +2294,11 @@ mod tests {
codex["account_quota_reset_history"][0]["outcome"],
json!("reset")
);
assert_eq!(codex["reset_credits"]["available_count"], json!(1u64));
assert_eq!(
codex["reset_credits"]["credits"],
json!([{"id": "credit-2", "expires_at": 30_000u64}])
);
}
}
@@ -4,8 +4,9 @@ use super::super::errors::{
use crate::handlers::admin::request::{AdminAppState, AdminProviderOAuthTemplate};
use aether_contracts::ProxySnapshot;
use aether_oauth::provider::providers::{
ClaudeCodeProviderOAuthAdapter, GenericProviderOAuthAdapter, CLAUDE_CODE_PROVIDER_TYPE,
CLAUDE_CODE_TOKEN_URL, CLAUDE_CODE_WEB_BASE_URL,
AntigravityProviderOAuthAdapter, ClaudeCodeProviderOAuthAdapter, GenericProviderOAuthAdapter,
ANTIGRAVITY_USER_INFO_URL, CLAUDE_CODE_PROVIDER_TYPE, CLAUDE_CODE_TOKEN_URL,
CLAUDE_CODE_WEB_BASE_URL,
};
use aether_oauth::provider::{
ProviderOAuthCookieAuthorizationInput, ProviderOAuthService, ProviderOAuthTransportContext,
@@ -39,7 +40,14 @@ fn provider_oauth_exchange_context(
fn provider_oauth_service_for_template(
template: AdminProviderOAuthTemplate,
token_url: String,
antigravity_user_info_url: String,
) -> Result<ProviderOAuthService, Response<Body>> {
if template.provider_type.eq_ignore_ascii_case("antigravity") {
let adapter = AntigravityProviderOAuthAdapter::default()
.with_token_url_override(token_url)
.with_user_info_url_override(antigravity_user_info_url);
return Ok(ProviderOAuthService::new().with_adapter(Arc::new(adapter)));
}
GenericProviderOAuthAdapter::for_provider_type(template.provider_type)
.map(|adapter| adapter.with_token_url_override(token_url))
.map(|adapter| ProviderOAuthService::new().with_adapter(Arc::new(adapter)))
@@ -71,7 +79,10 @@ pub(crate) async fn exchange_admin_provider_oauth_code(
proxy: Option<ProxySnapshot>,
) -> Result<serde_json::Value, Response<Body>> {
let token_url = state.provider_oauth_token_url(template.provider_type, template.token_url);
let service = provider_oauth_service_for_template(template, token_url)?;
let antigravity_user_info_url =
state.provider_oauth_token_url("antigravity_user_info", ANTIGRAVITY_USER_INFO_URL);
let service =
provider_oauth_service_for_template(template, token_url, antigravity_user_info_url)?;
let ctx = provider_oauth_exchange_context(template.provider_type, proxy);
let executor = crate::oauth::GatewayOAuthHttpExecutor::new(*state);
let result = service
@@ -99,7 +110,10 @@ pub(crate) async fn exchange_admin_provider_oauth_refresh_token(
proxy: Option<ProxySnapshot>,
) -> Result<serde_json::Value, Response<Body>> {
let token_url = state.provider_oauth_token_url(template.provider_type, template.token_url);
let service = provider_oauth_service_for_template(template, token_url)?;
let antigravity_user_info_url =
state.provider_oauth_token_url("antigravity_user_info", ANTIGRAVITY_USER_INFO_URL);
let service =
provider_oauth_service_for_template(template, token_url, antigravity_user_info_url)?;
let ctx = provider_oauth_exchange_context(template.provider_type, proxy);
let executor = crate::oauth::GatewayOAuthHttpExecutor::new(*state);
let input = aether_oauth::provider::ProviderOAuthImportInput {
@@ -14,6 +14,9 @@ use aether_data::repository::provider_catalog::InMemoryProviderCatalogReadReposi
use aether_data_contracts::repository::candidate_selection::{
StoredMinimalCandidateSelectionRow, StoredProviderModelMapping,
};
use aether_data_contracts::repository::candidates::{
RequestCandidateReadRepository, RequestCandidateStatus,
};
use aether_data_contracts::repository::provider_catalog::{
StoredProviderCatalogEndpoint, StoredProviderCatalogKey, StoredProviderCatalogProvider,
};
@@ -427,6 +430,112 @@ async fn gateway_stops_execution_runtime_stream_when_client_disconnects_impl() {
upstream_handle.abort();
}
#[test]
fn gateway_settles_stream_attempt_when_client_disconnects_before_first_byte() {
run_lifecycle_test(
"gateway_settles_stream_attempt_when_client_disconnects_before_first_byte",
gateway_settles_stream_attempt_when_client_disconnects_before_first_byte_impl,
);
}
async fn gateway_settles_stream_attempt_when_client_disconnects_before_first_byte_impl() {
// The execution runtime accepts the plan and then goes quiet, so the attempt
// is parked between its `pending` rows and the first upstream byte.
let execution_runtime = Router::new().route(
"/v1/execute/stream",
any(|_request: Request| async move {
tokio::time::sleep(std::time::Duration::from_secs(30)).await;
StatusCode::OK
}),
);
let (execution_runtime_url, execution_runtime_handle) = start_server(execution_runtime).await;
let auth_repository = Arc::new(InMemoryAuthApiKeySnapshotRepository::seed(vec![(
Some(hash_api_key("sk-client-openai-stream-precommit-disconnect")),
sample_local_openai_auth_snapshot(
"api-key-openai-lifecycle-local-1",
"user-openai-lifecycle-local-1",
),
)]));
let candidate_selection_repository =
Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(vec![
sample_local_openai_candidate_row(),
]));
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![sample_local_openai_provider()],
vec![sample_local_openai_endpoint()],
vec![sample_local_openai_key()],
));
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let gateway = build_router_with_state(
build_state_with_execution_runtime_override(execution_runtime_url)
.with_data_state_for_tests(
GatewayDataState::with_auth_candidate_selection_provider_catalog_and_request_candidate_repository_for_tests(
auth_repository,
candidate_selection_repository,
provider_catalog_repository,
Arc::clone(&request_candidate_repository),
DEVELOPMENT_ENCRYPTION_KEY,
),
),
);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let request = reqwest::Client::new()
.post(format!("{gateway_url}/v1/chat/completions"))
.header(http::header::CONTENT_TYPE, "application/json")
.header(
http::header::AUTHORIZATION,
"Bearer sk-client-openai-stream-precommit-disconnect",
)
.header(
TRACE_ID_HEADER,
"trace-openai-chat-stream-precommit-disconnect-123",
)
.body("{\"model\":\"gpt-5\",\"messages\":[],\"stream\":true}")
.send();
// Drop the in-flight request the way a downstream client does when its own
// first-byte timeout fires, before any response header exists.
assert!(
tokio::time::timeout(std::time::Duration::from_millis(750), request)
.await
.is_err(),
"the execution runtime should not have answered before the client gave up"
);
let mut stored_candidates = Vec::new();
for _ in 0..200 {
stored_candidates = request_candidate_repository
.list_by_request_id("trace-openai-chat-stream-precommit-disconnect-123")
.await
.expect("request candidate trace should read");
if stored_candidates
.iter()
.any(|candidate| candidate.status == RequestCandidateStatus::Cancelled)
{
break;
}
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
}
let cancelled = stored_candidates
.iter()
.find(|candidate| candidate.status == RequestCandidateStatus::Cancelled)
.unwrap_or_else(|| {
panic!("dropped stream attempt should settle as cancelled: {stored_candidates:?}")
});
assert_eq!(cancelled.status_code, Some(499));
assert_eq!(
cancelled.error_type.as_deref(),
Some("local_stream_attempt_cancelled")
);
assert!(cancelled.finished_at_unix_ms.is_some());
gateway_handle.abort();
execution_runtime_handle.abort();
}
#[test]
fn gateway_returns_error_body_when_prefetch_detects_embedded_stream_error() {
run_lifecycle_test(
@@ -1987,8 +1987,9 @@ fn admin_provider_oauth_quota_mod_stays_thin() {
"handlers/admin/provider/oauth/quota/antigravity.rs should import common quota helpers from shared.rs"
);
assert!(
quota_antigravity
.contains("use aether_provider_pool::build_antigravity_pool_quota_request;"),
quota_antigravity.contains("use aether_provider_pool::{")
&& quota_antigravity.contains("build_antigravity_pool_quota_request")
&& quota_antigravity.contains("build_antigravity_pool_quota_summary_request"),
"handlers/admin/provider/oauth/quota/antigravity.rs should delegate antigravity quota request construction to aether-provider-pool"
);
let quota_chatgpt_web = read_workspace_file(
@@ -1,9 +1,15 @@
use std::collections::BTreeMap;
use std::sync::{Arc, Mutex};
use aether_crypto::{encrypt_python_fernet_plaintext, DEVELOPMENT_ENCRYPTION_KEY};
use aether_crypto::{
decrypt_python_fernet_ciphertext, encrypt_python_fernet_plaintext, DEVELOPMENT_ENCRYPTION_KEY,
};
use aether_data::repository::global_models::InMemoryGlobalModelReadRepository;
use aether_data::repository::provider_catalog::InMemoryProviderCatalogReadRepository;
use aether_data::repository::proxy_nodes::InMemoryProxyNodeRepository;
use aether_data_contracts::repository::global_models::{
AdminProviderModelListQuery, GlobalModelReadRepository,
};
use aether_data_contracts::repository::provider_catalog::{
ProviderCatalogReadRepository, StoredProviderCatalogKey, StoredProviderCatalogProvider,
};
@@ -2308,6 +2314,12 @@ async fn gateway_refreshes_admin_provider_quota_locally_for_antigravity_with_tru
},
"gemini-2.5-pro": {
"displayName": "Gemini 2.5 Pro"
},
"gemini-3.7-flash-tiered": {
"displayName": "Gemini 3.7 Flash"
},
"chat_23310": {
"displayName": "Internal Chat"
}
}
}),
@@ -2409,6 +2421,7 @@ async fn gateway_refreshes_admin_provider_quota_locally_for_antigravity_with_tru
)],
vec![key],
));
let global_model_repository = Arc::new(InMemoryGlobalModelReadRepository::default());
let (upstream_url, upstream_handle) = start_server(upstream).await;
let (execution_runtime_url, execution_runtime_handle) = start_server(execution_runtime).await;
@@ -2418,6 +2431,7 @@ async fn gateway_refreshes_admin_provider_quota_locally_for_antigravity_with_tru
GatewayDataState::with_provider_catalog_repository_for_tests(
provider_catalog_repository.clone(),
)
.with_global_model_repository_for_tests(global_model_repository.clone())
.with_encryption_key_for_tests(DEVELOPMENT_ENCRYPTION_KEY),
),
);
@@ -2519,6 +2533,23 @@ async fn gateway_refreshes_admin_provider_quota_locally_for_antigravity_with_tru
.and_then(|value| value.get("remaining_fraction")),
Some(&json!(0.25))
);
let imported_provider_models = global_model_repository
.list_admin_provider_models(&AdminProviderModelListQuery {
provider_id: "provider-antigravity".to_string(),
is_active: None,
offset: 0,
limit: 100,
})
.await
.expect("imported Antigravity provider models should read");
let imported_model_names = imported_provider_models
.iter()
.map(|model| model.provider_model_name.as_str())
.collect::<std::collections::BTreeSet<_>>();
assert!(imported_model_names.contains("claude-sonnet-4"));
assert!(imported_model_names.contains("gemini-2.5-pro"));
assert!(imported_model_names.contains("gemini-3.7-flash-tiered"));
assert!(!imported_model_names.contains("chat_23310"));
assert_eq!(
reloaded[0]
.upstream_metadata
@@ -2,11 +2,9 @@ use std::sync::{Arc, Mutex};
use aether_data::repository::global_models::InMemoryGlobalModelReadRepository;
use aether_data::repository::provider_catalog::InMemoryProviderCatalogReadRepository;
use aether_data::repository::routing_profiles::InMemoryRoutingGroupRepository;
use aether_data_contracts::repository::global_models::{
AdminProviderModelListQuery, GlobalModelReadRepository,
};
use aether_data_contracts::repository::routing_profiles::StoredRoutingGroup;
use axum::body::Body;
use axum::routing::any;
use axum::{extract::Request, Router};
@@ -843,29 +841,6 @@ async fn gateway_handles_admin_global_model_routing_locally_with_trusted_admin_p
]),
);
let routing_group_repository = Arc::new(InMemoryRoutingGroupRepository::seed(
[StoredRoutingGroup {
id: "system-default".to_string(),
name: "system-default".to_string(),
description: Some("test system default routing strategy".to_string()),
enabled: true,
is_system_default: true,
sort_order: 0,
config_json: json!({
"default_policy": {
"priority_mode": "global_key",
"scheduling_mode": "fixed_order"
}
}),
version: 1,
created_at: 1,
updated_at: 1,
published_at: Some(1),
}],
std::iter::empty(),
std::iter::empty(),
));
let (upstream_url, upstream_handle) = start_server(upstream).await;
let gateway = build_router_with_state(
AppState::new()
@@ -875,7 +850,7 @@ async fn gateway_handles_admin_global_model_routing_locally_with_trusted_admin_p
provider_catalog_repository,
)
.with_global_model_repository_for_tests(global_model_repository)
.with_routing_group_repository_for_tests(routing_group_repository),
.with_system_default_routing_group_for_tests(),
),
);
let (gateway_url, gateway_handle) = start_server(gateway).await;
@@ -898,8 +873,8 @@ async fn gateway_handles_admin_global_model_routing_locally_with_trusted_admin_p
assert_eq!(payload["global_model_name"], "gpt-5");
assert_eq!(payload["display_name"], "GPT 5");
assert_eq!(payload["global_model_mappings"], json!(["gpt-5-upstream"]));
assert_eq!(payload["scheduling_mode"], "fixed_order");
assert_eq!(payload["priority_mode"], "global_key");
assert_eq!(payload["scheduling_mode"], "cache_affinity");
assert_eq!(payload["priority_mode"], "provider");
assert_eq!(payload["total_providers"], 2);
assert_eq!(payload["active_providers"], 2);
@@ -3912,6 +3912,175 @@ async fn gateway_completes_admin_provider_oauth_provider_locally_with_trusted_ad
upstream_handle.abort();
}
#[test]
fn gateway_names_new_antigravity_oauth_account_from_google_userinfo_email() {
run_admin_oauth_test(
"gateway_names_new_antigravity_oauth_account_from_google_userinfo_email",
gateway_names_new_antigravity_oauth_account_from_google_userinfo_email_impl,
);
}
async fn gateway_names_new_antigravity_oauth_account_from_google_userinfo_email_impl() {
let upstream_hits = Arc::new(Mutex::new(0usize));
let upstream_hits_clone = Arc::clone(&upstream_hits);
let upstream = Router::new().fallback(any(move |_request: Request| {
let upstream_hits_inner = Arc::clone(&upstream_hits_clone);
async move {
*upstream_hits_inner.lock().expect("mutex should lock") += 1;
(StatusCode::OK, Body::from("unexpected upstream hit"))
}
}));
let token_hits = Arc::new(Mutex::new(0usize));
let token_hits_clone = Arc::clone(&token_hits);
let user_info_hits = Arc::new(Mutex::new(0usize));
let user_info_hits_clone = Arc::clone(&user_info_hits);
let seen_user_info_authorization = Arc::new(Mutex::new(None::<String>));
let seen_user_info_authorization_clone = Arc::clone(&seen_user_info_authorization);
let google_server = Router::new()
.route(
"/oauth/token",
post(move || {
let token_hits_inner = Arc::clone(&token_hits_clone);
async move {
*token_hits_inner.lock().expect("mutex should lock") += 1;
Json(json!({
"access_token": "antigravity-access-token",
"refresh_token": "antigravity-refresh-token",
"token_type": "Bearer",
"expires_in": 3600,
"scope": "https://www.googleapis.com/auth/userinfo.email"
}))
}
}),
)
.route(
"/oauth/userinfo",
get(move |headers: HeaderMap| {
let user_info_hits_inner = Arc::clone(&user_info_hits_clone);
let seen_authorization_inner = Arc::clone(&seen_user_info_authorization_clone);
async move {
*user_info_hits_inner.lock().expect("mutex should lock") += 1;
*seen_authorization_inner.lock().expect("mutex should lock") = headers
.get(http::header::AUTHORIZATION)
.and_then(|value| value.to_str().ok())
.map(ToOwned::to_owned);
Json(json!({
"email": "[email protected]",
"verified_email": true,
"name": "Antigravity User"
}))
}
}),
);
let mut provider = sample_provider("provider-antigravity", "antigravity", 10);
provider.provider_type = "antigravity".to_string();
let endpoint = sample_endpoint(
"endpoint-antigravity",
"provider-antigravity",
"gemini:generate_content",
"https://daily-cloudcode-pa.googleapis.com",
);
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![provider],
vec![endpoint],
vec![],
));
let (upstream_url, upstream_handle) = start_server(upstream).await;
let (google_url, google_handle) = start_server(google_server).await;
let gateway = build_router_with_state(
AppState::new()
.expect("gateway should build")
.with_data_state_for_tests(
GatewayDataState::with_provider_catalog_repository_for_tests(
provider_catalog_repository.clone(),
)
.with_encryption_key_for_tests(DEVELOPMENT_ENCRYPTION_KEY),
)
.with_provider_oauth_state_entry_for_tests(
"nonce-antigravity-123",
json!({
"nonce": "nonce-antigravity-123",
"key_id": "",
"provider_id": "provider-antigravity",
"provider_type": "antigravity",
"pkce_verifier": "verifier-antigravity-123",
}),
)
.with_provider_oauth_token_url_for_tests(
"antigravity",
format!("{google_url}/oauth/token"),
)
.with_provider_oauth_token_url_for_tests(
"antigravity_user_info",
format!("{google_url}/oauth/userinfo"),
),
);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let response = reqwest::Client::new()
.post(format!(
"{gateway_url}/api/admin/provider-oauth/providers/provider-antigravity/complete"
))
.header(crate::constants::GATEWAY_HEADER, "rust-phase3b")
.header(TRUSTED_ADMIN_USER_ID_HEADER, "admin-user-123")
.header(TRUSTED_ADMIN_USER_ROLE_HEADER, "admin")
.header(TRUSTED_ADMIN_SESSION_ID_HEADER, "session-123")
.json(&json!({
"callback_url": "http://localhost:51121/oauth2callback?code=antigravity-code-123&state=nonce-antigravity-123"
}))
.send()
.await
.expect("request should succeed");
let status = response.status();
let payload: Value = response.json().await.expect("json body should parse");
assert_eq!(status, StatusCode::OK, "payload={payload}");
assert_eq!(payload["provider_type"], "antigravity");
assert_eq!(payload["email"], "[email protected]");
assert_eq!(payload["replaced"], false);
assert_eq!(*token_hits.lock().expect("mutex should lock"), 1);
assert_eq!(*user_info_hits.lock().expect("mutex should lock"), 1);
assert_eq!(
seen_user_info_authorization
.lock()
.expect("mutex should lock")
.as_deref(),
Some("Bearer antigravity-access-token")
);
assert_eq!(*upstream_hits.lock().expect("mutex should lock"), 0);
let key_id = payload["key_id"]
.as_str()
.expect("created key id should be returned")
.to_string();
let persisted_keys = provider_catalog_repository
.list_keys_by_ids(std::slice::from_ref(&key_id))
.await
.expect("created key should load");
let persisted = persisted_keys.first().expect("created key should exist");
assert_eq!(persisted.name, "[email protected]");
let decrypted_auth_config = decrypt_python_fernet_ciphertext(
DEVELOPMENT_ENCRYPTION_KEY,
persisted
.encrypted_auth_config
.as_deref()
.expect("auth config should be stored"),
)
.expect("auth config should decrypt");
let auth_config: Value =
serde_json::from_str(&decrypted_auth_config).expect("auth config json should parse");
assert_eq!(auth_config["email"], "[email protected]");
assert_eq!(auth_config["refresh_token"], "antigravity-refresh-token");
gateway_handle.abort();
google_handle.abort();
upstream_handle.abort();
drop(upstream_url);
}
#[test]
fn gateway_imports_admin_provider_oauth_refresh_token_locally_with_trusted_admin_principal() {
run_admin_oauth_test(
@@ -2184,7 +2184,7 @@ async fn gateway_handles_admin_system_config_locally_with_bearer_admin_session()
}),
);
let (upstream_url, upstream_handle) = start_server(upstream).await;
let (_upstream_url, upstream_handle) = start_server(upstream).await;
let state = AppState::new().expect("gateway should build");
let access_token = issue_test_admin_access_token(&state, "device-admin-config").await;
let gateway = build_router_with_state(state);
@@ -2208,6 +2208,44 @@ async fn gateway_handles_admin_system_config_locally_with_bearer_admin_session()
upstream_handle.abort();
}
#[tokio::test]
async fn gateway_rejects_removed_admin_system_provider_priority_mode_with_bearer_admin_session() {
let upstream_hits = Arc::new(Mutex::new(0usize));
let upstream_hits_clone = Arc::clone(&upstream_hits);
let upstream = Router::new().route(
"/api/admin/system/configs/provider_priority_mode",
any(move |_request: Request| {
let upstream_hits_inner = Arc::clone(&upstream_hits_clone);
async move {
*upstream_hits_inner.lock().expect("mutex should lock") += 1;
(StatusCode::OK, Body::from("unexpected upstream hit"))
}
}),
);
let (_upstream_url, upstream_handle) = start_server(upstream).await;
let state = AppState::new().expect("gateway should build");
let access_token = issue_test_admin_access_token(&state, "device-admin-config").await;
let gateway = build_router_with_state(state);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let response = reqwest::Client::new()
.get(format!(
"{gateway_url}/api/admin/system/configs/provider_priority_mode"
))
.header("authorization", format!("Bearer {access_token}"))
.header("x-client-device-id", "device-admin-config")
.send()
.await
.expect("request should succeed");
assert_eq!(response.status(), StatusCode::NOT_FOUND);
assert_eq!(*upstream_hits.lock().expect("mutex should lock"), 0);
gateway_handle.abort();
upstream_handle.abort();
}
#[tokio::test]
async fn gateway_sets_admin_system_config_locally_with_trusted_admin_principal() {
let upstream_hits = Arc::new(Mutex::new(0usize));
@@ -5,6 +5,7 @@ use super::{
InMemoryVideoTaskRepository, StoredAuthApiKeySnapshot, UpsertVideoTask, VideoTaskLookupKey,
VideoTaskReadRepository, VideoTaskStatus, VideoTaskWriteRepository, DEVELOPMENT_ENCRYPTION_KEY,
};
use crate::data::GatewayDataState;
use crate::image_capabilities::openai_image_gateway_max_generation_count;
use crate::tests::{
any, build_router_with_state, build_state_with_execution_runtime_override, json, start_server,
@@ -6,6 +6,7 @@ use super::{
InMemoryVideoTaskRepository, UpsertVideoTask, VideoTaskStatus, VideoTaskWriteRepository,
DEVELOPMENT_ENCRYPTION_KEY,
};
use crate::data::GatewayDataState;
use crate::tests::{
any, build_router, build_router_with_state, build_state_with_execution_runtime_override, json,
start_server, strip_sse_keepalive_comments, AppState, Arc, Body, HeaderValue, Json, Mutex,
@@ -1850,7 +1851,7 @@ async fn gateway_returns_internal_gateway_decision_sync_fallback_with_resolved_a
AppState::new()
.expect("gateway should build")
.with_data_state_for_tests(
crate::data::GatewayDataState::with_auth_api_key_reader_for_tests(auth_repository)
GatewayDataState::with_auth_api_key_reader_for_tests(auth_repository)
.with_system_default_routing_group_for_tests(),
),
);
+16
View File
@@ -27,6 +27,10 @@ use aether_data_contracts::repository::provider_catalog::{
StoredProviderCatalogEndpoint, StoredProviderCatalogKey, StoredProviderCatalogProvider,
};
use aether_runtime_state::{RedisClientConfig, RuntimeState};
use aether_scheduler_core::{
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope,
SchedulerAffinityScope,
};
use aether_test_support::ManagedRedisServer;
use sha2::{Digest, Sha256};
@@ -41,6 +45,18 @@ fn hash_api_key(value: &str) -> String {
format!("{:x}", hasher.finalize())
}
fn system_default_affinity_cache_key(api_key_id: &str, api_format: &str, model: &str) -> String {
let scope = SchedulerAffinityScope::new("system-default", Some(1));
build_scheduler_affinity_cache_key_for_api_key_id_with_client_session_and_scope(
api_key_id,
api_format,
model,
None,
Some(&scope),
)
.expect("system-default affinity cache key should build")
}
fn sample_auth_snapshot(
api_key_id: &str,
user_id: &str,