Merge remote-tracking branch 'origin/pr/613'

# Conflicts:
#	apps/aether-gateway/src/tests/usage/direct.rs
This commit is contained in:
elky
2026-06-02 19:27:50 +08:00
6 changed files with 623 additions and 169 deletions
@@ -3357,7 +3357,13 @@ async fn execute_stream_from_frame_stream(
"gateway skipped client stream flush after downstream disconnect"
);
}
if let Some(normalizer) = private_stream_normalizer.as_mut() {
// Buffered stream state is partial after a terminal failure; normal
// finish paths may synthesize successful terminal events.
let should_finish_stream_rewriters = terminal_failure.is_none();
if let Some(normalizer) = private_stream_normalizer
.as_mut()
.filter(|_| should_finish_stream_rewriters)
{
match normalizer.finish() {
Ok(normalized_chunk) if !normalized_chunk.is_empty() => {
let provider_private_error_body_json =
@@ -3391,13 +3397,12 @@ async fn execute_stream_from_frame_stream(
error = ?err,
"gateway failed to rewrite normalized private stream chunk during flush"
);
terminal_failure.get_or_insert_with(|| {
build_stream_failure_report(
"execution_runtime_stream_rewrite_flush_error",
format!("failed to rewrite normalized private stream chunk during flush: {err:?}"),
502,
)
});
let failure = build_stream_failure_report(
"execution_runtime_stream_rewrite_flush_error",
format!("failed to rewrite normalized private stream chunk during flush: {err:?}"),
502,
);
terminal_failure.get_or_insert(failure);
Vec::new()
}
}
@@ -3472,7 +3477,7 @@ async fn execute_stream_from_frame_stream(
}
}
}
if !downstream_dropped {
if !downstream_dropped && terminal_failure.is_none() {
if let Some(rewriter) = local_stream_rewriter.as_mut() {
match rewriter.finish() {
Ok(flushed_chunk) if !flushed_chunk.is_empty() => {
@@ -3898,8 +3903,9 @@ mod tests {
use std::time::{Duration, Instant};
use aether_contracts::{
ExecutionPlan, ExecutionStreamTerminalSummary, ExecutionTimeouts, RequestBody,
StandardizedUsage,
ExecutionError, ExecutionErrorKind, ExecutionPhase, ExecutionPlan,
ExecutionStreamTerminalSummary, ExecutionTimeouts, RequestBody, StandardizedUsage,
StreamFrame, StreamFramePayload, StreamFrameType,
};
use aether_data::repository::candidates::InMemoryRequestCandidateRepository;
use aether_data::repository::usage::InMemoryUsageReadRepository;
@@ -3996,6 +4002,12 @@ mod tests {
out
}
fn ndjson_frame(frame: StreamFrame) -> Bytes {
let mut bytes = serde_json::to_vec(&frame).expect("stream frame should serialize");
bytes.push(b'\n');
Bytes::from(bytes)
}
#[test]
fn merge_stream_terminal_summary_prefers_more_complete_observed_usage() {
let mut runtime_usage = StandardizedUsage::new();
@@ -5017,6 +5029,154 @@ mod tests {
assert_eq!(first.as_ref(), b": aether-keepalive\n\n");
}
#[tokio::test]
async fn execute_stream_from_frame_stream_does_not_finalize_rewritten_tool_call_after_midstream_error(
) {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let state = AppState::new()
.expect("app state should build")
.with_data_state_for_tests(
crate::data::GatewayDataState::with_request_candidate_and_usage_repository_for_tests(
Arc::clone(&request_candidate_repository),
Arc::clone(&usage_repository),
),
)
.with_usage_runtime_for_tests(UsageRuntimeConfig {
enabled: true,
..UsageRuntimeConfig::default()
});
let plan = ExecutionPlan {
request_id: "req-responses-tool-midstream-error".into(),
candidate_id: Some("cand-responses-tool-midstream-error".into()),
provider_name: Some("openai".into()),
provider_id: "provider-openai-responses".into(),
endpoint_id: "endpoint-openai-responses".into(),
key_id: "key-openai-responses".into(),
method: "POST".into(),
url: "https://api.openai.com/v1/responses".into(),
headers: BTreeMap::from([
("content-type".into(), "application/json".into()),
("accept".into(), "text/event-stream".into()),
]),
content_type: Some("application/json".into()),
content_encoding: None,
body: RequestBody::from_json(json!({
"model": "gpt-5.5",
"input": [],
"stream": true
})),
stream: true,
client_api_format: "claude:messages".into(),
provider_api_format: "openai:responses".into(),
model_name: Some("gpt-5.5".into()),
proxy: None,
transport_profile: None,
timeouts: None,
};
let upstream_chunk = concat!(
"event: response.created\n",
"data: {\"type\":\"response.created\",\"response\":{\"id\":\"resp_midstream_error\",\"model\":\"gpt-5.5\",\"status\":\"in_progress\"}}\n\n",
"event: response.output_item.added\n",
"data: {\"type\":\"response.output_item.added\",\"output_index\":0,\"item\":{\"type\":\"function_call\",\"id\":\"fc_1\",\"call_id\":\"call_1\",\"name\":\"lookup\",\"arguments\":\"\",\"status\":\"in_progress\"}}\n\n",
"event: response.function_call_arguments.delta\n",
"data: {\"type\":\"response.function_call_arguments.delta\",\"output_index\":0,\"item_id\":\"fc_1\",\"call_id\":\"call_1\",\"delta\":\"{\\\"query\\\":\\\"abc\"}\n\n"
);
let frame_stream = stream! {
yield Ok::<Bytes, std::io::Error>(ndjson_frame(StreamFrame {
frame_type: StreamFrameType::Headers,
payload: StreamFramePayload::Headers {
status_code: 200,
headers: BTreeMap::from([(
"content-type".to_string(),
"text/event-stream".to_string(),
)]),
},
}));
yield Ok::<Bytes, std::io::Error>(ndjson_frame(StreamFrame {
frame_type: StreamFrameType::Data,
payload: StreamFramePayload::Data {
chunk_b64: None,
text: Some(upstream_chunk.to_string()),
},
}));
yield Ok::<Bytes, std::io::Error>(ndjson_frame(StreamFrame {
frame_type: StreamFrameType::Error,
payload: StreamFramePayload::Error {
error: ExecutionError {
kind: ExecutionErrorKind::Internal,
phase: ExecutionPhase::StreamRead,
message: "error reading a body from connection: stream error received: unexpected internal error encountered".to_string(),
upstream_status: Some(200),
retryable: false,
failover_recommended: false,
},
},
}));
}
.boxed();
let response = execute_stream_from_frame_stream(
&state,
plan,
"trace-responses-tool-midstream-error",
&test_decision(),
"openai_responses_stream",
Some("openai_responses_stream_success".to_string()),
Some(json!({
"request_id": "req-responses-tool-midstream-error",
"candidate_id": "cand-responses-tool-midstream-error",
"candidate_index": 0,
"retry_index": 0,
"provider_api_format": "openai:responses",
"client_api_format": "claude:messages",
"needs_conversion": true,
})),
crate::clock::current_unix_ms(),
Instant::now(),
frame_stream,
None,
)
.await
.expect("execution should succeed")
.expect("execution should return a client response");
let body = to_bytes(response.into_body(), usize::MAX)
.await
.expect("response body should read");
let body_text = String::from_utf8(body.to_vec()).expect("body should be utf8");
assert!(body_text.contains("event: content_block_start"));
assert!(body_text.contains("event: content_block_delta"));
assert!(body_text.contains("\"type\":\"tool_use\""));
assert!(!body_text.contains("event: content_block_stop"));
assert!(!body_text.contains("event: message_delta"));
assert!(!body_text.contains("event: message_stop"));
assert!(!body_text.contains("\"stop_reason\":\"tool_use\""));
assert!(body_text.contains("\"error\""));
assert!(body_text.contains("unexpected internal error encountered"));
assert!(body_text.contains("data: [DONE]"));
let candidates = tokio::time::timeout(Duration::from_secs(1), async {
loop {
let candidates = request_candidate_repository
.list_by_request_id("req-responses-tool-midstream-error")
.await
.expect("request candidates should read");
if candidates
.first()
.is_some_and(|candidate| candidate.status == RequestCandidateStatus::Failed)
{
break candidates;
}
tokio::time::sleep(Duration::from_millis(10)).await;
}
})
.await
.expect("candidate should be marked failed");
assert_eq!(candidates[0].status_code, Some(200));
assert_eq!(candidates[0].error_type.as_deref(), Some("internal"));
}
#[tokio::test]
async fn openai_image_stream_ignores_plan_total_timeout() {
let state = AppState::new().expect("app state should build");
+128 -125
View File
@@ -154,74 +154,79 @@ async fn gateway_records_usage_for_execution_runtime_sync_when_runtime_enabled()
#[test]
fn gateway_records_pending_usage_before_execution_runtime_sync_result_arrives() {
run_async_test_on_large_stack("pending-usage-sync-before-runtime-result", async move {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let execution_request_started = Arc::new(tokio::sync::Notify::new());
let allow_execution_response = Arc::new(tokio::sync::Notify::new());
run_async_test_on_large_stack(
"gateway_records_pending_usage_before_execution_runtime_sync_result_arrives",
gateway_records_pending_usage_before_execution_runtime_sync_result_arrives_impl(),
);
}
let upstream = Router::new().route(
"/api/internal/gateway/report-sync",
any(|_request: Request| async move { Json(json!({"ok": true})) }),
);
async fn gateway_records_pending_usage_before_execution_runtime_sync_result_arrives_impl() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let execution_request_started = Arc::new(tokio::sync::Notify::new());
let allow_execution_response = Arc::new(tokio::sync::Notify::new());
let execution_runtime = Router::new().route(
"/v1/execute/sync",
any({
let upstream = Router::new().route(
"/api/internal/gateway/report-sync",
any(|_request: Request| async move { Json(json!({"ok": true})) }),
);
let execution_runtime = Router::new().route(
"/v1/execute/sync",
any({
let execution_request_started = Arc::clone(&execution_request_started);
let allow_execution_response = Arc::clone(&allow_execution_response);
move |_request: Request| {
let execution_request_started = Arc::clone(&execution_request_started);
let allow_execution_response = Arc::clone(&allow_execution_response);
move |_request: Request| {
let execution_request_started = Arc::clone(&execution_request_started);
let allow_execution_response = Arc::clone(&allow_execution_response);
async move {
execution_request_started.notify_one();
allow_execution_response.notified().await;
Json(json!({
"request_id": "req-usage-sync-pending-123",
"status_code": 200,
"headers": {
"content-type": "application/json"
},
"body": {
"json_body": {
"id": "chatcmpl-usage-sync-pending-123",
"usage": {
"input_tokens": 3,
"output_tokens": 5,
"total_tokens": 8
}
async move {
execution_request_started.notify_one();
allow_execution_response.notified().await;
Json(json!({
"request_id": "req-usage-sync-pending-123",
"status_code": 200,
"headers": {
"content-type": "application/json"
},
"body": {
"json_body": {
"id": "chatcmpl-usage-sync-pending-123",
"usage": {
"input_tokens": 3,
"output_tokens": 5,
"total_tokens": 8
}
},
"telemetry": {
"elapsed_ms": 45
}
}))
}
},
"telemetry": {
"elapsed_ms": 45
}
}))
}
}),
);
}
}),
);
let auth_repository = Arc::new(InMemoryAuthApiKeySnapshotRepository::seed(vec![(
Some(hash_api_key("sk-client-openai-usage-sync-pending")),
sample_local_openai_auth_snapshot(
"api-key-usage-sync-pending-123",
"user-usage-sync-pending-123",
),
)]));
let candidate_selection_repository =
Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(vec![
sample_local_openai_candidate_row(),
]));
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![sample_local_openai_provider()],
vec![sample_local_openai_endpoint()],
vec![sample_local_openai_key()],
));
let auth_repository = Arc::new(InMemoryAuthApiKeySnapshotRepository::seed(vec![(
Some(hash_api_key("sk-client-openai-usage-sync-pending")),
sample_local_openai_auth_snapshot(
"api-key-usage-sync-pending-123",
"user-usage-sync-pending-123",
),
)]));
let candidate_selection_repository =
Arc::new(InMemoryMinimalCandidateSelectionReadRepository::seed(vec![
sample_local_openai_candidate_row(),
]));
let provider_catalog_repository = Arc::new(InMemoryProviderCatalogReadRepository::seed(
vec![sample_local_openai_provider()],
vec![sample_local_openai_endpoint()],
vec![sample_local_openai_key()],
));
let (upstream_url, upstream_handle) = start_server(upstream).await;
let (execution_runtime_url, execution_runtime_handle) =
start_server(execution_runtime).await;
let gateway_state =
let (upstream_url, upstream_handle) = start_server(upstream).await;
let (execution_runtime_url, execution_runtime_handle) = start_server(execution_runtime).await;
let gateway_state =
build_state_with_execution_runtime_override(execution_runtime_url)
.with_data_state_for_tests(
GatewayDataState::with_auth_candidate_selection_provider_catalog_request_candidates_and_usage_for_tests(
@@ -237,76 +242,74 @@ fn gateway_records_pending_usage_before_execution_runtime_sync_result_arrives()
enabled: true,
..UsageRuntimeConfig::default()
});
let gateway = build_router_with_state(gateway_state);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let gateway = build_router_with_state(gateway_state);
let (gateway_url, gateway_handle) = start_server(gateway).await;
let request_task = tokio::spawn({
let gateway_url = gateway_url.clone();
async move {
let response = reqwest::Client::new()
.post(format!("{gateway_url}/v1/chat/completions"))
.header(http::header::CONTENT_TYPE, "application/json")
.header(
http::header::AUTHORIZATION,
"Bearer sk-client-openai-usage-sync-pending",
)
.header(TRACE_ID_HEADER, "req-usage-sync-pending-123")
.body("{\"model\":\"gpt-5\",\"messages\":[]}")
.send()
.await
.expect("request should succeed");
let status = response.status();
let body = response.text().await.expect("body should read");
(status, body)
}
});
execution_request_started.notified().await;
let mut pending = None;
for _ in 0..50 {
pending = usage_repository
.find_by_request_id("req-usage-sync-pending-123")
let request_task = tokio::spawn({
let gateway_url = gateway_url.clone();
async move {
let response = reqwest::Client::new()
.post(format!("{gateway_url}/v1/chat/completions"))
.header(http::header::CONTENT_TYPE, "application/json")
.header(
http::header::AUTHORIZATION,
"Bearer sk-client-openai-usage-sync-pending",
)
.header(TRACE_ID_HEADER, "req-usage-sync-pending-123")
.body("{\"model\":\"gpt-5\",\"messages\":[]}")
.send()
.await
.expect("usage lookup should succeed");
if pending
.as_ref()
.is_some_and(|stored| stored.status == "pending")
{
break;
}
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
.expect("request should succeed");
let status = response.status();
let body = response.text().await.expect("body should read");
(status, body)
}
let pending =
pending.expect("pending usage should be recorded before sync result resolves");
assert_eq!(pending.status, "pending");
assert_eq!(pending.billing_status, "pending");
assert_eq!(pending.response_time_ms, None);
allow_execution_response.notify_one();
let (status, _body) = request_task.await.expect("request task should join");
assert_eq!(status, StatusCode::OK);
let mut stored = None;
for _ in 0..50 {
stored = usage_repository
.find_by_request_id("req-usage-sync-pending-123")
.await
.expect("usage lookup should succeed");
if stored.as_ref().is_some_and(|row| row.status == "completed") {
break;
}
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
}
let stored = stored.expect("usage should be finalized");
assert_eq!(stored.status, "completed");
assert_eq!(stored.response_time_ms, Some(45));
gateway_handle.abort();
execution_runtime_handle.abort();
upstream_handle.abort();
});
execution_request_started.notified().await;
let mut pending = None;
for _ in 0..50 {
pending = usage_repository
.find_by_request_id("req-usage-sync-pending-123")
.await
.expect("usage lookup should succeed");
if pending
.as_ref()
.is_some_and(|stored| stored.status == "pending")
{
break;
}
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
}
let pending = pending.expect("pending usage should be recorded before sync result resolves");
assert_eq!(pending.status, "pending");
assert_eq!(pending.billing_status, "pending");
assert_eq!(pending.response_time_ms, None);
allow_execution_response.notify_one();
let (status, _body) = request_task.await.expect("request task should join");
assert_eq!(status, StatusCode::OK);
let mut stored = None;
for _ in 0..50 {
stored = usage_repository
.find_by_request_id("req-usage-sync-pending-123")
.await
.expect("usage lookup should succeed");
if stored.as_ref().is_some_and(|row| row.status == "completed") {
break;
}
tokio::time::sleep(std::time::Duration::from_millis(10)).await;
}
let stored = stored.expect("usage should be finalized");
assert_eq!(stored.status, "completed");
assert_eq!(stored.response_time_ms, Some(45));
gateway_handle.abort();
execution_runtime_handle.abort();
upstream_handle.abort();
}
#[tokio::test]
+9 -2
View File
@@ -559,8 +559,15 @@ async fn gateway_applies_system_max_request_body_size_to_local_openai_chat_sync_
upstream_handle.abort();
}
#[tokio::test]
async fn gateway_strips_request_and_response_bodies_when_request_record_level_is_base() {
#[test]
fn gateway_strips_request_and_response_bodies_when_request_record_level_is_base() {
run_async_test_on_large_stack(
"gateway_strips_request_and_response_bodies_when_request_record_level_is_base",
gateway_strips_request_and_response_bodies_when_request_record_level_is_base_impl(),
);
}
async fn gateway_strips_request_and_response_bodies_when_request_record_level_is_base_impl() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
+30 -2
View File
@@ -11,8 +11,36 @@ use super::{
};
use aether_data::repository::settlement::InMemorySettlementRepository;
#[tokio::test]
async fn gateway_settles_wallet_for_completed_execution_runtime_sync_usage() {
fn run_async_test_on_large_stack<F>(name: &'static str, future: F)
where
F: std::future::Future<Output = ()> + Send + 'static,
{
let handle = std::thread::Builder::new()
.name(name.to_string())
.stack_size(8 * 1024 * 1024)
.spawn(move || {
tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("tokio runtime should build")
.block_on(future);
})
.expect("large-stack test thread should spawn");
if let Err(payload) = handle.join() {
std::panic::resume_unwind(payload);
}
}
#[test]
fn gateway_settles_wallet_for_completed_execution_runtime_sync_usage() {
run_async_test_on_large_stack(
"gateway_settles_wallet_for_completed_execution_runtime_sync_usage",
gateway_settles_wallet_for_completed_execution_runtime_sync_usage_impl(),
);
}
async fn gateway_settles_wallet_for_completed_execution_runtime_sync_usage_impl() {
let usage_repository = Arc::new(InMemoryUsageReadRepository::default());
let request_candidate_repository = Arc::new(InMemoryRequestCandidateRepository::default());
let billing_repository = Arc::new(InMemoryBillingReadRepository::seed(vec![