feat: revamp analytics dashboards and harden database migrations

Add dashboard and overview analytics, health monitoring, provider expense tracking, and announcement updates across the gateway and frontend.

Keep schema migrations free of historical backfills while preserving automatic backfill execution. Bound migration deadlines, run schema preparation before Compose replacement, and anonymize deleted dashboard users.

Include the current documentation cleanup and regression coverage.
This commit is contained in:
elky
2026-10-01 11:48:17 +08:00
parent 60b89cc840
commit 066ea87d72
327 changed files with 31728 additions and 20645 deletions
@@ -181,6 +181,12 @@ async fn settle_cancelled_attempt(
usage_data.request_metadata.take(),
request_diagnostics.as_ref(),
);
usage_data.request_metadata = crate::usage::reporting::failure::with_analytics_failure(
usage_data.request_metadata.as_ref(),
"unknown",
"finalize",
"request_task_cancelled",
);
usage_data.status_code = Some(CLIENT_CANCELLED_STATUS_CODE);
usage_data.error_message = Some(error_message.to_string());
usage_data.error_category = Some("cancelled".to_string());
@@ -668,8 +668,22 @@ impl ExecutionAttemptLifecycle {
});
// 1. usage terminal
let analytics_context = if facts.provider.cancelled_by_provider() {
crate::usage::reporting::failure::with_analytics_failure(
payload.report_context.as_ref(),
"upstream",
"stream_read",
"provider_cancelled",
)
} else {
crate::usage::reporting::failure::stream_analytics_context(
payload.report_context.as_ref(),
&payload,
facts.delivery.is_aborted() && !facts.provider.is_terminal(),
)
};
let context_seed =
build_terminal_usage_context_seed(&self.plan, payload.report_context.as_ref());
build_terminal_usage_context_seed(&self.plan, analytics_context.as_ref());
let payload_seed = build_stream_terminal_usage_payload_seed(&payload);
let billing_void = settlement.billing.is_void();
let usage_runtime = Arc::clone(&state.usage_runtime);
@@ -1114,6 +1114,7 @@ fn grok_canonical_usage(usage: GrokUsageEstimate) -> StreamingCanonicalUsage {
fn grok_standardized_usage(usage: GrokUsageEstimate) -> StandardizedUsage {
let mut standardized = StandardizedUsage::new();
standardized.token_source = Some(aether_contracts::UsageTokenSource::Estimated);
standardized.input_tokens = i64::try_from(usage.input_tokens).unwrap_or(i64::MAX);
standardized.output_tokens = i64::try_from(usage.output_tokens).unwrap_or(i64::MAX);
standardized.reasoning_tokens = i64::try_from(usage.reasoning_tokens).unwrap_or(i64::MAX);
@@ -4574,6 +4575,101 @@ mod tests {
assert!(adapter.text.contains("[[1]](https://example.com/source"));
}
#[test]
fn grok_usage_reports_preserve_estimated_provenance_after_wire_roundtrip() {
use aether_usage_runtime::{
build_stream_terminal_usage_event, build_sync_terminal_usage_event,
GatewayStreamReportRequest, GatewaySyncReportRequest, UsageEventType,
};
for (format, report_prefix) in [
("openai:chat", "openai_chat"),
("openai:responses", "openai_responses"),
] {
let mut plan = sample_plan(
serde_json::json!({
"messages": [{"role": "user", "content": "hello"}]
}),
format,
);
plan.stream = false;
plan.provider_api_format = format.to_string();
// The trusted planner binds this hint to the Grok runtime adapter.
// Exercise its transport through the same serialized report as usage.
let context = serde_json::json!({
"provider_type": "grok",
"provider_api_format": format,
"client_api_format": format,
"usage_token_source": "estimated"
});
let collected = GrokCollected {
status_code: 200,
text: "hello back".to_string(),
thinking: "short reasoning".to_string(),
..GrokCollected::default()
};
let expected = grok_usage_estimate(&plan, &collected);
let result = grok_execution_result(&plan, collected, Some(&context));
let sync_report = GatewaySyncReportRequest {
trace_id: plan.request_id.clone(),
report_kind: format!("{report_prefix}_sync_success"),
report_context: Some(context.clone()),
status_code: result.status_code,
headers: result.headers,
body_json: result.body.and_then(|body| body.json_body),
client_body_json: None,
body_base64: None,
telemetry: result.telemetry,
};
let sync_report: GatewaySyncReportRequest =
serde_json::from_slice(&serde_json::to_vec(&sync_report).unwrap()).unwrap();
let sync_event = build_sync_terminal_usage_event(
&plan,
sync_report.report_context.as_ref(),
&sync_report,
)
.unwrap();
plan.stream = true;
let stream_report = GatewayStreamReportRequest {
trace_id: plan.request_id.clone(),
report_kind: format!("{report_prefix}_stream_success"),
report_context: Some(context),
status_code: 200,
headers: BTreeMap::new(),
provider_body_base64: None,
provider_body_state: None,
client_body_base64: None,
client_body_state: None,
terminal_summary: Some(super::grok_stream_terminal_summary(&plan, expected)),
telemetry: None,
};
let stream_report: GatewayStreamReportRequest =
serde_json::from_slice(&serde_json::to_vec(&stream_report).unwrap()).unwrap();
let stream_event = build_stream_terminal_usage_event(
&plan,
stream_report.report_context.as_ref(),
&stream_report,
)
.unwrap();
// Sync honors the response's explicit total. The existing stream
// summary has no explicit total, so its fallback also adds reasoning.
let sync_total = expected.input_tokens + expected.output_tokens;
let stream_total = sync_total + expected.reasoning_tokens;
for (event, expected_total) in [(sync_event, sync_total), (stream_event, stream_total)]
{
assert_eq!(event.event_type, UsageEventType::Completed, "{format}");
assert_eq!(event.data.input_tokens, Some(expected.input_tokens));
assert_eq!(event.data.output_tokens, Some(expected.output_tokens));
assert_eq!(event.data.total_tokens, Some(expected_total));
let metadata = event.data.request_metadata.unwrap();
assert_eq!(metadata["analytics_measurement"]["source"], "estimated");
assert!(metadata.get("usage_token_source").is_none());
}
}
}
#[test]
fn openai_chat_body_includes_estimated_usage() {
let plan = sample_plan(
@@ -12,7 +12,7 @@ use std::time::{Duration, Instant};
use aether_ai_serving::{AiAttemptExecutionOutcome, AiAttemptRetryScope};
use aether_contracts::{
ExecutionPlan, ExecutionResponseObservation, ExecutionStreamTerminalSummary,
ExecutionTelemetry, StandardizedUsage, StreamFrame, StreamFramePayload,
ExecutionTelemetry, StandardizedUsage, StreamFrame, StreamFramePayload, UsageTokenSource,
};
use aether_data_contracts::repository::candidates::{
RequestCandidateStatus, UpsertRequestCandidateRecord,
@@ -445,11 +445,15 @@ fn build_sync_terminal_usage_seeds(
report_context: Option<&serde_json::Value>,
payload: &GatewaySyncReportRequest,
) -> (TerminalUsageContextSeed, SyncTerminalUsagePayloadSeed) {
let analytics_context =
crate::usage::reporting::failure::sync_analytics_context(report_context, payload);
let report_context_with_diagnostics =
attach_current_request_diagnostics_to_report_context(report_context);
attach_current_request_diagnostics_to_report_context(analytics_context.as_ref());
let context_seed = build_terminal_usage_context_seed(
plan,
report_context_with_diagnostics.as_ref().or(report_context),
report_context_with_diagnostics
.as_ref()
.or(analytics_context.as_ref()),
);
let payload_seed = build_sync_terminal_usage_payload_seed(payload);
(context_seed, payload_seed)
@@ -586,7 +590,12 @@ async fn record_stream_terminal_usage(
cancelled: bool,
) {
crate::execution_runtime::mark_stream_candidate_watchdog_terminal_started();
let context_seed = build_terminal_usage_context_seed(plan, report_context);
let analytics_context = crate::usage::reporting::failure::stream_analytics_context(
report_context,
payload,
cancelled,
);
let context_seed = build_terminal_usage_context_seed(plan, analytics_context.as_ref());
let payload_seed = build_stream_terminal_usage_payload_seed(payload);
state
.usage_runtime
@@ -976,6 +985,9 @@ async fn maybe_apply_kiro_prompt_cache_usage_to_stream_summary(
usage.cache_read_tokens = 0;
if usage.input_tokens <= 0 {
usage.input_tokens = estimated_input_tokens as i64;
if usage.input_tokens > 0 {
mark_kiro_stream_estimated_usage(usage, report_context, false);
}
}
return;
}
@@ -984,6 +996,10 @@ async fn maybe_apply_kiro_prompt_cache_usage_to_stream_summary(
usage.input_tokens = kiro_billed_input_tokens(estimated_input_tokens, cache_usage) as i64;
usage.cache_creation_tokens = cache_usage.cache_creation_input_tokens as i64;
usage.cache_read_tokens = cache_usage.cache_read_input_tokens as i64;
if usage.input_tokens > 0 || usage.cache_creation_tokens > 0 || usage.cache_read_tokens > 0
{
mark_kiro_stream_estimated_usage(usage, report_context, false);
}
return;
}
@@ -996,12 +1012,18 @@ async fn maybe_apply_kiro_prompt_cache_usage_to_stream_summary(
cache_read_input_tokens: usage.cache_read_tokens.max(0) as u64,
},
) as i64;
if usage.input_tokens > 0 {
mark_kiro_stream_estimated_usage(usage, report_context, true);
}
}
return;
}
if usage.input_tokens <= 0 {
usage.input_tokens = estimated_input_tokens as i64;
if usage.input_tokens > 0 {
mark_kiro_stream_estimated_usage(usage, report_context, true);
}
}
let Some(profile) =
@@ -1024,6 +1046,35 @@ async fn maybe_apply_kiro_prompt_cache_usage_to_stream_summary(
usage.input_tokens = billed_input_tokens as i64;
usage.cache_creation_tokens = cache_usage.cache_creation_input_tokens as i64;
usage.cache_read_tokens = cache_usage.cache_read_input_tokens as i64;
mark_kiro_stream_estimated_usage(usage, report_context, false);
}
fn mark_kiro_stream_estimated_usage(
usage: &mut StandardizedUsage,
report_context: &Value,
retains_cache: bool,
) {
let retained_source = usage.token_source.unwrap_or_else(|| {
match report_context
.get("usage_token_source")
.and_then(Value::as_str)
{
Some("estimated") => UsageTokenSource::Estimated,
Some("mixed") => UsageTokenSource::Mixed,
_ => UsageTokenSource::Reported,
}
});
let retains_reported_tokens = retained_source != UsageTokenSource::Estimated
&& (usage.output_tokens > 0
|| usage.reasoning_tokens > 0
|| usage.cache_creation_ephemeral_5m_tokens > 0
|| usage.cache_creation_ephemeral_1h_tokens > 0
|| (retains_cache && (usage.cache_creation_tokens > 0 || usage.cache_read_tokens > 0)));
usage.token_source = Some(if retains_reported_tokens {
UsageTokenSource::Mixed
} else {
UsageTokenSource::Estimated
});
}
fn append_stream_capture_bytes(
@@ -3963,7 +4014,7 @@ async fn execute_execution_runtime_stream_inner(
let candidate_started_unix_secs = current_request_candidate_unix_ms();
let provider_in_flight_started_at = Instant::now();
let mut provider_pool_in_flight_guard =
match acquire_provider_pool_execution_guard(state, &plan).await? {
match acquire_provider_pool_execution_guard(state, &plan, report_context.as_ref()).await? {
ProviderPoolInFlightAdmission::Acquired(guard) => guard,
ProviderPoolInFlightAdmission::Saturated { limit } => {
record_local_runtime_candidate_skip_reason(
@@ -12293,6 +12344,10 @@ mod tests {
.expect("first usage should exist");
assert!(first_usage.cache_creation_tokens > 0);
assert_eq!(first_usage.cache_read_tokens, 0);
assert_eq!(
first_usage.token_source,
Some(aether_contracts::UsageTokenSource::Mixed)
);
let mut second_summary = Some(ExecutionStreamTerminalSummary {
standardized_usage: Some(StandardizedUsage {
@@ -12317,6 +12372,10 @@ mod tests {
assert_eq!(second_usage.cache_creation_tokens, 0);
assert!(second_usage.input_tokens < 6_000);
assert_eq!(second_usage.output_tokens, 19);
assert_eq!(
second_usage.token_source,
Some(aether_contracts::UsageTokenSource::Mixed)
);
}
#[tokio::test]
@@ -12512,6 +12571,49 @@ mod tests {
assert_eq!(usage.cache_creation_tokens, 0);
assert_eq!(usage.cache_read_tokens, 0);
assert_eq!(usage.output_tokens, 13);
assert_eq!(
usage.token_source,
Some(aether_contracts::UsageTokenSource::Mixed)
);
use aether_contracts::UsageTokenSource::{Estimated, Mixed};
for (hint, source, input, output, cache, expected) in [
(Some("estimated"), None, 0, 13, 0, Some(Estimated)),
(None, Some(Estimated), 0, 13, 0, Some(Estimated)),
(None, None, 0, 0, 200, Some(Mixed)),
(None, None, 0, 0, 0, Some(Estimated)),
(None, None, 50, 13, 0, None),
] {
let mut context = report_context.clone();
if let Some(hint) = hint {
context["usage_token_source"] = json!(hint);
}
let mut summary = Some(ExecutionStreamTerminalSummary {
standardized_usage: Some(StandardizedUsage {
token_source: source,
input_tokens: input,
output_tokens: output,
cache_read_tokens: cache,
..StandardizedUsage::new()
}),
..Default::default()
});
maybe_apply_kiro_prompt_cache_usage_to_stream_summary(
&state,
&plan,
Some(&context),
&mut summary,
)
.await;
let usage = summary.unwrap().standardized_usage.unwrap();
assert!(usage.input_tokens > 0);
assert_eq!(usage.output_tokens, output);
assert_eq!(usage.cache_read_tokens, cache);
assert_eq!(
usage.token_source, expected,
"hint={hint:?}, source={source:?}"
);
}
}
#[tokio::test]
@@ -12751,6 +12853,10 @@ mod tests {
assert_eq!(usage.cache_creation_tokens, 175);
assert_eq!(usage.cache_read_tokens, 24_463);
assert_eq!(usage.output_tokens, 167);
assert_eq!(
usage.token_source,
Some(aether_contracts::UsageTokenSource::Mixed)
);
}
#[tokio::test]
@@ -45,6 +45,7 @@ pub(super) struct StreamFailureReport {
honor_http_failover: bool,
extra_error_fields: Map<String, Value>,
provider_body_json: Option<Value>,
analytics_failure: Option<Value>,
}
#[derive(Serialize)]
@@ -133,6 +134,7 @@ impl StreamFailureReport {
honor_http_failover: _,
mut extra_error_fields,
provider_body_json,
analytics_failure: _,
} = self;
extra_error_fields.insert("type".to_string(), Value::String(error_type));
extra_error_fields.insert("message".to_string(), Value::String(error_message));
@@ -178,6 +180,7 @@ pub(super) fn build_stream_failure_report(
honor_http_failover: false,
extra_error_fields: Map::new(),
provider_body_json: None,
analytics_failure: None,
}
}
@@ -196,6 +199,7 @@ pub(super) fn build_stream_transport_failure_report(
honor_http_failover: false,
extra_error_fields: Map::new(),
provider_body_json: None,
analytics_failure: None,
}
}
@@ -241,6 +245,10 @@ pub(super) fn build_stream_failure_from_execution_error(
honor_http_failover: error.upstream_status.is_some(),
extra_error_fields: error_object,
provider_body_json: None,
analytics_failure: crate::usage::reporting::failure::execution_error_analytics_context(
None, error,
)
.and_then(|context| context.get("analytics_failure").cloned()),
}
}
@@ -271,6 +279,7 @@ pub(super) fn build_stream_failure_from_provider_error_body(
honor_http_failover: true,
extra_error_fields: Map::new(),
provider_body_json: Some(body_json.clone()),
analytics_failure: None,
}
}
@@ -334,6 +343,7 @@ fn build_stream_failure_sync_payload(
let status_code = failure.status_code;
let upstream_status_code = failure.upstream_status_code;
let transport_error = failure.transport_error;
let analytics_failure = failure.analytics_failure.clone();
let (body, client_body) = failure.into_body_jsons();
headers.retain(|name, _| {
!name.eq_ignore_ascii_case("content-encoding")
@@ -355,6 +365,9 @@ fn build_stream_failure_sync_payload(
.or(report_context);
let report_context = report_context.map(|mut context| {
if let Some(object) = context.as_object_mut() {
if let Some(failure) = analytics_failure {
object.insert("analytics_failure".into(), failure);
}
let response_headers = serde_json::to_value(&headers).unwrap_or(Value::Null);
if upstream_status_code.is_some() {
object.insert(
@@ -499,9 +512,11 @@ async fn record_stream_sync_failure(
);
if !matches!(handling, StreamFailureHandling::HonorLocalFailover) || !retrying_next_candidate {
crate::execution_runtime::mark_stream_candidate_watchdog_terminal_started();
let analytics_context =
crate::usage::reporting::failure::sync_analytics_context(report_context, payload);
let report_context_with_diagnostics =
attach_current_request_diagnostics_and_candidate_timing_to_report_context(
report_context,
analytics_context.as_ref(),
payload
.telemetry
.as_ref()
@@ -513,7 +528,9 @@ async fn record_stream_sync_failure(
);
let context_seed = build_terminal_usage_context_seed(
plan,
report_context_with_diagnostics.as_ref().or(report_context),
report_context_with_diagnostics
.as_ref()
.or(analytics_context.as_ref()),
);
let payload_seed = build_sync_terminal_usage_payload_seed(payload);
state
@@ -779,9 +796,13 @@ async fn handle_prefetch_transport_stream_failure(
&& matches!(analysis.decision, LocalFailoverDecision::RetryNextCandidate);
if !retrying_next_candidate {
crate::execution_runtime::mark_stream_candidate_watchdog_terminal_started();
let analytics_context = crate::usage::reporting::failure::sync_analytics_context(
payload.report_context.as_ref(),
&payload,
);
let report_context_with_diagnostics =
attach_current_request_diagnostics_and_candidate_timing_to_report_context(
payload.report_context.as_ref(),
analytics_context.as_ref(),
payload
.telemetry
.as_ref()
@@ -796,7 +817,7 @@ async fn handle_prefetch_transport_stream_failure(
plan,
report_context_with_diagnostics
.as_ref()
.or(payload.report_context.as_ref()),
.or(analytics_context.as_ref()),
);
let payload_seed = build_sync_terminal_usage_payload_seed(&payload);
state
@@ -243,7 +243,10 @@ impl SyncAttemptTerminalGuard {
record_sync_attempt_forced_terminal_state(
self.state.clone(),
self.plan.clone(),
self.report_context.clone(),
crate::usage::reporting::failure::gateway_error_analytics_context(
self.report_context.as_ref(),
error,
),
self.request_diagnostics.clone(),
self.candidate_started_unix_ms,
self.candidate_started_at,
@@ -317,6 +320,16 @@ async fn record_sync_attempt_forced_terminal_state(
let error_message = error_message.into();
let report_context =
attach_request_diagnostics_to_report_context(report_context, request_diagnostics.as_ref());
let report_context = if matches!(usage_event_type, UsageEventType::Cancelled) {
crate::usage::reporting::failure::with_analytics_failure(
report_context.as_ref(),
"unknown",
"finalize",
"request_task_cancelled",
)
} else {
report_context
};
let terminal_unix_ms = current_request_candidate_unix_ms();
let latency_ms = elapsed_ms_since(candidate_started_at);
record_local_request_candidate_status(
@@ -614,15 +627,19 @@ async fn record_sync_terminal_usage(
candidate_started_at: Instant,
candidate_first_byte_elapsed_ms: Option<u64>,
) {
let analytics_context =
crate::usage::reporting::failure::sync_analytics_context(report_context, payload);
let report_context_with_diagnostics =
attach_current_request_diagnostics_and_candidate_start_timing_to_report_context(
report_context,
analytics_context.as_ref(),
candidate_started_at,
candidate_first_byte_elapsed_ms,
);
let context_seed = build_terminal_usage_context_seed(
plan,
report_context_with_diagnostics.as_ref().or(report_context),
report_context_with_diagnostics
.as_ref()
.or(analytics_context.as_ref()),
);
let payload_seed = build_sync_terminal_usage_payload_seed(payload);
state
@@ -2074,37 +2091,38 @@ async fn execute_execution_runtime_sync_impl(
.unwrap_or_else(|| "-".to_string());
let candidate_started_at = Instant::now();
let candidate_started_unix_secs = current_request_candidate_unix_ms();
let _provider_pool_in_flight_guard = match acquire_provider_pool_execution_guard(state, &plan)
.await?
{
ProviderPoolInFlightAdmission::Acquired(guard) => guard,
ProviderPoolInFlightAdmission::Saturated { limit } => {
record_local_runtime_candidate_skip_reason(
state,
trace_id,
"provider_key_concurrency_limit_reached",
);
if let Some(retry_scope) = retry_scope_out.as_deref_mut() {
*retry_scope = AiAttemptRetryScope::Candidate;
let _provider_pool_in_flight_guard =
match acquire_provider_pool_execution_guard(state, &plan, report_context.as_ref()).await? {
ProviderPoolInFlightAdmission::Acquired(guard) => guard,
ProviderPoolInFlightAdmission::Saturated { limit } => {
record_local_runtime_candidate_skip_reason(
state,
trace_id,
"provider_key_concurrency_limit_reached",
);
if let Some(retry_scope) = retry_scope_out.as_deref_mut() {
*retry_scope = AiAttemptRetryScope::Candidate;
}
record_local_request_candidate_status(
state,
&plan,
report_context.as_ref(),
SchedulerRequestCandidateStatusUpdate {
status: RequestCandidateStatus::Skipped,
status_code: Some(StatusCode::TOO_MANY_REQUESTS.as_u16()),
error_type: Some("provider_key_concurrency_limit_reached".to_string()),
error_message: Some(format!(
"provider key concurrency limit reached: {limit}"
)),
latency_ms: Some(0),
started_at_unix_ms: Some(candidate_started_unix_secs),
finished_at_unix_ms: Some(candidate_started_unix_secs),
},
)
.await;
return Ok(None);
}
record_local_request_candidate_status(
state,
&plan,
report_context.as_ref(),
SchedulerRequestCandidateStatusUpdate {
status: RequestCandidateStatus::Skipped,
status_code: Some(StatusCode::TOO_MANY_REQUESTS.as_u16()),
error_type: Some("provider_key_concurrency_limit_reached".to_string()),
error_message: Some(format!("provider key concurrency limit reached: {limit}")),
latency_ms: Some(0),
started_at_unix_ms: Some(candidate_started_unix_secs),
finished_at_unix_ms: Some(candidate_started_unix_secs),
},
)
.await;
return Ok(None);
}
};
};
let lifecycle_seed = build_lifecycle_usage_seed(&plan, report_context.as_ref());
let usage_data = state.usage_lifecycle_data_state().as_ref().clone();
state
@@ -2804,6 +2822,9 @@ async fn execute_execution_runtime_sync_impl(
provider_response_observation.response_headers_observed_at_unix_ms,
&provider_response_observation.request_order_id,
);
if let Some(error) = result.error.as_ref() {
report_context = crate::usage::reporting::failure::execution_error_analytics_context(report_context.as_ref(), error);
}
if result.status_code >= 400 {
apply_local_execution_effect(
state,
@@ -130,6 +130,9 @@ pub(crate) async fn build_transport_error_stop_response(
None => serde_json::Map::new(),
};
request_metadata.insert("transport_error".to_string(), Value::Bool(true));
request_metadata.insert("analytics_failure".into(), json!({
"origin": "transport", "stage": "connect", "reason": "upstream_transport_error", "schema_version": 1,
}));
request_metadata.insert(
"transport_error_type".to_string(),
Value::String(error_type.to_string()),