mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-07 01:47:47 +08:00
chore: update gateway pressure observability
This commit is contained in:
@@ -0,0 +1,218 @@
|
||||
use aether_runtime::{MetricKind, MetricSample};
|
||||
|
||||
pub(crate) fn gateway_allocator_metric_samples() -> Vec<MetricSample> {
|
||||
match allocator_snapshot() {
|
||||
Some(snapshot) => snapshot.to_metric_samples(),
|
||||
None => unavailable_metric_samples(),
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default)]
|
||||
struct AllocatorSnapshot {
|
||||
allocated_bytes: u64,
|
||||
active_bytes: u64,
|
||||
resident_bytes: u64,
|
||||
mapped_bytes: u64,
|
||||
retained_bytes: u64,
|
||||
metadata_bytes: u64,
|
||||
}
|
||||
|
||||
impl AllocatorSnapshot {
|
||||
fn to_metric_samples(self) -> Vec<MetricSample> {
|
||||
vec![
|
||||
gauge(
|
||||
"gateway_allocator_observability_available",
|
||||
"Whether gateway allocator heap metrics were available for this scrape.",
|
||||
1,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_allocated_bytes",
|
||||
"Bytes currently allocated by the gateway allocator.",
|
||||
self.allocated_bytes,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_active_bytes",
|
||||
"Bytes in active pages managed by the gateway allocator.",
|
||||
self.active_bytes,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_resident_bytes",
|
||||
"Bytes resident in physical memory for the gateway allocator.",
|
||||
self.resident_bytes,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_mapped_bytes",
|
||||
"Bytes mapped by the gateway allocator.",
|
||||
self.mapped_bytes,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_retained_bytes",
|
||||
"Bytes retained by the gateway allocator for future use.",
|
||||
self.retained_bytes,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_metadata_bytes",
|
||||
"Bytes used for allocator metadata.",
|
||||
self.metadata_bytes,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_active_to_allocated_basis_points",
|
||||
"Active allocator bytes divided by allocated bytes in basis points.",
|
||||
ratio_basis_points(self.active_bytes, self.allocated_bytes),
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_resident_to_allocated_basis_points",
|
||||
"Resident allocator bytes divided by allocated bytes in basis points.",
|
||||
ratio_basis_points(self.resident_bytes, self.allocated_bytes),
|
||||
),
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
fn unavailable_metric_samples() -> Vec<MetricSample> {
|
||||
vec![
|
||||
gauge(
|
||||
"gateway_allocator_observability_available",
|
||||
"Whether gateway allocator heap metrics were available for this scrape.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_allocated_bytes",
|
||||
"Bytes currently allocated by the gateway allocator.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_active_bytes",
|
||||
"Bytes in active pages managed by the gateway allocator.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_resident_bytes",
|
||||
"Bytes resident in physical memory for the gateway allocator.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_mapped_bytes",
|
||||
"Bytes mapped by the gateway allocator.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_retained_bytes",
|
||||
"Bytes retained by the gateway allocator for future use.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_metadata_bytes",
|
||||
"Bytes used for allocator metadata.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_active_to_allocated_basis_points",
|
||||
"Active allocator bytes divided by allocated bytes in basis points.",
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_allocator_resident_to_allocated_basis_points",
|
||||
"Resident allocator bytes divided by allocated bytes in basis points.",
|
||||
0,
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
|
||||
fn allocator_snapshot() -> Option<AllocatorSnapshot> {
|
||||
refresh_jemalloc_epoch()?;
|
||||
Some(AllocatorSnapshot {
|
||||
allocated_bytes: read_jemalloc_stat("stats.allocated\0")?,
|
||||
active_bytes: read_jemalloc_stat("stats.active\0")?,
|
||||
resident_bytes: read_jemalloc_stat("stats.resident\0")?,
|
||||
mapped_bytes: read_jemalloc_stat("stats.mapped\0")?,
|
||||
retained_bytes: read_jemalloc_stat("stats.retained\0")?,
|
||||
metadata_bytes: read_jemalloc_stat("stats.metadata\0")?,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(not(all(feature = "jemalloc", not(target_env = "msvc"))))]
|
||||
fn allocator_snapshot() -> Option<AllocatorSnapshot> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
|
||||
fn refresh_jemalloc_epoch() -> Option<()> {
|
||||
let mut epoch = 1_u64;
|
||||
let result = unsafe {
|
||||
tikv_jemalloc_sys::mallctl(
|
||||
c"epoch".as_ptr(),
|
||||
std::ptr::null_mut(),
|
||||
std::ptr::null_mut(),
|
||||
(&mut epoch as *mut u64).cast(),
|
||||
std::mem::size_of::<u64>(),
|
||||
)
|
||||
};
|
||||
if result == 0 {
|
||||
Some(())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "jemalloc", not(target_env = "msvc")))]
|
||||
fn read_jemalloc_stat(name: &str) -> Option<u64> {
|
||||
let mut value = 0_usize;
|
||||
let mut size = std::mem::size_of::<usize>();
|
||||
let result = unsafe {
|
||||
tikv_jemalloc_sys::mallctl(
|
||||
name.as_ptr().cast(),
|
||||
(&mut value as *mut usize).cast(),
|
||||
&mut size,
|
||||
std::ptr::null_mut(),
|
||||
0,
|
||||
)
|
||||
};
|
||||
if result == 0 {
|
||||
Some(u64_from_usize(value))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn gauge(name: &'static str, help: &'static str, value: u64) -> MetricSample {
|
||||
MetricSample::new(name, help, MetricKind::Gauge, value)
|
||||
}
|
||||
|
||||
fn ratio_basis_points(numerator: u64, denominator: u64) -> u64 {
|
||||
if denominator == 0 {
|
||||
return 0;
|
||||
}
|
||||
numerator.saturating_mul(10_000) / denominator
|
||||
}
|
||||
|
||||
fn u64_from_usize(value: usize) -> u64 {
|
||||
u64::try_from(value).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{gateway_allocator_metric_samples, ratio_basis_points};
|
||||
|
||||
#[test]
|
||||
fn renders_allocator_metric_samples() {
|
||||
let samples = gateway_allocator_metric_samples();
|
||||
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_allocator_observability_available"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_allocator_allocated_bytes"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_allocator_active_to_allocated_basis_points"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn computes_ratio_basis_points() {
|
||||
assert_eq!(ratio_basis_points(150, 100), 15_000);
|
||||
assert_eq!(ratio_basis_points(1, 0), 0);
|
||||
}
|
||||
}
|
||||
@@ -252,6 +252,12 @@ impl UsageRuntimeAccess for GatewayDataState {
|
||||
GatewayDataState::usage_worker_queue(self)
|
||||
}
|
||||
|
||||
fn usage_worker_should_defer_for_database_pressure(&self) -> bool {
|
||||
self.database_pool_summary()
|
||||
.as_ref()
|
||||
.is_some_and(GatewayDataState::database_pool_summary_under_usage_worker_pressure)
|
||||
}
|
||||
|
||||
async fn body_capture_policy(&self) -> Result<UsageBodyCapturePolicy, DataLayerError> {
|
||||
let value = match GatewayDataState::find_system_config_value(self, REQUEST_RECORD_LEVEL_KEY)
|
||||
.await?
|
||||
|
||||
@@ -170,6 +170,25 @@ impl GatewayDataState {
|
||||
.and_then(|backends| backends.database_pool_summary())
|
||||
}
|
||||
|
||||
pub(crate) async fn postgres_observability_snapshot(
|
||||
&self,
|
||||
) -> Result<Option<aether_data::DatabasePostgresObservabilitySnapshot>, DataLayerError> {
|
||||
match &self.backends {
|
||||
Some(backends) => backends.postgres_observability_snapshot().await,
|
||||
None => Ok(None),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) async fn postgres_activity_groups(
|
||||
&self,
|
||||
limit: i64,
|
||||
) -> Result<Vec<aether_data::DatabasePostgresActivityGroup>, DataLayerError> {
|
||||
match &self.backends {
|
||||
Some(backends) => backends.postgres_activity_groups(limit).await,
|
||||
None => Ok(Vec::new()),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn database_pool_under_maintenance_pressure(&self) -> bool {
|
||||
self.database_pool_summary()
|
||||
.as_ref()
|
||||
@@ -182,6 +201,14 @@ impl GatewayDataState {
|
||||
summary.checked_out > 0 && summary.idle <= Self::maintenance_pool_idle_reserve(summary)
|
||||
}
|
||||
|
||||
pub(crate) fn database_pool_summary_under_usage_worker_pressure(
|
||||
summary: &aether_data::DatabasePoolSummary,
|
||||
) -> bool {
|
||||
summary.checked_out > 0
|
||||
&& (summary.checked_out >= summary.max_connections as usize
|
||||
|| summary.idle <= Self::usage_worker_pool_idle_reserve(summary))
|
||||
}
|
||||
|
||||
pub(crate) fn maintenance_pool_idle_reserve(
|
||||
summary: &aether_data::DatabasePoolSummary,
|
||||
) -> usize {
|
||||
@@ -201,6 +228,13 @@ impl GatewayDataState {
|
||||
ten_percent_ceil.clamp(2, 10).min(max_connections)
|
||||
}
|
||||
|
||||
fn usage_worker_pool_idle_reserve(summary: &aether_data::DatabasePoolSummary) -> usize {
|
||||
if summary.max_connections <= 1 {
|
||||
return 0;
|
||||
}
|
||||
1
|
||||
}
|
||||
|
||||
pub(crate) fn should_defer_maintenance_for_database_pool_pressure(
|
||||
&self,
|
||||
deferred_since: &mut Option<Instant>,
|
||||
|
||||
@@ -97,6 +97,39 @@ fn maintenance_pool_pressure_keeps_idle_reserve_for_foreground_work() {
|
||||
assert!(!GatewayDataState::database_pool_summary_under_maintenance_pressure(&idle));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usage_worker_pool_pressure_only_defers_near_pool_exhaustion() {
|
||||
let comfortable = aether_data::DatabasePoolSummary {
|
||||
driver: DatabaseDriver::Postgres,
|
||||
checked_out: 56,
|
||||
pool_size: 64,
|
||||
idle: 8,
|
||||
max_connections: 64,
|
||||
usage_rate: 87.5,
|
||||
};
|
||||
assert!(!GatewayDataState::database_pool_summary_under_usage_worker_pressure(&comfortable));
|
||||
|
||||
let last_idle_left = aether_data::DatabasePoolSummary {
|
||||
driver: DatabaseDriver::Postgres,
|
||||
checked_out: 63,
|
||||
pool_size: 64,
|
||||
idle: 1,
|
||||
max_connections: 64,
|
||||
usage_rate: 98.4375,
|
||||
};
|
||||
assert!(GatewayDataState::database_pool_summary_under_usage_worker_pressure(&last_idle_left));
|
||||
|
||||
let exhausted = aether_data::DatabasePoolSummary {
|
||||
driver: DatabaseDriver::Postgres,
|
||||
checked_out: 64,
|
||||
pool_size: 64,
|
||||
idle: 0,
|
||||
max_connections: 64,
|
||||
usage_rate: 100.0,
|
||||
};
|
||||
assert!(GatewayDataState::database_pool_summary_under_usage_worker_pressure(&exhausted));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn maintenance_pool_pressure_deferral_has_timeout() {
|
||||
let mut deferred_since = None;
|
||||
|
||||
@@ -1420,9 +1420,10 @@ impl DirectPassthroughFinalizerCore {
|
||||
}
|
||||
if !self.pending_recorded {
|
||||
self.pending_recorded = true;
|
||||
let usage_data = self.state.data.as_ref().clone();
|
||||
self.state
|
||||
.usage_runtime
|
||||
.record_pending_direct(self.state.data.as_ref(), self.lifecycle_seed.clone())
|
||||
.record_pending_direct(&usage_data, self.lifecycle_seed.clone())
|
||||
.await;
|
||||
}
|
||||
self.stream_started_recorded = true;
|
||||
@@ -2016,9 +2017,10 @@ async fn record_stream_pending_lifecycle(
|
||||
stage_trace: &mut RequestStageTrace,
|
||||
) {
|
||||
let usage_pending_started_at = Instant::now();
|
||||
let usage_data = state.data.as_ref().clone();
|
||||
state
|
||||
.usage_runtime
|
||||
.record_pending_direct(state.data.as_ref(), lifecycle_seed.clone())
|
||||
.record_pending_direct(&usage_data, lifecycle_seed.clone())
|
||||
.await;
|
||||
observe_gateway_stage_trace_ms(
|
||||
stage_trace,
|
||||
@@ -5250,10 +5252,11 @@ async fn execute_stream_from_frame_stream(
|
||||
telemetry.as_ref(),
|
||||
&mut usage_stream_telemetry,
|
||||
) {
|
||||
let usage_data = state_for_report.data.as_ref().clone();
|
||||
state_for_report
|
||||
.usage_runtime
|
||||
.record_stream_started_direct(
|
||||
state_for_report.data.as_ref(),
|
||||
&usage_data,
|
||||
&lifecycle_seed_for_report,
|
||||
status_code,
|
||||
usage_stream_telemetry.as_ref(),
|
||||
@@ -5458,10 +5461,11 @@ async fn execute_stream_from_frame_stream(
|
||||
);
|
||||
if should_refresh_stream_usage {
|
||||
if usage_frame_telemetry.ttfb_ms.is_some() {
|
||||
let usage_data = state_for_report.data.as_ref().clone();
|
||||
state_for_report
|
||||
.usage_runtime
|
||||
.record_stream_started_direct(
|
||||
state_for_report.data.as_ref(),
|
||||
&usage_data,
|
||||
&lifecycle_seed_for_report,
|
||||
status_code,
|
||||
Some(&usage_frame_telemetry),
|
||||
|
||||
@@ -1538,9 +1538,10 @@ async fn execute_execution_runtime_sync_impl(
|
||||
.unwrap_or_else(|| "-".to_string());
|
||||
let candidate_started_unix_secs = current_request_candidate_unix_ms();
|
||||
let lifecycle_seed = build_lifecycle_usage_seed(&plan, report_context.as_ref());
|
||||
let usage_data = state.data.as_ref().clone();
|
||||
state
|
||||
.usage_runtime
|
||||
.record_pending_direct(state.data.as_ref(), lifecycle_seed)
|
||||
.record_pending_direct(&usage_data, lifecycle_seed)
|
||||
.await;
|
||||
record_local_request_candidate_status(
|
||||
state,
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
|
||||
mod admin_api;
|
||||
mod ai_serving;
|
||||
mod allocator_metrics;
|
||||
mod api;
|
||||
mod async_task;
|
||||
mod audit;
|
||||
@@ -58,6 +59,7 @@ mod model_fetch;
|
||||
mod oauth;
|
||||
mod orchestration;
|
||||
mod privacy;
|
||||
mod process_metrics;
|
||||
mod provider_key_auth;
|
||||
mod provider_pool_demand;
|
||||
pub(crate) use aether_provider_transport as provider_transport;
|
||||
@@ -76,6 +78,7 @@ mod system_features;
|
||||
mod task_runtime;
|
||||
#[cfg(feature = "testkit")]
|
||||
pub mod testkit;
|
||||
mod tokio_metrics;
|
||||
mod tunnel;
|
||||
mod upstream_admission;
|
||||
mod usage;
|
||||
|
||||
+144
-22
@@ -250,6 +250,8 @@ const AUTO_USAGE_QUEUE_WORKERS_MIN: usize = 2;
|
||||
const AUTO_USAGE_QUEUE_WORKERS_REQUESTS_PER_WORKER: usize = 128;
|
||||
const AUTO_USAGE_QUEUE_WORKERS_DB_SHARE_ALL: usize = 4;
|
||||
const AUTO_USAGE_QUEUE_WORKERS_DB_SHARE_BACKGROUND: usize = 2;
|
||||
const AUTO_USAGE_WORKER_RECORD_DB_SHARE_ALL: usize = 8;
|
||||
const AUTO_USAGE_WORKER_RECORD_DB_SHARE_BACKGROUND: usize = 4;
|
||||
const MAX_USAGE_QUEUE_WORKERS: usize = 64;
|
||||
const DEFAULT_GATEWAY_LISTEN_BACKLOG: i32 = 65_535;
|
||||
const MIN_GATEWAY_LISTEN_BACKLOG: i32 = 128;
|
||||
@@ -325,6 +327,29 @@ fn usage_queue_worker_database_cap(
|
||||
.clamp(1, MAX_USAGE_QUEUE_WORKERS)
|
||||
}
|
||||
|
||||
fn usage_worker_record_concurrency_database_cap(
|
||||
node_role: NodeRoleArg,
|
||||
database: Option<&SqlDatabaseConfig>,
|
||||
) -> Option<usize> {
|
||||
let database = database?;
|
||||
if database.driver == DatabaseDriver::Sqlite {
|
||||
return Some(1);
|
||||
}
|
||||
|
||||
let divisor = if matches!(node_role, NodeRoleArg::Background) {
|
||||
AUTO_USAGE_WORKER_RECORD_DB_SHARE_BACKGROUND
|
||||
} else {
|
||||
AUTO_USAGE_WORKER_RECORD_DB_SHARE_ALL
|
||||
};
|
||||
let max_connections = database.pool.max_connections.max(1) as usize;
|
||||
Some(
|
||||
max_connections
|
||||
.checked_div(divisor.max(1))
|
||||
.unwrap_or(1)
|
||||
.clamp(1, MAX_USAGE_QUEUE_WORKERS),
|
||||
)
|
||||
}
|
||||
|
||||
fn automatic_usage_queue_workers_for_parallelism(
|
||||
parallelism: usize,
|
||||
node_role: NodeRoleArg,
|
||||
@@ -631,10 +656,19 @@ struct GatewayUsageArgs {
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_QUEUE_WORKER_MAX_COUNT",
|
||||
value_name = "COUNT"
|
||||
value_name = "COUNT",
|
||||
default_value = "32"
|
||||
)]
|
||||
queue_worker_max_count: Option<usize>,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_WORKER_RECORD_CONCURRENCY_LIMIT",
|
||||
value_name = "COUNT",
|
||||
default_value = "32"
|
||||
)]
|
||||
worker_record_concurrency_limit: Option<usize>,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_QUEUE_WORKER_SCALE_INTERVAL_MS",
|
||||
@@ -680,7 +714,7 @@ struct GatewayUsageArgs {
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_QUEUE_BATCH_SIZE",
|
||||
default_value_t = 500
|
||||
default_value_t = 128
|
||||
)]
|
||||
queue_batch_size: usize,
|
||||
|
||||
@@ -694,14 +728,14 @@ struct GatewayUsageArgs {
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_QUEUE_RECLAIM_IDLE_MS",
|
||||
default_value_t = 30_000
|
||||
default_value_t = 60_000
|
||||
)]
|
||||
queue_reclaim_idle_ms: u64,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_QUEUE_RECLAIM_COUNT",
|
||||
default_value_t = 500
|
||||
default_value_t = 128
|
||||
)]
|
||||
queue_reclaim_count: usize,
|
||||
|
||||
@@ -715,21 +749,28 @@ struct GatewayUsageArgs {
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_TERMINAL_ENQUEUE_MAX_IN_FLIGHT",
|
||||
default_value_t = 256
|
||||
default_value_t = 1_024
|
||||
)]
|
||||
terminal_enqueue_max_in_flight: u64,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_LIFECYCLE_ENQUEUE_MAX_IN_FLIGHT",
|
||||
default_value_t = 128
|
||||
default_value_t = 512
|
||||
)]
|
||||
lifecycle_enqueue_max_in_flight: u64,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_LIFECYCLE_ENQUEUE_DELAY_MS",
|
||||
default_value_t = 1_000
|
||||
)]
|
||||
lifecycle_enqueue_delay_ms: u64,
|
||||
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_RETRY_DEFERRED_LIFECYCLE_EVENTS",
|
||||
default_value_t = false
|
||||
default_value_t = true
|
||||
)]
|
||||
retry_deferred_lifecycle_events: bool,
|
||||
|
||||
@@ -743,7 +784,7 @@ struct GatewayUsageArgs {
|
||||
#[arg(
|
||||
long,
|
||||
env = "AETHER_GATEWAY_USAGE_ENQUEUE_RETRY_WORKERS",
|
||||
default_value_t = 4
|
||||
default_value_t = 8
|
||||
)]
|
||||
enqueue_retry_workers: usize,
|
||||
|
||||
@@ -791,10 +832,12 @@ impl GatewayUsageArgs {
|
||||
worker_count: usize,
|
||||
) -> usize {
|
||||
if !self.queue_worker_autoscale_enabled {
|
||||
return worker_count.max(1).min(MAX_USAGE_QUEUE_WORKERS);
|
||||
return worker_count.clamp(1, MAX_USAGE_QUEUE_WORKERS);
|
||||
}
|
||||
self.queue_worker_max_count
|
||||
.unwrap_or_else(|| usage_queue_worker_database_cap(node_role, database))
|
||||
.max(1)
|
||||
.min(usage_queue_worker_database_cap(node_role, database))
|
||||
.clamp(worker_count.max(1), MAX_USAGE_QUEUE_WORKERS)
|
||||
}
|
||||
|
||||
@@ -813,7 +856,39 @@ impl GatewayUsageArgs {
|
||||
Some(worker_max_count.clamp(1, MAX_USAGE_QUEUE_WORKERS))
|
||||
}
|
||||
|
||||
fn to_config(&self, worker_count: usize, worker_max_count: usize) -> UsageRuntimeConfig {
|
||||
fn effective_worker_record_concurrency_limit(
|
||||
&self,
|
||||
node_role: NodeRoleArg,
|
||||
database: Option<&SqlDatabaseConfig>,
|
||||
) -> Option<usize> {
|
||||
if let Some(limit) = self.worker_record_concurrency_limit {
|
||||
if limit == 0 {
|
||||
return None;
|
||||
}
|
||||
return Some(
|
||||
limit
|
||||
.min(MAX_USAGE_QUEUE_WORKERS)
|
||||
.min(
|
||||
usage_worker_record_concurrency_database_cap(node_role, database)
|
||||
.unwrap_or(MAX_USAGE_QUEUE_WORKERS),
|
||||
)
|
||||
.max(1),
|
||||
);
|
||||
}
|
||||
if !node_role.spawns_background_tasks()
|
||||
|| (!self.queue_terminal_events && !self.queue_lifecycle_events)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
usage_worker_record_concurrency_database_cap(node_role, database)
|
||||
}
|
||||
|
||||
fn to_config(
|
||||
&self,
|
||||
worker_count: usize,
|
||||
worker_max_count: usize,
|
||||
worker_record_concurrency_limit: Option<usize>,
|
||||
) -> UsageRuntimeConfig {
|
||||
UsageRuntimeConfig {
|
||||
enabled: true,
|
||||
queue_terminal_events: self.queue_terminal_events,
|
||||
@@ -821,6 +896,7 @@ impl GatewayUsageArgs {
|
||||
worker_count: worker_count.clamp(1, MAX_USAGE_QUEUE_WORKERS),
|
||||
worker_autoscale_enabled: self.queue_worker_autoscale_enabled,
|
||||
worker_max_count: worker_max_count.clamp(worker_count.max(1), MAX_USAGE_QUEUE_WORKERS),
|
||||
worker_record_concurrency_limit,
|
||||
worker_scale_interval_ms: self.queue_worker_scale_interval_ms.max(1),
|
||||
worker_idle_scale_down_ticks: self.queue_worker_idle_scale_down_ticks.max(1),
|
||||
stream_key: self.queue_stream_key.trim().to_string(),
|
||||
@@ -834,6 +910,7 @@ impl GatewayUsageArgs {
|
||||
reclaim_interval_ms: self.queue_reclaim_interval_ms.max(1),
|
||||
terminal_enqueue_max_in_flight: self.terminal_enqueue_max_in_flight.max(1),
|
||||
lifecycle_enqueue_max_in_flight: self.lifecycle_enqueue_max_in_flight.max(1),
|
||||
lifecycle_enqueue_delay_ms: self.lifecycle_enqueue_delay_ms,
|
||||
retry_deferred_lifecycle_events: self.retry_deferred_lifecycle_events,
|
||||
enqueue_retry_buffer_capacity: self.enqueue_retry_buffer_capacity.max(1),
|
||||
enqueue_retry_workers: self.enqueue_retry_workers.clamp(1, 64),
|
||||
@@ -1593,9 +1670,14 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
sql_database_config.as_ref(),
|
||||
usage_queue_workers,
|
||||
);
|
||||
let usage_config = args
|
||||
let usage_worker_record_concurrency_limit = args
|
||||
.usage
|
||||
.to_config(usage_queue_workers, usage_queue_worker_max_count);
|
||||
.effective_worker_record_concurrency_limit(args.node_role, sql_database_config.as_ref());
|
||||
let usage_config = args.usage.to_config(
|
||||
usage_queue_workers,
|
||||
usage_queue_worker_max_count,
|
||||
usage_worker_record_concurrency_limit,
|
||||
);
|
||||
let usage_blocking_stream_lanes = args.usage.runtime_state_blocking_stream_lanes(
|
||||
args.node_role,
|
||||
sql_database_config.as_ref(),
|
||||
@@ -1634,6 +1716,9 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
usage_queue_workers = usage_config.worker_count,
|
||||
usage_queue_worker_autoscale_enabled = usage_config.worker_autoscale_enabled,
|
||||
usage_queue_worker_max_count = usage_config.worker_max_count,
|
||||
usage_worker_record_concurrency_limit = usage_config
|
||||
.worker_record_concurrency_limit
|
||||
.unwrap_or_default(),
|
||||
usage_queue_request_concurrency_hint =
|
||||
usage_queue_request_concurrency_hint.unwrap_or_default(),
|
||||
usage_queue_request_concurrency_hint_source = if usage_queue_request_concurrency_hint.is_some() {
|
||||
@@ -1672,6 +1757,9 @@ async fn run() -> Result<(), Box<dyn std::error::Error>> {
|
||||
},
|
||||
usage_queue_worker_autoscale_enabled = usage_config.worker_autoscale_enabled,
|
||||
usage_queue_worker_max_count = usage_config.worker_max_count,
|
||||
usage_worker_record_concurrency_limit = usage_config
|
||||
.worker_record_concurrency_limit
|
||||
.unwrap_or_default(),
|
||||
usage_queue_request_concurrency_hint =
|
||||
usage_queue_request_concurrency_hint.unwrap_or_default(),
|
||||
usage_queue_request_concurrency_hint_source =
|
||||
@@ -2292,23 +2380,25 @@ mod tests {
|
||||
queue_lifecycle_events: true,
|
||||
queue_workers: Some(4),
|
||||
queue_worker_autoscale_enabled: true,
|
||||
queue_worker_max_count: None,
|
||||
queue_worker_max_count: Some(32),
|
||||
worker_record_concurrency_limit: Some(32),
|
||||
queue_worker_scale_interval_ms: 1_000,
|
||||
queue_worker_idle_scale_down_ticks: 30,
|
||||
queue_stream_key: "usage:events".to_string(),
|
||||
queue_group: "usage_consumers".to_string(),
|
||||
queue_dlq_stream_key: "usage:events:dlq".to_string(),
|
||||
queue_stream_maxlen: 200_000,
|
||||
queue_batch_size: 500,
|
||||
queue_batch_size: 128,
|
||||
queue_block_ms: 500,
|
||||
queue_reclaim_idle_ms: 30_000,
|
||||
queue_reclaim_count: 500,
|
||||
queue_reclaim_idle_ms: 60_000,
|
||||
queue_reclaim_count: 128,
|
||||
queue_reclaim_interval_ms: 5_000,
|
||||
terminal_enqueue_max_in_flight: 256,
|
||||
lifecycle_enqueue_max_in_flight: 128,
|
||||
retry_deferred_lifecycle_events: false,
|
||||
terminal_enqueue_max_in_flight: 1_024,
|
||||
lifecycle_enqueue_max_in_flight: 512,
|
||||
lifecycle_enqueue_delay_ms: 1_000,
|
||||
retry_deferred_lifecycle_events: true,
|
||||
enqueue_retry_buffer_capacity: 131_072,
|
||||
enqueue_retry_workers: 4,
|
||||
enqueue_retry_workers: 8,
|
||||
enqueue_retry_initial_backoff_ms: 3_000,
|
||||
enqueue_retry_max_backoff_ms: 10_000,
|
||||
},
|
||||
@@ -2526,7 +2616,7 @@ mod tests {
|
||||
);
|
||||
|
||||
assert_eq!(workers, 64);
|
||||
assert_eq!(args.usage.to_config(workers, 64).worker_count, 64);
|
||||
assert_eq!(args.usage.to_config(workers, 64, Some(8)).worker_count, 64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2566,7 +2656,7 @@ mod tests {
|
||||
let mut args = test_args();
|
||||
args.usage.queue_workers = None;
|
||||
args.usage.queue_worker_max_count = Some(32);
|
||||
let database = test_database(DatabaseDriver::Postgres, 100);
|
||||
let database = test_database(DatabaseDriver::Postgres, 200);
|
||||
|
||||
let workers =
|
||||
args.usage
|
||||
@@ -2579,6 +2669,38 @@ mod tests {
|
||||
assert_eq!(max_workers, 32);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gateway_usage_worker_record_concurrency_defaults_to_pool_reserve_share() {
|
||||
let args = test_args();
|
||||
let database = test_database(DatabaseDriver::Postgres, 64);
|
||||
|
||||
assert_eq!(
|
||||
args.usage
|
||||
.effective_worker_record_concurrency_limit(NodeRoleArg::All, Some(&database)),
|
||||
Some(8)
|
||||
);
|
||||
assert_eq!(
|
||||
args.usage.effective_worker_record_concurrency_limit(
|
||||
NodeRoleArg::Background,
|
||||
Some(&database)
|
||||
),
|
||||
Some(16)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gateway_usage_worker_record_concurrency_can_be_explicitly_disabled() {
|
||||
let mut args = test_args();
|
||||
args.usage.worker_record_concurrency_limit = Some(0);
|
||||
let database = test_database(DatabaseDriver::Postgres, 64);
|
||||
|
||||
assert_eq!(
|
||||
args.usage
|
||||
.effective_worker_record_concurrency_limit(NodeRoleArg::All, Some(&database)),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gateway_usage_queue_blocking_stream_lanes_only_expand_when_worker_can_spawn() {
|
||||
let database = test_database(DatabaseDriver::Postgres, 100);
|
||||
|
||||
@@ -30,5 +30,6 @@ pub(crate) use runtime::{
|
||||
ProxyUpgradeRolloutCancelSummary, ProxyUpgradeRolloutConflictClearSummary,
|
||||
ProxyUpgradeRolloutNodeActionSummary, ProxyUpgradeRolloutProbeConfig,
|
||||
ProxyUpgradeRolloutSkippedRestoreSummary, ProxyUpgradeRolloutStatus,
|
||||
ProxyUpgradeRolloutTrackedNodeState,
|
||||
ProxyUpgradeRolloutTrackedNodeState, UsageCounterFlushRuntimeMetrics,
|
||||
UsageCounterFlushWorkerConfig,
|
||||
};
|
||||
|
||||
@@ -116,6 +116,9 @@ pub(crate) use usage_cleanup::{
|
||||
preview_manual_usage_cleanup, ManualUsageCleanupMode, ManualUsageCleanupOptions,
|
||||
};
|
||||
use usage_counter_flush::*;
|
||||
pub(crate) use usage_counter_flush::{
|
||||
UsageCounterFlushRuntimeMetrics, UsageCounterFlushWorkerConfig,
|
||||
};
|
||||
use wallet_daily_usage::*;
|
||||
pub(crate) use workers::*;
|
||||
|
||||
@@ -136,12 +139,6 @@ const PROXY_NODE_STALE_MIN_GRACE_SECS: u64 = 15;
|
||||
const PROXY_NODE_STALE_MISSED_HEARTBEATS: u64 = 3;
|
||||
const POOL_MONITOR_INTERVAL: Duration = Duration::from_secs(5 * 60);
|
||||
const OAUTH_TOKEN_REFRESH_INTERVAL: Duration = Duration::from_secs(60);
|
||||
const USAGE_COUNTER_FLUSH_INTERVAL: Duration = Duration::from_secs(1);
|
||||
const USAGE_COUNTER_FLUSH_BATCH_SIZE: usize = 1_000;
|
||||
const USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT: usize = 20;
|
||||
const USAGE_COUNTER_DELTA_CLEANUP_INTERVAL: Duration = Duration::from_secs(60);
|
||||
const USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE: usize = 5_000;
|
||||
const USAGE_COUNTER_DELTA_RETENTION_SECS: u64 = 7 * 24 * 60 * 60;
|
||||
const PROVIDER_CHECKIN_CONCURRENCY: usize = 3;
|
||||
const PROVIDER_QUOTA_ALERT_CONCURRENCY: usize = 3;
|
||||
const PROVIDER_QUOTA_ALERT_INTERVAL: Duration = Duration::from_secs(5);
|
||||
|
||||
@@ -1,6 +1,300 @@
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
use std::time::Duration;
|
||||
|
||||
use crate::data::GatewayDataState;
|
||||
use aether_data::DataLayerError;
|
||||
use aether_data_contracts::repository::usage::UsageCounterFlushSummary;
|
||||
use aether_runtime::{MetricKind, MetricLabel, MetricSample};
|
||||
|
||||
const USAGE_COUNTER_FLUSH_INTERVAL_MS_ENV: &str = "AETHER_GATEWAY_USAGE_COUNTER_FLUSH_INTERVAL_MS";
|
||||
const USAGE_COUNTER_FLUSH_BATCH_SIZE_ENV: &str = "AETHER_GATEWAY_USAGE_COUNTER_FLUSH_BATCH_SIZE";
|
||||
const USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT_ENV: &str =
|
||||
"AETHER_GATEWAY_USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT";
|
||||
const USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS_ENV: &str =
|
||||
"AETHER_GATEWAY_USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS";
|
||||
const USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE_ENV: &str =
|
||||
"AETHER_GATEWAY_USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE";
|
||||
const USAGE_COUNTER_DELTA_RETENTION_SECS_ENV: &str =
|
||||
"AETHER_GATEWAY_USAGE_COUNTER_DELTA_RETENTION_SECS";
|
||||
|
||||
const DEFAULT_USAGE_COUNTER_FLUSH_INTERVAL_MS: u64 = 1_000;
|
||||
const DEFAULT_USAGE_COUNTER_FLUSH_BATCH_SIZE: usize = 1_000;
|
||||
const DEFAULT_USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT: usize = 20;
|
||||
const DEFAULT_USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS: u64 = 60_000;
|
||||
const DEFAULT_USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE: usize = 5_000;
|
||||
const DEFAULT_USAGE_COUNTER_DELTA_RETENTION_SECS: u64 = 7 * 24 * 60 * 60;
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub(crate) struct UsageCounterFlushWorkerConfig {
|
||||
pub(crate) flush_interval: Duration,
|
||||
pub(crate) flush_batch_size: usize,
|
||||
pub(crate) flush_catch_up_burst_limit: usize,
|
||||
pub(crate) cleanup_interval: Duration,
|
||||
pub(crate) cleanup_batch_size: usize,
|
||||
pub(crate) delta_retention_secs: u64,
|
||||
}
|
||||
|
||||
impl Default for UsageCounterFlushWorkerConfig {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
flush_interval: Duration::from_millis(DEFAULT_USAGE_COUNTER_FLUSH_INTERVAL_MS),
|
||||
flush_batch_size: DEFAULT_USAGE_COUNTER_FLUSH_BATCH_SIZE,
|
||||
flush_catch_up_burst_limit: DEFAULT_USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT,
|
||||
cleanup_interval: Duration::from_millis(
|
||||
DEFAULT_USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS,
|
||||
),
|
||||
cleanup_batch_size: DEFAULT_USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE,
|
||||
delta_retention_secs: DEFAULT_USAGE_COUNTER_DELTA_RETENTION_SECS,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl UsageCounterFlushWorkerConfig {
|
||||
pub(crate) fn from_env() -> Self {
|
||||
let defaults = Self::default();
|
||||
Self {
|
||||
flush_interval: Duration::from_millis(env_u64(
|
||||
USAGE_COUNTER_FLUSH_INTERVAL_MS_ENV,
|
||||
duration_millis_u64(defaults.flush_interval),
|
||||
)),
|
||||
flush_batch_size: env_usize(
|
||||
USAGE_COUNTER_FLUSH_BATCH_SIZE_ENV,
|
||||
defaults.flush_batch_size,
|
||||
),
|
||||
flush_catch_up_burst_limit: env_usize(
|
||||
USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT_ENV,
|
||||
defaults.flush_catch_up_burst_limit,
|
||||
),
|
||||
cleanup_interval: Duration::from_millis(env_u64(
|
||||
USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS_ENV,
|
||||
duration_millis_u64(defaults.cleanup_interval),
|
||||
)),
|
||||
cleanup_batch_size: env_usize(
|
||||
USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE_ENV,
|
||||
defaults.cleanup_batch_size,
|
||||
),
|
||||
delta_retention_secs: env_u64(
|
||||
USAGE_COUNTER_DELTA_RETENTION_SECS_ENV,
|
||||
defaults.delta_retention_secs,
|
||||
),
|
||||
}
|
||||
.normalized()
|
||||
}
|
||||
|
||||
fn normalized(mut self) -> Self {
|
||||
self.flush_interval = self.flush_interval.max(Duration::from_millis(1));
|
||||
self.flush_batch_size = self.flush_batch_size.max(1);
|
||||
self.flush_catch_up_burst_limit = self.flush_catch_up_burst_limit.max(1);
|
||||
self.cleanup_interval = self.cleanup_interval.max(Duration::from_millis(1));
|
||||
self.cleanup_batch_size = self.cleanup_batch_size.max(1);
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct UsageCounterFlushRuntimeMetrics {
|
||||
flush_batches_total: AtomicU64,
|
||||
flush_empty_batches_total: AtomicU64,
|
||||
flush_rows_claimed_total: AtomicU64,
|
||||
flush_api_key_targets_total: AtomicU64,
|
||||
flush_provider_api_key_targets_total: AtomicU64,
|
||||
flush_model_targets_total: AtomicU64,
|
||||
flush_provider_monthly_targets_total: AtomicU64,
|
||||
flush_proxy_node_targets_total: AtomicU64,
|
||||
flush_management_token_targets_total: AtomicU64,
|
||||
flush_api_key_last_used_targets_total: AtomicU64,
|
||||
flush_failed_batches_total: AtomicU64,
|
||||
flush_deferred_total: AtomicU64,
|
||||
cleanup_batches_total: AtomicU64,
|
||||
cleanup_rows_total: AtomicU64,
|
||||
cleanup_failed_batches_total: AtomicU64,
|
||||
cleanup_deferred_total: AtomicU64,
|
||||
}
|
||||
|
||||
impl UsageCounterFlushRuntimeMetrics {
|
||||
pub(crate) fn record_flush_success(&self, summary: &UsageCounterFlushSummary) {
|
||||
if summary.rows_claimed > 0 {
|
||||
self.flush_batches_total.fetch_add(1, Ordering::AcqRel);
|
||||
} else {
|
||||
self.flush_empty_batches_total
|
||||
.fetch_add(1, Ordering::AcqRel);
|
||||
}
|
||||
self.flush_rows_claimed_total
|
||||
.fetch_add(usize_to_u64(summary.rows_claimed), Ordering::AcqRel);
|
||||
self.flush_api_key_targets_total
|
||||
.fetch_add(usize_to_u64(summary.api_key_targets), Ordering::AcqRel);
|
||||
self.flush_provider_api_key_targets_total.fetch_add(
|
||||
usize_to_u64(summary.provider_api_key_targets),
|
||||
Ordering::AcqRel,
|
||||
);
|
||||
self.flush_model_targets_total
|
||||
.fetch_add(usize_to_u64(summary.model_targets), Ordering::AcqRel);
|
||||
self.flush_provider_monthly_targets_total.fetch_add(
|
||||
usize_to_u64(summary.provider_monthly_targets),
|
||||
Ordering::AcqRel,
|
||||
);
|
||||
self.flush_proxy_node_targets_total
|
||||
.fetch_add(usize_to_u64(summary.proxy_node_targets), Ordering::AcqRel);
|
||||
self.flush_management_token_targets_total.fetch_add(
|
||||
usize_to_u64(summary.management_token_targets),
|
||||
Ordering::AcqRel,
|
||||
);
|
||||
self.flush_api_key_last_used_targets_total.fetch_add(
|
||||
usize_to_u64(summary.api_key_last_used_targets),
|
||||
Ordering::AcqRel,
|
||||
);
|
||||
}
|
||||
|
||||
pub(crate) fn record_flush_failed(&self) {
|
||||
self.flush_failed_batches_total
|
||||
.fetch_add(1, Ordering::AcqRel);
|
||||
}
|
||||
|
||||
pub(crate) fn record_flush_deferred(&self) {
|
||||
self.flush_deferred_total.fetch_add(1, Ordering::AcqRel);
|
||||
}
|
||||
|
||||
pub(crate) fn record_cleanup_success(&self, rows_deleted: usize) {
|
||||
self.cleanup_batches_total.fetch_add(1, Ordering::AcqRel);
|
||||
self.cleanup_rows_total
|
||||
.fetch_add(usize_to_u64(rows_deleted), Ordering::AcqRel);
|
||||
}
|
||||
|
||||
pub(crate) fn record_cleanup_failed(&self) {
|
||||
self.cleanup_failed_batches_total
|
||||
.fetch_add(1, Ordering::AcqRel);
|
||||
}
|
||||
|
||||
pub(crate) fn record_cleanup_deferred(&self) {
|
||||
self.cleanup_deferred_total.fetch_add(1, Ordering::AcqRel);
|
||||
}
|
||||
|
||||
pub(crate) fn metric_samples(&self) -> Vec<MetricSample> {
|
||||
let mut samples = vec![
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_flush_batches_total",
|
||||
"Total non-empty usage counter outbox flush batches completed by this gateway process.",
|
||||
MetricKind::Counter,
|
||||
self.flush_batches_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_flush_empty_batches_total",
|
||||
"Total successful usage counter outbox flush checks that found no rows.",
|
||||
MetricKind::Counter,
|
||||
self.flush_empty_batches_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_flush_rows_claimed_total",
|
||||
"Total usage counter outbox rows claimed and processed by this gateway process.",
|
||||
MetricKind::Counter,
|
||||
self.flush_rows_claimed_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_flush_failed_batches_total",
|
||||
"Total usage counter outbox flush batches that failed in this gateway process.",
|
||||
MetricKind::Counter,
|
||||
self.flush_failed_batches_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_flush_deferred_total",
|
||||
"Total usage counter outbox flush ticks deferred because of database pool pressure.",
|
||||
MetricKind::Counter,
|
||||
self.flush_deferred_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_cleanup_batches_total",
|
||||
"Total usage counter outbox cleanup batches completed by this gateway process.",
|
||||
MetricKind::Counter,
|
||||
self.cleanup_batches_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_cleanup_rows_total",
|
||||
"Total processed usage counter outbox rows deleted by cleanup in this gateway process.",
|
||||
MetricKind::Counter,
|
||||
self.cleanup_rows_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_cleanup_failed_batches_total",
|
||||
"Total usage counter outbox cleanup batches that failed in this gateway process.",
|
||||
MetricKind::Counter,
|
||||
self.cleanup_failed_batches_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_cleanup_deferred_total",
|
||||
"Total usage counter outbox cleanup ticks deferred because of database pool pressure.",
|
||||
MetricKind::Counter,
|
||||
self.cleanup_deferred_total.load(Ordering::Acquire),
|
||||
),
|
||||
];
|
||||
for (kind, value) in [
|
||||
(
|
||||
"api_key",
|
||||
self.flush_api_key_targets_total.load(Ordering::Acquire),
|
||||
),
|
||||
(
|
||||
"provider_api_key",
|
||||
self.flush_provider_api_key_targets_total
|
||||
.load(Ordering::Acquire),
|
||||
),
|
||||
(
|
||||
"model",
|
||||
self.flush_model_targets_total.load(Ordering::Acquire),
|
||||
),
|
||||
(
|
||||
"provider_monthly",
|
||||
self.flush_provider_monthly_targets_total
|
||||
.load(Ordering::Acquire),
|
||||
),
|
||||
(
|
||||
"proxy_node",
|
||||
self.flush_proxy_node_targets_total.load(Ordering::Acquire),
|
||||
),
|
||||
(
|
||||
"management_token",
|
||||
self.flush_management_token_targets_total
|
||||
.load(Ordering::Acquire),
|
||||
),
|
||||
(
|
||||
"api_key_last_used",
|
||||
self.flush_api_key_last_used_targets_total
|
||||
.load(Ordering::Acquire),
|
||||
),
|
||||
] {
|
||||
samples.push(
|
||||
MetricSample::new(
|
||||
"usage_counter_outbox_flush_targets_total",
|
||||
"Total usage counter outbox aggregate targets updated by target kind.",
|
||||
MetricKind::Counter,
|
||||
value,
|
||||
)
|
||||
.with_labels(vec![MetricLabel::new("kind", kind)]),
|
||||
);
|
||||
}
|
||||
samples
|
||||
}
|
||||
}
|
||||
|
||||
fn usize_to_u64(value: usize) -> u64 {
|
||||
u64::try_from(value).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
fn duration_millis_u64(duration: Duration) -> u64 {
|
||||
u64::try_from(duration.as_millis()).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
fn env_u64(name: &str, default_value: u64) -> u64 {
|
||||
std::env::var(name)
|
||||
.ok()
|
||||
.and_then(|value| value.trim().parse::<u64>().ok())
|
||||
.unwrap_or(default_value)
|
||||
}
|
||||
|
||||
fn env_usize(name: &str, default_value: usize) -> usize {
|
||||
std::env::var(name)
|
||||
.ok()
|
||||
.and_then(|value| value.trim().parse::<usize>().ok())
|
||||
.unwrap_or(default_value)
|
||||
}
|
||||
|
||||
pub(crate) async fn run_usage_counter_flush_once(
|
||||
data: &GatewayDataState,
|
||||
@@ -19,3 +313,186 @@ pub(crate) async fn cleanup_processed_usage_counter_deltas_once(
|
||||
data.cleanup_processed_usage_counter_deltas(cutoff, batch_size)
|
||||
.await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::sync::{Mutex, OnceLock};
|
||||
use std::time::Duration;
|
||||
|
||||
use super::{
|
||||
UsageCounterFlushRuntimeMetrics, UsageCounterFlushSummary, UsageCounterFlushWorkerConfig,
|
||||
USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE_ENV, USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS_ENV,
|
||||
USAGE_COUNTER_DELTA_RETENTION_SECS_ENV, USAGE_COUNTER_FLUSH_BATCH_SIZE_ENV,
|
||||
USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT_ENV, USAGE_COUNTER_FLUSH_INTERVAL_MS_ENV,
|
||||
};
|
||||
|
||||
const CONFIG_ENV_KEYS: &[&str] = &[
|
||||
USAGE_COUNTER_FLUSH_INTERVAL_MS_ENV,
|
||||
USAGE_COUNTER_FLUSH_BATCH_SIZE_ENV,
|
||||
USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT_ENV,
|
||||
USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS_ENV,
|
||||
USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE_ENV,
|
||||
USAGE_COUNTER_DELTA_RETENTION_SECS_ENV,
|
||||
];
|
||||
|
||||
fn env_lock() -> &'static Mutex<()> {
|
||||
static LOCK: OnceLock<Mutex<()>> = OnceLock::new();
|
||||
LOCK.get_or_init(|| Mutex::new(()))
|
||||
}
|
||||
|
||||
struct EnvVarGuard {
|
||||
values: Vec<(&'static str, Option<String>)>,
|
||||
}
|
||||
|
||||
impl EnvVarGuard {
|
||||
fn new(keys: &[&'static str]) -> Self {
|
||||
let values = keys
|
||||
.iter()
|
||||
.map(|key| (*key, std::env::var(key).ok()))
|
||||
.collect();
|
||||
Self { values }
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for EnvVarGuard {
|
||||
fn drop(&mut self) {
|
||||
for (key, value) in &self.values {
|
||||
match value {
|
||||
Some(value) => std::env::set_var(key, value),
|
||||
None => std::env::remove_var(key),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn sample_value(samples: &[aether_runtime::MetricSample], name: &str) -> u64 {
|
||||
samples
|
||||
.iter()
|
||||
.find(|sample| sample.name == name)
|
||||
.map(|sample| sample.value)
|
||||
.expect("sample should exist")
|
||||
}
|
||||
|
||||
fn target_value(samples: &[aether_runtime::MetricSample], kind: &str) -> u64 {
|
||||
samples
|
||||
.iter()
|
||||
.find(|sample| {
|
||||
sample.name == "usage_counter_outbox_flush_targets_total"
|
||||
&& sample
|
||||
.labels
|
||||
.iter()
|
||||
.any(|label| label.key == "kind" && label.value == kind)
|
||||
})
|
||||
.map(|sample| sample.value)
|
||||
.expect("target sample should exist")
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usage_counter_flush_worker_config_reads_env() {
|
||||
let _lock = env_lock().lock().expect("env lock should not be poisoned");
|
||||
let _guard = EnvVarGuard::new(CONFIG_ENV_KEYS);
|
||||
std::env::set_var(USAGE_COUNTER_FLUSH_INTERVAL_MS_ENV, "100");
|
||||
std::env::set_var(USAGE_COUNTER_FLUSH_BATCH_SIZE_ENV, "2000");
|
||||
std::env::set_var(USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT_ENV, "50");
|
||||
std::env::set_var(USAGE_COUNTER_DELTA_CLEANUP_INTERVAL_MS_ENV, "250");
|
||||
std::env::set_var(USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE_ENV, "7000");
|
||||
std::env::set_var(USAGE_COUNTER_DELTA_RETENTION_SECS_ENV, "0");
|
||||
|
||||
let config = UsageCounterFlushWorkerConfig::from_env();
|
||||
|
||||
assert_eq!(config.flush_interval, Duration::from_millis(100));
|
||||
assert_eq!(config.flush_batch_size, 2000);
|
||||
assert_eq!(config.flush_catch_up_burst_limit, 50);
|
||||
assert_eq!(config.cleanup_interval, Duration::from_millis(250));
|
||||
assert_eq!(config.cleanup_batch_size, 7000);
|
||||
assert_eq!(config.delta_retention_secs, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usage_counter_flush_worker_config_normalizes_zero_values() {
|
||||
let _lock = env_lock().lock().expect("env lock should not be poisoned");
|
||||
let _guard = EnvVarGuard::new(CONFIG_ENV_KEYS);
|
||||
for key in CONFIG_ENV_KEYS {
|
||||
std::env::set_var(key, "0");
|
||||
}
|
||||
|
||||
let config = UsageCounterFlushWorkerConfig::from_env();
|
||||
|
||||
assert_eq!(config.flush_interval, Duration::from_millis(1));
|
||||
assert_eq!(config.flush_batch_size, 1);
|
||||
assert_eq!(config.flush_catch_up_burst_limit, 1);
|
||||
assert_eq!(config.cleanup_interval, Duration::from_millis(1));
|
||||
assert_eq!(config.cleanup_batch_size, 1);
|
||||
assert_eq!(config.delta_retention_secs, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usage_counter_flush_runtime_metrics_record_success_and_failures() {
|
||||
let metrics = UsageCounterFlushRuntimeMetrics::default();
|
||||
|
||||
metrics.record_flush_success(&UsageCounterFlushSummary {
|
||||
rows_claimed: 3,
|
||||
api_key_targets: 1,
|
||||
provider_api_key_targets: 2,
|
||||
model_targets: 3,
|
||||
provider_monthly_targets: 4,
|
||||
proxy_node_targets: 5,
|
||||
management_token_targets: 6,
|
||||
api_key_last_used_targets: 7,
|
||||
});
|
||||
metrics.record_flush_success(&UsageCounterFlushSummary::default());
|
||||
metrics.record_flush_failed();
|
||||
metrics.record_flush_deferred();
|
||||
metrics.record_cleanup_success(11);
|
||||
metrics.record_cleanup_failed();
|
||||
metrics.record_cleanup_deferred();
|
||||
|
||||
let samples = metrics.metric_samples();
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_flush_batches_total"),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_flush_empty_batches_total"),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_flush_rows_claimed_total"),
|
||||
3
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_flush_failed_batches_total"),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_flush_deferred_total"),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_cleanup_batches_total"),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_cleanup_rows_total"),
|
||||
11
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(
|
||||
&samples,
|
||||
"usage_counter_outbox_cleanup_failed_batches_total"
|
||||
),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
sample_value(&samples, "usage_counter_outbox_cleanup_deferred_total"),
|
||||
1
|
||||
);
|
||||
assert_eq!(target_value(&samples, "api_key"), 1);
|
||||
assert_eq!(target_value(&samples, "provider_api_key"), 2);
|
||||
assert_eq!(target_value(&samples, "model"), 3);
|
||||
assert_eq!(target_value(&samples, "provider_monthly"), 4);
|
||||
assert_eq!(target_value(&samples, "proxy_node"), 5);
|
||||
assert_eq!(target_value(&samples, "management_token"), 6);
|
||||
assert_eq!(target_value(&samples, "api_key_last_used"), 7);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -23,11 +23,9 @@ use super::{
|
||||
PROXY_NODE_METRICS_CLEANUP_HOUR, PROXY_NODE_METRICS_CLEANUP_MINUTE,
|
||||
PROXY_NODE_STALE_SWEEP_INTERVAL, PROXY_UPGRADE_ROLLOUT_INTERVAL,
|
||||
REQUEST_CANDIDATE_CLEANUP_INTERVAL, USAGE_CLEANUP_HOUR, USAGE_CLEANUP_MINUTE,
|
||||
USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE, USAGE_COUNTER_DELTA_CLEANUP_INTERVAL,
|
||||
USAGE_COUNTER_DELTA_RETENTION_SECS, USAGE_COUNTER_FLUSH_BATCH_SIZE,
|
||||
USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT, USAGE_COUNTER_FLUSH_INTERVAL,
|
||||
WALLET_DAILY_USAGE_AGGREGATION_HOUR, WALLET_DAILY_USAGE_AGGREGATION_MINUTE,
|
||||
};
|
||||
use super::{UsageCounterFlushRuntimeMetrics, UsageCounterFlushWorkerConfig};
|
||||
|
||||
const STATS_DAILY_CATCH_UP_BURST_LIMIT: usize = 14;
|
||||
const STATS_HOURLY_CATCH_UP_BURST_LIMIT: usize = 72;
|
||||
@@ -249,13 +247,26 @@ pub(crate) fn spawn_usage_cleanup_worker(
|
||||
|
||||
pub(crate) fn spawn_usage_counter_flush_worker(
|
||||
data: Arc<GatewayDataState>,
|
||||
metrics: Arc<UsageCounterFlushRuntimeMetrics>,
|
||||
) -> Option<tokio::task::JoinHandle<()>> {
|
||||
spawn_usage_counter_flush_worker_with_config(
|
||||
data,
|
||||
metrics,
|
||||
UsageCounterFlushWorkerConfig::from_env(),
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn spawn_usage_counter_flush_worker_with_config(
|
||||
data: Arc<GatewayDataState>,
|
||||
metrics: Arc<UsageCounterFlushRuntimeMetrics>,
|
||||
config: UsageCounterFlushWorkerConfig,
|
||||
) -> Option<tokio::task::JoinHandle<()>> {
|
||||
if !data.has_usage_counter_flush_backend() {
|
||||
return None;
|
||||
}
|
||||
|
||||
Some(tokio::spawn(async move {
|
||||
let mut interval = tokio::time::interval(USAGE_COUNTER_FLUSH_INTERVAL);
|
||||
let mut interval = tokio::time::interval(config.flush_interval);
|
||||
interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
|
||||
interval.tick().await;
|
||||
let mut last_delta_cleanup = tokio::time::Instant::now();
|
||||
@@ -268,47 +279,66 @@ pub(crate) fn spawn_usage_counter_flush_worker(
|
||||
"usage_counter_flush",
|
||||
&mut usage_counter_flush_deferred_since,
|
||||
) {
|
||||
metrics.record_flush_deferred();
|
||||
interval.tick().await;
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut batches = 0_usize;
|
||||
while batches < USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT {
|
||||
match run_usage_counter_flush_once(&data, USAGE_COUNTER_FLUSH_BATCH_SIZE).await {
|
||||
Ok(summary) if summary.rows_claimed > 0 => batches += 1,
|
||||
Ok(_) => break,
|
||||
while batches < config.flush_catch_up_burst_limit {
|
||||
match run_usage_counter_flush_once(&data, config.flush_batch_size).await {
|
||||
Ok(summary) if summary.rows_claimed > 0 => {
|
||||
metrics.record_flush_success(&summary);
|
||||
batches += 1;
|
||||
}
|
||||
Ok(summary) => {
|
||||
metrics.record_flush_success(&summary);
|
||||
break;
|
||||
}
|
||||
Err(err) => {
|
||||
metrics.record_flush_failed();
|
||||
log_maintenance_worker_failure("usage_counter_flush", "tick", &err);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if batches >= USAGE_COUNTER_FLUSH_CATCH_UP_BURST_LIMIT {
|
||||
if batches >= config.flush_catch_up_burst_limit {
|
||||
tokio::task::yield_now().await;
|
||||
continue;
|
||||
}
|
||||
|
||||
if last_delta_cleanup.elapsed() >= USAGE_COUNTER_DELTA_CLEANUP_INTERVAL {
|
||||
if last_delta_cleanup.elapsed() >= config.cleanup_interval {
|
||||
if should_defer_for_database_pressure(
|
||||
&data,
|
||||
"usage_counter_delta_cleanup",
|
||||
&mut usage_counter_delta_cleanup_deferred_since,
|
||||
) {
|
||||
metrics.record_cleanup_deferred();
|
||||
debug!(
|
||||
event_name = "maintenance_worker_deferred",
|
||||
log_type = "ops",
|
||||
worker = "usage_counter_delta_cleanup",
|
||||
"gateway maintenance worker deferred cleanup under database pressure"
|
||||
);
|
||||
} else if let Err(err) = cleanup_processed_usage_counter_deltas_once(
|
||||
&data,
|
||||
USAGE_COUNTER_DELTA_RETENTION_SECS,
|
||||
USAGE_COUNTER_DELTA_CLEANUP_BATCH_SIZE,
|
||||
)
|
||||
.await
|
||||
{
|
||||
log_maintenance_worker_failure("usage_counter_delta_cleanup", "tick", &err);
|
||||
} else {
|
||||
match cleanup_processed_usage_counter_deltas_once(
|
||||
&data,
|
||||
config.delta_retention_secs,
|
||||
config.cleanup_batch_size,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(rows_deleted) => metrics.record_cleanup_success(rows_deleted),
|
||||
Err(err) => {
|
||||
metrics.record_cleanup_failed();
|
||||
log_maintenance_worker_failure(
|
||||
"usage_counter_delta_cleanup",
|
||||
"tick",
|
||||
&err,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
last_delta_cleanup = tokio::time::Instant::now();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,787 @@
|
||||
use std::collections::HashSet;
|
||||
use std::sync::Mutex;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use aether_runtime::{MetricKind, MetricSample};
|
||||
use sysinfo::{get_current_pid, Networks, Pid, ProcessesToUpdate, System};
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default)]
|
||||
pub(crate) struct GatewayProcessResourceSnapshot {
|
||||
pub(crate) sampled_at_unix_secs: u64,
|
||||
pub(crate) system_cpu_usage_basis_points: u64,
|
||||
pub(crate) process_cpu_usage_basis_points: u64,
|
||||
pub(crate) memory_total_bytes: u64,
|
||||
pub(crate) memory_used_bytes: u64,
|
||||
pub(crate) memory_available_bytes: u64,
|
||||
pub(crate) memory_used_basis_points: u64,
|
||||
pub(crate) process_memory_bytes: u64,
|
||||
pub(crate) process_virtual_memory_bytes: u64,
|
||||
pub(crate) process_memory_basis_points: u64,
|
||||
pub(crate) process_uptime_secs: Option<u64>,
|
||||
pub(crate) process_threads: u64,
|
||||
pub(crate) fd_open_count: u64,
|
||||
pub(crate) fd_limit: u64,
|
||||
pub(crate) fd_usage_basis_points: u64,
|
||||
pub(crate) network_observability_available: u64,
|
||||
pub(crate) network_interface_count: u64,
|
||||
pub(crate) network_received_bytes_total: u64,
|
||||
pub(crate) network_transmitted_bytes_total: u64,
|
||||
pub(crate) network_received_packets_total: u64,
|
||||
pub(crate) network_transmitted_packets_total: u64,
|
||||
pub(crate) network_receive_errors_total: u64,
|
||||
pub(crate) network_transmit_errors_total: u64,
|
||||
pub(crate) network_receive_dropped_total: u64,
|
||||
pub(crate) network_transmit_dropped_total: u64,
|
||||
pub(crate) process_socket_fds: u64,
|
||||
pub(crate) tcp_state_observability_available: u64,
|
||||
pub(crate) host_tcp_connections: u64,
|
||||
pub(crate) host_tcp_established_connections: u64,
|
||||
pub(crate) host_tcp_listen_connections: u64,
|
||||
pub(crate) host_tcp_time_wait_connections: u64,
|
||||
pub(crate) host_tcp_syn_sent_connections: u64,
|
||||
pub(crate) host_tcp_syn_recv_connections: u64,
|
||||
pub(crate) host_tcp_close_wait_connections: u64,
|
||||
pub(crate) process_tcp_connections: u64,
|
||||
pub(crate) process_tcp_established_connections: u64,
|
||||
pub(crate) process_tcp_listen_connections: u64,
|
||||
pub(crate) process_tcp_time_wait_connections: u64,
|
||||
pub(crate) process_tcp_syn_sent_connections: u64,
|
||||
pub(crate) process_tcp_syn_recv_connections: u64,
|
||||
pub(crate) process_tcp_close_wait_connections: u64,
|
||||
}
|
||||
|
||||
impl GatewayProcessResourceSnapshot {
|
||||
pub(crate) fn to_metric_samples(self) -> Vec<MetricSample> {
|
||||
let mut samples = vec![
|
||||
MetricSample::new(
|
||||
"gateway_process_sampled_at_unix_secs",
|
||||
"Unix timestamp of the current gateway process resource sample.",
|
||||
MetricKind::Gauge,
|
||||
self.sampled_at_unix_secs,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_system_cpu_usage_basis_points",
|
||||
"Host CPU usage in basis points of percent, where 10000 means 100 percent.",
|
||||
MetricKind::Gauge,
|
||||
self.system_cpu_usage_basis_points,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_cpu_usage_basis_points",
|
||||
"Gateway process CPU usage in basis points of percent, where 10000 means 100 percent.",
|
||||
MetricKind::Gauge,
|
||||
self.process_cpu_usage_basis_points,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_system_memory_total_bytes",
|
||||
"Total host memory visible to the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.memory_total_bytes,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_system_memory_used_bytes",
|
||||
"Used host memory visible to the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.memory_used_bytes,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_system_memory_available_bytes",
|
||||
"Available host memory visible to the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.memory_available_bytes,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_system_memory_usage_basis_points",
|
||||
"Host memory usage in basis points, where 10000 means 100 percent.",
|
||||
MetricKind::Gauge,
|
||||
self.memory_used_basis_points,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_memory_bytes",
|
||||
"Gateway process resident memory bytes.",
|
||||
MetricKind::Gauge,
|
||||
self.process_memory_bytes,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_virtual_memory_bytes",
|
||||
"Gateway process virtual memory bytes.",
|
||||
MetricKind::Gauge,
|
||||
self.process_virtual_memory_bytes,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_memory_basis_points",
|
||||
"Gateway process resident memory as basis points of host memory.",
|
||||
MetricKind::Gauge,
|
||||
self.process_memory_basis_points,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_threads",
|
||||
"Current number of threads owned by the gateway process where available.",
|
||||
MetricKind::Gauge,
|
||||
self.process_threads,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_open_fds",
|
||||
"Current number of file descriptors opened by the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.fd_open_count,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_fd_limit",
|
||||
"Current soft file descriptor limit for the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.fd_limit,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_fd_usage_basis_points",
|
||||
"Gateway process file descriptor usage in basis points, where 10000 means 100 percent.",
|
||||
MetricKind::Gauge,
|
||||
self.fd_usage_basis_points,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_observability_available",
|
||||
"Whether host network interface counters are available.",
|
||||
MetricKind::Gauge,
|
||||
self.network_observability_available,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_interfaces",
|
||||
"Number of host network interfaces visible to the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.network_interface_count,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_received_bytes_total",
|
||||
"Total host network bytes received across visible interfaces.",
|
||||
MetricKind::Counter,
|
||||
self.network_received_bytes_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_transmitted_bytes_total",
|
||||
"Total host network bytes transmitted across visible interfaces.",
|
||||
MetricKind::Counter,
|
||||
self.network_transmitted_bytes_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_received_packets_total",
|
||||
"Total host network packets received across visible interfaces.",
|
||||
MetricKind::Counter,
|
||||
self.network_received_packets_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_transmitted_packets_total",
|
||||
"Total host network packets transmitted across visible interfaces.",
|
||||
MetricKind::Counter,
|
||||
self.network_transmitted_packets_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_receive_errors_total",
|
||||
"Total host network receive errors across visible interfaces.",
|
||||
MetricKind::Counter,
|
||||
self.network_receive_errors_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_transmit_errors_total",
|
||||
"Total host network transmit errors across visible interfaces.",
|
||||
MetricKind::Counter,
|
||||
self.network_transmit_errors_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_receive_dropped_total",
|
||||
"Total host network receive drops across visible interfaces where available.",
|
||||
MetricKind::Counter,
|
||||
self.network_receive_dropped_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_network_transmit_dropped_total",
|
||||
"Total host network transmit drops across visible interfaces where available.",
|
||||
MetricKind::Counter,
|
||||
self.network_transmit_dropped_total,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_socket_fds",
|
||||
"Current number of socket file descriptors opened by the gateway process.",
|
||||
MetricKind::Gauge,
|
||||
self.process_socket_fds,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_tcp_state_observability_available",
|
||||
"Whether Linux TCP state counters are available from procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.tcp_state_observability_available,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_connections",
|
||||
"Current host TCP connections visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_established_connections",
|
||||
"Current host TCP connections in ESTABLISHED state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_established_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_listen_connections",
|
||||
"Current host TCP sockets in LISTEN state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_listen_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_time_wait_connections",
|
||||
"Current host TCP connections in TIME_WAIT state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_time_wait_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_syn_sent_connections",
|
||||
"Current host TCP connections in SYN_SENT state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_syn_sent_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_syn_recv_connections",
|
||||
"Current host TCP connections in SYN_RECV state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_syn_recv_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_host_tcp_close_wait_connections",
|
||||
"Current host TCP connections in CLOSE_WAIT state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.host_tcp_close_wait_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_connections",
|
||||
"Current gateway process TCP connections visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_established_connections",
|
||||
"Current gateway process TCP connections in ESTABLISHED state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_established_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_listen_connections",
|
||||
"Current gateway process TCP sockets in LISTEN state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_listen_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_time_wait_connections",
|
||||
"Current gateway process TCP connections in TIME_WAIT state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_time_wait_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_syn_sent_connections",
|
||||
"Current gateway process TCP connections in SYN_SENT state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_syn_sent_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_syn_recv_connections",
|
||||
"Current gateway process TCP connections in SYN_RECV state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_syn_recv_connections,
|
||||
),
|
||||
MetricSample::new(
|
||||
"gateway_process_tcp_close_wait_connections",
|
||||
"Current gateway process TCP connections in CLOSE_WAIT state visible in procfs.",
|
||||
MetricKind::Gauge,
|
||||
self.process_tcp_close_wait_connections,
|
||||
),
|
||||
];
|
||||
|
||||
if let Some(process_uptime_secs) = self.process_uptime_secs {
|
||||
samples.push(MetricSample::new(
|
||||
"gateway_process_uptime_seconds",
|
||||
"Gateway process uptime in seconds.",
|
||||
MetricKind::Gauge,
|
||||
process_uptime_secs,
|
||||
));
|
||||
}
|
||||
|
||||
samples
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) struct GatewayProcessResourceMonitor {
|
||||
system: Mutex<System>,
|
||||
networks: Mutex<Networks>,
|
||||
current_pid: Option<Pid>,
|
||||
}
|
||||
|
||||
impl GatewayProcessResourceMonitor {
|
||||
pub(crate) fn new() -> Self {
|
||||
let mut system = System::new_all();
|
||||
let mut networks = Networks::new_with_refreshed_list();
|
||||
let current_pid = get_current_pid().ok();
|
||||
if let Some(pid) = current_pid {
|
||||
system.refresh_processes(ProcessesToUpdate::Some(&[pid]), true);
|
||||
}
|
||||
system.refresh_cpu_usage();
|
||||
system.refresh_memory();
|
||||
networks.refresh();
|
||||
Self {
|
||||
system: Mutex::new(system),
|
||||
networks: Mutex::new(networks),
|
||||
current_pid,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn snapshot(&self) -> GatewayProcessResourceSnapshot {
|
||||
let mut system = match self.system.lock() {
|
||||
Ok(guard) => guard,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
};
|
||||
|
||||
system.refresh_cpu_usage();
|
||||
system.refresh_memory();
|
||||
if let Some(pid) = self.current_pid {
|
||||
system.refresh_processes(ProcessesToUpdate::Some(&[pid]), true);
|
||||
}
|
||||
|
||||
let memory_total_bytes = system.total_memory();
|
||||
let memory_used_bytes = system.used_memory();
|
||||
let memory_available_bytes = system.available_memory();
|
||||
let (
|
||||
process_cpu_usage_basis_points,
|
||||
process_memory_bytes,
|
||||
process_virtual_memory_bytes,
|
||||
process_uptime_secs,
|
||||
) = self
|
||||
.current_pid
|
||||
.and_then(|pid| system.process(pid))
|
||||
.map(|process| {
|
||||
(
|
||||
percent_to_basis_points(process.cpu_usage() as f64),
|
||||
process.memory(),
|
||||
process.virtual_memory(),
|
||||
Some(process.run_time()),
|
||||
)
|
||||
})
|
||||
.unwrap_or((0, 0, 0, None));
|
||||
let fd_open_count = open_file_descriptors().unwrap_or(0);
|
||||
let fd_limit = file_descriptor_limit();
|
||||
let network = self.network_snapshot();
|
||||
let sockets = socket_snapshot().unwrap_or_default();
|
||||
|
||||
GatewayProcessResourceSnapshot {
|
||||
sampled_at_unix_secs: current_unix_secs(),
|
||||
system_cpu_usage_basis_points: percent_to_basis_points(system.global_cpu_usage() as f64),
|
||||
process_cpu_usage_basis_points,
|
||||
memory_total_bytes,
|
||||
memory_used_bytes,
|
||||
memory_available_bytes,
|
||||
memory_used_basis_points: ratio_to_basis_points(memory_used_bytes, memory_total_bytes),
|
||||
process_memory_bytes,
|
||||
process_virtual_memory_bytes,
|
||||
process_memory_basis_points: ratio_to_basis_points(
|
||||
process_memory_bytes,
|
||||
memory_total_bytes,
|
||||
),
|
||||
process_uptime_secs,
|
||||
process_threads: process_thread_count().unwrap_or(0),
|
||||
fd_open_count,
|
||||
fd_limit,
|
||||
fd_usage_basis_points: ratio_to_basis_points(fd_open_count, fd_limit),
|
||||
network_observability_available: network.observability_available,
|
||||
network_interface_count: network.interface_count,
|
||||
network_received_bytes_total: network.received_bytes_total,
|
||||
network_transmitted_bytes_total: network.transmitted_bytes_total,
|
||||
network_received_packets_total: network.received_packets_total,
|
||||
network_transmitted_packets_total: network.transmitted_packets_total,
|
||||
network_receive_errors_total: network.receive_errors_total,
|
||||
network_transmit_errors_total: network.transmit_errors_total,
|
||||
network_receive_dropped_total: network.receive_dropped_total,
|
||||
network_transmit_dropped_total: network.transmit_dropped_total,
|
||||
process_socket_fds: sockets.process_socket_fds,
|
||||
tcp_state_observability_available: sockets.tcp_state_observability_available,
|
||||
host_tcp_connections: sockets.host_tcp.total,
|
||||
host_tcp_established_connections: sockets.host_tcp.established,
|
||||
host_tcp_listen_connections: sockets.host_tcp.listen,
|
||||
host_tcp_time_wait_connections: sockets.host_tcp.time_wait,
|
||||
host_tcp_syn_sent_connections: sockets.host_tcp.syn_sent,
|
||||
host_tcp_syn_recv_connections: sockets.host_tcp.syn_recv,
|
||||
host_tcp_close_wait_connections: sockets.host_tcp.close_wait,
|
||||
process_tcp_connections: sockets.process_tcp.total,
|
||||
process_tcp_established_connections: sockets.process_tcp.established,
|
||||
process_tcp_listen_connections: sockets.process_tcp.listen,
|
||||
process_tcp_time_wait_connections: sockets.process_tcp.time_wait,
|
||||
process_tcp_syn_sent_connections: sockets.process_tcp.syn_sent,
|
||||
process_tcp_syn_recv_connections: sockets.process_tcp.syn_recv,
|
||||
process_tcp_close_wait_connections: sockets.process_tcp.close_wait,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn metric_samples(&self) -> Vec<MetricSample> {
|
||||
self.snapshot().to_metric_samples()
|
||||
}
|
||||
|
||||
fn network_snapshot(&self) -> GatewayNetworkSnapshot {
|
||||
let mut networks = match self.networks.lock() {
|
||||
Ok(guard) => guard,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
};
|
||||
networks.refresh();
|
||||
|
||||
let mut snapshot = GatewayNetworkSnapshot {
|
||||
observability_available: u64::from(!networks.list().is_empty()),
|
||||
interface_count: networks.list().len() as u64,
|
||||
..GatewayNetworkSnapshot::default()
|
||||
};
|
||||
for network in networks.list().values() {
|
||||
snapshot.received_bytes_total = snapshot
|
||||
.received_bytes_total
|
||||
.saturating_add(network.total_received());
|
||||
snapshot.transmitted_bytes_total = snapshot
|
||||
.transmitted_bytes_total
|
||||
.saturating_add(network.total_transmitted());
|
||||
snapshot.received_packets_total = snapshot
|
||||
.received_packets_total
|
||||
.saturating_add(network.total_packets_received());
|
||||
snapshot.transmitted_packets_total = snapshot
|
||||
.transmitted_packets_total
|
||||
.saturating_add(network.total_packets_transmitted());
|
||||
snapshot.receive_errors_total = snapshot
|
||||
.receive_errors_total
|
||||
.saturating_add(network.total_errors_on_received());
|
||||
snapshot.transmit_errors_total = snapshot
|
||||
.transmit_errors_total
|
||||
.saturating_add(network.total_errors_on_transmitted());
|
||||
}
|
||||
|
||||
let drops = network_drop_totals();
|
||||
snapshot.receive_dropped_total = drops.receive_dropped_total;
|
||||
snapshot.transmit_dropped_total = drops.transmit_dropped_total;
|
||||
snapshot
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for GatewayProcessResourceMonitor {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for GatewayProcessResourceMonitor {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("GatewayProcessResourceMonitor")
|
||||
.field("current_pid", &self.current_pid)
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
fn current_unix_secs() -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.map(|duration| duration.as_secs())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
fn percent_to_basis_points(value: f64) -> u64 {
|
||||
if !value.is_finite() || value.is_sign_negative() {
|
||||
0
|
||||
} else {
|
||||
(value * 100.0).round().clamp(0.0, u64::MAX as f64) as u64
|
||||
}
|
||||
}
|
||||
|
||||
fn ratio_to_basis_points(value: u64, total: u64) -> u64 {
|
||||
value.saturating_mul(10_000).checked_div(total).unwrap_or(0)
|
||||
}
|
||||
|
||||
fn open_file_descriptors() -> Option<u64> {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
for dir in ["/proc/self/fd", "/dev/fd"] {
|
||||
if let Ok(entries) = std::fs::read_dir(dir) {
|
||||
return Some(entries.count() as u64);
|
||||
}
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn file_descriptor_limit() -> u64 {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
let mut limit = libc::rlimit {
|
||||
rlim_cur: 0,
|
||||
rlim_max: 0,
|
||||
};
|
||||
let result = unsafe { libc::getrlimit(libc::RLIMIT_NOFILE, &mut limit) };
|
||||
if result == 0 {
|
||||
return limit.rlim_cur;
|
||||
}
|
||||
}
|
||||
0
|
||||
}
|
||||
|
||||
fn process_thread_count() -> Option<u64> {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
return std::fs::read_to_string("/proc/self/status")
|
||||
.ok()
|
||||
.and_then(|raw| parse_linux_process_thread_count(&raw));
|
||||
}
|
||||
|
||||
#[allow(unreachable_code)]
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn parse_linux_process_thread_count(raw: &str) -> Option<u64> {
|
||||
raw.lines().find_map(|line| {
|
||||
let value = line.strip_prefix("Threads:")?.trim();
|
||||
value.parse::<u64>().ok()
|
||||
})
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default)]
|
||||
struct GatewayNetworkSnapshot {
|
||||
observability_available: u64,
|
||||
interface_count: u64,
|
||||
received_bytes_total: u64,
|
||||
transmitted_bytes_total: u64,
|
||||
received_packets_total: u64,
|
||||
transmitted_packets_total: u64,
|
||||
receive_errors_total: u64,
|
||||
transmit_errors_total: u64,
|
||||
receive_dropped_total: u64,
|
||||
transmit_dropped_total: u64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default)]
|
||||
struct NetworkDropTotals {
|
||||
receive_dropped_total: u64,
|
||||
transmit_dropped_total: u64,
|
||||
}
|
||||
|
||||
fn network_drop_totals() -> NetworkDropTotals {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
return std::fs::read_to_string("/proc/net/dev")
|
||||
.ok()
|
||||
.map(|raw| parse_linux_network_drop_totals(&raw))
|
||||
.unwrap_or_default();
|
||||
}
|
||||
|
||||
#[allow(unreachable_code)]
|
||||
NetworkDropTotals::default()
|
||||
}
|
||||
|
||||
fn parse_linux_network_drop_totals(raw: &str) -> NetworkDropTotals {
|
||||
let mut totals = NetworkDropTotals::default();
|
||||
for line in raw.lines().skip(2) {
|
||||
let Some((_, counters)) = line.split_once(':') else {
|
||||
continue;
|
||||
};
|
||||
let fields: Vec<&str> = counters.split_whitespace().collect();
|
||||
if fields.len() < 12 {
|
||||
continue;
|
||||
}
|
||||
totals.receive_dropped_total = totals
|
||||
.receive_dropped_total
|
||||
.saturating_add(fields[3].parse::<u64>().unwrap_or_default());
|
||||
totals.transmit_dropped_total = totals
|
||||
.transmit_dropped_total
|
||||
.saturating_add(fields[11].parse::<u64>().unwrap_or_default());
|
||||
}
|
||||
totals
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default)]
|
||||
struct SocketSnapshot {
|
||||
process_socket_fds: u64,
|
||||
tcp_state_observability_available: u64,
|
||||
host_tcp: TcpStateCounts,
|
||||
process_tcp: TcpStateCounts,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
struct TcpStateCounts {
|
||||
total: u64,
|
||||
established: u64,
|
||||
listen: u64,
|
||||
time_wait: u64,
|
||||
syn_sent: u64,
|
||||
syn_recv: u64,
|
||||
close_wait: u64,
|
||||
}
|
||||
|
||||
impl TcpStateCounts {
|
||||
fn observe(&mut self, state: &str) {
|
||||
self.total = self.total.saturating_add(1);
|
||||
match state {
|
||||
"01" => self.established = self.established.saturating_add(1),
|
||||
"02" => self.syn_sent = self.syn_sent.saturating_add(1),
|
||||
"03" => self.syn_recv = self.syn_recv.saturating_add(1),
|
||||
"06" => self.time_wait = self.time_wait.saturating_add(1),
|
||||
"08" => self.close_wait = self.close_wait.saturating_add(1),
|
||||
"0A" | "0a" => self.listen = self.listen.saturating_add(1),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn socket_snapshot() -> Option<SocketSnapshot> {
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
let process_inodes = process_socket_inodes().unwrap_or_default();
|
||||
let mut snapshot = SocketSnapshot {
|
||||
process_socket_fds: process_inodes.len() as u64,
|
||||
..SocketSnapshot::default()
|
||||
};
|
||||
let mut observed_tcp_table = false;
|
||||
|
||||
for path in ["/proc/net/tcp", "/proc/net/tcp6"] {
|
||||
let Ok(raw) = std::fs::read_to_string(path) else {
|
||||
continue;
|
||||
};
|
||||
observed_tcp_table = true;
|
||||
observe_linux_tcp_table(&raw, &process_inodes, &mut snapshot);
|
||||
}
|
||||
|
||||
if observed_tcp_table {
|
||||
snapshot.tcp_state_observability_available = 1;
|
||||
return Some(snapshot);
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(unreachable_code)]
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn process_socket_inodes() -> Option<HashSet<u64>> {
|
||||
let mut inodes = HashSet::new();
|
||||
for entry in std::fs::read_dir("/proc/self/fd").ok()? {
|
||||
let Ok(entry) = entry else {
|
||||
continue;
|
||||
};
|
||||
let Ok(target) = std::fs::read_link(entry.path()) else {
|
||||
continue;
|
||||
};
|
||||
if let Some(inode) = parse_socket_inode(&target.to_string_lossy()) {
|
||||
inodes.insert(inode);
|
||||
}
|
||||
}
|
||||
Some(inodes)
|
||||
}
|
||||
|
||||
fn parse_socket_inode(target: &str) -> Option<u64> {
|
||||
target
|
||||
.strip_prefix("socket:[")
|
||||
.and_then(|value| value.strip_suffix(']'))
|
||||
.and_then(|value| value.parse::<u64>().ok())
|
||||
}
|
||||
|
||||
fn observe_linux_tcp_table(
|
||||
raw: &str,
|
||||
process_inodes: &HashSet<u64>,
|
||||
snapshot: &mut SocketSnapshot,
|
||||
) {
|
||||
for line in raw.lines().skip(1) {
|
||||
let fields: Vec<&str> = line.split_whitespace().collect();
|
||||
if fields.len() <= 9 {
|
||||
continue;
|
||||
}
|
||||
let state = fields[3];
|
||||
snapshot.host_tcp.observe(state);
|
||||
let inode = fields[9].parse::<u64>().unwrap_or_default();
|
||||
if process_inodes.contains(&inode) {
|
||||
snapshot.process_tcp.observe(state);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn parses_socket_inode_targets() {
|
||||
assert_eq!(parse_socket_inode("socket:[12345]"), Some(12345));
|
||||
assert_eq!(parse_socket_inode("anon_inode:[eventpoll]"), None);
|
||||
assert_eq!(parse_socket_inode("socket:12345"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_linux_tcp_table_state_counts() {
|
||||
let raw = "\
|
||||
sl local_address rem_address st tx_queue rx_queue tr tm->when retrnsmt uid timeout inode
|
||||
0: 0100007F:1F90 00000000:0000 0A 00000000:00000000 00:00000000 00000000 501 0 111 1 0000000000000000 100 0 0 10 0
|
||||
1: 0100007F:9C40 0100007F:1F90 01 00000000:00000000 00:00000000 00000000 501 0 222 1 0000000000000000 20 4 30 10 -1
|
||||
2: 0100007F:9C41 0100007F:1F90 08 00000000:00000000 00:00000000 00000000 501 0 333 1 0000000000000000 20 4 30 10 -1
|
||||
";
|
||||
let mut inodes = HashSet::new();
|
||||
inodes.insert(111);
|
||||
inodes.insert(333);
|
||||
let mut snapshot = SocketSnapshot::default();
|
||||
|
||||
observe_linux_tcp_table(raw, &inodes, &mut snapshot);
|
||||
|
||||
assert_eq!(snapshot.host_tcp.total, 3);
|
||||
assert_eq!(snapshot.host_tcp.listen, 1);
|
||||
assert_eq!(snapshot.host_tcp.established, 1);
|
||||
assert_eq!(snapshot.host_tcp.close_wait, 1);
|
||||
assert_eq!(snapshot.process_tcp.total, 2);
|
||||
assert_eq!(snapshot.process_tcp.listen, 1);
|
||||
assert_eq!(snapshot.process_tcp.close_wait, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_linux_network_drop_totals() {
|
||||
let raw = "\
|
||||
Inter-| Receive | Transmit
|
||||
face |bytes packets errs drop fifo frame compressed multicast|bytes packets errs drop fifo colls carrier compressed
|
||||
lo: 100 1 0 2 0 0 0 0 200 2 0 3 0 0 0 0
|
||||
eth0: 300 3 0 5 0 0 0 0 400 4 0 7 0 0 0 0
|
||||
";
|
||||
|
||||
let totals = parse_linux_network_drop_totals(raw);
|
||||
|
||||
assert_eq!(totals.receive_dropped_total, 7);
|
||||
assert_eq!(totals.transmit_dropped_total, 10);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn process_resource_monitor_renders_gateway_metrics() {
|
||||
let samples = GatewayProcessResourceMonitor::new().metric_samples();
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_process_memory_bytes"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_process_open_fds"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_process_fd_usage_basis_points"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_process_threads"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_process_socket_fds"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_network_observability_available"));
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
#[test]
|
||||
fn parses_linux_process_thread_count() {
|
||||
let raw = "\
|
||||
Name:\taether-gateway
|
||||
State:\tS (sleeping)
|
||||
Threads:\t42
|
||||
";
|
||||
|
||||
assert_eq!(parse_linux_process_thread_count(raw), Some(42));
|
||||
assert_eq!(parse_linux_process_thread_count("Name:\ttest\n"), None);
|
||||
}
|
||||
}
|
||||
@@ -17,9 +17,13 @@ const BATCH_SIZE_ENV: &str = "AETHER_GATEWAY_REQUEST_CANDIDATE_QUEUE_BATCH_SIZE"
|
||||
const FLUSH_INTERVAL_MS_ENV: &str = "AETHER_GATEWAY_REQUEST_CANDIDATE_QUEUE_FLUSH_INTERVAL_MS";
|
||||
const WORKERS_ENV: &str = "AETHER_GATEWAY_REQUEST_CANDIDATE_QUEUE_WORKERS";
|
||||
const QUEUE_FULL_ENV: &str = "AETHER_GATEWAY_REQUEST_CANDIDATE_QUEUE_FULL";
|
||||
const DB_WRITE_CONCURRENCY_LIMIT_ENV: &str =
|
||||
"AETHER_GATEWAY_REQUEST_CANDIDATE_DB_WRITE_CONCURRENCY_LIMIT";
|
||||
const DB_BATCH_SIZE_ENV: &str = "AETHER_GATEWAY_REQUEST_CANDIDATE_DB_BATCH_SIZE";
|
||||
|
||||
const DEFAULT_QUEUE_CAPACITY: usize = 65_536;
|
||||
const DEFAULT_BATCH_SIZE: usize = 512;
|
||||
const DEFAULT_DB_BATCH_SIZE: usize = 128;
|
||||
const DEFAULT_FLUSH_INTERVAL_MS: u64 = 50;
|
||||
const DEFAULT_WORKERS: usize = 2;
|
||||
const FAILED_FLUSH_RETRY_DELAY_MS: u64 = 25;
|
||||
@@ -41,8 +45,10 @@ pub(crate) struct RequestCandidateQueueConfig {
|
||||
pub(crate) mode: RequestCandidateWriteMode,
|
||||
pub(crate) capacity: usize,
|
||||
pub(crate) batch_size: usize,
|
||||
pub(crate) db_batch_size: usize,
|
||||
pub(crate) flush_interval: Duration,
|
||||
pub(crate) workers: usize,
|
||||
pub(crate) db_write_concurrency_limit: Option<usize>,
|
||||
pub(crate) full_policy: RequestCandidateQueueFullPolicy,
|
||||
}
|
||||
|
||||
@@ -52,8 +58,10 @@ impl Default for RequestCandidateQueueConfig {
|
||||
mode: RequestCandidateWriteMode::Sync,
|
||||
capacity: DEFAULT_QUEUE_CAPACITY,
|
||||
batch_size: DEFAULT_BATCH_SIZE,
|
||||
db_batch_size: DEFAULT_DB_BATCH_SIZE,
|
||||
flush_interval: Duration::from_millis(DEFAULT_FLUSH_INTERVAL_MS),
|
||||
workers: DEFAULT_WORKERS,
|
||||
db_write_concurrency_limit: None,
|
||||
full_policy: RequestCandidateQueueFullPolicy::Sync,
|
||||
}
|
||||
}
|
||||
@@ -69,9 +77,12 @@ impl RequestCandidateQueueConfig {
|
||||
};
|
||||
config.capacity = env_usize(QUEUE_CAPACITY_ENV, DEFAULT_QUEUE_CAPACITY).max(1);
|
||||
config.batch_size = env_usize(BATCH_SIZE_ENV, DEFAULT_BATCH_SIZE).max(1);
|
||||
config.db_batch_size = env_usize(DB_BATCH_SIZE_ENV, DEFAULT_DB_BATCH_SIZE).max(1);
|
||||
config.flush_interval =
|
||||
Duration::from_millis(env_u64(FLUSH_INTERVAL_MS_ENV, DEFAULT_FLUSH_INTERVAL_MS).max(1));
|
||||
config.workers = env_usize(WORKERS_ENV, DEFAULT_WORKERS).clamp(1, 32);
|
||||
config.db_write_concurrency_limit =
|
||||
env_optional_usize(DB_WRITE_CONCURRENCY_LIMIT_ENV).map(|limit| limit.clamp(1, 32));
|
||||
config.full_policy = match env_string(QUEUE_FULL_ENV).as_deref() {
|
||||
Some("drop") | Some("best_effort") | Some("best-effort") => {
|
||||
RequestCandidateQueueFullPolicy::Drop
|
||||
@@ -100,6 +111,9 @@ struct RequestCandidateQueueMetrics {
|
||||
flush_batches_total: AtomicU64,
|
||||
flush_sql_ops_total: AtomicU64,
|
||||
flush_sql_records_total: AtomicU64,
|
||||
db_write_in_flight: AtomicUsize,
|
||||
db_write_max_in_flight: AtomicUsize,
|
||||
db_write_wait_total: AtomicU64,
|
||||
compacted_total: AtomicU64,
|
||||
sync_fallback_total: AtomicU64,
|
||||
}
|
||||
@@ -110,6 +124,7 @@ pub(crate) struct RequestCandidateQueueRuntime {
|
||||
repository: Arc<dyn RequestCandidateWriteRepository>,
|
||||
config: RequestCandidateQueueConfig,
|
||||
metrics: Arc<RequestCandidateQueueMetrics>,
|
||||
db_write_gate: Option<Arc<RequestCandidateDbWriteGate>>,
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for RequestCandidateQueueRuntime {
|
||||
@@ -138,11 +153,16 @@ impl RequestCandidateQueueRuntime {
|
||||
senders.push(sender);
|
||||
receivers.push(receiver);
|
||||
}
|
||||
let db_write_gate = config
|
||||
.db_write_concurrency_limit
|
||||
.map(RequestCandidateDbWriteGate::new)
|
||||
.map(Arc::new);
|
||||
let runtime = Arc::new(Self {
|
||||
senders,
|
||||
repository,
|
||||
config,
|
||||
metrics: Arc::new(RequestCandidateQueueMetrics::default()),
|
||||
db_write_gate,
|
||||
});
|
||||
runtime.spawn_workers(receivers);
|
||||
runtime
|
||||
@@ -264,6 +284,36 @@ impl RequestCandidateQueueRuntime {
|
||||
MetricKind::Counter,
|
||||
self.metrics.flush_sql_records_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"request_candidate_queue_db_batch_size",
|
||||
"Maximum request candidate records submitted in one async DB batch upsert after compaction.",
|
||||
MetricKind::Gauge,
|
||||
self.config.db_batch_size as u64,
|
||||
),
|
||||
MetricSample::new(
|
||||
"request_candidate_queue_db_write_concurrency_limit",
|
||||
"Maximum concurrent request candidate async DB write batches; zero means unlimited.",
|
||||
MetricKind::Gauge,
|
||||
self.config.db_write_concurrency_limit.unwrap_or_default() as u64,
|
||||
),
|
||||
MetricSample::new(
|
||||
"request_candidate_queue_db_write_in_flight",
|
||||
"Current request candidate async DB write batches in flight.",
|
||||
MetricKind::Gauge,
|
||||
self.metrics.db_write_in_flight.load(Ordering::Acquire) as u64,
|
||||
),
|
||||
MetricSample::new(
|
||||
"request_candidate_queue_db_write_max_in_flight",
|
||||
"Maximum observed request candidate async DB write batches in flight.",
|
||||
MetricKind::Gauge,
|
||||
self.metrics.db_write_max_in_flight.load(Ordering::Acquire) as u64,
|
||||
),
|
||||
MetricSample::new(
|
||||
"request_candidate_queue_db_write_wait_total",
|
||||
"Total request candidate async DB write batches that had to wait for the DB write gate.",
|
||||
MetricKind::Counter,
|
||||
self.metrics.db_write_wait_total.load(Ordering::Acquire),
|
||||
),
|
||||
MetricSample::new(
|
||||
"request_candidate_queue_compacted_total",
|
||||
"Total request candidate records compacted before async persistence because a later queued record covered the same request candidate slot.",
|
||||
@@ -287,8 +337,17 @@ impl RequestCandidateQueueRuntime {
|
||||
let repository = Arc::clone(&self.repository);
|
||||
let config = self.config.clone();
|
||||
let metrics = Arc::clone(&self.metrics);
|
||||
let db_write_gate = self.db_write_gate.clone();
|
||||
tokio::spawn(async move {
|
||||
run_worker(repository, config, metrics, worker_index, receiver).await;
|
||||
run_worker(
|
||||
repository,
|
||||
config,
|
||||
metrics,
|
||||
db_write_gate,
|
||||
worker_index,
|
||||
receiver,
|
||||
)
|
||||
.await;
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -306,6 +365,7 @@ async fn run_worker(
|
||||
repository: Arc<dyn RequestCandidateWriteRepository>,
|
||||
config: RequestCandidateQueueConfig,
|
||||
metrics: Arc<RequestCandidateQueueMetrics>,
|
||||
db_write_gate: Option<Arc<RequestCandidateDbWriteGate>>,
|
||||
worker_index: usize,
|
||||
mut receiver: mpsc::Receiver<UpsertRequestCandidateRecord>,
|
||||
) {
|
||||
@@ -317,7 +377,14 @@ async fn run_worker(
|
||||
tokio::select! {
|
||||
_ = ticker.tick() => {
|
||||
if !batch.is_empty() {
|
||||
flush_batch(&repository, &metrics, worker_index, &mut batch).await;
|
||||
flush_batch(
|
||||
&repository,
|
||||
&config,
|
||||
&metrics,
|
||||
db_write_gate.as_ref(),
|
||||
worker_index,
|
||||
&mut batch,
|
||||
).await;
|
||||
}
|
||||
}
|
||||
received = receiver.recv() => {
|
||||
@@ -326,12 +393,26 @@ async fn run_worker(
|
||||
decrement_atomic_usize(&metrics.queued_current);
|
||||
batch.push(record);
|
||||
if batch.len() >= config.batch_size {
|
||||
flush_batch(&repository, &metrics, worker_index, &mut batch).await;
|
||||
flush_batch(
|
||||
&repository,
|
||||
&config,
|
||||
&metrics,
|
||||
db_write_gate.as_ref(),
|
||||
worker_index,
|
||||
&mut batch,
|
||||
).await;
|
||||
}
|
||||
}
|
||||
None => {
|
||||
if !batch.is_empty() {
|
||||
flush_batch(&repository, &metrics, worker_index, &mut batch).await;
|
||||
flush_batch(
|
||||
&repository,
|
||||
&config,
|
||||
&metrics,
|
||||
db_write_gate.as_ref(),
|
||||
worker_index,
|
||||
&mut batch,
|
||||
).await;
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -343,7 +424,9 @@ async fn run_worker(
|
||||
|
||||
async fn flush_batch(
|
||||
repository: &Arc<dyn RequestCandidateWriteRepository>,
|
||||
config: &RequestCandidateQueueConfig,
|
||||
metrics: &RequestCandidateQueueMetrics,
|
||||
db_write_gate: Option<&Arc<RequestCandidateDbWriteGate>>,
|
||||
worker_index: usize,
|
||||
batch: &mut Vec<UpsertRequestCandidateRecord>,
|
||||
) {
|
||||
@@ -360,42 +443,49 @@ async fn flush_batch(
|
||||
.fetch_add(compacted as u64, Ordering::AcqRel);
|
||||
}
|
||||
metrics.flush_batches_total.fetch_add(1, Ordering::AcqRel);
|
||||
let source_count = records
|
||||
.iter()
|
||||
.map(|record| record.source_count)
|
||||
.sum::<usize>();
|
||||
let record_count = records.len();
|
||||
let upsert_records = records
|
||||
.into_iter()
|
||||
.map(|record| record.record)
|
||||
.collect::<Vec<_>>();
|
||||
metrics.flush_sql_ops_total.fetch_add(1, Ordering::AcqRel);
|
||||
metrics
|
||||
.flush_sql_records_total
|
||||
.fetch_add(record_count as u64, Ordering::AcqRel);
|
||||
let mut failed = 0_u64;
|
||||
let mut retry_records = Vec::new();
|
||||
if let Err(err) = repository.upsert_many(upsert_records.clone()).await {
|
||||
failed = source_count as u64;
|
||||
decrement_atomic_usize_by(
|
||||
&metrics.pending_current,
|
||||
source_count.saturating_sub(record_count),
|
||||
);
|
||||
warn!(
|
||||
event_name = "request_candidate_async_flush_failed",
|
||||
log_type = "event",
|
||||
worker_index,
|
||||
record_count,
|
||||
source_count,
|
||||
error = ?err,
|
||||
"gateway failed to asynchronously persist request candidate batch"
|
||||
);
|
||||
retry_records = upsert_records;
|
||||
} else {
|
||||
|
||||
for chunk in records.chunks(config.db_batch_size) {
|
||||
let source_count = chunk
|
||||
.iter()
|
||||
.map(|record| record.source_count)
|
||||
.sum::<usize>();
|
||||
let record_count = chunk.len();
|
||||
let upsert_records = chunk
|
||||
.iter()
|
||||
.map(|record| record.record.clone())
|
||||
.collect::<Vec<_>>();
|
||||
metrics.flush_sql_ops_total.fetch_add(1, Ordering::AcqRel);
|
||||
metrics
|
||||
.flushed_total
|
||||
.fetch_add(source_count as u64, Ordering::AcqRel);
|
||||
decrement_atomic_usize_by(&metrics.pending_current, source_count);
|
||||
.flush_sql_records_total
|
||||
.fetch_add(record_count as u64, Ordering::AcqRel);
|
||||
let _db_write_permit = match db_write_gate {
|
||||
Some(gate) => Some(gate.acquire(metrics).await),
|
||||
None => None,
|
||||
};
|
||||
if let Err(err) = repository.upsert_many(upsert_records.clone()).await {
|
||||
failed = failed.saturating_add(source_count as u64);
|
||||
decrement_atomic_usize_by(
|
||||
&metrics.pending_current,
|
||||
source_count.saturating_sub(record_count),
|
||||
);
|
||||
warn!(
|
||||
event_name = "request_candidate_async_flush_failed",
|
||||
log_type = "event",
|
||||
worker_index,
|
||||
record_count,
|
||||
source_count,
|
||||
error = ?err,
|
||||
"gateway failed to asynchronously persist request candidate DB batch"
|
||||
);
|
||||
retry_records.extend(upsert_records);
|
||||
} else {
|
||||
metrics
|
||||
.flushed_total
|
||||
.fetch_add(source_count as u64, Ordering::AcqRel);
|
||||
decrement_atomic_usize_by(&metrics.pending_current, source_count);
|
||||
}
|
||||
}
|
||||
if failed > 0 {
|
||||
metrics
|
||||
@@ -476,6 +566,54 @@ fn worker_queue_capacity(total_capacity: usize, workers: usize, worker_index: us
|
||||
(base + usize::from(worker_index < remainder)).max(1)
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct RequestCandidateDbWriteGate {
|
||||
semaphore: tokio::sync::Semaphore,
|
||||
}
|
||||
|
||||
impl RequestCandidateDbWriteGate {
|
||||
fn new(limit: usize) -> Self {
|
||||
Self {
|
||||
semaphore: tokio::sync::Semaphore::new(limit.max(1)),
|
||||
}
|
||||
}
|
||||
|
||||
async fn acquire<'a>(
|
||||
&'a self,
|
||||
metrics: &'a RequestCandidateQueueMetrics,
|
||||
) -> RequestCandidateDbWritePermit<'a> {
|
||||
if self.semaphore.available_permits() == 0 {
|
||||
metrics.db_write_wait_total.fetch_add(1, Ordering::AcqRel);
|
||||
}
|
||||
let permit = self
|
||||
.semaphore
|
||||
.acquire()
|
||||
.await
|
||||
.expect("request candidate DB write gate semaphore should not be closed");
|
||||
let in_flight = metrics.db_write_in_flight.fetch_add(1, Ordering::AcqRel) + 1;
|
||||
metrics
|
||||
.db_write_max_in_flight
|
||||
.fetch_max(in_flight, Ordering::AcqRel);
|
||||
RequestCandidateDbWritePermit {
|
||||
metrics,
|
||||
_permit: permit,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct RequestCandidateDbWritePermit<'a> {
|
||||
metrics: &'a RequestCandidateQueueMetrics,
|
||||
_permit: tokio::sync::SemaphorePermit<'a>,
|
||||
}
|
||||
|
||||
impl Drop for RequestCandidateDbWritePermit<'_> {
|
||||
fn drop(&mut self) {
|
||||
self.metrics
|
||||
.db_write_in_flight
|
||||
.fetch_sub(1, Ordering::AcqRel);
|
||||
}
|
||||
}
|
||||
|
||||
fn merge_request_candidate_record(
|
||||
target: &mut UpsertRequestCandidateRecord,
|
||||
incoming: UpsertRequestCandidateRecord,
|
||||
@@ -601,6 +739,13 @@ fn env_usize(key: &str, default: usize) -> usize {
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
fn env_optional_usize(key: &str) -> Option<usize> {
|
||||
std::env::var(key)
|
||||
.ok()
|
||||
.and_then(|value| value.trim().parse::<usize>().ok())
|
||||
.filter(|value| *value > 0)
|
||||
}
|
||||
|
||||
fn env_u64(key: &str, default: u64) -> u64 {
|
||||
std::env::var(key)
|
||||
.ok()
|
||||
@@ -666,6 +811,8 @@ mod tests {
|
||||
inner: InMemoryRequestCandidateRepository,
|
||||
upsert_calls: AtomicUsize,
|
||||
upsert_many_calls: AtomicUsize,
|
||||
active_upsert_many: AtomicUsize,
|
||||
max_active_upsert_many: AtomicUsize,
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
@@ -683,10 +830,15 @@ mod tests {
|
||||
candidates: Vec<UpsertRequestCandidateRecord>,
|
||||
) -> Result<usize, DataLayerError> {
|
||||
self.upsert_many_calls.fetch_add(1, Ordering::AcqRel);
|
||||
let active = self.active_upsert_many.fetch_add(1, Ordering::AcqRel) + 1;
|
||||
self.max_active_upsert_many
|
||||
.fetch_max(active, Ordering::AcqRel);
|
||||
tokio::time::sleep(Duration::from_millis(30)).await;
|
||||
let count = candidates.len();
|
||||
for candidate in candidates {
|
||||
self.inner.upsert(candidate).await?;
|
||||
}
|
||||
self.active_upsert_many.fetch_sub(1, Ordering::AcqRel);
|
||||
Ok(count)
|
||||
}
|
||||
|
||||
@@ -826,8 +978,10 @@ mod tests {
|
||||
mode: super::RequestCandidateWriteMode::Async,
|
||||
capacity: 16,
|
||||
batch_size: 2,
|
||||
db_batch_size: 128,
|
||||
flush_interval: Duration::from_millis(10),
|
||||
workers: 1,
|
||||
db_write_concurrency_limit: None,
|
||||
full_policy: super::RequestCandidateQueueFullPolicy::Drop,
|
||||
},
|
||||
);
|
||||
@@ -858,8 +1012,10 @@ mod tests {
|
||||
mode: super::RequestCandidateWriteMode::Async,
|
||||
capacity: 16,
|
||||
batch_size: 4,
|
||||
db_batch_size: 128,
|
||||
flush_interval: Duration::from_millis(100),
|
||||
workers: 1,
|
||||
db_write_concurrency_limit: None,
|
||||
full_policy: super::RequestCandidateQueueFullPolicy::Drop,
|
||||
},
|
||||
);
|
||||
@@ -899,6 +1055,109 @@ mod tests {
|
||||
panic!("async request candidate queue did not finish batch flush in time");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn async_queue_splits_compacted_flush_into_db_batches() {
|
||||
let repository = Arc::new(CountingBatchRequestCandidateRepository::default());
|
||||
let runtime = RequestCandidateQueueRuntime::spawn(
|
||||
repository.clone(),
|
||||
RequestCandidateQueueConfig {
|
||||
mode: super::RequestCandidateWriteMode::Async,
|
||||
capacity: 16,
|
||||
batch_size: 5,
|
||||
db_batch_size: 2,
|
||||
flush_interval: Duration::from_millis(100),
|
||||
workers: 1,
|
||||
db_write_concurrency_limit: None,
|
||||
full_policy: super::RequestCandidateQueueFullPolicy::Drop,
|
||||
},
|
||||
);
|
||||
|
||||
for index in 0..5 {
|
||||
runtime
|
||||
.enqueue_or_fallback(record(
|
||||
"req-db-batch",
|
||||
index,
|
||||
0,
|
||||
RequestCandidateStatus::Success,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
for _ in 0..50 {
|
||||
if runtime.metrics.pending_current.load(Ordering::Acquire) == 0 {
|
||||
assert_eq!(repository.upsert_many_calls.load(Ordering::Acquire), 3);
|
||||
assert_eq!(
|
||||
runtime.metrics.flush_batches_total.load(Ordering::Acquire),
|
||||
1
|
||||
);
|
||||
assert_eq!(
|
||||
runtime.metrics.flush_sql_ops_total.load(Ordering::Acquire),
|
||||
3
|
||||
);
|
||||
assert_eq!(
|
||||
runtime
|
||||
.metrics
|
||||
.flush_sql_records_total
|
||||
.load(Ordering::Acquire),
|
||||
5
|
||||
);
|
||||
return;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(10)).await;
|
||||
}
|
||||
|
||||
panic!("async request candidate queue did not split DB batches in time");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn async_queue_db_write_gate_limits_concurrent_batch_writes() {
|
||||
let repository = Arc::new(CountingBatchRequestCandidateRepository::default());
|
||||
let runtime = RequestCandidateQueueRuntime::spawn(
|
||||
repository.clone(),
|
||||
RequestCandidateQueueConfig {
|
||||
mode: super::RequestCandidateWriteMode::Async,
|
||||
capacity: 64,
|
||||
batch_size: 1,
|
||||
db_batch_size: 128,
|
||||
flush_interval: Duration::from_millis(100),
|
||||
workers: 4,
|
||||
db_write_concurrency_limit: Some(2),
|
||||
full_policy: super::RequestCandidateQueueFullPolicy::Drop,
|
||||
},
|
||||
);
|
||||
|
||||
for index in 0..8 {
|
||||
runtime
|
||||
.enqueue_or_fallback(record(
|
||||
&format!("req-gate-{index}"),
|
||||
0,
|
||||
0,
|
||||
RequestCandidateStatus::Success,
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
for _ in 0..100 {
|
||||
if runtime.metrics.pending_current.load(Ordering::Acquire) == 0 {
|
||||
assert_eq!(repository.max_active_upsert_many.load(Ordering::Acquire), 2);
|
||||
assert_eq!(
|
||||
runtime
|
||||
.metrics
|
||||
.db_write_max_in_flight
|
||||
.load(Ordering::Acquire),
|
||||
2
|
||||
);
|
||||
assert!(runtime.metrics.db_write_wait_total.load(Ordering::Acquire) > 0);
|
||||
return;
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(10)).await;
|
||||
}
|
||||
|
||||
panic!("async request candidate queue did not finish gated writes in time");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn async_queue_preserves_same_slot_order_with_multiple_workers() {
|
||||
let repository = Arc::new(DelayedPendingRequestCandidateRepository::default());
|
||||
@@ -908,8 +1167,10 @@ mod tests {
|
||||
mode: super::RequestCandidateWriteMode::Async,
|
||||
capacity: 16,
|
||||
batch_size: 1,
|
||||
db_batch_size: 128,
|
||||
flush_interval: Duration::from_millis(100),
|
||||
workers: 2,
|
||||
db_write_concurrency_limit: None,
|
||||
full_policy: super::RequestCandidateQueueFullPolicy::Drop,
|
||||
},
|
||||
);
|
||||
|
||||
@@ -20,8 +20,10 @@ use super::super::cache::{
|
||||
};
|
||||
use super::super::data::GatewayDataState;
|
||||
use super::super::fallback_metrics;
|
||||
use super::super::maintenance::UsageCounterFlushRuntimeMetrics;
|
||||
use super::super::rate_limit::FrontdoorUserRpmLimiter;
|
||||
use super::super::request_candidate_queue::RequestCandidateQueueRuntime;
|
||||
use super::super::task_runtime::TaskSupervisorMetrics;
|
||||
use super::super::{provider_transport, usage};
|
||||
use super::{
|
||||
AdminBillingCollectorRecord, AdminBillingRuleRecord, AdminPaymentCallbackRecord,
|
||||
@@ -40,18 +42,22 @@ const MIN_LOCAL_EXECUTION_PLANNING_TIMEOUT_MS: u64 = 500;
|
||||
const MAX_LOCAL_EXECUTION_PLANNING_TIMEOUT_MS: u64 = 120_000;
|
||||
const LOCAL_EXECUTION_PLANNING_TIMEOUT_MS_ENV: &str =
|
||||
"AETHER_GATEWAY_LOCAL_EXECUTION_PLANNING_TIMEOUT_MS";
|
||||
const DEFAULT_AUTH_SNAPSHOT_LOAD_GATE_LIMIT: usize = 64;
|
||||
const DEFAULT_CANDIDATE_PLANNING_GATE_LIMIT: usize = 1024;
|
||||
const DEFAULT_UPSTREAM_EXECUTION_GATE_LIMIT: usize = 10_000;
|
||||
const DEFAULT_UPSTREAM_TARGET_GATE_LIMIT: usize = 10_000;
|
||||
const MAX_AUTH_SNAPSHOT_LOAD_GATE_LIMIT: usize = 1024;
|
||||
const MAX_CANDIDATE_PLANNING_GATE_LIMIT: usize = 8192;
|
||||
const MAX_UPSTREAM_EXECUTION_GATE_LIMIT: usize = 16_384;
|
||||
const MAX_UPSTREAM_TARGET_GATE_LIMIT: usize = 16_384;
|
||||
const AUTH_SNAPSHOT_LOAD_GATE_LIMIT_PER_CPU: usize = 16;
|
||||
const CANDIDATE_PLANNING_GATE_LIMIT_PER_CPU: usize = 256;
|
||||
const UPSTREAM_EXECUTION_GATE_LIMIT_PER_CPU: usize = 1024;
|
||||
const UPSTREAM_TARGET_GATE_LIMIT_PER_CPU: usize = 1024;
|
||||
const GATE_LIMIT_FD_RESERVE: usize = 128;
|
||||
const DEFAULT_INTERNAL_GATE_QUEUE_BUDGET_MS: u64 = 250;
|
||||
const MAX_INTERNAL_GATE_QUEUE_BUDGET_MS: u64 = 5_000;
|
||||
const AUTH_SNAPSHOT_LOAD_GATE_LIMIT_ENV: &str = "AETHER_GATEWAY_AUTH_SNAPSHOT_LOAD_GATE_LIMIT";
|
||||
const CANDIDATE_PLANNING_GATE_LIMIT_ENV: &str = "AETHER_GATEWAY_CANDIDATE_PLANNING_GATE_LIMIT";
|
||||
const UPSTREAM_EXECUTION_GATE_LIMIT_ENV: &str = "AETHER_GATEWAY_UPSTREAM_EXECUTION_GATE_LIMIT";
|
||||
const UPSTREAM_TARGET_GATE_LIMIT_ENV: &str = "AETHER_GATEWAY_UPSTREAM_TARGET_GATE_LIMIT";
|
||||
@@ -87,6 +93,7 @@ pub(crate) struct FrontdoorRuntimeGuardConfig {
|
||||
pub(crate) local_execution_planning_timeout: Duration,
|
||||
pub(crate) internal_gate_queue_budget: Duration,
|
||||
pub(crate) auth_capacity_cache_ttl: Duration,
|
||||
pub(crate) auth_snapshot_load_gate_limit: Option<usize>,
|
||||
pub(crate) candidate_planning_gate_limit: Option<usize>,
|
||||
pub(crate) upstream_execution_gate_limit: Option<usize>,
|
||||
pub(crate) upstream_target_gate_limit: Option<usize>,
|
||||
@@ -119,6 +126,7 @@ impl FrontdoorRuntimeGuardConfig {
|
||||
MIN_AUTH_CAPACITY_CACHE_TTL_MS,
|
||||
MAX_AUTH_CAPACITY_CACHE_TTL_MS,
|
||||
),
|
||||
auth_snapshot_load_gate_limit: auth_snapshot_load_gate_limit_from_env(),
|
||||
candidate_planning_gate_limit: candidate_planning_gate_limit_from_env(),
|
||||
upstream_execution_gate_limit: upstream_execution_gate_limit_from_env(),
|
||||
upstream_target_gate_limit: upstream_target_gate_limit_from_env(),
|
||||
@@ -137,6 +145,7 @@ impl FrontdoorRuntimeGuardConfig {
|
||||
DEFAULT_INTERNAL_GATE_QUEUE_BUDGET_MS,
|
||||
),
|
||||
auth_capacity_cache_ttl: Duration::from_millis(DEFAULT_AUTH_CAPACITY_CACHE_TTL_MS),
|
||||
auth_snapshot_load_gate_limit: Some(DEFAULT_AUTH_SNAPSHOT_LOAD_GATE_LIMIT),
|
||||
candidate_planning_gate_limit: Some(DEFAULT_CANDIDATE_PLANNING_GATE_LIMIT),
|
||||
upstream_execution_gate_limit: Some(DEFAULT_UPSTREAM_EXECUTION_GATE_LIMIT),
|
||||
upstream_target_gate_limit: Some(DEFAULT_UPSTREAM_TARGET_GATE_LIMIT),
|
||||
@@ -192,6 +201,13 @@ const CANDIDATE_PLANNING_GATE_AUTO_PROFILE: GateAutoProfile = GateAutoProfile {
|
||||
fd_divisor: None,
|
||||
};
|
||||
|
||||
const AUTH_SNAPSHOT_LOAD_GATE_AUTO_PROFILE: GateAutoProfile = GateAutoProfile {
|
||||
floor: DEFAULT_AUTH_SNAPSHOT_LOAD_GATE_LIMIT,
|
||||
cap: MAX_AUTH_SNAPSHOT_LOAD_GATE_LIMIT,
|
||||
per_cpu: AUTH_SNAPSHOT_LOAD_GATE_LIMIT_PER_CPU,
|
||||
fd_divisor: None,
|
||||
};
|
||||
|
||||
const UPSTREAM_EXECUTION_GATE_AUTO_PROFILE: GateAutoProfile = GateAutoProfile {
|
||||
floor: DEFAULT_UPSTREAM_EXECUTION_GATE_LIMIT,
|
||||
cap: MAX_UPSTREAM_EXECUTION_GATE_LIMIT,
|
||||
@@ -213,6 +229,13 @@ fn candidate_planning_gate_limit_from_env() -> Option<usize> {
|
||||
)
|
||||
}
|
||||
|
||||
fn auth_snapshot_load_gate_limit_from_env() -> Option<usize> {
|
||||
env_gate_limit(
|
||||
AUTH_SNAPSHOT_LOAD_GATE_LIMIT_ENV,
|
||||
AUTH_SNAPSHOT_LOAD_GATE_AUTO_PROFILE,
|
||||
)
|
||||
}
|
||||
|
||||
fn upstream_execution_gate_limit_from_env() -> Option<usize> {
|
||||
env_gate_limit(
|
||||
UPSTREAM_EXECUTION_GATE_LIMIT_ENV,
|
||||
@@ -315,6 +338,7 @@ pub struct AppState {
|
||||
pub(crate) video_task_poller: Option<VideoTaskPollerConfig>,
|
||||
pub(crate) frontdoor_runtime_guards: Arc<FrontdoorRuntimeGuardConfig>,
|
||||
pub(crate) request_gate: Option<Arc<ConcurrencyGate>>,
|
||||
pub(crate) auth_snapshot_load_gate: Option<Arc<ConcurrencyGate>>,
|
||||
pub(crate) candidate_planning_gate: Option<Arc<ConcurrencyGate>>,
|
||||
pub(crate) upstream_execution_gate: Option<Arc<ConcurrencyGate>>,
|
||||
pub(crate) upstream_target_admission: Arc<crate::upstream_admission::UpstreamTargetAdmission>,
|
||||
@@ -349,6 +373,9 @@ pub struct AppState {
|
||||
pub(crate) chat_pii_redaction_runtime_config_cache:
|
||||
crate::privacy::ChatPiiRedactionRuntimeConfigCacheHandle,
|
||||
pub(crate) fallback_metrics: Arc<fallback_metrics::GatewayFallbackMetrics>,
|
||||
pub(crate) usage_counter_flush_metrics: Arc<UsageCounterFlushRuntimeMetrics>,
|
||||
pub(crate) task_supervisor_metrics: TaskSupervisorMetrics,
|
||||
pub(crate) process_resource_monitor: Arc<crate::process_metrics::GatewayProcessResourceMonitor>,
|
||||
pub(crate) request_candidate_queue: Option<Arc<RequestCandidateQueueRuntime>>,
|
||||
pub(crate) frontdoor_cors: Option<Arc<FrontdoorCorsConfig>>,
|
||||
pub(crate) frontdoor_user_rpm: Arc<FrontdoorUserRpmLimiter>,
|
||||
@@ -454,6 +481,18 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn auth_snapshot_load_gate_auto_limit_uses_smaller_frontdoor_profile() {
|
||||
assert_eq!(
|
||||
parse_gate_limit_value(
|
||||
Some("auto"),
|
||||
AUTH_SNAPSHOT_LOAD_GATE_AUTO_PROFILE,
|
||||
TEST_CAPACITY
|
||||
),
|
||||
Some(192)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gate_limit_parser_accepts_fixed_numbers() {
|
||||
assert_eq!(
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -13,6 +13,18 @@ const AUTH_API_KEY_SNAPSHOT_RUNTIME_CACHE_TTL: Duration = Duration::from_secs(30
|
||||
use super::super::super::{AUTH_API_KEY_LAST_USED_MAX_ENTRIES, AUTH_API_KEY_LAST_USED_TTL};
|
||||
|
||||
impl AppState {
|
||||
async fn acquire_auth_snapshot_load_gate(
|
||||
&self,
|
||||
) -> Result<Option<aether_runtime::ConcurrencyPermit>, GatewayError> {
|
||||
let Some(gate) = self.auth_snapshot_load_gate.as_ref() else {
|
||||
return Ok(None);
|
||||
};
|
||||
gate.acquire()
|
||||
.await
|
||||
.map(Some)
|
||||
.map_err(|err| GatewayError::Internal(err.to_string()))
|
||||
}
|
||||
|
||||
pub(crate) async fn read_cached_auth_api_key_snapshot(
|
||||
&self,
|
||||
user_id: &str,
|
||||
@@ -28,6 +40,7 @@ impl AppState {
|
||||
cache_key,
|
||||
AUTH_API_KEY_SNAPSHOT_RUNTIME_CACHE_TTL,
|
||||
|| async move {
|
||||
let _permit = self.acquire_auth_snapshot_load_gate().await?;
|
||||
self.data
|
||||
.read_auth_api_key_snapshot(user_id, api_key_id, now_unix_secs)
|
||||
.await
|
||||
@@ -52,6 +65,7 @@ impl AppState {
|
||||
cache_key.clone(),
|
||||
AUTH_API_KEY_SNAPSHOT_RUNTIME_CACHE_TTL,
|
||||
|| async move {
|
||||
let _permit = self.acquire_auth_snapshot_load_gate().await?;
|
||||
self.data
|
||||
.read_auth_api_key_snapshot_by_key_hash(key_hash, now_unix_secs)
|
||||
.await
|
||||
|
||||
@@ -6,8 +6,8 @@ use aether_data_contracts::repository::background_tasks::{
|
||||
UpsertBackgroundTaskRun,
|
||||
};
|
||||
use aether_runtime::task::spawn_named;
|
||||
pub(crate) use aether_task_runtime::TaskSupervisor;
|
||||
use aether_task_runtime::{RetryPolicy, TaskDefinition, TaskKind};
|
||||
pub(crate) use aether_task_runtime::{TaskSupervisor, TaskSupervisorMetrics};
|
||||
use serde_json::Value;
|
||||
use tokio::task::JoinHandle;
|
||||
use tracing::warn;
|
||||
|
||||
@@ -343,6 +343,63 @@ async fn gateway_exposes_request_concurrency_metrics_impl() {
|
||||
assert!(body.contains("tunnel_proxy_connections 0"));
|
||||
assert!(body.contains("tunnel_nodes 0"));
|
||||
assert!(body.contains("tunnel_active_streams 0"));
|
||||
assert!(body.contains("gateway_process_cpu_usage_basis_points "));
|
||||
assert!(body.contains("gateway_process_memory_bytes "));
|
||||
assert!(body.contains("gateway_process_threads "));
|
||||
assert!(body.contains("gateway_process_open_fds "));
|
||||
assert!(body.contains("gateway_process_fd_limit "));
|
||||
assert!(body.contains("gateway_process_socket_fds "));
|
||||
assert!(body.contains("gateway_allocator_observability_available "));
|
||||
assert!(body.contains("gateway_allocator_allocated_bytes "));
|
||||
assert!(body.contains("gateway_allocator_active_bytes "));
|
||||
assert!(body.contains("gateway_allocator_resident_bytes "));
|
||||
assert!(body.contains("gateway_allocator_active_to_allocated_basis_points "));
|
||||
assert!(body.contains("gateway_network_observability_available "));
|
||||
assert!(body.contains("gateway_network_received_bytes_total "));
|
||||
assert!(body.contains("gateway_tcp_state_observability_available "));
|
||||
assert!(body.contains("gateway_host_tcp_established_connections "));
|
||||
assert!(body.contains("gateway_process_tcp_established_connections "));
|
||||
assert!(body.contains("postgres_observability_available{driver=\"postgres\"} 0"));
|
||||
assert!(body.contains("postgres_observability_unavailable{driver=\"postgres\"} 0"));
|
||||
assert!(body.contains("postgres_lock_waiting_connections{driver=\"postgres\"} 0"));
|
||||
assert!(body.contains("postgres_oldest_active_query_age_ms{driver=\"postgres\"} 0"));
|
||||
assert!(body.contains("postgres_oldest_transaction_age_ms{driver=\"postgres\"} 0"));
|
||||
assert!(body.contains("redis_runtime_enabled{backend=\"redis\"} 0"));
|
||||
assert!(body.contains("redis_runtime_health_unavailable{backend=\"redis\"} 0"));
|
||||
assert!(body.contains("redis_runtime_connected_clients{backend=\"redis\"} 0"));
|
||||
assert!(body.contains("redis_runtime_used_memory_bytes{backend=\"redis\"} 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_read_batches_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_read_entries_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_reclaimed_entries_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_acked_entries_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_dead_lettered_entries_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_process_failures_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_read_failures_total 0"));
|
||||
assert!(body.contains("usage_runtime_queue_worker_reclaim_failures_total 0"));
|
||||
assert!(body.contains("usage_queue_health_unavailable 0"));
|
||||
assert!(
|
||||
body.contains("usage_queue_enabled{stream=\"usage:events\",group=\"usage_consumers\"} 0")
|
||||
);
|
||||
assert!(body
|
||||
.contains("usage_queue_configured{stream=\"usage:events\",group=\"usage_consumers\"} 0"));
|
||||
assert!(body.contains("usage_queue_dlq_length{stream=\"usage:events:dlq\"} 0"));
|
||||
assert!(body.contains("usage_counter_health_unavailable 0"));
|
||||
assert!(body.contains("usage_counter_outbox_pending_rows 0"));
|
||||
assert!(body.contains("usage_counter_outbox_oldest_pending_age_seconds 0"));
|
||||
assert!(body.contains("usage_counter_outbox_flush_batches_total 0"));
|
||||
assert!(body.contains("usage_counter_outbox_flush_rows_claimed_total 0"));
|
||||
assert!(body.contains("usage_counter_outbox_flush_failed_batches_total 0"));
|
||||
assert!(body.contains("usage_counter_outbox_cleanup_rows_total 0"));
|
||||
assert!(body.contains("usage_counter_outbox_cleanup_failed_batches_total 0"));
|
||||
assert!(body.contains("gateway_background_tasks_active 0"));
|
||||
assert!(body.contains("gateway_background_tasks_supervised_total 0"));
|
||||
assert!(body.contains("gateway_background_tasks_unexpected_exits_total 0"));
|
||||
assert!(body.contains("gateway_background_tasks_panicked_total 0"));
|
||||
assert!(body.contains("gateway_background_tasks_aborted_total 0"));
|
||||
assert!(body.contains("gateway_tokio_runtime_observability_available 1"));
|
||||
assert!(body.contains("gateway_tokio_runtime_workers "));
|
||||
assert!(body.contains("gateway_tokio_runtime_alive_tasks "));
|
||||
assert!(body.contains("gateway_tokio_runtime_global_queue_depth "));
|
||||
|
||||
gateway_handle.abort();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
use aether_runtime::{MetricKind, MetricSample};
|
||||
|
||||
pub(crate) fn gateway_tokio_runtime_metric_samples() -> Vec<MetricSample> {
|
||||
let Ok(handle) = tokio::runtime::Handle::try_current() else {
|
||||
return vec![
|
||||
availability_sample(0),
|
||||
gauge(
|
||||
"gateway_tokio_runtime_workers",
|
||||
TOKIO_RUNTIME_WORKERS_HELP,
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_tokio_runtime_alive_tasks",
|
||||
TOKIO_RUNTIME_ALIVE_TASKS_HELP,
|
||||
0,
|
||||
),
|
||||
gauge(
|
||||
"gateway_tokio_runtime_global_queue_depth",
|
||||
TOKIO_RUNTIME_GLOBAL_QUEUE_DEPTH_HELP,
|
||||
0,
|
||||
),
|
||||
];
|
||||
};
|
||||
let metrics = handle.metrics();
|
||||
vec![
|
||||
availability_sample(1),
|
||||
gauge(
|
||||
"gateway_tokio_runtime_workers",
|
||||
TOKIO_RUNTIME_WORKERS_HELP,
|
||||
u64_from_usize(metrics.num_workers()),
|
||||
),
|
||||
gauge(
|
||||
"gateway_tokio_runtime_alive_tasks",
|
||||
TOKIO_RUNTIME_ALIVE_TASKS_HELP,
|
||||
u64_from_usize(metrics.num_alive_tasks()),
|
||||
),
|
||||
gauge(
|
||||
"gateway_tokio_runtime_global_queue_depth",
|
||||
TOKIO_RUNTIME_GLOBAL_QUEUE_DEPTH_HELP,
|
||||
u64_from_usize(metrics.global_queue_depth()),
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
const TOKIO_RUNTIME_WORKERS_HELP: &str =
|
||||
"Number of worker threads configured for the gateway Tokio runtime.";
|
||||
const TOKIO_RUNTIME_ALIVE_TASKS_HELP: &str =
|
||||
"Current number of alive tasks tracked by the gateway Tokio runtime.";
|
||||
const TOKIO_RUNTIME_GLOBAL_QUEUE_DEPTH_HELP: &str =
|
||||
"Current number of tasks waiting in the gateway Tokio runtime global queue.";
|
||||
|
||||
fn availability_sample(value: u64) -> MetricSample {
|
||||
gauge(
|
||||
"gateway_tokio_runtime_observability_available",
|
||||
"Whether gateway Tokio runtime metrics were available for this scrape.",
|
||||
value,
|
||||
)
|
||||
}
|
||||
|
||||
fn gauge(name: &'static str, help: &'static str, value: u64) -> MetricSample {
|
||||
MetricSample::new(name, help, MetricKind::Gauge, value)
|
||||
}
|
||||
|
||||
fn u64_from_usize(value: usize) -> u64 {
|
||||
u64::try_from(value).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::gateway_tokio_runtime_metric_samples;
|
||||
|
||||
#[tokio::test]
|
||||
async fn renders_tokio_runtime_metrics_inside_runtime() {
|
||||
let samples = gateway_tokio_runtime_metric_samples();
|
||||
|
||||
assert!(samples.iter().any(|sample| {
|
||||
sample.name == "gateway_tokio_runtime_observability_available" && sample.value == 1
|
||||
}));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_tokio_runtime_workers"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_tokio_runtime_alive_tasks"));
|
||||
assert!(samples
|
||||
.iter()
|
||||
.any(|sample| sample.name == "gateway_tokio_runtime_global_queue_depth"));
|
||||
}
|
||||
}
|
||||
@@ -5,11 +5,11 @@ pub(crate) mod write;
|
||||
|
||||
pub(crate) use aether_usage_runtime::UsageRuntime;
|
||||
pub use aether_usage_runtime::UsageRuntimeConfig;
|
||||
pub(crate) use aether_usage_runtime::UsageRuntimeMetricsSnapshot;
|
||||
pub(crate) use aether_usage_runtime::{
|
||||
now_ms, UsageEvent, UsageEventData, UsageEventType, UsageQueue, UsageRequestRecordLevel,
|
||||
USAGE_EVENT_VERSION,
|
||||
};
|
||||
pub(crate) use aether_usage_runtime::{UsageQueueHealthSnapshot, UsageRuntimeMetricsSnapshot};
|
||||
pub(crate) use reporting::{
|
||||
spawn_sync_report, submit_stream_report, submit_sync_report, GatewayStreamReportRequest,
|
||||
GatewaySyncReportRequest,
|
||||
|
||||
Reference in New Issue
Block a user