Files
Aether/crates/aether-data/runtime/schema/bootstrap/postgres/190_overview_analytics.sql
T
elky cb7b9c9ecd feat: unify user analytics and optimize overview aggregation
Merge user accounts and usage reporting into one page with a combined ranking and account table, shared precise time ranges, and simpler range labels.

Parse overview metadata once through a schema-only view migration and disable JIT locally for bucket rebuilds. Preserve automatic backfills.

Add redacted OAuth refresh diagnostics, bucket failure context, and regression coverage. Resolve strict Clippy warnings.
2026-10-05 00:28:31 +08:00

280 lines
16 KiB
PL/PgSQL

ALTER TABLE public.usage ADD COLUMN IF NOT EXISTS failure_origin text;
ALTER TABLE public.usage ADD COLUMN IF NOT EXISTS failure_stage text;
ALTER TABLE public.usage ADD COLUMN IF NOT EXISTS failure_reason text;
ALTER TABLE public.usage ADD COLUMN IF NOT EXISTS failure_schema_version integer;
ALTER TABLE public.usage_settlement_snapshots
ADD COLUMN IF NOT EXISTS quota_covered_amount_usd numeric(20,8),
ADD COLUMN IF NOT EXISTS wallet_consumed_amount_usd numeric(20,8),
ADD COLUMN IF NOT EXISTS wallet_debit_amount_usd numeric(20,8),
ADD COLUMN IF NOT EXISTS wallet_recharge_debit_usd numeric(20,8),
ADD COLUMN IF NOT EXISTS wallet_gift_debit_usd numeric(20,8),
ADD COLUMN IF NOT EXISTS wallet_overdraft_usd numeric(20,8),
ADD COLUMN IF NOT EXISTS allocation_schema_version integer,
ADD COLUMN IF NOT EXISTS allocation_status text;
CREATE TABLE IF NOT EXISTS public.usage_attribution_snapshots (
request_id text PRIMARY KEY,
actor_user_id text,
credential_owner_id text,
attribution_kind text NOT NULL DEFAULT 'unknown',
attribution_source text NOT NULL DEFAULT 'unknown',
record_kind text NOT NULL DEFAULT 'request',
parent_request_id text,
schema_version integer NOT NULL DEFAULT 1,
attribution_revision bigint NOT NULL DEFAULT 1,
recorded_at timestamptz NOT NULL DEFAULT NOW()
);
CREATE INDEX IF NOT EXISTS ix_usage_attribution_actor_request
ON public.usage_attribution_snapshots(actor_user_id, request_id);
-- This attribution table is new and empty; build both lookup indexes here.
CREATE INDEX IF NOT EXISTS ix_usage_attribution_owner_request
ON public.usage_attribution_snapshots(credential_owner_id, request_id);
CREATE TABLE IF NOT EXISTS public.stats_bucket_state (
projection_version text NOT NULL,
granularity text NOT NULL,
bucket_start timestamptz NOT NULL,
source_revision bigint NOT NULL DEFAULT 0,
built_revision bigint NOT NULL DEFAULT -1,
coverage_status text NOT NULL DEFAULT 'unbuilt',
built_at timestamptz,
last_error text,
last_failed_at timestamptz,
PRIMARY KEY (projection_version, granularity, bucket_start)
);
CREATE INDEX IF NOT EXISTS ix_stats_bucket_state_dirty
ON public.stats_bucket_state(bucket_start)
WHERE source_revision > built_revision;
CREATE TABLE IF NOT EXISTS public.stats_overview_hourly (
projection_version text NOT NULL,
bucket_start timestamptz NOT NULL,
dimensions jsonb NOT NULL,
metrics jsonb NOT NULL,
PRIMARY KEY(projection_version, bucket_start, dimensions)
);
CREATE TABLE IF NOT EXISTS public.stats_overview_daily (
projection_version text NOT NULL,
bucket_start timestamptz NOT NULL,
dimensions jsonb NOT NULL,
metrics jsonb NOT NULL,
PRIMARY KEY(projection_version, bucket_start, dimensions)
);
-- These triggers cover old writers, delayed settlement, and maintenance in the fact transaction.
-- Only future fact mutations enqueue work; no historical rows are backfilled.
-- Each writer owns its transaction's keys, so unrelated requests never contend
-- on the current hour/day's stats_bucket_state row.
CREATE TABLE IF NOT EXISTS public.stats_overview_dirty_events (
transaction_id bigint NOT NULL,
projection_version text NOT NULL,
granularity text NOT NULL,
bucket_start timestamptz NOT NULL,
unrecoverable boolean NOT NULL DEFAULT false,
PRIMARY KEY (transaction_id, projection_version, granularity, bucket_start)
);
CREATE INDEX IF NOT EXISTS ix_stats_overview_dirty_events_bucket
ON public.stats_overview_dirty_events(projection_version, granularity, bucket_start);
CREATE OR REPLACE FUNCTION public.overview_mark_usage_bucket() RETURNS trigger
LANGUAGE plpgsql AS $$
DECLARE old_time timestamptz; new_time timestamptz; bucket record;
BEGIN
IF TG_TABLE_NAME = 'usage' THEN
IF TG_OP <> 'INSERT' THEN old_time := OLD.created_at; END IF;
IF TG_OP <> 'DELETE' THEN new_time := NEW.created_at; END IF;
ELSE
IF TG_OP <> 'INSERT' THEN
SELECT created_at INTO old_time FROM public.usage WHERE request_id = OLD.request_id;
END IF;
IF TG_OP <> 'DELETE' THEN
SELECT created_at INTO new_time FROM public.usage WHERE request_id = NEW.request_id;
END IF;
END IF;
FOR bucket IN
SELECT DISTINCT g, date_trunc(g, t AT TIME ZONE 'UTC') AT TIME ZONE 'UTC' AS starts
FROM unnest(ARRAY[old_time, new_time]) t CROSS JOIN unnest(ARRAY['day','hour']) g
WHERE t IS NOT NULL ORDER BY g, starts
LOOP
INSERT INTO public.stats_overview_dirty_events
(transaction_id, projection_version, granularity, bucket_start, unrecoverable)
VALUES (txid_current(), 'overview-v2', bucket.g, bucket.starts,
TG_TABLE_NAME = 'usage' AND TG_OP = 'DELETE')
ON CONFLICT (transaction_id, projection_version, granularity, bucket_start)
DO UPDATE SET unrecoverable = stats_overview_dirty_events.unrecoverable OR EXCLUDED.unrecoverable;
END LOOP;
RETURN NULL;
END $$;
DROP TRIGGER IF EXISTS overview_usage_dirty ON public.usage;
CREATE TRIGGER overview_usage_dirty AFTER INSERT OR UPDATE OR DELETE ON public.usage
FOR EACH ROW EXECUTE FUNCTION public.overview_mark_usage_bucket();
DROP TRIGGER IF EXISTS overview_settlement_dirty ON public.usage_settlement_snapshots;
CREATE TRIGGER overview_settlement_dirty AFTER INSERT OR UPDATE OR DELETE ON public.usage_settlement_snapshots
FOR EACH ROW EXECUTE FUNCTION public.overview_mark_usage_bucket();
DROP TRIGGER IF EXISTS overview_attribution_dirty ON public.usage_attribution_snapshots;
CREATE TRIGGER overview_attribution_dirty AFTER INSERT OR UPDATE OR DELETE ON public.usage_attribution_snapshots
FOR EACH ROW EXECUTE FUNCTION public.overview_mark_usage_bucket();
CREATE OR REPLACE FUNCTION public.overview_capture_usage_identity() RETURNS trigger
LANGUAGE plpgsql AS $$
DECLARE actor text; owner text; kind text; source text; standalone boolean;
BEGIN
SELECT id INTO owner FROM public.users WHERE id=NEW.user_id AND NOT is_deleted;
SELECT COALESCE(k.is_standalone,
CASE WHEN jsonb_typeof(NEW.request_metadata::jsonb #> '{analytics_attribution,is_standalone}')='boolean'
THEN (NEW.request_metadata #>> '{analytics_attribution,is_standalone}')::boolean END,
CASE WHEN jsonb_typeof(NEW.request_metadata::jsonb->'api_key_is_standalone')='boolean'
THEN (NEW.request_metadata->>'api_key_is_standalone')::boolean END,
CASE WHEN a.attribution_source='user_account' THEN false
WHEN a.attribution_source='standalone_key' THEN true END,
CASE WHEN NEW.api_key_id IS NULL THEN false END)
INTO standalone FROM (SELECT 1) seed
LEFT JOIN public.api_keys k ON k.id=NEW.api_key_id
LEFT JOIN public.usage_attribution_snapshots a ON a.request_id=NEW.request_id;
kind := CASE WHEN owner IS NULL THEN 'unknown' WHEN standalone THEN 'standalone'
WHEN NOT standalone THEN 'employee' ELSE 'unknown' END;
actor := CASE WHEN kind='employee' THEN owner END;
source := CASE kind WHEN 'employee' THEN 'user_account' WHEN 'standalone' THEN 'standalone_key' ELSE 'unknown' END;
INSERT INTO public.usage_attribution_snapshots(request_id, actor_user_id, credential_owner_id,
attribution_kind, attribution_source, record_kind, parent_request_id, attribution_revision)
VALUES (NEW.request_id, actor, owner, kind, source,
COALESCE(NEW.request_metadata #>> '{analytics_attribution,record_kind}', 'request'),
NEW.request_metadata #>> '{analytics_attribution,parent_request_id}', 2)
ON CONFLICT (request_id) DO UPDATE SET actor_user_id = EXCLUDED.actor_user_id,
credential_owner_id = EXCLUDED.credential_owner_id, attribution_kind = EXCLUDED.attribution_kind,
attribution_source = EXCLUDED.attribution_source,
record_kind = COALESCE(NEW.request_metadata #>> '{analytics_attribution,record_kind}', usage_attribution_snapshots.record_kind),
parent_request_id = COALESCE(EXCLUDED.parent_request_id, usage_attribution_snapshots.parent_request_id),
attribution_revision = EXCLUDED.attribution_revision, recorded_at = NOW()
WHERE usage_attribution_snapshots.attribution_revision <= EXCLUDED.attribution_revision;
RETURN NULL;
END $$;
DROP TRIGGER IF EXISTS overview_usage_identity ON public.usage;
CREATE TRIGGER overview_usage_identity AFTER INSERT OR UPDATE OF user_id, api_key_id, request_metadata ON public.usage
FOR EACH ROW EXECUTE FUNCTION public.overview_capture_usage_identity();
CREATE OR REPLACE FUNCTION public.overview_capture_failure() RETURNS trigger
LANGUAGE plpgsql AS $$
BEGIN
IF NEW.status = 'completed' THEN
NEW.failure_origin := NULL; NEW.failure_stage := NULL; NEW.failure_reason := NULL;
NEW.failure_schema_version := NULL; RETURN NEW;
END IF;
IF NEW.request_metadata #>> '{analytics_failure,origin}' IN ('client','gateway','upstream','transport','unknown') THEN
NEW.failure_origin := NEW.request_metadata #>> '{analytics_failure,origin}';
NEW.failure_stage := NEW.request_metadata #>> '{analytics_failure,stage}';
NEW.failure_reason := NEW.request_metadata #>> '{analytics_failure,reason}';
NEW.failure_schema_version := 1;
END IF;
RETURN NEW;
END $$;
DROP TRIGGER IF EXISTS overview_usage_failure ON public.usage;
CREATE TRIGGER overview_usage_failure BEFORE INSERT OR UPDATE OF request_metadata, status ON public.usage
FOR EACH ROW EXECUTE FUNCTION public.overview_capture_failure();
CREATE OR REPLACE FUNCTION public.overview_anonymize_user() RETURNS trigger
LANGUAGE plpgsql AS $$
BEGIN
IF TG_OP = 'DELETE' OR NEW.is_deleted THEN
UPDATE public.usage_attribution_snapshots SET actor_user_id = NULL, credential_owner_id = NULL,
attribution_kind = 'unknown', attribution_source = 'unknown', attribution_revision = attribution_revision + 10
WHERE actor_user_id = OLD.id OR credential_owner_id = OLD.id;
UPDATE public.usage SET request_metadata = (request_metadata::jsonb #- '{analytics_attribution,actor_user_id}')::json
WHERE request_metadata #>> '{analytics_attribution,actor_user_id}' = OLD.id;
DELETE FROM public.stats_overview_hourly WHERE dimensions->>'actor_user_id'=OLD.id OR dimensions->>'credential_owner_id'=OLD.id;
DELETE FROM public.stats_overview_daily WHERE dimensions->>'actor_user_id'=OLD.id OR dimensions->>'credential_owner_id'=OLD.id;
END IF;
RETURN NULL;
END $$;
DROP TRIGGER IF EXISTS overview_user_anonymize ON public.users;
CREATE TRIGGER overview_user_anonymize AFTER DELETE OR UPDATE OF is_deleted ON public.users
FOR EACH ROW EXECUTE FUNCTION public.overview_anonymize_user();
CREATE OR REPLACE FUNCTION public.overview_delete_attribution() RETURNS trigger
LANGUAGE plpgsql AS $$
BEGIN
DELETE FROM public.usage_attribution_snapshots WHERE request_id = OLD.request_id;
RETURN OLD;
END $$;
DROP TRIGGER IF EXISTS overview_usage_delete_attribution ON public.usage;
CREATE TRIGGER overview_usage_delete_attribution BEFORE DELETE ON public.usage
FOR EACH ROW EXECUTE FUNCTION public.overview_delete_attribution();
CREATE OR REPLACE VIEW public.usage_analytics_facts_v1 AS
SELECT u.request_id, COALESCE(u.id, u.request_id) AS id, u.created_at,
CASE WHEN identity.owner_id IS NOT NULL AND identity.is_standalone=false THEN identity.owner_id END AS actor_user_id,
identity.owner_id AS credential_owner_id,
CASE WHEN identity.owner_id IS NULL THEN 'unknown' WHEN identity.is_standalone THEN 'standalone'
WHEN NOT identity.is_standalone THEN 'employee' ELSE 'unknown' END AS attribution_kind,
CASE WHEN identity.owner_id IS NULL THEN 'unknown' WHEN identity.is_standalone THEN 'standalone_key'
WHEN NOT identity.is_standalone THEN 'user_account' ELSE 'unknown' END AS attribution_source,
COALESCE(a.record_kind, 'request') AS record_kind, a.parent_request_id,
u.api_key_id, u.model, u.target_model, u.provider_id, u.provider_name,
u.api_format, u.endpoint_kind, u.request_type, u.is_stream, u.has_format_conversion,
u.status, u.status_code, u.error_category, u.failure_origin, u.failure_stage, u.failure_reason,
u.failure_schema_version, u.response_time_ms, u.first_byte_time_ms,
COALESCE(s.billing_status, u.billing_status) AS settlement_status,
COALESCE(metadata.value->'usage_available', 'true'::jsonb) <> 'false'::jsonb AS usage_available,
COALESCE(metadata.value->'usage_pricing_available', 'true'::jsonb) <> 'false'::jsonb
AND (s.billing_total_cost_usd IS NOT NULL OR COALESCE(s.billing_status, u.billing_status) = 'settled') AS pricing_available,
CASE WHEN COALESCE(metadata.value->'usage_available', 'true'::jsonb) <> 'false'::jsonb
THEN b.input_tokens END AS input_tokens,
CASE WHEN COALESCE(metadata.value->'usage_available', 'true'::jsonb) <> 'false'::jsonb
THEN b.output_tokens END AS output_tokens,
CASE WHEN COALESCE(metadata.value->'usage_available', 'true'::jsonb) <> 'false'::jsonb
THEN b.total_tokens END AS total_tokens,
CASE WHEN COALESCE(metadata.value->'usage_available', 'true'::jsonb) <> 'false'::jsonb
THEN b.cache_read_input_tokens END AS cache_read_input_tokens,
CASE WHEN COALESCE(metadata.value->'usage_available', 'true'::jsonb) <> 'false'::jsonb
THEN b.cache_creation_input_tokens END AS cache_creation_input_tokens,
CASE WHEN COALESCE(metadata.value->'usage_pricing_available', 'true'::jsonb) <> 'false'::jsonb
AND (s.billing_total_cost_usd IS NOT NULL OR COALESCE(s.billing_status, u.billing_status) = 'settled')
THEN round(COALESCE(s.billing_total_cost_usd::numeric, u.total_cost_usd::numeric), 8) END AS rated_amount,
CASE WHEN COALESCE(metadata.value->'usage_pricing_available', 'true'::jsonb) <> 'false'::jsonb
AND (s.billing_actual_total_cost_usd IS NOT NULL OR COALESCE(s.billing_status, u.billing_status) = 'settled')
THEN round(COALESCE(s.billing_actual_total_cost_usd::numeric, u.actual_total_cost_usd::numeric), 8) END AS billable_amount,
s.quota_covered_amount_usd AS quota_covered_amount,
s.wallet_consumed_amount_usd AS wallet_consumed_amount,
s.wallet_debit_amount_usd AS wallet_debit_amount,
s.wallet_recharge_debit_usd AS wallet_recharge_debit_amount,
s.wallet_gift_debit_usd AS wallet_gift_debit_amount,
s.wallet_overdraft_usd AS wallet_overdraft_amount,
s.allocation_status, s.finalized_at AS settled_at,
CASE WHEN s.billing_total_cost_usd IS NOT NULL THEN 'settlement_snapshot' ELSE 'legacy_float' END AS amount_source,
b.upstream_is_stream,
CASE WHEN metadata.value #>> '{analytics_measurement,source}' IN ('reported','estimated','mixed')
THEN metadata.value #>> '{analytics_measurement,source}' ELSE 'unknown' END AS token_source,
CASE WHEN COALESCE(metadata.value->'usage_available','true'::jsonb) <> 'false'::jsonb
AND COALESCE(metadata.value->'usage_pricing_available','true'::jsonb) <> 'false'::jsonb
AND s.input_price_per_1m IS NOT NULL AND s.billing_cache_read_cost_usd IS NOT NULL
THEN round(s.input_price_per_1m::numeric * b.cache_read_input_tokens::numeric / 1000000,8) END AS cache_estimated_full_cost_amount,
CASE WHEN COALESCE(metadata.value->'usage_available','true'::jsonb) <> 'false'::jsonb
AND COALESCE(metadata.value->'usage_pricing_available','true'::jsonb) <> 'false'::jsonb
AND s.input_price_per_1m IS NOT NULL AND s.billing_cache_read_cost_usd IS NOT NULL
THEN round(s.billing_cache_read_cost_usd::numeric,8) END AS cache_read_cost_amount,
CASE WHEN COALESCE(metadata.value->'usage_available','true'::jsonb) <> 'false'::jsonb
AND COALESCE(metadata.value->'usage_pricing_available','true'::jsonb) <> 'false'::jsonb
AND s.input_price_per_1m IS NOT NULL AND s.billing_cache_creation_cost_usd IS NOT NULL
THEN round(s.billing_cache_creation_cost_usd::numeric,8) END AS cache_creation_cost_amount
FROM public.usage u
-- OFFSET 0 keeps this projection from being flattened: large metadata is
-- detoasted and parsed once per request, rather than once per metric expression.
CROSS JOIN LATERAL (SELECT u.request_metadata::jsonb AS value OFFSET 0) metadata
LEFT JOIN public.usage_settlement_snapshots s USING (request_id)
LEFT JOIN public.usage_attribution_snapshots a USING (request_id)
JOIN public.usage_billing_facts b USING (request_id)
LEFT JOIN public.api_keys k ON k.id=u.api_key_id
CROSS JOIN LATERAL (
SELECT CASE WHEN a.request_id IS NOT NULL THEN a.credential_owner_id
WHEN EXISTS (SELECT 1 FROM public.users WHERE id=u.user_id AND NOT is_deleted) THEN u.user_id END AS owner_id,
COALESCE(k.is_standalone,
CASE WHEN jsonb_typeof(metadata.value #> '{analytics_attribution,is_standalone}')='boolean'
THEN (metadata.value #>> '{analytics_attribution,is_standalone}')::boolean END,
CASE WHEN jsonb_typeof(metadata.value->'api_key_is_standalone')='boolean'
THEN (metadata.value->>'api_key_is_standalone')::boolean END,
CASE WHEN a.attribution_source='user_account' THEN false
WHEN a.attribution_source='standalone_key' THEN true END,
CASE WHEN u.api_key_id IS NULL THEN false END) AS is_standalone
) identity;