From b296d46e9707823ba5defe784d1c86ebdf50a22a Mon Sep 17 00:00:00 2001 From: Kayphoon <109347466+Kayphoon@users.noreply.github.com> Date: Fri, 18 Sep 2026 04:44:47 +0000 Subject: [PATCH] feat(usage): surface Gemini thinkingConfig as reasoning effort MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Usage records show a reasoning badge next to the model name for OpenAI and Claude requests, but Gemini requests never got one. The extraction only read the OpenAI/Claude shapes (`reasoning_effort`, `reasoning.effort`, `output_config.effort`), while Gemini states its reasoning depth inside `generationConfig.thinkingConfig` — so nothing was written to the usage metadata and the list and detail views had no badge to render. Read the Gemini shape too, as a fallback after the existing three so the OpenAI and Claude paths are untouched: - `thinkingLevel` / `thinking_level` wins when present, trimmed and lowercased, with the protobuf enum prefix stripped so `THINKING_LEVEL_HIGH` resolves like `high`. - Otherwise `thinkingBudget` / `thinking_budget` goes through the existing shared budget ladder, yielding the same `low|medium|high|xhigh` vocabulary the badge already understands. - Both camelCase and snake_case spellings are read, so a captured client body and a converted provider body resolve to the same label. - `includeThoughts` alone is a visibility flag, not a depth, and produces no badge. Two cases are handled explicitly rather than through the shared ladder: - `thinkingBudget: 0` disables reasoning outright. The shared ladder maps `0..=1664` to `low`, which would report an explicitly disabled request as a shallow one, so it reports `none` instead. - `THINKING_LEVEL_UNSPECIFIED` is the enum's "no explicit level" member, not a depth; it is rejected rather than surfaced as an `unspecified` badge. The frontend needs no change: `UsageModelDisplay` already renders the badge whenever the fields are present, and keeps the `high -> xhigh` mapping format when the requested and upstream efforts differ. --- .../handlers/public/support/user_me_usage.rs | 23 ++ .../contracts/src/repository/usage/types.rs | 220 ++++++++++++++++++ .../runtime/src/request_metadata.rs | 33 +++ .../RequestDetailDrawer.pricing.spec.ts | 47 ++++ .../__tests__/UsageRecordsTable.spec.ts | 36 +++ 5 files changed, 359 insertions(+) diff --git a/apps/aether-gateway/src/handlers/public/support/user_me_usage.rs b/apps/aether-gateway/src/handlers/public/support/user_me_usage.rs index 5052254bc..b626a6312 100644 --- a/apps/aether-gateway/src/handlers/public/support/user_me_usage.rs +++ b/apps/aether-gateway/src/handlers/public/support/user_me_usage.rs @@ -1905,6 +1905,29 @@ mod tests { assert_eq!(active["reasoning_effort"], "max"); } + #[test] + fn user_usage_payloads_expose_gemini_thinking_config_reasoning_mapping() { + let item = StoredRequestUsageAudit { + request_body: Some(json!({ + "generationConfig": { + "thinkingConfig": { "includeThoughts": true, "thinkingLevel": "HIGH" } + } + })), + provider_request_body: Some(json!({ + "generationConfig": { "thinkingConfig": { "thinkingBudget": 8192 } } + })), + ..sample_usage("completed") + }; + + let record = build_users_me_usage_record_payload(&item, false, &BTreeMap::new(), false); + let active = build_users_me_usage_active_payload(&item); + + assert_eq!(record["requested_reasoning_effort"], "high"); + assert_eq!(active["requested_reasoning_effort"], "high"); + assert_eq!(record["reasoning_effort"], "xhigh"); + assert_eq!(active["reasoning_effort"], "xhigh"); + } + #[test] fn user_usage_payloads_expose_websocket_transport() { let item = StoredRequestUsageAudit { diff --git a/crates/aether-data/contracts/src/repository/usage/types.rs b/crates/aether-data/contracts/src/repository/usage/types.rs index a2b88639a..8d8ad98be 100644 --- a/crates/aether-data/contracts/src/repository/usage/types.rs +++ b/crates/aether-data/contracts/src/repository/usage/types.rs @@ -51,6 +51,75 @@ pub fn extract_provider_reasoning_effort_from_body(value: Option<&Value>) -> Opt .and_then(Value::as_str) }) .and_then(normalize_provider_reasoning_effort) + .or_else(|| { + // Gemini also nests its payload one level down, so both the flat + // `generateContent` body and the `v1internal` envelope that carries it are read. + extract_gemini_reasoning_effort_from_body(object).or_else(|| { + object + .get("request") + .and_then(Value::as_object) + .and_then(extract_gemini_reasoning_effort_from_body) + }) + }) +} + +/// Gemini `generateContent` states its reasoning depth inside +/// `generationConfig.thinkingConfig`, either as a symbolic `thinkingLevel` or as a token +/// `thinkingBudget`. Both camelCase and snake_case spellings are read so that a captured client +/// body and a converted provider body resolve to the same label. +/// +/// `includeThoughts` alone is a visibility flag, not a depth, so it never produces a label. +fn extract_gemini_reasoning_effort_from_body( + object: &serde_json::Map, +) -> Option { + let generation_config = object + .get("generationConfig") + .or_else(|| object.get("generation_config")) + .and_then(Value::as_object)?; + let thinking_config = generation_config + .get("thinkingConfig") + .or_else(|| generation_config.get("thinking_config")) + .and_then(Value::as_object)?; + + if let Some(level) = thinking_config + .get("thinkingLevel") + .or_else(|| thinking_config.get("thinking_level")) + .and_then(Value::as_str) + .and_then(normalize_gemini_thinking_level) + { + return Some(level); + } + + thinking_config + .get("thinkingBudget") + .or_else(|| thinking_config.get("thinking_budget")) + .and_then(Value::as_u64) + .map(|budget| { + // `0` disables reasoning outright. The shared budget ladder collapses it into `low`, + // which would report an explicitly disabled request as a shallow one. + if budget == 0 { + "none".to_string() + } else { + aether_ai_formats::formats::openai::shared::map_thinking_budget_to_openai_reasoning_effort(budget) + .to_string() + } + }) +} + +/// Gemini also emits the protobuf enum spelling (`THINKING_LEVEL_HIGH`); the level itself is what +/// the badge vocabulary understands, so the enum prefix is stripped before normalizing. +/// +/// `THINKING_LEVEL_UNSPECIFIED` is the enum's "no explicit level" member, not a depth. It is +/// rejected rather than surfaced, otherwise the badge would read `unspecified`. +fn normalize_gemini_thinking_level(value: &str) -> Option { + let normalized = value.trim().to_ascii_lowercase(); + let normalized = normalized + .strip_prefix("thinking_level_") + .unwrap_or(normalized.as_str()); + if normalized == "unspecified" { + return None; + } + normalize_provider_reasoning_effort(normalized) } fn normalize_provider_reasoning_effort(value: &str) -> Option { @@ -3239,6 +3308,157 @@ mod tests { assert_eq!(usage.provider_service_tier(), None); } + #[test] + fn gemini_thinking_level_supplies_reasoning_effort_for_both_body_spellings() { + let mut usage = sample_usage(); + usage.provider_request_body = Some(json!({ + "generationConfig": { + "thinkingConfig": { "includeThoughts": true, "thinkingLevel": "HIGH" } + } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("high")); + + // The converted provider body keeps snake_case keys, and the client body may carry the + // protobuf enum spelling. Both must land on the same badge vocabulary. + usage.provider_request_body = Some(json!({ + "generation_config": { + "thinking_config": { "thinking_level": "thinking_level_medium" } + } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("medium")); + + usage.provider_request_body = Some(json!({ + "generationConfig": { + "thinkingConfig": { "thinkingLevel": " low " } + } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("low")); + } + + #[test] + fn gemini_thinking_budget_supplies_reasoning_effort_without_collapsing_zero() { + let mut usage = sample_usage(); + usage.provider_request_body = Some(json!({ + "generationConfig": { "thinkingConfig": { "thinkingBudget": 8192 } } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("xhigh")); + + // `0` disables reasoning. The shared budget ladder maps 0..=1664 to `low`, which would + // report an explicitly disabled request as shallow, so the Gemini path reports `none`. + usage.provider_request_body = Some(json!({ + "generationConfig": { "thinkingConfig": { "thinkingBudget": 0 } } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("none")); + + usage.provider_request_body = Some(json!({ + "generation_config": { "thinking_config": { "thinking_budget": 1280 } } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("low")); + } + + #[test] + fn gemini_thinking_config_without_level_or_budget_yields_no_reasoning_effort() { + let mut usage = sample_usage(); + usage.provider_request_body = Some(json!({ + "generationConfig": { "thinkingConfig": { "includeThoughts": true } } + })); + + assert_eq!(usage.provider_reasoning_effort(), None); + + // A level-less, budget-less config must not fall back to metadata either: the captured + // body is authoritative and it says nothing about depth. + usage.request_metadata = Some(json!({ "provider_reasoning_effort": "max" })); + assert_eq!(usage.provider_reasoning_effort(), None); + + // `THINKING_LEVEL_UNSPECIFIED` is the enum's "no explicit level" member, not a depth. + usage.request_metadata = None; + usage.provider_request_body = Some(json!({ + "generationConfig": { + "thinkingConfig": { "thinkingLevel": "THINKING_LEVEL_UNSPECIFIED" } + } + })); + + assert_eq!(usage.provider_reasoning_effort(), None); + } + + /// The v1internal envelope nests the real `generateContent` payload under `request`. This is + /// the shape the Antigravity/Gemini CLI transports actually send upstream, so the extraction + /// has to descend into it or every converted `openai:chat -> gemini` request loses its badge. + #[test] + fn gemini_thinking_config_is_read_from_the_v1internal_envelope() { + let mut usage = sample_usage(); + usage.provider_request_body = Some(json!({ + "model": "gemini-3.8-flash-tiered", + "project": "aicode-consumers", + "requestId": "req-1", + "requestType": "agent", + "userAgent": "vscode/1.X.X (Antigravity/4.3.0)", + "request": { + "contents": [{ "role": "user", "parts": [{ "text": "hi" }] }], + "generationConfig": { + "maxOutputTokens": 65536, + "thinkingConfig": { "includeThoughts": true, "thinkingLevel": "high" } + } + } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("high")); + + usage.provider_request_body = Some(json!({ + "model": "gemini-3.8-flash-tiered", + "request": { + "generation_config": { + "thinking_config": { "include_thoughts": true, "thinking_budget": 32768 } + } + } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("xhigh")); + + usage.provider_request_body = Some(json!({ + "model": "gemini-3.8-flash-tiered", + "request": { + "generationConfig": { "thinkingConfig": { "thinkingBudget": 0 } } + } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("none")); + } + + /// A converted request that carries only `maxOutputTokens` must stay badge-less rather than + /// picking up a depth from somewhere else in the envelope. + #[test] + fn v1internal_envelope_without_thinking_config_yields_no_reasoning_effort() { + let mut usage = sample_usage(); + usage.provider_request_body = Some(json!({ + "model": "gemini-3.8-flash-tiered", + "project": "aicode-consumers", + "request": { + "contents": [{ "role": "user", "parts": [{ "text": "hi" }] }], + "generationConfig": { "maxOutputTokens": 65536 } + } + })); + + assert_eq!(usage.provider_reasoning_effort(), None); + } + + #[test] + fn gemini_thinking_config_does_not_shadow_explicit_effort_fields() { + let mut usage = sample_usage(); + usage.provider_request_body = Some(json!({ + "reasoning_effort": "max", + "generationConfig": { "thinkingConfig": { "thinkingLevel": "low" } } + })); + + assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("max")); + } + #[test] fn requested_and_provider_reasoning_efforts_remain_independent() { let mut usage = sample_usage(); diff --git a/crates/aether-usage/runtime/src/request_metadata.rs b/crates/aether-usage/runtime/src/request_metadata.rs index 33f7b5e0f..e7c2015eb 100644 --- a/crates/aether-usage/runtime/src/request_metadata.rs +++ b/crates/aether-usage/runtime/src/request_metadata.rs @@ -663,6 +663,39 @@ mod tests { assert!(cleared.get("requested_reasoning_effort").is_none()); } + #[test] + fn gemini_thinking_config_is_derived_into_client_and_provider_reasoning_metadata() { + let client_body = json!({ + "generationConfig": { + "thinkingConfig": { "includeThoughts": true, "thinkingLevel": "HIGH" } + } + }); + let provider_body = json!({ + "generation_config": { + "thinking_config": { "thinking_budget": 8192 } + } + }); + + let metadata = attach_client_request_body_metadata( + Some(json!({ "trace_id": "trace-1" })), + Some(&client_body), + ) + .expect("metadata should remain"); + assert_eq!(metadata["requested_reasoning_effort"], "high"); + + let metadata = attach_provider_request_body_metadata( + Some(metadata), + Some("gemini:generate_content"), + Some("gemini-3.8-flash"), + Some("gemini-3.8-flash"), + Some(&provider_body), + ) + .expect("metadata should remain"); + + assert_eq!(metadata["requested_reasoning_effort"], "high"); + assert_eq!(metadata["provider_reasoning_effort"], "xhigh"); + } + #[test] fn provider_request_body_metadata_uses_final_provider_body_as_source_of_truth() { let metadata = Some(json!({ diff --git a/frontend/src/features/usage/components/__tests__/RequestDetailDrawer.pricing.spec.ts b/frontend/src/features/usage/components/__tests__/RequestDetailDrawer.pricing.spec.ts index e72abbea3..91d38e160 100644 --- a/frontend/src/features/usage/components/__tests__/RequestDetailDrawer.pricing.spec.ts +++ b/frontend/src/features/usage/components/__tests__/RequestDetailDrawer.pricing.spec.ts @@ -631,6 +631,53 @@ describe('RequestDetailDrawer settlement pricing', () => { }) }) + it('shows the Gemini thinkingConfig reasoning effort in the model header', async () => { + apiMocks.getRequestDetail.mockResolvedValue({ + ...buildEmbeddingDetail(), + id: 'usage-gemini-thinking', + request_id: 'usage-gemini-thinking', + model: 'gemini-3.8-flash', + request_type: 'chat', + requested_reasoning_effort: 'xhigh', + reasoning_effort: 'high', + request_body: { + generationConfig: { + thinkingConfig: { includeThoughts: true, thinkingLevel: 'HIGH' }, + }, + }, + provider_request_body: { + generationConfig: { thinkingConfig: { thinkingBudget: 8192 } }, + }, + }) + + let isOpen!: Ref + const Host = defineComponent({ + setup() { + isOpen = ref(false) + return () => h(RequestDetailDrawer, { + isOpen: isOpen.value, + requestId: 'usage-gemini-thinking', + }) + }, + }) + + const root = document.createElement('div') + document.body.appendChild(root) + const app = createApp(Host) + app.mount(root) + mountedApps.push({ app, root }) + + isOpen.value = true + await nextTick() + + await vi.waitFor(() => { + expect(document.body.querySelector('[data-request-detail-model-display]')?.textContent) + .toContain('gemini-3.8-flash') + expect(document.body.querySelector('[data-request-detail-model-badge="reasoning"]')?.textContent?.trim()) + .toBe('xhigh -> high') + }) + }) + it('lets a newer final-provider summary clear facts cached from an earlier candidate', async () => { apiMocks.getRequestDetail.mockResolvedValue({ ...buildEmbeddingDetail(), diff --git a/frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts b/frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts index 58a8d5d1e..f207f59d5 100644 --- a/frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts +++ b/frontend/src/features/usage/components/__tests__/UsageRecordsTable.spec.ts @@ -439,6 +439,42 @@ describe('UsageRecordsTable', () => { .toBe('Fast') }) + it('shows the Gemini thinkingLevel reasoning effort next to the model name', () => { + const root = mountUsageRecordsTable([buildRecord({ + model: 'gemini-3.8-flash', + requested_reasoning_effort: 'high', + reasoning_effort: 'high', + })]) + + expect(root.textContent).toContain('gemini-3.8-flash') + const badge = root.querySelector('[data-usage-model-badge="reasoning"]') + expect(badge?.textContent?.trim()).toBe('high') + expect(badge?.getAttribute('title')).toBe('Reasoning: high') + }) + + it('shows the Gemini thinkingLevel mapping when request and provider disagree', () => { + const root = mountUsageRecordsTable([buildRecord({ + model: 'gemini-3.8-flash', + requested_reasoning_effort: 'xhigh', + reasoning_effort: 'high', + })]) + + expect(root.textContent).toContain('xhigh -> high') + expect(root.querySelector('[data-usage-model-badge="reasoning"]')?.textContent?.trim()) + .toBe('xhigh -> high') + }) + + it('shows a disabled Gemini thinkingBudget as none', () => { + const root = mountUsageRecordsTable([buildRecord({ + model: 'gemini-3.8-flash', + requested_reasoning_effort: null, + reasoning_effort: 'none', + })]) + + expect(root.querySelector('[data-usage-model-badge="reasoning"]')?.textContent?.trim()) + .toBe('none') + }) + it('shows request reasoning effort while the record is pending', () => { const root = mountUsageRecordsTable([buildRecord({ status: 'pending',