mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-06 01:17:46 +08:00
Merge pull request #879 from Kayphoon/feat/gemini-usage-reasoning-badges
feat(usage): surface Gemini thinkingConfig as reasoning effort
This commit is contained in:
@@ -1926,6 +1926,29 @@ mod tests {
|
||||
assert_eq!(active["reasoning_effort"], "max");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn user_usage_payloads_expose_gemini_thinking_config_reasoning_mapping() {
|
||||
let item = StoredRequestUsageAudit {
|
||||
request_body: Some(json!({
|
||||
"generationConfig": {
|
||||
"thinkingConfig": { "includeThoughts": true, "thinkingLevel": "HIGH" }
|
||||
}
|
||||
})),
|
||||
provider_request_body: Some(json!({
|
||||
"generationConfig": { "thinkingConfig": { "thinkingBudget": 8192 } }
|
||||
})),
|
||||
..sample_usage("completed")
|
||||
};
|
||||
|
||||
let record = build_users_me_usage_record_payload(&item, false, &BTreeMap::new(), false);
|
||||
let active = build_users_me_usage_active_payload(&item);
|
||||
|
||||
assert_eq!(record["requested_reasoning_effort"], "high");
|
||||
assert_eq!(active["requested_reasoning_effort"], "high");
|
||||
assert_eq!(record["reasoning_effort"], "xhigh");
|
||||
assert_eq!(active["reasoning_effort"], "xhigh");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn user_usage_payloads_expose_websocket_transport() {
|
||||
let item = StoredRequestUsageAudit {
|
||||
|
||||
@@ -53,6 +53,75 @@ pub fn extract_provider_reasoning_effort_from_body(value: Option<&Value>) -> Opt
|
||||
.and_then(Value::as_str)
|
||||
})
|
||||
.and_then(normalize_provider_reasoning_effort)
|
||||
.or_else(|| {
|
||||
// Gemini also nests its payload one level down, so both the flat
|
||||
// `generateContent` body and the `v1internal` envelope that carries it are read.
|
||||
extract_gemini_reasoning_effort_from_body(object).or_else(|| {
|
||||
object
|
||||
.get("request")
|
||||
.and_then(Value::as_object)
|
||||
.and_then(extract_gemini_reasoning_effort_from_body)
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// Gemini `generateContent` states its reasoning depth inside
|
||||
/// `generationConfig.thinkingConfig`, either as a symbolic `thinkingLevel` or as a token
|
||||
/// `thinkingBudget`. Both camelCase and snake_case spellings are read so that a captured client
|
||||
/// body and a converted provider body resolve to the same label.
|
||||
///
|
||||
/// `includeThoughts` alone is a visibility flag, not a depth, so it never produces a label.
|
||||
fn extract_gemini_reasoning_effort_from_body(
|
||||
object: &serde_json::Map<String, Value>,
|
||||
) -> Option<String> {
|
||||
let generation_config = object
|
||||
.get("generationConfig")
|
||||
.or_else(|| object.get("generation_config"))
|
||||
.and_then(Value::as_object)?;
|
||||
let thinking_config = generation_config
|
||||
.get("thinkingConfig")
|
||||
.or_else(|| generation_config.get("thinking_config"))
|
||||
.and_then(Value::as_object)?;
|
||||
|
||||
if let Some(level) = thinking_config
|
||||
.get("thinkingLevel")
|
||||
.or_else(|| thinking_config.get("thinking_level"))
|
||||
.and_then(Value::as_str)
|
||||
.and_then(normalize_gemini_thinking_level)
|
||||
{
|
||||
return Some(level);
|
||||
}
|
||||
|
||||
thinking_config
|
||||
.get("thinkingBudget")
|
||||
.or_else(|| thinking_config.get("thinking_budget"))
|
||||
.and_then(Value::as_u64)
|
||||
.map(|budget| {
|
||||
// `0` disables reasoning outright. The shared budget ladder collapses it into `low`,
|
||||
// which would report an explicitly disabled request as a shallow one.
|
||||
if budget == 0 {
|
||||
"none".to_string()
|
||||
} else {
|
||||
aether_ai_formats::formats::openai::shared::map_thinking_budget_to_openai_reasoning_effort(budget)
|
||||
.to_string()
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Gemini also emits the protobuf enum spelling (`THINKING_LEVEL_HIGH`); the level itself is what
|
||||
/// the badge vocabulary understands, so the enum prefix is stripped before normalizing.
|
||||
///
|
||||
/// `THINKING_LEVEL_UNSPECIFIED` is the enum's "no explicit level" member, not a depth. It is
|
||||
/// rejected rather than surfaced, otherwise the badge would read `unspecified`.
|
||||
fn normalize_gemini_thinking_level(value: &str) -> Option<String> {
|
||||
let normalized = value.trim().to_ascii_lowercase();
|
||||
let normalized = normalized
|
||||
.strip_prefix("thinking_level_")
|
||||
.unwrap_or(normalized.as_str());
|
||||
if normalized == "unspecified" {
|
||||
return None;
|
||||
}
|
||||
normalize_provider_reasoning_effort(normalized)
|
||||
}
|
||||
|
||||
fn normalize_provider_reasoning_effort(value: &str) -> Option<String> {
|
||||
@@ -3516,6 +3585,157 @@ mod tests {
|
||||
assert_eq!(usage.provider_service_tier(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gemini_thinking_level_supplies_reasoning_effort_for_both_body_spellings() {
|
||||
let mut usage = sample_usage();
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generationConfig": {
|
||||
"thinkingConfig": { "includeThoughts": true, "thinkingLevel": "HIGH" }
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("high"));
|
||||
|
||||
// The converted provider body keeps snake_case keys, and the client body may carry the
|
||||
// protobuf enum spelling. Both must land on the same badge vocabulary.
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generation_config": {
|
||||
"thinking_config": { "thinking_level": "thinking_level_medium" }
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("medium"));
|
||||
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generationConfig": {
|
||||
"thinkingConfig": { "thinkingLevel": " low " }
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("low"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gemini_thinking_budget_supplies_reasoning_effort_without_collapsing_zero() {
|
||||
let mut usage = sample_usage();
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generationConfig": { "thinkingConfig": { "thinkingBudget": 8192 } }
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("xhigh"));
|
||||
|
||||
// `0` disables reasoning. The shared budget ladder maps 0..=1664 to `low`, which would
|
||||
// report an explicitly disabled request as shallow, so the Gemini path reports `none`.
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generationConfig": { "thinkingConfig": { "thinkingBudget": 0 } }
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("none"));
|
||||
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generation_config": { "thinking_config": { "thinking_budget": 1280 } }
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("low"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gemini_thinking_config_without_level_or_budget_yields_no_reasoning_effort() {
|
||||
let mut usage = sample_usage();
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generationConfig": { "thinkingConfig": { "includeThoughts": true } }
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort(), None);
|
||||
|
||||
// A level-less, budget-less config must not fall back to metadata either: the captured
|
||||
// body is authoritative and it says nothing about depth.
|
||||
usage.request_metadata = Some(json!({ "provider_reasoning_effort": "max" }));
|
||||
assert_eq!(usage.provider_reasoning_effort(), None);
|
||||
|
||||
// `THINKING_LEVEL_UNSPECIFIED` is the enum's "no explicit level" member, not a depth.
|
||||
usage.request_metadata = None;
|
||||
usage.provider_request_body = Some(json!({
|
||||
"generationConfig": {
|
||||
"thinkingConfig": { "thinkingLevel": "THINKING_LEVEL_UNSPECIFIED" }
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort(), None);
|
||||
}
|
||||
|
||||
/// The v1internal envelope nests the real `generateContent` payload under `request`. This is
|
||||
/// the shape the Antigravity/Gemini CLI transports actually send upstream, so the extraction
|
||||
/// has to descend into it or every converted `openai:chat -> gemini` request loses its badge.
|
||||
#[test]
|
||||
fn gemini_thinking_config_is_read_from_the_v1internal_envelope() {
|
||||
let mut usage = sample_usage();
|
||||
usage.provider_request_body = Some(json!({
|
||||
"model": "gemini-3.8-flash-tiered",
|
||||
"project": "aicode-consumers",
|
||||
"requestId": "req-1",
|
||||
"requestType": "agent",
|
||||
"userAgent": "vscode/1.X.X (Antigravity/4.3.0)",
|
||||
"request": {
|
||||
"contents": [{ "role": "user", "parts": [{ "text": "hi" }] }],
|
||||
"generationConfig": {
|
||||
"maxOutputTokens": 65536,
|
||||
"thinkingConfig": { "includeThoughts": true, "thinkingLevel": "high" }
|
||||
}
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("high"));
|
||||
|
||||
usage.provider_request_body = Some(json!({
|
||||
"model": "gemini-3.8-flash-tiered",
|
||||
"request": {
|
||||
"generation_config": {
|
||||
"thinking_config": { "include_thoughts": true, "thinking_budget": 32768 }
|
||||
}
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("xhigh"));
|
||||
|
||||
usage.provider_request_body = Some(json!({
|
||||
"model": "gemini-3.8-flash-tiered",
|
||||
"request": {
|
||||
"generationConfig": { "thinkingConfig": { "thinkingBudget": 0 } }
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("none"));
|
||||
}
|
||||
|
||||
/// A converted request that carries only `maxOutputTokens` must stay badge-less rather than
|
||||
/// picking up a depth from somewhere else in the envelope.
|
||||
#[test]
|
||||
fn v1internal_envelope_without_thinking_config_yields_no_reasoning_effort() {
|
||||
let mut usage = sample_usage();
|
||||
usage.provider_request_body = Some(json!({
|
||||
"model": "gemini-3.8-flash-tiered",
|
||||
"project": "aicode-consumers",
|
||||
"request": {
|
||||
"contents": [{ "role": "user", "parts": [{ "text": "hi" }] }],
|
||||
"generationConfig": { "maxOutputTokens": 65536 }
|
||||
}
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gemini_thinking_config_does_not_shadow_explicit_effort_fields() {
|
||||
let mut usage = sample_usage();
|
||||
usage.provider_request_body = Some(json!({
|
||||
"reasoning_effort": "max",
|
||||
"generationConfig": { "thinkingConfig": { "thinkingLevel": "low" } }
|
||||
}));
|
||||
|
||||
assert_eq!(usage.provider_reasoning_effort().as_deref(), Some("max"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn requested_and_provider_reasoning_efforts_remain_independent() {
|
||||
let mut usage = sample_usage();
|
||||
|
||||
@@ -744,6 +744,39 @@ mod tests {
|
||||
assert!(cleared.get("requested_reasoning_effort").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gemini_thinking_config_is_derived_into_client_and_provider_reasoning_metadata() {
|
||||
let client_body = json!({
|
||||
"generationConfig": {
|
||||
"thinkingConfig": { "includeThoughts": true, "thinkingLevel": "HIGH" }
|
||||
}
|
||||
});
|
||||
let provider_body = json!({
|
||||
"generation_config": {
|
||||
"thinking_config": { "thinking_budget": 8192 }
|
||||
}
|
||||
});
|
||||
|
||||
let metadata = attach_client_request_body_metadata(
|
||||
Some(json!({ "trace_id": "trace-1" })),
|
||||
Some(&client_body),
|
||||
)
|
||||
.expect("metadata should remain");
|
||||
assert_eq!(metadata["requested_reasoning_effort"], "high");
|
||||
|
||||
let metadata = attach_provider_request_body_metadata(
|
||||
Some(metadata),
|
||||
Some("gemini:generate_content"),
|
||||
Some("gemini-3.8-flash"),
|
||||
Some("gemini-3.8-flash"),
|
||||
Some(&provider_body),
|
||||
)
|
||||
.expect("metadata should remain");
|
||||
|
||||
assert_eq!(metadata["requested_reasoning_effort"], "high");
|
||||
assert_eq!(metadata["provider_reasoning_effort"], "xhigh");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn provider_request_body_metadata_uses_final_provider_body_as_source_of_truth() {
|
||||
let metadata = Some(json!({
|
||||
|
||||
@@ -635,6 +635,53 @@ describe('RequestDetailDrawer settlement pricing', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('shows the Gemini thinkingConfig reasoning effort in the model header', async () => {
|
||||
apiMocks.getRequestDetail.mockResolvedValue({
|
||||
...buildEmbeddingDetail(),
|
||||
id: 'usage-gemini-thinking',
|
||||
request_id: 'usage-gemini-thinking',
|
||||
model: 'gemini-3.8-flash',
|
||||
request_type: 'chat',
|
||||
requested_reasoning_effort: 'xhigh',
|
||||
reasoning_effort: 'high',
|
||||
request_body: {
|
||||
generationConfig: {
|
||||
thinkingConfig: { includeThoughts: true, thinkingLevel: 'HIGH' },
|
||||
},
|
||||
},
|
||||
provider_request_body: {
|
||||
generationConfig: { thinkingConfig: { thinkingBudget: 8192 } },
|
||||
},
|
||||
})
|
||||
|
||||
let isOpen!: Ref<boolean>
|
||||
const Host = defineComponent({
|
||||
setup() {
|
||||
isOpen = ref(false)
|
||||
return () => h(RequestDetailDrawer, {
|
||||
isOpen: isOpen.value,
|
||||
requestId: 'usage-gemini-thinking',
|
||||
})
|
||||
},
|
||||
})
|
||||
|
||||
const root = document.createElement('div')
|
||||
document.body.appendChild(root)
|
||||
const app = createApp(Host)
|
||||
app.mount(root)
|
||||
mountedApps.push({ app, root })
|
||||
|
||||
isOpen.value = true
|
||||
await nextTick()
|
||||
|
||||
await vi.waitFor(() => {
|
||||
expect(document.body.querySelector('[data-request-detail-model-display]')?.textContent)
|
||||
.toContain('gemini-3.8-flash')
|
||||
expect(document.body.querySelector('[data-request-detail-model-badge="reasoning"]')?.textContent?.trim())
|
||||
.toBe('xhigh -> high')
|
||||
})
|
||||
})
|
||||
|
||||
it('lets a newer final-provider summary clear facts cached from an earlier candidate', async () => {
|
||||
apiMocks.getRequestDetail.mockResolvedValue({
|
||||
...buildEmbeddingDetail(),
|
||||
|
||||
@@ -440,6 +440,42 @@ describe('UsageRecordsTable', () => {
|
||||
.toBe('Fast')
|
||||
})
|
||||
|
||||
it('shows the Gemini thinkingLevel reasoning effort next to the model name', () => {
|
||||
const root = mountUsageRecordsTable([buildRecord({
|
||||
model: 'gemini-3.8-flash',
|
||||
requested_reasoning_effort: 'high',
|
||||
reasoning_effort: 'high',
|
||||
})])
|
||||
|
||||
expect(root.textContent).toContain('gemini-3.8-flash')
|
||||
const badge = root.querySelector('[data-usage-model-badge="reasoning"]')
|
||||
expect(badge?.textContent?.trim()).toBe('high')
|
||||
expect(badge?.getAttribute('title')).toBe('Reasoning: high')
|
||||
})
|
||||
|
||||
it('shows the Gemini thinkingLevel mapping when request and provider disagree', () => {
|
||||
const root = mountUsageRecordsTable([buildRecord({
|
||||
model: 'gemini-3.8-flash',
|
||||
requested_reasoning_effort: 'xhigh',
|
||||
reasoning_effort: 'high',
|
||||
})])
|
||||
|
||||
expect(root.textContent).toContain('xhigh -> high')
|
||||
expect(root.querySelector('[data-usage-model-badge="reasoning"]')?.textContent?.trim())
|
||||
.toBe('xhigh -> high')
|
||||
})
|
||||
|
||||
it('shows a disabled Gemini thinkingBudget as none', () => {
|
||||
const root = mountUsageRecordsTable([buildRecord({
|
||||
model: 'gemini-3.8-flash',
|
||||
requested_reasoning_effort: null,
|
||||
reasoning_effort: 'none',
|
||||
})])
|
||||
|
||||
expect(root.querySelector('[data-usage-model-badge="reasoning"]')?.textContent?.trim())
|
||||
.toBe('none')
|
||||
})
|
||||
|
||||
it('shows request reasoning effort while the record is pending', () => {
|
||||
const root = mountUsageRecordsTable([buildRecord({
|
||||
status: 'pending',
|
||||
|
||||
Reference in New Issue
Block a user