chore: pass fmt clippy and nextest checks

Format merged Rust sources, satisfy strict clippy lints, and restore documentation fixtures required by format registry tests.

Validation: cargo fmt --all --check; CI-scoped clippy with -D warnings; cargo nextest workspace excluding integration tests, 9747 passed and 47 skipped; targeted PostgreSQL analytics regression passed.
This commit is contained in:
elky
2026-10-02 17:35:08 +08:00
parent 625456ff24
commit 2075cd95de
13 changed files with 10656 additions and 80 deletions
@@ -267,9 +267,9 @@ impl History {
fn snapshot(&mut self, now_us: u64, started_at_us: i64) -> Value {
self.advance(now_us);
let mut providers: Vec<_> = self.providers.iter().collect();
providers.sort_unstable_by(|(a, _), (b, _)| a.cmp(b));
providers.sort_unstable_by_key(|(provider, _)| *provider);
let mut models: Vec<_> = self.models.iter().collect();
models.sort_unstable_by(|(a, _), (b, _)| a.cmp(b));
models.sort_unstable_by_key(|(model, _)| *model);
json!({
"observed_at": DateTime::from_timestamp_micros(started_at_us.saturating_add(self.through_us.min(i64::MAX as u64) as i64)),
"observed_from": DateTime::from_timestamp_micros(started_at_us),
@@ -180,7 +180,7 @@ pub(super) fn public_projection(
exclusion_policy: "client_cancelled_invalid_input_identity_or_quota_policy",
},
average_latency_ms: (metrics.latency_sample_count > 0)
.then(|| metrics.latency_sum_ms as f64 / metrics.latency_sample_count as f64),
.then(|| metrics.latency_sum_ms / metrics.latency_sample_count as f64),
latency_sample_count: metrics.latency_sample_count,
last_request_at: metrics.last_request_at_unix_ms.and_then(unix_ms_to_rfc3339),
timeline: Vec::new(),
@@ -1237,7 +1237,9 @@ async fn gateway_handles_admin_system_users_export_locally_with_trusted_admin_pr
assert!(payload["users"][0]["api_keys"][0]
.get("credential_kind")
.is_none());
assert!(payload["standalone_keys"][0].get("credential_kind").is_none());
assert!(payload["standalone_keys"][0]
.get("credential_kind")
.is_none());
assert!(payload["exported_at"].as_str().is_some());
assert_eq!(payload["user_groups"][0]["name"], "Restricted GPT");
assert!(payload["user_groups"][0].get("priority").is_none());
@@ -50,10 +50,10 @@ use chrono::{TimeZone, Utc};
const TEST_EMAIL_VERIFICATION_TOKEN: &str =
"test-email-verification-token-00000000000000000000000000000000";
#[path = "public_support/auth_cookie.rs"]
mod auth_cookie;
#[path = "public_support/announcement_user_list.rs"]
mod announcement_user_list;
#[path = "public_support/auth_cookie.rs"]
mod auth_cookie;
#[path = "public_support/dashboard.rs"]
mod dashboard;
#[path = "public_support/vscodex.rs"]
@@ -486,7 +486,10 @@ async fn live_overview_canonical_queries_and_dirty_rebuild() {
let state = sqlx::query("SELECT source_revision,built_revision FROM stats_bucket_state WHERE projection_version='overview-v2' AND granularity='hour' AND bucket_start=$1").bind(start).fetch_one(&pool).await.unwrap();
// Bucket state stays unchanged on the foreground write; pending events
// invalidate the projection before the background merger consumes them.
assert_eq!(state.get::<i64, _>("source_revision"), state.get::<i64, _>("built_revision"));
assert_eq!(
state.get::<i64, _>("source_revision"),
state.get::<i64, _>("built_revision")
);
let pending: bool = sqlx::query_scalar("SELECT EXISTS(SELECT 1 FROM stats_overview_dirty_events WHERE projection_version='overview-v2' AND granularity='hour' AND bucket_start=$1)")
.bind(start).fetch_one(&pool).await.unwrap();
assert!(pending);
@@ -607,12 +610,21 @@ async fn live_overview_dirty_events_merge_without_shared_writer_bucket_lock() {
.fetch_one(&pool)
.await
.unwrap();
assert_eq!(row, (2, "unrecoverable".into(), Some("retained usage facts were deleted".into())));
let remaining: i64 = sqlx::query_scalar("SELECT count(*) FROM stats_overview_dirty_events WHERE bucket_start=$1")
.bind(bucket)
.fetch_one(&pool)
.await
.unwrap();
assert_eq!(
row,
(
2,
"unrecoverable".into(),
Some("retained usage facts were deleted".into())
)
);
let remaining: i64 = sqlx::query_scalar(
"SELECT count(*) FROM stats_overview_dirty_events WHERE bucket_start=$1",
)
.bind(bucket)
.fetch_one(&pool)
.await
.unwrap();
assert_eq!(remaining, 0);
sqlx::query("DELETE FROM stats_bucket_state WHERE projection_version='overview-v2' AND granularity='hour' AND bucket_start=$1")
.bind(bucket)
@@ -63,8 +63,8 @@ mod analytics_tests;
mod attribution;
pub mod cleanup;
mod dashboard;
mod dashboard_summary;
mod dashboard_retention;
mod dashboard_summary;
#[cfg(test)]
mod dashboard_summary_tests;
mod health;
@@ -82,25 +82,29 @@ const FIND_USAGE_BODY_BLOB_BY_REF_SQL: &str = r#"SELECT CASE WHEN octet_length(p
const DELETE_USAGE_BODY_BLOB_SQL: &str = include_str!("queries/delete_usage_body_blob_sql.sql");
static USAGE_BODY_DECODE_SLOTS: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(4);
struct UsageAnalyticsDrilldown<'a> {
provider_id: Option<&'a str>,
api_key_id: Option<&'a str>,
request_id: Option<&'a str>,
attribution_kind: Option<&'a str>,
actor_user_id: Option<&'a str>,
endpoint_kind: Option<&'a str>,
request_type: Option<&'a str>,
slow_threshold_ms: Option<u64>,
has_format_conversion: Option<bool>,
}
fn push_usage_analytics_drilldown(
builder: &mut QueryBuilder<'_, Postgres>,
has_where: &mut bool,
provider_id: &Option<String>,
api_key_id: &Option<String>,
request_id: &Option<String>,
attribution_kind: &Option<String>,
actor_user_id: &Option<String>,
endpoint_kind: &Option<String>,
request_type: &Option<String>,
slow_threshold_ms: Option<u64>,
has_format_conversion: Option<bool>,
filters: UsageAnalyticsDrilldown<'_>,
) {
for (column, value) in [
("provider_id", provider_id),
("api_key_id", api_key_id),
("request_id", request_id),
("endpoint_kind", endpoint_kind),
("request_type", request_type),
("provider_id", filters.provider_id),
("api_key_id", filters.api_key_id),
("request_id", filters.request_id),
("endpoint_kind", filters.endpoint_kind),
("request_type", filters.request_type),
] {
if let Some(value) = value {
builder.push(if *has_where { " AND " } else { " WHERE " });
@@ -109,27 +113,27 @@ fn push_usage_analytics_drilldown(
.push("\"usage\".")
.push(column)
.push(" = ")
.push_bind(value.clone());
.push_bind(value.to_string());
}
}
for (column, value) in [
("attribution_kind", attribution_kind),
("actor_user_id", actor_user_id),
("attribution_kind", filters.attribution_kind),
("actor_user_id", filters.actor_user_id),
] {
if let Some(value) = value {
builder.push(if *has_where { " AND " } else { " WHERE " });
*has_where = true;
builder.push("EXISTS(SELECT 1 FROM usage_analytics_facts_v1 f WHERE f.request_id=\"usage\".request_id AND f.").push(column).push(" = ").push_bind(value.clone()).push(")");
builder.push("EXISTS(SELECT 1 FROM usage_analytics_facts_v1 f WHERE f.request_id=\"usage\".request_id AND f.").push(column).push(" = ").push_bind(value.to_string()).push(")");
}
}
if let Some(value) = slow_threshold_ms {
if let Some(value) = filters.slow_threshold_ms {
builder.push(if *has_where { " AND " } else { " WHERE " });
*has_where = true;
builder
.push("\"usage\".response_time_ms >= ")
.push_bind(i64::try_from(value).unwrap_or(i64::MAX));
}
if let Some(value) = has_format_conversion {
if let Some(value) = filters.has_format_conversion {
builder.push(if *has_where { " AND " } else { " WHERE " });
*has_where = true;
builder
@@ -3168,15 +3172,17 @@ ORDER BY request_count DESC, "usage".provider_name ASC
push_usage_analytics_drilldown(
&mut builder,
&mut has_where,
&query.provider_id,
&query.api_key_id,
&query.request_id,
&query.attribution_kind,
&query.actor_user_id,
&query.endpoint_kind,
&query.request_type,
query.slow_threshold_ms,
query.has_format_conversion,
UsageAnalyticsDrilldown {
provider_id: query.provider_id.as_deref(),
api_key_id: query.api_key_id.as_deref(),
request_id: query.request_id.as_deref(),
attribution_kind: query.attribution_kind.as_deref(),
actor_user_id: query.actor_user_id.as_deref(),
endpoint_kind: query.endpoint_kind.as_deref(),
request_type: query.request_type.as_deref(),
slow_threshold_ms: query.slow_threshold_ms,
has_format_conversion: query.has_format_conversion,
},
);
if let Some(user_id) = query.user_id.as_deref() {
builder.push(if has_where { " AND " } else { " WHERE " });
@@ -3294,15 +3300,17 @@ OR (\"usage\".error_message IS NOT NULL AND BTRIM(\"usage\".error_message) <> ''
push_usage_analytics_drilldown(
&mut builder,
&mut has_where,
&query.provider_id,
&query.api_key_id,
&query.request_id,
&query.attribution_kind,
&query.actor_user_id,
&query.endpoint_kind,
&query.request_type,
query.slow_threshold_ms,
query.has_format_conversion,
UsageAnalyticsDrilldown {
provider_id: query.provider_id.as_deref(),
api_key_id: query.api_key_id.as_deref(),
request_id: query.request_id.as_deref(),
attribution_kind: query.attribution_kind.as_deref(),
actor_user_id: query.actor_user_id.as_deref(),
endpoint_kind: query.endpoint_kind.as_deref(),
request_type: query.request_type.as_deref(),
slow_threshold_ms: query.slow_threshold_ms,
has_format_conversion: query.has_format_conversion,
},
);
if let Some(user_id) = query.user_id.as_deref() {
builder.push(if has_where { " AND " } else { " WHERE " });
@@ -3501,15 +3509,17 @@ OR (\"usage\".error_message IS NOT NULL AND BTRIM(\"usage\".error_message) <> ''
push_usage_analytics_drilldown(
&mut builder,
&mut has_where,
&query.provider_id,
&query.api_key_id,
&query.request_id,
&query.attribution_kind,
&query.actor_user_id,
&query.endpoint_kind,
&query.request_type,
query.slow_threshold_ms,
query.has_format_conversion,
UsageAnalyticsDrilldown {
provider_id: query.provider_id.as_deref(),
api_key_id: query.api_key_id.as_deref(),
request_id: query.request_id.as_deref(),
attribution_kind: query.attribution_kind.as_deref(),
actor_user_id: query.actor_user_id.as_deref(),
endpoint_kind: query.endpoint_kind.as_deref(),
request_type: query.request_type.as_deref(),
slow_threshold_ms: query.slow_threshold_ms,
has_format_conversion: query.has_format_conversion,
},
);
if let Some(user_id) = query.user_id.as_deref() {
builder.push(if has_where { " AND " } else { " WHERE " });
@@ -3616,15 +3626,17 @@ OR (\"usage\".error_message IS NOT NULL AND BTRIM(\"usage\".error_message) <> ''
push_usage_analytics_drilldown(
&mut builder,
&mut has_where,
&query.provider_id,
&query.api_key_id,
&query.request_id,
&query.attribution_kind,
&query.actor_user_id,
&query.endpoint_kind,
&query.request_type,
query.slow_threshold_ms,
query.has_format_conversion,
UsageAnalyticsDrilldown {
provider_id: query.provider_id.as_deref(),
api_key_id: query.api_key_id.as_deref(),
request_id: query.request_id.as_deref(),
attribution_kind: query.attribution_kind.as_deref(),
actor_user_id: query.actor_user_id.as_deref(),
endpoint_kind: query.endpoint_kind.as_deref(),
request_type: query.request_type.as_deref(),
slow_threshold_ms: query.slow_threshold_ms,
has_format_conversion: query.has_format_conversion,
},
);
if let Some(user_id) = query.user_id.as_deref() {
builder.push(if has_where { " AND " } else { " WHERE " });
@@ -10528,7 +10540,8 @@ impl UsageReadRepository for SqlxUsageReadRepository {
async fn query_dashboard_summary(
&self,
query: &aether_data_contracts::repository::usage::UsageDashboardAnalyticsQuery,
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError> {
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError>
{
Self::query_dashboard_summary(self, query).await
}
@@ -2,7 +2,7 @@ use super::analytics::{
analytics_additive_metrics_sql, dashboard_total_metrics_sql, push_analytics_filter,
};
use crate::error::SqlxResultExt;
use aether_data_contracts::{DataLayerError, repository::usage::*};
use aether_data_contracts::{repository::usage::*, DataLayerError};
use chrono::{DateTime, Utc};
use sqlx::{Postgres, QueryBuilder, Row};
@@ -27,13 +27,13 @@ use crate::lifecycle::bootstrap::postgres::{
EMPTY_DATABASE_SNAPSHOT_CUTOFF_VERSION, EMPTY_DATABASE_SNAPSHOT_SQL,
};
mod policy_nulls;
mod overview_dirty_events;
mod provider_expenses;
mod migration_deadlines;
mod overview_migration_safety;
mod legacy_overview_upgrade;
mod dashboard_user_anonymization;
mod legacy_overview_upgrade;
mod migration_deadlines;
mod overview_dirty_events;
mod overview_migration_safety;
mod policy_nulls;
mod provider_expenses;
/// A clean PostgreSQL database is bootstrapped from the schema snapshot first;
/// migrations after the privacy/security frontier are intentionally left
@@ -1285,7 +1285,8 @@ impl UsageReadRepository for InMemoryUsageReadRepository {
async fn query_dashboard_summary(
&self,
query: &aether_data_contracts::repository::usage::UsageDashboardAnalyticsQuery,
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError> {
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError>
{
self.dashboard_summary_query(query)
}
@@ -3472,7 +3473,9 @@ impl UsageWriteRepository for InMemoryUsageReadRepository {
finalized_at_unix_secs: usage.finalized_at_unix_secs,
};
self.dashboard_projection.write().expect("dashboard projection lock")
self.dashboard_projection
.write()
.expect("dashboard projection lock")
.record(&stored, &self.analytics_key_flags());
by_request_id.insert(stored.request_id.clone(), stored.clone());
if let Some(auth_api_keys) = self.auth_api_keys.as_ref() {
+238
View File
@@ -0,0 +1,238 @@
# Format Conversion Audit
Last audited: 2026-06-03
This audit tracks `source format -> Canonical -> target format` behavior. It is intentionally stricter than historical best-effort conversion.
Statuses:
- `native`: emitted as a target-native field without semantic change.
- `mapped`: converted through canonical/provider-specific mapping.
- `extension-preserved`: preserved in same-format canonical roundtrip or target-approved extension namespace.
- `transport-only`: audited client transport metadata intentionally omitted when the target has no compatible transport channel.
- `unaudited`: rejected because the source field is not in the audited provider schema inventory for cross-format conversion.
- `unsupported`: rejected because the request/field shape is outside the supported conversion surface, independent of schema drift.
- `lossy-blocked`: conversion fails closed.
- `invalid-enum`: conversion fails closed because a provider enum value is not valid for the target mapping.
Full schema field coverage is tracked in `docs/api/format-field-coverage-matrix.md`. That matrix is generated from the schema inventory in `docs/api/provider-interface-definitions.md` by `python3 docs/api/generate_format_field_coverage.py` and gives every documented OpenAI, Claude, and Gemini schema field a handling status. “Handled” means mapped, same-format/native preserved, extension-preserved, blocked with a structured error, or explicitly marked outside the canonical conversion surface.
Provider schema refresh is not a runtime dependency. Same-format runtime paths do not use this matrix; they bypass canonical conversion. Same-format canonical roundtrip must preserve unrecognized provider fields through provider extension namespaces. Cross-format conversion is capability-based: only explicitly mapped fields are emitted, and newly discovered or unknown provider fields fail closed with `UnauditedField` until a lossless mapping is audited.
## Implemented Boundary Changes
| Area | Current behavior |
| --- | --- |
| Pure conversion API | `convert_request_pure` and response equivalents do not apply model override or stream policy. |
| Legacy conversion API | `convert_request` / `convert_response` are retained for migration and may still use legacy context behavior. |
| Same-format provider path | Bypasses canonical conversion and copies the parsed JSON object before transport edits. |
| Cross-format same-format-provider path | Uses `convert_request_pure`, then applies model/body/stream edits in transport. |
| Conversion errors | Added `UnauditedField`, `UnsupportedField`, `InvalidEnumValue`, `LossyConversionBlocked`, and `InvalidTargetField`. |
| Reporting | Added `ConversionReport` with field statuses. Runtime reports remain conversion-operation oriented; exhaustive nested schema coverage is enforced by `format-field-coverage-matrix.md`. |
| Source schema coverage | Cross-format request conversion rejects unknown source root fields before emit. Every documented schema field is covered by the field coverage matrix. |
| Schema drift handling | Official schema changes are detected by regenerating the inventory/matrix. Runtime same-format remains passthrough; cross-format unknowns return `UnauditedField` until deliberately mapped. |
| Tool schema roundtrip | Claude `input_schema` and Gemini `functionDeclarations.parameters` preserve raw same-format schema through provider-specific extensions. |
| Tool result ids | Chat `tool_call_id`, Responses `call_id`, Claude `tool_use_id`, and Gemini `functionResponse.id` are mapped through canonical tool IDs. |
## OpenAI Chat -> OpenAI Responses
| Chat field | Canonical handling | Responses output | Status |
| --- | --- | --- | --- |
| `model` | request identity | `model` | native |
| `messages` | canonical messages/instructions | `input`, `instructions` | mapped |
| `max_tokens` | generation max tokens | `max_output_tokens` | mapped |
| `max_completion_tokens` | generation max tokens | `max_output_tokens` | mapped |
| `temperature` | generation | `temperature` | native |
| `top_p` | generation | `top_p` | native |
| `top_logprobs` | generation | `top_logprobs` | native |
| `n` | generation but no Responses equivalent | none | lossy-blocked |
| `stop` | generation but no Responses equivalent | none | lossy-blocked |
| `presence_penalty` | generation but no Responses equivalent | none | lossy-blocked |
| `frequency_penalty` | generation but no Responses equivalent | none | lossy-blocked |
| `seed` | generation but no Responses equivalent | none | lossy-blocked |
| `logprobs` | generation but no Responses equivalent | none | lossy-blocked |
| `stream` | OpenAI extension | `stream` | mapped if explicit |
| `stream_options` | Chat-specific extension | none | lossy-blocked |
| `tools[].function.name` | canonical tool | `tools[].name` | mapped |
| `tools[].function.description` | canonical tool | `tools[].description` | mapped |
| `tools[].function.parameters` | canonical tool | `tools[].parameters` | mapped |
| `tools[].function.strict` | canonical tool strict | `tools[].strict` | mapped, implemented |
| assistant `tool_calls[].id` | canonical tool use id | `function_call.call_id` | mapped, implemented |
| tool message `tool_call_id` | canonical tool result id | `function_call_output.call_id` | mapped, implemented |
| `tool_choice` | canonical tool choice | `tool_choice` | mapped |
| `parallel_tool_calls` | canonical bool | `parallel_tool_calls` | native |
| `metadata` | canonical metadata | `metadata` | native |
| `response_format` | canonical response format | `text.format` | mapped |
| `reasoning_effort` | OpenAI enum | `reasoning.effort` | mapped; invalid enum blocked |
| `verbosity` | OpenAI extension | `text.verbosity` | mapped |
| `store` | OpenAI extension | `store` | extension-preserved |
| `service_tier` | OpenAI extension | `service_tier` | extension-preserved |
| `safety_identifier` | OpenAI extension | `safety_identifier` | extension-preserved |
| `prompt_cache_key` | OpenAI extension | `prompt_cache_key` | extension-preserved |
| `user` | legacy Chat user field | none | lossy-blocked |
| unknown top-level fields | source schema guard | none | unaudited |
## OpenAI Responses -> OpenAI Chat
| Responses field | Canonical handling | Chat output | Status |
| --- | --- | --- | --- |
| `model` | request identity | `model` | native |
| `input` | canonical messages/content/tool I/O | `messages` | mapped |
| `instructions` | canonical instruction/system | `messages` system/developer | mapped |
| `max_output_tokens` | generation max tokens | `max_completion_tokens` | mapped |
| `temperature` | generation | `temperature` | native |
| `top_p` | generation | `top_p` | native |
| `top_logprobs` | generation | `top_logprobs` | native |
| `metadata` | canonical metadata | `metadata` | native |
| `client_metadata` | Responses client transport metadata | none | transport-only; omitted |
| `parallel_tool_calls` | canonical bool | `parallel_tool_calls` | native |
| `text.format` | canonical response format | `response_format` | mapped |
| `text.verbosity` | Responses extension | `verbosity` | mapped |
| `tools[].type=function` | canonical tool | `tools[].type=function` | mapped |
| `tools[].name` | canonical tool | `tools[].function.name` | mapped |
| `tools[].parameters` | canonical tool | `tools[].function.parameters` | mapped |
| `tools[].strict` | canonical tool strict | `tools[].function.strict` | mapped, implemented |
| `function_call.call_id` | canonical tool use id | `tool_calls[].id` | mapped, implemented |
| `function_call_output.call_id` | canonical tool result id | tool message `tool_call_id` | mapped, implemented |
| `tools[].type=custom` | raw Responses tool | none | lossy-blocked to Chat |
| `tools[].type=web_search*` | raw Responses tool | none | lossy-blocked to Chat |
| `tool_choice` | canonical tool choice | `tool_choice` | mapped |
| `reasoning.effort` | OpenAI enum | `reasoning_effort` | mapped; invalid enum blocked |
| `reasoning.summary` | Responses-only | none | lossy-blocked |
| `reasoning.budget_tokens` | Responses-only | none | lossy-blocked |
| `stream` | Responses request transport policy | none | lossy-blocked; target stream policy is transport-owned |
| `include` | Responses-only | none | lossy-blocked; legacy emitter no longer leaks |
| `previous_response_id` | Responses-only | none | lossy-blocked; legacy emitter no longer leaks |
| `truncation` | Responses-only | none | lossy-blocked |
| `prompt` | Responses-only | none | lossy-blocked |
| `conversation` | Responses-only | none | lossy-blocked |
| `background` | Responses-only | none | lossy-blocked |
| `max_tool_calls` | Responses-only | none | lossy-blocked |
| unknown top-level fields | source schema guard | none | unaudited |
## Claude Messages <-> OpenAI Chat / Responses
Claude to OpenAI Chat, Claude to OpenAI Responses, and the reverse directions are included in the field coverage matrix. Runtime strict guards cover request root fields, provider extension namespaces, thinking/cache/tool-result hazards, and target generation-field gaps. Fields without a lossless target equivalent fail closed instead of being dropped.
High-risk fields:
| Claude field | OpenAI target risk | Required status |
| --- | --- | --- |
| `system` with cache blocks | Chat/Responses system instructions | same-format preserved; cross-format `cache_control` loss is blocked |
| `thinking` | OpenAI reasoning | Claude request-level thinking config maps to OpenAI reasoning; message-level thinking blocks are blocked for Responses |
| `cache_control` | OpenAI content/tool extensions | same-format preserved; cross-format blocked when no target equivalent exists |
| `tools[].input_schema` | OpenAI tool parameters | mapped; raw same-format schema preservation implemented |
| `tool_choice.disable_parallel_tool_use` | OpenAI `parallel_tool_calls` | mapped, implemented |
| `tool_result` multi-block content | OpenAI tool output/content | same-format preserved; cross-format to Chat/Responses is lossy-blocked |
| `metadata` | OpenAI metadata | mapped when the target has metadata |
| `container`, `inference_geo`, `service_tier` | OpenAI target has no audited equivalent | lossy-blocked unless a target-approved mapping is added |
## Gemini GenerateContent <-> OpenAI Chat / Responses / Claude
Gemini to OpenAI Chat, Gemini to OpenAI Responses, Gemini to Claude, and reverse generation paths are included in the field coverage matrix. Gemini-only request fields are preserved same-format and blocked cross-format unless the target mapping is explicitly audited.
High-risk fields:
| Gemini field | Target risk | Required status |
| --- | --- | --- |
| `contents[].parts[].thoughtSignature` | OpenAI/Claude thinking | Chat/Claude preserve; Responses cross-format is lossy-blocked |
| `tools[].functionDeclarations` | OpenAI/Claude tool schema | mapped; raw same-format `parameters` preservation implemented |
| `toolConfig.functionCallingConfig.allowedFunctionNames` | OpenAI/Claude tool choice | single-name mapping implemented; multi-name input is lossy-blocked |
| `toolConfig.functionCallingConfig.mode` | OpenAI/Claude tool choice enum | valid enum required; invalid values fail with `InvalidEnumValue` |
| `generationConfig.thinkingConfig.thinkingLevel` | OpenAI/Claude reasoning effort | low/medium/high mapping implemented; invalid values fail closed |
| `safetySettings` | OpenAI/Claude no direct equivalent | lossy-blocked |
| `cachedContent` | OpenAI/Claude no direct equivalent | lossy-blocked |
| `codeExecution` | OpenAI/Claude tool/builtin mismatch | lossy-blocked |
| `generationConfig.responseModalities` | OpenAI/Claude modality mismatch | lossy-blocked |
| `functionResponse.id` | tool result id | conversion preserves id; Gemini upstream cleanup is transport-layer edit only |
## Embedding And Rerank
Embedding and rerank request parse/emit capability and strict target guards are implemented. Provider schema fields outside these canonical conversion surfaces are marked `not-in-conversion-surface` in the field coverage matrix instead of being left implicit.
Embedding source capability:
| Source format | Parsed request shape | Canonical fields | Status |
| --- | --- | --- | --- |
| OpenAI Embedding | `model`, `input`, `encoding_format`, `dimensions`, `user`, `parameters`, `task` | OpenAI-like embedding | mapped |
| Jina Embedding | OpenAI-like plus provider extension namespace | OpenAI-like embedding | mapped |
| Doubao Embedding | OpenAI-like `model` + text `input` | OpenAI-like embedding | mapped |
| Gemini Embedding | single `content.parts[].text` or batch `requests[]` | text input, `dimensions`, `task` | mapped |
| Aliyun Multimodal Embedding | `input.contents[]`, `parameters.dimension` | text/multimodal input, `dimensions`, `parameters` | mapped |
Embedding target guards:
| Target format | Accepted canonical fields | Blocked fields/cases | Status |
| --- | --- | --- | --- |
| OpenAI Embedding | text or token input, `encoding_format`, `dimensions`, `user` | multimodal input, `task`, generic `parameters` | lossy-blocked |
| Jina Embedding | text input, `dimensions`, `task`, `parameters` | token/multimodal input, `encoding_format`, `user` | lossy-blocked |
| Gemini Embedding | text input, `dimensions`, valid `taskType` | token/multimodal input, `encoding_format`, `user`, generic `parameters`, invalid `taskType` | lossy-blocked / invalid-enum |
| Doubao Embedding | text input, `dimensions` | token/multimodal input, `encoding_format`, `user`, `task`, generic `parameters` | lossy-blocked |
| Aliyun Multimodal Embedding | text or multimodal input, `dimensions`, `parameters` | token input, `encoding_format`, `user`, `task` | lossy-blocked |
Cross-format embedding invariants:
- Embedding formats can only convert to embedding formats.
- Unknown provider-specific embedding extension namespaces are blocked cross-format unless the namespace matches the target.
- Aliyun `parameters.dimension` maps to canonical `dimensions` and is not treated as generic `parameters`.
- Gemini batch embedding parse requires every batch item to share the same model, dimensions, and task.
Rerank first pass:
| Area | Current behavior | Status |
| --- | --- | --- |
| Source formats | OpenAI Rerank and Jina Rerank parse OpenAI-like `model`, `query`, `documents`, `top_n`, `return_documents` | mapped |
| Target formats | OpenAI Rerank and Jina Rerank emit OpenAI-like rerank bodies | mapped |
| Boundary | Rerank formats can only convert to rerank formats | lossy-blocked |
| Validation | Empty query/documents and `top_n=0` fail closed | invalid-target-field |
| Extensions | Unknown provider-specific rerank extension namespaces are blocked cross-format | unsupported |
## Sync Response Conversion
Cross-format sync response conversion now validates source stop/finish/status
enums before emitting a target body. Same-format runtime response passthrough is
still outside canonical conversion.
| Source field | Target risk | Current behavior | Status |
| --- | --- | --- | --- |
| Same-format response raw stop/status fields | canonical emitters would otherwise normalize unknown enum/status to default target stop values | raw OpenAI Chat `finish_reason`, OpenAI Responses `status`, Claude `stop_reason`/`stop_sequence`, and Gemini `finishReason` are preserved through provider extension metadata | extension-preserved |
| OpenAI Chat `choices[].finish_reason` | unknown value would otherwise emit as target normal stop | valid Chat enum required; unknown values fail with `InvalidEnumValue` | invalid-enum |
| OpenAI Responses `status` | `queued`, `in_progress`, and `cancelled` have no sync target equivalent | non-terminal valid states fail with `LossyConversionBlocked`; invalid states fail with `InvalidEnumValue` | lossy-blocked / invalid-enum |
| OpenAI Responses `incomplete_details.reason=content_filter` | previously mapped to max tokens/`length` | maps to canonical content filter and emits Chat `content_filter` / Claude `content_filtered` / Gemini `SAFETY` | mapped |
| Claude `stop_reason` | unknown value would otherwise emit as target normal stop | valid known stop enum required for cross-format conversion | invalid-enum |
| Gemini `candidates[].finishReason` | known-but-unmappable reasons would otherwise emit as target normal stop | mappable safety/max/stop reasons convert; known unmappable values such as `OTHER`, `MALFORMED_FUNCTION_CALL`, `UNEXPECTED_TOOL_CALL`, `MISSING_THOUGHT_SIGNATURE`, and `MALFORMED_RESPONSE` fail with `LossyConversionBlocked`; future unknown values fail with `InvalidEnumValue` | lossy-blocked / invalid-enum |
| Canonical `Unknown` stop reason | target emitters default to normal stop values | cross-format response conversion blocks canonical unknown stop reasons | lossy-blocked |
## Stream Conversion
Sixth batch first pass is implemented for unknown event handling and runtime
same-format boundaries. Sync response finish/status parity has a first strict
pass; stream finish-reason guardrails are implemented for unknown/unmappable
terminal reasons. Stream event schema fields are covered in the field coverage
matrix; provider-by-provider fixtures cover the runtime event behavior.
Current stream behavior:
| Area | Current behavior | Status |
| --- | --- | --- |
| Provider parsers | OpenAI Chat, OpenAI Responses, Claude, and Gemini unknown stream payloads become `CanonicalStreamEvent::UnknownEvent` | mapped |
| Cross-format stream matrix | Unknown canonical stream events emit a target-format error SSE with `unsupported_stream_event` and terminate conversion | lossy-blocked |
| Stream finish reason guard | Unknown OpenAI finish reasons, unknown Claude `stop_reason`, and Gemini known-but-unmappable `finishReason` values such as `OTHER` are preserved as raw canonical finish strings, then blocked by the matrix with `unsupported_finish_reason` | lossy-blocked |
| OpenAI Responses stream target | Canonical `length` and `content_filter` terminal reasons emit `response.incomplete` with `incomplete_details.reason=max_output_tokens` or `content_filter` instead of `response.completed` | mapped |
| Terminal observer | Unknown provider stream events increment `unknown_event_count`; OpenAI Responses failed events mark terminal error state | mapped |
| Stream -> sync aggregate | Unknown OpenAI Chat, OpenAI Responses, Claude, and Gemini stream events make the runtime finalize checked path return an error and block `body_json` fallback; legacy public aggregate helpers keep `Option` compatibility | lossy-blocked |
| Runtime strict fallback guard | `UnauditedField`, `InvalidEnumValue`, `UnsupportedField`, `LossyConversionBlocked`, and `InvalidTargetField` from registry response conversion are not allowed to fall through legacy conversion helpers | lossy-blocked |
| Runtime same-format stream | Same-format stream passthrough remains outside canonical conversion; stream policy edits are transport-layer only | native |
Stream fixture coverage:
| Provider stream | Covered fixture areas |
| --- | --- |
| OpenAI Chat | sync aggregation for text, tool call IDs/names/argument deltas, finish reason, and usage; cross-format unknown finish/event blocking |
| OpenAI Responses | text snapshot de-duplication, multi-part messages, reasoning/items, function calls, image generation calls, same-family stream sync, unknown event blocking |
| Claude Messages | thinking signatures, tool input deltas, cache/usage aggregation, media emission, unknown stop/event blocking |
| Gemini GenerateContent | text/media/signature aggregation, function calls/results, safety finish mapping, unknown parts/events, and unmappable finish reason blocking |
Matrix-level interception remains the authoritative runtime path for cross-format
unknown events; direct client emitters are covered only as provider/client
building blocks.
File diff suppressed because it is too large Load Diff
+548
View File
@@ -0,0 +1,548 @@
#!/usr/bin/env python3
"""Generate the provider schema field coverage matrix.
The input inventory is docs/api/provider-interface-definitions.md. Existing
coverage rows are reused so audited status/notes survive regeneration. Newly
introduced provider fields get conservative same-format/native and cross-format
fail-closed defaults until a human audits whether they deserve an explicit
mapping.
"""
from __future__ import annotations
import argparse
import dataclasses
from collections import Counter, defaultdict
from pathlib import Path
from typing import Iterable
ROOT = Path(__file__).resolve().parents[2]
DEFAULT_DEFINITIONS = ROOT / "docs/api/provider-interface-definitions.md"
DEFAULT_MATRIX = ROOT / "docs/api/format-field-coverage-matrix.md"
@dataclasses.dataclass(frozen=True)
class SourceField:
provider: str
schema: str
field: str
required: str
field_type: str
@dataclasses.dataclass(frozen=True)
class CoverageStatus:
surface: str
same_format_runtime: str
canonical_roundtrip: str
cross_format: str
notes: str
OPENAI_CHAT_MAPPED = {
"model",
"messages",
"max_tokens",
"max_completion_tokens",
"temperature",
"top_p",
"top_logprobs",
"tools",
"tool_choice",
"parallel_tool_calls",
"metadata",
"response_format",
"reasoning_effort",
"verbosity",
"store",
"service_tier",
"safety_identifier",
"prompt_cache_key",
"prompt_cache_retention",
"stream",
}
OPENAI_CHAT_BLOCKED = {
"n",
"stop",
"presence_penalty",
"frequency_penalty",
"seed",
"logprobs",
"stream_options",
"user",
"function_call",
"functions",
"logit_bias",
"modalities",
"prediction",
"audio",
"web_search_options",
}
OPENAI_RESPONSES_MAPPED = {
"model",
"input",
"instructions",
"max_output_tokens",
"temperature",
"top_p",
"top_logprobs",
"metadata",
"parallel_tool_calls",
"text",
"tools",
"tool_choice",
"reasoning",
"store",
"service_tier",
"safety_identifier",
"prompt_cache_key",
"prompt_cache_retention",
}
OPENAI_RESPONSES_BLOCKED = {
"include",
"previous_response_id",
"truncation",
"prompt",
"conversation",
"background",
"max_tool_calls",
"user",
"context_management",
"stream",
"stream_options",
}
CLAUDE_MAPPED_FIELDS = {
"id",
"type",
"role",
"text",
"content",
"source",
"name",
"description",
"input",
"input_schema",
"messages",
"model",
"max_tokens",
"system",
"temperature",
"top_p",
"top_k",
"stop_sequences",
"tool_choice",
"tools",
"metadata",
"thinking",
"output_config",
"usage",
"stop_reason",
"stop_sequence",
}
CLAUDE_PROVIDER_ONLY_FIELDS = {
"cache_control",
"container",
"inference_geo",
"service_tier",
"allowed_callers",
"allowed_domains",
"blocked_domains",
"defer_loading",
"max_uses",
"strict",
"user_location",
"citations",
"context",
"title",
"file_id",
"document_index",
"document_title",
"cited_text",
"caller",
}
def split_markdown_row(line: str) -> list[str]:
cells: list[str] = []
current: list[str] = []
escaped = False
for char in line:
if char == "|" and not escaped:
cells.append("".join(current).strip())
current.clear()
else:
current.append(char)
escaped = char == "\\" and not escaped
if escaped and char != "\\":
escaped = False
cells.append("".join(current).strip())
return cells
def strip_markdown_code(value: str) -> str:
value = value.strip()
if value.startswith("`") and value.endswith("`"):
value = value[1:-1]
return value.replace("\\|", "|")
def escape_markdown_cell(value: str) -> str:
return value.replace("|", "\\|")
def parse_schema_heading(line: str) -> str | None:
if not line.startswith("### `"):
return None
rest = line[len("### `") :]
schema, _, _ = rest.partition("`")
return schema or None
def parse_provider_definition_fields(definitions: str) -> list[SourceField]:
provider: str | None = None
schema: str | None = None
fields: list[SourceField] = []
for line in definitions.splitlines():
if line.startswith("## "):
if "OpenAI Schema" in line:
provider = "OpenAI"
elif "Claude / Anthropic TypeScript" in line:
provider = "Claude"
elif "Gemini Schema" in line:
provider = "Gemini"
else:
provider = None
schema = None
continue
if provider is None:
continue
if heading := parse_schema_heading(line):
schema = heading
continue
if schema is None or not line.startswith("| `"):
continue
cells = split_markdown_row(line)
if len(cells) < 4 or cells[2] not in {"是", "否"}:
continue
fields.append(
SourceField(
provider=provider,
schema=schema,
field=strip_markdown_code(cells[1]),
required=cells[2],
field_type=strip_markdown_code(cells[3]),
)
)
return fields
def parse_existing_coverage(
matrix: str,
) -> tuple[dict[tuple[str, str, str], CoverageStatus], dict[tuple[str, str], list[CoverageStatus]]]:
existing: dict[tuple[str, str, str], CoverageStatus] = {}
profiles: dict[tuple[str, str], list[CoverageStatus]] = defaultdict(list)
for line in matrix.splitlines():
if not line.startswith("| "):
continue
cells = split_markdown_row(line)
if len(cells) < 11 or cells[1] not in {"OpenAI", "Claude", "Gemini"}:
continue
status = CoverageStatus(
surface=cells[6],
same_format_runtime=cells[7],
canonical_roundtrip=cells[8],
cross_format=cells[9],
notes=cells[10],
)
provider = cells[1]
schema = strip_markdown_code(cells[2])
field = strip_markdown_code(cells[3])
existing[(provider, schema, field)] = status
profiles[(provider, schema)].append(status)
return existing, profiles
def most_common(values: Iterable[str]) -> str | None:
values = list(values)
if not values:
return None
return Counter(values).most_common(1)[0][0]
def openai_surface(schema: str) -> str:
if "CreateChatCompletion" in schema or "ChatCompletion" in schema:
return "openai:chat standard"
if "CreateEmbedding" in schema or "Embedding" in schema:
return "openai:embedding"
if any(token in schema for token in ("CreateImage", "EditImage", "Image", "Images")):
return "openai:image native-only"
if any(token in schema for token in ("Compact", "Compaction")):
return "openai:responses:compact native-only"
if any(
token in schema
for token in (
"Response",
"Input",
"Output",
"Tool",
"Reasoning",
"WebSearch",
"FileSearch",
"Computer",
"MCP",
"CodeInterpreter",
"Function",
"Custom",
"EasyInput",
"Prompt",
"Conversation",
"Annotation",
"Citation",
"LogProb",
"TopLogProb",
"Metadata",
"ServiceTier",
"Verbosity",
"TextResponse",
"ResponseFormat",
"Include",
"Modalities",
"ParallelToolCalls",
"StopConfiguration",
)
):
return "openai:responses standard"
return "openai auxiliary / not-in-conversion-surface"
def openai_default_status(field: SourceField, profile: list[CoverageStatus]) -> CoverageStatus:
surface = most_common(status.surface for status in profile) or openai_surface(field.schema)
if "not-in-conversion-surface" in surface or "native-only" in surface:
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip="not-in-conversion-surface",
cross_format="not-in-conversion-surface",
notes="not part of current canonical cross-format conversion; same-format runtime path remains provider-native when routed directly",
)
if field.schema == "CreateChatCompletionRequest":
if field.field in OPENAI_CHAT_MAPPED:
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip="mapped",
cross_format="mapped",
notes="Chat request field maps provider-specifically; target-incompatible cases fail closed",
)
if field.field in OPENAI_CHAT_BLOCKED:
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip="extension-preserved",
cross_format="lossy-blocked",
notes="Chat-only or provider-specific field has no audited lossless target equivalent",
)
if field.schema == "CreateResponse":
if field.field in OPENAI_RESPONSES_MAPPED:
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip="mapped",
cross_format="mapped",
notes="Responses request field maps provider-specifically; target-incompatible cases fail closed",
)
if field.field in OPENAI_RESPONSES_BLOCKED:
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip="extension-preserved",
cross_format="lossy-blocked",
notes="Responses-only field has no audited lossless Chat/Claude/Gemini target equivalent",
)
if profile:
cross_format = most_common(status.cross_format for status in profile) or "lossy-blocked"
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip=most_common(status.canonical_roundtrip for status in profile)
or "extension-preserved",
cross_format=cross_format,
notes=next(
(status.notes for status in profile if status.cross_format == cross_format),
"schema-level handling inherited from audited sibling fields",
),
)
return CoverageStatus(
surface=surface,
same_format_runtime="native",
canonical_roundtrip="extension-preserved",
cross_format="lossy-blocked",
notes="OpenAI documented field is preserved same-format; cross-format requires explicit target mapping or fails closed",
)
def claude_default_status(field: SourceField) -> CoverageStatus:
if "CountTokens" in field.schema:
return CoverageStatus(
surface="claude:messages/count_tokens native-only",
same_format_runtime="native",
canonical_roundtrip="not-in-conversion-surface",
cross_format="not-in-conversion-surface",
notes="count_tokens schemas are provider-native and outside canonical generation conversion",
)
if field.field in CLAUDE_MAPPED_FIELDS:
return CoverageStatus(
surface="claude:messages standard",
same_format_runtime="native",
canonical_roundtrip="mapped",
cross_format="mapped/lossy-blocked",
notes="Claude field maps where canonical and target support an equivalent; otherwise conversion fails closed",
)
if (
field.field in CLAUDE_PROVIDER_ONLY_FIELDS
or field.field.endswith("_tokens_details")
or "cache" in field.field
):
return CoverageStatus(
surface="claude:messages standard",
same_format_runtime="native",
canonical_roundtrip="extension-preserved",
cross_format="lossy-blocked",
notes="Claude provider-specific field is preserved same-format and blocked cross-format without an audited target equivalent",
)
return CoverageStatus(
surface="claude:messages standard",
same_format_runtime="native",
canonical_roundtrip="extension-preserved",
cross_format="lossy-blocked",
notes="Claude nested/provider-specific field is same-format preserved; cross-format requires explicit mapping or fails closed",
)
def gemini_default_status(field: SourceField, profile: list[CoverageStatus]) -> CoverageStatus:
if profile:
cross_format = most_common(status.cross_format for status in profile) or "lossy-blocked"
return CoverageStatus(
surface=most_common(status.surface for status in profile)
or "gemini:generate_content standard",
same_format_runtime="native",
canonical_roundtrip=most_common(status.canonical_roundtrip for status in profile)
or "extension-preserved",
cross_format=cross_format,
notes=next(
(status.notes for status in profile if status.cross_format == cross_format),
"Gemini field follows schema-level handling",
),
)
return CoverageStatus(
surface="gemini:generate_content standard",
same_format_runtime="native",
canonical_roundtrip="extension-preserved",
cross_format="lossy-blocked",
notes="Gemini documented field is preserved same-format; cross-format requires explicit mapping or fails closed",
)
def default_status(field: SourceField, profile: list[CoverageStatus]) -> CoverageStatus:
if field.provider == "OpenAI":
return openai_default_status(field, profile)
if field.provider == "Claude":
return claude_default_status(field)
if field.provider == "Gemini":
return gemini_default_status(field, profile)
raise ValueError(f"unsupported provider: {field.provider}")
def render_matrix(
fields: list[SourceField],
existing: dict[tuple[str, str, str], CoverageStatus],
profiles: dict[tuple[str, str], list[CoverageStatus]],
) -> str:
rows: list[str] = [
"# Format Field Coverage Matrix",
"",
"Last generated: 2026-06-03",
"",
"This file is generated from the schema inventory in `docs/api/provider-interface-definitions.md` and gives every documented schema field an explicit handling status. “处理到” here means the field is either mapped, preserved in same-format paths, rejected with a structured fail-closed error, or explicitly outside the current conversion surface. It does not mean every field can be cross-format converted.",
"",
"Provider schema updates do not require immediate conversion-code changes for runtime safety. Same-format runtime paths bypass canonical conversion, and same-format canonical roundtrip preserves provider extension fields. Cross-format conversion only enables fields with an audited semantic mapping; newly discovered or unknown provider fields default to structured fail-closed behavior until mapped.",
"",
"Regenerate with: `python3 docs/api/generate_format_field_coverage.py`.",
"",
"Statuses used in this matrix: `native`, `mapped`, `mapped/lossy-blocked`, `extension-preserved`, `unaudited`, `unsupported`, `invalid-enum`, `lossy-blocked`, `not-in-conversion-surface`.",
"",
"| Provider | Schema | Field | Required | Type | Surface | Same-Format Runtime | Canonical Roundtrip | Cross-Format | Notes |",
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
]
for field in fields:
status = existing.get(
(field.provider, field.schema, field.field),
default_status(field, profiles[(field.provider, field.schema)]),
)
rows.append(
"| "
+ " | ".join(
[
field.provider,
f"`{escape_markdown_cell(field.schema)}`",
f"`{escape_markdown_cell(field.field)}`",
field.required,
f"`{escape_markdown_cell(field.field_type)}`",
escape_markdown_cell(status.surface),
escape_markdown_cell(status.same_format_runtime),
escape_markdown_cell(status.canonical_roundtrip),
escape_markdown_cell(status.cross_format),
escape_markdown_cell(status.notes),
]
)
+ " |"
)
rows.extend(["", f"Total covered schema fields: {len(fields)}."])
return "\n".join(rows) + "\n"
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--definitions", type=Path, default=DEFAULT_DEFINITIONS)
parser.add_argument("--matrix", type=Path, default=DEFAULT_MATRIX)
parser.add_argument("--check", action="store_true")
args = parser.parse_args()
definitions = args.definitions.read_text()
current_matrix = args.matrix.read_text() if args.matrix.exists() else ""
fields = parse_provider_definition_fields(definitions)
existing, profiles = parse_existing_coverage(current_matrix)
next_matrix = render_matrix(fields, existing, profiles)
if args.check:
if current_matrix != next_matrix:
print(
f"{args.matrix} is not up to date; run "
"`python3 docs/api/generate_format_field_coverage.py`",
)
return 1
return 0
args.matrix.write_text(next_matrix)
print(f"wrote {len(fields)} field coverage rows to {args.matrix}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
File diff suppressed because it is too large Load Diff