mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-06 01:17:46 +08:00
chore: pass fmt clippy and nextest checks
Format merged Rust sources, satisfy strict clippy lints, and restore documentation fixtures required by format registry tests. Validation: cargo fmt --all --check; CI-scoped clippy with -D warnings; cargo nextest workspace excluding integration tests, 9747 passed and 47 skipped; targeted PostgreSQL analytics regression passed.
This commit is contained in:
@@ -267,9 +267,9 @@ impl History {
|
||||
fn snapshot(&mut self, now_us: u64, started_at_us: i64) -> Value {
|
||||
self.advance(now_us);
|
||||
let mut providers: Vec<_> = self.providers.iter().collect();
|
||||
providers.sort_unstable_by(|(a, _), (b, _)| a.cmp(b));
|
||||
providers.sort_unstable_by_key(|(provider, _)| *provider);
|
||||
let mut models: Vec<_> = self.models.iter().collect();
|
||||
models.sort_unstable_by(|(a, _), (b, _)| a.cmp(b));
|
||||
models.sort_unstable_by_key(|(model, _)| *model);
|
||||
json!({
|
||||
"observed_at": DateTime::from_timestamp_micros(started_at_us.saturating_add(self.through_us.min(i64::MAX as u64) as i64)),
|
||||
"observed_from": DateTime::from_timestamp_micros(started_at_us),
|
||||
|
||||
@@ -180,7 +180,7 @@ pub(super) fn public_projection(
|
||||
exclusion_policy: "client_cancelled_invalid_input_identity_or_quota_policy",
|
||||
},
|
||||
average_latency_ms: (metrics.latency_sample_count > 0)
|
||||
.then(|| metrics.latency_sum_ms as f64 / metrics.latency_sample_count as f64),
|
||||
.then(|| metrics.latency_sum_ms / metrics.latency_sample_count as f64),
|
||||
latency_sample_count: metrics.latency_sample_count,
|
||||
last_request_at: metrics.last_request_at_unix_ms.and_then(unix_ms_to_rfc3339),
|
||||
timeline: Vec::new(),
|
||||
|
||||
@@ -1237,7 +1237,9 @@ async fn gateway_handles_admin_system_users_export_locally_with_trusted_admin_pr
|
||||
assert!(payload["users"][0]["api_keys"][0]
|
||||
.get("credential_kind")
|
||||
.is_none());
|
||||
assert!(payload["standalone_keys"][0].get("credential_kind").is_none());
|
||||
assert!(payload["standalone_keys"][0]
|
||||
.get("credential_kind")
|
||||
.is_none());
|
||||
assert!(payload["exported_at"].as_str().is_some());
|
||||
assert_eq!(payload["user_groups"][0]["name"], "Restricted GPT");
|
||||
assert!(payload["user_groups"][0].get("priority").is_none());
|
||||
|
||||
@@ -50,10 +50,10 @@ use chrono::{TimeZone, Utc};
|
||||
const TEST_EMAIL_VERIFICATION_TOKEN: &str =
|
||||
"test-email-verification-token-00000000000000000000000000000000";
|
||||
|
||||
#[path = "public_support/auth_cookie.rs"]
|
||||
mod auth_cookie;
|
||||
#[path = "public_support/announcement_user_list.rs"]
|
||||
mod announcement_user_list;
|
||||
#[path = "public_support/auth_cookie.rs"]
|
||||
mod auth_cookie;
|
||||
#[path = "public_support/dashboard.rs"]
|
||||
mod dashboard;
|
||||
#[path = "public_support/vscodex.rs"]
|
||||
|
||||
@@ -486,7 +486,10 @@ async fn live_overview_canonical_queries_and_dirty_rebuild() {
|
||||
let state = sqlx::query("SELECT source_revision,built_revision FROM stats_bucket_state WHERE projection_version='overview-v2' AND granularity='hour' AND bucket_start=$1").bind(start).fetch_one(&pool).await.unwrap();
|
||||
// Bucket state stays unchanged on the foreground write; pending events
|
||||
// invalidate the projection before the background merger consumes them.
|
||||
assert_eq!(state.get::<i64, _>("source_revision"), state.get::<i64, _>("built_revision"));
|
||||
assert_eq!(
|
||||
state.get::<i64, _>("source_revision"),
|
||||
state.get::<i64, _>("built_revision")
|
||||
);
|
||||
let pending: bool = sqlx::query_scalar("SELECT EXISTS(SELECT 1 FROM stats_overview_dirty_events WHERE projection_version='overview-v2' AND granularity='hour' AND bucket_start=$1)")
|
||||
.bind(start).fetch_one(&pool).await.unwrap();
|
||||
assert!(pending);
|
||||
@@ -607,12 +610,21 @@ async fn live_overview_dirty_events_merge_without_shared_writer_bucket_lock() {
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(row, (2, "unrecoverable".into(), Some("retained usage facts were deleted".into())));
|
||||
let remaining: i64 = sqlx::query_scalar("SELECT count(*) FROM stats_overview_dirty_events WHERE bucket_start=$1")
|
||||
.bind(bucket)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
row,
|
||||
(
|
||||
2,
|
||||
"unrecoverable".into(),
|
||||
Some("retained usage facts were deleted".into())
|
||||
)
|
||||
);
|
||||
let remaining: i64 = sqlx::query_scalar(
|
||||
"SELECT count(*) FROM stats_overview_dirty_events WHERE bucket_start=$1",
|
||||
)
|
||||
.bind(bucket)
|
||||
.fetch_one(&pool)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(remaining, 0);
|
||||
sqlx::query("DELETE FROM stats_bucket_state WHERE projection_version='overview-v2' AND granularity='hour' AND bucket_start=$1")
|
||||
.bind(bucket)
|
||||
|
||||
@@ -63,8 +63,8 @@ mod analytics_tests;
|
||||
mod attribution;
|
||||
pub mod cleanup;
|
||||
mod dashboard;
|
||||
mod dashboard_summary;
|
||||
mod dashboard_retention;
|
||||
mod dashboard_summary;
|
||||
#[cfg(test)]
|
||||
mod dashboard_summary_tests;
|
||||
mod health;
|
||||
@@ -82,25 +82,29 @@ const FIND_USAGE_BODY_BLOB_BY_REF_SQL: &str = r#"SELECT CASE WHEN octet_length(p
|
||||
const DELETE_USAGE_BODY_BLOB_SQL: &str = include_str!("queries/delete_usage_body_blob_sql.sql");
|
||||
static USAGE_BODY_DECODE_SLOTS: tokio::sync::Semaphore = tokio::sync::Semaphore::const_new(4);
|
||||
|
||||
struct UsageAnalyticsDrilldown<'a> {
|
||||
provider_id: Option<&'a str>,
|
||||
api_key_id: Option<&'a str>,
|
||||
request_id: Option<&'a str>,
|
||||
attribution_kind: Option<&'a str>,
|
||||
actor_user_id: Option<&'a str>,
|
||||
endpoint_kind: Option<&'a str>,
|
||||
request_type: Option<&'a str>,
|
||||
slow_threshold_ms: Option<u64>,
|
||||
has_format_conversion: Option<bool>,
|
||||
}
|
||||
|
||||
fn push_usage_analytics_drilldown(
|
||||
builder: &mut QueryBuilder<'_, Postgres>,
|
||||
has_where: &mut bool,
|
||||
provider_id: &Option<String>,
|
||||
api_key_id: &Option<String>,
|
||||
request_id: &Option<String>,
|
||||
attribution_kind: &Option<String>,
|
||||
actor_user_id: &Option<String>,
|
||||
endpoint_kind: &Option<String>,
|
||||
request_type: &Option<String>,
|
||||
slow_threshold_ms: Option<u64>,
|
||||
has_format_conversion: Option<bool>,
|
||||
filters: UsageAnalyticsDrilldown<'_>,
|
||||
) {
|
||||
for (column, value) in [
|
||||
("provider_id", provider_id),
|
||||
("api_key_id", api_key_id),
|
||||
("request_id", request_id),
|
||||
("endpoint_kind", endpoint_kind),
|
||||
("request_type", request_type),
|
||||
("provider_id", filters.provider_id),
|
||||
("api_key_id", filters.api_key_id),
|
||||
("request_id", filters.request_id),
|
||||
("endpoint_kind", filters.endpoint_kind),
|
||||
("request_type", filters.request_type),
|
||||
] {
|
||||
if let Some(value) = value {
|
||||
builder.push(if *has_where { " AND " } else { " WHERE " });
|
||||
@@ -109,27 +113,27 @@ fn push_usage_analytics_drilldown(
|
||||
.push("\"usage\".")
|
||||
.push(column)
|
||||
.push(" = ")
|
||||
.push_bind(value.clone());
|
||||
.push_bind(value.to_string());
|
||||
}
|
||||
}
|
||||
for (column, value) in [
|
||||
("attribution_kind", attribution_kind),
|
||||
("actor_user_id", actor_user_id),
|
||||
("attribution_kind", filters.attribution_kind),
|
||||
("actor_user_id", filters.actor_user_id),
|
||||
] {
|
||||
if let Some(value) = value {
|
||||
builder.push(if *has_where { " AND " } else { " WHERE " });
|
||||
*has_where = true;
|
||||
builder.push("EXISTS(SELECT 1 FROM usage_analytics_facts_v1 f WHERE f.request_id=\"usage\".request_id AND f.").push(column).push(" = ").push_bind(value.clone()).push(")");
|
||||
builder.push("EXISTS(SELECT 1 FROM usage_analytics_facts_v1 f WHERE f.request_id=\"usage\".request_id AND f.").push(column).push(" = ").push_bind(value.to_string()).push(")");
|
||||
}
|
||||
}
|
||||
if let Some(value) = slow_threshold_ms {
|
||||
if let Some(value) = filters.slow_threshold_ms {
|
||||
builder.push(if *has_where { " AND " } else { " WHERE " });
|
||||
*has_where = true;
|
||||
builder
|
||||
.push("\"usage\".response_time_ms >= ")
|
||||
.push_bind(i64::try_from(value).unwrap_or(i64::MAX));
|
||||
}
|
||||
if let Some(value) = has_format_conversion {
|
||||
if let Some(value) = filters.has_format_conversion {
|
||||
builder.push(if *has_where { " AND " } else { " WHERE " });
|
||||
*has_where = true;
|
||||
builder
|
||||
@@ -3168,15 +3172,17 @@ ORDER BY request_count DESC, "usage".provider_name ASC
|
||||
push_usage_analytics_drilldown(
|
||||
&mut builder,
|
||||
&mut has_where,
|
||||
&query.provider_id,
|
||||
&query.api_key_id,
|
||||
&query.request_id,
|
||||
&query.attribution_kind,
|
||||
&query.actor_user_id,
|
||||
&query.endpoint_kind,
|
||||
&query.request_type,
|
||||
query.slow_threshold_ms,
|
||||
query.has_format_conversion,
|
||||
UsageAnalyticsDrilldown {
|
||||
provider_id: query.provider_id.as_deref(),
|
||||
api_key_id: query.api_key_id.as_deref(),
|
||||
request_id: query.request_id.as_deref(),
|
||||
attribution_kind: query.attribution_kind.as_deref(),
|
||||
actor_user_id: query.actor_user_id.as_deref(),
|
||||
endpoint_kind: query.endpoint_kind.as_deref(),
|
||||
request_type: query.request_type.as_deref(),
|
||||
slow_threshold_ms: query.slow_threshold_ms,
|
||||
has_format_conversion: query.has_format_conversion,
|
||||
},
|
||||
);
|
||||
if let Some(user_id) = query.user_id.as_deref() {
|
||||
builder.push(if has_where { " AND " } else { " WHERE " });
|
||||
@@ -3294,15 +3300,17 @@ OR (\"usage\".error_message IS NOT NULL AND BTRIM(\"usage\".error_message) <> ''
|
||||
push_usage_analytics_drilldown(
|
||||
&mut builder,
|
||||
&mut has_where,
|
||||
&query.provider_id,
|
||||
&query.api_key_id,
|
||||
&query.request_id,
|
||||
&query.attribution_kind,
|
||||
&query.actor_user_id,
|
||||
&query.endpoint_kind,
|
||||
&query.request_type,
|
||||
query.slow_threshold_ms,
|
||||
query.has_format_conversion,
|
||||
UsageAnalyticsDrilldown {
|
||||
provider_id: query.provider_id.as_deref(),
|
||||
api_key_id: query.api_key_id.as_deref(),
|
||||
request_id: query.request_id.as_deref(),
|
||||
attribution_kind: query.attribution_kind.as_deref(),
|
||||
actor_user_id: query.actor_user_id.as_deref(),
|
||||
endpoint_kind: query.endpoint_kind.as_deref(),
|
||||
request_type: query.request_type.as_deref(),
|
||||
slow_threshold_ms: query.slow_threshold_ms,
|
||||
has_format_conversion: query.has_format_conversion,
|
||||
},
|
||||
);
|
||||
if let Some(user_id) = query.user_id.as_deref() {
|
||||
builder.push(if has_where { " AND " } else { " WHERE " });
|
||||
@@ -3501,15 +3509,17 @@ OR (\"usage\".error_message IS NOT NULL AND BTRIM(\"usage\".error_message) <> ''
|
||||
push_usage_analytics_drilldown(
|
||||
&mut builder,
|
||||
&mut has_where,
|
||||
&query.provider_id,
|
||||
&query.api_key_id,
|
||||
&query.request_id,
|
||||
&query.attribution_kind,
|
||||
&query.actor_user_id,
|
||||
&query.endpoint_kind,
|
||||
&query.request_type,
|
||||
query.slow_threshold_ms,
|
||||
query.has_format_conversion,
|
||||
UsageAnalyticsDrilldown {
|
||||
provider_id: query.provider_id.as_deref(),
|
||||
api_key_id: query.api_key_id.as_deref(),
|
||||
request_id: query.request_id.as_deref(),
|
||||
attribution_kind: query.attribution_kind.as_deref(),
|
||||
actor_user_id: query.actor_user_id.as_deref(),
|
||||
endpoint_kind: query.endpoint_kind.as_deref(),
|
||||
request_type: query.request_type.as_deref(),
|
||||
slow_threshold_ms: query.slow_threshold_ms,
|
||||
has_format_conversion: query.has_format_conversion,
|
||||
},
|
||||
);
|
||||
if let Some(user_id) = query.user_id.as_deref() {
|
||||
builder.push(if has_where { " AND " } else { " WHERE " });
|
||||
@@ -3616,15 +3626,17 @@ OR (\"usage\".error_message IS NOT NULL AND BTRIM(\"usage\".error_message) <> ''
|
||||
push_usage_analytics_drilldown(
|
||||
&mut builder,
|
||||
&mut has_where,
|
||||
&query.provider_id,
|
||||
&query.api_key_id,
|
||||
&query.request_id,
|
||||
&query.attribution_kind,
|
||||
&query.actor_user_id,
|
||||
&query.endpoint_kind,
|
||||
&query.request_type,
|
||||
query.slow_threshold_ms,
|
||||
query.has_format_conversion,
|
||||
UsageAnalyticsDrilldown {
|
||||
provider_id: query.provider_id.as_deref(),
|
||||
api_key_id: query.api_key_id.as_deref(),
|
||||
request_id: query.request_id.as_deref(),
|
||||
attribution_kind: query.attribution_kind.as_deref(),
|
||||
actor_user_id: query.actor_user_id.as_deref(),
|
||||
endpoint_kind: query.endpoint_kind.as_deref(),
|
||||
request_type: query.request_type.as_deref(),
|
||||
slow_threshold_ms: query.slow_threshold_ms,
|
||||
has_format_conversion: query.has_format_conversion,
|
||||
},
|
||||
);
|
||||
if let Some(user_id) = query.user_id.as_deref() {
|
||||
builder.push(if has_where { " AND " } else { " WHERE " });
|
||||
@@ -10528,7 +10540,8 @@ impl UsageReadRepository for SqlxUsageReadRepository {
|
||||
async fn query_dashboard_summary(
|
||||
&self,
|
||||
query: &aether_data_contracts::repository::usage::UsageDashboardAnalyticsQuery,
|
||||
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError> {
|
||||
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError>
|
||||
{
|
||||
Self::query_dashboard_summary(self, query).await
|
||||
}
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@ use super::analytics::{
|
||||
analytics_additive_metrics_sql, dashboard_total_metrics_sql, push_analytics_filter,
|
||||
};
|
||||
use crate::error::SqlxResultExt;
|
||||
use aether_data_contracts::{DataLayerError, repository::usage::*};
|
||||
use aether_data_contracts::{repository::usage::*, DataLayerError};
|
||||
use chrono::{DateTime, Utc};
|
||||
use sqlx::{Postgres, QueryBuilder, Row};
|
||||
|
||||
|
||||
@@ -27,13 +27,13 @@ use crate::lifecycle::bootstrap::postgres::{
|
||||
EMPTY_DATABASE_SNAPSHOT_CUTOFF_VERSION, EMPTY_DATABASE_SNAPSHOT_SQL,
|
||||
};
|
||||
|
||||
mod policy_nulls;
|
||||
mod overview_dirty_events;
|
||||
mod provider_expenses;
|
||||
mod migration_deadlines;
|
||||
mod overview_migration_safety;
|
||||
mod legacy_overview_upgrade;
|
||||
mod dashboard_user_anonymization;
|
||||
mod legacy_overview_upgrade;
|
||||
mod migration_deadlines;
|
||||
mod overview_dirty_events;
|
||||
mod overview_migration_safety;
|
||||
mod policy_nulls;
|
||||
mod provider_expenses;
|
||||
|
||||
/// A clean PostgreSQL database is bootstrapped from the schema snapshot first;
|
||||
/// migrations after the privacy/security frontier are intentionally left
|
||||
|
||||
@@ -1285,7 +1285,8 @@ impl UsageReadRepository for InMemoryUsageReadRepository {
|
||||
async fn query_dashboard_summary(
|
||||
&self,
|
||||
query: &aether_data_contracts::repository::usage::UsageDashboardAnalyticsQuery,
|
||||
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError> {
|
||||
) -> Result<aether_data_contracts::repository::usage::StoredDashboardSummary, DataLayerError>
|
||||
{
|
||||
self.dashboard_summary_query(query)
|
||||
}
|
||||
|
||||
@@ -3472,7 +3473,9 @@ impl UsageWriteRepository for InMemoryUsageReadRepository {
|
||||
finalized_at_unix_secs: usage.finalized_at_unix_secs,
|
||||
};
|
||||
|
||||
self.dashboard_projection.write().expect("dashboard projection lock")
|
||||
self.dashboard_projection
|
||||
.write()
|
||||
.expect("dashboard projection lock")
|
||||
.record(&stored, &self.analytics_key_flags());
|
||||
by_request_id.insert(stored.request_id.clone(), stored.clone());
|
||||
if let Some(auth_api_keys) = self.auth_api_keys.as_ref() {
|
||||
|
||||
@@ -0,0 +1,238 @@
|
||||
# Format Conversion Audit
|
||||
|
||||
Last audited: 2026-06-03
|
||||
|
||||
This audit tracks `source format -> Canonical -> target format` behavior. It is intentionally stricter than historical best-effort conversion.
|
||||
|
||||
Statuses:
|
||||
|
||||
- `native`: emitted as a target-native field without semantic change.
|
||||
- `mapped`: converted through canonical/provider-specific mapping.
|
||||
- `extension-preserved`: preserved in same-format canonical roundtrip or target-approved extension namespace.
|
||||
- `transport-only`: audited client transport metadata intentionally omitted when the target has no compatible transport channel.
|
||||
- `unaudited`: rejected because the source field is not in the audited provider schema inventory for cross-format conversion.
|
||||
- `unsupported`: rejected because the request/field shape is outside the supported conversion surface, independent of schema drift.
|
||||
- `lossy-blocked`: conversion fails closed.
|
||||
- `invalid-enum`: conversion fails closed because a provider enum value is not valid for the target mapping.
|
||||
|
||||
Full schema field coverage is tracked in `docs/api/format-field-coverage-matrix.md`. That matrix is generated from the schema inventory in `docs/api/provider-interface-definitions.md` by `python3 docs/api/generate_format_field_coverage.py` and gives every documented OpenAI, Claude, and Gemini schema field a handling status. “Handled” means mapped, same-format/native preserved, extension-preserved, blocked with a structured error, or explicitly marked outside the canonical conversion surface.
|
||||
|
||||
Provider schema refresh is not a runtime dependency. Same-format runtime paths do not use this matrix; they bypass canonical conversion. Same-format canonical roundtrip must preserve unrecognized provider fields through provider extension namespaces. Cross-format conversion is capability-based: only explicitly mapped fields are emitted, and newly discovered or unknown provider fields fail closed with `UnauditedField` until a lossless mapping is audited.
|
||||
|
||||
## Implemented Boundary Changes
|
||||
|
||||
| Area | Current behavior |
|
||||
| --- | --- |
|
||||
| Pure conversion API | `convert_request_pure` and response equivalents do not apply model override or stream policy. |
|
||||
| Legacy conversion API | `convert_request` / `convert_response` are retained for migration and may still use legacy context behavior. |
|
||||
| Same-format provider path | Bypasses canonical conversion and copies the parsed JSON object before transport edits. |
|
||||
| Cross-format same-format-provider path | Uses `convert_request_pure`, then applies model/body/stream edits in transport. |
|
||||
| Conversion errors | Added `UnauditedField`, `UnsupportedField`, `InvalidEnumValue`, `LossyConversionBlocked`, and `InvalidTargetField`. |
|
||||
| Reporting | Added `ConversionReport` with field statuses. Runtime reports remain conversion-operation oriented; exhaustive nested schema coverage is enforced by `format-field-coverage-matrix.md`. |
|
||||
| Source schema coverage | Cross-format request conversion rejects unknown source root fields before emit. Every documented schema field is covered by the field coverage matrix. |
|
||||
| Schema drift handling | Official schema changes are detected by regenerating the inventory/matrix. Runtime same-format remains passthrough; cross-format unknowns return `UnauditedField` until deliberately mapped. |
|
||||
| Tool schema roundtrip | Claude `input_schema` and Gemini `functionDeclarations.parameters` preserve raw same-format schema through provider-specific extensions. |
|
||||
| Tool result ids | Chat `tool_call_id`, Responses `call_id`, Claude `tool_use_id`, and Gemini `functionResponse.id` are mapped through canonical tool IDs. |
|
||||
|
||||
## OpenAI Chat -> OpenAI Responses
|
||||
|
||||
| Chat field | Canonical handling | Responses output | Status |
|
||||
| --- | --- | --- | --- |
|
||||
| `model` | request identity | `model` | native |
|
||||
| `messages` | canonical messages/instructions | `input`, `instructions` | mapped |
|
||||
| `max_tokens` | generation max tokens | `max_output_tokens` | mapped |
|
||||
| `max_completion_tokens` | generation max tokens | `max_output_tokens` | mapped |
|
||||
| `temperature` | generation | `temperature` | native |
|
||||
| `top_p` | generation | `top_p` | native |
|
||||
| `top_logprobs` | generation | `top_logprobs` | native |
|
||||
| `n` | generation but no Responses equivalent | none | lossy-blocked |
|
||||
| `stop` | generation but no Responses equivalent | none | lossy-blocked |
|
||||
| `presence_penalty` | generation but no Responses equivalent | none | lossy-blocked |
|
||||
| `frequency_penalty` | generation but no Responses equivalent | none | lossy-blocked |
|
||||
| `seed` | generation but no Responses equivalent | none | lossy-blocked |
|
||||
| `logprobs` | generation but no Responses equivalent | none | lossy-blocked |
|
||||
| `stream` | OpenAI extension | `stream` | mapped if explicit |
|
||||
| `stream_options` | Chat-specific extension | none | lossy-blocked |
|
||||
| `tools[].function.name` | canonical tool | `tools[].name` | mapped |
|
||||
| `tools[].function.description` | canonical tool | `tools[].description` | mapped |
|
||||
| `tools[].function.parameters` | canonical tool | `tools[].parameters` | mapped |
|
||||
| `tools[].function.strict` | canonical tool strict | `tools[].strict` | mapped, implemented |
|
||||
| assistant `tool_calls[].id` | canonical tool use id | `function_call.call_id` | mapped, implemented |
|
||||
| tool message `tool_call_id` | canonical tool result id | `function_call_output.call_id` | mapped, implemented |
|
||||
| `tool_choice` | canonical tool choice | `tool_choice` | mapped |
|
||||
| `parallel_tool_calls` | canonical bool | `parallel_tool_calls` | native |
|
||||
| `metadata` | canonical metadata | `metadata` | native |
|
||||
| `response_format` | canonical response format | `text.format` | mapped |
|
||||
| `reasoning_effort` | OpenAI enum | `reasoning.effort` | mapped; invalid enum blocked |
|
||||
| `verbosity` | OpenAI extension | `text.verbosity` | mapped |
|
||||
| `store` | OpenAI extension | `store` | extension-preserved |
|
||||
| `service_tier` | OpenAI extension | `service_tier` | extension-preserved |
|
||||
| `safety_identifier` | OpenAI extension | `safety_identifier` | extension-preserved |
|
||||
| `prompt_cache_key` | OpenAI extension | `prompt_cache_key` | extension-preserved |
|
||||
| `user` | legacy Chat user field | none | lossy-blocked |
|
||||
| unknown top-level fields | source schema guard | none | unaudited |
|
||||
|
||||
## OpenAI Responses -> OpenAI Chat
|
||||
|
||||
| Responses field | Canonical handling | Chat output | Status |
|
||||
| --- | --- | --- | --- |
|
||||
| `model` | request identity | `model` | native |
|
||||
| `input` | canonical messages/content/tool I/O | `messages` | mapped |
|
||||
| `instructions` | canonical instruction/system | `messages` system/developer | mapped |
|
||||
| `max_output_tokens` | generation max tokens | `max_completion_tokens` | mapped |
|
||||
| `temperature` | generation | `temperature` | native |
|
||||
| `top_p` | generation | `top_p` | native |
|
||||
| `top_logprobs` | generation | `top_logprobs` | native |
|
||||
| `metadata` | canonical metadata | `metadata` | native |
|
||||
| `client_metadata` | Responses client transport metadata | none | transport-only; omitted |
|
||||
| `parallel_tool_calls` | canonical bool | `parallel_tool_calls` | native |
|
||||
| `text.format` | canonical response format | `response_format` | mapped |
|
||||
| `text.verbosity` | Responses extension | `verbosity` | mapped |
|
||||
| `tools[].type=function` | canonical tool | `tools[].type=function` | mapped |
|
||||
| `tools[].name` | canonical tool | `tools[].function.name` | mapped |
|
||||
| `tools[].parameters` | canonical tool | `tools[].function.parameters` | mapped |
|
||||
| `tools[].strict` | canonical tool strict | `tools[].function.strict` | mapped, implemented |
|
||||
| `function_call.call_id` | canonical tool use id | `tool_calls[].id` | mapped, implemented |
|
||||
| `function_call_output.call_id` | canonical tool result id | tool message `tool_call_id` | mapped, implemented |
|
||||
| `tools[].type=custom` | raw Responses tool | none | lossy-blocked to Chat |
|
||||
| `tools[].type=web_search*` | raw Responses tool | none | lossy-blocked to Chat |
|
||||
| `tool_choice` | canonical tool choice | `tool_choice` | mapped |
|
||||
| `reasoning.effort` | OpenAI enum | `reasoning_effort` | mapped; invalid enum blocked |
|
||||
| `reasoning.summary` | Responses-only | none | lossy-blocked |
|
||||
| `reasoning.budget_tokens` | Responses-only | none | lossy-blocked |
|
||||
| `stream` | Responses request transport policy | none | lossy-blocked; target stream policy is transport-owned |
|
||||
| `include` | Responses-only | none | lossy-blocked; legacy emitter no longer leaks |
|
||||
| `previous_response_id` | Responses-only | none | lossy-blocked; legacy emitter no longer leaks |
|
||||
| `truncation` | Responses-only | none | lossy-blocked |
|
||||
| `prompt` | Responses-only | none | lossy-blocked |
|
||||
| `conversation` | Responses-only | none | lossy-blocked |
|
||||
| `background` | Responses-only | none | lossy-blocked |
|
||||
| `max_tool_calls` | Responses-only | none | lossy-blocked |
|
||||
| unknown top-level fields | source schema guard | none | unaudited |
|
||||
|
||||
## Claude Messages <-> OpenAI Chat / Responses
|
||||
|
||||
Claude to OpenAI Chat, Claude to OpenAI Responses, and the reverse directions are included in the field coverage matrix. Runtime strict guards cover request root fields, provider extension namespaces, thinking/cache/tool-result hazards, and target generation-field gaps. Fields without a lossless target equivalent fail closed instead of being dropped.
|
||||
|
||||
High-risk fields:
|
||||
|
||||
| Claude field | OpenAI target risk | Required status |
|
||||
| --- | --- | --- |
|
||||
| `system` with cache blocks | Chat/Responses system instructions | same-format preserved; cross-format `cache_control` loss is blocked |
|
||||
| `thinking` | OpenAI reasoning | Claude request-level thinking config maps to OpenAI reasoning; message-level thinking blocks are blocked for Responses |
|
||||
| `cache_control` | OpenAI content/tool extensions | same-format preserved; cross-format blocked when no target equivalent exists |
|
||||
| `tools[].input_schema` | OpenAI tool parameters | mapped; raw same-format schema preservation implemented |
|
||||
| `tool_choice.disable_parallel_tool_use` | OpenAI `parallel_tool_calls` | mapped, implemented |
|
||||
| `tool_result` multi-block content | OpenAI tool output/content | same-format preserved; cross-format to Chat/Responses is lossy-blocked |
|
||||
| `metadata` | OpenAI metadata | mapped when the target has metadata |
|
||||
| `container`, `inference_geo`, `service_tier` | OpenAI target has no audited equivalent | lossy-blocked unless a target-approved mapping is added |
|
||||
|
||||
## Gemini GenerateContent <-> OpenAI Chat / Responses / Claude
|
||||
|
||||
Gemini to OpenAI Chat, Gemini to OpenAI Responses, Gemini to Claude, and reverse generation paths are included in the field coverage matrix. Gemini-only request fields are preserved same-format and blocked cross-format unless the target mapping is explicitly audited.
|
||||
|
||||
High-risk fields:
|
||||
|
||||
| Gemini field | Target risk | Required status |
|
||||
| --- | --- | --- |
|
||||
| `contents[].parts[].thoughtSignature` | OpenAI/Claude thinking | Chat/Claude preserve; Responses cross-format is lossy-blocked |
|
||||
| `tools[].functionDeclarations` | OpenAI/Claude tool schema | mapped; raw same-format `parameters` preservation implemented |
|
||||
| `toolConfig.functionCallingConfig.allowedFunctionNames` | OpenAI/Claude tool choice | single-name mapping implemented; multi-name input is lossy-blocked |
|
||||
| `toolConfig.functionCallingConfig.mode` | OpenAI/Claude tool choice enum | valid enum required; invalid values fail with `InvalidEnumValue` |
|
||||
| `generationConfig.thinkingConfig.thinkingLevel` | OpenAI/Claude reasoning effort | low/medium/high mapping implemented; invalid values fail closed |
|
||||
| `safetySettings` | OpenAI/Claude no direct equivalent | lossy-blocked |
|
||||
| `cachedContent` | OpenAI/Claude no direct equivalent | lossy-blocked |
|
||||
| `codeExecution` | OpenAI/Claude tool/builtin mismatch | lossy-blocked |
|
||||
| `generationConfig.responseModalities` | OpenAI/Claude modality mismatch | lossy-blocked |
|
||||
| `functionResponse.id` | tool result id | conversion preserves id; Gemini upstream cleanup is transport-layer edit only |
|
||||
|
||||
## Embedding And Rerank
|
||||
|
||||
Embedding and rerank request parse/emit capability and strict target guards are implemented. Provider schema fields outside these canonical conversion surfaces are marked `not-in-conversion-surface` in the field coverage matrix instead of being left implicit.
|
||||
|
||||
Embedding source capability:
|
||||
|
||||
| Source format | Parsed request shape | Canonical fields | Status |
|
||||
| --- | --- | --- | --- |
|
||||
| OpenAI Embedding | `model`, `input`, `encoding_format`, `dimensions`, `user`, `parameters`, `task` | OpenAI-like embedding | mapped |
|
||||
| Jina Embedding | OpenAI-like plus provider extension namespace | OpenAI-like embedding | mapped |
|
||||
| Doubao Embedding | OpenAI-like `model` + text `input` | OpenAI-like embedding | mapped |
|
||||
| Gemini Embedding | single `content.parts[].text` or batch `requests[]` | text input, `dimensions`, `task` | mapped |
|
||||
| Aliyun Multimodal Embedding | `input.contents[]`, `parameters.dimension` | text/multimodal input, `dimensions`, `parameters` | mapped |
|
||||
|
||||
Embedding target guards:
|
||||
|
||||
| Target format | Accepted canonical fields | Blocked fields/cases | Status |
|
||||
| --- | --- | --- | --- |
|
||||
| OpenAI Embedding | text or token input, `encoding_format`, `dimensions`, `user` | multimodal input, `task`, generic `parameters` | lossy-blocked |
|
||||
| Jina Embedding | text input, `dimensions`, `task`, `parameters` | token/multimodal input, `encoding_format`, `user` | lossy-blocked |
|
||||
| Gemini Embedding | text input, `dimensions`, valid `taskType` | token/multimodal input, `encoding_format`, `user`, generic `parameters`, invalid `taskType` | lossy-blocked / invalid-enum |
|
||||
| Doubao Embedding | text input, `dimensions` | token/multimodal input, `encoding_format`, `user`, `task`, generic `parameters` | lossy-blocked |
|
||||
| Aliyun Multimodal Embedding | text or multimodal input, `dimensions`, `parameters` | token input, `encoding_format`, `user`, `task` | lossy-blocked |
|
||||
|
||||
Cross-format embedding invariants:
|
||||
|
||||
- Embedding formats can only convert to embedding formats.
|
||||
- Unknown provider-specific embedding extension namespaces are blocked cross-format unless the namespace matches the target.
|
||||
- Aliyun `parameters.dimension` maps to canonical `dimensions` and is not treated as generic `parameters`.
|
||||
- Gemini batch embedding parse requires every batch item to share the same model, dimensions, and task.
|
||||
|
||||
Rerank first pass:
|
||||
|
||||
| Area | Current behavior | Status |
|
||||
| --- | --- | --- |
|
||||
| Source formats | OpenAI Rerank and Jina Rerank parse OpenAI-like `model`, `query`, `documents`, `top_n`, `return_documents` | mapped |
|
||||
| Target formats | OpenAI Rerank and Jina Rerank emit OpenAI-like rerank bodies | mapped |
|
||||
| Boundary | Rerank formats can only convert to rerank formats | lossy-blocked |
|
||||
| Validation | Empty query/documents and `top_n=0` fail closed | invalid-target-field |
|
||||
| Extensions | Unknown provider-specific rerank extension namespaces are blocked cross-format | unsupported |
|
||||
|
||||
## Sync Response Conversion
|
||||
|
||||
Cross-format sync response conversion now validates source stop/finish/status
|
||||
enums before emitting a target body. Same-format runtime response passthrough is
|
||||
still outside canonical conversion.
|
||||
|
||||
| Source field | Target risk | Current behavior | Status |
|
||||
| --- | --- | --- | --- |
|
||||
| Same-format response raw stop/status fields | canonical emitters would otherwise normalize unknown enum/status to default target stop values | raw OpenAI Chat `finish_reason`, OpenAI Responses `status`, Claude `stop_reason`/`stop_sequence`, and Gemini `finishReason` are preserved through provider extension metadata | extension-preserved |
|
||||
| OpenAI Chat `choices[].finish_reason` | unknown value would otherwise emit as target normal stop | valid Chat enum required; unknown values fail with `InvalidEnumValue` | invalid-enum |
|
||||
| OpenAI Responses `status` | `queued`, `in_progress`, and `cancelled` have no sync target equivalent | non-terminal valid states fail with `LossyConversionBlocked`; invalid states fail with `InvalidEnumValue` | lossy-blocked / invalid-enum |
|
||||
| OpenAI Responses `incomplete_details.reason=content_filter` | previously mapped to max tokens/`length` | maps to canonical content filter and emits Chat `content_filter` / Claude `content_filtered` / Gemini `SAFETY` | mapped |
|
||||
| Claude `stop_reason` | unknown value would otherwise emit as target normal stop | valid known stop enum required for cross-format conversion | invalid-enum |
|
||||
| Gemini `candidates[].finishReason` | known-but-unmappable reasons would otherwise emit as target normal stop | mappable safety/max/stop reasons convert; known unmappable values such as `OTHER`, `MALFORMED_FUNCTION_CALL`, `UNEXPECTED_TOOL_CALL`, `MISSING_THOUGHT_SIGNATURE`, and `MALFORMED_RESPONSE` fail with `LossyConversionBlocked`; future unknown values fail with `InvalidEnumValue` | lossy-blocked / invalid-enum |
|
||||
| Canonical `Unknown` stop reason | target emitters default to normal stop values | cross-format response conversion blocks canonical unknown stop reasons | lossy-blocked |
|
||||
|
||||
## Stream Conversion
|
||||
|
||||
Sixth batch first pass is implemented for unknown event handling and runtime
|
||||
same-format boundaries. Sync response finish/status parity has a first strict
|
||||
pass; stream finish-reason guardrails are implemented for unknown/unmappable
|
||||
terminal reasons. Stream event schema fields are covered in the field coverage
|
||||
matrix; provider-by-provider fixtures cover the runtime event behavior.
|
||||
|
||||
Current stream behavior:
|
||||
|
||||
| Area | Current behavior | Status |
|
||||
| --- | --- | --- |
|
||||
| Provider parsers | OpenAI Chat, OpenAI Responses, Claude, and Gemini unknown stream payloads become `CanonicalStreamEvent::UnknownEvent` | mapped |
|
||||
| Cross-format stream matrix | Unknown canonical stream events emit a target-format error SSE with `unsupported_stream_event` and terminate conversion | lossy-blocked |
|
||||
| Stream finish reason guard | Unknown OpenAI finish reasons, unknown Claude `stop_reason`, and Gemini known-but-unmappable `finishReason` values such as `OTHER` are preserved as raw canonical finish strings, then blocked by the matrix with `unsupported_finish_reason` | lossy-blocked |
|
||||
| OpenAI Responses stream target | Canonical `length` and `content_filter` terminal reasons emit `response.incomplete` with `incomplete_details.reason=max_output_tokens` or `content_filter` instead of `response.completed` | mapped |
|
||||
| Terminal observer | Unknown provider stream events increment `unknown_event_count`; OpenAI Responses failed events mark terminal error state | mapped |
|
||||
| Stream -> sync aggregate | Unknown OpenAI Chat, OpenAI Responses, Claude, and Gemini stream events make the runtime finalize checked path return an error and block `body_json` fallback; legacy public aggregate helpers keep `Option` compatibility | lossy-blocked |
|
||||
| Runtime strict fallback guard | `UnauditedField`, `InvalidEnumValue`, `UnsupportedField`, `LossyConversionBlocked`, and `InvalidTargetField` from registry response conversion are not allowed to fall through legacy conversion helpers | lossy-blocked |
|
||||
| Runtime same-format stream | Same-format stream passthrough remains outside canonical conversion; stream policy edits are transport-layer only | native |
|
||||
|
||||
Stream fixture coverage:
|
||||
|
||||
| Provider stream | Covered fixture areas |
|
||||
| --- | --- |
|
||||
| OpenAI Chat | sync aggregation for text, tool call IDs/names/argument deltas, finish reason, and usage; cross-format unknown finish/event blocking |
|
||||
| OpenAI Responses | text snapshot de-duplication, multi-part messages, reasoning/items, function calls, image generation calls, same-family stream sync, unknown event blocking |
|
||||
| Claude Messages | thinking signatures, tool input deltas, cache/usage aggregation, media emission, unknown stop/event blocking |
|
||||
| Gemini GenerateContent | text/media/signature aggregation, function calls/results, safety finish mapping, unknown parts/events, and unmappable finish reason blocking |
|
||||
|
||||
Matrix-level interception remains the authoritative runtime path for cross-format
|
||||
unknown events; direct client emitters are covered only as provider/client
|
||||
building blocks.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,548 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate the provider schema field coverage matrix.
|
||||
|
||||
The input inventory is docs/api/provider-interface-definitions.md. Existing
|
||||
coverage rows are reused so audited status/notes survive regeneration. Newly
|
||||
introduced provider fields get conservative same-format/native and cross-format
|
||||
fail-closed defaults until a human audits whether they deserve an explicit
|
||||
mapping.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import dataclasses
|
||||
from collections import Counter, defaultdict
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
DEFAULT_DEFINITIONS = ROOT / "docs/api/provider-interface-definitions.md"
|
||||
DEFAULT_MATRIX = ROOT / "docs/api/format-field-coverage-matrix.md"
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
class SourceField:
|
||||
provider: str
|
||||
schema: str
|
||||
field: str
|
||||
required: str
|
||||
field_type: str
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
class CoverageStatus:
|
||||
surface: str
|
||||
same_format_runtime: str
|
||||
canonical_roundtrip: str
|
||||
cross_format: str
|
||||
notes: str
|
||||
|
||||
|
||||
OPENAI_CHAT_MAPPED = {
|
||||
"model",
|
||||
"messages",
|
||||
"max_tokens",
|
||||
"max_completion_tokens",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"top_logprobs",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"metadata",
|
||||
"response_format",
|
||||
"reasoning_effort",
|
||||
"verbosity",
|
||||
"store",
|
||||
"service_tier",
|
||||
"safety_identifier",
|
||||
"prompt_cache_key",
|
||||
"prompt_cache_retention",
|
||||
"stream",
|
||||
}
|
||||
|
||||
OPENAI_CHAT_BLOCKED = {
|
||||
"n",
|
||||
"stop",
|
||||
"presence_penalty",
|
||||
"frequency_penalty",
|
||||
"seed",
|
||||
"logprobs",
|
||||
"stream_options",
|
||||
"user",
|
||||
"function_call",
|
||||
"functions",
|
||||
"logit_bias",
|
||||
"modalities",
|
||||
"prediction",
|
||||
"audio",
|
||||
"web_search_options",
|
||||
}
|
||||
|
||||
OPENAI_RESPONSES_MAPPED = {
|
||||
"model",
|
||||
"input",
|
||||
"instructions",
|
||||
"max_output_tokens",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"top_logprobs",
|
||||
"metadata",
|
||||
"parallel_tool_calls",
|
||||
"text",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"reasoning",
|
||||
"store",
|
||||
"service_tier",
|
||||
"safety_identifier",
|
||||
"prompt_cache_key",
|
||||
"prompt_cache_retention",
|
||||
}
|
||||
|
||||
OPENAI_RESPONSES_BLOCKED = {
|
||||
"include",
|
||||
"previous_response_id",
|
||||
"truncation",
|
||||
"prompt",
|
||||
"conversation",
|
||||
"background",
|
||||
"max_tool_calls",
|
||||
"user",
|
||||
"context_management",
|
||||
"stream",
|
||||
"stream_options",
|
||||
}
|
||||
|
||||
CLAUDE_MAPPED_FIELDS = {
|
||||
"id",
|
||||
"type",
|
||||
"role",
|
||||
"text",
|
||||
"content",
|
||||
"source",
|
||||
"name",
|
||||
"description",
|
||||
"input",
|
||||
"input_schema",
|
||||
"messages",
|
||||
"model",
|
||||
"max_tokens",
|
||||
"system",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"top_k",
|
||||
"stop_sequences",
|
||||
"tool_choice",
|
||||
"tools",
|
||||
"metadata",
|
||||
"thinking",
|
||||
"output_config",
|
||||
"usage",
|
||||
"stop_reason",
|
||||
"stop_sequence",
|
||||
}
|
||||
|
||||
CLAUDE_PROVIDER_ONLY_FIELDS = {
|
||||
"cache_control",
|
||||
"container",
|
||||
"inference_geo",
|
||||
"service_tier",
|
||||
"allowed_callers",
|
||||
"allowed_domains",
|
||||
"blocked_domains",
|
||||
"defer_loading",
|
||||
"max_uses",
|
||||
"strict",
|
||||
"user_location",
|
||||
"citations",
|
||||
"context",
|
||||
"title",
|
||||
"file_id",
|
||||
"document_index",
|
||||
"document_title",
|
||||
"cited_text",
|
||||
"caller",
|
||||
}
|
||||
|
||||
|
||||
def split_markdown_row(line: str) -> list[str]:
|
||||
cells: list[str] = []
|
||||
current: list[str] = []
|
||||
escaped = False
|
||||
for char in line:
|
||||
if char == "|" and not escaped:
|
||||
cells.append("".join(current).strip())
|
||||
current.clear()
|
||||
else:
|
||||
current.append(char)
|
||||
escaped = char == "\\" and not escaped
|
||||
if escaped and char != "\\":
|
||||
escaped = False
|
||||
cells.append("".join(current).strip())
|
||||
return cells
|
||||
|
||||
|
||||
def strip_markdown_code(value: str) -> str:
|
||||
value = value.strip()
|
||||
if value.startswith("`") and value.endswith("`"):
|
||||
value = value[1:-1]
|
||||
return value.replace("\\|", "|")
|
||||
|
||||
|
||||
def escape_markdown_cell(value: str) -> str:
|
||||
return value.replace("|", "\\|")
|
||||
|
||||
|
||||
def parse_schema_heading(line: str) -> str | None:
|
||||
if not line.startswith("### `"):
|
||||
return None
|
||||
rest = line[len("### `") :]
|
||||
schema, _, _ = rest.partition("`")
|
||||
return schema or None
|
||||
|
||||
|
||||
def parse_provider_definition_fields(definitions: str) -> list[SourceField]:
|
||||
provider: str | None = None
|
||||
schema: str | None = None
|
||||
fields: list[SourceField] = []
|
||||
|
||||
for line in definitions.splitlines():
|
||||
if line.startswith("## "):
|
||||
if "OpenAI Schema" in line:
|
||||
provider = "OpenAI"
|
||||
elif "Claude / Anthropic TypeScript" in line:
|
||||
provider = "Claude"
|
||||
elif "Gemini Schema" in line:
|
||||
provider = "Gemini"
|
||||
else:
|
||||
provider = None
|
||||
schema = None
|
||||
continue
|
||||
|
||||
if provider is None:
|
||||
continue
|
||||
|
||||
if heading := parse_schema_heading(line):
|
||||
schema = heading
|
||||
continue
|
||||
|
||||
if schema is None or not line.startswith("| `"):
|
||||
continue
|
||||
|
||||
cells = split_markdown_row(line)
|
||||
if len(cells) < 4 or cells[2] not in {"是", "否"}:
|
||||
continue
|
||||
fields.append(
|
||||
SourceField(
|
||||
provider=provider,
|
||||
schema=schema,
|
||||
field=strip_markdown_code(cells[1]),
|
||||
required=cells[2],
|
||||
field_type=strip_markdown_code(cells[3]),
|
||||
)
|
||||
)
|
||||
|
||||
return fields
|
||||
|
||||
|
||||
def parse_existing_coverage(
|
||||
matrix: str,
|
||||
) -> tuple[dict[tuple[str, str, str], CoverageStatus], dict[tuple[str, str], list[CoverageStatus]]]:
|
||||
existing: dict[tuple[str, str, str], CoverageStatus] = {}
|
||||
profiles: dict[tuple[str, str], list[CoverageStatus]] = defaultdict(list)
|
||||
|
||||
for line in matrix.splitlines():
|
||||
if not line.startswith("| "):
|
||||
continue
|
||||
cells = split_markdown_row(line)
|
||||
if len(cells) < 11 or cells[1] not in {"OpenAI", "Claude", "Gemini"}:
|
||||
continue
|
||||
status = CoverageStatus(
|
||||
surface=cells[6],
|
||||
same_format_runtime=cells[7],
|
||||
canonical_roundtrip=cells[8],
|
||||
cross_format=cells[9],
|
||||
notes=cells[10],
|
||||
)
|
||||
provider = cells[1]
|
||||
schema = strip_markdown_code(cells[2])
|
||||
field = strip_markdown_code(cells[3])
|
||||
existing[(provider, schema, field)] = status
|
||||
profiles[(provider, schema)].append(status)
|
||||
|
||||
return existing, profiles
|
||||
|
||||
|
||||
def most_common(values: Iterable[str]) -> str | None:
|
||||
values = list(values)
|
||||
if not values:
|
||||
return None
|
||||
return Counter(values).most_common(1)[0][0]
|
||||
|
||||
|
||||
def openai_surface(schema: str) -> str:
|
||||
if "CreateChatCompletion" in schema or "ChatCompletion" in schema:
|
||||
return "openai:chat standard"
|
||||
if "CreateEmbedding" in schema or "Embedding" in schema:
|
||||
return "openai:embedding"
|
||||
if any(token in schema for token in ("CreateImage", "EditImage", "Image", "Images")):
|
||||
return "openai:image native-only"
|
||||
if any(token in schema for token in ("Compact", "Compaction")):
|
||||
return "openai:responses:compact native-only"
|
||||
if any(
|
||||
token in schema
|
||||
for token in (
|
||||
"Response",
|
||||
"Input",
|
||||
"Output",
|
||||
"Tool",
|
||||
"Reasoning",
|
||||
"WebSearch",
|
||||
"FileSearch",
|
||||
"Computer",
|
||||
"MCP",
|
||||
"CodeInterpreter",
|
||||
"Function",
|
||||
"Custom",
|
||||
"EasyInput",
|
||||
"Prompt",
|
||||
"Conversation",
|
||||
"Annotation",
|
||||
"Citation",
|
||||
"LogProb",
|
||||
"TopLogProb",
|
||||
"Metadata",
|
||||
"ServiceTier",
|
||||
"Verbosity",
|
||||
"TextResponse",
|
||||
"ResponseFormat",
|
||||
"Include",
|
||||
"Modalities",
|
||||
"ParallelToolCalls",
|
||||
"StopConfiguration",
|
||||
)
|
||||
):
|
||||
return "openai:responses standard"
|
||||
return "openai auxiliary / not-in-conversion-surface"
|
||||
|
||||
|
||||
def openai_default_status(field: SourceField, profile: list[CoverageStatus]) -> CoverageStatus:
|
||||
surface = most_common(status.surface for status in profile) or openai_surface(field.schema)
|
||||
if "not-in-conversion-surface" in surface or "native-only" in surface:
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="not-in-conversion-surface",
|
||||
cross_format="not-in-conversion-surface",
|
||||
notes="not part of current canonical cross-format conversion; same-format runtime path remains provider-native when routed directly",
|
||||
)
|
||||
if field.schema == "CreateChatCompletionRequest":
|
||||
if field.field in OPENAI_CHAT_MAPPED:
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="mapped",
|
||||
cross_format="mapped",
|
||||
notes="Chat request field maps provider-specifically; target-incompatible cases fail closed",
|
||||
)
|
||||
if field.field in OPENAI_CHAT_BLOCKED:
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="extension-preserved",
|
||||
cross_format="lossy-blocked",
|
||||
notes="Chat-only or provider-specific field has no audited lossless target equivalent",
|
||||
)
|
||||
if field.schema == "CreateResponse":
|
||||
if field.field in OPENAI_RESPONSES_MAPPED:
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="mapped",
|
||||
cross_format="mapped",
|
||||
notes="Responses request field maps provider-specifically; target-incompatible cases fail closed",
|
||||
)
|
||||
if field.field in OPENAI_RESPONSES_BLOCKED:
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="extension-preserved",
|
||||
cross_format="lossy-blocked",
|
||||
notes="Responses-only field has no audited lossless Chat/Claude/Gemini target equivalent",
|
||||
)
|
||||
if profile:
|
||||
cross_format = most_common(status.cross_format for status in profile) or "lossy-blocked"
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip=most_common(status.canonical_roundtrip for status in profile)
|
||||
or "extension-preserved",
|
||||
cross_format=cross_format,
|
||||
notes=next(
|
||||
(status.notes for status in profile if status.cross_format == cross_format),
|
||||
"schema-level handling inherited from audited sibling fields",
|
||||
),
|
||||
)
|
||||
return CoverageStatus(
|
||||
surface=surface,
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="extension-preserved",
|
||||
cross_format="lossy-blocked",
|
||||
notes="OpenAI documented field is preserved same-format; cross-format requires explicit target mapping or fails closed",
|
||||
)
|
||||
|
||||
|
||||
def claude_default_status(field: SourceField) -> CoverageStatus:
|
||||
if "CountTokens" in field.schema:
|
||||
return CoverageStatus(
|
||||
surface="claude:messages/count_tokens native-only",
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="not-in-conversion-surface",
|
||||
cross_format="not-in-conversion-surface",
|
||||
notes="count_tokens schemas are provider-native and outside canonical generation conversion",
|
||||
)
|
||||
if field.field in CLAUDE_MAPPED_FIELDS:
|
||||
return CoverageStatus(
|
||||
surface="claude:messages standard",
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="mapped",
|
||||
cross_format="mapped/lossy-blocked",
|
||||
notes="Claude field maps where canonical and target support an equivalent; otherwise conversion fails closed",
|
||||
)
|
||||
if (
|
||||
field.field in CLAUDE_PROVIDER_ONLY_FIELDS
|
||||
or field.field.endswith("_tokens_details")
|
||||
or "cache" in field.field
|
||||
):
|
||||
return CoverageStatus(
|
||||
surface="claude:messages standard",
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="extension-preserved",
|
||||
cross_format="lossy-blocked",
|
||||
notes="Claude provider-specific field is preserved same-format and blocked cross-format without an audited target equivalent",
|
||||
)
|
||||
return CoverageStatus(
|
||||
surface="claude:messages standard",
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="extension-preserved",
|
||||
cross_format="lossy-blocked",
|
||||
notes="Claude nested/provider-specific field is same-format preserved; cross-format requires explicit mapping or fails closed",
|
||||
)
|
||||
|
||||
|
||||
def gemini_default_status(field: SourceField, profile: list[CoverageStatus]) -> CoverageStatus:
|
||||
if profile:
|
||||
cross_format = most_common(status.cross_format for status in profile) or "lossy-blocked"
|
||||
return CoverageStatus(
|
||||
surface=most_common(status.surface for status in profile)
|
||||
or "gemini:generate_content standard",
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip=most_common(status.canonical_roundtrip for status in profile)
|
||||
or "extension-preserved",
|
||||
cross_format=cross_format,
|
||||
notes=next(
|
||||
(status.notes for status in profile if status.cross_format == cross_format),
|
||||
"Gemini field follows schema-level handling",
|
||||
),
|
||||
)
|
||||
return CoverageStatus(
|
||||
surface="gemini:generate_content standard",
|
||||
same_format_runtime="native",
|
||||
canonical_roundtrip="extension-preserved",
|
||||
cross_format="lossy-blocked",
|
||||
notes="Gemini documented field is preserved same-format; cross-format requires explicit mapping or fails closed",
|
||||
)
|
||||
|
||||
|
||||
def default_status(field: SourceField, profile: list[CoverageStatus]) -> CoverageStatus:
|
||||
if field.provider == "OpenAI":
|
||||
return openai_default_status(field, profile)
|
||||
if field.provider == "Claude":
|
||||
return claude_default_status(field)
|
||||
if field.provider == "Gemini":
|
||||
return gemini_default_status(field, profile)
|
||||
raise ValueError(f"unsupported provider: {field.provider}")
|
||||
|
||||
|
||||
def render_matrix(
|
||||
fields: list[SourceField],
|
||||
existing: dict[tuple[str, str, str], CoverageStatus],
|
||||
profiles: dict[tuple[str, str], list[CoverageStatus]],
|
||||
) -> str:
|
||||
rows: list[str] = [
|
||||
"# Format Field Coverage Matrix",
|
||||
"",
|
||||
"Last generated: 2026-06-03",
|
||||
"",
|
||||
"This file is generated from the schema inventory in `docs/api/provider-interface-definitions.md` and gives every documented schema field an explicit handling status. “处理到” here means the field is either mapped, preserved in same-format paths, rejected with a structured fail-closed error, or explicitly outside the current conversion surface. It does not mean every field can be cross-format converted.",
|
||||
"",
|
||||
"Provider schema updates do not require immediate conversion-code changes for runtime safety. Same-format runtime paths bypass canonical conversion, and same-format canonical roundtrip preserves provider extension fields. Cross-format conversion only enables fields with an audited semantic mapping; newly discovered or unknown provider fields default to structured fail-closed behavior until mapped.",
|
||||
"",
|
||||
"Regenerate with: `python3 docs/api/generate_format_field_coverage.py`.",
|
||||
"",
|
||||
"Statuses used in this matrix: `native`, `mapped`, `mapped/lossy-blocked`, `extension-preserved`, `unaudited`, `unsupported`, `invalid-enum`, `lossy-blocked`, `not-in-conversion-surface`.",
|
||||
"",
|
||||
"| Provider | Schema | Field | Required | Type | Surface | Same-Format Runtime | Canonical Roundtrip | Cross-Format | Notes |",
|
||||
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
|
||||
]
|
||||
|
||||
for field in fields:
|
||||
status = existing.get(
|
||||
(field.provider, field.schema, field.field),
|
||||
default_status(field, profiles[(field.provider, field.schema)]),
|
||||
)
|
||||
rows.append(
|
||||
"| "
|
||||
+ " | ".join(
|
||||
[
|
||||
field.provider,
|
||||
f"`{escape_markdown_cell(field.schema)}`",
|
||||
f"`{escape_markdown_cell(field.field)}`",
|
||||
field.required,
|
||||
f"`{escape_markdown_cell(field.field_type)}`",
|
||||
escape_markdown_cell(status.surface),
|
||||
escape_markdown_cell(status.same_format_runtime),
|
||||
escape_markdown_cell(status.canonical_roundtrip),
|
||||
escape_markdown_cell(status.cross_format),
|
||||
escape_markdown_cell(status.notes),
|
||||
]
|
||||
)
|
||||
+ " |"
|
||||
)
|
||||
|
||||
rows.extend(["", f"Total covered schema fields: {len(fields)}."])
|
||||
return "\n".join(rows) + "\n"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--definitions", type=Path, default=DEFAULT_DEFINITIONS)
|
||||
parser.add_argument("--matrix", type=Path, default=DEFAULT_MATRIX)
|
||||
parser.add_argument("--check", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
definitions = args.definitions.read_text()
|
||||
current_matrix = args.matrix.read_text() if args.matrix.exists() else ""
|
||||
fields = parse_provider_definition_fields(definitions)
|
||||
existing, profiles = parse_existing_coverage(current_matrix)
|
||||
next_matrix = render_matrix(fields, existing, profiles)
|
||||
|
||||
if args.check:
|
||||
if current_matrix != next_matrix:
|
||||
print(
|
||||
f"{args.matrix} is not up to date; run "
|
||||
"`python3 docs/api/generate_format_field_coverage.py`",
|
||||
)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
args.matrix.write_text(next_matrix)
|
||||
print(f"wrote {len(fields)} field coverage rows to {args.matrix}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user