mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-03 16:07:46 +08:00
feat(ai-formats): deliver Gemini grounding to every client as native citations
Gemini runs `googleSearch` inside Google. The search leaves no client-visible tool call, and the evidence arrives only as `candidates[].groundingMetadata`. Every cross-format target dropped it wholesale, so a grounded answer reached OpenAI- and Claude-shaped clients as prose that names its sources with nothing structured behind it: no `annotations`, no `citations`, no `url_citation`. Callers that verify grounding — the common "did this model actually search?" check — saw a 200 with no evidence and had to treat the answer as ungrounded. Adapters now normalise `groundingMetadata` into neutral citations and each target renders its own family's standard shape: `url_citation` annotations for `openai:chat` and `openai:responses`, and `web_search_result_location` citations on the text block for `claude:messages`. Gemini reports segment bounds as UTF-8 byte offsets while both targets count characters, so the bounds are converted rather than copied. Streaming is covered too, since that is what grounded traffic actually uses. A new `CanonicalStreamEvent::Citations` carries the neutral list once the answer text is whole — the offsets index into the finished answer, so it rides just ahead of `Finish` rather than as a delta per chunk — and each client emitter renders it: `delta.annotations` chunks, `response.output_text.annotation.added` events (also kept on the finished message item so clients that only read `response.completed` see them), and `citations_delta` content block deltas. For reference, CLIProxyAPI projects grounding only in its antigravity→Claude translator, and only when the client declared a typed `web_search_*` tool; its OpenAI and plain Gemini translators have no grounding handling at all. The citation shape here matches theirs, but the coverage is deliberately wider: all three targets, streaming and non-streaming, with no dependency on a declared tool. Co-Authored-By: Claude Opus 5 <[email protected]>
This commit is contained in:
@@ -2,6 +2,7 @@ use std::collections::BTreeMap;
|
||||
|
||||
use serde_json::{json, Map, Value};
|
||||
|
||||
use crate::formats::shared::citations::canonical_citations_to_claude_citations;
|
||||
use crate::formats::shared::response::{
|
||||
build_generated_tool_call_id, canonicalize_tool_arguments,
|
||||
remove_empty_pages_from_tool_arguments,
|
||||
@@ -773,6 +774,33 @@ impl ClaudeClientEmitter {
|
||||
name,
|
||||
content,
|
||||
} => self.emit_tool_result_block(index, tool_use_id, name, content),
|
||||
CanonicalStreamEvent::Citations(citations) => {
|
||||
let citations = canonical_citations_to_claude_citations(&citations);
|
||||
if citations.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
// Citations belong to the answer text. If a tool call or a
|
||||
// thinking block closed it, open a fresh text block rather than
|
||||
// hang the evidence off an unrelated one.
|
||||
let mut out = self.ensure_text_block()?;
|
||||
let Some(ClaudeOpenBlock::Text { block_index }) = self.open_block else {
|
||||
return Ok(out);
|
||||
};
|
||||
for citation in citations {
|
||||
out.extend(encode_json_sse(
|
||||
Some("content_block_delta"),
|
||||
&json!({
|
||||
"type": "content_block_delta",
|
||||
"index": block_index,
|
||||
"delta": {
|
||||
"type": "citations_delta",
|
||||
"citation": citation,
|
||||
}
|
||||
}),
|
||||
)?);
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
CanonicalStreamEvent::UnknownEvent(_) => Ok(Vec::new()),
|
||||
CanonicalStreamEvent::Finish {
|
||||
finish_reason,
|
||||
|
||||
@@ -2,15 +2,176 @@ use serde_json::{json, Map, Value};
|
||||
|
||||
use crate::{
|
||||
formats::context::FormatContext,
|
||||
formats::shared::citations::{
|
||||
canonical_citation, canonical_citations_to_claude_citations,
|
||||
canonical_citations_to_openai_annotations,
|
||||
},
|
||||
protocol::canonical::{
|
||||
canonical_extension_object_mut, canonical_usage_total_input_tokens,
|
||||
canonical_usage_total_tokens_for_inclusive_input, gemini_extensions,
|
||||
gemini_part_to_canonical_block, gemini_stop_reason_to_canonical, gemini_usage_to_canonical,
|
||||
CanonicalContentBlock, CanonicalResponse, CanonicalResponseOutput, CanonicalRole,
|
||||
CanonicalStopReason, CanonicalUsage,
|
||||
CanonicalStopReason, CanonicalUsage, CLAUDE_EXTENSION_NAMESPACE,
|
||||
OPENAI_RESPONSES_EXTENSION_NAMESPACE,
|
||||
},
|
||||
};
|
||||
|
||||
/// Project Gemini grounding metadata onto the answer text as structured
|
||||
/// citations.
|
||||
///
|
||||
/// Native `googleSearch` grounding runs inside Google, so there is no
|
||||
/// client-visible tool call and the evidence only exists in
|
||||
/// `candidates[].groundingMetadata`. Cross-format targets used to drop that
|
||||
/// wholesale, leaving callers with prose that names its sources but nothing a
|
||||
/// client can render or verify. Every grounded span is therefore emitted twice,
|
||||
/// each time in the target family's own standard shape: OpenAI `url_citation`
|
||||
/// annotations and Claude `web_search_result_location` citations. Both ride
|
||||
/// extension namespaces the respective emitters already merge onto the text
|
||||
/// block, so no target has to learn anything Gemini-specific.
|
||||
fn attach_gemini_grounding_citations(
|
||||
candidate: &Map<String, Value>,
|
||||
content: &mut [CanonicalContentBlock],
|
||||
) {
|
||||
let Some(grounding) = gemini_candidate_grounding(candidate) else {
|
||||
return;
|
||||
};
|
||||
let Some(block) = content.iter_mut().find(|block| {
|
||||
matches!(block, CanonicalContentBlock::Text { text, .. } if !text.trim().is_empty())
|
||||
}) else {
|
||||
return;
|
||||
};
|
||||
let CanonicalContentBlock::Text { text, extensions } = block else {
|
||||
return;
|
||||
};
|
||||
|
||||
let citations = gemini_grounding_citations(grounding, text);
|
||||
if citations.is_empty() {
|
||||
return;
|
||||
}
|
||||
let annotations = canonical_citations_to_openai_annotations(&citations);
|
||||
let claude_citations = canonical_citations_to_claude_citations(&citations);
|
||||
canonical_extension_object_mut(extensions, OPENAI_RESPONSES_EXTENSION_NAMESPACE)
|
||||
.entry("annotations".to_string())
|
||||
.or_insert_with(|| Value::Array(annotations));
|
||||
canonical_extension_object_mut(extensions, CLAUDE_EXTENSION_NAMESPACE)
|
||||
.entry("citations".to_string())
|
||||
.or_insert_with(|| Value::Array(claude_citations));
|
||||
}
|
||||
|
||||
pub(crate) fn gemini_candidate_grounding(candidate: &Map<String, Value>) -> Option<&Value> {
|
||||
candidate
|
||||
.get("groundingMetadata")
|
||||
.or_else(|| candidate.get("grounding_metadata"))
|
||||
}
|
||||
|
||||
/// Normalise `groundingMetadata` into neutral citations against `text`.
|
||||
///
|
||||
/// Gemini reports segment bounds as UTF-8 byte offsets while every target
|
||||
/// counts characters, so the bounds are converted rather than copied.
|
||||
pub(crate) fn gemini_grounding_citations(grounding: &Value, text: &str) -> Vec<Value> {
|
||||
let chunks = grounding
|
||||
.get("groundingChunks")
|
||||
.and_then(Value::as_array)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or_default();
|
||||
if chunks.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
let supports = grounding
|
||||
.get("groundingSupports")
|
||||
.and_then(Value::as_array)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or_default();
|
||||
|
||||
let mut citations = Vec::new();
|
||||
for support in supports {
|
||||
let segment = support.get("segment");
|
||||
let start = segment
|
||||
.and_then(|segment| segment.get("startIndex"))
|
||||
.and_then(Value::as_u64)
|
||||
.unwrap_or(0);
|
||||
let end = segment
|
||||
.and_then(|segment| segment.get("endIndex"))
|
||||
.and_then(Value::as_u64);
|
||||
let start_byte = gemini_clamped_byte_offset(text, start);
|
||||
let end_byte = end
|
||||
.map(|end| gemini_clamped_byte_offset(text, end))
|
||||
.filter(|end| *end >= start_byte);
|
||||
let cited_text = segment
|
||||
.and_then(|segment| segment.get("text"))
|
||||
.and_then(Value::as_str)
|
||||
.or_else(|| end_byte.map(|end| &text[start_byte..end]))
|
||||
.map(str::trim)
|
||||
.filter(|cited_text| !cited_text.is_empty());
|
||||
let indices = support
|
||||
.get("groundingChunkIndices")
|
||||
.and_then(Value::as_array)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or_default();
|
||||
for index in indices {
|
||||
let Some(chunk) = index
|
||||
.as_u64()
|
||||
.and_then(|index| usize::try_from(index).ok())
|
||||
.and_then(|index| chunks.get(index))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let Some((uri, title)) = gemini_grounding_chunk_source(chunk) else {
|
||||
continue;
|
||||
};
|
||||
citations.push(canonical_citation(
|
||||
uri,
|
||||
title,
|
||||
Some(text[..start_byte].chars().count()),
|
||||
end_byte.map(|end| text[..end].chars().count()),
|
||||
cited_text,
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// `groundingSupports` is optional; without it the chunks are still the
|
||||
// evidence, just unanchored.
|
||||
if citations.is_empty() {
|
||||
for chunk in chunks {
|
||||
let Some((uri, title)) = gemini_grounding_chunk_source(chunk) else {
|
||||
continue;
|
||||
};
|
||||
citations.push(canonical_citation(uri, title, None, None, None));
|
||||
}
|
||||
}
|
||||
citations
|
||||
}
|
||||
|
||||
fn gemini_grounding_chunk_source(chunk: &Value) -> Option<(&str, Option<&str>)> {
|
||||
let source = chunk.get("web").or_else(|| chunk.get("retrievedContext"))?;
|
||||
let uri = source
|
||||
.get("uri")
|
||||
.or_else(|| source.get("url"))
|
||||
.and_then(Value::as_str)
|
||||
.map(str::trim)
|
||||
.filter(|uri| !uri.is_empty())?;
|
||||
let title = source
|
||||
.get("title")
|
||||
.and_then(Value::as_str)
|
||||
.map(str::trim)
|
||||
.filter(|title| !title.is_empty());
|
||||
Some((uri, title))
|
||||
}
|
||||
|
||||
/// Gemini offsets are byte counts into the UTF-8 answer. A truncated or stale
|
||||
/// offset must not panic the conversion, so snap it into range and back onto a
|
||||
/// character boundary.
|
||||
fn gemini_clamped_byte_offset(text: &str, byte_offset: u64) -> usize {
|
||||
let mut offset = usize::try_from(byte_offset)
|
||||
.unwrap_or(text.len())
|
||||
.min(text.len());
|
||||
while offset > 0 && !text.is_char_boundary(offset) {
|
||||
offset -= 1;
|
||||
}
|
||||
offset
|
||||
}
|
||||
|
||||
pub fn from(body: &Value, _ctx: &FormatContext) -> Option<CanonicalResponse> {
|
||||
from_raw(body)
|
||||
}
|
||||
@@ -37,11 +198,12 @@ pub fn from_raw(body_json: &Value) -> Option<CanonicalResponse> {
|
||||
.and_then(Value::as_array)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or(&[]);
|
||||
let content = parts
|
||||
let mut content = parts
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, part)| gemini_part_to_canonical_block(part, index))
|
||||
.collect::<Vec<_>>();
|
||||
attach_gemini_grounding_citations(candidate_object, &mut content);
|
||||
let mut stop_reason = candidate_object
|
||||
.get("finishReason")
|
||||
.or_else(|| candidate_object.get("finish_reason"))
|
||||
@@ -433,6 +595,49 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::CanonicalContentBlock;
|
||||
|
||||
/// Gemini omits `groundingSupports` when it cannot anchor the answer to a
|
||||
/// span. The sources are still real, so they must survive unanchored
|
||||
/// rather than be dropped for lacking offsets.
|
||||
#[test]
|
||||
fn grounding_without_supports_still_yields_unanchored_citations() {
|
||||
let body = json!({
|
||||
"responseId": "resp-unanchored",
|
||||
"candidates": [{
|
||||
"content": {"role": "model", "parts": [{"text": "Rust 1.95 is current."}]},
|
||||
"finishReason": "STOP",
|
||||
"groundingMetadata": {
|
||||
"groundingChunks": [
|
||||
{"web": {"uri": "https://blog.rust-lang.org/", "title": "Rust Blog"}},
|
||||
{"web": {"title": "no uri here"}}
|
||||
]
|
||||
}
|
||||
}]
|
||||
});
|
||||
|
||||
let canonical = from_raw(&body).expect("canonical");
|
||||
let CanonicalContentBlock::Text { extensions, .. } = &canonical.outputs[0].content[0]
|
||||
else {
|
||||
panic!("expected a text block");
|
||||
};
|
||||
|
||||
assert_eq!(
|
||||
extensions["claude"]["citations"],
|
||||
json!([{
|
||||
"type": "web_search_result_location",
|
||||
"url": "https://blog.rust-lang.org/",
|
||||
"title": "Rust Blog",
|
||||
}])
|
||||
);
|
||||
assert_eq!(
|
||||
extensions["openai_responses"]["annotations"],
|
||||
json!([{
|
||||
"type": "url_citation",
|
||||
"url": "https://blog.rust-lang.org/",
|
||||
"title": "Rust Blog",
|
||||
}])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gemini_response_without_visible_parts_is_not_success() {
|
||||
let body = json!({
|
||||
|
||||
@@ -2,6 +2,9 @@ use std::collections::BTreeMap;
|
||||
|
||||
use serde_json::{json, Map, Value};
|
||||
|
||||
use crate::formats::gemini::generate_content::response::{
|
||||
gemini_candidate_grounding, gemini_grounding_citations,
|
||||
};
|
||||
use crate::formats::shared::response::{build_generated_tool_call_id, canonicalize_tool_arguments};
|
||||
use crate::formats::shared::sse::encode_json_sse;
|
||||
use crate::formats::shared::stream_core::common::*;
|
||||
@@ -36,6 +39,10 @@ pub struct GeminiProviderState {
|
||||
content_parts: BTreeMap<usize, CanonicalContentPart>,
|
||||
tool_calls: BTreeMap<usize, GeminiProviderToolState>,
|
||||
tool_results: BTreeMap<usize, GeminiProviderToolResultState>,
|
||||
/// Last `groundingMetadata` seen. Gemini resends it cumulatively, so the
|
||||
/// newest copy is the complete one; citations are emitted once at finish,
|
||||
/// when the answer text they index into is whole.
|
||||
grounding: Option<Value>,
|
||||
}
|
||||
|
||||
impl GeminiProviderState {
|
||||
@@ -68,6 +75,31 @@ impl GeminiProviderState {
|
||||
self.started = true;
|
||||
}
|
||||
|
||||
/// Turn the grounding metadata collected over the stream into citations.
|
||||
///
|
||||
/// The offsets Gemini reports index into the finished answer, so this can
|
||||
/// only run once the text is complete — hence a single frame just ahead of
|
||||
/// `Finish` rather than a delta per chunk.
|
||||
fn push_citations_frame(&mut self, id: &str, model: &str, out: &mut Vec<CanonicalStreamFrame>) {
|
||||
let Some(grounding) = self.grounding.take() else {
|
||||
return;
|
||||
};
|
||||
let text = self
|
||||
.text_parts
|
||||
.values()
|
||||
.map(String::as_str)
|
||||
.collect::<String>();
|
||||
let citations = gemini_grounding_citations(&grounding, &text);
|
||||
if citations.is_empty() {
|
||||
return;
|
||||
}
|
||||
out.push(CanonicalStreamFrame {
|
||||
id: id.to_string(),
|
||||
model: model.to_string(),
|
||||
event: CanonicalStreamEvent::Citations(citations),
|
||||
});
|
||||
}
|
||||
|
||||
fn unknown_frame(&self, report_context: &Value, payload: Value) -> CanonicalStreamFrame {
|
||||
let (id, model) = self.identity(report_context);
|
||||
CanonicalStreamFrame {
|
||||
@@ -120,6 +152,11 @@ impl GeminiProviderState {
|
||||
response_model.as_str(),
|
||||
event_object.get("usageMetadata"),
|
||||
);
|
||||
if !self.terminal_observation_only {
|
||||
if let Some(grounding) = gemini_candidate_grounding(candidate_object) {
|
||||
self.grounding = Some(grounding.clone());
|
||||
}
|
||||
}
|
||||
let Some(content) = candidate_object.get("content").and_then(Value::as_object) else {
|
||||
if let Some(payload) = terminal_error {
|
||||
out.push(self.unknown_frame(report_context, payload));
|
||||
@@ -361,6 +398,7 @@ impl GeminiProviderState {
|
||||
if has_tool_calls && finish_reason.as_deref().is_none_or(|value| value == "stop") {
|
||||
finish_reason = Some("tool_calls".to_string());
|
||||
}
|
||||
self.push_citations_frame(&id, &model, &mut out);
|
||||
out.push(CanonicalStreamFrame {
|
||||
id,
|
||||
model,
|
||||
@@ -385,14 +423,17 @@ impl GeminiProviderState {
|
||||
}
|
||||
self.finished = true;
|
||||
let (id, model) = self.identity(report_context);
|
||||
Ok(vec![CanonicalStreamFrame {
|
||||
let mut out = Vec::new();
|
||||
self.push_citations_frame(&id, &model, &mut out);
|
||||
out.push(CanonicalStreamFrame {
|
||||
id,
|
||||
model,
|
||||
event: CanonicalStreamEvent::Finish {
|
||||
finish_reason: None,
|
||||
usage: None,
|
||||
},
|
||||
}])
|
||||
});
|
||||
Ok(out)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -674,6 +715,9 @@ impl GeminiClientEmitter {
|
||||
None,
|
||||
None,
|
||||
),
|
||||
// Only Gemini produces citations today, and a Gemini-to-Gemini
|
||||
// stream keeps its own `groundingMetadata` on the passthrough path.
|
||||
CanonicalStreamEvent::Citations(_) => Ok(Vec::new()),
|
||||
CanonicalStreamEvent::UnknownEvent(_) => Ok(Vec::new()),
|
||||
CanonicalStreamEvent::Finish {
|
||||
finish_reason,
|
||||
|
||||
@@ -13,6 +13,7 @@ use crate::formats::openai::responses::{
|
||||
},
|
||||
GeminiToolSignatureCarrierDirection,
|
||||
};
|
||||
use crate::formats::shared::citations::canonical_citations_to_openai_annotations;
|
||||
use crate::formats::shared::response::build_generated_tool_call_id;
|
||||
use crate::formats::shared::sse::{encode_done_sse, encode_json_sse};
|
||||
use crate::formats::shared::stream_core::common::*;
|
||||
@@ -2161,6 +2162,9 @@ pub struct OpenAIResponsesClientEmitter {
|
||||
text_item_started: bool,
|
||||
text_part_started: bool,
|
||||
message_output_index: Option<usize>,
|
||||
/// Citations projected onto the answer text, kept on the finished message
|
||||
/// item so non-incremental clients see them too.
|
||||
annotations: Vec<Value>,
|
||||
text: String,
|
||||
reasoning: String,
|
||||
reasoning_part: String,
|
||||
@@ -2286,6 +2290,28 @@ impl OpenAIChatClientEmitter {
|
||||
}))?);
|
||||
Ok(out)
|
||||
}
|
||||
CanonicalStreamEvent::Citations(citations) => {
|
||||
let annotations = canonical_citations_to_openai_annotations(&citations);
|
||||
if annotations.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let mut out = self.ensure_started()?;
|
||||
out.extend(self.encode_chunk(json!({
|
||||
"id": self.response_id
|
||||
.as_deref()
|
||||
.unwrap_or("chatcmpl-local-stream"),
|
||||
"object": "chat.completion.chunk",
|
||||
"model": self.model.as_deref().unwrap_or("unknown"),
|
||||
"choices": [{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"annotations": annotations,
|
||||
},
|
||||
"finish_reason": Value::Null
|
||||
}]
|
||||
}))?);
|
||||
Ok(out)
|
||||
}
|
||||
CanonicalStreamEvent::ReasoningSignature(_) => Ok(Vec::new()),
|
||||
CanonicalStreamEvent::ContentPart(part) => {
|
||||
let placeholder = openai_stream_placeholder_for_content_part(&part);
|
||||
@@ -2894,7 +2920,7 @@ impl OpenAIResponsesClientEmitter {
|
||||
"part": {
|
||||
"type": "output_text",
|
||||
"text": self.text.as_str(),
|
||||
"annotations": [],
|
||||
"annotations": self.annotations.as_slice(),
|
||||
}
|
||||
}),
|
||||
)?);
|
||||
@@ -2913,7 +2939,7 @@ impl OpenAIResponsesClientEmitter {
|
||||
"content": [{
|
||||
"type": "output_text",
|
||||
"text": self.text.as_str(),
|
||||
"annotations": [],
|
||||
"annotations": self.annotations.as_slice(),
|
||||
}],
|
||||
}
|
||||
}),
|
||||
@@ -3128,7 +3154,7 @@ impl OpenAIResponsesClientEmitter {
|
||||
"content": [{
|
||||
"type": "output_text",
|
||||
"text": self.text.as_str(),
|
||||
"annotations": [],
|
||||
"annotations": self.annotations.as_slice(),
|
||||
}],
|
||||
}),
|
||||
));
|
||||
@@ -3351,6 +3377,35 @@ impl OpenAIResponsesClientEmitter {
|
||||
out.extend(self.encode_reasoning_text_delta(&text)?);
|
||||
Ok(out)
|
||||
}
|
||||
CanonicalStreamEvent::Citations(citations) => {
|
||||
let annotations = canonical_citations_to_openai_annotations(&citations);
|
||||
if annotations.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
// The text item has to exist before an annotation can point at
|
||||
// it, and the annotations are also kept on the item itself so
|
||||
// clients that only read `response.completed` still see them.
|
||||
let mut out = self.ensure_text_item_started()?;
|
||||
let item_id = self.message_item_id();
|
||||
let output_index = self.message_output_index.unwrap_or(0);
|
||||
for annotation in annotations {
|
||||
let annotation_index = self.annotations.len();
|
||||
self.annotations.push(annotation.clone());
|
||||
out.extend(self.encode_response_event(
|
||||
"response.output_text.annotation.added",
|
||||
json!({
|
||||
"type": "response.output_text.annotation.added",
|
||||
"response_id": self.response_id(),
|
||||
"output_index": output_index,
|
||||
"item_id": item_id,
|
||||
"content_index": 0,
|
||||
"annotation_index": annotation_index,
|
||||
"annotation": annotation,
|
||||
}),
|
||||
)?);
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
CanonicalStreamEvent::ReasoningSummaryDone => {
|
||||
// Close the current reasoning part and reset state so the next
|
||||
// ReasoningDelta starts a fresh part within the same item.
|
||||
|
||||
@@ -3594,6 +3594,108 @@ mod tests {
|
||||
.any(|field| field.field == "messages"));
|
||||
}
|
||||
|
||||
/// Gemini runs `googleSearch` server-side, so the only trace of the search
|
||||
/// is `groundingMetadata`. Clients on the other formats have to receive it
|
||||
/// as their own native citations or the answer arrives unverifiable.
|
||||
#[test]
|
||||
fn gemini_grounding_reaches_every_cross_format_client_as_citations() {
|
||||
let gemini = grounded_gemini_response();
|
||||
|
||||
for target in ["openai:chat", "openai:responses"] {
|
||||
let converted =
|
||||
convert_response_pure("gemini:generate_content", target, &gemini).expect(target);
|
||||
let body = serde_json::to_string(&converted.value).expect("serialize");
|
||||
let annotations = find_first_array(&converted.value, "annotations")
|
||||
.unwrap_or_else(|| panic!("{target} dropped the grounding metadata: {body}"));
|
||||
assert_eq!(
|
||||
annotations,
|
||||
&json!([{
|
||||
"type": "url_citation",
|
||||
"url": "https://time.gov/",
|
||||
"title": "time.gov",
|
||||
"start_index": 0,
|
||||
"end_index": 9,
|
||||
}]),
|
||||
"{target} annotations"
|
||||
);
|
||||
}
|
||||
|
||||
let converted =
|
||||
convert_response_pure("gemini:generate_content", "claude:messages", &gemini)
|
||||
.expect("claude:messages");
|
||||
let body = serde_json::to_string(&converted.value).expect("serialize");
|
||||
let citations = find_first_array(&converted.value, "citations")
|
||||
.unwrap_or_else(|| panic!("claude:messages dropped the grounding metadata: {body}"));
|
||||
assert_eq!(
|
||||
citations,
|
||||
&json!([{
|
||||
"type": "web_search_result_location",
|
||||
"url": "https://time.gov/",
|
||||
"title": "time.gov",
|
||||
"cited_text": "今天是 2026",
|
||||
}])
|
||||
);
|
||||
}
|
||||
|
||||
/// The grounded span is reported in UTF-8 bytes but every target counts
|
||||
/// characters, so a multi-byte answer must not shift the citation.
|
||||
#[test]
|
||||
fn gemini_grounding_offsets_are_converted_from_bytes_to_characters() {
|
||||
let converted = convert_response_pure(
|
||||
"gemini:generate_content",
|
||||
"openai:chat",
|
||||
&grounded_gemini_response(),
|
||||
)
|
||||
.expect("convert");
|
||||
let annotation =
|
||||
&find_first_array(&converted.value, "annotations").expect("annotations")[0];
|
||||
|
||||
// "今天是 2026 " is 15 bytes but 9 characters.
|
||||
assert_eq!(annotation["end_index"], json!(9));
|
||||
}
|
||||
|
||||
fn grounded_gemini_response() -> serde_json::Value {
|
||||
json!({
|
||||
"responseId": "resp_grounded",
|
||||
"modelVersion": "gemini-3.8-flash",
|
||||
"candidates": [{
|
||||
"index": 0,
|
||||
"finishReason": "STOP",
|
||||
"groundingMetadata": {
|
||||
"webSearchQueries": ["current UTC date"],
|
||||
"groundingChunks": [{
|
||||
"web": {"uri": "https://time.gov/", "title": "time.gov"}
|
||||
}],
|
||||
"groundingSupports": [{
|
||||
"segment": {"startIndex": 0, "endIndex": 15},
|
||||
"groundingChunkIndices": [0]
|
||||
}]
|
||||
},
|
||||
"content": {"parts": [{"text": "今天是 2026 年"}]}
|
||||
}]
|
||||
})
|
||||
}
|
||||
|
||||
fn find_first_array<'a>(
|
||||
value: &'a serde_json::Value,
|
||||
key: &str,
|
||||
) -> Option<&'a serde_json::Value> {
|
||||
match value {
|
||||
serde_json::Value::Object(object) => {
|
||||
if let Some(found) = object.get(key).filter(|found| found.is_array()) {
|
||||
return Some(found);
|
||||
}
|
||||
object
|
||||
.values()
|
||||
.find_map(|value| find_first_array(value, key))
|
||||
}
|
||||
serde_json::Value::Array(items) => {
|
||||
items.iter().find_map(|item| find_first_array(item, key))
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn runtime_responses_to_gemini_rejects_mixed_tools_for_gemini_two() {
|
||||
let body = json!({
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
//! Provider-neutral source citations.
|
||||
//!
|
||||
//! Some providers ground an answer server-side (Gemini's native `googleSearch`
|
||||
//! is the motivating case): the search leaves no client-visible tool call, and
|
||||
//! the evidence arrives only as provider-specific metadata alongside the text.
|
||||
//! Dropping it leaves callers with prose that names its sources but nothing
|
||||
//! they can render, link, or verify.
|
||||
//!
|
||||
//! Adapters therefore normalise that metadata into the neutral citation shape
|
||||
//! below, and each target renders it into its own family's standard shape.
|
||||
//! Neither side has to learn the other's vocabulary.
|
||||
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
/// Build one neutral citation.
|
||||
///
|
||||
/// `start_index` / `end_index` are character offsets into the answer text —
|
||||
/// providers that report byte offsets convert before calling. Every field but
|
||||
/// `url` is optional, because providers routinely ground an answer without
|
||||
/// anchoring it to a span.
|
||||
pub(crate) fn canonical_citation(
|
||||
url: &str,
|
||||
title: Option<&str>,
|
||||
start_index: Option<usize>,
|
||||
end_index: Option<usize>,
|
||||
cited_text: Option<&str>,
|
||||
) -> Value {
|
||||
let mut citation = Map::new();
|
||||
citation.insert("url".to_string(), Value::String(url.to_string()));
|
||||
if let Some(title) = title {
|
||||
citation.insert("title".to_string(), Value::String(title.to_string()));
|
||||
}
|
||||
if let Some(start_index) = start_index {
|
||||
citation.insert("start_index".to_string(), Value::from(start_index as u64));
|
||||
}
|
||||
if let Some(end_index) = end_index {
|
||||
citation.insert("end_index".to_string(), Value::from(end_index as u64));
|
||||
}
|
||||
if let Some(cited_text) = cited_text {
|
||||
citation.insert(
|
||||
"cited_text".to_string(),
|
||||
Value::String(cited_text.to_string()),
|
||||
);
|
||||
}
|
||||
Value::Object(citation)
|
||||
}
|
||||
|
||||
fn citation_string<'a>(citation: &'a Value, key: &str) -> Option<&'a str> {
|
||||
citation
|
||||
.get(key)
|
||||
.and_then(Value::as_str)
|
||||
.map(str::trim)
|
||||
.filter(|value| !value.is_empty())
|
||||
}
|
||||
|
||||
/// Render a neutral citation as an OpenAI `url_citation` annotation, the shape
|
||||
/// both `chat.completions` and `responses` attach to assistant text.
|
||||
pub(crate) fn canonical_citation_to_openai_annotation(citation: &Value) -> Option<Value> {
|
||||
let url = citation_string(citation, "url")?;
|
||||
let mut annotation = Map::new();
|
||||
annotation.insert(
|
||||
"type".to_string(),
|
||||
Value::String("url_citation".to_string()),
|
||||
);
|
||||
annotation.insert("url".to_string(), Value::String(url.to_string()));
|
||||
if let Some(title) = citation_string(citation, "title") {
|
||||
annotation.insert("title".to_string(), Value::String(title.to_string()));
|
||||
}
|
||||
for key in ["start_index", "end_index"] {
|
||||
if let Some(index) = citation.get(key).and_then(Value::as_u64) {
|
||||
annotation.insert(key.to_string(), Value::from(index));
|
||||
}
|
||||
}
|
||||
Some(Value::Object(annotation))
|
||||
}
|
||||
|
||||
/// Render a neutral citation as a Claude `web_search_result_location`, the
|
||||
/// shape Claude puts in a text block's `citations`.
|
||||
pub(crate) fn canonical_citation_to_claude_citation(citation: &Value) -> Option<Value> {
|
||||
let url = citation_string(citation, "url")?;
|
||||
let mut out = Map::new();
|
||||
out.insert(
|
||||
"type".to_string(),
|
||||
Value::String("web_search_result_location".to_string()),
|
||||
);
|
||||
out.insert("url".to_string(), Value::String(url.to_string()));
|
||||
if let Some(title) = citation_string(citation, "title") {
|
||||
out.insert("title".to_string(), Value::String(title.to_string()));
|
||||
}
|
||||
if let Some(cited_text) = citation_string(citation, "cited_text") {
|
||||
out.insert(
|
||||
"cited_text".to_string(),
|
||||
Value::String(cited_text.to_string()),
|
||||
);
|
||||
}
|
||||
Some(Value::Object(out))
|
||||
}
|
||||
|
||||
/// Render every citation that carries a usable URL.
|
||||
pub(crate) fn canonical_citations_to_openai_annotations(citations: &[Value]) -> Vec<Value> {
|
||||
citations
|
||||
.iter()
|
||||
.filter_map(canonical_citation_to_openai_annotation)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Render every citation that carries a usable URL.
|
||||
pub(crate) fn canonical_citations_to_claude_citations(citations: &[Value]) -> Vec<Value> {
|
||||
citations
|
||||
.iter()
|
||||
.filter_map(canonical_citation_to_claude_citation)
|
||||
.collect()
|
||||
}
|
||||
@@ -6,6 +6,7 @@ use std::fmt;
|
||||
/// a base64 field cannot trigger an unchecked allocation before parsing.
|
||||
pub(crate) const MAX_SYNC_REPORT_BODY_BYTES: usize = 64 * 1024 * 1024;
|
||||
|
||||
pub mod citations;
|
||||
pub mod error_body;
|
||||
pub mod family;
|
||||
pub mod image_bridge;
|
||||
|
||||
@@ -875,6 +875,104 @@ mod tests {
|
||||
format!("event: {event}\n").into_bytes()
|
||||
}
|
||||
|
||||
/// Gemini runs `googleSearch` inside Google, so a grounded streaming answer
|
||||
/// carries its evidence as `groundingMetadata` on the final chunk and never
|
||||
/// as a tool call. Each client family has to receive it in its own citation
|
||||
/// shape, or the answer streams out unverifiable.
|
||||
#[test]
|
||||
fn streams_gemini_grounding_to_every_client_as_native_citations() {
|
||||
let text = "今天是 2026 年";
|
||||
let first = json!({
|
||||
"responseId": "resp_grounded",
|
||||
"modelVersion": "gemini-3.8-flash",
|
||||
"candidates": [{
|
||||
"index": 0,
|
||||
"content": {"role": "model", "parts": [{"text": text}]}
|
||||
}]
|
||||
});
|
||||
let last = json!({
|
||||
"responseId": "resp_grounded",
|
||||
"modelVersion": "gemini-3.8-flash",
|
||||
"candidates": [{
|
||||
"index": 0,
|
||||
"finishReason": "STOP",
|
||||
"content": {"role": "model", "parts": [{"text": text}]},
|
||||
"groundingMetadata": {
|
||||
"webSearchQueries": ["current UTC date"],
|
||||
"groundingChunks": [{
|
||||
"web": {"uri": "https://time.gov/", "title": "time.gov"}
|
||||
}],
|
||||
"groundingSupports": [{
|
||||
"segment": {"startIndex": 0, "endIndex": 15},
|
||||
"groundingChunkIndices": [0]
|
||||
}]
|
||||
}
|
||||
}]
|
||||
});
|
||||
|
||||
for (client_api_format, marker) in [
|
||||
("openai:chat", "\"annotations\":[{\"type\":\"url_citation\""),
|
||||
(
|
||||
"openai:responses",
|
||||
"event: response.output_text.annotation.added\n",
|
||||
),
|
||||
("claude:messages", "\"type\":\"citations_delta\""),
|
||||
] {
|
||||
let context = report_context("gemini:generate_content", client_api_format);
|
||||
let mut matrix = StreamingStandardFormatMatrix::default();
|
||||
let mut output = matrix
|
||||
.transform_line(&context, data_line(first.clone()))
|
||||
.expect("text chunk");
|
||||
output.extend(
|
||||
matrix
|
||||
.transform_line(&context, data_line(last.clone()))
|
||||
.expect("grounded chunk"),
|
||||
);
|
||||
output.extend(matrix.finish(&context).expect("finish"));
|
||||
let sse = String::from_utf8(output).expect("valid SSE");
|
||||
|
||||
assert!(
|
||||
sse.contains(marker),
|
||||
"{client_api_format} missing citations: {sse}"
|
||||
);
|
||||
assert!(
|
||||
sse.contains("https://time.gov/"),
|
||||
"{client_api_format} missing source url: {sse}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The citation frame is emitted once the answer is whole, so a provider
|
||||
/// that closes the stream without a `finishReason` must still deliver it.
|
||||
#[test]
|
||||
fn streams_gemini_grounding_even_when_the_provider_never_sends_a_finish_reason() {
|
||||
let context = report_context("gemini:generate_content", "openai:chat");
|
||||
let mut matrix = StreamingStandardFormatMatrix::default();
|
||||
let mut output = matrix
|
||||
.transform_line(
|
||||
&context,
|
||||
data_line(json!({
|
||||
"responseId": "resp_grounded",
|
||||
"modelVersion": "gemini-3.8-flash",
|
||||
"candidates": [{
|
||||
"index": 0,
|
||||
"content": {"role": "model", "parts": [{"text": "grounded"}]},
|
||||
"groundingMetadata": {
|
||||
"groundingChunks": [{"web": {"uri": "https://time.gov/"}}]
|
||||
}
|
||||
}]
|
||||
})),
|
||||
)
|
||||
.expect("grounded chunk");
|
||||
output.extend(matrix.finish(&context).expect("finish"));
|
||||
let sse = String::from_utf8(output).expect("valid SSE");
|
||||
|
||||
assert!(
|
||||
sse.contains("url_citation") && sse.contains("https://time.gov/"),
|
||||
"{sse}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn terminal_observer_marks_malformed_gemini_function_call_as_failure() {
|
||||
let context = report_context("gemini:generate_content", "openai:responses");
|
||||
|
||||
@@ -3662,6 +3662,11 @@ fn try_aggregate_gemini_stream_sync_response(
|
||||
CanonicalStreamEvent::TextDelta(text) => {
|
||||
append_gemini_text_part(&mut parts, text, false);
|
||||
}
|
||||
// This rebuilds a raw Gemini body, and every non-`content`
|
||||
// candidate key — `groundingMetadata` included — is already
|
||||
// copied across above. Projecting it into citations is the
|
||||
// job of whoever converts that body onward.
|
||||
CanonicalStreamEvent::Citations(_) => {}
|
||||
CanonicalStreamEvent::ReasoningDelta(text) => {
|
||||
append_gemini_text_part(&mut parts, text, true);
|
||||
}
|
||||
|
||||
@@ -16,6 +16,7 @@ pub use crate::protocol::stream::{CanonicalStreamEvent, CanonicalStreamFrame};
|
||||
|
||||
pub(crate) const OPENAI_RESPONSES_EXTENSION_NAMESPACE: &str = "openai_responses";
|
||||
pub(crate) const OPENAI_RESPONSES_LEGACY_EXTENSION_NAMESPACE: &str = "openai_cli";
|
||||
pub(crate) const CLAUDE_EXTENSION_NAMESPACE: &str = "claude";
|
||||
const AETHER_EXTENSION_NAMESPACE: &str = "aether";
|
||||
const CLAUDE_MESSAGES_REQUEST_SOURCE_MARKER: &str = "claude_messages_request";
|
||||
const CLAUDE_SYSTEM_SOURCE_MARKER: &str = "claude_system";
|
||||
@@ -5879,7 +5880,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
let mut out = Map::new();
|
||||
out.insert("type".to_string(), Value::String("text".to_string()));
|
||||
out.insert("text".to_string(), Value::String(text.clone()));
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::Thinking {
|
||||
@@ -5899,7 +5904,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
Value::String("redacted_thinking".to_string()),
|
||||
);
|
||||
out.insert("data".to_string(), Value::String(data.clone()));
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
return Some(Some(Value::Object(out)));
|
||||
}
|
||||
if !matches!(role, CanonicalRole::Assistant) {
|
||||
@@ -5920,7 +5929,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
if let Some(signature) = signature.as_ref().filter(|value| !value.is_empty()) {
|
||||
out.insert("signature".to_string(), Value::String(signature.clone()));
|
||||
}
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::Image {
|
||||
@@ -5945,7 +5958,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
"source".to_string(),
|
||||
claude_source_value(media_type.as_deref(), data.as_deref(), url.as_deref())?,
|
||||
);
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::File {
|
||||
@@ -5968,7 +5985,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
"source".to_string(),
|
||||
claude_source_value(media_type.as_deref(), data.as_deref(), file_url.as_deref())?,
|
||||
);
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::Audio {
|
||||
@@ -5988,7 +6009,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
None,
|
||||
)?,
|
||||
);
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::ToolUse {
|
||||
@@ -6006,7 +6031,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
);
|
||||
out.insert("name".to_string(), Value::String(name.clone()));
|
||||
out.insert("input".to_string(), input);
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::ToolResult {
|
||||
@@ -6035,7 +6064,11 @@ pub(crate) fn canonical_block_to_claude(
|
||||
if *is_error {
|
||||
out.insert("is_error".to_string(), Value::Bool(true));
|
||||
}
|
||||
out.extend(namespace_extension_object(extensions, "claude", &out));
|
||||
out.extend(namespace_extension_object(
|
||||
extensions,
|
||||
CLAUDE_EXTENSION_NAMESPACE,
|
||||
&out,
|
||||
));
|
||||
Some(Some(Value::Object(out)))
|
||||
}
|
||||
CanonicalContentBlock::Unknown {
|
||||
|
||||
@@ -74,6 +74,14 @@ pub enum CanonicalStreamEvent {
|
||||
name: Option<String>,
|
||||
content: String,
|
||||
},
|
||||
/// Provider-neutral source citations for the answer text streamed so far.
|
||||
///
|
||||
/// Emitted once, just before `Finish`, by providers that ground an answer
|
||||
/// server-side and report the evidence as metadata instead of a tool call.
|
||||
/// Each entry carries `url` plus optional `title`, `cited_text` and
|
||||
/// `start_index`/`end_index` character offsets; every target renders them
|
||||
/// into its own family's citation shape.
|
||||
Citations(Vec<Value>),
|
||||
UnknownEvent(Value),
|
||||
Finish {
|
||||
finish_reason: Option<String>,
|
||||
|
||||
Reference in New Issue
Block a user