mirror of
https://github.com/fawney19/Aether.git
synced 2026-10-04 00:17:45 +08:00
fix(responses): keep raw reasoning on content only
Raw chain-of-thought was written to both `content` (`reasoning_text`) and `summary` (`summary_text`), and the stream emitter sent the same delta on `response.reasoning_text.delta` *and* `response.reasoning_summary_text.delta`. Clients that render both channels therefore printed every thinking chunk twice — most visibly the Codex CLI, whose thinking panel repeated itself. OpenAI keeps the two channels distinct: `content` carries the raw CoT while `summary` is the summarised view. Emit the thinking on `content` only: - `openai_responses_reasoning_text_fields` becomes `openai_responses_reasoning_text_parts`, returning just the `content` array; reasoning items keep `summary: []` (or a provider-supplied summary). - The Responses stream emitter emits `response.reasoning_text.delta` / `.done` and no longer mirrors them onto the summary events. The reasoning `output_item.added` no longer announces a `reasoning_summary_part`. - The provider-state reasoning reader accepts `content` (`reasoning_text`) first and falls back to `summary`, so it also understands items produced by older Aether versions; its state field is renamed accordingly. - Non-streaming builders (Chat -> Responses, manual Responses response, Grok gateway) place the thinking on `content` and leave `summary` empty. Tests cover the raw thinking appearing exactly once in the emitted stream.
This commit is contained in:
@@ -3203,10 +3203,7 @@ fn openai_responses_body(
|
||||
"id": openai_responses_synthetic_reasoning_item_id(&response_id, 0),
|
||||
"type": "reasoning",
|
||||
"status": "completed",
|
||||
"summary": [{
|
||||
"type": "summary_text",
|
||||
"text": thinking,
|
||||
}],
|
||||
"summary": [],
|
||||
"content": [{
|
||||
"type": "reasoning_text",
|
||||
"text": thinking,
|
||||
@@ -4640,10 +4637,7 @@ mod tests {
|
||||
body["output"][0]["content"][0]["text"],
|
||||
serde_json::json!("short reasoning")
|
||||
);
|
||||
assert_eq!(
|
||||
body["output"][0]["summary"][0]["text"],
|
||||
serde_json::json!("short reasoning")
|
||||
);
|
||||
assert_eq!(body["output"][0]["summary"], serde_json::json!([]));
|
||||
assert_eq!(body["output"][1]["type"], serde_json::json!("message"));
|
||||
assert!(body["output"][1]["id"]
|
||||
.as_str()
|
||||
@@ -4827,7 +4821,12 @@ mod tests {
|
||||
|
||||
assert!(body.contains("event: response.created"));
|
||||
assert!(body.contains("event: response.in_progress"));
|
||||
assert!(body.contains("event: response.reasoning_summary_part.added"));
|
||||
// Thinking must stay off the summary channel or clients that render
|
||||
// both (Codex) print the raw chain-of-thought twice.
|
||||
assert!(!body.contains("event: response.reasoning_summary_part.added"));
|
||||
assert!(!body.contains("event: response.reasoning_summary_text.delta"));
|
||||
assert!(!body.contains("event: response.reasoning_summary_text.done"));
|
||||
assert!(body.contains("\"type\":\"reasoning_text\""));
|
||||
assert!(body.contains("event: response.content_part.added"));
|
||||
assert!(body.contains("event: response.output_text.done"));
|
||||
assert!(body.contains("event: response.completed"));
|
||||
|
||||
@@ -11880,7 +11880,11 @@ mod tests {
|
||||
.expect("response body should read");
|
||||
let body = String::from_utf8(body.to_vec()).expect("response body should be utf8");
|
||||
assert!(
|
||||
body.contains("event: response.reasoning_summary_text.delta\n"),
|
||||
body.contains("event: response.reasoning_text.delta\n"),
|
||||
"{body}"
|
||||
);
|
||||
assert!(
|
||||
!body.contains("event: response.reasoning_summary_text.delta\n"),
|
||||
"{body}"
|
||||
);
|
||||
assert!(
|
||||
|
||||
Reference in New Issue
Block a user