fix(chat): an edit after a broken edit call no longer fails, found with Opus 5.5

- Claude Opus 5.5 sent an edit with invalid JSON, then the same edit
  again. The first call's streamed preview was never undone: its input
  has no operations, and the undo sat behind that check. The second edit
  then started from the preview, failed on a duplicate id, and the model
  had to try a third time. The undo now runs first, and an edit that
  starts in the same render uses the undone diagram.
- The SDK passes an invalid tool call's error as a string, which was
  wrapped as a provider error. streamErrorText keeps it as the text the
  model reads.
- Bedrock's "on-demand throughput isn't supported" gets the model id hint.
- The thinking header uses the page language ("Thought for 1 second" in
  English), from the dictionary entries that were already there.
This commit is contained in:
dayuan.jiang
2026-10-04 21:18:54 +09:00
parent 65e3dd1dde
commit 2ac4c54eb2
9 changed files with 271 additions and 18 deletions
+2 -7
View File
@@ -38,7 +38,7 @@ import {
setTraceOutput,
wrapWithObserve,
} from "@/lib/langfuse"
import { classifyLLMError, isToolCallError } from "@/lib/llm-errors"
import { classifyLLMError, streamErrorText } from "@/lib/llm-errors"
import {
resolveMaxOutputTokens,
withOutputTokenLimitFallback,
@@ -724,12 +724,7 @@ Call this tool to get shape names and usage syntax for a specific library.`,
const response = result.toUIMessageStreamResponse({
sendReasoning: true,
// The same text goes back to the model when its tool call was
// invalid, so it can fix it: keep that one as it is
onError: (error) =>
isToolCallError(error)
? (error as Error).message
: JSON.stringify(classifyLLMError(error)),
onError: streamErrorText,
messageMetadata: ({ part }) => {
if (part.type === "finish") {
const usage = (part as any).totalUsage
+36 -9
View File
@@ -26,6 +26,7 @@ import {
ReasoningContent,
ReasoningTrigger,
} from "@/components/ai-elements/reasoning"
import { Shimmer } from "@/components/ai-elements/shimmer"
import { ChatLobby } from "@/components/chat/ChatLobby"
import { TemplateCreateDialog } from "@/components/chat/TemplateCreateDialog"
import { ToolCallCard } from "@/components/chat/ToolCallCard"
@@ -193,6 +194,23 @@ export function ChatMessageDisplay({
currentInput = "",
}: ChatMessageDisplayProps) {
const dict = useDictionary()
// The thinking header in the page language
const thinkingMessage = (isStreaming: boolean, duration?: number) => {
if (isStreaming || duration === 0) {
return <Shimmer duration={1}>{dict.reasoning.thinking}</Shimmer>
}
if (duration === undefined) return <p>{dict.reasoning.thoughtBrief}</p>
return (
<p>
{duration === 1
? dict.reasoning.thoughtForOne
: dict.reasoning.thoughtFor.replace(
"{duration}",
String(duration),
)}
</p>
)
}
const { chartXML, loadDiagram: onDisplayChart } = useDiagram()
const messagesEndRef = useRef<HTMLDivElement>(null)
const scrollTopRef = useRef<HTMLDivElement>(null)
@@ -404,6 +422,10 @@ export function ChatMessageDisplay({
// Previous messages are already processed and won't change
const messagesToProcess =
messages.length > 0 ? [messages[messages.length - 1]] : []
// The diagram without streamed previews. Undoing a failed edit's
// preview below changes it before chartXML catches up, and an edit
// streaming right after must start from the undone diagram.
let baseXml = chartXML
messagesToProcess.forEach((message) => {
// Messages restored from a saved session were applied before it was
@@ -460,13 +482,12 @@ export function ChatMessageDisplay({
// Handle edit_diagram streaming - apply operations incrementally for preview
// Uses shared editDiagramOriginalXmlRef to coordinate with tool handler
if (
part.type === "tool-edit_diagram" &&
input?.operations
) {
if (part.type === "tool-edit_diagram") {
// Failed or stopped: if the original XML is still
// stored, the tool handler never ran (user pressed
// stop), so undo the streamed preview here.
// stored, the tool handler never ran (invalid
// JSON, or the user pressed stop), so undo the
// streamed preview here. Invalid JSON leaves no
// operations in the input, so check this first.
if (state === "output-error") {
const originalXml =
editDiagramOriginalXmlRef.current.get(
@@ -477,9 +498,11 @@ export function ChatMessageDisplay({
toolCallId,
)
onDisplayChart(originalXml, true)
baseXml = originalXml
}
return
}
if (!input?.operations) return
if (state !== "input-streaming") {
// Input complete: the tool handler applies the
@@ -506,7 +529,7 @@ export function ChatMessageDisplay({
toolCallId,
)
) {
if (!chartXML) {
if (!baseXml) {
console.warn(
"[edit_diagram streaming] No chart XML available",
)
@@ -514,7 +537,7 @@ export function ChatMessageDisplay({
}
editDiagramOriginalXmlRef.current.set(
toolCallId,
chartXML,
baseXml,
)
}
const originalXml =
@@ -739,7 +762,11 @@ export function ChatMessageDisplay({
!isRestoredMessage
}
>
<ReasoningTrigger />
<ReasoningTrigger
getThinkingMessage={
thinkingMessage
}
/>
<ReasoningContent>
{
reasoningPart.text
+1
View File
@@ -248,6 +248,7 @@
"reasoning": {
"thinking": "Thinking...",
"thoughtFor": "Thought for {duration} seconds",
"thoughtForOne": "Thought for 1 second",
"thoughtBrief": "Thought for a few seconds"
},
"dev": {
+1
View File
@@ -248,6 +248,7 @@
"reasoning": {
"thinking": "考え中...",
"thoughtFor": "{duration} 秒考えました",
"thoughtForOne": "1 秒考えました",
"thoughtBrief": "数秒考えました"
},
"dev": {
+1
View File
@@ -248,6 +248,7 @@
"reasoning": {
"thinking": "思考中...",
"thoughtFor": "思考了 {duration} 秒",
"thoughtForOne": "思考了 1 秒",
"thoughtBrief": "思考了幾秒鐘"
},
"dev": {
+1
View File
@@ -248,6 +248,7 @@
"reasoning": {
"thinking": "思考中...",
"thoughtFor": "思考了 {duration} 秒",
"thoughtForOne": "思考了 1 秒",
"thoughtBrief": "思考了几秒钟"
},
"dev": {
+13
View File
@@ -49,6 +49,8 @@ const SPECIFIC_TEXTS: Array<[RegExp, LLMErrorCode]> = [
],
// Bedrock, when the output limit cut the tool call's JSON short
[/toolUse\.input is invalid/i, "output_truncated"],
// Bedrock, for a model id without the inference profile prefix
[/on-demand throughput isn.t supported/i, "model_not_found"],
[
/insufficient[_ ]quota|insufficient balance|exceeded your current quota|credit balance is too low|余额不足/i,
"insufficient_quota",
@@ -105,6 +107,17 @@ function problemDetail(body: string): string | undefined {
}
}
/**
* The error text for the chat stream: what went wrong with the provider as
* JSON for the hint, or the text the model must read to fix a tool call.
*/
export function streamErrorText(error: unknown): string {
// The SDK passes an invalid tool call's error as a plain string
if (typeof error === "string") return error
if (isToolCallError(error)) return (error as Error).message
return JSON.stringify(classifyLLMError(error))
}
/**
* Model and tool errors the SDK sends back to the model as the tool result,
* so it can fix its call. Their text has to stay as it is.
+135
View File
@@ -137,6 +137,34 @@ test("edit_diagram applies all operations or none", async ({ page: p }) => {
await expect(canvas.getByText("Broken", { exact: true })).toHaveCount(0)
})
test("the thinking header is in the page language", async ({ page: p }) => {
const events = [
{ type: "start" },
{ type: "reasoning-start", id: "r1" },
{ type: "reasoning-delta", id: "r1", delta: "Plan the boxes" },
{ type: "reasoning-end", id: "r1" },
{ type: "text-start", id: "t1" },
{ type: "text-delta", id: "t1", delta: "Done" },
{ type: "text-end", id: "t1" },
{ type: "finish" },
]
await p.route("**/api/chat", (route) =>
route.fulfill({
status: 200,
contentType: "text/event-stream",
body: `${events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")}data: [DONE]\n\n`,
}),
)
await p.goto("/zh", { waitUntil: "networkidle" })
await getIframe(p).waitFor({ state: "visible", timeout: 30000 })
await sendMessage(p, "画两个框")
await expect(p.getByText("Plan the boxes")).toBeAttached({
timeout: 15000,
})
await expect(p.getByText(/^思考/)).toBeVisible()
await expect(p.getByText(/^Thought for|^Thinking/)).toHaveCount(0)
})
test("blank text before a tool call shows no empty bubble", async ({
page: p,
}) => {
@@ -163,3 +191,110 @@ test("blank text before a tool call shows no empty bubble", async ({
// Assistant text bubbles have this background
await expect(p.locator("div.rounded-2xl.bg-muted\\/60")).toHaveCount(0)
})
test("an edit right after a broken edit call starts from the real diagram", async ({
page: p,
}) => {
// Seen with Claude Opus 5.5: the first edit call had invalid JSON, the
// server rejected it, and the model sent the same edit again at once.
// The second edit must not see the first one's streamed preview.
const sse = (events: object[]) =>
events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")
const edit = {
operations: [
{
operation: "add",
cell_id: "c",
new_xml: cell("c", "Gamma", 400),
},
],
}
const deltas = (id: string) =>
(JSON.stringify(edit).match(/[\s\S]{1,40}/g) ?? []).map((d) => ({
type: "tool-input-delta",
toolCallId: id,
inputTextDelta: d,
}))
const start = (id: string) => ({
type: "tool-input-start",
toolCallId: id,
toolName: "edit_diagram",
})
// Each inner array is sent as one network chunk, 300 ms apart, so the
// throttled UI renders between chunks like with a real model
const replies = [
[streamedToolCall("display_diagram", { xml: cell("a", "Alpha", 40) })],
[
sse([
{ type: "start" },
{ type: "start-step" },
start("e1"),
...deltas("e1"),
]),
sse([
{
type: "tool-input-error",
toolCallId: "e1",
toolName: "edit_diagram",
input: "{broken",
errorText: "JSON parsing failed",
},
{
type: "tool-output-error",
toolCallId: "e1",
errorText: "JSON parsing failed",
},
{ type: "finish-step" },
{ type: "start-step" },
start("e2"),
...deltas("e2"),
]),
`${sse([
{
type: "tool-input-available",
toolCallId: "e2",
toolName: "edit_diagram",
input: edit,
},
{ type: "finish-step" },
{ type: "finish" },
])}data: [DONE]\n\n`,
],
]
await p.addInitScript((replies) => {
const realFetch = window.fetch
let n = 0
window.fetch = async (input, init) => {
const url =
typeof input === "string" ? input : (input as Request).url
if (!url.endsWith("/api/chat")) return realFetch(input, init)
const chunks = replies[n++] ?? [
'data: {"type":"start"}\n\ndata: {"type":"finish"}\n\ndata: [DONE]\n\n',
]
const body = new ReadableStream({
async start(controller) {
for (const chunk of chunks) {
controller.enqueue(new TextEncoder().encode(chunk))
await new Promise((r) => setTimeout(r, 300))
}
controller.close()
},
})
return new Response(body, {
headers: { "content-type": "text/event-stream" },
})
}
}, replies)
await p.goto("/", { waitUntil: "networkidle" })
await getIframe(p).waitFor({ state: "visible", timeout: 30000 })
const canvas = p.frameLocator("iframe")
await sendMessage(p, "Draw a box")
await waitForCompleteCount(p, 1)
await sendMessage(p, "Add another box")
await waitForCompleteCount(p, 2)
await expect(canvas.getByText("Gamma", { exact: true })).toBeVisible({
timeout: 15000,
})
await expect(p.getByText(/No changes were made/)).toHaveCount(0)
})
+81 -2
View File
@@ -1,7 +1,20 @@
// @vitest-environment node
import { APICallError, InvalidToolInputError, RetryError } from "ai"
import {
APICallError,
InvalidToolInputError,
RetryError,
simulateReadableStream,
streamText,
tool,
} from "ai"
import { MockLanguageModelV3 } from "ai/test"
import { describe, expect, it } from "vitest"
import { classifyLLMError, isToolCallError } from "@/lib/llm-errors"
import { z } from "zod"
import {
classifyLLMError,
isToolCallError,
streamErrorText,
} from "@/lib/llm-errors"
const apiError = (statusCode: number, message: string, responseBody = "") =>
new APICallError({
@@ -89,6 +102,14 @@ describe("classifyLLMError", () => {
expect(classifyLLMError(timeout).code).toBe("timeout")
})
it("points to the model id when Bedrock wants an inference profile", () => {
const error = apiError(
400,
"Invocation of model ID anthropic.claude-sonnet-5-5 with on-demand throughput isn’t supported. Retry your request with the ID or ARN of an inference profile that contains this model.",
)
expect(classifyLLMError(error).code).toBe("model_not_found")
})
it("names a network error the SDK wrapped", () => {
const error = new APICallError({
message:
@@ -131,6 +152,64 @@ describe("classifyLLMError", () => {
})
})
describe("streamErrorText", () => {
it("keeps the text of a tool call the model got wrong", async () => {
// Seen with Claude Opus 5.5: a quote left unescaped in the input
const model = new MockLanguageModelV3({
doStream: (async () => ({
stream: simulateReadableStream({
chunks: [
{
type: "tool-call",
toolCallId: "c1",
toolName: "edit_diagram",
input: '{"operations": [{"new_xml": "as="x""}]}',
},
{
type: "finish",
finishReason: {
unified: "tool-calls",
raw: "tool_use",
},
usage: {
inputTokens: { total: 1 },
outputTokens: { total: 1 },
},
},
],
}),
})) as any,
})
const result = streamText({
model: model as any,
prompt: "edit",
tools: {
edit_diagram: tool({
inputSchema: z.object({ operations: z.array(z.any()) }),
}),
},
})
const errors: string[] = []
for await (const chunk of result.toUIMessageStream({
onError: streamErrorText,
})) {
if ("errorText" in chunk) errors.push(chunk.errorText)
}
expect(errors.length).toBeGreaterThan(0)
for (const text of errors) {
expect(text).toMatch(/^Invalid input for tool edit_diagram/)
}
})
it("classifies a provider error", () => {
expect(JSON.parse(streamErrorText(apiError(401, "bad key")))).toEqual({
type: "provider",
code: "invalid_api_key",
message: "bad key",
})
})
})
describe("isToolCallError", () => {
it("spots errors the model must see unchanged", () => {
const invalid = new InvalidToolInputError({