fix(chat): an edit after a broken edit call no longer fails, found with Opus 5.5

- Claude Opus 5.5 sent an edit with invalid JSON, then the same edit
  again. The first call's streamed preview was never undone: its input
  has no operations, and the undo sat behind that check. The second edit
  then started from the preview, failed on a duplicate id, and the model
  had to try a third time. The undo now runs first, and an edit that
  starts in the same render uses the undone diagram.
- The SDK passes an invalid tool call's error as a string, which was
  wrapped as a provider error. streamErrorText keeps it as the text the
  model reads.
- Bedrock's "on-demand throughput isn't supported" gets the model id hint.
- The thinking header uses the page language ("Thought for 1 second" in
  English), from the dictionary entries that were already there.
This commit is contained in:
dayuan.jiang
2026-10-04 21:18:54 +09:00
parent 65e3dd1dde
commit 2ac4c54eb2
9 changed files with 271 additions and 18 deletions
+135
View File
@@ -137,6 +137,34 @@ test("edit_diagram applies all operations or none", async ({ page: p }) => {
await expect(canvas.getByText("Broken", { exact: true })).toHaveCount(0)
})
test("the thinking header is in the page language", async ({ page: p }) => {
const events = [
{ type: "start" },
{ type: "reasoning-start", id: "r1" },
{ type: "reasoning-delta", id: "r1", delta: "Plan the boxes" },
{ type: "reasoning-end", id: "r1" },
{ type: "text-start", id: "t1" },
{ type: "text-delta", id: "t1", delta: "Done" },
{ type: "text-end", id: "t1" },
{ type: "finish" },
]
await p.route("**/api/chat", (route) =>
route.fulfill({
status: 200,
contentType: "text/event-stream",
body: `${events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")}data: [DONE]\n\n`,
}),
)
await p.goto("/zh", { waitUntil: "networkidle" })
await getIframe(p).waitFor({ state: "visible", timeout: 30000 })
await sendMessage(p, "画两个框")
await expect(p.getByText("Plan the boxes")).toBeAttached({
timeout: 15000,
})
await expect(p.getByText(/^思考/)).toBeVisible()
await expect(p.getByText(/^Thought for|^Thinking/)).toHaveCount(0)
})
test("blank text before a tool call shows no empty bubble", async ({
page: p,
}) => {
@@ -163,3 +191,110 @@ test("blank text before a tool call shows no empty bubble", async ({
// Assistant text bubbles have this background
await expect(p.locator("div.rounded-2xl.bg-muted\\/60")).toHaveCount(0)
})
test("an edit right after a broken edit call starts from the real diagram", async ({
page: p,
}) => {
// Seen with Claude Opus 5.5: the first edit call had invalid JSON, the
// server rejected it, and the model sent the same edit again at once.
// The second edit must not see the first one's streamed preview.
const sse = (events: object[]) =>
events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")
const edit = {
operations: [
{
operation: "add",
cell_id: "c",
new_xml: cell("c", "Gamma", 400),
},
],
}
const deltas = (id: string) =>
(JSON.stringify(edit).match(/[\s\S]{1,40}/g) ?? []).map((d) => ({
type: "tool-input-delta",
toolCallId: id,
inputTextDelta: d,
}))
const start = (id: string) => ({
type: "tool-input-start",
toolCallId: id,
toolName: "edit_diagram",
})
// Each inner array is sent as one network chunk, 300 ms apart, so the
// throttled UI renders between chunks like with a real model
const replies = [
[streamedToolCall("display_diagram", { xml: cell("a", "Alpha", 40) })],
[
sse([
{ type: "start" },
{ type: "start-step" },
start("e1"),
...deltas("e1"),
]),
sse([
{
type: "tool-input-error",
toolCallId: "e1",
toolName: "edit_diagram",
input: "{broken",
errorText: "JSON parsing failed",
},
{
type: "tool-output-error",
toolCallId: "e1",
errorText: "JSON parsing failed",
},
{ type: "finish-step" },
{ type: "start-step" },
start("e2"),
...deltas("e2"),
]),
`${sse([
{
type: "tool-input-available",
toolCallId: "e2",
toolName: "edit_diagram",
input: edit,
},
{ type: "finish-step" },
{ type: "finish" },
])}data: [DONE]\n\n`,
],
]
await p.addInitScript((replies) => {
const realFetch = window.fetch
let n = 0
window.fetch = async (input, init) => {
const url =
typeof input === "string" ? input : (input as Request).url
if (!url.endsWith("/api/chat")) return realFetch(input, init)
const chunks = replies[n++] ?? [
'data: {"type":"start"}\n\ndata: {"type":"finish"}\n\ndata: [DONE]\n\n',
]
const body = new ReadableStream({
async start(controller) {
for (const chunk of chunks) {
controller.enqueue(new TextEncoder().encode(chunk))
await new Promise((r) => setTimeout(r, 300))
}
controller.close()
},
})
return new Response(body, {
headers: { "content-type": "text/event-stream" },
})
}
}, replies)
await p.goto("/", { waitUntil: "networkidle" })
await getIframe(p).waitFor({ state: "visible", timeout: 30000 })
const canvas = p.frameLocator("iframe")
await sendMessage(p, "Draw a box")
await waitForCompleteCount(p, 1)
await sendMessage(p, "Add another box")
await waitForCompleteCount(p, 2)
await expect(canvas.getByText("Gamma", { exact: true })).toBeVisible({
timeout: 15000,
})
await expect(p.getByText(/No changes were made/)).toHaveCount(0)
})
+81 -2
View File
@@ -1,7 +1,20 @@
// @vitest-environment node
import { APICallError, InvalidToolInputError, RetryError } from "ai"
import {
APICallError,
InvalidToolInputError,
RetryError,
simulateReadableStream,
streamText,
tool,
} from "ai"
import { MockLanguageModelV3 } from "ai/test"
import { describe, expect, it } from "vitest"
import { classifyLLMError, isToolCallError } from "@/lib/llm-errors"
import { z } from "zod"
import {
classifyLLMError,
isToolCallError,
streamErrorText,
} from "@/lib/llm-errors"
const apiError = (statusCode: number, message: string, responseBody = "") =>
new APICallError({
@@ -89,6 +102,14 @@ describe("classifyLLMError", () => {
expect(classifyLLMError(timeout).code).toBe("timeout")
})
it("points to the model id when Bedrock wants an inference profile", () => {
const error = apiError(
400,
"Invocation of model ID anthropic.claude-sonnet-5-5 with on-demand throughput isn’t supported. Retry your request with the ID or ARN of an inference profile that contains this model.",
)
expect(classifyLLMError(error).code).toBe("model_not_found")
})
it("names a network error the SDK wrapped", () => {
const error = new APICallError({
message:
@@ -131,6 +152,64 @@ describe("classifyLLMError", () => {
})
})
describe("streamErrorText", () => {
it("keeps the text of a tool call the model got wrong", async () => {
// Seen with Claude Opus 5.5: a quote left unescaped in the input
const model = new MockLanguageModelV3({
doStream: (async () => ({
stream: simulateReadableStream({
chunks: [
{
type: "tool-call",
toolCallId: "c1",
toolName: "edit_diagram",
input: '{"operations": [{"new_xml": "as="x""}]}',
},
{
type: "finish",
finishReason: {
unified: "tool-calls",
raw: "tool_use",
},
usage: {
inputTokens: { total: 1 },
outputTokens: { total: 1 },
},
},
],
}),
})) as any,
})
const result = streamText({
model: model as any,
prompt: "edit",
tools: {
edit_diagram: tool({
inputSchema: z.object({ operations: z.array(z.any()) }),
}),
},
})
const errors: string[] = []
for await (const chunk of result.toUIMessageStream({
onError: streamErrorText,
})) {
if ("errorText" in chunk) errors.push(chunk.errorText)
}
expect(errors.length).toBeGreaterThan(0)
for (const text of errors) {
expect(text).toMatch(/^Invalid input for tool edit_diagram/)
}
})
it("classifies a provider error", () => {
expect(JSON.parse(streamErrorText(apiError(401, "bad key")))).toEqual({
type: "provider",
code: "invalid_api_key",
message: "bad key",
})
})
})
describe("isToolCallError", () => {
it("spots errors the model must see unchanged", () => {
const invalid = new InvalidToolInputError({