mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-10-08 18:57:47 +08:00
fix(chat): an edit after a broken edit call no longer fails, found with Opus 5.5
- Claude Opus 5.5 sent an edit with invalid JSON, then the same edit
again. The first call's streamed preview was never undone: its input
has no operations, and the undo sat behind that check. The second edit
then started from the preview, failed on a duplicate id, and the model
had to try a third time. The undo now runs first, and an edit that
starts in the same render uses the undone diagram.
- The SDK passes an invalid tool call's error as a string, which was
wrapped as a provider error. streamErrorText keeps it as the text the
model reads.
- Bedrock's "on-demand throughput isn't supported" gets the model id hint.
- The thinking header uses the page language ("Thought for 1 second" in
English), from the dictionary entries that were already there.
This commit is contained in:
@@ -38,7 +38,7 @@ import {
|
||||
setTraceOutput,
|
||||
wrapWithObserve,
|
||||
} from "@/lib/langfuse"
|
||||
import { classifyLLMError, isToolCallError } from "@/lib/llm-errors"
|
||||
import { classifyLLMError, streamErrorText } from "@/lib/llm-errors"
|
||||
import {
|
||||
resolveMaxOutputTokens,
|
||||
withOutputTokenLimitFallback,
|
||||
@@ -724,12 +724,7 @@ Call this tool to get shape names and usage syntax for a specific library.`,
|
||||
|
||||
const response = result.toUIMessageStreamResponse({
|
||||
sendReasoning: true,
|
||||
// The same text goes back to the model when its tool call was
|
||||
// invalid, so it can fix it: keep that one as it is
|
||||
onError: (error) =>
|
||||
isToolCallError(error)
|
||||
? (error as Error).message
|
||||
: JSON.stringify(classifyLLMError(error)),
|
||||
onError: streamErrorText,
|
||||
messageMetadata: ({ part }) => {
|
||||
if (part.type === "finish") {
|
||||
const usage = (part as any).totalUsage
|
||||
|
||||
@@ -26,6 +26,7 @@ import {
|
||||
ReasoningContent,
|
||||
ReasoningTrigger,
|
||||
} from "@/components/ai-elements/reasoning"
|
||||
import { Shimmer } from "@/components/ai-elements/shimmer"
|
||||
import { ChatLobby } from "@/components/chat/ChatLobby"
|
||||
import { TemplateCreateDialog } from "@/components/chat/TemplateCreateDialog"
|
||||
import { ToolCallCard } from "@/components/chat/ToolCallCard"
|
||||
@@ -193,6 +194,23 @@ export function ChatMessageDisplay({
|
||||
currentInput = "",
|
||||
}: ChatMessageDisplayProps) {
|
||||
const dict = useDictionary()
|
||||
// The thinking header in the page language
|
||||
const thinkingMessage = (isStreaming: boolean, duration?: number) => {
|
||||
if (isStreaming || duration === 0) {
|
||||
return <Shimmer duration={1}>{dict.reasoning.thinking}</Shimmer>
|
||||
}
|
||||
if (duration === undefined) return <p>{dict.reasoning.thoughtBrief}</p>
|
||||
return (
|
||||
<p>
|
||||
{duration === 1
|
||||
? dict.reasoning.thoughtForOne
|
||||
: dict.reasoning.thoughtFor.replace(
|
||||
"{duration}",
|
||||
String(duration),
|
||||
)}
|
||||
</p>
|
||||
)
|
||||
}
|
||||
const { chartXML, loadDiagram: onDisplayChart } = useDiagram()
|
||||
const messagesEndRef = useRef<HTMLDivElement>(null)
|
||||
const scrollTopRef = useRef<HTMLDivElement>(null)
|
||||
@@ -404,6 +422,10 @@ export function ChatMessageDisplay({
|
||||
// Previous messages are already processed and won't change
|
||||
const messagesToProcess =
|
||||
messages.length > 0 ? [messages[messages.length - 1]] : []
|
||||
// The diagram without streamed previews. Undoing a failed edit's
|
||||
// preview below changes it before chartXML catches up, and an edit
|
||||
// streaming right after must start from the undone diagram.
|
||||
let baseXml = chartXML
|
||||
|
||||
messagesToProcess.forEach((message) => {
|
||||
// Messages restored from a saved session were applied before it was
|
||||
@@ -460,13 +482,12 @@ export function ChatMessageDisplay({
|
||||
|
||||
// Handle edit_diagram streaming - apply operations incrementally for preview
|
||||
// Uses shared editDiagramOriginalXmlRef to coordinate with tool handler
|
||||
if (
|
||||
part.type === "tool-edit_diagram" &&
|
||||
input?.operations
|
||||
) {
|
||||
if (part.type === "tool-edit_diagram") {
|
||||
// Failed or stopped: if the original XML is still
|
||||
// stored, the tool handler never ran (user pressed
|
||||
// stop), so undo the streamed preview here.
|
||||
// stored, the tool handler never ran (invalid
|
||||
// JSON, or the user pressed stop), so undo the
|
||||
// streamed preview here. Invalid JSON leaves no
|
||||
// operations in the input, so check this first.
|
||||
if (state === "output-error") {
|
||||
const originalXml =
|
||||
editDiagramOriginalXmlRef.current.get(
|
||||
@@ -477,9 +498,11 @@ export function ChatMessageDisplay({
|
||||
toolCallId,
|
||||
)
|
||||
onDisplayChart(originalXml, true)
|
||||
baseXml = originalXml
|
||||
}
|
||||
return
|
||||
}
|
||||
if (!input?.operations) return
|
||||
|
||||
if (state !== "input-streaming") {
|
||||
// Input complete: the tool handler applies the
|
||||
@@ -506,7 +529,7 @@ export function ChatMessageDisplay({
|
||||
toolCallId,
|
||||
)
|
||||
) {
|
||||
if (!chartXML) {
|
||||
if (!baseXml) {
|
||||
console.warn(
|
||||
"[edit_diagram streaming] No chart XML available",
|
||||
)
|
||||
@@ -514,7 +537,7 @@ export function ChatMessageDisplay({
|
||||
}
|
||||
editDiagramOriginalXmlRef.current.set(
|
||||
toolCallId,
|
||||
chartXML,
|
||||
baseXml,
|
||||
)
|
||||
}
|
||||
const originalXml =
|
||||
@@ -739,7 +762,11 @@ export function ChatMessageDisplay({
|
||||
!isRestoredMessage
|
||||
}
|
||||
>
|
||||
<ReasoningTrigger />
|
||||
<ReasoningTrigger
|
||||
getThinkingMessage={
|
||||
thinkingMessage
|
||||
}
|
||||
/>
|
||||
<ReasoningContent>
|
||||
{
|
||||
reasoningPart.text
|
||||
|
||||
@@ -248,6 +248,7 @@
|
||||
"reasoning": {
|
||||
"thinking": "Thinking...",
|
||||
"thoughtFor": "Thought for {duration} seconds",
|
||||
"thoughtForOne": "Thought for 1 second",
|
||||
"thoughtBrief": "Thought for a few seconds"
|
||||
},
|
||||
"dev": {
|
||||
|
||||
@@ -248,6 +248,7 @@
|
||||
"reasoning": {
|
||||
"thinking": "考え中...",
|
||||
"thoughtFor": "{duration} 秒考えました",
|
||||
"thoughtForOne": "1 秒考えました",
|
||||
"thoughtBrief": "数秒考えました"
|
||||
},
|
||||
"dev": {
|
||||
|
||||
@@ -248,6 +248,7 @@
|
||||
"reasoning": {
|
||||
"thinking": "思考中...",
|
||||
"thoughtFor": "思考了 {duration} 秒",
|
||||
"thoughtForOne": "思考了 1 秒",
|
||||
"thoughtBrief": "思考了幾秒鐘"
|
||||
},
|
||||
"dev": {
|
||||
|
||||
@@ -248,6 +248,7 @@
|
||||
"reasoning": {
|
||||
"thinking": "思考中...",
|
||||
"thoughtFor": "思考了 {duration} 秒",
|
||||
"thoughtForOne": "思考了 1 秒",
|
||||
"thoughtBrief": "思考了几秒钟"
|
||||
},
|
||||
"dev": {
|
||||
|
||||
@@ -49,6 +49,8 @@ const SPECIFIC_TEXTS: Array<[RegExp, LLMErrorCode]> = [
|
||||
],
|
||||
// Bedrock, when the output limit cut the tool call's JSON short
|
||||
[/toolUse\.input is invalid/i, "output_truncated"],
|
||||
// Bedrock, for a model id without the inference profile prefix
|
||||
[/on-demand throughput isn.t supported/i, "model_not_found"],
|
||||
[
|
||||
/insufficient[_ ]quota|insufficient balance|exceeded your current quota|credit balance is too low|余额不足/i,
|
||||
"insufficient_quota",
|
||||
@@ -105,6 +107,17 @@ function problemDetail(body: string): string | undefined {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The error text for the chat stream: what went wrong with the provider as
|
||||
* JSON for the hint, or the text the model must read to fix a tool call.
|
||||
*/
|
||||
export function streamErrorText(error: unknown): string {
|
||||
// The SDK passes an invalid tool call's error as a plain string
|
||||
if (typeof error === "string") return error
|
||||
if (isToolCallError(error)) return (error as Error).message
|
||||
return JSON.stringify(classifyLLMError(error))
|
||||
}
|
||||
|
||||
/**
|
||||
* Model and tool errors the SDK sends back to the model as the tool result,
|
||||
* so it can fix its call. Their text has to stay as it is.
|
||||
|
||||
@@ -137,6 +137,34 @@ test("edit_diagram applies all operations or none", async ({ page: p }) => {
|
||||
await expect(canvas.getByText("Broken", { exact: true })).toHaveCount(0)
|
||||
})
|
||||
|
||||
test("the thinking header is in the page language", async ({ page: p }) => {
|
||||
const events = [
|
||||
{ type: "start" },
|
||||
{ type: "reasoning-start", id: "r1" },
|
||||
{ type: "reasoning-delta", id: "r1", delta: "Plan the boxes" },
|
||||
{ type: "reasoning-end", id: "r1" },
|
||||
{ type: "text-start", id: "t1" },
|
||||
{ type: "text-delta", id: "t1", delta: "Done" },
|
||||
{ type: "text-end", id: "t1" },
|
||||
{ type: "finish" },
|
||||
]
|
||||
await p.route("**/api/chat", (route) =>
|
||||
route.fulfill({
|
||||
status: 200,
|
||||
contentType: "text/event-stream",
|
||||
body: `${events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")}data: [DONE]\n\n`,
|
||||
}),
|
||||
)
|
||||
await p.goto("/zh", { waitUntil: "networkidle" })
|
||||
await getIframe(p).waitFor({ state: "visible", timeout: 30000 })
|
||||
await sendMessage(p, "画两个框")
|
||||
await expect(p.getByText("Plan the boxes")).toBeAttached({
|
||||
timeout: 15000,
|
||||
})
|
||||
await expect(p.getByText(/^思考/)).toBeVisible()
|
||||
await expect(p.getByText(/^Thought for|^Thinking/)).toHaveCount(0)
|
||||
})
|
||||
|
||||
test("blank text before a tool call shows no empty bubble", async ({
|
||||
page: p,
|
||||
}) => {
|
||||
@@ -163,3 +191,110 @@ test("blank text before a tool call shows no empty bubble", async ({
|
||||
// Assistant text bubbles have this background
|
||||
await expect(p.locator("div.rounded-2xl.bg-muted\\/60")).toHaveCount(0)
|
||||
})
|
||||
|
||||
test("an edit right after a broken edit call starts from the real diagram", async ({
|
||||
page: p,
|
||||
}) => {
|
||||
// Seen with Claude Opus 5.5: the first edit call had invalid JSON, the
|
||||
// server rejected it, and the model sent the same edit again at once.
|
||||
// The second edit must not see the first one's streamed preview.
|
||||
const sse = (events: object[]) =>
|
||||
events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")
|
||||
const edit = {
|
||||
operations: [
|
||||
{
|
||||
operation: "add",
|
||||
cell_id: "c",
|
||||
new_xml: cell("c", "Gamma", 400),
|
||||
},
|
||||
],
|
||||
}
|
||||
const deltas = (id: string) =>
|
||||
(JSON.stringify(edit).match(/[\s\S]{1,40}/g) ?? []).map((d) => ({
|
||||
type: "tool-input-delta",
|
||||
toolCallId: id,
|
||||
inputTextDelta: d,
|
||||
}))
|
||||
const start = (id: string) => ({
|
||||
type: "tool-input-start",
|
||||
toolCallId: id,
|
||||
toolName: "edit_diagram",
|
||||
})
|
||||
// Each inner array is sent as one network chunk, 300 ms apart, so the
|
||||
// throttled UI renders between chunks like with a real model
|
||||
const replies = [
|
||||
[streamedToolCall("display_diagram", { xml: cell("a", "Alpha", 40) })],
|
||||
[
|
||||
sse([
|
||||
{ type: "start" },
|
||||
{ type: "start-step" },
|
||||
start("e1"),
|
||||
...deltas("e1"),
|
||||
]),
|
||||
sse([
|
||||
{
|
||||
type: "tool-input-error",
|
||||
toolCallId: "e1",
|
||||
toolName: "edit_diagram",
|
||||
input: "{broken",
|
||||
errorText: "JSON parsing failed",
|
||||
},
|
||||
{
|
||||
type: "tool-output-error",
|
||||
toolCallId: "e1",
|
||||
errorText: "JSON parsing failed",
|
||||
},
|
||||
{ type: "finish-step" },
|
||||
{ type: "start-step" },
|
||||
start("e2"),
|
||||
...deltas("e2"),
|
||||
]),
|
||||
`${sse([
|
||||
{
|
||||
type: "tool-input-available",
|
||||
toolCallId: "e2",
|
||||
toolName: "edit_diagram",
|
||||
input: edit,
|
||||
},
|
||||
{ type: "finish-step" },
|
||||
{ type: "finish" },
|
||||
])}data: [DONE]\n\n`,
|
||||
],
|
||||
]
|
||||
await p.addInitScript((replies) => {
|
||||
const realFetch = window.fetch
|
||||
let n = 0
|
||||
window.fetch = async (input, init) => {
|
||||
const url =
|
||||
typeof input === "string" ? input : (input as Request).url
|
||||
if (!url.endsWith("/api/chat")) return realFetch(input, init)
|
||||
const chunks = replies[n++] ?? [
|
||||
'data: {"type":"start"}\n\ndata: {"type":"finish"}\n\ndata: [DONE]\n\n',
|
||||
]
|
||||
const body = new ReadableStream({
|
||||
async start(controller) {
|
||||
for (const chunk of chunks) {
|
||||
controller.enqueue(new TextEncoder().encode(chunk))
|
||||
await new Promise((r) => setTimeout(r, 300))
|
||||
}
|
||||
controller.close()
|
||||
},
|
||||
})
|
||||
return new Response(body, {
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
})
|
||||
}
|
||||
}, replies)
|
||||
await p.goto("/", { waitUntil: "networkidle" })
|
||||
await getIframe(p).waitFor({ state: "visible", timeout: 30000 })
|
||||
const canvas = p.frameLocator("iframe")
|
||||
|
||||
await sendMessage(p, "Draw a box")
|
||||
await waitForCompleteCount(p, 1)
|
||||
await sendMessage(p, "Add another box")
|
||||
await waitForCompleteCount(p, 2)
|
||||
await expect(canvas.getByText("Gamma", { exact: true })).toBeVisible({
|
||||
timeout: 15000,
|
||||
})
|
||||
await expect(p.getByText(/No changes were made/)).toHaveCount(0)
|
||||
})
|
||||
|
||||
@@ -1,7 +1,20 @@
|
||||
// @vitest-environment node
|
||||
import { APICallError, InvalidToolInputError, RetryError } from "ai"
|
||||
import {
|
||||
APICallError,
|
||||
InvalidToolInputError,
|
||||
RetryError,
|
||||
simulateReadableStream,
|
||||
streamText,
|
||||
tool,
|
||||
} from "ai"
|
||||
import { MockLanguageModelV3 } from "ai/test"
|
||||
import { describe, expect, it } from "vitest"
|
||||
import { classifyLLMError, isToolCallError } from "@/lib/llm-errors"
|
||||
import { z } from "zod"
|
||||
import {
|
||||
classifyLLMError,
|
||||
isToolCallError,
|
||||
streamErrorText,
|
||||
} from "@/lib/llm-errors"
|
||||
|
||||
const apiError = (statusCode: number, message: string, responseBody = "") =>
|
||||
new APICallError({
|
||||
@@ -89,6 +102,14 @@ describe("classifyLLMError", () => {
|
||||
expect(classifyLLMError(timeout).code).toBe("timeout")
|
||||
})
|
||||
|
||||
it("points to the model id when Bedrock wants an inference profile", () => {
|
||||
const error = apiError(
|
||||
400,
|
||||
"Invocation of model ID anthropic.claude-sonnet-5-5 with on-demand throughput isn’t supported. Retry your request with the ID or ARN of an inference profile that contains this model.",
|
||||
)
|
||||
expect(classifyLLMError(error).code).toBe("model_not_found")
|
||||
})
|
||||
|
||||
it("names a network error the SDK wrapped", () => {
|
||||
const error = new APICallError({
|
||||
message:
|
||||
@@ -131,6 +152,64 @@ describe("classifyLLMError", () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe("streamErrorText", () => {
|
||||
it("keeps the text of a tool call the model got wrong", async () => {
|
||||
// Seen with Claude Opus 5.5: a quote left unescaped in the input
|
||||
const model = new MockLanguageModelV3({
|
||||
doStream: (async () => ({
|
||||
stream: simulateReadableStream({
|
||||
chunks: [
|
||||
{
|
||||
type: "tool-call",
|
||||
toolCallId: "c1",
|
||||
toolName: "edit_diagram",
|
||||
input: '{"operations": [{"new_xml": "as="x""}]}',
|
||||
},
|
||||
{
|
||||
type: "finish",
|
||||
finishReason: {
|
||||
unified: "tool-calls",
|
||||
raw: "tool_use",
|
||||
},
|
||||
usage: {
|
||||
inputTokens: { total: 1 },
|
||||
outputTokens: { total: 1 },
|
||||
},
|
||||
},
|
||||
],
|
||||
}),
|
||||
})) as any,
|
||||
})
|
||||
const result = streamText({
|
||||
model: model as any,
|
||||
prompt: "edit",
|
||||
tools: {
|
||||
edit_diagram: tool({
|
||||
inputSchema: z.object({ operations: z.array(z.any()) }),
|
||||
}),
|
||||
},
|
||||
})
|
||||
const errors: string[] = []
|
||||
for await (const chunk of result.toUIMessageStream({
|
||||
onError: streamErrorText,
|
||||
})) {
|
||||
if ("errorText" in chunk) errors.push(chunk.errorText)
|
||||
}
|
||||
expect(errors.length).toBeGreaterThan(0)
|
||||
for (const text of errors) {
|
||||
expect(text).toMatch(/^Invalid input for tool edit_diagram/)
|
||||
}
|
||||
})
|
||||
|
||||
it("classifies a provider error", () => {
|
||||
expect(JSON.parse(streamErrorText(apiError(401, "bad key")))).toEqual({
|
||||
type: "provider",
|
||||
code: "invalid_api_key",
|
||||
message: "bad key",
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
describe("isToolCallError", () => {
|
||||
it("spots errors the model must see unchanged", () => {
|
||||
const invalid = new InvalidToolInputError({
|
||||
|
||||
Reference in New Issue
Block a user