diff --git a/app/api/chat/route.ts b/app/api/chat/route.ts index e11b789e..cececdf5 100644 --- a/app/api/chat/route.ts +++ b/app/api/chat/route.ts @@ -38,7 +38,7 @@ import { setTraceOutput, wrapWithObserve, } from "@/lib/langfuse" -import { classifyLLMError, isToolCallError } from "@/lib/llm-errors" +import { classifyLLMError, streamErrorText } from "@/lib/llm-errors" import { resolveMaxOutputTokens, withOutputTokenLimitFallback, @@ -724,12 +724,7 @@ Call this tool to get shape names and usage syntax for a specific library.`, const response = result.toUIMessageStreamResponse({ sendReasoning: true, - // The same text goes back to the model when its tool call was - // invalid, so it can fix it: keep that one as it is - onError: (error) => - isToolCallError(error) - ? (error as Error).message - : JSON.stringify(classifyLLMError(error)), + onError: streamErrorText, messageMetadata: ({ part }) => { if (part.type === "finish") { const usage = (part as any).totalUsage diff --git a/components/chat-message-display.tsx b/components/chat-message-display.tsx index d86abadb..477c9f65 100644 --- a/components/chat-message-display.tsx +++ b/components/chat-message-display.tsx @@ -26,6 +26,7 @@ import { ReasoningContent, ReasoningTrigger, } from "@/components/ai-elements/reasoning" +import { Shimmer } from "@/components/ai-elements/shimmer" import { ChatLobby } from "@/components/chat/ChatLobby" import { TemplateCreateDialog } from "@/components/chat/TemplateCreateDialog" import { ToolCallCard } from "@/components/chat/ToolCallCard" @@ -193,6 +194,23 @@ export function ChatMessageDisplay({ currentInput = "", }: ChatMessageDisplayProps) { const dict = useDictionary() + // The thinking header in the page language + const thinkingMessage = (isStreaming: boolean, duration?: number) => { + if (isStreaming || duration === 0) { + return {dict.reasoning.thinking} + } + if (duration === undefined) return

{dict.reasoning.thoughtBrief}

+ return ( +

+ {duration === 1 + ? dict.reasoning.thoughtForOne + : dict.reasoning.thoughtFor.replace( + "{duration}", + String(duration), + )} +

+ ) + } const { chartXML, loadDiagram: onDisplayChart } = useDiagram() const messagesEndRef = useRef(null) const scrollTopRef = useRef(null) @@ -404,6 +422,10 @@ export function ChatMessageDisplay({ // Previous messages are already processed and won't change const messagesToProcess = messages.length > 0 ? [messages[messages.length - 1]] : [] + // The diagram without streamed previews. Undoing a failed edit's + // preview below changes it before chartXML catches up, and an edit + // streaming right after must start from the undone diagram. + let baseXml = chartXML messagesToProcess.forEach((message) => { // Messages restored from a saved session were applied before it was @@ -460,13 +482,12 @@ export function ChatMessageDisplay({ // Handle edit_diagram streaming - apply operations incrementally for preview // Uses shared editDiagramOriginalXmlRef to coordinate with tool handler - if ( - part.type === "tool-edit_diagram" && - input?.operations - ) { + if (part.type === "tool-edit_diagram") { // Failed or stopped: if the original XML is still - // stored, the tool handler never ran (user pressed - // stop), so undo the streamed preview here. + // stored, the tool handler never ran (invalid + // JSON, or the user pressed stop), so undo the + // streamed preview here. Invalid JSON leaves no + // operations in the input, so check this first. if (state === "output-error") { const originalXml = editDiagramOriginalXmlRef.current.get( @@ -477,9 +498,11 @@ export function ChatMessageDisplay({ toolCallId, ) onDisplayChart(originalXml, true) + baseXml = originalXml } return } + if (!input?.operations) return if (state !== "input-streaming") { // Input complete: the tool handler applies the @@ -506,7 +529,7 @@ export function ChatMessageDisplay({ toolCallId, ) ) { - if (!chartXML) { + if (!baseXml) { console.warn( "[edit_diagram streaming] No chart XML available", ) @@ -514,7 +537,7 @@ export function ChatMessageDisplay({ } editDiagramOriginalXmlRef.current.set( toolCallId, - chartXML, + baseXml, ) } const originalXml = @@ -739,7 +762,11 @@ export function ChatMessageDisplay({ !isRestoredMessage } > - + { reasoningPart.text diff --git a/lib/i18n/dictionaries/en.json b/lib/i18n/dictionaries/en.json index 27277be8..d1b7c41d 100644 --- a/lib/i18n/dictionaries/en.json +++ b/lib/i18n/dictionaries/en.json @@ -248,6 +248,7 @@ "reasoning": { "thinking": "Thinking...", "thoughtFor": "Thought for {duration} seconds", + "thoughtForOne": "Thought for 1 second", "thoughtBrief": "Thought for a few seconds" }, "dev": { diff --git a/lib/i18n/dictionaries/ja.json b/lib/i18n/dictionaries/ja.json index fcb1bcc1..87861452 100644 --- a/lib/i18n/dictionaries/ja.json +++ b/lib/i18n/dictionaries/ja.json @@ -248,6 +248,7 @@ "reasoning": { "thinking": "考え中...", "thoughtFor": "{duration} 秒考えました", + "thoughtForOne": "1 秒考えました", "thoughtBrief": "数秒考えました" }, "dev": { diff --git a/lib/i18n/dictionaries/zh-Hant.json b/lib/i18n/dictionaries/zh-Hant.json index 40ecc53f..cd6d8bb9 100644 --- a/lib/i18n/dictionaries/zh-Hant.json +++ b/lib/i18n/dictionaries/zh-Hant.json @@ -248,6 +248,7 @@ "reasoning": { "thinking": "思考中...", "thoughtFor": "思考了 {duration} 秒", + "thoughtForOne": "思考了 1 秒", "thoughtBrief": "思考了幾秒鐘" }, "dev": { diff --git a/lib/i18n/dictionaries/zh.json b/lib/i18n/dictionaries/zh.json index 3a2d5039..2c048fa4 100644 --- a/lib/i18n/dictionaries/zh.json +++ b/lib/i18n/dictionaries/zh.json @@ -248,6 +248,7 @@ "reasoning": { "thinking": "思考中...", "thoughtFor": "思考了 {duration} 秒", + "thoughtForOne": "思考了 1 秒", "thoughtBrief": "思考了几秒钟" }, "dev": { diff --git a/lib/llm-errors.ts b/lib/llm-errors.ts index fb87d809..87fb15d0 100644 --- a/lib/llm-errors.ts +++ b/lib/llm-errors.ts @@ -49,6 +49,8 @@ const SPECIFIC_TEXTS: Array<[RegExp, LLMErrorCode]> = [ ], // Bedrock, when the output limit cut the tool call's JSON short [/toolUse\.input is invalid/i, "output_truncated"], + // Bedrock, for a model id without the inference profile prefix + [/on-demand throughput isn.t supported/i, "model_not_found"], [ /insufficient[_ ]quota|insufficient balance|exceeded your current quota|credit balance is too low|余额不足/i, "insufficient_quota", @@ -105,6 +107,17 @@ function problemDetail(body: string): string | undefined { } } +/** + * The error text for the chat stream: what went wrong with the provider as + * JSON for the hint, or the text the model must read to fix a tool call. + */ +export function streamErrorText(error: unknown): string { + // The SDK passes an invalid tool call's error as a plain string + if (typeof error === "string") return error + if (isToolCallError(error)) return (error as Error).message + return JSON.stringify(classifyLLMError(error)) +} + /** * Model and tool errors the SDK sends back to the model as the tool result, * so it can fix its call. Their text has to stay as it is. diff --git a/tests/e2e/diagram-content.spec.ts b/tests/e2e/diagram-content.spec.ts index f5368b7c..70766f2a 100644 --- a/tests/e2e/diagram-content.spec.ts +++ b/tests/e2e/diagram-content.spec.ts @@ -137,6 +137,34 @@ test("edit_diagram applies all operations or none", async ({ page: p }) => { await expect(canvas.getByText("Broken", { exact: true })).toHaveCount(0) }) +test("the thinking header is in the page language", async ({ page: p }) => { + const events = [ + { type: "start" }, + { type: "reasoning-start", id: "r1" }, + { type: "reasoning-delta", id: "r1", delta: "Plan the boxes" }, + { type: "reasoning-end", id: "r1" }, + { type: "text-start", id: "t1" }, + { type: "text-delta", id: "t1", delta: "Done" }, + { type: "text-end", id: "t1" }, + { type: "finish" }, + ] + await p.route("**/api/chat", (route) => + route.fulfill({ + status: 200, + contentType: "text/event-stream", + body: `${events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("")}data: [DONE]\n\n`, + }), + ) + await p.goto("/zh", { waitUntil: "networkidle" }) + await getIframe(p).waitFor({ state: "visible", timeout: 30000 }) + await sendMessage(p, "画两个框") + await expect(p.getByText("Plan the boxes")).toBeAttached({ + timeout: 15000, + }) + await expect(p.getByText(/^思考/)).toBeVisible() + await expect(p.getByText(/^Thought for|^Thinking/)).toHaveCount(0) +}) + test("blank text before a tool call shows no empty bubble", async ({ page: p, }) => { @@ -163,3 +191,110 @@ test("blank text before a tool call shows no empty bubble", async ({ // Assistant text bubbles have this background await expect(p.locator("div.rounded-2xl.bg-muted\\/60")).toHaveCount(0) }) + +test("an edit right after a broken edit call starts from the real diagram", async ({ + page: p, +}) => { + // Seen with Claude Opus 5.5: the first edit call had invalid JSON, the + // server rejected it, and the model sent the same edit again at once. + // The second edit must not see the first one's streamed preview. + const sse = (events: object[]) => + events.map((e) => `data: ${JSON.stringify(e)}\n\n`).join("") + const edit = { + operations: [ + { + operation: "add", + cell_id: "c", + new_xml: cell("c", "Gamma", 400), + }, + ], + } + const deltas = (id: string) => + (JSON.stringify(edit).match(/[\s\S]{1,40}/g) ?? []).map((d) => ({ + type: "tool-input-delta", + toolCallId: id, + inputTextDelta: d, + })) + const start = (id: string) => ({ + type: "tool-input-start", + toolCallId: id, + toolName: "edit_diagram", + }) + // Each inner array is sent as one network chunk, 300 ms apart, so the + // throttled UI renders between chunks like with a real model + const replies = [ + [streamedToolCall("display_diagram", { xml: cell("a", "Alpha", 40) })], + [ + sse([ + { type: "start" }, + { type: "start-step" }, + start("e1"), + ...deltas("e1"), + ]), + sse([ + { + type: "tool-input-error", + toolCallId: "e1", + toolName: "edit_diagram", + input: "{broken", + errorText: "JSON parsing failed", + }, + { + type: "tool-output-error", + toolCallId: "e1", + errorText: "JSON parsing failed", + }, + { type: "finish-step" }, + { type: "start-step" }, + start("e2"), + ...deltas("e2"), + ]), + `${sse([ + { + type: "tool-input-available", + toolCallId: "e2", + toolName: "edit_diagram", + input: edit, + }, + { type: "finish-step" }, + { type: "finish" }, + ])}data: [DONE]\n\n`, + ], + ] + await p.addInitScript((replies) => { + const realFetch = window.fetch + let n = 0 + window.fetch = async (input, init) => { + const url = + typeof input === "string" ? input : (input as Request).url + if (!url.endsWith("/api/chat")) return realFetch(input, init) + const chunks = replies[n++] ?? [ + 'data: {"type":"start"}\n\ndata: {"type":"finish"}\n\ndata: [DONE]\n\n', + ] + const body = new ReadableStream({ + async start(controller) { + for (const chunk of chunks) { + controller.enqueue(new TextEncoder().encode(chunk)) + await new Promise((r) => setTimeout(r, 300)) + } + controller.close() + }, + }) + return new Response(body, { + headers: { "content-type": "text/event-stream" }, + }) + } + }, replies) + await p.goto("/", { waitUntil: "networkidle" }) + await getIframe(p).waitFor({ state: "visible", timeout: 30000 }) + const canvas = p.frameLocator("iframe") + + await sendMessage(p, "Draw a box") + await waitForCompleteCount(p, 1) + await sendMessage(p, "Add another box") + await waitForCompleteCount(p, 2) + await expect(canvas.getByText("Gamma", { exact: true })).toBeVisible({ + timeout: 15000, + }) + await expect(p.getByText(/No changes were made/)).toHaveCount(0) +}) diff --git a/tests/unit/llm-errors.test.ts b/tests/unit/llm-errors.test.ts index 65903cf9..07be0d0b 100644 --- a/tests/unit/llm-errors.test.ts +++ b/tests/unit/llm-errors.test.ts @@ -1,7 +1,20 @@ // @vitest-environment node -import { APICallError, InvalidToolInputError, RetryError } from "ai" +import { + APICallError, + InvalidToolInputError, + RetryError, + simulateReadableStream, + streamText, + tool, +} from "ai" +import { MockLanguageModelV3 } from "ai/test" import { describe, expect, it } from "vitest" -import { classifyLLMError, isToolCallError } from "@/lib/llm-errors" +import { z } from "zod" +import { + classifyLLMError, + isToolCallError, + streamErrorText, +} from "@/lib/llm-errors" const apiError = (statusCode: number, message: string, responseBody = "") => new APICallError({ @@ -89,6 +102,14 @@ describe("classifyLLMError", () => { expect(classifyLLMError(timeout).code).toBe("timeout") }) + it("points to the model id when Bedrock wants an inference profile", () => { + const error = apiError( + 400, + "Invocation of model ID anthropic.claude-sonnet-5-5 with on-demand throughput isn’t supported. Retry your request with the ID or ARN of an inference profile that contains this model.", + ) + expect(classifyLLMError(error).code).toBe("model_not_found") + }) + it("names a network error the SDK wrapped", () => { const error = new APICallError({ message: @@ -131,6 +152,64 @@ describe("classifyLLMError", () => { }) }) +describe("streamErrorText", () => { + it("keeps the text of a tool call the model got wrong", async () => { + // Seen with Claude Opus 5.5: a quote left unescaped in the input + const model = new MockLanguageModelV3({ + doStream: (async () => ({ + stream: simulateReadableStream({ + chunks: [ + { + type: "tool-call", + toolCallId: "c1", + toolName: "edit_diagram", + input: '{"operations": [{"new_xml": "as="x""}]}', + }, + { + type: "finish", + finishReason: { + unified: "tool-calls", + raw: "tool_use", + }, + usage: { + inputTokens: { total: 1 }, + outputTokens: { total: 1 }, + }, + }, + ], + }), + })) as any, + }) + const result = streamText({ + model: model as any, + prompt: "edit", + tools: { + edit_diagram: tool({ + inputSchema: z.object({ operations: z.array(z.any()) }), + }), + }, + }) + const errors: string[] = [] + for await (const chunk of result.toUIMessageStream({ + onError: streamErrorText, + })) { + if ("errorText" in chunk) errors.push(chunk.errorText) + } + expect(errors.length).toBeGreaterThan(0) + for (const text of errors) { + expect(text).toMatch(/^Invalid input for tool edit_diagram/) + } + }) + + it("classifies a provider error", () => { + expect(JSON.parse(streamErrorText(apiError(401, "bad key")))).toEqual({ + type: "provider", + code: "invalid_api_key", + message: "bad key", + }) + }) +}) + describe("isToolCallError", () => { it("spots errors the model must see unchanged", () => { const invalid = new InvalidToolInputError({