mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-10-10 19:49:52 +08:00
fix: older defects (batch C) and the second batch's review
Chats: - New Chat right after an answer saves that chat once. Saves run one at a time and read the chat on screen when their turn comes; a save scheduled for a chat that is no longer on screen is dropped. A chat whose id was still on its way to the URL no longer comes back after New Chat (the next answer went into it). - Crossing the 768 px breakpoint keeps the chat panel: a streaming answer, unsaved messages and attachments stay. The panel gets the sizes of each side, and a panel collapsed on desktop opens on mobile. - The chat's export waits for its own reply: an edit's history export still on its way no longer answers it with the older diagram, and two file saves at once no longer swap results. - A second edit in one answer is previewed on the first edit's result. - Stop also ends a running screenshot check; a chat that cannot be saved (storage full) can be left with "Continue without saving". - Small diagrams with shapes count as diagrams; the tool card no longer crashes on malformed operations. Quota and providers: - Requests that reach the server's own endpoints count toward the quota: EdgeOne (always its own endpoint now), a private base URL whatever key header is sent, keyless Ollama without a URL. With the quota on, a redirect is followed only to a public address. The output cap applies to these requests too. - Stop records the tokens of the steps that finished; the screenshot check counts its tokens without counting a request. - EdgeOne configured only by AI_PROVIDER works, also in the admin Test, which forwards the access code. Azure set up only in the admin panel works in chat. The Test sends a Bedrock session token. - The admin panel's Test of an entry without a URL uses the server's URL as the server does (no private address check for it); the admin panel no longer writes an Ollama URL. MCP server: - Write tools and start_session run one at a time, so two at once never drop each other's change; a cancelled call waiting its turn is skipped. get_diagram and export_diagram keep the session they started with. - Export to .drawio first gets the user's latest edits from the browser. - History thumbnails: one that arrives after the next AI write is dropped; a sync reply keeps the image; a version that changed only page settings is its own entry. - A diagram over the 10 MB limit is saved without its image, or the user is told to download it (the server now answers 413 instead of cutting the connection). - Labels holding text like id='1' or parent='1' are no longer read as attributes (a layer or a parent was deleted). A broken bare <mxGraphModel> file is refused. - After a sync reply the tab no longer sends its autosave copy again. Desktop and files: - A newer switch of the same preset is not rolled back by an older one that failed. .env values with escaped quotes are read whole. - MCP saved files: a file that could not be read stays protected while a folder without permission hides it, and is saved again once deleted. - The desktop app reports "no chats" only when the count was read and no model settings are stored.
This commit is contained in:
@@ -48,6 +48,7 @@ export async function POST(req: Request) {
|
||||
sameEndpoint && stored ? [stored] : [],
|
||||
)
|
||||
|
||||
const serverUrl = globalBaseUrl(resolved.provider)
|
||||
return validateModel(
|
||||
new Request(new URL("/api/validate-model", req.url), {
|
||||
method: "POST",
|
||||
@@ -55,14 +56,23 @@ export async function POST(req: Request) {
|
||||
"Content-Type": "application/json",
|
||||
// Checked again there, in place of an access code
|
||||
"x-admin-password": req.headers.get("x-admin-password") || "",
|
||||
// The EdgeOne function checks the access code and Pages
|
||||
// cookies, and its URL is built from the page's origin
|
||||
"x-access-code": req.headers.get("x-access-code") || "",
|
||||
cookie: req.headers.get("cookie") || "",
|
||||
...(req.headers.get("origin") && {
|
||||
origin: req.headers.get("origin") as string,
|
||||
}),
|
||||
},
|
||||
body: JSON.stringify({
|
||||
provider: resolved.provider,
|
||||
apiKey: resolved.apiKey,
|
||||
// Without a URL of its own, chat sends the entry's key to
|
||||
// the server's <P>_BASE_URL: test that endpoint, not
|
||||
// another one
|
||||
baseUrl: resolved.baseUrl || globalBaseUrl(resolved.provider),
|
||||
// another one. It is the server's own, which chat uses
|
||||
// without the checks for a URL a user typed.
|
||||
baseUrl: resolved.baseUrl || serverUrl,
|
||||
...(!resolved.baseUrl && serverUrl && { serverBaseUrl: true }),
|
||||
modelId: body.modelId,
|
||||
awsAccessKeyId: resolved.awsAccessKeyId,
|
||||
awsSecretAccessKey: resolved.awsSecretAccessKey,
|
||||
|
||||
+60
-25
@@ -13,6 +13,7 @@ import { z } from "zod"
|
||||
import { checkAccessCode, rejectCrossSite } from "@/lib/access-code"
|
||||
import {
|
||||
CACHE_POINT,
|
||||
edgeOneEndpoint,
|
||||
getAIModel,
|
||||
getServerProvider,
|
||||
SINGLE_SYSTEM_PROVIDERS,
|
||||
@@ -189,15 +190,16 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
}
|
||||
|
||||
// A server model's provider comes from its config: for one set up in
|
||||
// the admin panel the header holds the provider name's slug
|
||||
const isEdgeOne = (serverModelConfig.provider || provider) === "edgeone"
|
||||
// the admin panel the header holds the provider name's slug. Without
|
||||
// either, the server's own AI_PROVIDER.
|
||||
const isEdgeOne =
|
||||
(serverModelConfig.provider || provider || getServerProvider()) ===
|
||||
"edgeone"
|
||||
|
||||
// For EdgeOne provider, construct full URL from request origin
|
||||
// because createOpenAI needs absolute URL, not relative path
|
||||
if (isEdgeOne && !baseUrl) {
|
||||
const origin = req.headers.get("origin") || new URL(req.url).origin
|
||||
baseUrl = `${origin}/api/edgeai`
|
||||
}
|
||||
// EdgeOne is this deployment's own function, whatever URL the request
|
||||
// names: another host would get the user's EdgeOne cookies, and the
|
||||
// quota counts it. Absolute, as the SDK needs.
|
||||
if (isEdgeOne) baseUrl = edgeOneEndpoint(req)
|
||||
|
||||
// Same rule as validate-model: with ALLOW_PRIVATE_URLS=false a request may
|
||||
// not point the server at a private or internal address
|
||||
@@ -212,8 +214,12 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
const cookieHeader = req.headers.get("cookie")
|
||||
|
||||
const clientOverrides = {
|
||||
// Server model provider takes precedence over client header
|
||||
provider: serverModelConfig.provider || provider,
|
||||
// Server model provider takes precedence over client header; EdgeOne
|
||||
// named only in AI_PROVIDER is named here, for its own base URL
|
||||
provider:
|
||||
serverModelConfig.provider ||
|
||||
provider ||
|
||||
(isEdgeOne ? "edgeone" : null),
|
||||
baseUrl,
|
||||
apiKey: req.headers.get("x-ai-api-key"),
|
||||
// A server model runs the model it was configured with, whatever the header says
|
||||
@@ -274,18 +280,24 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
|
||||
// === SERVER-SIDE QUOTA CHECK START ===
|
||||
// Quota is opt-in (DYNAMODB_QUOTA_TABLE) and counts what runs on the
|
||||
// server's keys, or on its keyless Ollama or EdgeOne. Decided by the key
|
||||
// actually used: a key header the provider never reads must not skip it.
|
||||
// EdgeOne never reads one; keyless Ollama at a private address is the
|
||||
// server's own network.
|
||||
// server's keys, or on the server's own endpoints: EdgeOne, its keyless
|
||||
// Ollama, and anything at a private address (the server's network,
|
||||
// which ignores a dummy key header). Bedrock and EdgeOne never use the
|
||||
// base URL header. In the desktop app every endpoint is the user's.
|
||||
const clientBaseUrl = normalizeBaseUrl(
|
||||
req.headers.get("x-ai-base-url") ?? "",
|
||||
)
|
||||
const usesClientBaseUrl =
|
||||
resolvedProvider !== "bedrock" && resolvedProvider !== "edgeone"
|
||||
const onServerEndpoint =
|
||||
(resolvedProvider === "edgeone" && !clientBaseUrl) ||
|
||||
(resolvedProvider === "ollama" &&
|
||||
!clientOverrides.apiKey &&
|
||||
(!clientBaseUrl || (await isPrivateUrl(clientBaseUrl))))
|
||||
process.env.NEXT_AI_DRAWIO_DESKTOP !== "1" &&
|
||||
(resolvedProvider === "edgeone" ||
|
||||
(resolvedProvider === "ollama" &&
|
||||
!clientBaseUrl &&
|
||||
!clientOverrides.apiKey) ||
|
||||
(usesClientBaseUrl &&
|
||||
!!clientBaseUrl &&
|
||||
(await isPrivateUrl(clientBaseUrl))))
|
||||
const countsQuota =
|
||||
isQuotaEnabled() &&
|
||||
(onServerCredentials || onServerEndpoint) &&
|
||||
@@ -317,11 +329,11 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
)
|
||||
|
||||
// The user setting can raise the budget only on their own key (in the
|
||||
// desktop app every key is the user's); on the server's keys it can only
|
||||
// lower it
|
||||
// desktop app every key is the user's); on the server's keys or own
|
||||
// endpoints it can only lower it
|
||||
const maxOutputTokens = resolveMaxOutputTokens(
|
||||
req.headers.get("x-max-output-tokens"),
|
||||
onServerCredentials,
|
||||
onServerCredentials || onServerEndpoint,
|
||||
)
|
||||
console.log(`[maxOutputTokens] ${maxOutputTokens}`)
|
||||
|
||||
@@ -353,8 +365,13 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
${userInputText}
|
||||
"""`
|
||||
|
||||
// Convert UIMessages to ModelMessages and add system message
|
||||
const modelMessages = await convertToModelMessages(messages)
|
||||
// Convert UIMessages to ModelMessages and add system message. A tool
|
||||
// call that never got its result (the user stopped while it ran) is
|
||||
// left out: the SDK would refuse this and every later request of the
|
||||
// chat (MissingToolResultsError)
|
||||
const modelMessages = await convertToModelMessages(messages, {
|
||||
ignoreIncompleteToolCalls: true,
|
||||
})
|
||||
|
||||
// DEBUG_LLM_PAYLOAD=true logs the incoming message structure
|
||||
if (DEBUG_LLM_PAYLOAD) {
|
||||
@@ -541,6 +558,8 @@ IMPORTANT: The "Current diagram XML" is the SINGLE SOURCE OF TRUTH for what's on
|
||||
|
||||
const allMessages = [...systemMessages, ...enhancedMessages]
|
||||
|
||||
// Set by onAbort, which records the finished steps' tokens itself
|
||||
let stopped = false
|
||||
const result = streamText({
|
||||
model,
|
||||
// The system messages carry cache points, so they go in messages.
|
||||
@@ -606,7 +625,7 @@ IMPORTANT: The "Current diagram XML" is the SINGLE SOURCE OF TRUTH for what's on
|
||||
// Record token usage for server-side quota tracking (if enabled)
|
||||
// Use totalUsage (cumulative across all steps) instead of usage (final step only)
|
||||
// inputTokens already includes cache reads and writes in AI SDK 6
|
||||
if (countsQuota && totalUsage) {
|
||||
if (countsQuota && totalUsage && !stopped) {
|
||||
const totalTokens =
|
||||
(totalUsage.inputTokens || 0) +
|
||||
(totalUsage.outputTokens || 0)
|
||||
@@ -618,7 +637,23 @@ IMPORTANT: The "Current diagram XML" is the SINGLE SOURCE OF TRUTH for what's on
|
||||
console.error(error) // what AI SDK does without an onError
|
||||
endTrace()
|
||||
},
|
||||
onAbort: () => endTrace(),
|
||||
onAbort: ({ steps }) => {
|
||||
stopped = true
|
||||
endTrace()
|
||||
// Stopped (or disconnected) after some steps finished: their
|
||||
// tokens were used, or stopping every request after a costly
|
||||
// first step would get around the token limits
|
||||
if (countsQuota) {
|
||||
const tokens = steps.reduce(
|
||||
(sum, step) =>
|
||||
sum +
|
||||
(step.usage.inputTokens || 0) +
|
||||
(step.usage.outputTokens || 0),
|
||||
0,
|
||||
)
|
||||
if (tokens > 0) recordTokenUsage(userId, tokens)
|
||||
}
|
||||
},
|
||||
tools: {
|
||||
// Client-side tool that will be executed on the client
|
||||
display_diagram: {
|
||||
|
||||
@@ -6,6 +6,12 @@
|
||||
import { Output, streamText } from "ai"
|
||||
import { checkAccessCode, rejectCrossSite } from "@/lib/access-code"
|
||||
import { getValidationModel } from "@/lib/ai-providers"
|
||||
import {
|
||||
checkAndIncrementRequest,
|
||||
isQuotaEnabled,
|
||||
recordTokenUsage,
|
||||
} from "@/lib/dynamo-quota-manager"
|
||||
import { getUserIdFromRequest } from "@/lib/user-id"
|
||||
import { VALIDATION_SYSTEM_PROMPT } from "@/lib/validation-prompts"
|
||||
import {
|
||||
type ValidationResult,
|
||||
@@ -78,6 +84,35 @@ export async function POST(req: Request): Promise<Response> {
|
||||
)
|
||||
}
|
||||
|
||||
// It runs the server's vision model: with the quota on, the daily
|
||||
// and per-minute token limits apply, and its tokens are counted. Not
|
||||
// the request limit, which is for chats: the day's last chat still
|
||||
// gets its check, and a check does not count as a chat.
|
||||
const userId = getUserIdFromRequest(req)
|
||||
const countsQuota = isQuotaEnabled() && userId !== "anonymous"
|
||||
if (countsQuota) {
|
||||
const quotaCheck = await checkAndIncrementRequest(
|
||||
userId,
|
||||
{
|
||||
requests: 0,
|
||||
tokens: Number(process.env.DAILY_TOKEN_LIMIT) || 200000,
|
||||
tpm: Number(process.env.TPM_LIMIT) || 20000,
|
||||
},
|
||||
0,
|
||||
)
|
||||
if (!quotaCheck.allowed) {
|
||||
return Response.json(
|
||||
{
|
||||
error: quotaCheck.error,
|
||||
type: quotaCheck.type,
|
||||
used: quotaCheck.used,
|
||||
limit: quotaCheck.limit,
|
||||
},
|
||||
{ status: 429 },
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Get the validation model
|
||||
let model
|
||||
try {
|
||||
@@ -120,7 +155,14 @@ export async function POST(req: Request): Promise<Response> {
|
||||
],
|
||||
maxOutputTokens: 1024,
|
||||
abortSignal: AbortSignal.timeout(timeout),
|
||||
onFinish: ({ output }) => {
|
||||
onFinish: ({ output, totalUsage }) => {
|
||||
if (countsQuota && totalUsage) {
|
||||
recordTokenUsage(
|
||||
userId,
|
||||
(totalUsage.inputTokens || 0) +
|
||||
(totalUsage.outputTokens || 0),
|
||||
)
|
||||
}
|
||||
if (sessionId && output) {
|
||||
console.log(
|
||||
`[validate-diagram] Session ${sessionId}: valid=${output.valid}, issues=${output.issues?.length ?? 0}`,
|
||||
|
||||
@@ -3,7 +3,12 @@ import { NextResponse } from "next/server"
|
||||
import { z } from "zod"
|
||||
import { checkAccessCode, rejectCrossSite } from "@/lib/access-code"
|
||||
import { checkAdminAuth } from "@/lib/admin/auth"
|
||||
import { getAIModel, usesServerCredentials } from "@/lib/ai-providers"
|
||||
import {
|
||||
edgeOneEndpoint,
|
||||
getAIModel,
|
||||
globalBaseUrl,
|
||||
usesServerCredentials,
|
||||
} from "@/lib/ai-providers"
|
||||
import { classifyLLMError } from "@/lib/llm-errors"
|
||||
import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection"
|
||||
import type { ProviderName } from "@/lib/types/model-config"
|
||||
@@ -19,8 +24,11 @@ interface ValidateRequest {
|
||||
awsAccessKeyId?: string
|
||||
awsSecretAccessKey?: string
|
||||
awsRegion?: string
|
||||
awsSessionToken?: string
|
||||
// Vertex AI specific
|
||||
vertexApiKey?: string // Express Mode API key
|
||||
// Set by the admin panel's Test: baseUrl is the server's <P>_BASE_URL
|
||||
serverBaseUrl?: boolean
|
||||
}
|
||||
|
||||
const TEST_TIMEOUT_MS = 15_000
|
||||
@@ -47,11 +55,11 @@ export async function POST(req: Request) {
|
||||
const {
|
||||
provider,
|
||||
apiKey,
|
||||
baseUrl,
|
||||
modelId,
|
||||
awsAccessKeyId,
|
||||
awsSecretAccessKey,
|
||||
awsRegion,
|
||||
awsSessionToken,
|
||||
// Note: Express Mode only needs vertexApiKey
|
||||
vertexApiKey,
|
||||
} = body
|
||||
@@ -62,9 +70,26 @@ export async function POST(req: Request) {
|
||||
{ status: 400 },
|
||||
)
|
||||
}
|
||||
// EdgeOne is this site's own function, as in the chat; the admin
|
||||
// panel's Test sends no URL, and a relative one cannot be fetched
|
||||
const baseUrl =
|
||||
provider === "edgeone" ? edgeOneEndpoint(req) : body.baseUrl
|
||||
// The admin panel's Test of an entry without a URL sends the
|
||||
// server's own <P>_BASE_URL, which chat uses as it is: not a URL a
|
||||
// user chose, so no private-address or redirect rules
|
||||
const serverUrl =
|
||||
body.serverBaseUrl === true &&
|
||||
!!baseUrl &&
|
||||
baseUrl === globalBaseUrl(provider) &&
|
||||
!checkAdminAuth(req)
|
||||
|
||||
// SECURITY: Block SSRF attacks via custom baseUrl
|
||||
if (baseUrl && !allowPrivateUrls() && (await isPrivateUrl(baseUrl))) {
|
||||
if (
|
||||
baseUrl &&
|
||||
!serverUrl &&
|
||||
!allowPrivateUrls() &&
|
||||
(await isPrivateUrl(baseUrl))
|
||||
) {
|
||||
return NextResponse.json(
|
||||
{ valid: false, error: "Invalid base URL" },
|
||||
{ status: 400 },
|
||||
@@ -122,9 +147,12 @@ export async function POST(req: Request) {
|
||||
modelId,
|
||||
apiKey,
|
||||
baseUrl,
|
||||
trustedBaseUrl: serverUrl,
|
||||
awsAccessKeyId,
|
||||
awsSecretAccessKey,
|
||||
awsRegion,
|
||||
// Temporary AWS credentials need it, as in the chat
|
||||
awsSessionToken,
|
||||
vertexApiKey,
|
||||
// EdgeOne checks the Pages cookies and the access code
|
||||
...(provider === "edgeone" && {
|
||||
|
||||
Reference in New Issue
Block a user