mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-10-09 03:07:46 +08:00
fix(server): count quota by the key actually used, and more review fixes
Found by the PR review, each with a test that failed first: - Quota: any key header skipped it, even one the provider never reads (x-aws-access-key-id with OpenAI), so a request ran on the server's key without being counted. The check now runs after the model is resolved and uses usesServerCredentials. On main already. - usesServerCredentials read the raw base URL; "/" cleans up to none, so an Ollama request ran on the server's key past the server-model check. - SGLang's default 127.0.0.1:8000 only fills the settings form. Chat and the model list used it as a real address, so the server called its own machine even with private URLs blocked. Now a base URL is required. - With a user's OpenAI key and no base URL, the SDK read the server's OPENAI_BASE_URL. The official endpoint is now passed. On main already. - The Test button refused nothing on the server's keys (Ollama Cloud), and a 15 s timeout reported "connected, no tool call". - The model list for Ollama without a base URL came from ollama.com while chat went to the server's Ollama. - Bedrock's "Too many tokens, please wait" counted as context too long. - On the server's keys the provider's error text stays in the server log; it can name the server's AWS account, role or internal hosts. - Desktop app: the preset keys are the user's own (NEXT_AI_DRAWIO_DESKTOP), so Max Output Tokens can be raised and keyless models in settings work again. A launch that found the remembered port taken no longer replaces it, which hid the user's chats and settings for good.
This commit is contained in:
+31
-39
@@ -133,36 +133,6 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
userId: userId,
|
||||
})
|
||||
|
||||
// === SERVER-SIDE QUOTA CHECK START ===
|
||||
// Quota is opt-in: only enabled when DYNAMODB_QUOTA_TABLE env var is set
|
||||
const hasOwnApiKey = !!(
|
||||
req.headers.get("x-ai-provider") &&
|
||||
(req.headers.get("x-ai-api-key") ||
|
||||
req.headers.get("x-aws-access-key-id") ||
|
||||
req.headers.get("x-vertex-api-key"))
|
||||
)
|
||||
|
||||
// Skip quota check if: quota disabled, user has own API key, or is anonymous
|
||||
if (isQuotaEnabled() && !hasOwnApiKey && userId !== "anonymous") {
|
||||
const quotaCheck = await checkAndIncrementRequest(userId, {
|
||||
requests: Number(process.env.DAILY_REQUEST_LIMIT) || 10,
|
||||
tokens: Number(process.env.DAILY_TOKEN_LIMIT) || 200000,
|
||||
tpm: Number(process.env.TPM_LIMIT) || 20000,
|
||||
})
|
||||
if (!quotaCheck.allowed) {
|
||||
return Response.json(
|
||||
{
|
||||
error: quotaCheck.error,
|
||||
type: quotaCheck.type,
|
||||
used: quotaCheck.used,
|
||||
limit: quotaCheck.limit,
|
||||
},
|
||||
{ status: 429 },
|
||||
)
|
||||
}
|
||||
}
|
||||
// === SERVER-SIDE QUOTA CHECK END ===
|
||||
|
||||
// === FILE VALIDATION START ===
|
||||
const fileValidation = validateFileParts(messages)
|
||||
if (!fileValidation.valid) {
|
||||
@@ -292,14 +262,41 @@ async function handleChatRequest(req: Request): Promise<Response> {
|
||||
)
|
||||
}
|
||||
|
||||
// === SERVER-SIDE QUOTA CHECK START ===
|
||||
// Quota is opt-in (DYNAMODB_QUOTA_TABLE) and counts what runs on the
|
||||
// server's keys. Decided by the key actually used: a key header the
|
||||
// provider never reads must not skip it.
|
||||
const countsQuota =
|
||||
isQuotaEnabled() && onServerCredentials && userId !== "anonymous"
|
||||
if (countsQuota) {
|
||||
const quotaCheck = await checkAndIncrementRequest(userId, {
|
||||
requests: Number(process.env.DAILY_REQUEST_LIMIT) || 10,
|
||||
tokens: Number(process.env.DAILY_TOKEN_LIMIT) || 200000,
|
||||
tpm: Number(process.env.TPM_LIMIT) || 20000,
|
||||
})
|
||||
if (!quotaCheck.allowed) {
|
||||
return Response.json(
|
||||
{
|
||||
error: quotaCheck.error,
|
||||
type: quotaCheck.type,
|
||||
used: quotaCheck.used,
|
||||
limit: quotaCheck.limit,
|
||||
},
|
||||
{ status: 429 },
|
||||
)
|
||||
}
|
||||
}
|
||||
// === SERVER-SIDE QUOTA CHECK END ===
|
||||
|
||||
// Retry once if the provider rejects the requested budget, or (newer
|
||||
// Claude models) the sampling or thinking settings
|
||||
const model = withOutputTokenLimitFallback(
|
||||
withDeprecatedParamsFallback(baseModel),
|
||||
)
|
||||
|
||||
// The user setting can raise the budget only on their own key (desktop users
|
||||
// can still raise it themselves); on the server's keys it can only lower it
|
||||
// The user setting can raise the budget only on their own key (in the
|
||||
// desktop app every key is the user's); on the server's keys it can only
|
||||
// lower it
|
||||
const maxOutputTokens = resolveMaxOutputTokens(
|
||||
req.headers.get("x-max-output-tokens"),
|
||||
onServerCredentials,
|
||||
@@ -587,12 +584,7 @@ IMPORTANT: The "Current diagram XML" is the SINGLE SOURCE OF TRUTH for what's on
|
||||
// Record token usage for server-side quota tracking (if enabled)
|
||||
// Use totalUsage (cumulative across all steps) instead of usage (final step only)
|
||||
// inputTokens already includes cache reads and writes in AI SDK 6
|
||||
if (
|
||||
isQuotaEnabled() &&
|
||||
!hasOwnApiKey &&
|
||||
userId !== "anonymous" &&
|
||||
totalUsage
|
||||
) {
|
||||
if (countsQuota && totalUsage) {
|
||||
const totalTokens =
|
||||
(totalUsage.inputTokens || 0) +
|
||||
(totalUsage.outputTokens || 0)
|
||||
@@ -724,7 +716,7 @@ Call this tool to get shape names and usage syntax for a specific library.`,
|
||||
|
||||
const response = result.toUIMessageStreamResponse({
|
||||
sendReasoning: true,
|
||||
onError: streamErrorText,
|
||||
onError: (error) => streamErrorText(error, onServerCredentials),
|
||||
messageMetadata: ({ part }) => {
|
||||
if (part.type === "finish") {
|
||||
const usage = (part as any).totalUsage
|
||||
|
||||
@@ -2,14 +2,15 @@ import { streamText, tool } from "ai"
|
||||
import { NextResponse } from "next/server"
|
||||
import { z } from "zod"
|
||||
import { checkAccessCode } from "@/lib/access-code"
|
||||
import { getAIModel } from "@/lib/ai-providers"
|
||||
import { getAIModel, usesServerCredentials } from "@/lib/ai-providers"
|
||||
import { classifyLLMError } from "@/lib/llm-errors"
|
||||
import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection"
|
||||
import type { ProviderName } from "@/lib/types/model-config"
|
||||
|
||||
export const runtime = "nodejs"
|
||||
|
||||
interface ValidateRequest {
|
||||
provider: string
|
||||
provider: ProviderName
|
||||
apiKey: string
|
||||
baseUrl?: string
|
||||
modelId: string
|
||||
@@ -93,6 +94,22 @@ export async function POST(req: Request) {
|
||||
{ status: 400 },
|
||||
)
|
||||
}
|
||||
// The Test button checks the user's own provider. On the server's
|
||||
// keys (Ollama Cloud without a key or URL) anyone could run any model.
|
||||
if (
|
||||
usesServerCredentials(provider, {
|
||||
apiKey,
|
||||
baseUrl,
|
||||
awsAccessKeyId,
|
||||
awsSecretAccessKey,
|
||||
vertexApiKey,
|
||||
})
|
||||
) {
|
||||
return NextResponse.json(
|
||||
{ valid: false, error: "API key is required" },
|
||||
{ status: 400 },
|
||||
)
|
||||
}
|
||||
|
||||
// The same model the chat would use. A client base URL makes it
|
||||
// refuse redirects to internal hosts.
|
||||
@@ -130,6 +147,14 @@ export async function POST(req: Request) {
|
||||
let finishReason: string | undefined
|
||||
for await (const part of result.fullStream) {
|
||||
if (part.type === "error") throw part.error
|
||||
// The timeout ends the stream with an abort part, not an error
|
||||
if (part.type === "abort") {
|
||||
const timeout = new Error(
|
||||
`The model did not answer within ${TEST_TIMEOUT_MS / 1000} s.`,
|
||||
)
|
||||
timeout.name = "TimeoutError"
|
||||
throw timeout
|
||||
}
|
||||
if (part.type === "tool-call") {
|
||||
calledTool = true
|
||||
break
|
||||
|
||||
Reference in New Issue
Block a user