Files
next-ai-draw-io/lib/output-token-limit.ts
T
dayuan.jiang 366480426d fix(chat): close credential leaks and harden the chat route
- Vertex: a client-supplied base URL only works with the client's own Vertex key
- Accept only data: URLs for file parts in every message, so the server never downloads them
- Output budget retry accounts for the thinking budget Bedrock/Anthropic add, and reads
  Volcengine, DashScope, SGLang and vLLM rejections; falls back to 16000 once
- x-max-output-tokens can only lower the budget on server credentials
- On server credentials only server models or AI_MODEL entries can be used
- Drop tool results together with the invalid tool calls they belong to
- Count quota tokens as input + output (cached tokens were counted twice)
- Private-URL check for custom base URLs, end Langfuse traces on error/abort/early return
- Fix repairToolCall ordering and placeholder, align edit_diagram prompt with operations
- Panel Bedrock keys are read from ADMIN_AWS_*; forward the access code to EdgeOne
- isMinimalDiagram only treats root cells as an empty canvas
2026-10-03 17:45:41 +09:00

225 lines
8.3 KiB
TypeScript

import { wrapLanguageModel } from "ai"
type WrappedModel = ReturnType<typeof wrapLanguageModel>
/**
* Default output budget for a chat turn.
*
* This has to cover thinking + prose + the tool call, because reasoning models
* spend it in that order. Measured on deepseek-v4-flash: refining an existing
* diagram burned 16000 tokens on thinking alone and the request ended with
* finishReason "length" before display_diagram was ever called (issue #924).
* 64000 leaves room for the plan and the XML in one turn.
*/
export const DEFAULT_MAX_OUTPUT_TOKENS = 64000
/** Ceiling for the user-supplied override, to catch typos like an extra zero. */
export const MAX_OUTPUT_TOKENS_LIMIT = 200000
/**
* Below this a diagram cannot come out whole, so a retry would just produce
* truncated XML instead of the provider's error. Better to surface the error.
*/
const MIN_USABLE_OUTPUT_TOKENS = 1024
/**
* Retry budget when a rejection names the budget parameter but no number we can
* read. It is the default from before 64000, which these providers ran with.
*/
const FALLBACK_OUTPUT_TOKENS = 16000
/** Status codes that can carry a complaint about the requested budget. */
const BUDGET_REJECTION_STATUSES = new Set([400, 422])
function usableLimit(value: number): number | null {
return value >= MIN_USABLE_OUTPUT_TOKENS ? value : null
}
/** Message and body of an error that may be about the budget, or null. */
function rejectionText(error: unknown): string | null {
const err = error as {
message?: unknown
responseBody?: unknown
statusCode?: unknown
}
// An auth or rate-limit failure is not about the budget, so leave it alone.
if (
typeof err?.statusCode === "number" &&
!BUDGET_REJECTION_STATUSES.has(err.statusCode)
) {
return null
}
const text = [
typeof err?.message === "string" ? err.message : "",
typeof err?.responseBody === "string" ? err.responseBody : "",
].join(" ")
return text.trim() ? text : null
}
/**
* A budget this large exceeds what some models accept. Providers reject it with a
* 400 that names the real limit, so we parse the number out and retry once
* instead of failing the turn.
*
* Formats seen in the wild:
* - Bedrock: "The maximum tokens you requested exceeds the model limit of 4096."
* - OpenRouter: "This endpoint's maximum context length is 64000 tokens. However,
* you requested about 64025 tokens (25 of text input, 64000 in the output)."
* Note this one is an input+output ceiling, so the input has to be subtracted.
* vLLM and SGLang send the same kind of ceiling, with the input written as
* "6000 in the messages", "has 6000 input tokens" or "6000 tokens from the input".
* - Anthropic: "max_tokens: 200000 > 64000, which is the maximum allowed..."
* - OpenAI: "This model supports at most 16384 completion tokens"
* - Volcengine Ark: "The parameter `max_tokens` specified in the request are not
* valid: integer above maximum value, expected a value <= 32768, but got 64000"
* - DashScope: "Range of max_tokens should be [1, 8192]"
*
* Every pattern names tokens explicitly. A generic one (an earlier draft matched
* "lower than N") would reinterpret unrelated failures, and retrying on a bogus
* number turns a readable error into an empty diagram.
*/
function readCeiling(text: string): number | null {
// Combined input+output ceiling: subtract the input the provider counted,
// plus a small margin because its estimate is approximate.
const context = text.match(/maximum context length (?:is|of) (\d+)/i)
if (context) {
const input =
text.match(/(\d+) of text input/i) ||
text.match(/(\d+) in the messages/i) ||
text.match(/(\d+) tokens from the input/i) ||
text.match(/(\d+) input tokens/i)
return Number(context[1]) - (input ? Number(input[1]) : 0) - 1024
}
const output =
text.match(/model limit of (\d+)/i) ||
text.match(/> (\d+), which is the maximum/i) ||
text.match(/at most (\d+) completion tokens/i) ||
text.match(/max_\w*tokens.*?expected a value (?:<=|\\u003c=) (\d+)/i) ||
text.match(/Range of max_tokens should be \[1,\s*(\d+)\]/i)
return output ? Number(output[1]) : null
}
/** The usable output ceiling named in a rejection, or null. */
export function parseOutputTokenLimit(error: unknown): number | null {
const text = rejectionText(error)
const ceiling = text ? readCeiling(text) : null
return ceiling === null ? null : usableLimit(ceiling)
}
/**
* Thinking budget the provider adds on top of maxOutputTokens. Bedrock and
* Anthropic send maxOutputTokens + budgetTokens as max_tokens, so a ceiling in
* their rejection covers both.
*/
function thinkingBudget(providerOptions: unknown): number {
const options = providerOptions as
| {
bedrock?: {
reasoningConfig?: { type?: string; budgetTokens?: unknown }
}
anthropic?: {
thinking?: { type?: string; budgetTokens?: unknown }
}
}
| undefined
const config =
options?.bedrock?.reasoningConfig ?? options?.anthropic?.thinking
return config?.type === "enabled" && typeof config.budgetTokens === "number"
? config.budgetTokens
: 0
}
/**
* The budget to retry with after a rejection, or null to surface the error.
*/
export function retryOutputTokens(
error: unknown,
params: { maxOutputTokens?: number; providerOptions?: unknown },
): number | null {
const requested = params.maxOutputTokens
const text = rejectionText(error)
if (!requested || !text) return null
const ceiling = readCeiling(text)
if (ceiling !== null) {
// The ceiling applies to what was actually sent, thinking included,
// so the retry has to leave room for the thinking too.
const thinking = thinkingBudget(params.providerOptions)
if (ceiling >= requested + thinking) return null
return usableLimit(ceiling - thinking)
}
// Names the budget parameter, but in a format we cannot read a number from
if (/max_\w*tokens/i.test(text) && requested > FALLBACK_OUTPUT_TOKENS) {
return FALLBACK_OUTPUT_TOKENS
}
return null
}
/**
* Retry the stream once with a smaller budget when the provider rejects the
* requested one. Without this, raising the default breaks every model whose
* ceiling is below it (measured: bedrock claude-3-haiku 4096, nova-lite 10000,
* openrouter deepseek-r1 64000 shared with the input).
*/
export function withOutputTokenLimitFallback(
model: WrappedModel,
): WrappedModel {
return wrapLanguageModel({
model,
middleware: {
specificationVersion: "v3",
async wrapStream({ doStream, params, model: inner }) {
try {
return await doStream()
} catch (error) {
const retry = retryOutputTokens(error, params)
if (!retry) throw error
console.warn(
`[maxOutputTokens] ${params.maxOutputTokens} rejected, retrying with ${retry}`,
)
return await inner.doStream({
...params,
maxOutputTokens: retry,
})
}
},
},
})
}
function validBudget(value: string | null | undefined): number | null {
const parsed = Number(value)
return Number.isInteger(parsed) &&
parsed > 0 &&
parsed <= MAX_OUTPUT_TOKENS_LIMIT
? parsed
: null
}
/**
* Resolve the output budget: user setting (sent as a header so it works in the
* desktop app too), then server env, then the default. Both sources go through
* the same validation, so a typo in either falls back instead of reaching the
* provider.
*
* On the server's credentials the user setting can only lower the server value,
* so MAX_OUTPUT_TOKENS keeps capping what the server pays for.
*/
export function resolveMaxOutputTokens(
headerValue: string | null,
usesServerCredentials: boolean,
): number {
const header = validBudget(headerValue)
const server =
validBudget(process.env.MAX_OUTPUT_TOKENS) ?? DEFAULT_MAX_OUTPUT_TOKENS
if (header === null) return server
return usesServerCredentials ? Math.min(header, server) : header
}