Files
next-ai-draw-io/lib/ai-providers.ts
T
dayuan.jiang c75f74a5a0 fix: what the third round broke, and the first batch's review
MCP preview after the server lost a session (it expired, or the MCP
process restarted):
- Every server state has an id, made when the state is created. The tab
  notices a new id even when the version numbers happen to match, and
  every push names the state it was based on, so one based on a lost state
  is refused, also when it comes before the tab's first poll (the server
  recovers the saved file first).
- The tab keeps the newest canvas XML, saved or not. When the server knows
  nothing (no file) or exactly what the tab last saved, the canvas wins and
  is saved, so edits made while the server was down are kept. Otherwise
  the server's diagram (an AI write the tab missed, a cleared document
  that was saved) is shown and the tab's copy goes to History.
- Late answers to an old state's push or poll are dropped; a failed push
  says the server is unreachable; Download as .drawio saves the canvas.

Settings and server:
- Saved providers this version does not know stay in storage with their
  keys, and sending no longer trips over them.
- The desktop "Ollama (Local)" preset with a key goes to local Ollama
  again; a server model's Ollama URL variable is read; the admin panel
  writes Ollama Cloud's URL for a key without one.
- Provider error texts show again in the desktop app and for EdgeOne.
- .env: a quoted value followed by a comment ending in a quote is read as
  dotenv reads it; unquoted values are unchanged.
- Desktop app: the next launch opens the port where a chat was last
  saved; a launch elsewhere that saves nothing does not move it, and a
  page with no chats lets the next launch try the other port once.
- The Test button no longer stays busy after another tab changed the key.
- A completed append_diagram is no longer undone by an earlier failed
  edit's preview; a file read once in vain is saved again once it is read
  or gone.

From the first batch's review:
- The admin panel's Test of an entry without a URL now tests the server's
  <P>_BASE_URL, where chat sends the entry's key; chat is unchanged (the
  first fix rerouted working setups).
- The model list ends downloads that are too large, accepts answers
  without a body, and keeps the "redirects are not allowed" explanation.
- A test covers the preview's History rendering.
2026-10-05 17:33:36 +09:00

1207 lines
46 KiB
TypeScript

import { createAmazonBedrock } from "@ai-sdk/amazon-bedrock"
import { createAnthropic } from "@ai-sdk/anthropic"
import { createAzure } from "@ai-sdk/azure"
import { createDeepSeek } from "@ai-sdk/deepseek"
import { createGoogleGenerativeAI } from "@ai-sdk/google"
import { createVertex } from "@ai-sdk/google-vertex"
import { createOpenAI } from "@ai-sdk/openai"
import { createOpenAICompatible } from "@ai-sdk/openai-compatible"
import { createAihubmix } from "@aihubmix/ai-sdk-provider"
import { fromNodeProviderChain } from "@aws-sdk/credential-providers"
import { createOpenRouter } from "@openrouter/ai-sdk-provider"
import {
createGateway,
defaultSettingsMiddleware,
extractReasoningMiddleware,
type LanguageModel,
wrapLanguageModel,
} from "ai"
import { createOllama } from "ollama-ai-provider-v2"
import {
adminProvidersToConfig,
loadAdminProviders,
} from "@/lib/admin/providers"
import { redirectGuardedFetch } from "@/lib/ssrf-protection"
import {
normalizeBaseUrl,
PROVIDER_INFO,
type ProviderName,
} from "@/lib/types/model-config"
export type { ProviderName }
export const AIHUBMIX_APP_CODE = "MSBS9675"
interface ModelConfig {
model: any
providerOptions?: any
modelId: string
provider: ProviderName
}
// Providers that only support a single system message
export const SINGLE_SYSTEM_PROVIDERS = new Set<ProviderName>([
"minimax",
"glm",
"qwen",
"kimi",
"qiniu",
"novita",
"mimo",
])
/**
* Normalize MiniMax base URL for AI SDK compatibility.
* MiniMax supports Anthropic-compatible and OpenAI-compatible endpoints.
*/
export function normalizeMiniMaxBaseURL(rawUrl: string): {
baseURL: string
isAnthropicCompatible: boolean
} {
const isAnthropicCompatible = rawUrl.includes("/anthropic")
let baseURL = rawUrl.replace(/\/$/, "")
if (isAnthropicCompatible) {
if (!baseURL.endsWith("/anthropic/v1")) {
if (baseURL.endsWith("/anthropic")) {
baseURL = `${baseURL}/v1`
} else {
baseURL = `${baseURL}/anthropic/v1`
}
}
} else {
if (!baseURL.endsWith("/v1")) {
baseURL = `${baseURL}/v1`
}
}
return { baseURL, isAnthropicCompatible }
}
export function isAihubmixStandardBaseURL(
rawUrl: string | null | undefined,
): boolean {
if (!rawUrl) return true
const baseURL = rawUrl.replace(/\/+$/, "")
return (
baseURL === "https://aihubmix.com" ||
baseURL === "https://aihubmix.com/v1"
)
}
export interface ClientOverrides {
provider?: string | null
baseUrl?: string | null
apiKey?: string | null
modelId?: string | null
// AWS Bedrock credentials
awsAccessKeyId?: string | null
awsSecretAccessKey?: string | null
awsRegion?: string | null
awsSessionToken?: string | null
// Vertex AI config
vertexApiKey?: string | null // Express Mode API key
// Custom headers (e.g., for EdgeOne cookie auth)
headers?: Record<string, string>
// Custom env var name(s) for server models
// Can be a single string or array of strings for load balancing
apiKeyEnv?: string | string[]
baseUrlEnv?: string
}
// Bedrock provider options for Anthropic beta features
const BEDROCK_ANTHROPIC_BETA = {
bedrock: {
anthropicBeta: ["fine-grained-tool-streaming-2025-05-14"],
},
}
/**
* Resolve baseURL based on whether user is providing their own API key.
* When user provides their own API key, we should NOT fall back to server's
* baseURL environment variable - user credentials should only be sent to
* user-specified endpoints or official provider endpoints.
*
* @param userApiKey - User-provided API key (if any)
* @param userBaseUrl - User-provided base URL (if any)
* @param serverBaseUrl - Server's base URL from environment variable
* @param defaultBaseUrl - Provider's official/default base URL (optional)
* @returns The resolved base URL to use
*/
export function resolveBaseURL(
userApiKey: string | null | undefined,
userBaseUrl: string | null | undefined,
serverBaseUrl: string | undefined,
defaultBaseUrl?: string,
): string | undefined {
if (userApiKey) {
// User provides their own API key - only use user's baseUrl or default
return userBaseUrl || defaultBaseUrl || undefined
}
// No user API key - fall back to server config
return userBaseUrl || serverBaseUrl || defaultBaseUrl || undefined
}
/**
* Resolve API key from custom env var name or default env var.
* Supports multiple API keys per provider via ai-models.json apiKeyEnv config.
* When multiple keys are configured, randomly selects one for load balancing.
*
* Priority:
* 1. User-provided API key (overrides.apiKey)
* 2. Custom env var(s) from ai-models.json (overrides.apiKeyEnv)
* - If array, randomly picks one with a valid value
* 3. Default provider env var (defaultEnvVar)
*/
function resolveApiKey(
overrides: ClientOverrides | undefined,
defaultEnvVar: string,
): string | undefined {
if (overrides?.apiKey) return overrides.apiKey
if (overrides?.apiKeyEnv) {
// Handle array of env var names - randomly select one
if (Array.isArray(overrides.apiKeyEnv)) {
// Filter to only env vars that have values
const validEnvVars = overrides.apiKeyEnv.filter(
(envVar) => process.env[envVar],
)
if (validEnvVars.length > 0) {
// Randomly select one
const selectedEnvVar =
validEnvVars[
Math.floor(Math.random() * validEnvVars.length)
]
console.log(
`[API Key Routing] Selected ${selectedEnvVar} from ${validEnvVars.length} available keys`,
)
return process.env[selectedEnvVar]
}
} else {
return process.env[overrides.apiKeyEnv]
}
}
return process.env[defaultEnvVar]
}
/**
* Resolve base URL from custom env var name or default env var.
* Supports multiple base URLs per provider via ai-models.json baseUrlEnv config.
*/
function resolveBaseUrlEnv(
overrides: ClientOverrides | undefined,
defaultEnvVar: string,
): string | undefined {
if (overrides?.baseUrlEnv) return process.env[overrides.baseUrlEnv]
return process.env[defaultEnvVar]
}
/**
* Safely parse integer from environment variable with validation
*/
function parseIntSafe(
value: string | undefined,
varName: string,
min?: number,
max?: number,
): number | undefined {
if (!value) return undefined
const parsed = Number.parseInt(value, 10)
if (Number.isNaN(parsed)) {
throw new Error(`${varName} must be a valid integer, got: ${value}`)
}
if (min !== undefined && parsed < min) {
throw new Error(`${varName} must be >= ${min}, got: ${parsed}`)
}
if (max !== undefined && parsed > max) {
throw new Error(`${varName} must be <= ${max}, got: ${parsed}`)
}
return parsed
}
/**
* GOOGLE_TOP_K and GOOGLE_TOP_P. They are call settings, so they go on the
* model through a middleware: as Google provider options they were dropped.
*/
function googleSamplingSettings(): { topK?: number; topP?: number } {
const settings: { topK?: number; topP?: number } = {}
const topK = parseIntSafe(process.env.GOOGLE_TOP_K, "GOOGLE_TOP_K", 1, 100)
if (topK) settings.topK = topK
if (process.env.GOOGLE_TOP_P) {
const topP = Number.parseFloat(process.env.GOOGLE_TOP_P)
if (Number.isNaN(topP) || topP < 0 || topP > 1) {
throw new Error(
`GOOGLE_TOP_P must be a number between 0 and 1, got: ${process.env.GOOGLE_TOP_P}`,
)
}
settings.topP = topP
}
return settings
}
/**
* Build provider-specific options from environment variables
* Supports various AI SDK providers with their unique configuration options
*
* Environment variables:
* - OPENAI_REASONING_EFFORT: OpenAI reasoning effort level (minimal/low/medium/high) - for the o-series and gpt-5 or later
* - OPENAI_REASONING_SUMMARY: OpenAI reasoning summary (auto/detailed) - auto-enabled for the o-series and gpt-5 or later
* - ANTHROPIC_THINKING_BUDGET_TOKENS: Anthropic thinking budget in tokens (1024-64000)
* - ANTHROPIC_THINKING_TYPE: Anthropic thinking type (enabled)
* - GOOGLE_THINKING_BUDGET: Google Gemini 2.5 thinking budget in tokens (1024-100000)
* - GOOGLE_THINKING_LEVEL: Google Gemini 3 thinking level (low/high)
* - GOOGLE_VERTEX_THINKING_BUDGET: Vertex AI Gemini 2.5 thinking budget in tokens (1024-100000)
* - GOOGLE_VERTEX_THINKING_LEVEL: Vertex AI Gemini 3 thinking level (low/high)
* - AZURE_REASONING_EFFORT: Azure/OpenAI reasoning effort (low/medium/high)
* - AZURE_REASONING_SUMMARY: Azure reasoning summary (none/brief/detailed)
* - BEDROCK_REASONING_BUDGET_TOKENS: Bedrock Claude reasoning budget in tokens (1024-64000)
* - BEDROCK_REASONING_EFFORT: Bedrock Nova reasoning effort (low/medium/high)
* - OLLAMA_ENABLE_THINKING: Enable Ollama thinking mode (set to "true")
*/
function buildProviderOptions(
provider: ProviderName,
modelId?: string,
): Record<string, any> | undefined {
const options: Record<string, any> = {}
switch (provider) {
case "openai": {
const reasoningEffort = process.env.OPENAI_REASONING_EFFORT
const reasoningSummary = process.env.OPENAI_REASONING_SUMMARY
// Reasoning models (the o-series, gpt-5 and later) need
// reasoningSummary to return thoughts
if (modelId && /^(o\d|gpt-([5-9]|[1-9]\d))/.test(modelId)) {
options.openai = {
// Auto-enable reasoning summary for reasoning models
// Use 'auto' as default since not all models support 'detailed'
reasoningSummary:
(reasoningSummary as "auto" | "detailed") || "auto",
}
// Optionally configure reasoning effort
if (reasoningEffort) {
options.openai.reasoningEffort = reasoningEffort as
| "minimal"
| "low"
| "medium"
| "high"
}
} else if (reasoningEffort || reasoningSummary) {
// Non-reasoning models: only apply if explicitly configured
options.openai = {}
if (reasoningEffort) {
options.openai.reasoningEffort = reasoningEffort as
| "minimal"
| "low"
| "medium"
| "high"
}
if (reasoningSummary) {
options.openai.reasoningSummary = reasoningSummary as
| "auto"
| "detailed"
}
}
break
}
case "anthropic": {
const thinkingBudget = parseIntSafe(
process.env.ANTHROPIC_THINKING_BUDGET_TOKENS,
"ANTHROPIC_THINKING_BUDGET_TOKENS",
1024,
64000,
)
const thinkingType =
process.env.ANTHROPIC_THINKING_TYPE || "enabled"
if (thinkingBudget) {
options.anthropic = {
thinking: {
type: thinkingType,
budgetTokens: thinkingBudget,
},
}
}
break
}
case "google": {
const thinkingBudgetVal = parseIntSafe(
process.env.GOOGLE_THINKING_BUDGET,
"GOOGLE_THINKING_BUDGET",
1024,
100000,
)
const thinkingLevel = process.env.GOOGLE_THINKING_LEVEL
// Google Gemini 2.5/3 models think by default, but need includeThoughts: true
// to return the reasoning in the response
if (
modelId &&
(modelId.includes("gemini-2") ||
modelId.includes("gemini-3") ||
modelId.includes("gemini2") ||
modelId.includes("gemini3"))
) {
const thinkingConfig: Record<string, any> = {
includeThoughts: true,
}
// Optionally configure thinking budget or level
if (
thinkingBudgetVal &&
(modelId.includes("2.5") || modelId.includes("2-5"))
) {
thinkingConfig.thinkingBudget = thinkingBudgetVal
} else if (
thinkingLevel &&
(modelId.includes("gemini-3") ||
modelId.includes("gemini3"))
) {
thinkingConfig.thinkingLevel = thinkingLevel as
| "low"
| "high"
}
options.google = { thinkingConfig }
}
break
}
case "vertexai": {
const thinkingBudget = parseIntSafe(
process.env.GOOGLE_VERTEX_THINKING_BUDGET,
"GOOGLE_VERTEX_THINKING_BUDGET",
1024,
100000,
)
const thinkingLevel = process.env.GOOGLE_VERTEX_THINKING_LEVEL
if (
modelId &&
(modelId.includes("gemini-2") ||
modelId.includes("gemini-3") ||
modelId.includes("gemini2") ||
modelId.includes("gemini3"))
) {
const thinkingConfig: Record<string, any> = {
includeThoughts: true,
}
const isGemini3 =
modelId?.includes("gemini-3") ||
modelId?.includes("gemini3")
const isGemini25 =
modelId?.includes("2.5") || modelId?.includes("2-5")
if (isGemini3 && thinkingLevel) {
// Vertex AI provider in AI SDK supports more granular levels (minimal/low/medium/high)
thinkingConfig.thinkingLevel = thinkingLevel as
| "minimal"
| "low"
| "medium"
| "high"
} else if (isGemini25 && thinkingBudget) {
thinkingConfig.thinkingBudget = thinkingBudget
}
options.google = { thinkingConfig }
}
break
}
case "azure": {
const reasoningEffort = process.env.AZURE_REASONING_EFFORT
const reasoningSummary = process.env.AZURE_REASONING_SUMMARY
if (reasoningEffort || reasoningSummary) {
options.azure = {}
if (reasoningEffort) {
options.azure.reasoningEffort = reasoningEffort as
| "low"
| "medium"
| "high"
}
if (reasoningSummary) {
options.azure.reasoningSummary = reasoningSummary as
| "none"
| "brief"
| "detailed"
}
}
break
}
case "bedrock": {
const budgetTokens = parseIntSafe(
process.env.BEDROCK_REASONING_BUDGET_TOKENS,
"BEDROCK_REASONING_BUDGET_TOKENS",
1024,
64000,
)
const reasoningEffort = process.env.BEDROCK_REASONING_EFFORT
// Bedrock reasoning ONLY for Claude and Nova models
// Other models (MiniMax, etc.) don't support reasoningConfig
if (
modelId &&
(budgetTokens || reasoningEffort) &&
(modelId.includes("claude") ||
modelId.includes("anthropic") ||
modelId.includes("nova") ||
modelId.includes("amazon"))
) {
const reasoningConfig: Record<string, any> = { type: "enabled" }
// Claude models: use budgetTokens (1024-64000)
if (
budgetTokens &&
(modelId.includes("claude") ||
modelId.includes("anthropic"))
) {
reasoningConfig.budgetTokens = budgetTokens
}
// Nova models: use maxReasoningEffort (low/medium/high)
else if (
reasoningEffort &&
(modelId.includes("nova") || modelId.includes("amazon"))
) {
reasoningConfig.maxReasoningEffort = reasoningEffort as
| "low"
| "medium"
| "high"
}
options.bedrock = { reasoningConfig }
}
break
}
case "ollama": {
const enableThinking = process.env.OLLAMA_ENABLE_THINKING
// Ollama supports reasoning with think: true for models like qwen3
if (enableThinking === "true") {
options.ollama = { think: true }
}
break
}
default:
break
}
return Object.keys(options).length > 0 ? options : undefined
}
// Map of provider to required environment variable
export const PROVIDER_ENV_VARS: Record<ProviderName, string | null> = {
bedrock: null, // AWS SDK auto-uses IAM role on AWS, or env vars locally
openai: "OPENAI_API_KEY",
anthropic: "ANTHROPIC_API_KEY",
google: "GOOGLE_GENERATIVE_AI_API_KEY",
vertexai: "GOOGLE_VERTEX_API_KEY",
azure: "AZURE_API_KEY",
ollama: null, // No credentials needed for local Ollama
openrouter: "OPENROUTER_API_KEY",
aihubmix: "AIHUBMIX_API_KEY",
deepseek: "DEEPSEEK_API_KEY",
siliconflow: "SILICONFLOW_API_KEY",
sglang: "SGLANG_API_KEY",
gateway: "AI_GATEWAY_API_KEY",
edgeone: null, // No credentials needed - uses EdgeOne Edge AI
doubao: "DOUBAO_API_KEY",
modelscope: "MODELSCOPE_API_KEY",
glm: "GLM_API_KEY",
qwen: "QWEN_API_KEY",
qiniu: "QINIU_API_KEY",
kimi: "KIMI_API_KEY",
minimax: "MINIMAX_API_KEY",
novita: "NOVITA_API_KEY",
mimo: "MIMO_API_KEY",
atlascloud: "ATLASCLOUD_API_KEY",
}
/**
* Auto-detect provider based on available API keys
* Returns the provider if exactly one is configured, otherwise null
*/
function detectProvider(): ProviderName | null {
const configuredProviders: ProviderName[] = []
for (const [provider, envVar] of Object.entries(PROVIDER_ENV_VARS)) {
if (envVar === null) {
// Skip ollama - it doesn't require credentials
continue
}
// Anthropic accepts ANTHROPIC_AUTH_TOKEN (Bearer auth) as alternative to ANTHROPIC_API_KEY
const hasCredential =
provider === "anthropic"
? !!(
process.env.ANTHROPIC_API_KEY ||
process.env.ANTHROPIC_AUTH_TOKEN
)
: !!process.env[envVar]
if (hasCredential) {
// Azure requires additional config (baseURL or resourceName)
if (provider === "azure") {
const hasBaseUrl = !!process.env.AZURE_BASE_URL
const hasResourceName = !!process.env.AZURE_RESOURCE_NAME
if (hasBaseUrl || hasResourceName) {
configuredProviders.push(provider as ProviderName)
}
} else {
configuredProviders.push(provider as ProviderName)
}
}
}
if (configuredProviders.length === 1) {
return configuredProviders[0]
}
return null
}
/**
* Validate that required API keys are present for the selected provider
* @param provider - The provider to validate
* @param customApiKeyEnv - Optional custom env var name(s) (from ai-models.json apiKeyEnv)
*/
function validateProviderCredentials(
provider: ProviderName,
customApiKeyEnv?: string | string[],
): void {
// Handle array of env var names - at least one must be set
if (Array.isArray(customApiKeyEnv)) {
const hasAnyKey = customApiKeyEnv.some((envVar) => process.env[envVar])
if (!hasAnyKey) {
throw new Error(
`At least one of [${customApiKeyEnv.join(", ")}] environment variables is required for ${provider} provider. ` +
`Please set at least one in your .env.local file.`,
)
}
return
}
// Anthropic accepts ANTHROPIC_AUTH_TOKEN (Bearer auth) as alternative to ANTHROPIC_API_KEY
if (provider === "anthropic" && !customApiKeyEnv) {
const hasCredential = !!(
process.env.ANTHROPIC_API_KEY || process.env.ANTHROPIC_AUTH_TOKEN
)
if (!hasCredential) {
throw new Error(
`Either ANTHROPIC_API_KEY or ANTHROPIC_AUTH_TOKEN environment variable is required for anthropic provider. ` +
`Please set one in your .env.local file.`,
)
}
} else {
// Use custom env var name if provided, otherwise use default
const requiredVar = customApiKeyEnv || PROVIDER_ENV_VARS[provider]
if (requiredVar && !process.env[requiredVar]) {
throw new Error(
`${requiredVar} environment variable is required for ${provider} provider. ` +
`Please set it in your .env.local file.`,
)
}
}
// Azure requires either AZURE_BASE_URL or AZURE_RESOURCE_NAME in addition to API key
if (provider === "azure") {
const hasBaseUrl = !!process.env.AZURE_BASE_URL
const hasResourceName = !!process.env.AZURE_RESOURCE_NAME
if (!hasBaseUrl && !hasResourceName) {
throw new Error(
`Azure requires either AZURE_BASE_URL or AZURE_RESOURCE_NAME to be set. ` +
`Please set one in your .env.local file.`,
)
}
}
}
/** AWS's Bedrock endpoint for a region, as the Bedrock SDK builds it */
function bedrockRuntimeUrl(region: string): string {
const suffix =
[
["cn-", "amazonaws.com.cn"],
["us-iso-", "c2s.ic.gov"],
["us-isob-", "sc2s.sgov.gov"],
["eu-isoe-", "cloud.adc-e.uk"],
["us-isof-", "csp.hci.ic.gov"],
["eusc-", "amazonaws.eu"],
].find(([prefix]) => region.startsWith(prefix))?.[1] ?? "amazonaws.com"
return `https://bedrock-runtime.${region}.${suffix}`
}
/**
* Providers whose SDK has the official endpoint built in. The others are
* OpenAI-compatible APIs (or Anthropic) that are called at
* PROVIDER_INFO.defaultBaseUrl unless a base URL is configured.
*/
const SDK_KNOWS_ENDPOINT = new Set<ProviderName>([
"openai",
"google",
"azure",
"openrouter",
"gateway",
"deepseek",
])
/** Where and how to call a provider, once credentials are resolved */
interface Endpoint {
apiKey?: string
baseURL?: string
headers?: Record<string, string>
fetch?: typeof fetch
authToken?: string // Anthropic Bearer auth
resourceName?: string // Azure
// baseURL comes from the settings or env, not the provider's default
configuredBaseURL?: boolean
}
/**
* An OpenAI-compatible chat model. includeUsage asks for token usage in the
* stream, which quota tracking needs. Some of these models write their
* reasoning inside <think> tags; that text becomes reasoning, not reply.
*/
function compatibleModel(
provider: ProviderName,
modelId: string,
e: Endpoint,
): LanguageModel {
const model = createOpenAICompatible({
name: provider,
apiKey: e.apiKey,
baseURL: e.baseURL ?? "",
...(e.headers && { headers: e.headers }),
...(e.fetch && { fetch: e.fetch }),
includeUsage: true,
})(modelId)
return wrapLanguageModel({
model,
middleware: extractReasoningMiddleware({ tagName: "think" }),
})
}
/** Create the model for a provider. Credentials are already resolved. */
function createModel(
provider: ProviderName,
modelId: string,
e: Endpoint,
): LanguageModel {
const opts = {
apiKey: e.apiKey,
...(e.baseURL && { baseURL: e.baseURL }),
...(e.fetch && { fetch: e.fetch }),
}
switch (provider) {
case "openai": {
const openaiProvider = createOpenAI(opts)
// A configured base URL is usually a proxy that only has Chat
// Completions; without one the Responses API is used, which
// returns reasoning for the o-series and gpt-5 or later
return e.configuredBaseURL
? openaiProvider.chat(modelId)
: openaiProvider(modelId)
}
case "anthropic":
// The provider streams tool input per tool (eager_input_streaming),
// which replaced the fine-grained-tool-streaming beta header
return createAnthropic({
...(e.authToken
? { authToken: e.authToken }
: { apiKey: e.apiKey }),
baseURL: e.baseURL,
...(e.fetch && { fetch: e.fetch }),
})(modelId)
case "google": {
const model = createGoogleGenerativeAI(opts)(modelId)
const sampling = googleSamplingSettings()
return Object.keys(sampling).length > 0
? wrapLanguageModel({
model,
middleware: defaultSettingsMiddleware({
settings: sampling,
}),
})
: model
}
case "azure":
// baseURL takes precedence over resourceName per SDK behavior
return createAzure({
...opts,
...(!e.baseURL &&
e.resourceName && { resourceName: e.resourceName }),
})(modelId)
case "openrouter":
return createOpenRouter(opts)(modelId)
case "gateway":
// Without a key or URL the SDK uses Vercel's endpoint and OIDC
return createGateway(opts)(modelId)
case "deepseek":
case "kimi":
case "mimo":
// Kimi and MiMo return reasoning_content like DeepSeek and need it
// passed back in multi-turn tool calls (MiMo answers 400 otherwise)
return createDeepSeek(opts)(modelId)
case "doubao": {
// DeepSeek and Kimi models on Doubao use reasoning_content too
const lower = modelId.toLowerCase()
return lower.includes("deepseek") || lower.includes("kimi")
? createDeepSeek(opts)(modelId)
: compatibleModel(provider, modelId, e)
}
case "aihubmix":
return isAihubmixStandardBaseURL(e.baseURL)
? createAihubmix({
apiKey: e.apiKey,
appCode: AIHUBMIX_APP_CODE,
})(modelId)
: compatibleModel(provider, modelId, e)
case "minimax": {
const { baseURL, isAnthropicCompatible } = normalizeMiniMaxBaseURL(
e.baseURL as string,
)
return isAnthropicCompatible
? createAnthropic({
apiKey: e.apiKey,
baseURL,
...(e.fetch && { fetch: e.fetch }),
})(modelId)
: compatibleModel(provider, modelId, { ...e, baseURL })
}
default:
// siliconflow, sglang, modelscope, glm, qwen, qiniu, novita,
// atlascloud, edgeone
return compatibleModel(provider, modelId, e)
}
}
/**
* Get the AI model for a chat request: the client's own provider and
* credentials, or the server's (AI_PROVIDER, AI_MODEL and each provider's
* <NAME>_API_KEY / <NAME>_BASE_URL, see env.example). The settings test
* button uses the same function, so a passing test means the chat works.
*/
export function getAIModel(clientOverrides?: ClientOverrides): ModelConfig {
// Drop an endpoint path pasted along with the client's base URL
const overrides = clientOverrides?.baseUrl
? {
...clientOverrides,
baseUrl: normalizeBaseUrl(clientOverrides.baseUrl),
}
: clientOverrides
// SECURITY: Prevent SSRF attacks (GHSA-9qf7-mprq-9qgm)
// If a custom baseUrl is provided, an API key MUST also be provided.
// This prevents attackers from redirecting server API keys to malicious endpoints.
// Exception: EdgeOne doesn't require API keys.
// Ollama is exempt only when no server OLLAMA_API_KEY is configured;
// when it IS configured, the outer guard also enforces client apiKey for custom baseUrls.
if (
overrides?.baseUrl &&
!overrides?.apiKey &&
!(overrides?.provider === "vertexai" && overrides?.vertexApiKey) &&
overrides?.provider !== "edgeone" &&
!(overrides?.provider === "ollama" && !process.env.OLLAMA_API_KEY)
) {
throw new Error(
`API key is required when using a custom base URL. ` +
`Please provide your own API key in Settings.`,
)
}
// Check if client is providing their own provider override
const isClientOverride = !!(
overrides?.provider &&
(overrides?.apiKey ||
(overrides?.provider === "vertexai" && overrides?.vertexApiKey))
)
// Use client override if provided, otherwise fall back to env vars.
// AI_MODEL may be comma-separated (multi-model fallback); pick the first.
const envModel = process.env.AI_MODEL?.split(",")[0]?.trim() || undefined
const modelId = overrides?.modelId || envModel
if (!modelId) {
if (isClientOverride) {
throw new Error(
`Model ID is required when using custom AI provider. Please specify a model in Settings.`,
)
}
throw new Error(
`AI_MODEL environment variable is required. Example: AI_MODEL=claude-sonnet-4-5`,
)
}
// Determine provider: client override > explicit config > auto-detect > error
let provider: ProviderName
if (overrides?.provider) {
// Validate client-provided provider
if (!Object.hasOwn(PROVIDER_INFO, overrides.provider)) {
throw new Error(
`Invalid provider: ${overrides.provider}. Allowed providers: ${Object.keys(PROVIDER_INFO).join(", ")}`,
)
}
provider = overrides.provider as ProviderName
} else if (process.env.AI_PROVIDER) {
provider = process.env.AI_PROVIDER as ProviderName
} else {
const detected = detectProvider()
if (detected) {
provider = detected
console.log(`[AI Provider] Auto-detected provider: ${provider}`)
} else {
// List configured providers for better error message
const configured = Object.entries(PROVIDER_ENV_VARS)
.filter(([, envVar]) => envVar && process.env[envVar as string])
.map(([p]) => p)
if (configured.length === 0) {
const keys = Object.entries(PROVIDER_ENV_VARS)
.filter(([, envVar]) => envVar)
.map(([p, envVar]) => `- ${envVar} for ${p}`)
throw new Error(
`No AI provider configured. Please set one of the following API keys in your .env.local file:\n` +
`${keys.join("\n")}\n` +
`- AWS_ACCESS_KEY_ID for bedrock\n` +
`Or set AI_PROVIDER=ollama for local Ollama.`,
)
}
throw new Error(
`Multiple AI providers configured (${configured.join(", ")}). ` +
`Please set AI_PROVIDER to specify which one to use.`,
)
}
}
if (!Object.hasOwn(PROVIDER_INFO, provider)) {
throw new Error(
`Unknown AI provider: ${provider}. Supported providers: ${Object.keys(PROVIDER_INFO).join(", ")}`,
)
}
// Only validate server credentials if client isn't providing their own API key
if (!isClientOverride) {
validateProviderCredentials(provider, overrides?.apiKeyEnv)
}
console.log(`[AI Provider] Initializing ${provider} with model: ${modelId}`)
// Requests to a base URL the client chose must not follow redirects
const guardedFetch = overrides?.baseUrl ? redirectGuardedFetch() : undefined
// Build provider-specific options from environment variables
let providerOptions = buildProviderOptions(provider, modelId)
let model: LanguageModel
switch (provider) {
case "bedrock": {
// Use client-provided credentials if available, otherwise fall back to IAM/env vars
const hasClientCredentials =
overrides?.awsAccessKeyId && overrides?.awsSecretAccessKey
// Keys from the admin panel. The ADMIN_ names keep them out of the
// default AWS credential chain, which other clients such as the
// DynamoDB quota manager use with their own credentials.
const adminAccessKeyId = process.env.ADMIN_AWS_ACCESS_KEY_ID
const adminSecretAccessKey = process.env.ADMIN_AWS_SECRET_ACCESS_KEY
// The region becomes part of the endpoint's host name, so a
// request's region must be a region name, or it could send the
// server's credentials to another host
if (
overrides?.awsRegion &&
!/^[a-z]{2,4}(-[a-z]+)+-\d{1,2}$/.test(overrides.awsRegion)
) {
throw Object.assign(
new Error(`Invalid AWS region "${overrides.awsRegion}"`),
{ statusCode: 400 },
)
}
const bedrockRegion =
overrides?.awsRegion ||
process.env.ADMIN_AWS_REGION ||
process.env.AWS_REGION ||
"us-west-2"
const bedrockProvider = hasClientCredentials
? createAmazonBedrock({
region: bedrockRegion,
accessKeyId: overrides.awsAccessKeyId as string,
secretAccessKey: overrides.awsSecretAccessKey as string,
...(overrides?.awsSessionToken && {
sessionToken: overrides.awsSessionToken,
}),
// Without an apiKey the SDK reads the server's
// AWS_BEARER_TOKEN_BEDROCK, which wins over the keys
apiKey: "",
// Without a baseURL it reads the server's
// AWS_ENDPOINT_URL_BEDROCK_RUNTIME / AWS_ENDPOINT_URL
baseURL: bedrockRuntimeUrl(bedrockRegion),
})
: adminAccessKeyId && adminSecretAccessKey
? createAmazonBedrock({
region: bedrockRegion,
accessKeyId: adminAccessKeyId,
secretAccessKey: adminSecretAccessKey,
// The keys the admin panel's Test button checked
apiKey: "",
})
: createAmazonBedrock({
region: bedrockRegion,
credentialProvider: fromNodeProviderChain(),
})
model = bedrockProvider(modelId)
// Add Anthropic beta options if using Claude models via Bedrock
if (modelId.includes("anthropic.claude")) {
// Deep merge to preserve both anthropicBeta and reasoningConfig
providerOptions = {
bedrock: {
...BEDROCK_ANTHROPIC_BETA.bedrock,
...(providerOptions?.bedrock || {}),
},
}
}
break
}
case "vertexai": {
// Express Mode: Use API key for authentication
// SECURITY: a client base URL only ever gets the client's key, so the
// server's GOOGLE_VERTEX_API_KEY is never sent to a client-chosen host
const vertexApiKey = overrides?.baseUrl
? overrides.vertexApiKey
: overrides?.vertexApiKey || process.env.GOOGLE_VERTEX_API_KEY
if (!vertexApiKey) {
throw new Error(
"Vertex AI requires an API key for Express Mode. " +
"Get one from Google Cloud Console or set GOOGLE_VERTEX_API_KEY environment variable.",
)
}
// Support custom base URL from env or client override.
// A client key only goes to the client's URL or the official one.
const baseURL = resolveBaseURL(
overrides?.vertexApiKey,
overrides?.baseUrl,
process.env.GOOGLE_VERTEX_BASE_URL,
)
model = createVertex({
apiKey: vertexApiKey,
...(baseURL && { baseURL }),
...(guardedFetch && { fetch: guardedFetch }),
})(modelId)
break
}
case "ollama": {
// SECURITY: When client provides a custom base URL, only use
// client-provided API key. Never fall back to server OLLAMA_API_KEY
// to prevent leaking server credentials to user-controlled endpoints.
const apiKey = overrides?.baseUrl
? overrides?.apiKey || undefined
: resolveApiKey(overrides, "OLLAMA_API_KEY")
// Like other providers, a user's key never goes to the server's
// base URL: without a URL of their own it goes to Ollama Cloud.
// The server's key goes to OLLAMA_BASE_URL (or a server model's
// own variable), else to the SDK's local default: the desktop
// app's "Ollama (Local)" preset puts its key field there too.
const baseURL =
overrides?.baseUrl ||
(overrides?.apiKey
? PROVIDER_INFO.ollama.defaultBaseUrl
: resolveBaseUrlEnv(overrides, "OLLAMA_BASE_URL"))
model = createOllama({
...(baseURL && { baseURL }),
...(apiKey && {
headers: { Authorization: `Bearer ${apiKey}` },
}),
...(guardedFetch && { fetch: guardedFetch }),
})(modelId)
break
}
case "edgeone":
// EdgeOne Pages Edge AI, an OpenAI-compatible API without a key.
// The SDK appends /chat/completions to the base URL. Cookies
// (eo_token, eo_time) and the access code authenticate the call.
model = compatibleModel(provider, modelId, {
apiKey: "edgeone",
baseURL: overrides?.baseUrl || "/api/edgeai",
headers: overrides?.headers,
fetch: guardedFetch,
})
break
default: {
// Every other provider takes an API key and a base URL from
// <NAME>_API_KEY / <NAME>_BASE_URL (or a server model's apiKeyEnv)
const apiKey = resolveApiKey(
overrides,
PROVIDER_ENV_VARS[provider] as string,
)
const baseUrlEnv =
provider === "gateway"
? "AI_GATEWAY_BASE_URL"
: `${provider.toUpperCase()}_BASE_URL`
// A local default (SGLang's 127.0.0.1) only fills the settings
// form; the server must not call its own machine for it. With a
// user's key the OpenAI SDK would read the server's
// OPENAI_BASE_URL, so name the official endpoint.
const defaultUrl = PROVIDER_INFO[provider].defaultBaseUrl
const publicDefault = defaultUrl?.startsWith("https://")
? defaultUrl
: undefined
const configuredBaseURL = resolveBaseURL(
overrides?.apiKey,
overrides?.baseUrl,
resolveBaseUrlEnv(overrides, baseUrlEnv),
)
const baseURL =
configuredBaseURL ||
(SDK_KNOWS_ENDPOINT.has(provider) &&
!(provider === "openai" && overrides?.apiKey)
? undefined
: publicDefault)
// With a user's Azure key the SDK would read the server's
// AZURE_RESOURCE_NAME
if (
!baseURL &&
(!SDK_KNOWS_ENDPOINT.has(provider) ||
(provider === "azure" && overrides?.apiKey))
) {
throw new Error(
`${PROVIDER_INFO[provider].label} needs a base URL. Add it in the model settings.`,
)
}
model = createModel(provider, modelId, {
apiKey,
baseURL,
configuredBaseURL: !!configuredBaseURL,
fetch: guardedFetch,
// Bearer auth for Anthropic when there is no API key
authToken:
provider === "anthropic" && !apiKey
? process.env.ANTHROPIC_AUTH_TOKEN
: undefined,
// Only the server's own resource; a client key needs its URL
resourceName:
provider === "azure" && !overrides?.apiKey
? process.env.AZURE_RESOURCE_NAME
: undefined,
})
}
}
return { model, providerOptions, modelId, provider }
}
/**
* The server's <P>_BASE_URL for a provider, which getAIModel uses for a
* server model without a URL variable of its own (an admin panel entry
* without a URL). Bedrock, EdgeOne and Ollama (the panel writes
* OLLAMA_BASE_URL itself) have none.
*/
export function globalBaseUrl(provider: ProviderName): string | undefined {
if (["bedrock", "edgeone", "ollama"].includes(provider)) return undefined
const name =
provider === "vertexai"
? "GOOGLE_VERTEX_BASE_URL"
: provider === "gateway"
? "AI_GATEWAY_BASE_URL"
: `${provider.toUpperCase()}_BASE_URL`
return process.env[name] || undefined
}
/** The provider of the server's own config: AI_PROVIDER, or the one with a key */
export function getServerProvider(): ProviderName | null {
return (process.env.AI_PROVIDER as ProviderName) || detectProvider()
}
/**
* Whether the call is paid for by the server's own credentials (env keys or
* IAM role) rather than credentials sent with the request. Mirrors which key
* each branch of getAIModel ends up using.
*/
export function usesServerCredentials(
provider: ProviderName,
overrides?: ClientOverrides,
): boolean {
// The desktop app's local server holds the user's own preset keys
if (process.env.NEXT_AI_DRAWIO_DESKTOP === "1") return false
// Cleaned like getAIModel does: "/" means no base URL
const baseUrl = normalizeBaseUrl(overrides?.baseUrl ?? "")
switch (provider) {
case "bedrock":
return !(overrides?.awsAccessKeyId && overrides?.awsSecretAccessKey)
case "vertexai":
return !overrides?.vertexApiKey
case "edgeone":
// The platform's own endpoint, no key involved
return false
case "ollama":
// Only a server key costs money; a keyless local server or the
// client's own server does not
return (
!baseUrl &&
!overrides?.apiKey &&
!!(overrides?.apiKeyEnv || process.env.OLLAMA_API_KEY)
)
default:
return !overrides?.apiKey
}
}
/**
* Prompt cache breakpoint for Claude, set on a message's providerOptions.
* Each provider reads only its own key; OpenRouter also reads the
* anthropic one.
*/
export const CACHE_POINT = {
bedrock: { cachePoint: { type: "default" } },
anthropic: { cacheControl: { type: "ephemeral" } },
}
/**
* Check if a model supports prompt caching: Claude models, on Bedrock,
* the Anthropic API or OpenRouter (see CACHE_POINT).
*/
export function supportsPromptCaching(modelId: string): boolean {
return (
modelId.includes("claude") ||
modelId.includes("anthropic") ||
modelId.startsWith("us.anthropic") ||
modelId.startsWith("eu.anthropic")
)
}
/**
* Get the AI model for diagram validation.
* Uses VALIDATION_MODEL env var if set, otherwise falls back to AI_MODEL.
*
* Note: we no longer guess whether the model supports image input from its
* name — that heuristic misfired on newer models (see issue #874). If a
* configured validation model can't handle images, the API call simply errors
* and the validate-diagram route falls back to "valid".
*/
export function getValidationModel(): ReturnType<typeof getAIModel>["model"] {
// AI_MODEL may be comma-separated (multi-model fallback); pick the first.
const envFallback = process.env.AI_MODEL?.split(",")[0]?.trim() || undefined
const modelId = process.env.VALIDATION_MODEL || envFallback
if (!modelId) {
throw new Error(
"No validation model configured. Set VALIDATION_MODEL or AI_MODEL.",
)
}
// A default set in the admin panel becomes AI_PROVIDER/AI_MODEL, but its key
// lives in an ADMIN_-prefixed env var. Point at it the way the chat route
// does for server models, or the standard env var is required instead.
const panelDefault = adminProvidersToConfig(
loadAdminProviders(),
).providers.find((p) => p.default && p.provider === process.env.AI_PROVIDER)
const { model } = getAIModel({
modelId,
apiKeyEnv: panelDefault?.apiKeyEnv,
baseUrlEnv: panelDefault?.baseUrlEnv,
})
return model
}