mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-10-05 01:07:56 +08:00
Found by the PR review, each with a test that failed first: - Quota: any key header skipped it, even one the provider never reads (x-aws-access-key-id with OpenAI), so a request ran on the server's key without being counted. The check now runs after the model is resolved and uses usesServerCredentials. On main already. - usesServerCredentials read the raw base URL; "/" cleans up to none, so an Ollama request ran on the server's key past the server-model check. - SGLang's default 127.0.0.1:8000 only fills the settings form. Chat and the model list used it as a real address, so the server called its own machine even with private URLs blocked. Now a base URL is required. - With a user's OpenAI key and no base URL, the SDK read the server's OPENAI_BASE_URL. The official endpoint is now passed. On main already. - The Test button refused nothing on the server's keys (Ollama Cloud), and a 15 s timeout reported "connected, no tool call". - The model list for Ollama without a base URL came from ollama.com while chat went to the server's Ollama. - Bedrock's "Too many tokens, please wait" counted as context too long. - On the server's keys the provider's error text stays in the server log; it can name the server's AWS account, role or internal hosts. - Desktop app: the preset keys are the user's own (NEXT_AI_DRAWIO_DESKTOP), so Max Output Tokens can be raised and keyless models in settings work again. A launch that found the remembered port taken no longer replaces it, which hid the user's chats and settings for good.
1132 lines
42 KiB
TypeScript
1132 lines
42 KiB
TypeScript
import { createAmazonBedrock } from "@ai-sdk/amazon-bedrock"
|
|
import { createAnthropic } from "@ai-sdk/anthropic"
|
|
import { createAzure } from "@ai-sdk/azure"
|
|
import { createDeepSeek } from "@ai-sdk/deepseek"
|
|
import { createGoogleGenerativeAI } from "@ai-sdk/google"
|
|
import { createVertex } from "@ai-sdk/google-vertex"
|
|
import { createOpenAI } from "@ai-sdk/openai"
|
|
import { createOpenAICompatible } from "@ai-sdk/openai-compatible"
|
|
import { createAihubmix } from "@aihubmix/ai-sdk-provider"
|
|
import { fromNodeProviderChain } from "@aws-sdk/credential-providers"
|
|
import { createOpenRouter } from "@openrouter/ai-sdk-provider"
|
|
import {
|
|
createGateway,
|
|
defaultSettingsMiddleware,
|
|
extractReasoningMiddleware,
|
|
type LanguageModel,
|
|
wrapLanguageModel,
|
|
} from "ai"
|
|
import { createOllama } from "ollama-ai-provider-v2"
|
|
import {
|
|
adminProvidersToConfig,
|
|
loadAdminProviders,
|
|
} from "@/lib/admin/providers"
|
|
import { redirectGuardedFetch } from "@/lib/ssrf-protection"
|
|
import {
|
|
normalizeBaseUrl,
|
|
PROVIDER_INFO,
|
|
type ProviderName,
|
|
} from "@/lib/types/model-config"
|
|
|
|
export type { ProviderName }
|
|
|
|
export const AIHUBMIX_APP_CODE = "MSBS9675"
|
|
|
|
interface ModelConfig {
|
|
model: any
|
|
providerOptions?: any
|
|
modelId: string
|
|
provider: ProviderName
|
|
}
|
|
|
|
// Providers that only support a single system message
|
|
export const SINGLE_SYSTEM_PROVIDERS = new Set<ProviderName>([
|
|
"minimax",
|
|
"glm",
|
|
"qwen",
|
|
"kimi",
|
|
"qiniu",
|
|
"novita",
|
|
"mimo",
|
|
])
|
|
|
|
/**
|
|
* Normalize MiniMax base URL for AI SDK compatibility.
|
|
* MiniMax supports Anthropic-compatible and OpenAI-compatible endpoints.
|
|
*/
|
|
export function normalizeMiniMaxBaseURL(rawUrl: string): {
|
|
baseURL: string
|
|
isAnthropicCompatible: boolean
|
|
} {
|
|
const isAnthropicCompatible = rawUrl.includes("/anthropic")
|
|
let baseURL = rawUrl.replace(/\/$/, "")
|
|
if (isAnthropicCompatible) {
|
|
if (!baseURL.endsWith("/anthropic/v1")) {
|
|
if (baseURL.endsWith("/anthropic")) {
|
|
baseURL = `${baseURL}/v1`
|
|
} else {
|
|
baseURL = `${baseURL}/anthropic/v1`
|
|
}
|
|
}
|
|
} else {
|
|
if (!baseURL.endsWith("/v1")) {
|
|
baseURL = `${baseURL}/v1`
|
|
}
|
|
}
|
|
return { baseURL, isAnthropicCompatible }
|
|
}
|
|
|
|
export function isAihubmixStandardBaseURL(
|
|
rawUrl: string | null | undefined,
|
|
): boolean {
|
|
if (!rawUrl) return true
|
|
|
|
const baseURL = rawUrl.replace(/\/+$/, "")
|
|
return (
|
|
baseURL === "https://aihubmix.com" ||
|
|
baseURL === "https://aihubmix.com/v1"
|
|
)
|
|
}
|
|
|
|
export interface ClientOverrides {
|
|
provider?: string | null
|
|
baseUrl?: string | null
|
|
apiKey?: string | null
|
|
modelId?: string | null
|
|
// AWS Bedrock credentials
|
|
awsAccessKeyId?: string | null
|
|
awsSecretAccessKey?: string | null
|
|
awsRegion?: string | null
|
|
awsSessionToken?: string | null
|
|
// Vertex AI config
|
|
vertexApiKey?: string | null // Express Mode API key
|
|
// Custom headers (e.g., for EdgeOne cookie auth)
|
|
headers?: Record<string, string>
|
|
// Custom env var name(s) for server models
|
|
// Can be a single string or array of strings for load balancing
|
|
apiKeyEnv?: string | string[]
|
|
baseUrlEnv?: string
|
|
}
|
|
|
|
// Bedrock provider options for Anthropic beta features
|
|
const BEDROCK_ANTHROPIC_BETA = {
|
|
bedrock: {
|
|
anthropicBeta: ["fine-grained-tool-streaming-2025-05-14"],
|
|
},
|
|
}
|
|
|
|
/**
|
|
* Resolve baseURL based on whether user is providing their own API key.
|
|
* When user provides their own API key, we should NOT fall back to server's
|
|
* baseURL environment variable - user credentials should only be sent to
|
|
* user-specified endpoints or official provider endpoints.
|
|
*
|
|
* @param userApiKey - User-provided API key (if any)
|
|
* @param userBaseUrl - User-provided base URL (if any)
|
|
* @param serverBaseUrl - Server's base URL from environment variable
|
|
* @param defaultBaseUrl - Provider's official/default base URL (optional)
|
|
* @returns The resolved base URL to use
|
|
*/
|
|
export function resolveBaseURL(
|
|
userApiKey: string | null | undefined,
|
|
userBaseUrl: string | null | undefined,
|
|
serverBaseUrl: string | undefined,
|
|
defaultBaseUrl?: string,
|
|
): string | undefined {
|
|
if (userApiKey) {
|
|
// User provides their own API key - only use user's baseUrl or default
|
|
return userBaseUrl || defaultBaseUrl || undefined
|
|
}
|
|
// No user API key - fall back to server config
|
|
return userBaseUrl || serverBaseUrl || defaultBaseUrl || undefined
|
|
}
|
|
|
|
/**
|
|
* Resolve API key from custom env var name or default env var.
|
|
* Supports multiple API keys per provider via ai-models.json apiKeyEnv config.
|
|
* When multiple keys are configured, randomly selects one for load balancing.
|
|
*
|
|
* Priority:
|
|
* 1. User-provided API key (overrides.apiKey)
|
|
* 2. Custom env var(s) from ai-models.json (overrides.apiKeyEnv)
|
|
* - If array, randomly picks one with a valid value
|
|
* 3. Default provider env var (defaultEnvVar)
|
|
*/
|
|
function resolveApiKey(
|
|
overrides: ClientOverrides | undefined,
|
|
defaultEnvVar: string,
|
|
): string | undefined {
|
|
if (overrides?.apiKey) return overrides.apiKey
|
|
|
|
if (overrides?.apiKeyEnv) {
|
|
// Handle array of env var names - randomly select one
|
|
if (Array.isArray(overrides.apiKeyEnv)) {
|
|
// Filter to only env vars that have values
|
|
const validEnvVars = overrides.apiKeyEnv.filter(
|
|
(envVar) => process.env[envVar],
|
|
)
|
|
if (validEnvVars.length > 0) {
|
|
// Randomly select one
|
|
const selectedEnvVar =
|
|
validEnvVars[
|
|
Math.floor(Math.random() * validEnvVars.length)
|
|
]
|
|
console.log(
|
|
`[API Key Routing] Selected ${selectedEnvVar} from ${validEnvVars.length} available keys`,
|
|
)
|
|
return process.env[selectedEnvVar]
|
|
}
|
|
} else {
|
|
return process.env[overrides.apiKeyEnv]
|
|
}
|
|
}
|
|
|
|
return process.env[defaultEnvVar]
|
|
}
|
|
|
|
/**
|
|
* Resolve base URL from custom env var name or default env var.
|
|
* Supports multiple base URLs per provider via ai-models.json baseUrlEnv config.
|
|
*/
|
|
function resolveBaseUrlEnv(
|
|
overrides: ClientOverrides | undefined,
|
|
defaultEnvVar: string,
|
|
): string | undefined {
|
|
if (overrides?.baseUrlEnv) return process.env[overrides.baseUrlEnv]
|
|
return process.env[defaultEnvVar]
|
|
}
|
|
|
|
/**
|
|
* Safely parse integer from environment variable with validation
|
|
*/
|
|
function parseIntSafe(
|
|
value: string | undefined,
|
|
varName: string,
|
|
min?: number,
|
|
max?: number,
|
|
): number | undefined {
|
|
if (!value) return undefined
|
|
const parsed = Number.parseInt(value, 10)
|
|
if (Number.isNaN(parsed)) {
|
|
throw new Error(`${varName} must be a valid integer, got: ${value}`)
|
|
}
|
|
if (min !== undefined && parsed < min) {
|
|
throw new Error(`${varName} must be >= ${min}, got: ${parsed}`)
|
|
}
|
|
if (max !== undefined && parsed > max) {
|
|
throw new Error(`${varName} must be <= ${max}, got: ${parsed}`)
|
|
}
|
|
return parsed
|
|
}
|
|
|
|
/**
|
|
* GOOGLE_TOP_K and GOOGLE_TOP_P. They are call settings, so they go on the
|
|
* model through a middleware: as Google provider options they were dropped.
|
|
*/
|
|
function googleSamplingSettings(): { topK?: number; topP?: number } {
|
|
const settings: { topK?: number; topP?: number } = {}
|
|
const topK = parseIntSafe(process.env.GOOGLE_TOP_K, "GOOGLE_TOP_K", 1, 100)
|
|
if (topK) settings.topK = topK
|
|
if (process.env.GOOGLE_TOP_P) {
|
|
const topP = Number.parseFloat(process.env.GOOGLE_TOP_P)
|
|
if (Number.isNaN(topP) || topP < 0 || topP > 1) {
|
|
throw new Error(
|
|
`GOOGLE_TOP_P must be a number between 0 and 1, got: ${process.env.GOOGLE_TOP_P}`,
|
|
)
|
|
}
|
|
settings.topP = topP
|
|
}
|
|
return settings
|
|
}
|
|
|
|
/**
|
|
* Build provider-specific options from environment variables
|
|
* Supports various AI SDK providers with their unique configuration options
|
|
*
|
|
* Environment variables:
|
|
* - OPENAI_REASONING_EFFORT: OpenAI reasoning effort level (minimal/low/medium/high) - for the o-series and gpt-5 or later
|
|
* - OPENAI_REASONING_SUMMARY: OpenAI reasoning summary (auto/detailed) - auto-enabled for the o-series and gpt-5 or later
|
|
* - ANTHROPIC_THINKING_BUDGET_TOKENS: Anthropic thinking budget in tokens (1024-64000)
|
|
* - ANTHROPIC_THINKING_TYPE: Anthropic thinking type (enabled)
|
|
* - GOOGLE_THINKING_BUDGET: Google Gemini 2.5 thinking budget in tokens (1024-100000)
|
|
* - GOOGLE_THINKING_LEVEL: Google Gemini 3 thinking level (low/high)
|
|
* - GOOGLE_VERTEX_THINKING_BUDGET: Vertex AI Gemini 2.5 thinking budget in tokens (1024-100000)
|
|
* - GOOGLE_VERTEX_THINKING_LEVEL: Vertex AI Gemini 3 thinking level (low/high)
|
|
* - AZURE_REASONING_EFFORT: Azure/OpenAI reasoning effort (low/medium/high)
|
|
* - AZURE_REASONING_SUMMARY: Azure reasoning summary (none/brief/detailed)
|
|
* - BEDROCK_REASONING_BUDGET_TOKENS: Bedrock Claude reasoning budget in tokens (1024-64000)
|
|
* - BEDROCK_REASONING_EFFORT: Bedrock Nova reasoning effort (low/medium/high)
|
|
* - OLLAMA_ENABLE_THINKING: Enable Ollama thinking mode (set to "true")
|
|
*/
|
|
function buildProviderOptions(
|
|
provider: ProviderName,
|
|
modelId?: string,
|
|
): Record<string, any> | undefined {
|
|
const options: Record<string, any> = {}
|
|
|
|
switch (provider) {
|
|
case "openai": {
|
|
const reasoningEffort = process.env.OPENAI_REASONING_EFFORT
|
|
const reasoningSummary = process.env.OPENAI_REASONING_SUMMARY
|
|
|
|
// Reasoning models (the o-series, gpt-5 and later) need
|
|
// reasoningSummary to return thoughts
|
|
if (modelId && /^(o\d|gpt-([5-9]|[1-9]\d))/.test(modelId)) {
|
|
options.openai = {
|
|
// Auto-enable reasoning summary for reasoning models
|
|
// Use 'auto' as default since not all models support 'detailed'
|
|
reasoningSummary:
|
|
(reasoningSummary as "auto" | "detailed") || "auto",
|
|
}
|
|
|
|
// Optionally configure reasoning effort
|
|
if (reasoningEffort) {
|
|
options.openai.reasoningEffort = reasoningEffort as
|
|
| "minimal"
|
|
| "low"
|
|
| "medium"
|
|
| "high"
|
|
}
|
|
} else if (reasoningEffort || reasoningSummary) {
|
|
// Non-reasoning models: only apply if explicitly configured
|
|
options.openai = {}
|
|
if (reasoningEffort) {
|
|
options.openai.reasoningEffort = reasoningEffort as
|
|
| "minimal"
|
|
| "low"
|
|
| "medium"
|
|
| "high"
|
|
}
|
|
if (reasoningSummary) {
|
|
options.openai.reasoningSummary = reasoningSummary as
|
|
| "auto"
|
|
| "detailed"
|
|
}
|
|
}
|
|
break
|
|
}
|
|
|
|
case "anthropic": {
|
|
const thinkingBudget = parseIntSafe(
|
|
process.env.ANTHROPIC_THINKING_BUDGET_TOKENS,
|
|
"ANTHROPIC_THINKING_BUDGET_TOKENS",
|
|
1024,
|
|
64000,
|
|
)
|
|
const thinkingType =
|
|
process.env.ANTHROPIC_THINKING_TYPE || "enabled"
|
|
|
|
if (thinkingBudget) {
|
|
options.anthropic = {
|
|
thinking: {
|
|
type: thinkingType,
|
|
budgetTokens: thinkingBudget,
|
|
},
|
|
}
|
|
}
|
|
break
|
|
}
|
|
|
|
case "google": {
|
|
const thinkingBudgetVal = parseIntSafe(
|
|
process.env.GOOGLE_THINKING_BUDGET,
|
|
"GOOGLE_THINKING_BUDGET",
|
|
1024,
|
|
100000,
|
|
)
|
|
const thinkingLevel = process.env.GOOGLE_THINKING_LEVEL
|
|
|
|
// Google Gemini 2.5/3 models think by default, but need includeThoughts: true
|
|
// to return the reasoning in the response
|
|
if (
|
|
modelId &&
|
|
(modelId.includes("gemini-2") ||
|
|
modelId.includes("gemini-3") ||
|
|
modelId.includes("gemini2") ||
|
|
modelId.includes("gemini3"))
|
|
) {
|
|
const thinkingConfig: Record<string, any> = {
|
|
includeThoughts: true,
|
|
}
|
|
|
|
// Optionally configure thinking budget or level
|
|
if (
|
|
thinkingBudgetVal &&
|
|
(modelId.includes("2.5") || modelId.includes("2-5"))
|
|
) {
|
|
thinkingConfig.thinkingBudget = thinkingBudgetVal
|
|
} else if (
|
|
thinkingLevel &&
|
|
(modelId.includes("gemini-3") ||
|
|
modelId.includes("gemini3"))
|
|
) {
|
|
thinkingConfig.thinkingLevel = thinkingLevel as
|
|
| "low"
|
|
| "high"
|
|
}
|
|
|
|
options.google = { thinkingConfig }
|
|
}
|
|
break
|
|
}
|
|
case "vertexai": {
|
|
const thinkingBudget = parseIntSafe(
|
|
process.env.GOOGLE_VERTEX_THINKING_BUDGET,
|
|
"GOOGLE_VERTEX_THINKING_BUDGET",
|
|
1024,
|
|
100000,
|
|
)
|
|
const thinkingLevel = process.env.GOOGLE_VERTEX_THINKING_LEVEL
|
|
|
|
if (
|
|
modelId &&
|
|
(modelId.includes("gemini-2") ||
|
|
modelId.includes("gemini-3") ||
|
|
modelId.includes("gemini2") ||
|
|
modelId.includes("gemini3"))
|
|
) {
|
|
const thinkingConfig: Record<string, any> = {
|
|
includeThoughts: true,
|
|
}
|
|
|
|
const isGemini3 =
|
|
modelId?.includes("gemini-3") ||
|
|
modelId?.includes("gemini3")
|
|
const isGemini25 =
|
|
modelId?.includes("2.5") || modelId?.includes("2-5")
|
|
|
|
if (isGemini3 && thinkingLevel) {
|
|
// Vertex AI provider in AI SDK supports more granular levels (minimal/low/medium/high)
|
|
thinkingConfig.thinkingLevel = thinkingLevel as
|
|
| "minimal"
|
|
| "low"
|
|
| "medium"
|
|
| "high"
|
|
} else if (isGemini25 && thinkingBudget) {
|
|
thinkingConfig.thinkingBudget = thinkingBudget
|
|
}
|
|
options.google = { thinkingConfig }
|
|
}
|
|
break
|
|
}
|
|
case "azure": {
|
|
const reasoningEffort = process.env.AZURE_REASONING_EFFORT
|
|
const reasoningSummary = process.env.AZURE_REASONING_SUMMARY
|
|
|
|
if (reasoningEffort || reasoningSummary) {
|
|
options.azure = {}
|
|
if (reasoningEffort) {
|
|
options.azure.reasoningEffort = reasoningEffort as
|
|
| "low"
|
|
| "medium"
|
|
| "high"
|
|
}
|
|
if (reasoningSummary) {
|
|
options.azure.reasoningSummary = reasoningSummary as
|
|
| "none"
|
|
| "brief"
|
|
| "detailed"
|
|
}
|
|
}
|
|
break
|
|
}
|
|
|
|
case "bedrock": {
|
|
const budgetTokens = parseIntSafe(
|
|
process.env.BEDROCK_REASONING_BUDGET_TOKENS,
|
|
"BEDROCK_REASONING_BUDGET_TOKENS",
|
|
1024,
|
|
64000,
|
|
)
|
|
const reasoningEffort = process.env.BEDROCK_REASONING_EFFORT
|
|
|
|
// Bedrock reasoning ONLY for Claude and Nova models
|
|
// Other models (MiniMax, etc.) don't support reasoningConfig
|
|
if (
|
|
modelId &&
|
|
(budgetTokens || reasoningEffort) &&
|
|
(modelId.includes("claude") ||
|
|
modelId.includes("anthropic") ||
|
|
modelId.includes("nova") ||
|
|
modelId.includes("amazon"))
|
|
) {
|
|
const reasoningConfig: Record<string, any> = { type: "enabled" }
|
|
|
|
// Claude models: use budgetTokens (1024-64000)
|
|
if (
|
|
budgetTokens &&
|
|
(modelId.includes("claude") ||
|
|
modelId.includes("anthropic"))
|
|
) {
|
|
reasoningConfig.budgetTokens = budgetTokens
|
|
}
|
|
// Nova models: use maxReasoningEffort (low/medium/high)
|
|
else if (
|
|
reasoningEffort &&
|
|
(modelId.includes("nova") || modelId.includes("amazon"))
|
|
) {
|
|
reasoningConfig.maxReasoningEffort = reasoningEffort as
|
|
| "low"
|
|
| "medium"
|
|
| "high"
|
|
}
|
|
|
|
options.bedrock = { reasoningConfig }
|
|
}
|
|
break
|
|
}
|
|
|
|
case "ollama": {
|
|
const enableThinking = process.env.OLLAMA_ENABLE_THINKING
|
|
// Ollama supports reasoning with think: true for models like qwen3
|
|
if (enableThinking === "true") {
|
|
options.ollama = { think: true }
|
|
}
|
|
break
|
|
}
|
|
|
|
default:
|
|
break
|
|
}
|
|
|
|
return Object.keys(options).length > 0 ? options : undefined
|
|
}
|
|
|
|
// Map of provider to required environment variable
|
|
export const PROVIDER_ENV_VARS: Record<ProviderName, string | null> = {
|
|
bedrock: null, // AWS SDK auto-uses IAM role on AWS, or env vars locally
|
|
openai: "OPENAI_API_KEY",
|
|
anthropic: "ANTHROPIC_API_KEY",
|
|
google: "GOOGLE_GENERATIVE_AI_API_KEY",
|
|
vertexai: "GOOGLE_VERTEX_API_KEY",
|
|
azure: "AZURE_API_KEY",
|
|
ollama: null, // No credentials needed for local Ollama
|
|
openrouter: "OPENROUTER_API_KEY",
|
|
aihubmix: "AIHUBMIX_API_KEY",
|
|
deepseek: "DEEPSEEK_API_KEY",
|
|
siliconflow: "SILICONFLOW_API_KEY",
|
|
sglang: "SGLANG_API_KEY",
|
|
gateway: "AI_GATEWAY_API_KEY",
|
|
edgeone: null, // No credentials needed - uses EdgeOne Edge AI
|
|
doubao: "DOUBAO_API_KEY",
|
|
modelscope: "MODELSCOPE_API_KEY",
|
|
glm: "GLM_API_KEY",
|
|
qwen: "QWEN_API_KEY",
|
|
qiniu: "QINIU_API_KEY",
|
|
kimi: "KIMI_API_KEY",
|
|
minimax: "MINIMAX_API_KEY",
|
|
novita: "NOVITA_API_KEY",
|
|
mimo: "MIMO_API_KEY",
|
|
atlascloud: "ATLASCLOUD_API_KEY",
|
|
}
|
|
|
|
/**
|
|
* Auto-detect provider based on available API keys
|
|
* Returns the provider if exactly one is configured, otherwise null
|
|
*/
|
|
function detectProvider(): ProviderName | null {
|
|
const configuredProviders: ProviderName[] = []
|
|
|
|
for (const [provider, envVar] of Object.entries(PROVIDER_ENV_VARS)) {
|
|
if (envVar === null) {
|
|
// Skip ollama - it doesn't require credentials
|
|
continue
|
|
}
|
|
// Anthropic accepts ANTHROPIC_AUTH_TOKEN (Bearer auth) as alternative to ANTHROPIC_API_KEY
|
|
const hasCredential =
|
|
provider === "anthropic"
|
|
? !!(
|
|
process.env.ANTHROPIC_API_KEY ||
|
|
process.env.ANTHROPIC_AUTH_TOKEN
|
|
)
|
|
: !!process.env[envVar]
|
|
if (hasCredential) {
|
|
// Azure requires additional config (baseURL or resourceName)
|
|
if (provider === "azure") {
|
|
const hasBaseUrl = !!process.env.AZURE_BASE_URL
|
|
const hasResourceName = !!process.env.AZURE_RESOURCE_NAME
|
|
if (hasBaseUrl || hasResourceName) {
|
|
configuredProviders.push(provider as ProviderName)
|
|
}
|
|
} else {
|
|
configuredProviders.push(provider as ProviderName)
|
|
}
|
|
}
|
|
}
|
|
|
|
if (configuredProviders.length === 1) {
|
|
return configuredProviders[0]
|
|
}
|
|
|
|
return null
|
|
}
|
|
|
|
/**
|
|
* Validate that required API keys are present for the selected provider
|
|
* @param provider - The provider to validate
|
|
* @param customApiKeyEnv - Optional custom env var name(s) (from ai-models.json apiKeyEnv)
|
|
*/
|
|
function validateProviderCredentials(
|
|
provider: ProviderName,
|
|
customApiKeyEnv?: string | string[],
|
|
): void {
|
|
// Handle array of env var names - at least one must be set
|
|
if (Array.isArray(customApiKeyEnv)) {
|
|
const hasAnyKey = customApiKeyEnv.some((envVar) => process.env[envVar])
|
|
if (!hasAnyKey) {
|
|
throw new Error(
|
|
`At least one of [${customApiKeyEnv.join(", ")}] environment variables is required for ${provider} provider. ` +
|
|
`Please set at least one in your .env.local file.`,
|
|
)
|
|
}
|
|
return
|
|
}
|
|
|
|
// Anthropic accepts ANTHROPIC_AUTH_TOKEN (Bearer auth) as alternative to ANTHROPIC_API_KEY
|
|
if (provider === "anthropic" && !customApiKeyEnv) {
|
|
const hasCredential = !!(
|
|
process.env.ANTHROPIC_API_KEY || process.env.ANTHROPIC_AUTH_TOKEN
|
|
)
|
|
if (!hasCredential) {
|
|
throw new Error(
|
|
`Either ANTHROPIC_API_KEY or ANTHROPIC_AUTH_TOKEN environment variable is required for anthropic provider. ` +
|
|
`Please set one in your .env.local file.`,
|
|
)
|
|
}
|
|
} else {
|
|
// Use custom env var name if provided, otherwise use default
|
|
const requiredVar = customApiKeyEnv || PROVIDER_ENV_VARS[provider]
|
|
if (requiredVar && !process.env[requiredVar]) {
|
|
throw new Error(
|
|
`${requiredVar} environment variable is required for ${provider} provider. ` +
|
|
`Please set it in your .env.local file.`,
|
|
)
|
|
}
|
|
}
|
|
|
|
// Azure requires either AZURE_BASE_URL or AZURE_RESOURCE_NAME in addition to API key
|
|
if (provider === "azure") {
|
|
const hasBaseUrl = !!process.env.AZURE_BASE_URL
|
|
const hasResourceName = !!process.env.AZURE_RESOURCE_NAME
|
|
if (!hasBaseUrl && !hasResourceName) {
|
|
throw new Error(
|
|
`Azure requires either AZURE_BASE_URL or AZURE_RESOURCE_NAME to be set. ` +
|
|
`Please set one in your .env.local file.`,
|
|
)
|
|
}
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Providers whose SDK has the official endpoint built in. The others are
|
|
* OpenAI-compatible APIs (or Anthropic) that are called at
|
|
* PROVIDER_INFO.defaultBaseUrl unless a base URL is configured.
|
|
*/
|
|
const SDK_KNOWS_ENDPOINT = new Set<ProviderName>([
|
|
"openai",
|
|
"google",
|
|
"azure",
|
|
"openrouter",
|
|
"gateway",
|
|
"deepseek",
|
|
])
|
|
|
|
/** Where and how to call a provider, once credentials are resolved */
|
|
interface Endpoint {
|
|
apiKey?: string
|
|
baseURL?: string
|
|
headers?: Record<string, string>
|
|
fetch?: typeof fetch
|
|
authToken?: string // Anthropic Bearer auth
|
|
resourceName?: string // Azure
|
|
}
|
|
|
|
/**
|
|
* An OpenAI-compatible chat model. includeUsage asks for token usage in the
|
|
* stream, which quota tracking needs. Some of these models write their
|
|
* reasoning inside <think> tags; that text becomes reasoning, not reply.
|
|
*/
|
|
function compatibleModel(
|
|
provider: ProviderName,
|
|
modelId: string,
|
|
e: Endpoint,
|
|
): LanguageModel {
|
|
const model = createOpenAICompatible({
|
|
name: provider,
|
|
apiKey: e.apiKey,
|
|
baseURL: e.baseURL ?? "",
|
|
...(e.headers && { headers: e.headers }),
|
|
...(e.fetch && { fetch: e.fetch }),
|
|
includeUsage: true,
|
|
})(modelId)
|
|
return wrapLanguageModel({
|
|
model,
|
|
middleware: extractReasoningMiddleware({ tagName: "think" }),
|
|
})
|
|
}
|
|
|
|
/** Create the model for a provider. Credentials are already resolved. */
|
|
function createModel(
|
|
provider: ProviderName,
|
|
modelId: string,
|
|
e: Endpoint,
|
|
): LanguageModel {
|
|
const opts = {
|
|
apiKey: e.apiKey,
|
|
...(e.baseURL && { baseURL: e.baseURL }),
|
|
...(e.fetch && { fetch: e.fetch }),
|
|
}
|
|
switch (provider) {
|
|
case "openai": {
|
|
const openaiProvider = createOpenAI(opts)
|
|
// A custom base URL is usually a proxy that only has Chat
|
|
// Completions; the official endpoint uses the Responses API,
|
|
// which returns reasoning for the o-series and gpt-5 or later
|
|
return e.baseURL &&
|
|
e.baseURL !== PROVIDER_INFO.openai.defaultBaseUrl
|
|
? openaiProvider.chat(modelId)
|
|
: openaiProvider(modelId)
|
|
}
|
|
case "anthropic":
|
|
// The provider streams tool input per tool (eager_input_streaming),
|
|
// which replaced the fine-grained-tool-streaming beta header
|
|
return createAnthropic({
|
|
...(e.authToken
|
|
? { authToken: e.authToken }
|
|
: { apiKey: e.apiKey }),
|
|
baseURL: e.baseURL,
|
|
...(e.fetch && { fetch: e.fetch }),
|
|
})(modelId)
|
|
case "google": {
|
|
const model = createGoogleGenerativeAI(opts)(modelId)
|
|
const sampling = googleSamplingSettings()
|
|
return Object.keys(sampling).length > 0
|
|
? wrapLanguageModel({
|
|
model,
|
|
middleware: defaultSettingsMiddleware({
|
|
settings: sampling,
|
|
}),
|
|
})
|
|
: model
|
|
}
|
|
case "azure":
|
|
// baseURL takes precedence over resourceName per SDK behavior
|
|
return createAzure({
|
|
...opts,
|
|
...(!e.baseURL &&
|
|
e.resourceName && { resourceName: e.resourceName }),
|
|
})(modelId)
|
|
case "openrouter":
|
|
return createOpenRouter(opts)(modelId)
|
|
case "gateway":
|
|
// Without a key or URL the SDK uses Vercel's endpoint and OIDC
|
|
return createGateway(opts)(modelId)
|
|
case "deepseek":
|
|
case "kimi":
|
|
case "mimo":
|
|
// Kimi and MiMo return reasoning_content like DeepSeek and need it
|
|
// passed back in multi-turn tool calls (MiMo answers 400 otherwise)
|
|
return createDeepSeek(opts)(modelId)
|
|
case "doubao": {
|
|
// DeepSeek and Kimi models on Doubao use reasoning_content too
|
|
const lower = modelId.toLowerCase()
|
|
return lower.includes("deepseek") || lower.includes("kimi")
|
|
? createDeepSeek(opts)(modelId)
|
|
: compatibleModel(provider, modelId, e)
|
|
}
|
|
case "aihubmix":
|
|
return isAihubmixStandardBaseURL(e.baseURL)
|
|
? createAihubmix({
|
|
apiKey: e.apiKey,
|
|
appCode: AIHUBMIX_APP_CODE,
|
|
})(modelId)
|
|
: compatibleModel(provider, modelId, e)
|
|
case "minimax": {
|
|
const { baseURL, isAnthropicCompatible } = normalizeMiniMaxBaseURL(
|
|
e.baseURL as string,
|
|
)
|
|
return isAnthropicCompatible
|
|
? createAnthropic({
|
|
apiKey: e.apiKey,
|
|
baseURL,
|
|
...(e.fetch && { fetch: e.fetch }),
|
|
})(modelId)
|
|
: compatibleModel(provider, modelId, { ...e, baseURL })
|
|
}
|
|
default:
|
|
// siliconflow, sglang, modelscope, glm, qwen, qiniu, novita,
|
|
// atlascloud, edgeone
|
|
return compatibleModel(provider, modelId, e)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Get the AI model for a chat request: the client's own provider and
|
|
* credentials, or the server's (AI_PROVIDER, AI_MODEL and each provider's
|
|
* <NAME>_API_KEY / <NAME>_BASE_URL, see env.example). The settings test
|
|
* button uses the same function, so a passing test means the chat works.
|
|
*/
|
|
export function getAIModel(clientOverrides?: ClientOverrides): ModelConfig {
|
|
// Drop an endpoint path pasted along with the client's base URL
|
|
const overrides = clientOverrides?.baseUrl
|
|
? {
|
|
...clientOverrides,
|
|
baseUrl: normalizeBaseUrl(clientOverrides.baseUrl),
|
|
}
|
|
: clientOverrides
|
|
// SECURITY: Prevent SSRF attacks (GHSA-9qf7-mprq-9qgm)
|
|
// If a custom baseUrl is provided, an API key MUST also be provided.
|
|
// This prevents attackers from redirecting server API keys to malicious endpoints.
|
|
// Exception: EdgeOne doesn't require API keys.
|
|
// Ollama is exempt only when no server OLLAMA_API_KEY is configured;
|
|
// when it IS configured, the outer guard also enforces client apiKey for custom baseUrls.
|
|
if (
|
|
overrides?.baseUrl &&
|
|
!overrides?.apiKey &&
|
|
!(overrides?.provider === "vertexai" && overrides?.vertexApiKey) &&
|
|
overrides?.provider !== "edgeone" &&
|
|
!(overrides?.provider === "ollama" && !process.env.OLLAMA_API_KEY)
|
|
) {
|
|
throw new Error(
|
|
`API key is required when using a custom base URL. ` +
|
|
`Please provide your own API key in Settings.`,
|
|
)
|
|
}
|
|
|
|
// Check if client is providing their own provider override
|
|
const isClientOverride = !!(
|
|
overrides?.provider &&
|
|
(overrides?.apiKey ||
|
|
(overrides?.provider === "vertexai" && overrides?.vertexApiKey))
|
|
)
|
|
|
|
// Use client override if provided, otherwise fall back to env vars.
|
|
// AI_MODEL may be comma-separated (multi-model fallback); pick the first.
|
|
const envModel = process.env.AI_MODEL?.split(",")[0]?.trim() || undefined
|
|
const modelId = overrides?.modelId || envModel
|
|
|
|
if (!modelId) {
|
|
if (isClientOverride) {
|
|
throw new Error(
|
|
`Model ID is required when using custom AI provider. Please specify a model in Settings.`,
|
|
)
|
|
}
|
|
throw new Error(
|
|
`AI_MODEL environment variable is required. Example: AI_MODEL=claude-sonnet-4-5`,
|
|
)
|
|
}
|
|
|
|
// Determine provider: client override > explicit config > auto-detect > error
|
|
let provider: ProviderName
|
|
if (overrides?.provider) {
|
|
// Validate client-provided provider
|
|
if (!Object.hasOwn(PROVIDER_INFO, overrides.provider)) {
|
|
throw new Error(
|
|
`Invalid provider: ${overrides.provider}. Allowed providers: ${Object.keys(PROVIDER_INFO).join(", ")}`,
|
|
)
|
|
}
|
|
provider = overrides.provider as ProviderName
|
|
} else if (process.env.AI_PROVIDER) {
|
|
provider = process.env.AI_PROVIDER as ProviderName
|
|
} else {
|
|
const detected = detectProvider()
|
|
if (detected) {
|
|
provider = detected
|
|
console.log(`[AI Provider] Auto-detected provider: ${provider}`)
|
|
} else {
|
|
// List configured providers for better error message
|
|
const configured = Object.entries(PROVIDER_ENV_VARS)
|
|
.filter(([, envVar]) => envVar && process.env[envVar as string])
|
|
.map(([p]) => p)
|
|
|
|
if (configured.length === 0) {
|
|
const keys = Object.entries(PROVIDER_ENV_VARS)
|
|
.filter(([, envVar]) => envVar)
|
|
.map(([p, envVar]) => `- ${envVar} for ${p}`)
|
|
throw new Error(
|
|
`No AI provider configured. Please set one of the following API keys in your .env.local file:\n` +
|
|
`${keys.join("\n")}\n` +
|
|
`- AWS_ACCESS_KEY_ID for bedrock\n` +
|
|
`Or set AI_PROVIDER=ollama for local Ollama.`,
|
|
)
|
|
}
|
|
throw new Error(
|
|
`Multiple AI providers configured (${configured.join(", ")}). ` +
|
|
`Please set AI_PROVIDER to specify which one to use.`,
|
|
)
|
|
}
|
|
}
|
|
if (!Object.hasOwn(PROVIDER_INFO, provider)) {
|
|
throw new Error(
|
|
`Unknown AI provider: ${provider}. Supported providers: ${Object.keys(PROVIDER_INFO).join(", ")}`,
|
|
)
|
|
}
|
|
|
|
// Only validate server credentials if client isn't providing their own API key
|
|
if (!isClientOverride) {
|
|
validateProviderCredentials(provider, overrides?.apiKeyEnv)
|
|
}
|
|
|
|
console.log(`[AI Provider] Initializing ${provider} with model: ${modelId}`)
|
|
|
|
// Requests to a base URL the client chose must not follow redirects
|
|
const guardedFetch = overrides?.baseUrl ? redirectGuardedFetch() : undefined
|
|
// Build provider-specific options from environment variables
|
|
let providerOptions = buildProviderOptions(provider, modelId)
|
|
let model: LanguageModel
|
|
|
|
switch (provider) {
|
|
case "bedrock": {
|
|
// Use client-provided credentials if available, otherwise fall back to IAM/env vars
|
|
const hasClientCredentials =
|
|
overrides?.awsAccessKeyId && overrides?.awsSecretAccessKey
|
|
// Keys from the admin panel. The ADMIN_ names keep them out of the
|
|
// default AWS credential chain, which other clients such as the
|
|
// DynamoDB quota manager use with their own credentials.
|
|
const adminAccessKeyId = process.env.ADMIN_AWS_ACCESS_KEY_ID
|
|
const adminSecretAccessKey = process.env.ADMIN_AWS_SECRET_ACCESS_KEY
|
|
const bedrockRegion =
|
|
overrides?.awsRegion ||
|
|
process.env.ADMIN_AWS_REGION ||
|
|
process.env.AWS_REGION ||
|
|
"us-west-2"
|
|
|
|
const bedrockProvider = hasClientCredentials
|
|
? createAmazonBedrock({
|
|
region: bedrockRegion,
|
|
accessKeyId: overrides.awsAccessKeyId as string,
|
|
secretAccessKey: overrides.awsSecretAccessKey as string,
|
|
...(overrides?.awsSessionToken && {
|
|
sessionToken: overrides.awsSessionToken,
|
|
}),
|
|
})
|
|
: adminAccessKeyId && adminSecretAccessKey
|
|
? createAmazonBedrock({
|
|
region: bedrockRegion,
|
|
accessKeyId: adminAccessKeyId,
|
|
secretAccessKey: adminSecretAccessKey,
|
|
})
|
|
: createAmazonBedrock({
|
|
region: bedrockRegion,
|
|
credentialProvider: fromNodeProviderChain(),
|
|
})
|
|
model = bedrockProvider(modelId)
|
|
// Add Anthropic beta options if using Claude models via Bedrock
|
|
if (modelId.includes("anthropic.claude")) {
|
|
// Deep merge to preserve both anthropicBeta and reasoningConfig
|
|
providerOptions = {
|
|
bedrock: {
|
|
...BEDROCK_ANTHROPIC_BETA.bedrock,
|
|
...(providerOptions?.bedrock || {}),
|
|
},
|
|
}
|
|
}
|
|
break
|
|
}
|
|
|
|
case "vertexai": {
|
|
// Express Mode: Use API key for authentication
|
|
// SECURITY: a client base URL only ever gets the client's key, so the
|
|
// server's GOOGLE_VERTEX_API_KEY is never sent to a client-chosen host
|
|
const vertexApiKey = overrides?.baseUrl
|
|
? overrides.vertexApiKey
|
|
: overrides?.vertexApiKey || process.env.GOOGLE_VERTEX_API_KEY
|
|
|
|
if (!vertexApiKey) {
|
|
throw new Error(
|
|
"Vertex AI requires an API key for Express Mode. " +
|
|
"Get one from Google Cloud Console or set GOOGLE_VERTEX_API_KEY environment variable.",
|
|
)
|
|
}
|
|
|
|
// Support custom base URL from env or client override.
|
|
// A client key only goes to the client's URL or the official one.
|
|
const baseURL = resolveBaseURL(
|
|
overrides?.vertexApiKey,
|
|
overrides?.baseUrl,
|
|
process.env.GOOGLE_VERTEX_BASE_URL,
|
|
)
|
|
model = createVertex({
|
|
apiKey: vertexApiKey,
|
|
...(baseURL && { baseURL }),
|
|
...(guardedFetch && { fetch: guardedFetch }),
|
|
})(modelId)
|
|
break
|
|
}
|
|
|
|
case "ollama": {
|
|
const baseURL = overrides?.baseUrl || process.env.OLLAMA_BASE_URL
|
|
// SECURITY: When client provides a custom base URL, only use
|
|
// client-provided API key. Never fall back to server OLLAMA_API_KEY
|
|
// to prevent leaking server credentials to user-controlled endpoints.
|
|
const apiKey = overrides?.baseUrl
|
|
? overrides?.apiKey || undefined
|
|
: resolveApiKey(overrides, "OLLAMA_API_KEY")
|
|
model = createOllama({
|
|
...(baseURL && { baseURL }),
|
|
...(apiKey && {
|
|
headers: { Authorization: `Bearer ${apiKey}` },
|
|
}),
|
|
...(guardedFetch && { fetch: guardedFetch }),
|
|
})(modelId)
|
|
break
|
|
}
|
|
|
|
case "edgeone":
|
|
// EdgeOne Pages Edge AI, an OpenAI-compatible API without a key.
|
|
// The SDK appends /chat/completions to the base URL. Cookies
|
|
// (eo_token, eo_time) and the access code authenticate the call.
|
|
model = compatibleModel(provider, modelId, {
|
|
apiKey: "edgeone",
|
|
baseURL: overrides?.baseUrl || "/api/edgeai",
|
|
headers: overrides?.headers,
|
|
fetch: guardedFetch,
|
|
})
|
|
break
|
|
|
|
default: {
|
|
// Every other provider takes an API key and a base URL from
|
|
// <NAME>_API_KEY / <NAME>_BASE_URL (or a server model's apiKeyEnv)
|
|
const apiKey = resolveApiKey(
|
|
overrides,
|
|
PROVIDER_ENV_VARS[provider] as string,
|
|
)
|
|
const baseUrlEnv =
|
|
provider === "gateway"
|
|
? "AI_GATEWAY_BASE_URL"
|
|
: `${provider.toUpperCase()}_BASE_URL`
|
|
// A local default (SGLang's 127.0.0.1) only fills the settings
|
|
// form; the server must not call its own machine for it. With a
|
|
// user's key the OpenAI SDK would read the server's
|
|
// OPENAI_BASE_URL, so name the official endpoint.
|
|
const defaultUrl = PROVIDER_INFO[provider].defaultBaseUrl
|
|
const publicDefault = defaultUrl?.startsWith("https://")
|
|
? defaultUrl
|
|
: undefined
|
|
const baseURL = resolveBaseURL(
|
|
overrides?.apiKey,
|
|
overrides?.baseUrl,
|
|
resolveBaseUrlEnv(overrides, baseUrlEnv),
|
|
SDK_KNOWS_ENDPOINT.has(provider) &&
|
|
!(provider === "openai" && overrides?.apiKey)
|
|
? undefined
|
|
: publicDefault,
|
|
)
|
|
if (!baseURL && !SDK_KNOWS_ENDPOINT.has(provider)) {
|
|
throw new Error(
|
|
`${PROVIDER_INFO[provider].label} needs a base URL. Add it in the model settings.`,
|
|
)
|
|
}
|
|
model = createModel(provider, modelId, {
|
|
apiKey,
|
|
baseURL,
|
|
fetch: guardedFetch,
|
|
// Bearer auth for Anthropic when there is no API key
|
|
authToken:
|
|
provider === "anthropic" && !apiKey
|
|
? process.env.ANTHROPIC_AUTH_TOKEN
|
|
: undefined,
|
|
// Only the server's own resource; a client key needs its URL
|
|
resourceName:
|
|
provider === "azure" && !overrides?.apiKey
|
|
? process.env.AZURE_RESOURCE_NAME
|
|
: undefined,
|
|
})
|
|
}
|
|
}
|
|
|
|
return { model, providerOptions, modelId, provider }
|
|
}
|
|
|
|
/**
|
|
* Whether the call is paid for by the server's own credentials (env keys or
|
|
* IAM role) rather than credentials sent with the request. Mirrors which key
|
|
* each branch of getAIModel ends up using.
|
|
*/
|
|
export function usesServerCredentials(
|
|
provider: ProviderName,
|
|
overrides?: ClientOverrides,
|
|
): boolean {
|
|
// The desktop app's local server holds the user's own preset keys
|
|
if (process.env.NEXT_AI_DRAWIO_DESKTOP === "1") return false
|
|
// Cleaned like getAIModel does: "/" means no base URL
|
|
const baseUrl = normalizeBaseUrl(overrides?.baseUrl ?? "")
|
|
switch (provider) {
|
|
case "bedrock":
|
|
return !(overrides?.awsAccessKeyId && overrides?.awsSecretAccessKey)
|
|
case "vertexai":
|
|
return !overrides?.vertexApiKey
|
|
case "edgeone":
|
|
// The platform's own endpoint, no key involved
|
|
return false
|
|
case "ollama":
|
|
// Only a server key costs money; a keyless local server or the
|
|
// client's own server does not
|
|
return (
|
|
!baseUrl &&
|
|
!overrides?.apiKey &&
|
|
!!(overrides?.apiKeyEnv || process.env.OLLAMA_API_KEY)
|
|
)
|
|
default:
|
|
return !overrides?.apiKey
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Prompt cache breakpoint for Claude, set on a message's providerOptions.
|
|
* Each provider reads only its own key; OpenRouter also reads the
|
|
* anthropic one.
|
|
*/
|
|
export const CACHE_POINT = {
|
|
bedrock: { cachePoint: { type: "default" } },
|
|
anthropic: { cacheControl: { type: "ephemeral" } },
|
|
}
|
|
|
|
/**
|
|
* Check if a model supports prompt caching: Claude models, on Bedrock,
|
|
* the Anthropic API or OpenRouter (see CACHE_POINT).
|
|
*/
|
|
export function supportsPromptCaching(modelId: string): boolean {
|
|
return (
|
|
modelId.includes("claude") ||
|
|
modelId.includes("anthropic") ||
|
|
modelId.startsWith("us.anthropic") ||
|
|
modelId.startsWith("eu.anthropic")
|
|
)
|
|
}
|
|
|
|
/**
|
|
* Get the AI model for diagram validation.
|
|
* Uses VALIDATION_MODEL env var if set, otherwise falls back to AI_MODEL.
|
|
*
|
|
* Note: we no longer guess whether the model supports image input from its
|
|
* name — that heuristic misfired on newer models (see issue #874). If a
|
|
* configured validation model can't handle images, the API call simply errors
|
|
* and the validate-diagram route falls back to "valid".
|
|
*/
|
|
export function getValidationModel(): ReturnType<typeof getAIModel>["model"] {
|
|
// AI_MODEL may be comma-separated (multi-model fallback); pick the first.
|
|
const envFallback = process.env.AI_MODEL?.split(",")[0]?.trim() || undefined
|
|
const modelId = process.env.VALIDATION_MODEL || envFallback
|
|
|
|
if (!modelId) {
|
|
throw new Error(
|
|
"No validation model configured. Set VALIDATION_MODEL or AI_MODEL.",
|
|
)
|
|
}
|
|
|
|
// A default set in the admin panel becomes AI_PROVIDER/AI_MODEL, but its key
|
|
// lives in an ADMIN_-prefixed env var. Point at it the way the chat route
|
|
// does for server models, or the standard env var is required instead.
|
|
const panelDefault = adminProvidersToConfig(
|
|
loadAdminProviders(),
|
|
).providers.find((p) => p.default && p.provider === process.env.AI_PROVIDER)
|
|
|
|
const { model } = getAIModel({
|
|
modelId,
|
|
apiKeyEnv: panelDefault?.apiKeyEnv,
|
|
baseUrlEnv: panelDefault?.baseUrlEnv,
|
|
})
|
|
return model
|
|
}
|