diff --git a/app/api/validate-model/route.ts b/app/api/validate-model/route.ts index 6f3b5b90..0cf1eda2 100644 --- a/app/api/validate-model/route.ts +++ b/app/api/validate-model/route.ts @@ -15,6 +15,7 @@ import { isAihubmixStandardBaseURL, normalizeMiniMaxBaseURL, } from "@/lib/ai-providers" +import { resolveModelValidationMaxTokens } from "@/lib/model-validation" import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection" import { PROVIDER_INFO, type ProviderName } from "@/lib/types/model-config" @@ -31,6 +32,9 @@ interface ValidateRequest { awsRegion?: string // Vertex AI specific vertexApiKey?: string // Express Mode API key + // Optional test token budget override + maxTokens?: number + max_tokens?: number } export async function POST(req: Request) { @@ -47,6 +51,11 @@ export async function POST(req: Request) { // Note: Express Mode only needs vertexApiKey vertexApiKey, } = body + const validationMaxTokens = resolveModelValidationMaxTokens( + body.maxTokens, + body.max_tokens, + process.env.MODEL_VALIDATION_MAX_TOKENS, + ) if (!provider || !modelId) { return NextResponse.json( @@ -299,7 +308,7 @@ export async function POST(req: Request) { messages: [ { role: "user", content: "Say 'OK'" }, ], - max_tokens: 20, + max_tokens: validationMaxTokens, stream: true, enable_thinking: false, }), @@ -413,7 +422,7 @@ export async function POST(req: Request) { await generateText({ model, prompt: "Say 'OK'", - maxOutputTokens: 20, + maxOutputTokens: validationMaxTokens, }) const responseTime = Date.now() - startTime diff --git a/env.example b/env.example index fbf1d499..088f81fb 100644 --- a/env.example +++ b/env.example @@ -121,6 +121,11 @@ AI_MODEL=global.anthropic.claude-sonnet-4-5-20250929-v1:0 # Leave unset for models that don't support temperature (e.g., GPT-5.1 reasoning models) # TEMPERATURE=0 +# Model Validation (Optional) +# Token budget used when testing model availability. Increase for reasoning +# models that may emit thinking tokens before final content. +# MODEL_VALIDATION_MAX_TOKENS=1000 + # Access Control (Optional) # ACCESS_CODE_LIST=your-secret-code,another-code diff --git a/lib/admin/settings-registry.ts b/lib/admin/settings-registry.ts index 6419b3c5..fad1b99d 100644 --- a/lib/admin/settings-registry.ts +++ b/lib/admin/settings-registry.ts @@ -88,6 +88,17 @@ export const SETTINGS_REGISTRY: SettingDef[] = [ label: "Max Output Tokens", min: 1, }, + { + key: "MODEL_VALIDATION_MAX_TOKENS", + group: "generation", + type: "number", + label: "Model Validation Max Tokens", + description: + "Token budget for model test requests. Increase for reasoning models that may emit thinking tokens first.", + min: 1, + max: 64000, + default: "1000", + }, // ── Access Control ─────────────────────────────────────────────── { diff --git a/lib/i18n/dictionaries/en.json b/lib/i18n/dictionaries/en.json index 0de5b7b3..5a6abcd9 100644 --- a/lib/i18n/dictionaries/en.json +++ b/lib/i18n/dictionaries/en.json @@ -494,6 +494,10 @@ "MAX_OUTPUT_TOKENS": { "label": "Max Output Tokens" }, + "MODEL_VALIDATION_MAX_TOKENS": { + "label": "Model Validation Max Tokens", + "description": "Token budget for model test requests. Increase for reasoning models that may emit thinking tokens first." + }, "ACCESS_CODE_LIST": { "label": "Access Codes", "description": "Comma-separated list. Users must enter one to chat. Empty = open access." diff --git a/lib/i18n/dictionaries/ja.json b/lib/i18n/dictionaries/ja.json index 262d7c6a..b5b63135 100644 --- a/lib/i18n/dictionaries/ja.json +++ b/lib/i18n/dictionaries/ja.json @@ -494,6 +494,10 @@ "MAX_OUTPUT_TOKENS": { "label": "最大出力トークン数" }, + "MODEL_VALIDATION_MAX_TOKENS": { + "label": "モデル検証の最大トークン数", + "description": "モデルテストリクエストのトークン予算。推論モデルが先に思考トークンを出す場合は増やしてください。" + }, "ACCESS_CODE_LIST": { "label": "アクセスコード", "description": "カンマ区切りのリスト。チャットにはいずれかの入力が必要です。空 = オープンアクセス。" diff --git a/lib/i18n/dictionaries/zh-Hant.json b/lib/i18n/dictionaries/zh-Hant.json index 586dcc44..27715e97 100644 --- a/lib/i18n/dictionaries/zh-Hant.json +++ b/lib/i18n/dictionaries/zh-Hant.json @@ -494,6 +494,10 @@ "MAX_OUTPUT_TOKENS": { "label": "最大輸出 token 數" }, + "MODEL_VALIDATION_MAX_TOKENS": { + "label": "模型驗證最大 token 數", + "description": "模型測試請求的 token 預算。推理模型可能先輸出思考 token,可適當調大。" + }, "ACCESS_CODE_LIST": { "label": "存取碼", "description": "以逗號分隔的清單。使用者需輸入其中之一才能聊天。留空 = 開放存取。" diff --git a/lib/i18n/dictionaries/zh.json b/lib/i18n/dictionaries/zh.json index 38c64049..e617fb99 100644 --- a/lib/i18n/dictionaries/zh.json +++ b/lib/i18n/dictionaries/zh.json @@ -494,6 +494,10 @@ "MAX_OUTPUT_TOKENS": { "label": "最大输出 token 数" }, + "MODEL_VALIDATION_MAX_TOKENS": { + "label": "模型验证最大 token 数", + "description": "模型测试请求的 token 预算。推理模型可能先输出思考 token,可适当调大。" + }, "ACCESS_CODE_LIST": { "label": "访问码", "description": "以逗号分隔的列表。用户需输入其中之一才能聊天。留空 = 开放访问。" diff --git a/lib/model-validation.ts b/lib/model-validation.ts new file mode 100644 index 00000000..2bf98453 --- /dev/null +++ b/lib/model-validation.ts @@ -0,0 +1,28 @@ +export const DEFAULT_MODEL_VALIDATION_MAX_TOKENS = 1000 +export const MAX_MODEL_VALIDATION_MAX_TOKENS = 64000 + +export function parseModelValidationMaxTokens(value: unknown): number | null { + if (value === undefined || value === null || value === "") return null + + const numeric = + typeof value === "number" ? value : Number(String(value).trim()) + + if ( + !Number.isFinite(numeric) || + !Number.isInteger(numeric) || + numeric < 1 + ) { + return null + } + + return Math.min(numeric, MAX_MODEL_VALIDATION_MAX_TOKENS) +} + +export function resolveModelValidationMaxTokens(...sources: unknown[]): number { + for (const source of sources) { + const parsed = parseModelValidationMaxTokens(source) + if (parsed !== null) return parsed + } + + return DEFAULT_MODEL_VALIDATION_MAX_TOKENS +} diff --git a/tests/unit/model-validation.test.ts b/tests/unit/model-validation.test.ts new file mode 100644 index 00000000..867621d6 --- /dev/null +++ b/tests/unit/model-validation.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it } from "vitest" +import { + DEFAULT_MODEL_VALIDATION_MAX_TOKENS, + MAX_MODEL_VALIDATION_MAX_TOKENS, + parseModelValidationMaxTokens, + resolveModelValidationMaxTokens, +} from "@/lib/model-validation" + +describe("model validation token budget", () => { + it("defaults to a larger reasoning-friendly budget", () => { + expect(resolveModelValidationMaxTokens()).toBe( + DEFAULT_MODEL_VALIDATION_MAX_TOKENS, + ) + }) + + it("prefers the first valid source", () => { + expect(resolveModelValidationMaxTokens(2000, 3000)).toBe(2000) + expect(resolveModelValidationMaxTokens("bad", "3000")).toBe(3000) + }) + + it("rejects invalid values", () => { + expect(parseModelValidationMaxTokens(0)).toBeNull() + expect(parseModelValidationMaxTokens(-1)).toBeNull() + expect(parseModelValidationMaxTokens(1.5)).toBeNull() + expect(parseModelValidationMaxTokens("abc")).toBeNull() + }) + + it("caps very large values", () => { + expect( + resolveModelValidationMaxTokens( + MAX_MODEL_VALIDATION_MAX_TOKENS + 1, + ), + ).toBe(MAX_MODEL_VALIDATION_MAX_TOKENS) + }) +})