fix: make model validation token budget configurable

This commit is contained in:
chaochaoweb3
2026-07-02 15:48:44 +08:00
parent 5bfd7b2468
commit c1e4b1c9bf
9 changed files with 106 additions and 2 deletions
+11 -2
View File
@@ -15,6 +15,7 @@ import {
isAihubmixStandardBaseURL,
normalizeMiniMaxBaseURL,
} from "@/lib/ai-providers"
import { resolveModelValidationMaxTokens } from "@/lib/model-validation"
import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection"
import { PROVIDER_INFO, type ProviderName } from "@/lib/types/model-config"
@@ -31,6 +32,9 @@ interface ValidateRequest {
awsRegion?: string
// Vertex AI specific
vertexApiKey?: string // Express Mode API key
// Optional test token budget override
maxTokens?: number
max_tokens?: number
}
export async function POST(req: Request) {
@@ -47,6 +51,11 @@ export async function POST(req: Request) {
// Note: Express Mode only needs vertexApiKey
vertexApiKey,
} = body
const validationMaxTokens = resolveModelValidationMaxTokens(
body.maxTokens,
body.max_tokens,
process.env.MODEL_VALIDATION_MAX_TOKENS,
)
if (!provider || !modelId) {
return NextResponse.json(
@@ -299,7 +308,7 @@ export async function POST(req: Request) {
messages: [
{ role: "user", content: "Say 'OK'" },
],
max_tokens: 20,
max_tokens: validationMaxTokens,
stream: true,
enable_thinking: false,
}),
@@ -413,7 +422,7 @@ export async function POST(req: Request) {
await generateText({
model,
prompt: "Say 'OK'",
maxOutputTokens: 20,
maxOutputTokens: validationMaxTokens,
})
const responseTime = Date.now() - startTime
+5
View File
@@ -121,6 +121,11 @@ AI_MODEL=global.anthropic.claude-sonnet-4-5-20250929-v1:0
# Leave unset for models that don't support temperature (e.g., GPT-5.1 reasoning models)
# TEMPERATURE=0
# Model Validation (Optional)
# Token budget used when testing model availability. Increase for reasoning
# models that may emit thinking tokens before final content.
# MODEL_VALIDATION_MAX_TOKENS=1000
# Access Control (Optional)
# ACCESS_CODE_LIST=your-secret-code,another-code
+11
View File
@@ -88,6 +88,17 @@ export const SETTINGS_REGISTRY: SettingDef[] = [
label: "Max Output Tokens",
min: 1,
},
{
key: "MODEL_VALIDATION_MAX_TOKENS",
group: "generation",
type: "number",
label: "Model Validation Max Tokens",
description:
"Token budget for model test requests. Increase for reasoning models that may emit thinking tokens first.",
min: 1,
max: 64000,
default: "1000",
},
// ── Access Control ───────────────────────────────────────────────
{
+4
View File
@@ -494,6 +494,10 @@
"MAX_OUTPUT_TOKENS": {
"label": "Max Output Tokens"
},
"MODEL_VALIDATION_MAX_TOKENS": {
"label": "Model Validation Max Tokens",
"description": "Token budget for model test requests. Increase for reasoning models that may emit thinking tokens first."
},
"ACCESS_CODE_LIST": {
"label": "Access Codes",
"description": "Comma-separated list. Users must enter one to chat. Empty = open access."
+4
View File
@@ -494,6 +494,10 @@
"MAX_OUTPUT_TOKENS": {
"label": "最大出力トークン数"
},
"MODEL_VALIDATION_MAX_TOKENS": {
"label": "モデル検証の最大トークン数",
"description": "モデルテストリクエストのトークン予算。推論モデルが先に思考トークンを出す場合は増やしてください。"
},
"ACCESS_CODE_LIST": {
"label": "アクセスコード",
"description": "カンマ区切りのリスト。チャットにはいずれかの入力が必要です。空 = オープンアクセス。"
+4
View File
@@ -494,6 +494,10 @@
"MAX_OUTPUT_TOKENS": {
"label": "最大輸出 token 數"
},
"MODEL_VALIDATION_MAX_TOKENS": {
"label": "模型驗證最大 token 數",
"description": "模型測試請求的 token 預算。推理模型可能先輸出思考 token,可適當調大。"
},
"ACCESS_CODE_LIST": {
"label": "存取碼",
"description": "以逗號分隔的清單。使用者需輸入其中之一才能聊天。留空 = 開放存取。"
+4
View File
@@ -494,6 +494,10 @@
"MAX_OUTPUT_TOKENS": {
"label": "最大输出 token 数"
},
"MODEL_VALIDATION_MAX_TOKENS": {
"label": "模型验证最大 token 数",
"description": "模型测试请求的 token 预算。推理模型可能先输出思考 token,可适当调大。"
},
"ACCESS_CODE_LIST": {
"label": "访问码",
"description": "以逗号分隔的列表。用户需输入其中之一才能聊天。留空 = 开放访问。"
+28
View File
@@ -0,0 +1,28 @@
export const DEFAULT_MODEL_VALIDATION_MAX_TOKENS = 1000
export const MAX_MODEL_VALIDATION_MAX_TOKENS = 64000
export function parseModelValidationMaxTokens(value: unknown): number | null {
if (value === undefined || value === null || value === "") return null
const numeric =
typeof value === "number" ? value : Number(String(value).trim())
if (
!Number.isFinite(numeric) ||
!Number.isInteger(numeric) ||
numeric < 1
) {
return null
}
return Math.min(numeric, MAX_MODEL_VALIDATION_MAX_TOKENS)
}
export function resolveModelValidationMaxTokens(...sources: unknown[]): number {
for (const source of sources) {
const parsed = parseModelValidationMaxTokens(source)
if (parsed !== null) return parsed
}
return DEFAULT_MODEL_VALIDATION_MAX_TOKENS
}
+35
View File
@@ -0,0 +1,35 @@
import { describe, expect, it } from "vitest"
import {
DEFAULT_MODEL_VALIDATION_MAX_TOKENS,
MAX_MODEL_VALIDATION_MAX_TOKENS,
parseModelValidationMaxTokens,
resolveModelValidationMaxTokens,
} from "@/lib/model-validation"
describe("model validation token budget", () => {
it("defaults to a larger reasoning-friendly budget", () => {
expect(resolveModelValidationMaxTokens()).toBe(
DEFAULT_MODEL_VALIDATION_MAX_TOKENS,
)
})
it("prefers the first valid source", () => {
expect(resolveModelValidationMaxTokens(2000, 3000)).toBe(2000)
expect(resolveModelValidationMaxTokens("bad", "3000")).toBe(3000)
})
it("rejects invalid values", () => {
expect(parseModelValidationMaxTokens(0)).toBeNull()
expect(parseModelValidationMaxTokens(-1)).toBeNull()
expect(parseModelValidationMaxTokens(1.5)).toBeNull()
expect(parseModelValidationMaxTokens("abc")).toBeNull()
})
it("caps very large values", () => {
expect(
resolveModelValidationMaxTokens(
MAX_MODEL_VALIDATION_MAX_TOKENS + 1,
),
).toBe(MAX_MODEL_VALIDATION_MAX_TOKENS)
})
})