mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-10-08 18:57:47 +08:00
fix: make model validation token budget configurable
This commit is contained in:
@@ -15,6 +15,7 @@ import {
|
||||
isAihubmixStandardBaseURL,
|
||||
normalizeMiniMaxBaseURL,
|
||||
} from "@/lib/ai-providers"
|
||||
import { resolveModelValidationMaxTokens } from "@/lib/model-validation"
|
||||
import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection"
|
||||
import { PROVIDER_INFO, type ProviderName } from "@/lib/types/model-config"
|
||||
|
||||
@@ -31,6 +32,9 @@ interface ValidateRequest {
|
||||
awsRegion?: string
|
||||
// Vertex AI specific
|
||||
vertexApiKey?: string // Express Mode API key
|
||||
// Optional test token budget override
|
||||
maxTokens?: number
|
||||
max_tokens?: number
|
||||
}
|
||||
|
||||
export async function POST(req: Request) {
|
||||
@@ -47,6 +51,11 @@ export async function POST(req: Request) {
|
||||
// Note: Express Mode only needs vertexApiKey
|
||||
vertexApiKey,
|
||||
} = body
|
||||
const validationMaxTokens = resolveModelValidationMaxTokens(
|
||||
body.maxTokens,
|
||||
body.max_tokens,
|
||||
process.env.MODEL_VALIDATION_MAX_TOKENS,
|
||||
)
|
||||
|
||||
if (!provider || !modelId) {
|
||||
return NextResponse.json(
|
||||
@@ -299,7 +308,7 @@ export async function POST(req: Request) {
|
||||
messages: [
|
||||
{ role: "user", content: "Say 'OK'" },
|
||||
],
|
||||
max_tokens: 20,
|
||||
max_tokens: validationMaxTokens,
|
||||
stream: true,
|
||||
enable_thinking: false,
|
||||
}),
|
||||
@@ -413,7 +422,7 @@ export async function POST(req: Request) {
|
||||
await generateText({
|
||||
model,
|
||||
prompt: "Say 'OK'",
|
||||
maxOutputTokens: 20,
|
||||
maxOutputTokens: validationMaxTokens,
|
||||
})
|
||||
const responseTime = Date.now() - startTime
|
||||
|
||||
|
||||
@@ -121,6 +121,11 @@ AI_MODEL=global.anthropic.claude-sonnet-4-5-20250929-v1:0
|
||||
# Leave unset for models that don't support temperature (e.g., GPT-5.1 reasoning models)
|
||||
# TEMPERATURE=0
|
||||
|
||||
# Model Validation (Optional)
|
||||
# Token budget used when testing model availability. Increase for reasoning
|
||||
# models that may emit thinking tokens before final content.
|
||||
# MODEL_VALIDATION_MAX_TOKENS=1000
|
||||
|
||||
# Access Control (Optional)
|
||||
# ACCESS_CODE_LIST=your-secret-code,another-code
|
||||
|
||||
|
||||
@@ -88,6 +88,17 @@ export const SETTINGS_REGISTRY: SettingDef[] = [
|
||||
label: "Max Output Tokens",
|
||||
min: 1,
|
||||
},
|
||||
{
|
||||
key: "MODEL_VALIDATION_MAX_TOKENS",
|
||||
group: "generation",
|
||||
type: "number",
|
||||
label: "Model Validation Max Tokens",
|
||||
description:
|
||||
"Token budget for model test requests. Increase for reasoning models that may emit thinking tokens first.",
|
||||
min: 1,
|
||||
max: 64000,
|
||||
default: "1000",
|
||||
},
|
||||
|
||||
// ── Access Control ───────────────────────────────────────────────
|
||||
{
|
||||
|
||||
@@ -494,6 +494,10 @@
|
||||
"MAX_OUTPUT_TOKENS": {
|
||||
"label": "Max Output Tokens"
|
||||
},
|
||||
"MODEL_VALIDATION_MAX_TOKENS": {
|
||||
"label": "Model Validation Max Tokens",
|
||||
"description": "Token budget for model test requests. Increase for reasoning models that may emit thinking tokens first."
|
||||
},
|
||||
"ACCESS_CODE_LIST": {
|
||||
"label": "Access Codes",
|
||||
"description": "Comma-separated list. Users must enter one to chat. Empty = open access."
|
||||
|
||||
@@ -494,6 +494,10 @@
|
||||
"MAX_OUTPUT_TOKENS": {
|
||||
"label": "最大出力トークン数"
|
||||
},
|
||||
"MODEL_VALIDATION_MAX_TOKENS": {
|
||||
"label": "モデル検証の最大トークン数",
|
||||
"description": "モデルテストリクエストのトークン予算。推論モデルが先に思考トークンを出す場合は増やしてください。"
|
||||
},
|
||||
"ACCESS_CODE_LIST": {
|
||||
"label": "アクセスコード",
|
||||
"description": "カンマ区切りのリスト。チャットにはいずれかの入力が必要です。空 = オープンアクセス。"
|
||||
|
||||
@@ -494,6 +494,10 @@
|
||||
"MAX_OUTPUT_TOKENS": {
|
||||
"label": "最大輸出 token 數"
|
||||
},
|
||||
"MODEL_VALIDATION_MAX_TOKENS": {
|
||||
"label": "模型驗證最大 token 數",
|
||||
"description": "模型測試請求的 token 預算。推理模型可能先輸出思考 token,可適當調大。"
|
||||
},
|
||||
"ACCESS_CODE_LIST": {
|
||||
"label": "存取碼",
|
||||
"description": "以逗號分隔的清單。使用者需輸入其中之一才能聊天。留空 = 開放存取。"
|
||||
|
||||
@@ -494,6 +494,10 @@
|
||||
"MAX_OUTPUT_TOKENS": {
|
||||
"label": "最大输出 token 数"
|
||||
},
|
||||
"MODEL_VALIDATION_MAX_TOKENS": {
|
||||
"label": "模型验证最大 token 数",
|
||||
"description": "模型测试请求的 token 预算。推理模型可能先输出思考 token,可适当调大。"
|
||||
},
|
||||
"ACCESS_CODE_LIST": {
|
||||
"label": "访问码",
|
||||
"description": "以逗号分隔的列表。用户需输入其中之一才能聊天。留空 = 开放访问。"
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
export const DEFAULT_MODEL_VALIDATION_MAX_TOKENS = 1000
|
||||
export const MAX_MODEL_VALIDATION_MAX_TOKENS = 64000
|
||||
|
||||
export function parseModelValidationMaxTokens(value: unknown): number | null {
|
||||
if (value === undefined || value === null || value === "") return null
|
||||
|
||||
const numeric =
|
||||
typeof value === "number" ? value : Number(String(value).trim())
|
||||
|
||||
if (
|
||||
!Number.isFinite(numeric) ||
|
||||
!Number.isInteger(numeric) ||
|
||||
numeric < 1
|
||||
) {
|
||||
return null
|
||||
}
|
||||
|
||||
return Math.min(numeric, MAX_MODEL_VALIDATION_MAX_TOKENS)
|
||||
}
|
||||
|
||||
export function resolveModelValidationMaxTokens(...sources: unknown[]): number {
|
||||
for (const source of sources) {
|
||||
const parsed = parseModelValidationMaxTokens(source)
|
||||
if (parsed !== null) return parsed
|
||||
}
|
||||
|
||||
return DEFAULT_MODEL_VALIDATION_MAX_TOKENS
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
import { describe, expect, it } from "vitest"
|
||||
import {
|
||||
DEFAULT_MODEL_VALIDATION_MAX_TOKENS,
|
||||
MAX_MODEL_VALIDATION_MAX_TOKENS,
|
||||
parseModelValidationMaxTokens,
|
||||
resolveModelValidationMaxTokens,
|
||||
} from "@/lib/model-validation"
|
||||
|
||||
describe("model validation token budget", () => {
|
||||
it("defaults to a larger reasoning-friendly budget", () => {
|
||||
expect(resolveModelValidationMaxTokens()).toBe(
|
||||
DEFAULT_MODEL_VALIDATION_MAX_TOKENS,
|
||||
)
|
||||
})
|
||||
|
||||
it("prefers the first valid source", () => {
|
||||
expect(resolveModelValidationMaxTokens(2000, 3000)).toBe(2000)
|
||||
expect(resolveModelValidationMaxTokens("bad", "3000")).toBe(3000)
|
||||
})
|
||||
|
||||
it("rejects invalid values", () => {
|
||||
expect(parseModelValidationMaxTokens(0)).toBeNull()
|
||||
expect(parseModelValidationMaxTokens(-1)).toBeNull()
|
||||
expect(parseModelValidationMaxTokens(1.5)).toBeNull()
|
||||
expect(parseModelValidationMaxTokens("abc")).toBeNull()
|
||||
})
|
||||
|
||||
it("caps very large values", () => {
|
||||
expect(
|
||||
resolveModelValidationMaxTokens(
|
||||
MAX_MODEL_VALIDATION_MAX_TOKENS + 1,
|
||||
),
|
||||
).toBe(MAX_MODEL_VALIDATION_MAX_TOKENS)
|
||||
})
|
||||
})
|
||||
Reference in New Issue
Block a user