mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-09-01 17:10:24 +08:00
supportsImageInput() guessed multimodal capability from the model id string. The heuristic misfired on newer models (e.g. kimi-k3.6, qwen36), either wrongly rejecting images for capable models or letting them through. The AI SDK does not emit a warning when an OpenAI-compatible endpoint silently drops an image, so the guess was the only signal — but an unreliable one. Drop the detection entirely and let the real provider error surface instead (already translated to a friendly message in chat-panel.tsx). Validation falls back to "valid" on any model error. - Remove supportsImageInput() and its pre-send check in chat route - Drop the vision-capability throw in getValidationModel() - Remove the corresponding unit tests
This commit is contained in:
@@ -1419,77 +1419,14 @@ export function supportsPromptCaching(modelId: string): boolean {
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a model supports image/vision input.
|
||||
* Some models silently drop image parts without error (AI SDK warning only).
|
||||
*/
|
||||
export function supportsImageInput(modelId: string): boolean {
|
||||
const lowerModelId = modelId.toLowerCase()
|
||||
|
||||
// Helper to check if model has vision capability indicator
|
||||
const hasVisionIndicator =
|
||||
lowerModelId.includes("vision") || lowerModelId.includes("vl")
|
||||
|
||||
// Models that DON'T support image/vision input (unless vision variant)
|
||||
// Kimi K2 doesn't support images, but K2.5 does
|
||||
// Only block kimi-k2 specifically, not other Kimi models
|
||||
if (
|
||||
(lowerModelId.includes("kimi-k2") ||
|
||||
lowerModelId.includes("kimi_k2")) &&
|
||||
!hasVisionIndicator &&
|
||||
!lowerModelId.includes("2.5") &&
|
||||
!lowerModelId.includes("k2.5")
|
||||
) {
|
||||
return false
|
||||
}
|
||||
|
||||
// Moonshot text models (moonshot-v1 series are text-only)
|
||||
if (lowerModelId.includes("moonshot-v1") && !hasVisionIndicator) {
|
||||
return false
|
||||
}
|
||||
|
||||
// MiniMax text models (MiniMax-M2.x series are text-only; M3 supports image input)
|
||||
if (
|
||||
lowerModelId.includes("minimax") &&
|
||||
!hasVisionIndicator &&
|
||||
!lowerModelId.includes("m3")
|
||||
) {
|
||||
return false
|
||||
}
|
||||
|
||||
// DeepSeek text models (not vision variants)
|
||||
if (lowerModelId.includes("deepseek") && !hasVisionIndicator) {
|
||||
return false
|
||||
}
|
||||
|
||||
// Qwen text models (not vision variants like qwen-vl)
|
||||
// Qwen3.5 series (qwen3.5, qwen3.5-plus, qwen3.5-flash) natively support image input
|
||||
// QvQ (Qwen Visual QA) models are vision models — exclude them even when prefixed with "qwen/"
|
||||
if (
|
||||
lowerModelId.includes("qwen") &&
|
||||
!hasVisionIndicator &&
|
||||
!lowerModelId.includes("qwen3.5") &&
|
||||
!lowerModelId.includes("qvq")
|
||||
) {
|
||||
return false
|
||||
}
|
||||
|
||||
// GLM text models (not vision variants)
|
||||
// GLM vision models: glm-4v, glm-4v-9b, glm-4.1v-9b-thinking
|
||||
if (lowerModelId.includes("glm") && !hasVisionIndicator) {
|
||||
if (!/[\d.]v/.test(lowerModelId)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// Default: assume model supports images
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the AI model for diagram validation.
|
||||
* Uses VALIDATION_MODEL env var if set, otherwise falls back to AI_MODEL.
|
||||
* Throws if the model doesn't support image input.
|
||||
*
|
||||
* Note: we no longer guess whether the model supports image input from its
|
||||
* name — that heuristic misfired on newer models (see issue #874). If a
|
||||
* configured validation model can't handle images, the API call simply errors
|
||||
* and the validate-diagram route falls back to "valid".
|
||||
*/
|
||||
export function getValidationModel(): ReturnType<typeof getAIModel>["model"] {
|
||||
// AI_MODEL may be comma-separated (multi-model fallback); pick the first.
|
||||
@@ -1502,12 +1439,6 @@ export function getValidationModel(): ReturnType<typeof getAIModel>["model"] {
|
||||
)
|
||||
}
|
||||
|
||||
if (!supportsImageInput(modelId)) {
|
||||
throw new Error(
|
||||
`Validation requires a vision-capable model. Model "${modelId}" does not support image input.`,
|
||||
)
|
||||
}
|
||||
|
||||
const { model } = getAIModel({ modelId })
|
||||
return model
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user