Compare commits

..
Author SHA1 Message Date
dayuan.jiang 886e748aa7 fix: preserve charset detection and block CGNAT range in parse-url SSRF fix
Follow-up to the multi-reviewer review of the SSRF fix:

- Restore charset handling lost when switching from extract() to
  response.text(): non-UTF-8 pages (Shift_JIS/GBK/EUC/Big5, common on CJK
  sites this project targets) decoded as mojibake. Now read the body as
  bytes, detect charset from Content-Type / <meta charset>, and decode
  with TextDecoder before extractFromHtml.
- Wrap extractFromHtml in try/catch: it throws (not returns null) on
  empty/non-HTML bodies, which previously surfaced as a 500 instead of the
  intended 400.
- Add 100.64.0.0/10 (RFC 6598 CGNAT) to isPrivateIp; it is routable inside
  some cloud internal networks and was a residual SSRF target.
- Add tests for CGNAT, its boundaries, 0.0.0.0, and DNS-resolved IPv6.
2026-06-28 00:32:24 +09:00
dayuan.jiang 73862f6108 fix: resolve DNS before SSRF check and block redirects in parse-url
isPrivateUrl() did string-only hostname matching and never resolved DNS,
so a public-looking name that maps to an internal IP (e.g.
127-0-0-1.sslip.io -> 127.0.0.1) passed the check while fetch/extract
later resolved it and reached internal services (GHSA-wqcv-5qvx-vx75).

- isPrivateUrl is now async: it keeps the fast string/literal-IP path,
  then resolves the hostname via DNS and rejects if any address is private.
- parse-url now fetches the page itself with redirect: "error" and parses
  via extractFromHtml(), since article-extractor follows redirects
  internally and drops a redirect option, which allowed a public URL to
  302 to an internal host.
- Update validate-model call site to await; add regression tests.
2026-06-27 18:03:28 +09:00
24 changed files with 237 additions and 846 deletions
-67
View File
@@ -1,67 +0,0 @@
name: Publish MCP Server
# Publishes @next-ai-drawio/mcp-server to npm via OIDC trusted publishing
# (no token, no OTP). Triggers when packages/mcp-server changes on main;
# skips silently if the package.json version is already on npm — so a
# release is just "bump the version in a PR and merge".
on:
push:
branches:
- main
paths:
- "packages/mcp-server/**"
workflow_dispatch:
permissions:
contents: read
id-token: write # OIDC token for npm trusted publishing
concurrency:
group: publish-mcp
cancel-in-progress: false
jobs:
publish:
runs-on: ubuntu-latest
defaults:
run:
working-directory: packages/mcp-server
steps:
- name: Checkout
uses: actions/checkout@v6
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version: 24
cache: "npm"
cache-dependency-path: packages/mcp-server/package-lock.json
registry-url: "https://registry.npmjs.org"
# Trusted publishing requires npm >= 11.5.1
- name: Update npm
run: npm install -g npm@latest
- name: Check if version is already published
id: version
run: |
LOCAL=$(node -p "require('./package.json').version")
if npm view "@next-ai-drawio/mcp-server@${LOCAL}" version >/dev/null 2>&1; then
echo "Version ${LOCAL} already on npm - nothing to publish"
echo "publish=false" >> "$GITHUB_OUTPUT"
else
echo "Version ${LOCAL} not on npm - publishing"
echo "publish=true" >> "$GITHUB_OUTPUT"
fi
- name: Install dependencies
if: steps.version.outputs.publish == 'true'
run: npm ci
- name: Test
if: steps.version.outputs.publish == 'true'
run: npm test
- name: Publish to npm
if: steps.version.outputs.publish == 'true'
run: npm publish
+11 -4
View File
@@ -15,6 +15,7 @@ import { z } from "zod"
import {
getAIModel,
SINGLE_SYSTEM_PROVIDERS,
supportsImageInput,
supportsPromptCaching,
} from "@/lib/ai-providers"
import { findCachedResponse } from "@/lib/cached-responses"
@@ -265,10 +266,16 @@ async function handleChatRequest(req: Request): Promise<Response> {
lastUserMessage?.parts?.filter((part: any) => part.type === "file") ||
[]
// Note: we used to pre-emptively reject images for models we guessed were
// text-only (by name matching). That heuristic misfired on newer models
// (see issue #874), so we now let the request through and surface the real
// provider error if the model genuinely can't accept images.
// Check if user is sending images to a model that doesn't support them
// AI SDK silently drops unsupported parts, so we need to catch this early
if (fileParts.length > 0 && !supportsImageInput(modelId)) {
return Response.json(
{
error: `The model "${modelId}" does not support image input. Please use a vision-capable model (e.g., GPT-4o, Claude, Gemini) or remove the image.`,
},
{ status: 400 },
)
}
// User input only - XML is now in a separate cached system message
const formattedUserInput = `User input:
+2 -3
View File
@@ -372,13 +372,12 @@ export async function POST(req: Request) {
break
}
// GLM, Qwen, Kimi, Qiniu, Novita, MiMo - OpenAI compatible
// GLM, Qwen, Kimi, Qiniu, Novita - OpenAI compatible
case "glm":
case "qwen":
case "kimi":
case "qiniu":
case "novita":
case "mimo": {
case "novita": {
const baseURL =
baseUrl ||
PROVIDER_INFO[provider as ProviderName]?.defaultBaseUrl ||
@@ -249,11 +249,6 @@ export function ProviderCredentialsFields({
{dict.modelConfig.minimaxBaseUrlHint}
</p>
)}
{provider === "mimo" && (
<p className="text-xs text-muted-foreground">
{dict.modelConfig.mimoBaseUrlHint}
</p>
)}
</div>
</>
)}
+1 -14
View File
@@ -308,19 +308,6 @@ AI_MODEL=your_model_id
QINIU_BASE_URL=https://your-custom-endpoint
```
### MiMo (小米)
```bash
MIMO_API_KEY=your_api_key
AI_MODEL=mimo-v2.5-pro
```
可选的自定义端点(Token Plan 订阅用户请设置专属 Base URL):
```bash
MIMO_BASE_URL=https://token-plan-cn.xiaomimimo.com/v1
```
## 自动检测
如果您只配置了**一个**提供商的 API 密钥,系统将自动检测并使用该提供商。无需设置 `AI_PROVIDER`。
@@ -328,7 +315,7 @@ MIMO_BASE_URL=https://token-plan-cn.xiaomimimo.com/v1
如果您配置了**多个** API 密钥,则必须显式设置 `AI_PROVIDER`:
```bash
AI_PROVIDER=google # 或:openai, anthropic, aihubmix, deepseek, siliconflow, doubao, azure, bedrock, openrouter, ollama, gateway, sglang, modelscope, minimax, glm, qwen, kimi, qiniu, mimo
AI_PROVIDER=google # 或:openai, anthropic, aihubmix, deepseek, siliconflow, doubao, azure, bedrock, openrouter, ollama, gateway, sglang, modelscope, minimax, glm, qwen, kimi, qiniu
```
## 服务端多模型配置
+1 -14
View File
@@ -323,19 +323,6 @@ Optional custom endpoint:
QINIU_BASE_URL=https://your-custom-endpoint
```
### MiMo (Xiaomi)
```bash
MIMO_API_KEY=your_api_key
AI_MODEL=mimo-v2.5-pro
```
Optional custom endpoint (Token Plan subscribers should set their dedicated Base URL):
```bash
MIMO_BASE_URL=https://token-plan-cn.xiaomimimo.com/v1
```
## Auto-Detection
If you only configure **one** provider's API key, the system will automatically detect and use that provider. No need to set `AI_PROVIDER`.
@@ -343,7 +330,7 @@ If you only configure **one** provider's API key, the system will automatically
If you configure **multiple** API keys, you must explicitly set `AI_PROVIDER`:
```bash
AI_PROVIDER=google # or: openai, anthropic, aihubmix, deepseek, siliconflow, doubao, azure, bedrock, openrouter, ollama, gateway, sglang, modelscope, minimax, glm, qwen, kimi, qiniu, mimo
AI_PROVIDER=google # or: openai, anthropic, aihubmix, deepseek, siliconflow, doubao, azure, bedrock, openrouter, ollama, gateway, sglang, modelscope, minimax, glm, qwen, kimi, qiniu
```
## Server-Side Multi-Model Configuration
+1 -14
View File
@@ -308,19 +308,6 @@ AI_MODEL=your_model_id
QINIU_BASE_URL=https://your-custom-endpoint
```
### MiMo (Xiaomi)
```bash
MIMO_API_KEY=your_api_key
AI_MODEL=mimo-v2.5-pro
```
オプションのカスタムエンドポイント(Token Plan 加入者は専用の Base URL を設定してください):
```bash
MIMO_BASE_URL=https://token-plan-cn.xiaomimimo.com/v1
```
## 自動検出
**1つ**のプロバイダーの API キーのみを設定した場合、システムはそのプロバイダーを自動的に検出して使用します。`AI_PROVIDER` を設定する必要はありません。
@@ -328,7 +315,7 @@ MIMO_BASE_URL=https://token-plan-cn.xiaomimimo.com/v1
**複数**の API キーを設定する場合は、`AI_PROVIDER` を明示的に設定する必要があります:
```bash
AI_PROVIDER=google # または: openai, anthropic, aihubmix, deepseek, siliconflow, doubao, azure, bedrock, openrouter, ollama, gateway, sglang, modelscope, minimax, glm, qwen, kimi, qiniu, mimo
AI_PROVIDER=google # または: openai, anthropic, aihubmix, deepseek, siliconflow, doubao, azure, bedrock, openrouter, ollama, gateway, sglang, modelscope, minimax, glm, qwen, kimi, qiniu
```
## サーバーサイドマルチモデル設定
-5
View File
@@ -189,8 +189,3 @@ AI_MODEL=global.anthropic.claude-sonnet-4-5-20250929-v1:0
# Get your API key from: https://novita.ai/dashboard/key
# NOVITA_API_KEY=your_novita_api_key
# NOVITA_BASE_URL=https://api.novita.ai/openai # Optional, default
# MiMo (Xiaomi) Configuration (Optional)
# Get your API key from: https://platform.xiaomimimo.com/
# MIMO_API_KEY=your_mimo_api_key
# MIMO_BASE_URL=https://api.xiaomimimo.com/v1 # Optional, default. Token Plan users: https://token-plan-cn.xiaomimimo.com/v1
+76 -28
View File
@@ -32,7 +32,6 @@ export const SINGLE_SYSTEM_PROVIDERS = new Set<ProviderName>([
"kimi",
"qiniu",
"novita",
"mimo",
])
/**
@@ -117,7 +116,6 @@ const ALLOWED_CLIENT_PROVIDERS: ProviderName[] = [
"kimi",
"minimax",
"novita",
"mimo",
]
// Bedrock provider options for Anthropic beta features
@@ -542,8 +540,7 @@ function buildProviderOptions(
case "qwen":
case "kimi":
case "qiniu":
case "novita":
case "mimo": {
case "novita": {
// These providers don't have reasoning configs in AI SDK yet
// Gateway passes through to underlying providers which handle their own configs
break
@@ -580,7 +577,6 @@ export const PROVIDER_ENV_VARS: Record<ProviderName, string | null> = {
kimi: "KIMI_API_KEY",
minimax: "MINIMAX_API_KEY",
novita: "NOVITA_API_KEY",
mimo: "MIMO_API_KEY",
}
/**
@@ -1350,23 +1346,6 @@ export function getAIModel(overrides?: ClientOverrides): ModelConfig {
break
}
case "mimo": {
const apiKey = resolveApiKey(overrides, "MIMO_API_KEY")
const baseURL = resolveBaseURL(
overrides?.apiKey,
overrides?.baseUrl,
resolveBaseUrlEnv(overrides, "MIMO_BASE_URL"),
PROVIDER_INFO.mimo?.defaultBaseUrl,
)
// Use createDeepSeek to properly handle reasoning_content for MiMo
// thinking models (e.g., mimo-v2.5-pro). MiMo's API requires
// reasoning_content to be passed back during multi-turn tool calls
// (returns 400 otherwise), same convention as DeepSeek and Kimi.
const mimoProvider = createDeepSeek({ apiKey, baseURL })
model = mimoProvider(modelId)
break
}
case "glm":
case "qwen":
case "qiniu":
@@ -1414,7 +1393,7 @@ export function getAIModel(overrides?: ClientOverrides): ModelConfig {
default:
throw new Error(
`Unknown AI provider: ${provider}. Supported providers: bedrock, openai, anthropic, google, azure, ollama, openrouter, aihubmix, deepseek, siliconflow, sglang, gateway, edgeone, doubao, modelscope, glm, qwen, qiniu, kimi, minimax, novita, mimo`,
`Unknown AI provider: ${provider}. Supported providers: bedrock, openai, anthropic, google, azure, ollama, openrouter, aihubmix, deepseek, siliconflow, sglang, gateway, edgeone, doubao, modelscope, glm, qwen, qiniu, kimi, minimax, novita`,
)
}
@@ -1440,14 +1419,77 @@ export function supportsPromptCaching(modelId: string): boolean {
)
}
/**
* Check if a model supports image/vision input.
* Some models silently drop image parts without error (AI SDK warning only).
*/
export function supportsImageInput(modelId: string): boolean {
const lowerModelId = modelId.toLowerCase()
// Helper to check if model has vision capability indicator
const hasVisionIndicator =
lowerModelId.includes("vision") || lowerModelId.includes("vl")
// Models that DON'T support image/vision input (unless vision variant)
// Kimi K2 doesn't support images, but K2.5 does
// Only block kimi-k2 specifically, not other Kimi models
if (
(lowerModelId.includes("kimi-k2") ||
lowerModelId.includes("kimi_k2")) &&
!hasVisionIndicator &&
!lowerModelId.includes("2.5") &&
!lowerModelId.includes("k2.5")
) {
return false
}
// Moonshot text models (moonshot-v1 series are text-only)
if (lowerModelId.includes("moonshot-v1") && !hasVisionIndicator) {
return false
}
// MiniMax text models (MiniMax-M2.x series are text-only; M3 supports image input)
if (
lowerModelId.includes("minimax") &&
!hasVisionIndicator &&
!lowerModelId.includes("m3")
) {
return false
}
// DeepSeek text models (not vision variants)
if (lowerModelId.includes("deepseek") && !hasVisionIndicator) {
return false
}
// Qwen text models (not vision variants like qwen-vl)
// Qwen3.5 series (qwen3.5, qwen3.5-plus, qwen3.5-flash) natively support image input
// QvQ (Qwen Visual QA) models are vision models — exclude them even when prefixed with "qwen/"
if (
lowerModelId.includes("qwen") &&
!hasVisionIndicator &&
!lowerModelId.includes("qwen3.5") &&
!lowerModelId.includes("qvq")
) {
return false
}
// GLM text models (not vision variants)
// GLM vision models: glm-4v, glm-4v-9b, glm-4.1v-9b-thinking
if (lowerModelId.includes("glm") && !hasVisionIndicator) {
if (!/[\d.]v/.test(lowerModelId)) {
return false
}
}
// Default: assume model supports images
return true
}
/**
* Get the AI model for diagram validation.
* Uses VALIDATION_MODEL env var if set, otherwise falls back to AI_MODEL.
*
* Note: we no longer guess whether the model supports image input from its
* name — that heuristic misfired on newer models (see issue #874). If a
* configured validation model can't handle images, the API call simply errors
* and the validate-diagram route falls back to "valid".
* Throws if the model doesn't support image input.
*/
export function getValidationModel(): ReturnType<typeof getAIModel>["model"] {
// AI_MODEL may be comma-separated (multi-model fallback); pick the first.
@@ -1460,6 +1502,12 @@ export function getValidationModel(): ReturnType<typeof getAIModel>["model"] {
)
}
if (!supportsImageInput(modelId)) {
throw new Error(
`Validation requires a vision-capable model. Model "${modelId}" does not support image input.`,
)
}
const { model } = getAIModel({ modelId })
return model
}
+1 -3
View File
@@ -34,8 +34,7 @@
"glm": "GLM",
"qwen": "Qwen",
"kimi": "Kimi",
"qiniu": "Qiniu",
"mimo": "MiMo (Xiaomi)"
"qiniu": "Qiniu"
},
"chat": {
"placeholder": "Describe your diagram or upload a file...",
@@ -372,7 +371,6 @@
"baseUrlWithExample": "Base URL (optional, e.g. {example})",
"customEndpoint": "Custom endpoint URL",
"minimaxBaseUrlHint": "Use /anthropic for Anthropic-compatible API (recommended), or /v1 for OpenAI-compatible API",
"mimoBaseUrlHint": "Default works with pay-as-you-go keys (sk-...). Token Plan subscribers (tp-... keys) must set https://token-plan-cn.xiaomimimo.com/v1",
"models": "Models",
"customModelId": "Custom model ID...",
"allAdded": "All added",
+1 -3
View File
@@ -34,8 +34,7 @@
"glm": "GLM",
"qwen": "Qwen",
"kimi": "Kimi",
"qiniu": "Qiniu",
"mimo": "MiMo (Xiaomi)"
"qiniu": "Qiniu"
},
"chat": {
"placeholder": "ダイアグラムを説明するか、ファイルをアップロード...",
@@ -326,7 +325,6 @@
"baseUrlWithExample": "ベース URL(オプション、例: {example})",
"customEndpoint": "カスタムエンドポイント URL",
"minimaxBaseUrlHint": "/anthropic で Anthropic 互換 API(推奨)、または /v1 で OpenAI 互換 API を使用",
"mimoBaseUrlHint": "デフォルトは従量課金キー(sk-...)用です。Token Plan 加入者(tp-... キー)は https://token-plan-cn.xiaomimimo.com/v1 を設定してください",
"models": "モデル",
"customModelId": "カスタムモデル ID...",
"allAdded": "すべて追加済み",
+1 -3
View File
@@ -34,8 +34,7 @@
"glm": "GLM",
"qwen": "Qwen",
"kimi": "Kimi",
"qiniu": "Qiniu",
"mimo": "MiMo (小米)"
"qiniu": "Qiniu"
},
"chat": {
"placeholder": "描述您的圖表或上傳檔案...",
@@ -372,7 +371,6 @@
"baseUrlWithExample": "基礎 URL(可選,例如 {example})",
"customEndpoint": "自訂端點 URL",
"minimaxBaseUrlHint": "使用 /anthropic 端點為 Anthropic 相容 API(推薦),或使用 /v1 端點為 OpenAI 相容 API",
"mimoBaseUrlHint": "預設地址適用於按量付費金鑰(sk-...)。Token Plan 訂閱用戶(tp-... 金鑰)請設定為 https://token-plan-cn.xiaomimimo.com/v1",
"models": "模型",
"customModelId": "自訂模型 ID...",
"allAdded": "已全部新增",
+1 -3
View File
@@ -34,8 +34,7 @@
"glm": "GLM",
"qwen": "Qwen",
"kimi": "Kimi",
"qiniu": "Qiniu",
"mimo": "MiMo (小米)"
"qiniu": "Qiniu"
},
"chat": {
"placeholder": "描述您的图表或上传文件...",
@@ -372,7 +371,6 @@
"baseUrlWithExample": "基础 URL(可选,例如 {example})",
"customEndpoint": "自定义端点 URL",
"minimaxBaseUrlHint": "使用 /anthropic 端点为 Anthropic 兼容 API(推荐),或使用 /v1 端点为 OpenAI 兼容 API",
"mimoBaseUrlHint": "默认地址适用于按量付费密钥(sk-...)。Token Plan 订阅用户(tp-... 密钥)请设置为 https://token-plan-cn.xiaomimimo.com/v1",
"models": "模型",
"customModelId": "自定义模型 ID...",
"allAdded": "已全部添加",
-7
View File
@@ -23,7 +23,6 @@ export type ProviderName =
| "kimi"
| "minimax"
| "novita"
| "mimo"
// Individual model configuration
export interface ModelConfig {
@@ -115,7 +114,6 @@ export const PROVIDER_LOGO_MAP: Record<string, string> = {
modelscope: "modelscope",
minimax: "minimax",
novita: "novita",
mimo: "xiaomi",
}
// Provider metadata
@@ -202,10 +200,6 @@ export const PROVIDER_INFO: Record<
label: "Novita AI",
defaultBaseUrl: "https://api.novita.ai/openai",
},
mimo: {
label: "MiMo (Xiaomi)",
defaultBaseUrl: "https://api.xiaomimimo.com/v1",
},
}
// Suggested models per provider for quick add
@@ -443,7 +437,6 @@ export const SUGGESTED_MODELS: Partial<Record<ProviderName, string[]>> = {
"moonshotai/kimi-k2.6",
"deepseek/deepseek-v4-flash",
],
mimo: ["mimo-v2.5-pro", "mimo-v2.5"],
}
// Helper to generate UUID
+1 -6
View File
@@ -116,14 +116,9 @@ Use the standard MCP configuration with:
|------|-------------|
| `start_session` | Opens browser with real-time diagram preview |
| `create_new_diagram` | Create a new diagram from XML (requires `xml` argument) |
| `load_diagram` | Load a `.drawio` file from disk into the session (handles compressed files) |
| `edit_diagram` | Edit diagram by ID-based operations (update/add/delete cells) |
| `get_diagram` | Get the current diagram XML |
| `export_diagram` | Save diagram to a `.drawio`, `.png`, or `.svg` file |
| `list_pages` | List every page (tab) with id, name, index, and cell count |
| `add_page` | Append a new page without touching existing ones |
| `rename_page` | Rename a page |
| `delete_page` | Delete a page (refuses to delete the last one) |
| `export_diagram` | Save diagram to a `.drawio` file |
## How It Works
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@next-ai-drawio/mcp-server",
"version": "0.2.3",
"version": "0.2.1",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@next-ai-drawio/mcp-server",
"version": "0.2.3",
"version": "0.2.1",
"license": "Apache-2.0",
"dependencies": {
"@modelcontextprotocol/sdk": "^1.0.4",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@next-ai-drawio/mcp-server",
"version": "0.2.3",
"version": "0.2.1",
"description": "MCP server for Next AI Draw.io - AI-powered diagram generation with real-time browser preview",
"type": "module",
"main": "dist/index.js",
-102
View File
@@ -1,102 +0,0 @@
/**
* Workflow gate for edit_diagram.
*
* Instead of a wall-clock timeout (the old 30s rule rejected slow-but-correct
* clients, see #885), we compare content: `lastSeenXml` is the state-store
* XML the model last saw (get_diagram) or wrote itself (create_new_diagram /
* edit_diagram / page CRUD). The store only changes on server writes or
* browser pushes (user autosave, sync exports), so if the live store still
* matches `lastSeenXml`, nothing happened that the model hasn't seen — the
* edit is safe no matter how much time passed.
*
* "Matches" is structural, not byte-for-byte: draw.io re-serialises the
* document when it pushes state back (different attribute order, pretty-
* printed whitespace, regenerated diagram ids, viewport attributes like
* dx/dy/pageWidth on <mxGraphModel>, a different mxfile host). None of that
* is a user edit, so the fingerprint keeps only what a user can actually
* change: the set of pages, each page's name, and each page's cell tree
* (tags + sorted attributes + text). Byte equality is kept as a fast path.
*/
import { isMxGraphModel, normalizeToMxfile, parseMxfile } from "./pages.js"
export type EditGateResult =
| { ok: true }
| { ok: false; reason: "no-context" | "stale" }
/**
* Canonical serialisation of an element subtree: tag + attributes sorted by
* name + child elements in order + non-whitespace text. Whitespace-only text
* nodes (pretty-printing) are dropped.
*/
function canonicalizeElement(el: Element): string {
const attrs = Array.from(el.attributes)
.map((a) => `${a.name}=${JSON.stringify(a.value)}`)
.sort()
.join(" ")
let children = ""
for (const child of Array.from(el.childNodes)) {
if (child.nodeType === 1) {
children += canonicalizeElement(child as Element)
} else if (child.nodeType === 3 || child.nodeType === 4) {
const text = (child.textContent ?? "").trim()
if (text) children += JSON.stringify(text)
}
}
return `<${el.tagName} ${attrs}>${children}</${el.tagName}>`
}
/**
* Structural fingerprint of a diagram document: page names + each page's
* <root> subtree, ignoring everything draw.io rewrites on re-serialisation
* (mxfile/mxGraphModel attributes, diagram ids, formatting). A bare
* <mxGraphModel> fingerprints identically to its single-page mxfile wrapping.
* Unparseable input falls back to the trimmed raw string, degrading to the
* plain string comparison.
*
* `includeNames=false` drops page names from the fingerprint — used when the
* other side of a comparison is a bare <mxGraphModel>, which carries no page
* name at all (normalizeToMxfile would invent "Page-1", falsely mismatching
* any real page name).
*/
export function contentFingerprint(xml: string, includeNames = true): string {
const normalized = normalizeToMxfile(xml)
const doc = normalized ? parseMxfile(normalized) : null
if (!doc) return xml.trim()
const pages: string[] = []
doc.querySelectorAll("diagram").forEach((d) => {
const name = includeNames ? d.getAttribute("name") || "" : ""
const root = d.querySelector("root")
// No <root> means the page content is not plain XML (e.g. draw.io's
// compressed format) — fingerprint the raw text instead.
const body = root
? canonicalizeElement(root)
: (d.textContent || "").trim()
pages.push(`${name}=${body}`)
})
return pages.join("\n")
}
export function checkEditGate(
lastSeenXml: string,
liveXml: string,
): EditGateResult {
// Model never fetched or produced any diagram state in this session.
if (!lastSeenXml) return { ok: false, reason: "no-context" }
// Browser state moved since the model last looked (e.g. manual user
// edits): force a re-fetch so update/delete operations don't build on
// stale cell contents. An empty liveXml means the store has no entry to
// compare against, so there is nothing newer to have missed.
if (liveXml && liveXml !== lastSeenXml) {
// A bare <mxGraphModel> on either side carries no page name, so
// comparing names would mismatch against anything not called
// "Page-1". Compare cell trees only in that case.
const includeNames =
!isMxGraphModel(liveXml) && !isMxGraphModel(lastSeenXml)
if (
contentFingerprint(liveXml, includeNames) !==
contentFingerprint(lastSeenXml, includeNames)
)
return { ok: false, reason: "stale" }
}
return { ok: true }
}
+53 -202
View File
@@ -36,7 +36,6 @@ class XMLSerializerPolyfill {
}
;(globalThis as any).XMLSerializer = XMLSerializerPolyfill
import { createRequire } from "node:module"
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
import open from "open"
@@ -45,7 +44,6 @@ import {
applyDiagramOperations,
type DiagramOperation,
} from "./diagram-operations.js"
import { checkEditGate } from "./edit-gate.js"
import { addHistory } from "./history.js"
import {
getState,
@@ -56,7 +54,6 @@ import {
startHttpServer,
waitForSync,
} from "./http-server.js"
import { parseDrawioFileContent } from "./load-diagram.js"
import { log } from "./logger.js"
import {
addPageToDoc,
@@ -82,25 +79,13 @@ let currentSession: {
id: string
xml: string
version: number
// The exact state-store XML the model last saw (get_diagram) or wrote
// itself (create/edit/page CRUD). The store only changes on server
// writes or browser pushes (user autosave / sync), so edit_diagram can
// detect unseen user edits by comparing the live store against this.
// Empty = no diagram context established yet.
lastSeenXml: string
lastGetDiagramTime: number // Track when get_diagram was last called (for enforcing workflow)
} | null = null
// Create MCP server. The version reported in the MCP handshake is read from
// package.json so it can never drift from the published npm version again
// (it sat hardcoded at stale values for most of this package's history).
// Both src/ (tsx dev) and dist/ (published build) live one level below the
// package root, so the relative path works in either runtime.
const require = createRequire(import.meta.url)
const packageVersion: string = require("../package.json").version
// Create MCP server
const server = new McpServer({
name: "next-ai-drawio",
version: packageVersion,
version: "0.3.0",
})
// Shared Zod schema fragment for page-targeting parameters.
@@ -173,21 +158,21 @@ server.prompt(
1. Call start_session to open the browser preview
2. Use create_new_diagram with either a bare <mxGraphModel> (single page) or a full <mxfile> with one or more <diagram> children (multi-page)
## Opening an Existing .drawio File
- Use load_diagram with the file path — the server reads and decompresses the file itself; don't read it and pass the XML through create_new_diagram
- After loading, call get_diagram once before editing (you haven't seen the file's cell IDs yet)
## Working with Multiple Pages
- Use list_pages to discover existing pages (id, name, index)
- Use add_page to append a new page (without losing existing ones — unlike create_new_diagram which REPLACES everything)
- Use rename_page / delete_page for management
- edit_diagram, get_diagram, and export_diagram all accept optional page_id / page_name / page_index — when omitted they target the first page
## Editing a Page (add / update / delete cells)
1. Call edit_diagram with your operations, optionally with a page selector
2. If you don't know the current cell IDs or structure, call get_diagram first
3. For add/update, provide the cell_id and complete mxCell XML
4. No need to call get_diagram before every edit: the server rejects the edit (with no side effects) if the user changed the diagram in the browser since you last saw it, and tells you to call get_diagram once and retry
## Adding Elements to an Existing Page
1. Use edit_diagram with "add" operation, optionally with a page selector
2. Provide a unique cell_id and complete mxCell XML
3. No need to call get_diagram first - the server fetches latest state automatically
## Modifying or Deleting Existing Elements
1. FIRST call get_diagram to see current cell IDs and page structure
2. THEN call edit_diagram with "update" or "delete" operations
3. For update, provide the cell_id and complete new mxCell XML
## Important Notes
- create_new_diagram REPLACES the entire document, including ALL pages - only use for new diagrams. Use add_page to add a tab without losing existing content.
@@ -220,7 +205,7 @@ server.registerTool(
id: sessionId,
xml: "",
version: 0,
lastSeenXml: "",
lastGetDiagramTime: 0,
}
// Open browser
@@ -391,12 +376,10 @@ COMMON STYLES:
// Update session state
currentSession.xml = xml
currentSession.version++
currentSession.lastGetDiagramTime = Date.now()
// Push to embedded server state. The model just authored this
// exact XML, so record it as seen — edit_diagram may follow
// without a redundant get_diagram round-trip.
// Push to embedded server state
setState(currentSession.id, xml)
currentSession.lastSeenXml = xml
// Save AI result (no SVG yet - will be captured by browser)
addHistory(currentSession.id, xml, "")
@@ -434,136 +417,19 @@ COMMON STYLES:
},
)
// Tool: load_diagram
server.registerTool(
"load_diagram",
{
description:
"Load a .drawio file from disk into the current session, REPLACING the entire diagram (all pages). " +
"The server reads the file directly — you do NOT need to read the file yourself or pass its XML through create_new_diagram. " +
"Handles both plain-XML and draw.io's compressed save format.\n\n" +
"After loading, call get_diagram before edit_diagram — you haven't seen the file's cell IDs or structure yet.",
inputSchema: {
path: z
.string()
.describe(
"Path to the .drawio file to load (e.g., ./diagram.drawio)",
),
},
},
async ({ path }) => {
try {
if (!currentSession) {
return {
content: [
{
type: "text",
text: "Error: No active session. Please call start_session first.",
},
],
isError: true,
}
}
const fs = await import("node:fs/promises")
const nodePath = await import("node:path")
const absolutePath = nodePath.resolve(path)
let content: string
try {
content = await fs.readFile(absolutePath, "utf-8")
} catch (e) {
const msg = e instanceof Error ? e.message : String(e)
return {
content: [
{
type: "text",
text: `Error: Cannot read file ${absolutePath}: ${msg}`,
},
],
isError: true,
}
}
const loaded = parseDrawioFileContent(content)
if (!loaded.ok) {
return {
content: [{ type: "text", text: `Error: ${loaded.error}` }],
isError: true,
}
}
const xml = loaded.xml
log.info(
`Loading diagram from ${absolutePath} (${xml.length} chars)`,
)
// Save the user's current state before replacing (same flow as
// create_new_diagram).
const browserState = getState(currentSession.id)
if (browserState?.xml) {
currentSession.xml = browserState.xml
}
if (currentSession.xml) {
addHistory(
currentSession.id,
currentSession.xml,
browserState?.svg || "",
)
}
currentSession.xml = xml
currentSession.version++
setState(currentSession.id, xml)
// Deliberately NOT marking the loaded XML as seen: the model only
// supplied a path, so it doesn't know the file's cell IDs. The
// edit gate will require one get_diagram before edits.
currentSession.lastSeenXml = ""
addHistory(currentSession.id, xml, "")
const doc = parseMxfile(xml)
const pages = doc ? listPagesFromDoc(doc) : []
const pageSummary =
pages.length > 0
? `Pages (${pages.length}): ${pages.map((p) => `[${p.index}] id=${p.id} name="${p.name}" cells=${p.cellCount}`).join(" | ")}`
: "no pages parsed"
log.info(`Diagram loaded from file (${pageSummary})`)
return {
content: [
{
type: "text",
text: `Diagram loaded from ${absolutePath}!\n\nThe diagram is now visible in your browser.\n\n${pageSummary}\n\nCall get_diagram before edit_diagram — you haven't seen this file's cell IDs yet.`,
},
],
}
} catch (error) {
const message =
error instanceof Error ? error.message : String(error)
log.error("load_diagram failed:", message)
return {
content: [{ type: "text", text: `Error: ${message}` }],
isError: true,
}
}
},
)
// Tool: edit_diagram
server.registerTool(
"edit_diagram",
{
description:
"Edit a specific page in the current diagram by ID-based operations (update/add/delete cells).\n\n" +
"Freshness: the server remembers the last diagram state you have seen, and rejects this call " +
"only if the user edited the diagram in the browser since then. You do NOT need to call " +
"get_diagram before every edit — if your view is stale, the call is rejected (with no side " +
"effects) and the error tells you to call get_diagram once and retry.\n\n" +
"Call get_diagram first only when you don't know the current diagram content (cell IDs, " +
"structure) — e.g. the diagram wasn't created in this conversation, or you're unsure your " +
"memory of it is accurate.\n\n" +
"⚠️ REQUIRED: You MUST call get_diagram BEFORE this tool!\n" +
"This fetches the latest state from the browser including any manual user edits.\n" +
"Skipping get_diagram WILL cause user's changes to be LOST.\n\n" +
"Workflow:\n" +
"1. Call get_diagram to see current cell IDs, page structure, and active page\n" +
"2. Use the returned XML to construct your edit operations\n" +
"3. Call edit_diagram with your operations and (optionally) a page selector\n\n" +
"Multi-page targeting:\n" +
"- page_id / page_name / page_index are optional; when all omitted, the FIRST page is targeted\n" +
"- Use list_pages to discover what pages exist\n\n" +
@@ -616,6 +482,27 @@ server.registerTool(
}
}
// Enforce workflow: require get_diagram to be called first
const timeSinceGet = Date.now() - currentSession.lastGetDiagramTime
if (timeSinceGet > 30000) {
// 30 seconds
log.warn(
"edit_diagram called without recent get_diagram - rejecting to prevent data loss",
)
return {
content: [
{
type: "text",
text:
"Error: You must call get_diagram first before edit_diagram.\n\n" +
"This ensures you have the latest diagram state including any manual edits the user made in the browser. " +
"Please call get_diagram, then use that XML to construct your edit operations.",
},
],
isError: true,
}
}
// Fetch latest state from browser. Re-normalise to mxfile: the
// embed/sync path can hand back a bare <mxGraphModel>, and adopting
// it verbatim would silently strip a multi-page document down to
@@ -639,37 +526,6 @@ server.registerTool(
}
}
// Enforce workflow: the model must have seen the current diagram
// state. Content comparison instead of a wall-clock timeout —
// slow reasoning between get_diagram and edit_diagram is fine as
// long as nothing changed in the browser meanwhile (#885).
const gate = checkEditGate(
currentSession.lastSeenXml,
browserState?.xml ?? "",
)
if (!gate.ok) {
log.warn(
gate.reason === "stale"
? "edit_diagram called with unseen browser changes - rejecting to prevent data loss"
: "edit_diagram called without get_diagram - rejecting to prevent data loss",
)
return {
content: [
{
type: "text",
text:
gate.reason === "stale"
? "Error: The diagram changed in the browser since you last fetched it (e.g. manual user edits).\n\n" +
"Call get_diagram to see the latest state, then rebuild your edit operations on top of it."
: "Error: You must call get_diagram first before edit_diagram.\n\n" +
"This ensures you have the latest diagram state including any manual edits the user made in the browser. " +
"Please call get_diagram, then use that XML to construct your edit operations.",
},
],
isError: true,
}
}
const pageSelector = pickPageSelector({
page_id,
page_name,
@@ -745,10 +601,8 @@ server.registerTool(
currentSession.xml = result
currentSession.version++
// Push to embedded server; the pushed XML is now the latest
// state the model has seen.
// Push to embedded server
setState(currentSession.id, result)
currentSession.lastSeenXml = result
// Save AI result (no SVG yet - will be captured by browser)
addHistory(currentSession.id, result, "")
@@ -787,9 +641,8 @@ server.registerTool(
{
description:
"Get the current diagram XML (fetches latest from browser, including user's manual edits). " +
"Call this when you don't know the current diagram content (cell IDs, pages, structure) — " +
"e.g. before editing a diagram you didn't create in this conversation, or after edit_diagram " +
"was rejected because the user changed the diagram in the browser.\n\n" +
"Call this BEFORE edit_diagram if you need to update or delete existing elements, " +
"so you can see the current cell IDs, pages, and structure.\n\n" +
"Returns the full <mxfile> by default. If a page selector is provided, returns just that page's <mxGraphModel> embedded in a one-page <mxfile> wrapper.",
inputSchema: {
...pageSelectorSchema,
@@ -824,6 +677,9 @@ server.registerTool(
}
}
// Mark that get_diagram was called (for edit_diagram workflow check)
currentSession.lastGetDiagramTime = Date.now()
// Fetch latest state from browser, re-normalising to mxfile so a
// bare <mxGraphModel> pushed back by the embed/sync path doesn't
// strip page structure (see edit_diagram for the same guard).
@@ -844,11 +700,6 @@ server.registerTool(
}
}
// The model is now looking at the current state. Record the raw
// store value — the gate's fast path is plain string equality
// against the store, with a structural comparison as fallback.
currentSession.lastSeenXml = browserState?.xml || currentSession.xml
const pageSelector = pickPageSelector({
page_id,
page_name,
@@ -1219,11 +1070,11 @@ async function loadMxfileForMutation(): Promise<
addHistory(sessionRef.id, sessionRef.xml, browserState?.svg || "")
sessionRef.xml = newXml
sessionRef.version++
// Page CRUD updates the structure that get_diagram would return,
// so refresh the workflow timestamp — subsequent edit_diagram
// calls don't need a redundant get_diagram round-trip.
sessionRef.lastGetDiagramTime = Date.now()
setState(sessionRef.id, newXml)
// The model just wrote this exact state, so mark it as seen —
// subsequent edit_diagram calls don't need a redundant
// get_diagram round-trip.
sessionRef.lastSeenXml = newXml
addHistory(sessionRef.id, newXml, "")
},
}
-101
View File
@@ -1,101 +0,0 @@
/**
* File-loading helpers for the load_diagram tool.
*
* A .drawio file is an <mxfile> whose <diagram> children hold each page's
* <mxGraphModel> either as plain XML or — draw.io's default save format —
* compressed: encodeURIComponent(xml) → raw deflate → base64 as the
* diagram's text content. The rest of the server assumes plain XML inside
* every <diagram>, so loading decompresses all pages up front.
*/
import { inflateRawSync } from "node:zlib"
import { DOMParser } from "linkedom"
import {
isMxFile,
isMxGraphModel,
normalizeToMxfile,
parseMxfile,
serializeMxfile,
} from "./pages.js"
export type LoadResult =
| { ok: true; xml: string }
| { ok: false; error: string }
/**
* Decode one compressed page body (base64 → raw deflate → URI-decode).
* Returns null if the text isn't in that format.
*/
export function decompressPageContent(compressed: string): string | null {
try {
const inflated = inflateRawSync(
Buffer.from(compressed.trim(), "base64"),
).toString("utf-8")
try {
return decodeURIComponent(inflated)
} catch {
// Not URI-encoded (older files) — the inflated text is the XML.
return inflated
}
} catch {
return null
}
}
/**
* Parse the content of a .drawio file into the canonical session shape:
* an <mxfile> whose every page holds plain <mxGraphModel> XML. Accepts a
* bare <mxGraphModel> (wrapped into a one-page mxfile) and decompresses
* any compressed pages.
*/
export function parseDrawioFileContent(content: string): LoadResult {
const trimmed = content.trim()
if (!trimmed) return { ok: false, error: "File is empty." }
if (isMxGraphModel(trimmed)) {
const normalized = normalizeToMxfile(trimmed)
return normalized
? { ok: true, xml: normalized }
: { ok: false, error: "Failed to parse <mxGraphModel> XML." }
}
if (!isMxFile(trimmed)) {
return {
ok: false,
error: "Not a draw.io file: expected an <mxfile> or <mxGraphModel> root element.",
}
}
const doc = parseMxfile(trimmed)
if (!doc) return { ok: false, error: "Failed to parse <mxfile> XML." }
let decompressedAny = false
for (const d of Array.from(doc.querySelectorAll("diagram"))) {
if (d.querySelector("mxGraphModel")) continue
const text = (d.textContent || "").trim()
if (!text) continue // an empty page is valid
const pageLabel =
d.getAttribute("name") || d.getAttribute("id") || "unnamed"
const xml = decompressPageContent(text)
if (!xml || !isMxGraphModel(xml)) {
return {
ok: false,
error: `Page "${pageLabel}" has content that is neither plain <mxGraphModel> XML nor draw.io's compressed format.`,
}
}
const inner = new DOMParser().parseFromString(xml, "text/xml")
if (
inner.querySelector("parsererror") ||
inner.documentElement?.tagName !== "mxGraphModel"
) {
return {
ok: false,
error: `Page "${pageLabel}" decompressed but its XML failed to parse.`,
}
}
d.textContent = ""
d.appendChild(
doc.importNode(inner.documentElement as unknown as Node, true),
)
decompressedAny = true
}
// Nothing changed — keep the file's own serialisation.
return { ok: true, xml: decompressedAny ? serializeMxfile(doc) : trimmed }
}
-132
View File
@@ -1,132 +0,0 @@
/**
* Unit tests for the edit_diagram workflow gate (edit-gate.ts).
*
* The gate replaced the old 30-second wall-clock rule (#885): an edit is
* allowed when the model has seen the current browser state, no matter how
* long ago — and rejected when the browser state moved since. "Seen" is
* judged structurally, so draw.io's re-serialisation of the same content
* (attribute order, whitespace, viewport attributes, wrapper shape) never
* reads as a user edit.
*/
import { DOMParser } from "linkedom"
import { beforeAll, describe, expect, it } from "vitest"
beforeAll(() => {
;(globalThis as any).DOMParser = DOMParser
})
import { checkEditGate, contentFingerprint } from "../src/edit-gate.js"
const XML_A = `<mxfile host="app.diagrams.net"><diagram id="p1" name="Page-1"><mxGraphModel><root><mxCell id="0"/><mxCell id="1" parent="0"/><mxCell id="box1" value="Hello" style="rounded=0;" vertex="1" parent="1"><mxGeometry x="40" y="40" width="120" height="60" as="geometry"/></mxCell></root></mxGraphModel></diagram></mxfile>`
// The same document as draw.io re-serialises it on autosave: different host,
// regenerated diagram id, viewport attributes on mxGraphModel, re-ordered
// cell attributes, pretty-printed whitespace.
const XML_A_RESERIALIZED = `<mxfile host="embed.diagrams.net">
<diagram id="regenerated-id" name="Page-1">
<mxGraphModel dx="1596" dy="743" grid="1" pageWidth="827" pageHeight="1169">
<root>
<mxCell id="0" />
<mxCell id="1" parent="0" />
<mxCell id="box1" parent="1" style="rounded=0;" value="Hello" vertex="1">
<mxGeometry height="60" width="120" x="40" y="40" as="geometry" />
</mxCell>
</root>
</mxGraphModel>
</diagram>
</mxfile>`
// A real user edit: box1 moved to a different position.
const XML_B = XML_A.replace('x="40" y="40"', 'x="300" y="200"')
// Bare mxGraphModel with identical page content to XML_A.
const XML_A_BARE = `<mxGraphModel><root><mxCell id="0"/><mxCell id="1" parent="0"/><mxCell id="box1" value="Hello" style="rounded=0;" vertex="1" parent="1"><mxGeometry x="40" y="40" width="120" height="60" as="geometry"/></mxCell></root></mxGraphModel>`
describe("checkEditGate", () => {
it("rejects when no diagram context was ever established", () => {
expect(checkEditGate("", XML_A)).toEqual({
ok: false,
reason: "no-context",
})
})
it("allows when the browser state is exactly what the model saw", () => {
expect(checkEditGate(XML_A, XML_A)).toEqual({ ok: true })
})
it("allows when the browser state is a re-serialisation of the same content", () => {
expect(checkEditGate(XML_A, XML_A_RESERIALIZED)).toEqual({ ok: true })
})
it("rejects when a cell actually changed", () => {
expect(checkEditGate(XML_A, XML_B)).toEqual({
ok: false,
reason: "stale",
})
})
it("rejects a real edit even when wrapped in re-serialisation noise", () => {
const movedAndReserialized = XML_A_RESERIALIZED.replace(
'x="40" y="40"',
'x="300" y="200"',
)
expect(checkEditGate(XML_A, movedAndReserialized)).toEqual({
ok: false,
reason: "stale",
})
})
it("allows when the store has no live entry to compare against", () => {
expect(checkEditGate(XML_A, "")).toEqual({ ok: true })
})
// A bare <mxGraphModel> push carries no page name, so the gate must not
// compare the invented "Page-1" wrapper name against the real one.
it("allows a bare mxGraphModel push when the page has a custom name", () => {
const seenRenamed = XML_A.replace('name="Page-1"', 'name="Arch"')
expect(checkEditGate(seenRenamed, XML_A_BARE)).toEqual({ ok: true })
})
it("still rejects a bare mxGraphModel push whose cells changed", () => {
const seenRenamed = XML_A.replace('name="Page-1"', 'name="Arch"')
const bareMoved = XML_A_BARE.replace('x="40" y="40"', 'x="300" y="200"')
expect(checkEditGate(seenRenamed, bareMoved)).toEqual({
ok: false,
reason: "stale",
})
})
})
describe("contentFingerprint", () => {
it("is invariant under draw.io re-serialisation", () => {
expect(contentFingerprint(XML_A)).toBe(
contentFingerprint(XML_A_RESERIALIZED),
)
})
it("treats a bare mxGraphModel like its one-page mxfile wrapping", () => {
expect(contentFingerprint(XML_A_BARE)).toBe(contentFingerprint(XML_A))
})
it("changes when a cell attribute changes", () => {
expect(contentFingerprint(XML_A)).not.toBe(contentFingerprint(XML_B))
})
it("changes when a page is renamed", () => {
const renamed = XML_A.replace('name="Page-1"', 'name="Renamed"')
expect(contentFingerprint(XML_A)).not.toBe(contentFingerprint(renamed))
})
it("changes when a page is added", () => {
const twoPages = XML_A.replace(
"</mxfile>",
`<diagram id="p2" name="Page-2"><mxGraphModel><root><mxCell id="0"/><mxCell id="1" parent="0"/></root></mxGraphModel></diagram></mxfile>`,
)
expect(contentFingerprint(XML_A)).not.toBe(contentFingerprint(twoPages))
})
it("falls back to the raw string for unparseable input", () => {
expect(contentFingerprint("not xml at all")).toBe("not xml at all")
})
})
@@ -1,126 +0,0 @@
/**
* Unit tests for load_diagram's file parsing (load-diagram.ts).
*
* A .drawio file stores each page's <mxGraphModel> either as plain XML or
* as draw.io's compressed default (encodeURIComponent → raw deflate →
* base64 text content). The loader must produce the canonical session
* shape: an <mxfile> whose every page is plain XML.
*/
import { deflateRawSync } from "node:zlib"
import { DOMParser } from "linkedom"
import { beforeAll, describe, expect, it } from "vitest"
// Install the DOM polyfills exactly as index.ts does at runtime.
beforeAll(() => {
;(globalThis as any).DOMParser = DOMParser
class XMLSerializerPolyfill {
serializeToString(node: any): string {
if (node.outerHTML !== undefined) return node.outerHTML
if (node.documentElement) return node.documentElement.outerHTML
return ""
}
}
;(globalThis as any).XMLSerializer = XMLSerializerPolyfill
})
import {
decompressPageContent,
parseDrawioFileContent,
} from "../src/load-diagram.js"
const MODEL_XML = `<mxGraphModel><root><mxCell id="0"/><mxCell id="1" parent="0"/><mxCell id="box1" value="Hello" style="rounded=0;" vertex="1" parent="1"><mxGeometry x="40" y="40" width="120" height="60" as="geometry"/></mxCell></root></mxGraphModel>`
/** Compress a page body exactly the way draw.io does when saving. */
function drawioCompress(xml: string): string {
return deflateRawSync(
Buffer.from(encodeURIComponent(xml), "utf-8"),
).toString("base64")
}
const PLAIN_MXFILE = `<mxfile host="app.diagrams.net"><diagram id="p1" name="Page-1">${MODEL_XML}</diagram></mxfile>`
const COMPRESSED_MXFILE = `<mxfile host="app.diagrams.net" compressed="true"><diagram id="p1" name="Page-1">${drawioCompress(MODEL_XML)}</diagram></mxfile>`
describe("decompressPageContent", () => {
it("round-trips draw.io's compressed format", () => {
expect(decompressPageContent(drawioCompress(MODEL_XML))).toBe(MODEL_XML)
})
it("handles non-URI-encoded legacy payloads", () => {
const legacy = deflateRawSync(Buffer.from(MODEL_XML, "utf-8")).toString(
"base64",
)
expect(decompressPageContent(legacy)).toBe(MODEL_XML)
})
it("returns null for garbage", () => {
expect(decompressPageContent("not base64 deflate")).toBeNull()
})
})
describe("parseDrawioFileContent", () => {
it("passes a plain-XML mxfile through unchanged", () => {
const r = parseDrawioFileContent(PLAIN_MXFILE)
expect(r).toEqual({ ok: true, xml: PLAIN_MXFILE })
})
it("wraps a bare mxGraphModel into a one-page mxfile", () => {
const r = parseDrawioFileContent(MODEL_XML)
expect(r.ok).toBe(true)
if (r.ok) {
expect(r.xml).toContain("<mxfile")
expect(r.xml).toContain('value="Hello"')
}
})
it("decompresses a compressed mxfile into plain XML pages", () => {
const r = parseDrawioFileContent(COMPRESSED_MXFILE)
expect(r.ok).toBe(true)
if (r.ok) {
expect(r.xml).toContain("<mxGraphModel")
expect(r.xml).toContain('value="Hello"')
// The compressed blob must be gone.
expect(r.xml).not.toContain(drawioCompress(MODEL_XML))
}
})
it("decompresses only the compressed pages of a mixed file", () => {
const mixed = `<mxfile><diagram id="a" name="Plain">${MODEL_XML}</diagram><diagram id="b" name="Squeezed">${drawioCompress(MODEL_XML)}</diagram></mxfile>`
const r = parseDrawioFileContent(mixed)
expect(r.ok).toBe(true)
if (r.ok) {
const doc = new DOMParser().parseFromString(r.xml, "text/xml")
const diagrams = Array.from(
doc.querySelectorAll("diagram"),
) as Element[]
expect(diagrams).toHaveLength(2)
for (const d of diagrams) {
expect(d.querySelector("mxGraphModel")).not.toBeNull()
}
}
})
it("keeps empty pages as-is", () => {
const withEmpty = `<mxfile><diagram id="a" name="Page-1">${MODEL_XML}</diagram><diagram id="b" name="Empty"></diagram></mxfile>`
const r = parseDrawioFileContent(withEmpty)
expect(r).toEqual({ ok: true, xml: withEmpty })
})
it("rejects empty files", () => {
const r = parseDrawioFileContent(" ")
expect(r.ok).toBe(false)
})
it("rejects non-drawio content", () => {
const r = parseDrawioFileContent("<svg><rect/></svg>")
expect(r.ok).toBe(false)
if (!r.ok) expect(r.error).toContain("Not a draw.io file")
})
it("rejects a page whose content is neither XML nor compressed", () => {
const bad = `<mxfile><diagram id="a" name="Broken">!!! not a diagram !!!</diagram></mxfile>`
const r = parseDrawioFileContent(bad)
expect(r.ok).toBe(false)
if (!r.ok) expect(r.error).toContain('"Broken"')
})
})
@@ -31,7 +31,6 @@ const tsxBin = path.resolve(
const EXPECTED_TOOLS = [
"start_session",
"create_new_diagram",
"load_diagram",
"edit_diagram",
"get_diagram",
"export_diagram",
+84
View File
@@ -3,6 +3,7 @@ import {
getAIModel,
isAihubmixStandardBaseURL,
resolveBaseURL,
supportsImageInput,
supportsPromptCaching,
} from "@/lib/ai-providers"
import { extractAihubmixModelIds } from "@/lib/aihubmix-models"
@@ -182,6 +183,89 @@ describe("supportsPromptCaching", () => {
})
})
describe("supportsImageInput", () => {
it("returns true for models with vision capability", () => {
expect(supportsImageInput("gpt-4-vision")).toBe(true)
expect(supportsImageInput("qwen-vl")).toBe(true)
expect(supportsImageInput("deepseek-vl")).toBe(true)
})
it("returns false for Kimi K2 models without vision", () => {
expect(supportsImageInput("kimi-k2")).toBe(false)
expect(supportsImageInput("moonshot/kimi-k2")).toBe(false)
})
it("returns true for Kimi K2.5 models (supports vision)", () => {
expect(supportsImageInput("kimi-k2.5")).toBe(true)
expect(supportsImageInput("moonshotai/kimi-k2.5")).toBe(true)
})
it("returns false for Moonshot v1 text models", () => {
expect(supportsImageInput("moonshot-v1-8k")).toBe(false)
expect(supportsImageInput("moonshot-v1-32k")).toBe(false)
expect(supportsImageInput("moonshot-v1-128k")).toBe(false)
})
it("returns false for MiniMax M2 text models", () => {
expect(supportsImageInput("MiniMax-M2.7")).toBe(false)
expect(supportsImageInput("MiniMax-M2.7-highspeed")).toBe(false)
expect(supportsImageInput("MiniMax-M2")).toBe(false)
})
it("returns true for MiniMax M3 (supports image input)", () => {
expect(supportsImageInput("MiniMax-M3")).toBe(true)
})
it("returns false for DeepSeek text models", () => {
expect(supportsImageInput("deepseek-chat")).toBe(false)
expect(supportsImageInput("deepseek-coder")).toBe(false)
})
it("returns false for Qwen text models", () => {
expect(supportsImageInput("qwen-turbo")).toBe(false)
expect(supportsImageInput("qwen-plus")).toBe(false)
expect(supportsImageInput("qwen3-max")).toBe(false)
})
it("returns true for Qwen vision models", () => {
expect(supportsImageInput("qwen-vl")).toBe(true)
expect(supportsImageInput("Qwen3.5")).toBe(true)
expect(supportsImageInput("qwen3.5")).toBe(true)
expect(supportsImageInput("qwen3.5-plus")).toBe(true)
expect(supportsImageInput("qwen3.5-flash")).toBe(true)
expect(supportsImageInput("qwen3-vl-plus")).toBe(true)
expect(supportsImageInput("qwen3-vl-flash")).toBe(true)
})
it("returns true for QvQ (Qwen Visual QA) models including OpenRouter-prefixed names", () => {
expect(supportsImageInput("qvq-72b-preview")).toBe(true)
expect(supportsImageInput("qvq-max")).toBe(true)
expect(supportsImageInput("qwen/qvq-72b-preview")).toBe(true)
expect(supportsImageInput("qwen/qvq-max")).toBe(true)
})
it("returns false for GLM text models", () => {
expect(supportsImageInput("glm-4")).toBe(false)
expect(supportsImageInput("glm-4-plus")).toBe(false)
expect(supportsImageInput("glm-4-flash")).toBe(false)
expect(supportsImageInput("glm-4-long")).toBe(false)
expect(supportsImageInput("glm-4.7")).toBe(false)
expect(supportsImageInput("glm-5")).toBe(false)
})
it("returns true for GLM vision models", () => {
expect(supportsImageInput("glm-4v")).toBe(true)
expect(supportsImageInput("glm-4v-9b")).toBe(true)
expect(supportsImageInput("glm-4.1v-9b-thinking")).toBe(true)
})
it("returns true for Claude and GPT models by default", () => {
expect(supportsImageInput("claude-sonnet-4-5")).toBe(true)
expect(supportsImageInput("gpt-4o")).toBe(true)
expect(supportsImageInput("gemini-pro")).toBe(true)
})
})
vi.mock("ollama-ai-provider-v2", () => {
const mockModel = { modelId: "test-model" }
const mockProviderFn = vi.fn(() => mockModel)