Files
next-ai-draw-io/packages/mcp-server/src/xml-validation.ts
T
dayuan.jiang 1c51a9f31f feat(diagram): compact cells, position and size as attributes on the mxCell
The model now writes a shape as one self-closing mxCell with x, y, w and h
and an edge as one with source and target; compact-cells.ts fills in
vertex, edge, parent="1" and the mxGeometry element after validation. The
long form stays accepted, so waypoints, edge labels and anything else keep
their mxGeometry. The diagram shown to the model (the system context, the
edit error texts and the MCP get_diagram result) is folded into the same
notation, so the model reads what it is asked to write; the canvas and the
saved file stay standard draw.io XML. Folding then expanding the 652 valid
production diagrams gives the same cells back for every one of them.

The auto-fix gains one step for the slip this notation invites: a missing
closing quote before the self-closing slash (h="112/>).

Prompts and the shared examples use the compact form; the single waypoint
example keeps the long form on purpose.

Measured on the five start-screen examples, same prompt, against the
named-styles build: gpt-6-luna 19% fewer output tokens, Opus 5.5 26% fewer
(35% and 41% against main). Every shape was written compactly on the first
try; edits used it too and all applied.
2026-10-10 20:27:51 +09:00

1132 lines
40 KiB
TypeScript

/**
* XML Validation and Auto-Fix for draw.io diagrams
* Copied from lib/utils.ts to avoid cross-package imports
*/
import { readAttributes } from "./xml-attributes.ts"
import { getXmlSyntaxError } from "./xml-syntax.ts"
// ============================================================================
// Constants
// ============================================================================
/** Maximum XML size to process (1MB) - larger XMLs may cause performance issues */
const MAX_XML_SIZE = 1_000_000
/** Structural attributes that should not be duplicated in draw.io */
const STRUCTURAL_ATTRS = [
"edge",
"parent",
"source",
"target",
"vertex",
"connectable",
]
/** Valid XML entity names */
const VALID_ENTITIES = new Set(["lt", "gt", "amp", "quot", "apos"])
/** Element names draw.io understands (case-sensitive) */
const VALID_DRAWIO_TAGS = new Set([
"mxfile",
"diagram",
"mxGraphModel",
"root",
"mxCell",
"mxGeometry",
"mxPoint",
"Array",
"Object",
"mxRectangle",
// Wrappers draw.io writes for cells with links, tooltips or data
"UserObject",
"object",
])
// ============================================================================
// XML Parsing Helpers
// ============================================================================
interface ParsedTag {
tag: string
tagName: string
isClosing: boolean
isSelfClosing: boolean
startIndex: number
endIndex: number
}
/**
* Parse XML tags while properly handling quoted strings
*/
function parseXmlTags(xml: string): ParsedTag[] {
const tags: ParsedTag[] = []
let i = 0
while (i < xml.length) {
const tagStart = xml.indexOf("<", i)
if (tagStart === -1) break
// Find matching > by tracking quotes
let tagEnd = tagStart + 1
let inQuote = false
let quoteChar = ""
while (tagEnd < xml.length) {
const c = xml[tagEnd]
if (inQuote) {
if (c === quoteChar) inQuote = false
} else {
if (c === '"' || c === "'") {
inQuote = true
quoteChar = c
} else if (c === ">") {
break
}
}
tagEnd++
}
if (tagEnd >= xml.length) break
const tag = xml.substring(tagStart, tagEnd + 1)
i = tagEnd + 1
const tagMatch = /^<(\/?)([a-zA-Z][a-zA-Z0-9:_-]*)/.exec(tag)
if (!tagMatch) continue
tags.push({
tag,
tagName: tagMatch[2],
isClosing: tagMatch[1] === "/",
isSelfClosing: tag.endsWith("/>"),
startIndex: tagStart,
endIndex: tagEnd,
})
}
return tags
}
/**
* Returns a function telling whether a position lies inside a quoted
* attribute value. Positions must be queried in increasing order.
*/
function createQuoteTracker(str: string): (pos: number) => boolean {
let i = 0
let inQuote = false
let quoteChar = ""
return (pos: number) => {
for (; i < pos && i < str.length; i++) {
const c = str[i]
if (inQuote) {
if (c === quoteChar) inQuote = false
} else if (c === '"' || c === "'") {
// Only quotes that follow "=" open an attribute value
let j = i - 1
while (j >= 0 && /\s/.test(str[j])) j--
if (j >= 0 && str[j] === "=") {
inQuote = true
quoteChar = c
}
}
}
return inQuote
}
}
/** Rewrite every opening tag with fn, leaving text and closing tags as is. */
function replaceInOpeningTags(
xml: string,
fn: (tag: string) => string,
): string {
let out = ""
let last = 0
for (const { tag, isClosing, startIndex, endIndex } of parseXmlTags(xml)) {
if (isClosing) continue
out += xml.slice(last, startIndex) + fn(tag)
last = endIndex + 1
}
return out + xml.slice(last)
}
// ============================================================================
// Validation Helper Functions
// ============================================================================
/** Check for duplicate structural attributes in a tag */
function checkDuplicateAttributes(xml: string): string | null {
const structuralSet = new Set(STRUCTURAL_ATTRS)
for (const [tag] of xml.matchAll(/<[^>]+>/g)) {
const attributes = new Map<string, number>()
for (const { name } of readAttributes(tag)) {
attributes.set(name, (attributes.get(name) || 0) + 1)
}
const duplicates = Array.from(attributes.entries())
.filter(([name, count]) => count > 1 && structuralSet.has(name))
.map(([name]) => name)
if (duplicates.length > 0) {
return `Invalid XML: Duplicate structural attribute(s): ${duplicates.join(", ")}. Remove duplicate attributes.`
}
}
return null
}
/**
* Check for duplicate IDs in XML.
*
* For multi-page documents (<mxfile> with multiple <diagram> children), cell
* IDs are unique **within a page**, not across the whole document — drawio
* legitimately reuses "0" and "1" for the root cells of every page. So we
* scope the cell-ID uniqueness check per <diagram>, and additionally check
* that the <diagram> ids themselves are unique.
*
* The legacy regex-based check is kept as a fallback for non-mxfile inputs.
*/
function checkDuplicateIds(xml: string): string | null {
// The DOM-aware path only matters for <mxfile> wrappers; for legacy
// bare <mxGraphModel> inputs (the overwhelming majority of historic
// traffic), the cheap regex fallback at the bottom is enough. A quick
// string check avoids paying the DOMParser cost on every call.
const mightBeMxFile = /<mxfile[\s>]/i.test(xml)
// Try DOM-aware, page-scoped check first when the input looks mxfile-ish.
if (mightBeMxFile)
try {
const doc = new DOMParser().parseFromString(xml, "text/xml")
const rootEl = doc.documentElement
if (rootEl && rootEl.tagName === "mxfile") {
const diagrams = doc.querySelectorAll("diagram")
// 1) <diagram> ids must be unique across the file.
const diagramIds = new Map<string, number>()
diagrams.forEach((d) => {
const id = d.getAttribute("id")
if (id) diagramIds.set(id, (diagramIds.get(id) || 0) + 1)
})
const dupDiagrams = Array.from(diagramIds.entries())
.filter(([, c]) => c > 1)
.map(([id]) => `'${id}'`)
if (dupDiagrams.length > 0) {
return `Invalid XML: Found duplicate <diagram> id(s): ${dupDiagrams.slice(0, 3).join(", ")}. Each page must have a unique id.`
}
// 2) Within each page, cell ids must be unique. A cell with
// a link or custom data is a UserObject/object holding the
// id, around an mxCell whose own id does not count.
for (let i = 0; i < diagrams.length; i++) {
const diagram = diagrams[i]
const pageId = diagram.getAttribute("id") || `(index ${i})`
const cells = diagram.querySelectorAll(
"mxCell, UserObject, object",
)
const cellIds = new Map<string, number>()
cells.forEach((c) => {
const wrapped =
c.tagName === "mxCell" &&
/^(UserObject|object)$/.test(
c.parentElement?.tagName ?? "",
)
const id = c.getAttribute("id")
if (id && !wrapped)
cellIds.set(id, (cellIds.get(id) || 0) + 1)
})
const dups = Array.from(cellIds.entries())
.filter(([, c]) => c > 1)
.map(([id, count]) => `'${id}' (${count}x)`)
if (dups.length > 0) {
return `Invalid XML: Found duplicate cell ID(s) in page "${pageId}": ${dups.slice(0, 3).join(", ")}. All mxCell ids must be unique within a page.`
}
}
return null
}
} catch {
// fall through to regex
}
// Legacy regex-based check for bare <mxGraphModel> inputs.
const idPattern = /\bid\s*=\s*["']([^"']+)["']/gi
const ids = new Map<string, number>()
let idMatch
while ((idMatch = idPattern.exec(xml)) !== null) {
const id = idMatch[1]
ids.set(id, (ids.get(id) || 0) + 1)
}
const duplicateIds = Array.from(ids.entries())
.filter(([, count]) => count > 1)
.map(([id, count]) => `'${id}' (${count}x)`)
if (duplicateIds.length > 0) {
return `Invalid XML: Found duplicate ID(s): ${duplicateIds.slice(0, 3).join(", ")}. All id attributes must be unique.`
}
return null
}
/** Check for tag mismatches using parsed tags */
function checkTagMismatches(xml: string): string | null {
const xmlWithoutComments = xml.replace(/<!--[\s\S]*?-->/g, "")
const tags = parseXmlTags(xmlWithoutComments)
const tagStack: string[] = []
for (const { tagName, isClosing, isSelfClosing } of tags) {
if (isClosing) {
if (tagStack.length === 0) {
return `Invalid XML: Closing tag </${tagName}> without matching opening tag`
}
const expected = tagStack.pop()
if (expected?.toLowerCase() !== tagName.toLowerCase()) {
return `Invalid XML: Expected closing tag </${expected}> but found </${tagName}>`
}
} else if (!isSelfClosing) {
tagStack.push(tagName)
}
}
if (tagStack.length > 0) {
return `Invalid XML: Document has ${tagStack.length} unclosed tag(s): ${tagStack.join(", ")}`
}
return null
}
/** Check for invalid character references */
function checkCharacterReferences(xml: string): string | null {
const charRefPattern = /&#x?[^;]+;?/g
let charMatch
while ((charMatch = charRefPattern.exec(xml)) !== null) {
const ref = charMatch[0]
if (ref.startsWith("&#x")) {
if (!ref.endsWith(";")) {
return `Invalid XML: Missing semicolon after hex reference: ${ref}`
}
const hexDigits = ref.substring(3, ref.length - 1)
if (hexDigits.length === 0 || !/^[0-9a-fA-F]+$/.test(hexDigits)) {
return `Invalid XML: Invalid hex character reference: ${ref}`
}
} else if (ref.startsWith("&#")) {
if (!ref.endsWith(";")) {
return `Invalid XML: Missing semicolon after decimal reference: ${ref}`
}
const decDigits = ref.substring(2, ref.length - 1)
if (decDigits.length === 0 || !/^[0-9]+$/.test(decDigits)) {
return `Invalid XML: Invalid decimal character reference: ${ref}`
}
}
}
return null
}
/** Check for invalid entity references */
function checkEntityReferences(xml: string): string | null {
const xmlWithoutComments = xml.replace(/<!--[\s\S]*?-->/g, "")
const bareAmpPattern = /&(?!(?:lt|gt|amp|quot|apos|#))/g
if (bareAmpPattern.test(xmlWithoutComments)) {
return "Invalid XML: Found unescaped & character(s). Replace & with &amp;"
}
const invalidEntityPattern = /&([a-zA-Z][a-zA-Z0-9]*);/g
let entityMatch
while (
(entityMatch = invalidEntityPattern.exec(xmlWithoutComments)) !== null
) {
if (!VALID_ENTITIES.has(entityMatch[1])) {
return `Invalid XML: Invalid entity reference: &${entityMatch[1]}; - use only valid XML entities (lt, gt, amp, quot, apos)`
}
}
return null
}
/** Check for nested mxCell tags using regex */
function checkNestedMxCells(xml: string): string | null {
const cellTagPattern = /<\/?mxCell[^>]*>/g
const cellStack: number[] = []
let cellMatch
while ((cellMatch = cellTagPattern.exec(xml)) !== null) {
const tag = cellMatch[0]
if (tag.startsWith("</mxCell>")) {
if (cellStack.length > 0) cellStack.pop()
} else if (!tag.endsWith("/>")) {
const isLabelOrGeometry =
/\sas\s*=\s*["'](valueLabel|geometry)["']/.test(tag)
if (!isLabelOrGeometry) {
cellStack.push(cellMatch.index)
if (cellStack.length > 1) {
return "Invalid XML: Found nested mxCell tags. Cells should be siblings, not nested inside other mxCell elements."
}
}
}
}
return null
}
/** Check for element names draw.io does not know (e.g. a lowercase <mxcell>) */
function checkUnknownElements(xml: string): string | null {
const tags = parseXmlTags(xml.replace(/<!--[\s\S]*?-->/g, ""))
for (const { tagName } of tags) {
if (!VALID_DRAWIO_TAGS.has(tagName)) {
return `Invalid XML: Unknown element <${tagName}>. draw.io only understands ${Array.from(VALID_DRAWIO_TAGS).join(", ")} (names are case-sensitive).`
}
}
return null
}
/**
* Find <mxPoint> elements without an "as" attribute outside <Array
* as="points">. draw.io rejects them with "Could not add object mxPoint".
*/
function findOrphanMxPoints(
xml: string,
): Array<{ start: number; end: number }> {
const arrays: Array<[number, number]> = []
// (?<!\/) skips an empty <Array/>, which has no points inside
for (const m of xml.matchAll(/<Array\b[^>]*(?<!\/)>[\s\S]*?<\/Array>/g)) {
arrays.push([m.index, m.index + m[0].length])
}
const orphans: Array<{ start: number; end: number }> = []
for (const m of xml.matchAll(/<mxPoint\b[^>]*?(?:\/>|>\s*<\/mxPoint>)/g)) {
if (/\sas\s*=/.test(m[0])) continue
if (arrays.some(([s, e]) => m.index > s && m.index < e)) continue
orphans.push({ start: m.index, end: m.index + m[0].length })
}
return orphans
}
// ============================================================================
// Main Validation Function
// ============================================================================
/**
* Validates draw.io XML structure for common issues
* Uses DOM parsing + additional regex checks for high accuracy
* @param xml - The XML string to validate
* @param opts.strict - Also reject unknown element names and orphan
* <mxPoint>s. Used for XML the model wrote, not for files or browser state.
* @returns null if valid, error message string if invalid
*/
/** The first non-blank text under el, skipping a page's compressed data */
function findTextBetweenTags(el: Element | null): string | null {
if (!el) return null
// A <diagram> with only text holds the page compressed
const compressed = el.tagName === "diagram" && el.children.length === 0
for (const node of Array.from(el.childNodes)) {
if (node.nodeType === 1) {
const text = findTextBetweenTags(node as Element)
if (text) return text
} else if (
// Text, or a CDATA section (draw.io reads it as text too)
(node.nodeType === 3 || node.nodeType === 4) &&
!compressed
) {
const text = node.textContent?.trim()
if (text) return text.slice(0, 40)
}
}
return null
}
export function validateMxCellStructure(
xml: string,
opts: { strict?: boolean } = {},
): string | null {
// Size check for performance
if (xml.length > MAX_XML_SIZE) {
console.warn(
`[validateMxCellStructure] XML size (${xml.length}) exceeds ${MAX_XML_SIZE} bytes, may cause performance issues`,
)
}
// 0. DOM-based checks. Syntax errors are caught by the strict check at
// the end: linkedom's DOMParser never reports them.
try {
const parser = new DOMParser()
const doc = parser.parseFromString(xml, "text/xml")
// DOM-based checks for nested mxCell
const allCells = doc.querySelectorAll("mxCell")
for (const cell of allCells) {
if (cell.parentElement?.tagName === "mxCell") {
const id = cell.getAttribute("id") || "unknown"
return `Invalid XML: Found nested mxCell (id="${id}"). Cells should be siblings, not nested inside other mxCell elements.`
}
}
// draw.io reads any text inside a page as compressed page data and
// then fails to open the page
if (!doc.querySelector("parsererror")) {
const text = findTextBetweenTags(doc.documentElement)
if (text) {
return `Invalid XML: Found text "${text}" between tags. Labels belong in the value attribute; remove any other text between tags.`
}
}
} catch (error) {
console.warn(
"[validateMxCellStructure] DOMParser threw unexpected error, falling back to regex validation:",
error,
)
}
// 1. Check for CDATA wrapper (invalid at document root)
if (/^\s*<!\[CDATA\[/.test(xml)) {
return "Invalid XML: XML is wrapped in CDATA section - remove <![CDATA[ from start and ]]> from end"
}
// 2. Check for duplicate structural attributes
const dupAttrError = checkDuplicateAttributes(xml)
if (dupAttrError) {
return dupAttrError
}
// 3. Check for unescaped < in attribute values
const attrValuePattern = /=\s*"([^"]*)"/g
let attrValMatch
while ((attrValMatch = attrValuePattern.exec(xml)) !== null) {
const value = attrValMatch[1]
if (/</.test(value) && !/&lt;/.test(value)) {
return "Invalid XML: Unescaped < character in attribute values. Replace < with &lt;"
}
}
// 4. Check for duplicate IDs
const dupIdError = checkDuplicateIds(xml)
if (dupIdError) {
return dupIdError
}
// 5. Check for tag mismatches
const tagMismatchError = checkTagMismatches(xml)
if (tagMismatchError) {
return tagMismatchError
}
// 6. Check invalid character references
const charRefError = checkCharacterReferences(xml)
if (charRefError) {
return charRefError
}
// 7. Check for invalid comment syntax (-- inside comments)
const commentPattern = /<!--([\s\S]*?)-->/g
let commentMatch
while ((commentMatch = commentPattern.exec(xml)) !== null) {
if (/--/.test(commentMatch[1])) {
return "Invalid XML: Comment contains -- (double hyphen) which is not allowed"
}
}
// 8. Check for unescaped entity references and invalid entity names
const entityError = checkEntityReferences(xml)
if (entityError) {
return entityError
}
// 9. Check for empty id attributes on mxCell
if (/<mxCell[^>]*\sid\s*=\s*["']\s*["'][^>]*>/g.test(xml)) {
return "Invalid XML: Found mxCell element(s) with empty id attribute"
}
// 10. Check for nested mxCell tags
const nestedCellError = checkNestedMxCells(xml)
if (nestedCellError) {
return nestedCellError
}
if (opts.strict) {
const unknownError = checkUnknownElements(xml)
if (unknownError) {
return unknownError
}
if (findOrphanMxPoints(xml).length > 0) {
return 'Invalid XML: Found <mxPoint> without an "as" attribute outside <Array as="points">. Put waypoints inside <Array as="points"> or remove the point.'
}
}
// 11. Strict XML syntax check, run last so the checks above can give
// more specific messages. Catches what they miss, e.g. duplicate or
// unquoted attributes, which make draw.io refuse to load the diagram.
const syntaxError = getXmlSyntaxError(xml)
if (syntaxError) {
return `Invalid XML: syntax error at ${syntaxError} Escape special characters in attribute values (&lt; for <, &amp; for &, &quot; for "), quote every attribute value, and do not repeat an attribute.`
}
return null
}
// ============================================================================
// Auto-Fix Function
// ============================================================================
/**
* Attempts to auto-fix common XML issues in draw.io diagrams
* @param xml - The XML string to fix
* @returns Object with fixed XML and list of fixes applied
*/
export function autoFixXml(xml: string): { fixed: string; fixes: string[] } {
let fixed = xml
const fixes: string[] = []
// 0. Fix JSON-escaped XML
if (/=\\"/.test(fixed)) {
fixed = fixed.replace(/\\"/g, '"')
fixed = fixed.replace(/\\n/g, "\n")
fixes.push("Fixed JSON-escaped XML")
}
// 0b. Literal \n, \t or \r between tags, from escaping the XML twice
const unescaped = fixed.replace(/>(?:\s|\\[nrt])+</g, (gap) =>
gap.replace(/\\n/g, "\n").replace(/\\t/g, "\t").replace(/\\r/g, ""),
)
if (unescaped !== fixed) {
fixed = unescaped
fixes.push("Replaced literal \\n between tags with line breaks")
}
// 1. Remove CDATA wrapper
if (/^\s*<!\[CDATA\[/.test(fixed)) {
fixed = fixed.replace(/^\s*<!\[CDATA\[/, "").replace(/\]\]>\s*$/, "")
fixes.push("Removed CDATA wrapper")
}
// 2. Remove text before XML declaration or root element
const xmlStart = fixed.search(/<(\?xml|mxGraphModel|mxfile)/i)
if (xmlStart > 0 && !/^<[a-zA-Z]/.test(fixed.trim())) {
fixed = fixed.substring(xmlStart)
fixes.push("Removed text before XML root")
}
// 3. Fix duplicate attributes
let dupAttrFixed = false
const structural = new Set(STRUCTURAL_ATTRS)
fixed = fixed.replace(/<[^>]+>/g, (tag) => {
// Keep the first of each, drop the later ones
const seen = new Set<string>()
let newTag = ""
let last = 0
for (const attr of readAttributes(tag)) {
if (!structural.has(attr.name)) continue
if (!seen.has(attr.name)) {
seen.add(attr.name)
continue
}
newTag += tag.slice(last, attr.start)
last = attr.end
dupAttrFixed = true
}
return newTag + tag.slice(last)
})
if (dupAttrFixed) {
fixes.push("Removed duplicate structural attributes")
}
// 4. Fix unescaped & characters
const ampersandPattern =
/&(?!(?:lt|gt|amp|quot|apos|#[0-9]+|#x[0-9a-fA-F]+);)/g
if (ampersandPattern.test(fixed)) {
fixed = fixed.replace(
/&(?!(?:lt|gt|amp|quot|apos|#[0-9]+|#x[0-9a-fA-F]+);)/g,
"&amp;",
)
fixes.push("Escaped unescaped & characters")
}
// 5. Fix invalid entity names (double-escaping)
const invalidEntities = [
{ pattern: /&ampquot;/g, replacement: "&quot;", name: "&ampquot;" },
{ pattern: /&amplt;/g, replacement: "&lt;", name: "&amplt;" },
{ pattern: /&ampgt;/g, replacement: "&gt;", name: "&ampgt;" },
{ pattern: /&ampapos;/g, replacement: "&apos;", name: "&ampapos;" },
{ pattern: /&ampamp;/g, replacement: "&amp;", name: "&ampamp;" },
]
for (const { pattern, replacement, name } of invalidEntities) {
if (pattern.test(fixed)) {
fixed = fixed.replace(pattern, replacement)
fixes.push(`Fixed double-escaped entity ${name}`)
}
}
// 6. Fix malformed attribute quotes (name=&quot;value&quot;). Quoted
// values are matched first and kept, so &quot; inside a rich-text
// label like value="&lt;font style=&quot;...&quot;&gt;" is left alone.
let quotesFixed = false
fixed = replaceInOpeningTags(fixed, (tag) =>
tag.replace(
/("[^"]*"|'[^']*')|(\s[a-zA-Z][a-zA-Z0-9_:-]*)=&quot;([^&]*?)&quot;/g,
(match, quoted, name, value) => {
if (quoted) return match
quotesFixed = true
return `${name}="${value}"`
},
),
)
if (quotesFixed) {
fixes.push("Fixed malformed attribute quotes")
}
// 6b. Missing closing quote on a numeric attribute right before the
// self-closing slash (h="112/>), which a compact cell ends with. Numbers
// only, so a quote that belongs to a label is never touched.
const quoteBeforeSlash = /=("-?\d+(?:\.\d+)?)\/>/g
if (quoteBeforeSlash.test(fixed)) {
fixed = fixed.replace(quoteBeforeSlash, '=$1"/>')
fixes.push("Added a missing closing quote before />")
}
// 7. Fix malformed closing tags
const malformedClosingTag = /<\/([a-zA-Z][a-zA-Z0-9]*)\s*\/>/g
if (malformedClosingTag.test(fixed)) {
fixed = fixed.replace(/<\/([a-zA-Z][a-zA-Z0-9]*)\s*\/>/g, "</$1>")
fixes.push("Fixed malformed closing tags")
}
// 8. Fix missing space between attributes (id="2"vertex="1"). Every
// quoted value is consumed whole, so quotes always pair up within one
// attribute.
let spaceAdded = false
fixed = replaceInOpeningTags(fixed, (tag) =>
tag.replace(
/("[^"]*"|'[^']*')([a-zA-Z_:])?/g,
(match, quoted, next) => {
if (!next) return match
spaceAdded = true
return `${quoted} ${next}`
},
),
)
if (spaceAdded) {
fixes.push("Added missing space between attributes")
}
// 9. Fix unescaped quotes in style color values
const quotedColorPattern = /;([a-zA-Z]*[Cc]olor)="#/
if (quotedColorPattern.test(fixed)) {
fixed = fixed.replace(/;([a-zA-Z]*[Cc]olor)="#/g, ";$1=#")
fixes.push("Removed quotes around color values in style")
}
// 10. Fix unescaped < and > in attribute values
// < is required to be escaped, > is not strictly required but we escape for consistency
const attrPattern = /(=\s*")([^"]*?)(<)([^"]*?)(")/g
let attrMatch
let hasUnescapedLt = false
while ((attrMatch = attrPattern.exec(fixed)) !== null) {
if (!attrMatch[3].startsWith("&lt;")) {
hasUnescapedLt = true
break
}
}
if (hasUnescapedLt) {
fixed = fixed.replace(/=\s*"([^"]*)"/g, (_match, value) => {
const escaped = value.replace(/</g, "&lt;").replace(/>/g, "&gt;")
return `="${escaped}"`
})
fixes.push("Escaped <> characters in attribute values")
}
// 11. Fix invalid hex character references
const invalidHexRefs: string[] = []
fixed = fixed.replace(/&#x([^;]*);/g, (match, hex) => {
if (/^[0-9a-fA-F]+$/.test(hex) && hex.length > 0) {
return match
}
invalidHexRefs.push(match)
return ""
})
if (invalidHexRefs.length > 0) {
fixes.push(
`Removed ${invalidHexRefs.length} invalid hex character reference(s)`,
)
}
// 12. Fix invalid decimal character references
const invalidDecRefs: string[] = []
fixed = fixed.replace(/&#([^x][^;]*);/g, (match, dec) => {
if (/^[0-9]+$/.test(dec) && dec.length > 0) {
return match
}
invalidDecRefs.push(match)
return ""
})
if (invalidDecRefs.length > 0) {
fixes.push(
`Removed ${invalidDecRefs.length} invalid decimal character reference(s)`,
)
}
// 13. Fix invalid comment syntax
fixed = fixed.replace(/<!--([\s\S]*?)-->/g, (match, content) => {
if (/--/.test(content)) {
let fixedContent = content
while (/--/.test(fixedContent)) {
fixedContent = fixedContent.replace(/--/g, "-")
}
fixes.push("Fixed invalid comment syntax")
return `<!--${fixedContent}-->`
}
return match
})
// 14. Fix <Cell> tags to <mxCell>
const hasCellTags = /<\/?Cell[\s>]/i.test(fixed)
if (hasCellTags) {
fixed = fixed.replace(/<Cell(\s)/gi, "<mxCell$1")
fixed = fixed.replace(/<Cell>/gi, "<mxCell>")
fixed = fixed.replace(/<\/Cell>/gi, "</mxCell>")
fixes.push("Fixed <Cell> tags to <mxCell>")
}
// 15. Fix closing tag typos and wrong tag case, e.g. <mxcell> (MUST run
// before foreign tag removal, which would otherwise delete them)
const before15 = fixed
fixed = fixed.replace(/<\/mxElement>/gi, "</mxCell>")
if (fixed !== before15) {
fixes.push("Fixed typo </mxElement> to </mxCell>")
}
for (const name of ["mxCell", "mxGeometry", "mxPoint", "mxGraphModel"]) {
let changed = false
fixed = fixed.replace(
new RegExp(`<(/?)${name}(?=[\\s/>])`, "gi"),
(match, slash) => {
const right = `<${slash}${name}`
if (match !== right) changed = true
return right
},
)
if (changed) {
fixes.push(`Fixed tag case of <${name}>`)
}
}
// 16. Remove non-draw.io tags (after the case fixes above). Removes only
// the exact tag occurrences and skips quoted attribute values, so a stray
// <mxGraph/> never takes <mxGraphModel> with it and <b> inside
// value="..." stays.
const isInsideQuotesFor16 = createQuoteTracker(fixed)
const foreignTagPattern = /<\/?([a-zA-Z][a-zA-Z0-9_]*)[^>]*>/g
let foreignMatch
const foreignTags = new Set<string>()
const foreignTagPositions: Array<{ start: number; end: number }> = []
while ((foreignMatch = foreignTagPattern.exec(fixed)) !== null) {
const tagName = foreignMatch[1]
if (VALID_DRAWIO_TAGS.has(tagName)) continue
if (isInsideQuotesFor16(foreignMatch.index)) continue
foreignTags.add(tagName)
foreignTagPositions.push({
start: foreignMatch.index,
end: foreignMatch.index + foreignMatch[0].length,
})
}
if (foreignTagPositions.length > 0) {
// Remove from the end so earlier positions stay valid
for (const { start, end } of foreignTagPositions.reverse()) {
fixed = fixed.slice(0, start) + fixed.slice(end)
}
fixes.push(
`Removed foreign tags: ${Array.from(foreignTags).join(", ")}`,
)
}
// 16b. Remove orphan <mxPoint>s (no "as" attribute, not inside
// <Array as="points">), which draw.io refuses to load
const orphanPoints = findOrphanMxPoints(fixed)
if (orphanPoints.length > 0) {
for (const { start, end } of orphanPoints.reverse()) {
fixed = fixed.slice(0, start) + fixed.slice(end)
}
fixes.push(`Removed ${orphanPoints.length} orphan <mxPoint>(s)`)
}
// 17. Fix unclosed tags
const tagStack: string[] = []
const parsedTags = parseXmlTags(fixed)
for (const { tagName, isClosing, isSelfClosing } of parsedTags) {
if (isClosing) {
const lastIdx = tagStack.lastIndexOf(tagName)
if (lastIdx !== -1) {
tagStack.splice(lastIdx, 1)
}
} else if (!isSelfClosing) {
tagStack.push(tagName)
}
}
if (tagStack.length > 0) {
const tagsToClose: string[] = []
for (const tagName of tagStack.reverse()) {
const openCount = (
fixed.match(new RegExp(`<${tagName}[\\s>]`, "gi")) || []
).length
const closeCount = (
fixed.match(new RegExp(`</${tagName}>`, "gi")) || []
).length
if (openCount > closeCount) {
tagsToClose.push(tagName)
}
}
if (tagsToClose.length > 0) {
const closingTags = tagsToClose.map((t) => `</${t}>`).join("\n")
fixed = fixed.trimEnd() + "\n" + closingTags
fixes.push(
`Closed ${tagsToClose.length} unclosed tag(s): ${tagsToClose.join(", ")}`,
)
}
}
// 18. Remove extra closing tags. Counts only draw.io tags outside quoted
// attribute values (value="<b>Title</b>" holds HTML, not elements).
const tagCounts = new Map<
string,
{ opens: number; closes: number; selfClosing: number }
>()
const fullTagPattern = /<(\/?[a-zA-Z][a-zA-Z0-9]*)[^>]*>/g
const isInsideQuotesFor18 = createQuoteTracker(fixed)
let tagCountMatch
while ((tagCountMatch = fullTagPattern.exec(fixed)) !== null) {
if (isInsideQuotesFor18(tagCountMatch.index)) continue
const fullMatch = tagCountMatch[0]
const tagPart = tagCountMatch[1]
const isClosing = tagPart.startsWith("/")
const isSelfClosing = fullMatch.endsWith("/>")
const tagName = isClosing ? tagPart.slice(1) : tagPart
if (!VALID_DRAWIO_TAGS.has(tagName)) continue
let counts = tagCounts.get(tagName)
if (!counts) {
counts = { opens: 0, closes: 0, selfClosing: 0 }
tagCounts.set(tagName, counts)
}
if (isClosing) {
counts.closes++
} else if (isSelfClosing) {
counts.selfClosing++
} else {
counts.opens++
}
}
for (const [tagName, counts] of tagCounts) {
const extraCloses = counts.closes - counts.opens
if (extraCloses > 0) {
let removed = 0
const closeTagPattern = new RegExp(`</${tagName}>`, "g")
const matches = [...fixed.matchAll(closeTagPattern)]
for (
let i = matches.length - 1;
i >= 0 && removed < extraCloses;
i--
) {
const match = matches[i]
const idx = match.index ?? 0
fixed = fixed.slice(0, idx) + fixed.slice(idx + match[0].length)
removed++
}
if (removed > 0) {
fixes.push(
`Removed ${removed} extra </${tagName}> closing tag(s)`,
)
}
}
}
// 19. Remove trailing garbage after last XML tag
const closingTagPattern = /<\/[a-zA-Z][a-zA-Z0-9]*>|\/>/g
let lastValidTagEnd = -1
let closingMatch
while ((closingMatch = closingTagPattern.exec(fixed)) !== null) {
lastValidTagEnd = closingMatch.index + closingMatch[0].length
}
if (lastValidTagEnd > 0 && lastValidTagEnd < fixed.length) {
const trailing = fixed.slice(lastValidTagEnd).trim()
if (trailing) {
fixed = fixed.slice(0, lastValidTagEnd)
fixes.push("Removed trailing garbage after last XML tag")
}
}
// 20. Fix nested mxCell by flattening
const lines = fixed.split("\n")
let newLines: string[] = []
let nestedFixed = 0
let extraClosingToRemove = 0
for (let i = 0; i < lines.length; i++) {
const line = lines[i]
const nextLine = lines[i + 1]
if (
nextLine &&
/<mxCell\s/.test(line) &&
/<mxCell\s/.test(nextLine) &&
!line.includes("/>") &&
!nextLine.includes("/>")
) {
const id1 = line.match(/\bid\s*=\s*["']([^"']+)["']/)?.[1]
const id2 = nextLine.match(/\bid\s*=\s*["']([^"']+)["']/)?.[1]
if (id1 && id1 === id2) {
nestedFixed++
extraClosingToRemove++
continue
}
}
if (extraClosingToRemove > 0 && /^\s*<\/mxCell>\s*$/.test(line)) {
extraClosingToRemove--
continue
}
newLines.push(line)
}
if (nestedFixed > 0) {
fixed = newLines.join("\n")
fixes.push(`Flattened ${nestedFixed} duplicate-ID nested mxCell(s)`)
}
// 21. Fix true nested mxCell (different IDs). Runs only when the nesting
// check finds real nesting, because this line-based rewrite can break
// valid cells written over several lines.
const lines2 = checkNestedMxCells(fixed) ? fixed.split("\n") : []
newLines = []
let trueNestedFixed = 0
let cellDepth = 0
let pendingCloseRemoval = 0
for (let i = 0; i < lines2.length; i++) {
const line = lines2[i]
const trimmed = line.trim()
// A line holding a whole cell (<mxCell ...>...</mxCell>) opens nothing
const isOpenCell =
/<mxCell\s/.test(trimmed) &&
!trimmed.endsWith("/>") &&
!trimmed.endsWith("</mxCell>")
const isCloseCell = trimmed === "</mxCell>"
if (isOpenCell) {
if (cellDepth > 0) {
const indent = line.match(/^(\s*)/)?.[1] || ""
newLines.push(indent + "</mxCell>")
trueNestedFixed++
pendingCloseRemoval++
}
cellDepth = 1
newLines.push(line)
} else if (isCloseCell) {
if (pendingCloseRemoval > 0) {
pendingCloseRemoval--
} else {
cellDepth = Math.max(0, cellDepth - 1)
newLines.push(line)
}
} else {
newLines.push(line)
}
}
if (trueNestedFixed > 0) {
fixed = newLines.join("\n")
fixes.push(`Fixed ${trueNestedFixed} true nested mxCell(s)`)
}
// 22. Fix duplicate IDs by appending suffix.
// Skipped for multi-page <mxfile> documents — cell ids "0" and "1" repeat
// across pages legitimately (every page has its own <root> with id="0"/"1"
// sentinel cells). Renaming them would break drawio's parent references.
// For mxfile inputs, duplicate-id validation is page-scoped in
// checkDuplicateIds() and a true duplicate produces a hard error rather
// than a silent rename.
if (!/<mxfile[\s>]/i.test(fixed)) {
const seenIds = new Map<string, number>()
const duplicateIds: string[] = []
const idPattern = /\bid\s*=\s*["']([^"']+)["']/gi
let idMatch
while ((idMatch = idPattern.exec(fixed)) !== null) {
const id = idMatch[1]
seenIds.set(id, (seenIds.get(id) || 0) + 1)
}
for (const [id, count] of seenIds) {
if (count > 1) duplicateIds.push(id)
}
if (duplicateIds.length > 0) {
const idCounters = new Map<string, number>()
// Rebuild from the captured parts so only the value changes (an id
// like "d" or "i" also occurs in the attribute name itself)
fixed = fixed.replace(
/(\bid\s*=\s*["'])([^"']+)(["'])/gi,
(match, before, id, after) => {
if (!duplicateIds.includes(id)) return match
const count = idCounters.get(id) || 0
idCounters.set(id, count + 1)
if (count === 0) return match
return `${before}${id}_dup${count}${after}`
},
)
fixes.push(`Renamed ${duplicateIds.length} duplicate ID(s)`)
}
}
// 23. Fix empty id attributes
let emptyIdCount = 0
fixed = fixed.replace(
/<mxCell([^>]*)\sid\s*=\s*["']\s*["']([^>]*)>/g,
(_match, before, after) => {
emptyIdCount++
const newId = `cell_${Date.now()}_${emptyIdCount}`
return `<mxCell${before} id="${newId}"${after}>`
},
)
if (emptyIdCount > 0) {
fixes.push(`Generated ${emptyIdCount} missing ID(s)`)
}
return { fixed, fixes }
}
// ============================================================================
// Combined Validation and Fix
// ============================================================================
/**
* Validates XML and attempts to fix if invalid. By default runs the strict
* checks (unknown elements, orphan mxPoints), meant for XML the model wrote.
* Pass strict: false for a diagram that also holds the user's own content.
* @param xml - The XML string to validate and potentially fix
* @returns Object with validation result, fixed XML if applicable, and fixes applied
*/
export function validateAndFixXml(
xml: string,
{ strict = true }: { strict?: boolean } = {},
): {
valid: boolean
error: string | null
fixed: string | null
fixes: string[]
} {
// First validation attempt
let error = validateMxCellStructure(xml, { strict })
if (!error) {
return { valid: true, error: null, fixed: null, fixes: [] }
}
// Try to fix
const { fixed, fixes } = autoFixXml(xml)
// Validate the fixed version
error = validateMxCellStructure(fixed, { strict })
if (!error) {
return { valid: true, error: null, fixed, fixes }
}
// Still invalid after fixes
return {
valid: false,
error,
fixed: fixes.length > 0 ? fixed : null,
fixes,
}
}