2025-12-06 12:46:40 +09:00
import { type ClassValue , clsx } from "clsx"
import * as pako from "pako"
2025-03-19 06:04:06 +00:00
import { twMerge } from "tailwind-merge"
2026-01-10 14:06:17 +09:00
import type { DiagramOperation } from "@/components/chat/types"
export type { DiagramOperation }
2025-03-19 06:04:06 +00:00
export function cn (... inputs : ClassValue []) {
2025-12-06 12:46:40 +09:00
return twMerge ( clsx ( inputs ))
2025-03-19 06:04:06 +00:00
}
2025-03-22 15:45:49 +00:00
2026-01-04 10:25:19 +09:00
// ============================================================================
// Diagram Constants
// ============================================================================
/**
* Minimum length for a "real" diagram XML (not just empty template).
* Empty mxfile templates are ~147-300 chars; real diagrams are larger.
*/
export const MIN_REAL_DIAGRAM_LENGTH = 300
/**
* Check if diagram XML represents a real diagram (not just empty template).
* @param xml - The diagram XML string to check
* @returns true if the XML is a real diagram with content
*/
export function isRealDiagram ( xml : string | undefined | null ) : boolean {
return !! xml && xml . length > MIN_REAL_DIAGRAM_LENGTH
}
2025-12-13 23:31:01 +09:00
// ============================================================================
// XML Validation/Fix Constants
// ============================================================================
/** Maximum XML size to process (1MB) - larger XMLs may cause performance issues */
const MAX_XML_SIZE = 1 _000_000
/** Maximum iterations for aggressive cell dropping to prevent infinite loops */
const MAX_DROP_ITERATIONS = 10
/** Structural attributes that should not be duplicated in draw.io */
const STRUCTURAL_ATTRS = [
"edge" ,
"parent" ,
"source" ,
"target" ,
"vertex" ,
"connectable" ,
]
/** Valid XML entity names */
const VALID_ENTITIES = new Set ([ "lt" , "gt" , "amp" , "quot" , "apos" ])
2025-12-14 14:04:44 +09:00
// ============================================================================
// mxCell XML Helpers
// ============================================================================
/**
* Check if mxCell XML output is complete (not truncated).
* Complete XML ends with a self-closing tag (/>) or closing mxCell tag.
2025-12-24 09:31:54 +09:00
* Uses a robust approach that handles any LLM provider's wrapper tags
* by finding the last valid mxCell ending and checking if suffix is just closing tags.
2025-12-14 14:04:44 +09:00
* @param xml - The XML string to check (can be undefined/null)
* @returns true if XML appears complete, false if truncated or empty
*/
export function isMxCellXmlComplete ( xml : string | undefined | null ) : boolean {
2025-12-24 09:31:54 +09:00
const trimmed = xml ? . trim () || ""
2025-12-14 14:04:44 +09:00
if ( ! trimmed ) return false
2025-12-14 19:38:40 +09:00
2025-12-24 09:31:54 +09:00
// Find position of last complete mxCell ending (either /> or </mxCell>)
const lastSelfClose = trimmed . lastIndexOf ( "/>" )
const lastMxCellClose = trimmed . lastIndexOf ( "</mxCell>" )
2025-12-14 19:38:40 +09:00
2025-12-24 09:31:54 +09:00
const lastValidEnd = Math . max ( lastSelfClose , lastMxCellClose )
// No valid ending found at all
if ( lastValidEnd === - 1 ) return false
2026-10-03 17:45:25 +09:00
// If the last mxCell has no </mxCell> after it, it must be self-closing.
// Otherwise the trailing "/>" belongs to a child such as <mxGeometry .../>
// and the output was cut off before the cell was closed.
const lastCellStart = trimmed . lastIndexOf ( "<mxCell" )
if (
lastCellStart > lastMxCellClose &&
! /^<mxCell\b[^<]*\/>/ . test ( trimmed . slice ( lastCellStart ))
) {
return false
}
2025-12-24 09:31:54 +09:00
// Check what comes after the last valid ending
// For />: add 2 chars, for </mxCell>: add 9 chars
const endOffset = lastMxCellClose > lastSelfClose ? 9 : 2
const suffix = trimmed . slice ( lastValidEnd + endOffset )
// If suffix is empty or only contains closing tags (any provider's wrapper) or whitespace, it's complete
// This regex matches any sequence of closing XML tags like </foo>, </bar>, </| DSML| xyz>
return /^(\s*<\/[^>]+>)*\s*$/ . test ( suffix )
2025-12-14 14:04:44 +09:00
}
2025-12-23 18:54:03 +09:00
/**
* Extract only complete mxCell elements from partial/streaming XML.
* This allows progressive rendering during streaming by ignoring incomplete trailing elements.
* @param xml - The partial XML string (may contain incomplete trailing mxCell)
* @returns XML string containing only complete mxCell elements
*/
export function extractCompleteMxCells ( xml : string | undefined | null ) : string {
if ( ! xml ) return ""
2026-10-03 17:45:25 +09:00
// Match self-closing <mxCell ... /> or <mxCell ...>...</mxCell>, in document order.
// The lazy [^>]*? tries "/>" first, so a self-closing cell never swallows
// the following cells up to the next </mxCell>.
const cellPattern = /<mxCell\b[^>]*?(?:\/>|>[\s\S]*?<\/mxCell>)/g
2025-12-23 18:54:03 +09:00
2026-10-03 17:45:25 +09:00
return ( xml . match ( cellPattern ) || []). join ( "\n" )
2025-12-23 18:54:03 +09:00
}
2025-12-13 23:31:01 +09:00
// ============================================================================
// XML Parsing Helpers
// ============================================================================
interface ParsedTag {
tag : string
tagName : string
isClosing : boolean
isSelfClosing : boolean
startIndex : number
endIndex : number
}
/**
* Parse XML tags while properly handling quoted strings
* This is a shared utility used by both validation and fixing logic
*/
function parseXmlTags ( xml : string ) : ParsedTag [] {
const tags : ParsedTag [] = []
let i = 0
while ( i < xml . length ) {
const tagStart = xml . indexOf ( "<" , i )
if ( tagStart === - 1 ) break
// Find matching > by tracking quotes
let tagEnd = tagStart + 1
let inQuote = false
let quoteChar = ""
while ( tagEnd < xml . length ) {
const c = xml [ tagEnd ]
if ( inQuote ) {
if ( c === quoteChar ) inQuote = false
} else {
if ( c === '"' || c === "'" ) {
inQuote = true
quoteChar = c
} else if ( c === ">" ) {
break
}
}
tagEnd ++
}
if ( tagEnd >= xml . length ) break
const tag = xml . substring ( tagStart , tagEnd + 1 )
i = tagEnd + 1
const tagMatch = /^<(\/?)([a-zA-Z][a-zA-Z0-9:_-]*)/ . exec ( tag )
if ( ! tagMatch ) continue
tags . push ({
tag ,
tagName : tagMatch [ 2 ],
isClosing : tagMatch [ 1 ] === "/" ,
isSelfClosing : tag.endsWith ( "/>" ),
startIndex : tagStart ,
endIndex : tagEnd ,
})
}
return tags
}
2025-08-31 20:52:04 +09:00
/**
* Format XML string with proper indentation and line breaks
* @param xml - The XML string to format
* @param indent - The indentation string (default: ' ')
* @returns Formatted XML string
*/
2025-12-06 12:46:40 +09:00
export function formatXML ( xml : string , indent : string = " " ) : string {
let formatted = ""
let pad = 0
2025-08-31 20:52:04 +09:00
2025-12-06 12:46:40 +09:00
// Remove existing whitespace between tags
xml = xml . replace ( />\s*</g , "><" ). trim ()
2025-08-31 20:52:04 +09:00
2025-12-06 12:46:40 +09:00
// Split on tags
const tags = xml . split ( /(?=<)|(?<=>)/g ). filter ( Boolean )
2025-08-31 20:52:04 +09:00
2025-12-06 12:46:40 +09:00
tags . forEach (( node ) => {
if ( node . match ( /^<\/\w/ )) {
// Closing tag - decrease indent
pad = Math . max ( 0 , pad - 1 )
formatted += indent . repeat ( pad ) + node + "\n"
} else if ( node . match ( /^<\w[^>]*[^/]>.*$/ )) {
// Opening tag
formatted += indent . repeat ( pad ) + node
// Only add newline if next item is a tag
const nextIndex = tags . indexOf ( node ) + 1
if ( nextIndex < tags . length && tags [ nextIndex ]. startsWith ( "<" )) {
formatted += "\n"
if ( ! node . match ( /^<\w[^>]*\/>$/ )) {
pad ++
}
}
} else if ( node . match ( /^<\w[^>]*\/>$/ )) {
// Self-closing tag
formatted += indent . repeat ( pad ) + node + "\n"
} else if ( node . startsWith ( "<" )) {
// Other tags (like <?xml)
formatted += indent . repeat ( pad ) + node + "\n"
} else {
// Text content
formatted += node
2025-08-31 20:52:04 +09:00
}
2025-12-06 12:46:40 +09:00
})
2025-08-31 20:52:04 +09:00
2025-12-06 12:46:40 +09:00
return formatted . trim ()
2025-08-31 20:52:04 +09:00
}
2025-12-06 12:46:40 +09:00
/**
2025-03-25 08:56:24 +00:00
* Efficiently converts a potentially incomplete XML string to a legal XML string by closing any open tags properly.
* Additionally, if an <mxCell> tag does not have an mxGeometry child (e.g. <mxCell id="3">),
* it removes that tag from the output.
2025-12-07 00:40:19 +09:00
* Also removes orphaned <mxPoint> elements that aren't inside <Array> or don't have proper 'as' attribute.
2025-03-22 15:45:49 +00:00
* @param xmlString The potentially incomplete XML string
2025-03-25 08:56:24 +00:00
* @returns A legal XML string with properly closed tags and removed incomplete mxCell elements.
2025-03-22 15:45:49 +00:00
*/
export function convertToLegalXml ( xmlString : string ) : string {
2025-12-06 12:46:40 +09:00
// This regex will match either self-closing <mxCell .../> or a block element
// <mxCell ...> ... </mxCell>. Unfinished ones are left out because they don't match.
const regex = /<mxCell\b[^>]*(?:\/>|>([\s\S]*?)<\/mxCell>)/g
let match : RegExpExecArray | null
let result = "<root>\n"
2025-03-22 15:45:49 +00:00
2025-12-06 12:46:40 +09:00
while (( match = regex . exec ( xmlString )) !== null ) {
// match[0] contains the entire matched mxCell block
2025-12-07 00:40:19 +09:00
let cellContent = match [ 0 ]
// Remove orphaned <mxPoint> elements that are directly inside <mxGeometry>
// without an 'as' attribute (like as="sourcePoint", as="targetPoint")
// and not inside <Array as="points">
// These cause "Could not add object mxPoint" errors in draw.io
// First check if there's an <Array as="points"> - if so, keep all mxPoints inside it
const hasArrayPoints = /<Array\s+as="points">/ . test ( cellContent )
if ( ! hasArrayPoints ) {
// Remove mxPoint elements without 'as' attribute
cellContent = cellContent . replace (
/<mxPoint\b[^>]*\/>/g ,
( pointMatch ) => {
// Keep if it has an 'as' attribute
if ( /\sas=/ . test ( pointMatch )) {
return pointMatch
}
// Remove orphaned mxPoint
return ""
},
)
}
2025-12-14 19:38:40 +09:00
// Fix unescaped & characters in attribute values (but not valid entities)
// This prevents DOMParser from failing on content like "semantic & missing-step"
cellContent = cellContent . replace (
/&(?!(?:lt|gt|amp|quot|apos|#[0-9]+|#x[0-9a-fA-F]+);)/g ,
"&" ,
)
2025-12-24 09:31:54 +09:00
// Fix unescaped < and > in attribute values for XML parsing
// HTML content in value attributes (e.g., <b>Title</b>) needs to be escaped
// This is critical because DOMParser will fail on unescaped < > in attributes
if ( /=\s*"[^"]*<[^"]*"/ . test ( cellContent )) {
cellContent = cellContent . replace (
/=\s*"([^"]*)"/g ,
( _match , value ) => {
const escaped = value
. replace ( /</g , "<" )
. replace ( />/g , ">" )
return `=" ${ escaped } "`
},
)
}
2025-12-06 12:46:40 +09:00
// Indent each line of the matched block for readability.
2025-12-07 00:40:19 +09:00
const formatted = cellContent
2025-12-06 12:46:40 +09:00
. split ( "\n" )
. map (( line ) => " " + line . trim ())
2025-12-07 00:40:19 +09:00
. filter (( line ) => line . trim ()) // Remove empty lines from removed mxPoints
2025-12-06 12:46:40 +09:00
. join ( "\n" )
result += formatted + "\n"
}
result += "</root>"
2025-03-22 15:45:49 +00:00
2025-12-06 12:46:40 +09:00
return result
2025-03-25 02:24:12 +00:00
}
2025-12-09 15:53:59 +09:00
/**
* Wrap XML content with the full mxfile structure required by draw.io.
2025-12-14 14:04:44 +09:00
* Always adds root cells (id="0" and id="1") automatically.
* If input already contains root cells, they are removed to avoid duplication.
* LLM should only generate mxCell elements starting from id="2".
* @param xml - The XML string (bare mxCells, <root>, <mxGraphModel>, or full <mxfile>)
* @returns Full mxfile-wrapped XML string with root cells included
2025-12-09 15:53:59 +09:00
*/
export function wrapWithMxFile ( xml : string ) : string {
2025-12-14 14:04:44 +09:00
const ROOT_CELLS = '<mxCell id="0"/><mxCell id="1" parent="0"/>'
if ( ! xml || ! xml . trim ()) {
return `<mxfile><diagram name="Page-1" id="page-1"><mxGraphModel><root> ${ ROOT_CELLS } </root></mxGraphModel></diagram></mxfile>`
2025-12-09 15:53:59 +09:00
}
// Already has full structure
if ( xml . includes ( "<mxfile" )) {
return xml
}
// Has mxGraphModel but not mxfile
if ( xml . includes ( "<mxGraphModel" )) {
return `<mxfile><diagram name="Page-1" id="page-1"> ${ xml } </diagram></mxfile>`
}
2025-12-14 14:04:44 +09:00
// Has <root> wrapper - extract inner content
let content = xml
if ( xml . includes ( "<root>" )) {
content = xml . replace ( /<\/?root>/g , "" ). trim ()
}
2025-12-24 09:31:54 +09:00
// Strip trailing LLM wrapper tags (from any provider: Anthropic, DeepSeek, etc.)
// Find the last valid mxCell ending and remove everything after it
const lastSelfClose = content . lastIndexOf ( "/>" )
const lastMxCellClose = content . lastIndexOf ( "</mxCell>" )
const lastValidEnd = Math . max ( lastSelfClose , lastMxCellClose )
if ( lastValidEnd !== - 1 ) {
const endOffset = lastMxCellClose > lastSelfClose ? 9 : 2
const suffix = content . slice ( lastValidEnd + endOffset )
// If suffix is only closing tags (wrapper tags), strip it
if ( /^(\s*<\/[^>]+>)*\s*$/ . test ( suffix )) {
content = content . slice ( 0 , lastValidEnd + endOffset )
}
}
2025-12-14 14:04:44 +09:00
// Remove any existing root cells from content (LLM shouldn't include them, but handle it gracefully)
// Use flexible patterns that match both self-closing (/>) and non-self-closing (></mxCell>) formats
content = content
. replace ( /<mxCell[^>]*\bid=["']0["'][^>]*(?:\/>|><\/mxCell>)/g , "" )
. replace ( /<mxCell[^>]*\bid=["']1["'][^>]*(?:\/>|><\/mxCell>)/g , "" )
. trim ()
return `<mxfile><diagram name="Page-1" id="page-1"><mxGraphModel><root> ${ ROOT_CELLS }${ content } </root></mxGraphModel></diagram></mxfile>`
2025-12-09 15:53:59 +09:00
}
2025-03-25 08:56:24 +00:00
/**
* Replace nodes in a Draw.io XML diagram
* @param currentXML - The original Draw.io XML string
* @param nodes - The XML string containing new nodes to replace in the diagram
* @returns The updated XML string with replaced nodes
*/
export function replaceNodes ( currentXML : string , nodes : string ) : string {
2025-12-06 12:46:40 +09:00
// Check for valid inputs
if ( ! currentXML || ! nodes ) {
throw new Error ( "Both currentXML and nodes must be provided" )
2025-03-25 08:56:24 +00:00
}
2025-12-06 12:46:40 +09:00
try {
// Parse the XML strings to create DOM objects
const parser = new DOMParser ()
const currentDoc = parser . parseFromString ( currentXML , "text/xml" )
2025-03-25 08:56:24 +00:00
2025-12-06 12:46:40 +09:00
// Handle nodes input - if it doesn't contain <root>, wrap it
let nodesString = nodes
if ( ! nodes . includes ( "<root>" )) {
nodesString = `<root> ${ nodes } </root>`
}
2025-03-25 08:56:24 +00:00
2025-12-06 12:46:40 +09:00
const nodesDoc = parser . parseFromString ( nodesString , "text/xml" )
2025-03-25 08:56:24 +00:00
2025-12-06 12:46:40 +09:00
// Find the root element in the current document
let currentRoot = currentDoc . querySelector ( "mxGraphModel > root" )
if ( ! currentRoot ) {
// If no root element is found, create the proper structure
const mxGraphModel =
currentDoc . querySelector ( "mxGraphModel" ) ||
currentDoc . createElement ( "mxGraphModel" )
if ( ! currentDoc . contains ( mxGraphModel )) {
currentDoc . appendChild ( mxGraphModel )
}
currentRoot = currentDoc . createElement ( "root" )
mxGraphModel . appendChild ( currentRoot )
}
// Find the root element in the nodes document
const nodesRoot = nodesDoc . querySelector ( "root" )
if ( ! nodesRoot ) {
throw new Error (
"Invalid nodes: Could not find or create <root> element" ,
)
}
// Clear all existing child elements from the current root
while ( currentRoot . firstChild ) {
currentRoot . removeChild ( currentRoot . firstChild )
}
// Ensure the base cells exist
const hasCell0 = Array . from ( nodesRoot . childNodes ). some (
( node ) =>
node . nodeName === "mxCell" &&
( node as Element ). getAttribute ( "id" ) === "0" ,
)
const hasCell1 = Array . from ( nodesRoot . childNodes ). some (
( node ) =>
node . nodeName === "mxCell" &&
( node as Element ). getAttribute ( "id" ) === "1" ,
)
// Copy all child nodes from the nodes root to the current root
Array . from ( nodesRoot . childNodes ). forEach (( node ) => {
const importedNode = currentDoc . importNode ( node , true )
currentRoot . appendChild ( importedNode )
})
// Add default cells if they don't exist
if ( ! hasCell0 ) {
const cell0 = currentDoc . createElement ( "mxCell" )
cell0 . setAttribute ( "id" , "0" )
currentRoot . insertBefore ( cell0 , currentRoot . firstChild )
}
if ( ! hasCell1 ) {
const cell1 = currentDoc . createElement ( "mxCell" )
cell1 . setAttribute ( "id" , "1" )
cell1 . setAttribute ( "parent" , "0" )
// Insert after cell0 if possible
const cell0 = currentRoot . querySelector ( 'mxCell[id="0"]' )
2025-12-06 16:18:26 +09:00
if ( cell0 ? . nextSibling ) {
2025-12-06 12:46:40 +09:00
currentRoot . insertBefore ( cell1 , cell0 . nextSibling )
} else {
currentRoot . appendChild ( cell1 )
}
}
// Convert the modified DOM back to a string
const serializer = new XMLSerializer ()
return serializer . serializeToString ( currentDoc )
} catch ( error ) {
throw new Error ( `Error replacing nodes: ${ error } ` )
2025-03-25 08:56:24 +00:00
}
2025-03-27 06:45:38 +00:00
}
2025-12-15 14:22:56 +09:00
// ============================================================================
// ID-based Diagram Operations
// ============================================================================
export interface OperationError {
type : "update" | "add" | "delete"
cellId : string
message : string
}
export interface ApplyOperationsResult {
result : string
errors : OperationError []
2025-12-04 13:26:06 +09:00
}
2026-10-03 17:45:25 +09:00
/**
* draw.io wraps cells that have links, tooltips or custom data in
* <object>/<UserObject>, and the wrapper carries the id instead of the mxCell.
*/
function getCellWrapper ( cell : Element ) : Element | null {
const parent = cell . parentElement
return parent ? . tagName === "object" || parent ? . tagName === "UserObject"
? parent
: null
}
/** Id of a cell, read from its wrapper when the mxCell has none */
function getCellId ( cell : Element ) : string | null {
return (
cell . getAttribute ( "id" ) ||
getCellWrapper ( cell ) ? . getAttribute ( "id" ) ||
null
)
}
/** Element to replace or remove for a cell (the wrapper if there is one) */
function getCellNode ( cell : Element ) : Element {
return getCellWrapper ( cell ) || cell
}
2025-12-04 13:26:06 +09:00
/**
2025-12-15 14:22:56 +09:00
* Apply diagram operations (update/add/delete) using ID-based lookup.
* This replaces the text-matching approach with direct DOM manipulation.
*
* @param xmlContent - The full mxfile XML content
* @param operations - Array of operations to apply
* @returns Object with result XML and any errors
2025-12-04 13:26:06 +09:00
*/
2025-12-15 14:22:56 +09:00
export function applyDiagramOperations (
2025-12-06 12:46:40 +09:00
xmlContent : string ,
2025-12-15 14:22:56 +09:00
operations : DiagramOperation [],
) : ApplyOperationsResult {
const errors : OperationError [] = []
2025-03-27 06:45:38 +00:00
2025-12-15 14:22:56 +09:00
// Parse the XML
const parser = new DOMParser ()
const doc = parser . parseFromString ( xmlContent , "text/xml" )
2025-08-31 20:52:04 +09:00
2025-12-15 14:22:56 +09:00
// Check for parse errors
const parseError = doc . querySelector ( "parsererror" )
if ( parseError ) {
return {
result : xmlContent ,
errors : [
{
type : "update" ,
cellId : "" ,
message : `XML parse error: ${ parseError . textContent } ` ,
},
],
2025-08-31 20:52:04 +09:00
}
2025-12-04 22:56:59 +09:00
}
2025-12-15 14:22:56 +09:00
// Find the root element (inside mxGraphModel)
const root = doc . querySelector ( "root" )
if ( ! root ) {
return {
result : xmlContent ,
errors : [
{
type : "update" ,
cellId : "" ,
message : "Could not find <root> element in XML" ,
},
],
}
}
2026-10-03 17:45:25 +09:00
// Build a map of cell IDs to elements (wrapper elements for wrapped cells)
2025-12-15 14:22:56 +09:00
const cellMap = new Map < string , Element >()
root . querySelectorAll ( "mxCell" ). forEach (( cell ) => {
2026-10-03 17:45:25 +09:00
const id = getCellId ( cell )
if ( id ) cellMap . set ( id , getCellNode ( cell ))
2025-12-15 14:22:56 +09:00
})
2026-10-03 17:45:25 +09:00
// Cells removed by delete operations in this batch
const deletedIds = new Set < string >()
2025-12-15 14:22:56 +09:00
// Process each operation
for ( const op of operations ) {
2025-12-25 13:19:04 +09:00
if ( op . operation === "update" ) {
2025-12-15 14:22:56 +09:00
const existingCell = cellMap . get ( op . cell_id )
if ( ! existingCell ) {
errors . push ({
type : "update" ,
cellId : op.cell_id ,
message : `Cell with id=" ${ op . cell_id } " not found` ,
})
continue
}
if ( ! op . new_xml ) {
errors . push ({
type : "update" ,
cellId : op.cell_id ,
message : "new_xml is required for update operation" ,
})
continue
}
// Parse the new XML
const newDoc = parser . parseFromString (
`<wrapper> ${ op . new_xml } </wrapper>` ,
"text/xml" ,
)
const newCell = newDoc . querySelector ( "mxCell" )
if ( ! newCell ) {
errors . push ({
type : "update" ,
cellId : op.cell_id ,
message : "new_xml must contain an mxCell element" ,
})
continue
}
// Validate ID matches
2026-10-03 17:45:25 +09:00
const newCellId = getCellId ( newCell )
2025-12-15 14:22:56 +09:00
if ( newCellId !== op . cell_id ) {
errors . push ({
type : "update" ,
cellId : op.cell_id ,
message : `ID mismatch: cell_id is " ${ op . cell_id } " but new_xml has id=" ${ newCellId } "` ,
})
continue
}
2026-10-03 17:45:25 +09:00
// Import and replace the node (with its wrapper, if any)
const importedNode = doc . importNode ( getCellNode ( newCell ), true )
2025-12-15 14:22:56 +09:00
existingCell . parentNode ? . replaceChild ( importedNode , existingCell )
// Update the map with the new element
cellMap . set ( op . cell_id , importedNode )
2025-12-25 13:19:04 +09:00
} else if ( op . operation === "add" ) {
2025-12-15 14:22:56 +09:00
// Check if ID already exists
if ( cellMap . has ( op . cell_id )) {
errors . push ({
type : "add" ,
cellId : op.cell_id ,
message : `Cell with id=" ${ op . cell_id } " already exists` ,
})
continue
}
if ( ! op . new_xml ) {
errors . push ({
type : "add" ,
cellId : op.cell_id ,
message : "new_xml is required for add operation" ,
})
continue
}
// Parse the new XML
const newDoc = parser . parseFromString (
`<wrapper> ${ op . new_xml } </wrapper>` ,
"text/xml" ,
)
const newCell = newDoc . querySelector ( "mxCell" )
if ( ! newCell ) {
errors . push ({
type : "add" ,
cellId : op.cell_id ,
message : "new_xml must contain an mxCell element" ,
})
continue
}
// Validate ID matches
2026-10-03 17:45:25 +09:00
const newCellId = getCellId ( newCell )
2025-12-15 14:22:56 +09:00
if ( newCellId !== op . cell_id ) {
errors . push ({
type : "add" ,
cellId : op.cell_id ,
message : `ID mismatch: cell_id is " ${ op . cell_id } " but new_xml has id=" ${ newCellId } "` ,
})
continue
}
2026-10-03 17:45:25 +09:00
// Import and append the node (with its wrapper, if any)
const importedNode = doc . importNode ( getCellNode ( newCell ), true )
2025-12-15 14:22:56 +09:00
root . appendChild ( importedNode )
// Add to map
cellMap . set ( op . cell_id , importedNode )
2025-12-25 13:19:04 +09:00
} else if ( op . operation === "delete" ) {
2025-12-30 00:03:30 +09:00
// Protect root cells from deletion
if ( op . cell_id === "0" || op . cell_id === "1" ) {
2025-12-15 14:22:56 +09:00
errors . push ({
type : "delete" ,
cellId : op.cell_id ,
2025-12-30 00:03:30 +09:00
message : `Cannot delete root cell " ${ op . cell_id } "` ,
2025-12-15 14:22:56 +09:00
})
continue
}
2025-12-30 00:03:30 +09:00
const existingCell = cellMap . get ( op . cell_id )
if ( ! existingCell ) {
2026-10-03 17:45:25 +09:00
// Cells cascade-deleted earlier in this batch are skipped silently
// (AI may redundantly list children/edges)
if ( ! deletedIds . has ( op . cell_id )) {
errors . push ({
type : "delete" ,
cellId : op.cell_id ,
message : `Cell with id=" ${ op . cell_id } " not found` ,
})
}
2025-12-30 00:03:30 +09:00
continue
}
// Cascade delete: collect all cells to delete (children + edges + self)
const cellsToDelete = new Set < string >()
// Recursive function to find all descendants
const collectDescendants = ( cellId : string ) => {
if ( cellsToDelete . has ( cellId )) return
cellsToDelete . add ( cellId )
// Find children (cells where parent === cellId)
const children = root . querySelectorAll (
`mxCell[parent=" ${ cellId } "]` ,
)
children . forEach (( child ) => {
2026-10-03 17:45:25 +09:00
const childId = getCellId ( child )
2025-12-30 00:03:30 +09:00
if ( childId && childId !== "0" && childId !== "1" ) {
collectDescendants ( childId )
}
})
}
// Collect the target cell and all its descendants
collectDescendants ( op . cell_id )
// Find edges referencing any of the cells to be deleted
// Also recursively collect children of those edges (e.g., edge labels)
for ( const cellId of cellsToDelete ) {
const referencingEdges = root . querySelectorAll (
`mxCell[source=" ${ cellId } "], mxCell[target=" ${ cellId } "]` ,
)
referencingEdges . forEach (( edge ) => {
2026-10-03 17:45:25 +09:00
const edgeId = getCellId ( edge )
2025-12-30 00:03:30 +09:00
// Protect root cells from being added via edge references
if ( edgeId && edgeId !== "0" && edgeId !== "1" ) {
// Recurse to collect edge's children (like labels)
collectDescendants ( edgeId )
}
})
}
// Log what will be deleted
if ( cellsToDelete . size > 1 ) {
console . log (
`[applyDiagramOperations] Cascade delete " ${ op . cell_id } " → deleting ${ cellsToDelete . size } cells: ${ Array . from ( cellsToDelete ). join ( ", " ) } ` ,
2025-12-15 14:22:56 +09:00
)
}
2025-12-30 00:03:30 +09:00
// Delete all collected cells
for ( const cellId of cellsToDelete ) {
const cell = cellMap . get ( cellId )
if ( cell ) {
cell . parentNode ? . removeChild ( cell )
cellMap . delete ( cellId )
2026-10-03 17:45:25 +09:00
deletedIds . add ( cellId )
2025-12-30 00:03:30 +09:00
}
}
2025-12-15 14:22:56 +09:00
}
}
// Serialize back to string
const serializer = new XMLSerializer ()
const result = serializer . serializeToString ( doc )
return { result , errors }
2025-08-31 20:52:04 +09:00
}
2025-03-27 06:45:38 +00:00
2025-12-13 23:31:01 +09:00
// ============================================================================
// Validation Helper Functions
// ============================================================================
2025-12-03 16:14:53 +09:00
2025-12-13 23:31:01 +09:00
/** Check for duplicate structural attributes in a tag */
function checkDuplicateAttributes ( xml : string ) : string | null {
const structuralSet = new Set ( STRUCTURAL_ATTRS )
2025-12-13 15:00:28 +09:00
const tagPattern = /<[^>]+>/g
let tagMatch
while (( tagMatch = tagPattern . exec ( xml )) !== null ) {
const tag = tagMatch [ 0 ]
const attrPattern = /\s([a-zA-Z_:][a-zA-Z0-9_:.-]*)\s*=/g
const attributes = new Map < string , number >()
let attrMatch
while (( attrMatch = attrPattern . exec ( tag )) !== null ) {
const attrName = attrMatch [ 1 ]
attributes . set ( attrName , ( attributes . get ( attrName ) || 0 ) + 1 )
}
const duplicates = Array . from ( attributes . entries ())
2025-12-13 23:31:01 +09:00
. filter (([ name , count ]) => count > 1 && structuralSet . has ( name ))
2025-12-13 15:00:28 +09:00
. map (([ name ]) => name )
if ( duplicates . length > 0 ) {
return `Invalid XML: Duplicate structural attribute(s): ${ duplicates . join ( ", " ) } . Remove duplicate attributes.`
}
}
2025-12-13 23:31:01 +09:00
return null
}
2025-12-13 15:00:28 +09:00
2026-10-03 17:45:25 +09:00
/** Matches one <diagram> page of a document (the last one may be unclosed) */
const PAGE_PATTERN = /<diagram\b[\s\S]*?(?:<\/diagram>|$)/g
const ID_ATTR_PATTERN = /\bid\s*=\s*["']([^"']+)["']/gi
/**
* Split XML into pages. Ids only need to be unique within a page: every
* page of a multi-page document has its own root cells "0" and "1".
*/
function splitPages ( xml : string ) : string [] {
return xml . match ( PAGE_PATTERN ) || [ xml ]
}
/** Ids that appear more than once, with their counts */
function findDuplicateIds ( xml : string ) : Map < string , number > {
2025-12-13 15:00:28 +09:00
const ids = new Map < string , number >()
2026-10-03 17:45:25 +09:00
for ( const match of xml . matchAll ( ID_ATTR_PATTERN )) {
ids . set ( match [ 1 ], ( ids . get ( match [ 1 ]) || 0 ) + 1 )
2025-12-13 15:00:28 +09:00
}
2026-10-03 17:45:25 +09:00
return new Map ( Array . from ( ids ). filter (([, count ]) => count > 1 ))
}
/** Check for duplicate IDs in XML (per page) */
function checkDuplicateIds ( xml : string ) : string | null {
for ( const page of splitPages ( xml )) {
const duplicateIds = Array . from ( findDuplicateIds ( page )). map (
([ id , count ]) => `' ${ id } ' ( ${ count } x)` ,
)
if ( duplicateIds . length > 0 ) {
return `Invalid XML: Found duplicate ID(s): ${ duplicateIds . slice ( 0 , 3 ). join ( ", " ) } . All id attributes must be unique.`
}
2025-12-03 16:14:53 +09:00
}
2025-12-13 23:31:01 +09:00
return null
}
2025-12-03 16:14:53 +09:00
2026-10-03 17:45:25 +09:00
/** Rename repeated ids in one page (keeps the first occurrence) */
function renameDuplicateIds ( xml : string ) : { xml : string ; renamed : number } {
const duplicateIds = findDuplicateIds ( xml )
if ( duplicateIds . size === 0 ) return { xml , renamed : 0 }
const idCounters = new Map < string , number >()
const renamedXml = xml . replace ( ID_ATTR_PATTERN , ( match , id ) => {
if ( ! duplicateIds . has ( id )) return match
const count = idCounters . get ( id ) || 0
idCounters . set ( id , count + 1 )
if ( count === 0 ) return match // Keep first occurrence
// Rename subsequent occurrences (the id sits just before the closing quote)
return ` ${ match . slice ( 0 , - id . length - 1 ) }${ id } _dup ${ count }${ match . slice ( - 1 ) } `
})
return { xml : renamedXml , renamed : duplicateIds.size }
}
/**
* Returns a function telling whether a position is inside a quoted attribute
* value. Positions must be queried in increasing order: the scan resumes where
* it stopped instead of starting over, which keeps large documents fast.
*/
function createQuoteTracker ( str : string ) : ( pos : number ) => boolean {
let i = 0
let inQuote = false
let quoteChar = ""
return ( pos : number ) => {
for (; i < pos && i < str . length ; i ++ ) {
const c = str [ i ]
if ( inQuote ) {
if ( c === quoteChar ) inQuote = false
} else if ( c === '"' || c === "'" ) {
// Only quotes that follow "=" open an attribute value
let j = i - 1
while ( j >= 0 && /\s/ . test ( str [ j ])) j --
if ( j >= 0 && str [ j ] === "=" ) {
inQuote = true
quoteChar = c
}
}
}
return inQuote
}
}
2025-12-13 23:31:01 +09:00
/** Check for tag mismatches using parsed tags */
function checkTagMismatches ( xml : string ) : string | null {
2025-12-13 15:00:28 +09:00
const xmlWithoutComments = xml . replace ( /<!--[\s\S]*?-->/g , "" )
2025-12-13 23:31:01 +09:00
const tags = parseXmlTags ( xmlWithoutComments )
2025-12-13 15:00:28 +09:00
const tagStack : string [] = []
2025-12-03 16:14:53 +09:00
2025-12-13 23:31:01 +09:00
for ( const { tagName , isClosing , isSelfClosing } of tags ) {
2025-12-13 15:00:28 +09:00
if ( isClosing ) {
if ( tagStack . length === 0 ) {
return `Invalid XML: Closing tag </ ${ tagName } > without matching opening tag`
}
const expected = tagStack . pop ()
if ( expected ? . toLowerCase () !== tagName . toLowerCase ()) {
return `Invalid XML: Expected closing tag </ ${ expected } > but found </ ${ tagName } >`
}
} else if ( ! isSelfClosing ) {
tagStack . push ( tagName )
}
}
if ( tagStack . length > 0 ) {
return `Invalid XML: Document has ${ tagStack . length } unclosed tag(s): ${ tagStack . join ( ", " ) } `
}
2025-12-13 23:31:01 +09:00
return null
}
2025-12-13 15:00:28 +09:00
2025-12-13 23:31:01 +09:00
/** Check for invalid character references */
function checkCharacterReferences ( xml : string ) : string | null {
2025-12-13 15:00:28 +09:00
const charRefPattern = /&#x?[^;]+;?/g
let charMatch
while (( charMatch = charRefPattern . exec ( xml )) !== null ) {
const ref = charMatch [ 0 ]
if ( ref . startsWith ( "&#x" )) {
if ( ! ref . endsWith ( ";" )) {
return `Invalid XML: Missing semicolon after hex reference: ${ ref } `
}
const hexDigits = ref . substring ( 3 , ref . length - 1 )
if ( hexDigits . length === 0 || ! /^[0-9a-fA-F]+$/ . test ( hexDigits )) {
return `Invalid XML: Invalid hex character reference: ${ ref } `
}
} else if ( ref . startsWith ( "&#" )) {
if ( ! ref . endsWith ( ";" )) {
return `Invalid XML: Missing semicolon after decimal reference: ${ ref } `
}
const decDigits = ref . substring ( 2 , ref . length - 1 )
if ( decDigits . length === 0 || ! /^[0-9]+$/ . test ( decDigits )) {
return `Invalid XML: Invalid decimal character reference: ${ ref } `
2025-12-07 00:40:19 +09:00
}
}
2025-12-13 15:00:28 +09:00
}
2025-12-13 23:31:01 +09:00
return null
}
2025-12-07 00:40:19 +09:00
2025-12-13 23:31:01 +09:00
/** Check for invalid entity references */
function checkEntityReferences ( xml : string ) : string | null {
const xmlWithoutComments = xml . replace ( /<!--[\s\S]*?-->/g , "" )
2025-12-13 15:00:28 +09:00
const bareAmpPattern = /&(?!(?:lt|gt|amp|quot|apos|#))/g
if ( bareAmpPattern . test ( xmlWithoutComments )) {
return "Invalid XML: Found unescaped & character(s). Replace & with &"
}
const invalidEntityPattern = /&([a-zA-Z][a-zA-Z0-9]*);/g
let entityMatch
while (
( entityMatch = invalidEntityPattern . exec ( xmlWithoutComments )) !== null
) {
2025-12-13 23:31:01 +09:00
if ( ! VALID_ENTITIES . has ( entityMatch [ 1 ])) {
2025-12-13 15:00:28 +09:00
return `Invalid XML: Invalid entity reference: & ${ entityMatch [ 1 ] } ; - use only valid XML entities (lt, gt, amp, quot, apos)`
}
}
2025-12-13 23:31:01 +09:00
return null
}
2025-12-13 15:00:28 +09:00
2025-12-13 23:31:01 +09:00
/** Check for nested mxCell tags using regex */
function checkNestedMxCells ( xml : string ) : string | null {
2025-12-13 15:00:28 +09:00
const cellTagPattern = /<\/?mxCell[^>]*>/g
const cellStack : number [] = []
let cellMatch
while (( cellMatch = cellTagPattern . exec ( xml )) !== null ) {
const tag = cellMatch [ 0 ]
if ( tag . startsWith ( "</mxCell>" )) {
if ( cellStack . length > 0 ) cellStack . pop ()
} else if ( ! tag . endsWith ( "/>" )) {
const isLabelOrGeometry =
/\sas\s*=\s*["'](valueLabel|geometry)["']/ . test ( tag )
if ( ! isLabelOrGeometry ) {
cellStack . push ( cellMatch . index )
if ( cellStack . length > 1 ) {
return "Invalid XML: Found nested mxCell tags. Cells should be siblings, not nested inside other mxCell elements."
}
}
}
2025-12-07 00:40:19 +09:00
}
2025-12-13 23:31:01 +09:00
return null
}
/**
* Validates draw.io XML structure for common issues
* Uses DOM parsing + additional regex checks for high accuracy
* @param xml - The XML string to validate
* @returns null if valid, error message string if invalid
*/
export function validateMxCellStructure ( xml : string ) : string | null {
// Size check for performance
if ( xml . length > MAX_XML_SIZE ) {
console . warn (
`[validateMxCellStructure] XML size ( ${ xml . length } ) exceeds ${ MAX_XML_SIZE } bytes, may cause performance issues` ,
)
}
// 0. First use DOM parser to catch syntax errors (most accurate)
try {
const parser = new DOMParser ()
const doc = parser . parseFromString ( xml , "text/xml" )
const parseError = doc . querySelector ( "parsererror" )
if ( parseError ) {
return `Invalid XML: The XML contains syntax errors (likely unescaped special characters like <, >, & in attribute values). Please escape special characters: use < for <, > for >, & for &, " for ". Regenerate the diagram with properly escaped values.`
}
// DOM-based checks for nested mxCell
const allCells = doc . querySelectorAll ( "mxCell" )
for ( const cell of allCells ) {
if ( cell . parentElement ? . tagName === "mxCell" ) {
const id = cell . getAttribute ( "id" ) || "unknown"
return `Invalid XML: Found nested mxCell (id=" ${ id } "). Cells should be siblings, not nested inside other mxCell elements.`
}
}
} catch ( error ) {
// Log unexpected DOMParser errors before falling back to regex checks
console . warn (
"[validateMxCellStructure] DOMParser threw unexpected error, falling back to regex validation:" ,
error ,
)
}
// 1. Check for CDATA wrapper (invalid at document root)
if ( /^\s*<!\[CDATA\[/ . test ( xml )) {
return "Invalid XML: XML is wrapped in CDATA section - remove <![CDATA[ from start and ]]> from end"
}
// 2. Check for duplicate structural attributes
const dupAttrError = checkDuplicateAttributes ( xml )
2025-12-14 21:23:14 +09:00
if ( dupAttrError ) {
return dupAttrError
}
2025-12-13 23:31:01 +09:00
// 3. Check for unescaped < in attribute values
const attrValuePattern = /=\s*"([^"]*)"/g
let attrValMatch
while (( attrValMatch = attrValuePattern . exec ( xml )) !== null ) {
const value = attrValMatch [ 1 ]
if ( /</ . test ( value ) && ! /</ . test ( value )) {
return "Invalid XML: Unescaped < character in attribute values. Replace < with <"
}
}
// 4. Check for duplicate IDs
const dupIdError = checkDuplicateIds ( xml )
2025-12-14 21:23:14 +09:00
if ( dupIdError ) {
return dupIdError
}
2025-12-13 23:31:01 +09:00
// 5. Check for tag mismatches
const tagMismatchError = checkTagMismatches ( xml )
2025-12-14 21:23:14 +09:00
if ( tagMismatchError ) {
return tagMismatchError
}
2025-12-13 23:31:01 +09:00
// 6. Check invalid character references
const charRefError = checkCharacterReferences ( xml )
2025-12-14 21:23:14 +09:00
if ( charRefError ) {
return charRefError
}
2025-12-13 23:31:01 +09:00
// 7. Check for invalid comment syntax (-- inside comments)
const commentPattern = /<!--([\s\S]*?)-->/g
let commentMatch
while (( commentMatch = commentPattern . exec ( xml )) !== null ) {
if ( /--/ . test ( commentMatch [ 1 ])) {
return "Invalid XML: Comment contains -- (double hyphen) which is not allowed"
}
}
// 8. Check for unescaped entity references and invalid entity names
const entityError = checkEntityReferences ( xml )
2025-12-14 21:23:14 +09:00
if ( entityError ) {
return entityError
}
2025-12-13 23:31:01 +09:00
// 9. Check for empty id attributes on mxCell
if ( /<mxCell[^>]*\sid\s*=\s*["']\s*["'][^>]*>/g . test ( xml )) {
return "Invalid XML: Found mxCell element(s) with empty id attribute"
}
// 10. Check for nested mxCell tags
const nestedCellError = checkNestedMxCells ( xml )
2025-12-14 21:23:14 +09:00
if ( nestedCellError ) {
return nestedCellError
}
2025-12-07 00:40:19 +09:00
2025-12-06 12:46:40 +09:00
return null
2025-12-03 16:14:53 +09:00
}
2025-12-13 15:00:28 +09:00
/**
* Attempts to auto-fix common XML issues in draw.io diagrams
* @param xml - The XML string to fix
* @returns Object with fixed XML and list of fixes applied
*/
export function autoFixXml ( xml : string ) : { fixed : string ; fixes : string [] } {
let fixed = xml
const fixes : string [] = []
2025-12-13 23:31:01 +09:00
// 0. Fix JSON-escaped XML (common when XML is stored in JSON without unescaping)
// Only apply when we see JSON-escaped attribute patterns like =\"value\"
// Don't apply to legitimate \n in value attributes (draw.io uses these for line breaks)
if ( /=\\"/ . test ( fixed )) {
// Replace literal \" with actual quotes
fixed = fixed . replace ( /\\"/g , '"' )
// Replace literal \n with actual newlines (only after confirming JSON-escaped)
fixed = fixed . replace ( /\\n/g , "\n" )
fixes . push ( "Fixed JSON-escaped XML" )
2025-12-13 16:02:56 +09:00
}
// 1. Remove CDATA wrapper (MUST be before text-before-root check)
2025-12-13 15:00:28 +09:00
if ( /^\s*<!\[CDATA\[/ . test ( fixed )) {
fixed = fixed . replace ( /^\s*<!\[CDATA\[/ , "" ). replace ( /\]\]>\s*$/ , "" )
fixes . push ( "Removed CDATA wrapper" )
}
2025-12-24 09:31:54 +09:00
// 1b. Strip trailing LLM wrapper tags (DeepSeek, Anthropic, etc.)
// These are closing tags after the last valid mxCell that break XML parsing
const lastSelfClose = fixed . lastIndexOf ( "/>" )
const lastMxCellClose = fixed . lastIndexOf ( "</mxCell>" )
const lastValidEnd = Math . max ( lastSelfClose , lastMxCellClose )
if ( lastValidEnd !== - 1 ) {
const endOffset = lastMxCellClose > lastSelfClose ? 9 : 2
const suffix = fixed . slice ( lastValidEnd + endOffset )
// If suffix contains only closing tags (wrapper tags) or whitespace, strip it
if ( /^(\s*<\/[^>]+>)+\s*$/ . test ( suffix )) {
fixed = fixed . slice ( 0 , lastValidEnd + endOffset )
fixes . push ( "Stripped trailing LLM wrapper tags" )
}
}
2025-12-13 16:02:56 +09:00
// 2. Remove text before XML declaration or root element (only if it's garbage text, not valid XML)
const xmlStart = fixed . search ( /<(\?xml|mxGraphModel|mxfile)/i )
if ( xmlStart > 0 && ! /^<[a-zA-Z]/ . test ( fixed . trim ())) {
fixed = fixed . substring ( xmlStart )
fixes . push ( "Removed text before XML root" )
}
2025-12-13 15:00:28 +09:00
// 2. Fix duplicate attributes (keep first occurrence, remove duplicates)
let dupAttrFixed = false
fixed = fixed . replace ( /<[^>]+>/g , ( tag ) => {
let newTag = tag
2025-12-13 23:31:01 +09:00
for ( const attr of STRUCTURAL_ATTRS ) {
2025-12-13 15:00:28 +09:00
// Find all occurrences of this attribute
const attrRegex = new RegExp (
`\\s ${ attr } \\s*=\\s*["'][^"']*["']` ,
"gi" ,
)
const matches = tag . match ( attrRegex )
if ( matches && matches . length > 1 ) {
// Keep first, remove others
let firstKept = false
newTag = newTag . replace ( attrRegex , ( m ) => {
if ( ! firstKept ) {
firstKept = true
return m
}
dupAttrFixed = true
return ""
})
}
}
return newTag
})
if ( dupAttrFixed ) {
fixes . push ( "Removed duplicate structural attributes" )
}
// 3. Fix unescaped & characters (but not valid entities)
// Match & not followed by valid entity pattern
const ampersandPattern =
/&(?!(?:lt|gt|amp|quot|apos|#[0-9]+|#x[0-9a-fA-F]+);)/g
if ( ampersandPattern . test ( fixed )) {
fixed = fixed . replace (
/&(?!(?:lt|gt|amp|quot|apos|#[0-9]+|#x[0-9a-fA-F]+);)/g ,
"&" ,
)
fixes . push ( "Escaped unescaped & characters" )
}
// 3. Fix invalid entity names like &quot; -> "
// Common mistake: double-escaping
const invalidEntities = [
{ pattern : /&quot;/g , replacement : """ , name : "&quot;" },
{ pattern : /&lt;/g , replacement : "<" , name : "&lt;" },
{ pattern : /&gt;/g , replacement : ">" , name : "&gt;" },
{ pattern : /&apos;/g , replacement : "'" , name : "&apos;" },
{ pattern : /&amp;/g , replacement : "&" , name : "&amp;" },
]
for ( const { pattern , replacement , name } of invalidEntities ) {
if ( pattern . test ( fixed )) {
fixed = fixed . replace ( pattern , replacement )
fixes . push ( `Fixed double-escaped entity ${ name } ` )
}
}
2025-12-13 23:31:01 +09:00
// 3b. Fix malformed attribute values where " is used as delimiter instead of actual quotes
// Pattern: attr="value" should become attr="value" (the " was meant to be the quote delimiter)
// This commonly happens with dashPattern="1 1;"
2026-10-03 17:45:25 +09:00
// Matches inside another attribute value are kept: rich text labels like
// value="<font color="#ff0000">..." are valid.
const isInsideQuotesFor3b = createQuoteTracker ( fixed )
let malformedQuotesFixed = false
fixed = fixed . replace (
/(\s[a-zA-Z][a-zA-Z0-9_:-]*)="([^&]*?)"/g ,
( match : string , attr : string , value : string , offset : number ) => {
if ( isInsideQuotesFor3b ( offset )) return match
malformedQuotesFixed = true
return ` ${ attr } =" ${ value } "`
},
)
if ( malformedQuotesFixed ) {
2025-12-13 15:00:28 +09:00
fixes . push (
2025-12-13 23:31:01 +09:00
'Fixed malformed attribute quotes (="..." to ="...")' ,
2025-12-13 15:00:28 +09:00
)
}
2025-12-13 23:31:01 +09:00
// 3c. Fix malformed closing tags like </tag/> -> </tag>
const malformedClosingTag = /<\/([a-zA-Z][a-zA-Z0-9]*)\s*\/>/g
if ( malformedClosingTag . test ( fixed )) {
fixed = fixed . replace ( /<\/([a-zA-Z][a-zA-Z0-9]*)\s*\/>/g , "</$1>" )
fixes . push ( "Fixed malformed closing tags (</tag/> to </tag>)" )
}
// 3d. Fix missing space between attributes like vertex="1"parent="1"
2026-10-03 17:45:25 +09:00
// Requires name=" right after the quote, so the opening quote of a value
// such as style="rounded=1;..." is not mistaken for a closing one.
const missingSpacePattern = /"([a-zA-Z_:][\w:.-]*=")/g
2025-12-13 23:31:01 +09:00
if ( missingSpacePattern . test ( fixed )) {
2026-10-03 17:45:25 +09:00
fixed = fixed . replace ( missingSpacePattern , '" $1' )
2025-12-13 23:31:01 +09:00
fixes . push ( "Added missing space between attributes" )
}
// 3e. Fix unescaped quotes in style color values like fillColor="#fff2e6"
// The " after Color= prematurely ends the style attribute. Remove it.
// Pattern: ;fillColor="#fff → ;fillColor=#fff (remove first ", keep second as style closer)
const quotedColorPattern = /;([a-zA-Z]*[Cc]olor)="#/
if ( quotedColorPattern . test ( fixed )) {
fixed = fixed . replace ( /;([a-zA-Z]*[Cc]olor)="#/g , ";$1=#" )
fixes . push ( "Removed quotes around color values in style" )
}
2025-12-24 09:31:54 +09:00
// 4. Fix unescaped < and > in attribute values
// < is required to be escaped, > is not strictly required but we escape for consistency
2025-12-13 15:00:28 +09:00
const attrPattern = /(=\s*")([^"]*?)(<)([^"]*?)(")/g
let attrMatch
let hasUnescapedLt = false
while (( attrMatch = attrPattern . exec ( fixed )) !== null ) {
if ( ! attrMatch [ 3 ]. startsWith ( "<" )) {
hasUnescapedLt = true
break
}
}
if ( hasUnescapedLt ) {
2025-12-24 09:31:54 +09:00
// Replace < and > with < and > inside attribute values
2025-12-13 23:31:01 +09:00
fixed = fixed . replace ( /=\s*"([^"]*)"/g , ( _match , value ) => {
2025-12-24 09:31:54 +09:00
const escaped = value . replace ( /</g , "<" ). replace ( />/g , ">" )
2025-12-13 15:00:28 +09:00
return `=" ${ escaped } "`
})
2025-12-24 09:31:54 +09:00
fixes . push ( "Escaped <> characters in attribute values" )
2025-12-13 15:00:28 +09:00
}
// 5. Fix invalid character references (remove malformed ones)
// Pattern: &#x followed by non-hex chars before ;
const invalidHexRefs : string [] = []
fixed = fixed . replace ( /&#x([^;]*);/g , ( match , hex ) => {
if ( /^[0-9a-fA-F]+$/ . test ( hex ) && hex . length > 0 ) {
return match // Valid hex ref, keep it
}
invalidHexRefs . push ( match )
return "" // Remove invalid ref
})
if ( invalidHexRefs . length > 0 ) {
fixes . push (
`Removed ${ invalidHexRefs . length } invalid hex character reference(s)` ,
)
}
// 6. Fix invalid decimal character references
const invalidDecRefs : string [] = []
fixed = fixed . replace ( /&#([^x][^;]*);/g , ( match , dec ) => {
if ( /^[0-9]+$/ . test ( dec ) && dec . length > 0 ) {
return match // Valid decimal ref, keep it
}
invalidDecRefs . push ( match )
return "" // Remove invalid ref
})
if ( invalidDecRefs . length > 0 ) {
fixes . push (
`Removed ${ invalidDecRefs . length } invalid decimal character reference(s)` ,
)
}
// 7. Fix invalid comment syntax (replace -- with - repeatedly until none left)
fixed = fixed . replace ( /<!--([\s\S]*?)-->/g , ( match , content ) => {
if ( /--/ . test ( content )) {
// Keep replacing until no double hyphens remain
let fixedContent = content
while ( /--/ . test ( fixedContent )) {
fixedContent = fixedContent . replace ( /--/g , "-" )
}
fixes . push ( "Fixed invalid comment syntax (removed double hyphens)" )
return `<!-- ${ fixedContent } -->`
}
return match
})
// 8. Fix <Cell> tags that should be <mxCell> (common LLM mistake)
// This handles both opening and closing tags
const hasCellTags = /<\/?Cell[\s>]/i . test ( fixed )
if ( hasCellTags ) {
2025-12-14 19:38:40 +09:00
console . log ( "[autoFixXml] Step 8: Found <Cell> tags to fix" )
const beforeFix = fixed
2025-12-13 15:00:28 +09:00
fixed = fixed . replace ( /<Cell(\s)/gi , "<mxCell$1" )
fixed = fixed . replace ( /<Cell>/gi , "<mxCell>" )
fixed = fixed . replace ( /<\/Cell>/gi , "</mxCell>" )
2025-12-14 19:38:40 +09:00
if ( beforeFix !== fixed ) {
console . log ( "[autoFixXml] Step 8: Fixed <Cell> tags" )
}
2025-12-13 15:00:28 +09:00
fixes . push ( "Fixed <Cell> tags to <mxCell>" )
}
2025-12-21 00:32:51 +09:00
// 8b. Fix common closing tag typos (MUST run before foreign tag removal)
const tagTypos = [
{ wrong : /<\/mxElement>/gi , right : "</mxCell>" , name : "</mxElement>" },
{ wrong : /<\/mxcell>/g , right : "</mxCell>" , name : "</mxcell>" }, // case sensitivity
{
wrong : /<\/mxgeometry>/g ,
right : "</mxGeometry>" ,
name : "</mxgeometry>" ,
},
{ wrong : /<\/mxpoint>/g , right : "</mxPoint>" , name : "</mxpoint>" },
{
wrong : /<\/mxgraphmodel>/gi ,
right : "</mxGraphModel>" ,
name : "</mxgraphmodel>" ,
},
]
for ( const { wrong , right , name } of tagTypos ) {
const before = fixed
fixed = fixed . replace ( wrong , right )
if ( fixed !== before ) {
fixes . push ( `Fixed typo ${ name } to ${ right } ` )
}
}
// 8c. Remove non-draw.io tags (after typo fixes so lowercase variants are fixed first)
2025-12-24 09:31:54 +09:00
// IMPORTANT: Only remove tags at the element level, NOT inside quoted attribute values
// Tags like <b>, <br> inside value="<b>text</b>" should be preserved (they're HTML content)
2025-12-14 19:38:40 +09:00
const validDrawioTags = new Set ([
"mxfile" ,
"diagram" ,
"mxGraphModel" ,
"root" ,
"mxCell" ,
"mxGeometry" ,
"mxPoint" ,
"Array" ,
"Object" ,
2026-10-03 17:45:25 +09:00
// Wrappers of cells with links, tooltips or custom data
"object" ,
"UserObject" ,
2025-12-14 19:38:40 +09:00
"mxRectangle" ,
])
2025-12-24 09:31:54 +09:00
2026-10-03 17:45:25 +09:00
const isInsideQuotesFor8c = createQuoteTracker ( fixed )
2025-12-14 19:38:40 +09:00
const foreignTagPattern = /<\/?([a-zA-Z][a-zA-Z0-9_]*)[^>]*>/g
let foreignMatch
const foreignTags = new Set < string >()
2025-12-24 09:31:54 +09:00
const foreignTagPositions : Array < {
tag : string
start : number
end : number
} > = []
2025-12-14 19:38:40 +09:00
while (( foreignMatch = foreignTagPattern . exec ( fixed )) !== null ) {
const tagName = foreignMatch [ 1 ]
2025-12-24 09:31:54 +09:00
// Skip if this is a valid draw.io tag
if ( validDrawioTags . has ( tagName )) continue
// Skip if this tag is inside a quoted attribute value
2026-10-03 17:45:25 +09:00
if ( isInsideQuotesFor8c ( foreignMatch . index )) continue
2025-12-24 09:31:54 +09:00
foreignTags . add ( tagName )
foreignTagPositions . push ({
tag : tagName ,
start : foreignMatch.index ,
end : foreignMatch.index + foreignMatch [ 0 ]. length ,
})
2025-12-14 19:38:40 +09:00
}
2025-12-24 09:31:54 +09:00
if ( foreignTagPositions . length > 0 ) {
// Remove tags from end to start to preserve indices
foreignTagPositions . sort (( a , b ) => b . start - a . start )
for ( const { start , end } of foreignTagPositions ) {
fixed = fixed . slice ( 0 , start ) + fixed . slice ( end )
2025-12-14 19:38:40 +09:00
}
fixes . push (
`Removed foreign tags: ${ Array . from ( foreignTags ). join ( ", " ) } ` ,
)
}
2025-12-13 15:00:28 +09:00
// 10. Fix unclosed tags by appending missing closing tags
2025-12-13 23:31:01 +09:00
// Use parseXmlTags helper to track open tags
2025-12-13 15:00:28 +09:00
const tagStack : string [] = []
2025-12-13 23:31:01 +09:00
const parsedTags = parseXmlTags ( fixed )
2025-12-13 15:00:28 +09:00
2025-12-13 23:31:01 +09:00
for ( const { tagName , isClosing , isSelfClosing } of parsedTags ) {
2025-12-13 15:00:28 +09:00
if ( isClosing ) {
// Find matching opening tag (may not be the last one if there's mismatch)
const lastIdx = tagStack . lastIndexOf ( tagName )
if ( lastIdx !== - 1 ) {
tagStack . splice ( lastIdx , 1 )
}
} else if ( ! isSelfClosing ) {
tagStack . push ( tagName )
}
}
// If there are unclosed tags, append closing tags in reverse order
// But first verify with simple count that they're actually unclosed
if ( tagStack . length > 0 ) {
const tagsToClose : string [] = []
for ( const tagName of tagStack . reverse ()) {
// Simple count check: only close if opens > closes
const openCount = (
fixed . match ( new RegExp ( `< ${ tagName } [\\s>]` , "gi" )) || []
). length
const closeCount = (
fixed . match ( new RegExp ( `</ ${ tagName } >` , "gi" )) || []
). length
if ( openCount > closeCount ) {
tagsToClose . push ( tagName )
}
}
if ( tagsToClose . length > 0 ) {
const closingTags = tagsToClose . map (( t ) => `</ ${ t } >` ). join ( "\n" )
fixed = fixed . trimEnd () + "\n" + closingTags
fixes . push (
`Closed ${ tagsToClose . length } unclosed tag(s): ${ tagsToClose . join ( ", " ) } ` ,
)
}
}
2025-12-14 19:38:40 +09:00
// 10b. Remove extra closing tags (more closes than opens)
// Need to properly count self-closing tags (they don't need closing tags)
2025-12-24 09:31:54 +09:00
// IMPORTANT: Only count tags at element level, NOT inside quoted attribute values
2025-12-14 19:38:40 +09:00
const tagCounts = new Map <
string ,
{ opens : number ; closes : number ; selfClosing : number }
> ()
// Match full tags to detect self-closing by checking if ends with />
const fullTagPattern = /<(\/?[a-zA-Z][a-zA-Z0-9]*)[^>]*>/g
2026-10-03 17:45:25 +09:00
const isInsideQuotesFor10b = createQuoteTracker ( fixed )
2025-12-14 19:38:40 +09:00
let tagCountMatch
while (( tagCountMatch = fullTagPattern . exec ( fixed )) !== null ) {
2025-12-24 09:31:54 +09:00
// Skip tags inside quoted attribute values (e.g., value="<b>Title</b>")
2026-10-03 17:45:25 +09:00
if ( isInsideQuotesFor10b ( tagCountMatch . index )) continue
2025-12-24 09:31:54 +09:00
2025-12-14 19:38:40 +09:00
const fullMatch = tagCountMatch [ 0 ] // e.g., "<mxCell .../>" or "</mxCell>"
const tagPart = tagCountMatch [ 1 ] // e.g., "mxCell" or "/mxCell"
const isClosing = tagPart . startsWith ( "/" )
const isSelfClosing = fullMatch . endsWith ( "/>" )
const tagName = isClosing ? tagPart . slice ( 1 ) : tagPart
2025-12-24 09:31:54 +09:00
// Only count valid draw.io tags - skip partial/invalid tags like "mx" from streaming
if ( ! validDrawioTags . has ( tagName )) continue
2025-12-14 19:38:40 +09:00
let counts = tagCounts . get ( tagName )
if ( ! counts ) {
counts = { opens : 0 , closes : 0 , selfClosing : 0 }
tagCounts . set ( tagName , counts )
}
if ( isClosing ) {
counts . closes ++
} else if ( isSelfClosing ) {
counts . selfClosing ++
} else {
counts . opens ++
}
}
// Log tag counts for debugging
for ( const [ tagName , counts ] of tagCounts ) {
if (
tagName === "mxCell" ||
tagName === "mxGeometry" ||
counts . opens !== counts . closes
) {
console . log (
`[autoFixXml] Step 10b: ${ tagName } - opens: ${ counts . opens } , closes: ${ counts . closes } , selfClosing: ${ counts . selfClosing } ` ,
)
}
}
// Find tags with extra closing tags (self-closing tags are balanced, don't need closing)
for ( const [ tagName , counts ] of tagCounts ) {
const extraCloses = counts . closes - counts . opens // Only compare opens vs closes (self-closing are balanced)
if ( extraCloses > 0 ) {
console . log (
`[autoFixXml] Step 10b: ${ tagName } has ${ counts . opens } opens, ${ counts . closes } closes, removing ${ extraCloses } extra` ,
)
// Remove extra closing tags from the end
let removed = 0
const closeTagPattern = new RegExp ( `</ ${ tagName } >` , "g" )
const matches = [... fixed . matchAll ( closeTagPattern )]
// Remove from the end (last occurrences are likely the extras)
for (
let i = matches . length - 1 ;
i >= 0 && removed < extraCloses ;
i --
) {
const match = matches [ i ]
const idx = match . index ?? 0
fixed = fixed . slice ( 0 , idx ) + fixed . slice ( idx + match [ 0 ]. length )
removed ++
}
if ( removed > 0 ) {
console . log (
`[autoFixXml] Step 10b: Removed ${ removed } extra </ ${ tagName } >` ,
)
fixes . push (
`Removed ${ removed } extra </ ${ tagName } > closing tag(s)` ,
)
}
}
}
// 10c. Remove trailing garbage after last XML tag (e.g., stray backslashes, text)
// Find the last valid closing tag or self-closing tag
const closingTagPattern = /<\/[a-zA-Z][a-zA-Z0-9]*>|\/>/g
let lastValidTagEnd = - 1
let closingMatch
while (( closingMatch = closingTagPattern . exec ( fixed )) !== null ) {
lastValidTagEnd = closingMatch . index + closingMatch [ 0 ]. length
}
if ( lastValidTagEnd > 0 && lastValidTagEnd < fixed . length ) {
const trailing = fixed . slice ( lastValidTagEnd ). trim ()
if ( trailing ) {
fixed = fixed . slice ( 0 , lastValidTagEnd )
fixes . push ( "Removed trailing garbage after last XML tag" )
}
}
2025-12-13 15:00:28 +09:00
// 11. Fix nested mxCell by flattening
// Pattern A: <mxCell id="X">...<mxCell id="X">...</mxCell></mxCell> (duplicate ID)
// Pattern B: <mxCell id="X">...<mxCell id="Y">...</mxCell></mxCell> (different ID - true nesting)
2026-10-03 17:45:25 +09:00
// These passes work line by line and would break valid cells written on a
// single line, so each one runs only when cells are really nested.
if ( checkNestedMxCells ( fixed )) {
const lines = fixed . split ( "\n" )
const newLines : string [] = []
let nestedFixed = 0
let extraClosingToRemove = 0
2025-12-13 15:00:28 +09:00
2026-10-03 17:45:25 +09:00
// First pass: fix duplicate ID nesting (same as before)
for ( let i = 0 ; i < lines . length ; i ++ ) {
const line = lines [ i ]
const nextLine = lines [ i + 1 ]
2025-12-13 15:00:28 +09:00
2026-10-03 17:45:25 +09:00
// Check if current line and next line are both mxCell opening tags with same ID
if (
nextLine &&
/<mxCell\s/ . test ( line ) &&
/<mxCell\s/ . test ( nextLine ) &&
! line . includes ( "/>" ) &&
! nextLine . includes ( "/>" )
) {
const id1 = line . match ( /\bid\s*=\s*["']([^"']+)["']/ ) ? .[ 1 ]
const id2 = nextLine . match ( /\bid\s*=\s*["']([^"']+)["']/ ) ? .[ 1 ]
2025-12-13 15:00:28 +09:00
2026-10-03 17:45:25 +09:00
if ( id1 && id1 === id2 ) {
nestedFixed ++
extraClosingToRemove ++ // Need to remove one </mxCell> later
continue // Skip this duplicate opening line
}
2025-12-13 15:00:28 +09:00
}
2026-10-03 17:45:25 +09:00
// Remove extra </mxCell> if we have pending removals
if ( extraClosingToRemove > 0 && /^\s*<\/mxCell>\s*$/ . test ( line )) {
extraClosingToRemove --
continue // Skip this closing tag
2025-12-13 15:00:28 +09:00
}
2026-10-03 17:45:25 +09:00
2025-12-13 15:00:28 +09:00
newLines . push ( line )
2026-10-03 17:45:25 +09:00
}
if ( nestedFixed > 0 ) {
fixed = newLines . join ( "\n" )
fixes . push ( `Flattened ${ nestedFixed } duplicate-ID nested mxCell(s)` )
}
}
if ( checkNestedMxCells ( fixed )) {
// Second pass: fix true nesting (different IDs)
// Insert </mxCell> before nested child to close parent
const lines2 = fixed . split ( "\n" )
const newLines : string [] = []
let trueNestedFixed = 0
let cellDepth = 0
let pendingCloseRemoval = 0
for ( let i = 0 ; i < lines2 . length ; i ++ ) {
const line = lines2 [ i ]
const trimmed = line . trim ()
// Track mxCell depth
const isOpenCell =
/<mxCell\s/ . test ( trimmed ) && ! trimmed . endsWith ( "/>" )
const isCloseCell = trimmed === "</mxCell>"
if ( isOpenCell ) {
if ( cellDepth > 0 ) {
// Found nested cell - insert closing tag for parent before this line
const indent = line . match ( /^(\s*)/ ) ? .[ 1 ] || ""
newLines . push ( indent + "</mxCell>" )
trueNestedFixed ++
pendingCloseRemoval ++ // Need to remove one </mxCell> later
}
cellDepth = 1 // Reset to 1 since we just opened a new cell
newLines . push ( line )
} else if ( isCloseCell ) {
if ( pendingCloseRemoval > 0 ) {
pendingCloseRemoval --
// Skip this extra closing tag
} else {
cellDepth = Math . max ( 0 , cellDepth - 1 )
newLines . push ( line )
}
2025-12-13 15:00:28 +09:00
} else {
newLines . push ( line )
}
2026-10-03 17:45:25 +09:00
}
if ( trueNestedFixed > 0 ) {
fixed = newLines . join ( "\n" )
fixes . push ( `Fixed ${ trueNestedFixed } true nested mxCell(s)` )
2025-12-13 15:00:28 +09:00
}
}
2026-10-03 17:45:25 +09:00
// 12. Fix duplicate IDs by appending suffix, page by page (ids such as the
// root cells "0" and "1" legitimately repeat across pages)
let renamedIds = 0
const renamePage = ( page : string ) => {
const { xml : renamed , renamed : count } = renameDuplicateIds ( page )
renamedIds += count
return renamed
2025-12-13 15:00:28 +09:00
}
2026-10-03 17:45:25 +09:00
fixed = /<diagram\b/ . test ( fixed )
? fixed . replace ( PAGE_PATTERN , renamePage )
: renamePage ( fixed )
if ( renamedIds > 0 ) {
fixes . push ( `Renamed ${ renamedIds } duplicate ID(s)` )
2025-12-13 15:00:28 +09:00
}
// 9. Fix empty id attributes by generating unique IDs
let emptyIdCount = 0
fixed = fixed . replace (
/<mxCell([^>]*)\sid\s*=\s*["']\s*["']([^>]*)>/g ,
2025-12-13 23:31:01 +09:00
( _match , before , after ) => {
2025-12-13 15:00:28 +09:00
emptyIdCount ++
const newId = `cell_ ${ Date . now () } _ ${ emptyIdCount } `
return `<mxCell ${ before } id=" ${ newId } " ${ after } >`
},
)
if ( emptyIdCount > 0 ) {
fixes . push ( `Generated ${ emptyIdCount } missing ID(s)` )
}
2025-12-13 16:02:56 +09:00
// 13. Aggressive: drop broken mxCell elements that can't be fixed
// Only do this if DOM parser still finds errors after all other fixes
if ( typeof DOMParser !== "undefined" ) {
let droppedCells = 0
2025-12-13 23:31:01 +09:00
let maxIterations = MAX_DROP_ITERATIONS
2025-12-13 16:02:56 +09:00
while ( maxIterations -- > 0 ) {
const parser = new DOMParser ()
const doc = parser . parseFromString ( fixed , "text/xml" )
const parseError = doc . querySelector ( "parsererror" )
if ( ! parseError ) break // Valid now!
const errText = parseError . textContent || ""
const match = errText . match ( /(\d+):\d+:/ )
if ( ! match ) break
const errLine = parseInt ( match [ 1 ], 10 ) - 1
const lines = fixed . split ( "\n" )
// Find the mxCell containing this error line
let cellStart = errLine
let cellEnd = errLine
// Go back to find <mxCell
while ( cellStart > 0 && ! lines [ cellStart ]. includes ( "<mxCell" )) {
cellStart --
}
// Go forward to find </mxCell> or />
while ( cellEnd < lines . length - 1 ) {
if (
lines [ cellEnd ]. includes ( "</mxCell>" ) ||
lines [ cellEnd ]. trim (). endsWith ( "/>" )
) {
break
}
cellEnd ++
}
// Remove these lines
lines . splice ( cellStart , cellEnd - cellStart + 1 )
fixed = lines . join ( "\n" )
droppedCells ++
}
if ( droppedCells > 0 ) {
fixes . push ( `Dropped ${ droppedCells } unfixable mxCell element(s)` )
}
}
2025-12-13 15:00:28 +09:00
return { fixed , fixes }
}
/**
* Validates XML and attempts to fix if invalid
* @param xml - The XML string to validate and potentially fix
* @returns Object with validation result, fixed XML if applicable, and fixes applied
*/
export function validateAndFixXml ( xml : string ) : {
valid : boolean
error : string | null
fixed : string | null
fixes : string []
} {
// First validation attempt
let error = validateMxCellStructure ( xml )
if ( ! error ) {
return { valid : true , error : null , fixed : null , fixes : [] }
}
// Try to fix
const { fixed , fixes } = autoFixXml ( xml )
2025-12-14 19:38:40 +09:00
console . log ( "[validateAndFixXml] Fixes applied:" , fixes )
2025-12-13 15:00:28 +09:00
// Validate the fixed version
error = validateMxCellStructure ( fixed )
2025-12-14 19:38:40 +09:00
if ( error ) {
console . log ( "[validateAndFixXml] Still invalid after fix:" , error )
}
2025-12-13 15:00:28 +09:00
if ( ! error ) {
return { valid : true , error : null , fixed , fixes }
}
2025-12-14 19:38:40 +09:00
// Still invalid after fixes - but return the partially fixed XML
// so we can see what was fixed and what error remains
return {
valid : false ,
error ,
fixed : fixes.length > 0 ? fixed : null ,
fixes ,
}
2025-12-13 15:00:28 +09:00
}
2026-10-03 17:45:25 +09:00
/**
* Decode an xmlsvg export (SVG data URL) into uncompressed diagram XML.
* Only the first page is returned; for the full multi-page document use the
* autosaved chartXML instead.
*/
2025-03-27 06:45:38 +00:00
export function extractDiagramXML ( xml_svg_string : string ) : string {
2025-12-06 12:46:40 +09:00
try {
// 1. Parse the SVG string (using built-in DOMParser in a browser-like environment)
const svgString = atob ( xml_svg_string . slice ( 26 ))
const parser = new DOMParser ()
const svgDoc = parser . parseFromString ( svgString , "image/svg+xml" )
const svgElement = svgDoc . querySelector ( "svg" )
2025-03-27 06:45:38 +00:00
2025-12-06 12:46:40 +09:00
if ( ! svgElement ) {
throw new Error ( "No SVG element found in the input string." )
}
// 2. Extract the 'content' attribute
const encodedContent = svgElement . getAttribute ( "content" )
if ( ! encodedContent ) {
throw new Error ( "SVG element does not have a 'content' attribute." )
}
// 3. Decode HTML entities (using a minimal function)
function decodeHtmlEntities ( str : string ) {
const textarea = document . createElement ( "textarea" ) // Use built-in element
textarea . innerHTML = str
return textarea . value
}
const xmlContent = decodeHtmlEntities ( encodedContent )
// 4. Parse the XML content
const xmlDoc = parser . parseFromString ( xmlContent , "text/xml" )
const diagramElement = xmlDoc . querySelector ( "diagram" )
if ( ! diagramElement ) {
throw new Error ( "No diagram element found" )
}
// 5. Extract base64 encoded data
const base64EncodedData = diagramElement . textContent
if ( ! base64EncodedData ) {
throw new Error ( "No encoded data found in the diagram element" )
}
// 6. Decode base64 data
const binaryString = atob ( base64EncodedData )
// 7. Convert binary string to Uint8Array
const len = binaryString . length
const bytes = new Uint8Array ( len )
for ( let i = 0 ; i < len ; i ++ ) {
bytes [ i ] = binaryString . charCodeAt ( i )
}
// 8. Decompress data using pako (equivalent to zlib.decompress with wbits=-15)
const decompressedData = pako . inflate ( bytes , { windowBits : - 15 })
// 9. Convert the decompressed data to a string
const decoder = new TextDecoder ( "utf-8" )
const decodedString = decoder . decode ( decompressedData )
// Decode URL-encoded content (equivalent to Python's urllib.parse.unquote)
const urlDecodedString = decodeURIComponent ( decodedString )
return urlDecodedString
} catch ( error ) {
console . error ( "Error extracting diagram XML:" , error )
throw error // Re-throw for caller handling
2025-03-27 06:45:38 +00:00
}
}