diff --git a/app/api/parse-url/route.ts b/app/api/parse-url/route.ts index 40014547..794b711e 100644 --- a/app/api/parse-url/route.ts +++ b/app/api/parse-url/route.ts @@ -7,6 +7,31 @@ const MAX_CONTENT_LENGTH = 150000 // Match PDF limit const EXTRACT_TIMEOUT_MS = 15000 const USER_AGENT = "Mozilla/5.0 (compatible; NextAIDrawio/1.0)" +// Detect the page's charset so non-UTF-8 pages (Shift_JIS/GBK/EUC/Big5, common +// on CJK sites) are decoded correctly. Response.text() always assumes UTF-8 and +// would produce mojibake; the article-extractor library does the same detection +// when it fetches the page itself, which we no longer rely on. +function detectCharset( + contentType: string | null, + buffer: ArrayBuffer, +): string { + // 1. HTTP Content-Type header charset (most authoritative). + const headerCharset = contentType?.match(/charset=([^;]+)/i)?.[1]?.trim() + // 2. / in the first bytes of the document. + const head = new TextDecoder("utf-8").decode(buffer.slice(0, 4096)) + const metaCharset = + head.match(/]+charset=["']?\s*([\w-]+)/i)?.[1] || + head.match(/]+content=["'][^"']*charset=([\w-]+)/i)?.[1] + const charset = (headerCharset || metaCharset || "utf-8").toLowerCase() + // TextDecoder throws on unknown encoding labels; fall back to UTF-8. + try { + new TextDecoder(charset) + return charset + } catch { + return "utf-8" + } +} + export async function POST(req: Request) { try { const { url } = await req.json() @@ -72,7 +97,9 @@ export async function POST(req: Request) { ) } - html = await response.text() + const buffer = await response.arrayBuffer() + const charset = detectCharset(contentType, buffer) + html = new TextDecoder(charset).decode(buffer) } catch (err: any) { if (err?.name === "AbortError") { return NextResponse.json( @@ -90,7 +117,14 @@ export async function POST(req: Request) { clearTimeout(timeoutId) } - const article = await extractFromHtml(html, url) + // extractFromHtml throws (not returns null) on empty/non-HTML bodies, + // so map any parse error to the same 400 as the no-content case. + let article: Awaited> + try { + article = await extractFromHtml(html, url) + } catch { + article = null + } if (!article || !article.content) { return NextResponse.json( diff --git a/electron/main/history-recovery.ts b/electron/main/history-recovery.ts new file mode 100644 index 00000000..2c37f2c4 --- /dev/null +++ b/electron/main/history-recovery.ts @@ -0,0 +1,118 @@ +import fs from "node:fs" +import path from "node:path" +import { app } from "electron" + +/** + * Recover chat history / templates that were orphaned by an origin change. + * + * Browser storage (IndexedDB, localStorage) is scoped per origin + * (scheme://host:port). Older releases served the app from + * `http://localhost:61337`; v0.4.14+ switched the host to `127.0.0.1` + * (and the default port to 13370), which is a *different* origin. The old + * data is still on disk but the new origin can't see it, so users perceive + * their chat history and template library as gone (issue #875). + * + * Chromium stores each origin's IndexedDB databases in a single directory + * named `http__.indexeddb.leveldb`. All IndexedDB databases for + * one origin (both `next-ai-drawio` sessions and `next-ai-drawio-templates`) + * live in that one directory. So if the origin the window is about to load + * has no data yet, we copy the richest orphaned origin's directory into it. + * + * This is a best-effort, one-time operation: + * - It only copies when the target origin is empty, so it never clobbers + * data the user created on the current origin. + * - A marker file makes it run at most once, so deleting all history won't + * resurrect old data on the next launch. + * - It must run BEFORE the window loads the URL, while the target leveldb is + * still unlocked. The Next.js server process does not touch IndexedDB + * (that is renderer/Chromium state), so running after startNextServer is + * fine as long as it is before createWindow. + */ +export function recoverOrphanedHistory(serverUrl: string): void { + try { + const userData = app.getPath("userData") + const marker = path.join(userData, ".idb-history-recovered") + if (fs.existsSync(marker)) return + + const idbDir = path.join(userData, "IndexedDB") + if (!fs.existsSync(idbDir)) { + writeMarker(marker) + return + } + + const { hostname, port } = new URL(serverUrl) + const targetName = `http_${hostname}_${port}.indexeddb.leveldb` + const targetPath = path.join(idbDir, targetName) + + // If the active origin already has data, leave it untouched. + if (fs.existsSync(targetPath) && idbDataSize(targetPath) > 0) { + writeMarker(marker) + return + } + + // Find the richest orphaned origin directory (excluding the target). + const candidates = fs + .readdirSync(idbDir) + .filter( + (name) => + name.endsWith(".indexeddb.leveldb") && name !== targetName, + ) + .map((name) => { + const dir = path.join(idbDir, name) + return { name, dir, size: idbDataSize(dir) } + }) + .filter((c) => c.size > 0) + .sort((a, b) => b.size - a.size) + + if (candidates.length === 0) { + writeMarker(marker) + return + } + + const best = candidates[0] + // Copy the whole leveldb directory (CURRENT/MANIFEST/*.ldb/*.log) + // into the active origin's directory name. The source origin is never + // opened by this app and old versions aren't running (single-instance + // lock), so it is quiescent and safe to copy. + fs.cpSync(best.dir, targetPath, { recursive: true }) + console.log( + `Recovered chat history: copied ${best.name} -> ${targetName} (${best.size} bytes)`, + ) + writeMarker(marker) + } catch (error) { + // Never block startup on recovery failure. + console.error("History recovery failed:", error) + } +} + +/** + * Size of actual IndexedDB data in a leveldb directory. + * + * Counts only data-bearing files: `*.ldb` (compacted tables) and numbered + * write-ahead logs like `000003.log`. Deliberately ignores leveldb's + * human-readable `LOG`/`LOG.old` info files, which exist even for an empty + * database and would otherwise make an empty origin look non-empty. + */ +function idbDataSize(dir: string): number { + let total = 0 + try { + if (!fs.statSync(dir).isDirectory()) return 0 + for (const file of fs.readdirSync(dir)) { + if (/\.ldb$/i.test(file) || /^\d+\.log$/i.test(file)) { + total += fs.statSync(path.join(dir, file)).size + } + } + } catch { + return 0 + } + return total +} + +function writeMarker(marker: string): void { + try { + fs.writeFileSync(marker, new Date().toISOString()) + } catch { + // Ignore — worst case we re-check on next launch (the empty-target + // guard still prevents clobbering existing data). + } +} diff --git a/electron/main/index.ts b/electron/main/index.ts index 6c613da2..348ec951 100644 --- a/electron/main/index.ts +++ b/electron/main/index.ts @@ -2,6 +2,7 @@ import { app, BrowserWindow, dialog, shell } from "electron" import { buildAppMenu } from "./app-menu" import { getCurrentPresetEnv } from "./config-manager" import { loadEnvFile } from "./env-loader" +import { recoverOrphanedHistory } from "./history-recovery" import { registerIpcHandlers } from "./ipc-handlers" import { startNextServer, stopNextServer } from "./next-server" import { applyProxyToEnv } from "./proxy-manager" @@ -56,6 +57,13 @@ if (!gotTheLock) { serverUrl = await startNextServer() } + // Recover chat history orphaned by an earlier origin change + // (host/port). Must run before the window loads the URL, while the + // target IndexedDB directory is still unlocked. (#875) + if (!isDev) { + recoverOrphanedHistory(serverUrl) + } + // Create main window createWindow(serverUrl) } catch (error) { diff --git a/lib/ssrf-protection.ts b/lib/ssrf-protection.ts index f9fad85c..6b43ddf2 100644 --- a/lib/ssrf-protection.ts +++ b/lib/ssrf-protection.ts @@ -41,6 +41,7 @@ function isPrivateIp(ip: string): boolean { if (a === 169 && b === 254) return true // 169.254.0.0/16 (link-local) if (a === 127) return true // 127.0.0.0/8 (loopback) if (a === 0) return true // 0.0.0.0/8 + if (a === 100 && b >= 64 && b <= 127) return true // 100.64.0.0/10 (CGNAT, used by some cloud internal networks) } return false diff --git a/tests/unit/ssrf-protection.test.ts b/tests/unit/ssrf-protection.test.ts index 53b18303..a05eab30 100644 --- a/tests/unit/ssrf-protection.test.ts +++ b/tests/unit/ssrf-protection.test.ts @@ -31,9 +31,26 @@ describe("isPrivateUrl", () => { expect(await isPrivateUrl("http://10.0.0.5/")).toBe(true) expect(await isPrivateUrl("http://192.168.1.1/")).toBe(true) expect(await isPrivateUrl("http://169.254.169.254/")).toBe(true) + expect(await isPrivateUrl("http://0.0.0.0/")).toBe(true) + // 100.64.0.0/10 CGNAT (RFC 6598), routable in some cloud internal nets + expect(await isPrivateUrl("http://100.64.0.1/")).toBe(true) + expect(await isPrivateUrl("http://100.127.255.255/")).toBe(true) expect(lookupMock).not.toHaveBeenCalled() }) + it("treats CGNAT boundaries correctly", async () => { + // 100.63.x and 100.128.x are outside 100.64.0.0/10 → public + lookupMock.mockResolvedValue([{ address: "100.63.255.255", family: 4 }]) + expect(await isPrivateUrl("http://just-below.example/")).toBe(false) + lookupMock.mockResolvedValue([{ address: "100.128.0.1", family: 4 }]) + expect(await isPrivateUrl("http://just-above.example/")).toBe(false) + }) + + it("blocks a hostname that resolves to a private IPv6 address", async () => { + lookupMock.mockResolvedValue([{ address: "fd00::1", family: 6 }]) + expect(await isPrivateUrl("http://v6.example.com/")).toBe(true) + }) + it("allows public URLs that resolve to public IPs", async () => { lookupMock.mockResolvedValue([{ address: "93.184.216.34", family: 4 }]) expect(await isPrivateUrl("https://example.com/article")).toBe(false)