2025-12-11 14:28:02 +09:00
|
|
|
"use client"
|
|
|
|
|
|
|
|
|
|
import { useState } from "react"
|
|
|
|
|
import { toast } from "sonner"
|
|
|
|
|
import {
|
|
|
|
|
extractPdfText,
|
|
|
|
|
extractTextFileContent,
|
|
|
|
|
isPdfFile,
|
|
|
|
|
isTextFile,
|
|
|
|
|
MAX_EXTRACTED_CHARS,
|
|
|
|
|
} from "@/lib/pdf-utils"
|
|
|
|
|
|
|
|
|
|
export interface FileData {
|
|
|
|
|
text: string
|
|
|
|
|
charCount: number
|
|
|
|
|
isExtracting: boolean
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Hook for processing file uploads, especially PDFs and text files.
|
|
|
|
|
* Handles text extraction, character limit validation, and cleanup.
|
|
|
|
|
*/
|
|
|
|
|
export function useFileProcessor() {
|
|
|
|
|
const [files, setFiles] = useState<File[]>([])
|
|
|
|
|
const [pdfData, setPdfData] = useState<Map<File, FileData>>(new Map())
|
|
|
|
|
|
|
|
|
|
const handleFileChange = async (newFiles: File[]) => {
|
|
|
|
|
setFiles(newFiles)
|
|
|
|
|
|
2026-10-03 17:45:27 +09:00
|
|
|
const pending = newFiles.filter(
|
|
|
|
|
(file) =>
|
|
|
|
|
(isPdfFile(file) || isTextFile(file)) && !pdfData.has(file),
|
|
|
|
|
)
|
2025-12-11 14:28:02 +09:00
|
|
|
|
2026-10-03 17:45:27 +09:00
|
|
|
// Before any await: drop data for removed files and mark every new
|
|
|
|
|
// file as extracting, so queued files also block sending
|
|
|
|
|
setPdfData((prev) => {
|
|
|
|
|
const next = new Map<File, FileData>()
|
|
|
|
|
for (const file of newFiles) {
|
|
|
|
|
const existing = prev.get(file)
|
|
|
|
|
if (existing) next.set(file, existing)
|
|
|
|
|
}
|
|
|
|
|
for (const file of pending) {
|
|
|
|
|
next.set(file, { text: "", charCount: 0, isExtracting: true })
|
|
|
|
|
}
|
|
|
|
|
return next
|
|
|
|
|
})
|
2025-12-11 14:28:02 +09:00
|
|
|
|
2026-10-03 17:45:27 +09:00
|
|
|
// Extract one file at a time
|
|
|
|
|
for (const file of pending) {
|
|
|
|
|
try {
|
|
|
|
|
let text: string
|
|
|
|
|
if (isPdfFile(file)) {
|
|
|
|
|
text = await extractPdfText(file)
|
|
|
|
|
} else {
|
|
|
|
|
text = await extractTextFileContent(file)
|
|
|
|
|
}
|
2025-12-11 14:28:02 +09:00
|
|
|
|
2026-10-03 17:45:27 +09:00
|
|
|
// Check character limit
|
|
|
|
|
if (text.length > MAX_EXTRACTED_CHARS) {
|
|
|
|
|
const limitK = MAX_EXTRACTED_CHARS / 1000
|
|
|
|
|
toast.error(
|
|
|
|
|
`${file.name}: Content exceeds ${limitK}k character limit (${(text.length / 1000).toFixed(1)}k chars)`,
|
|
|
|
|
)
|
2025-12-11 14:28:02 +09:00
|
|
|
setPdfData((prev) => {
|
|
|
|
|
const next = new Map(prev)
|
|
|
|
|
next.delete(file)
|
|
|
|
|
return next
|
|
|
|
|
})
|
2026-10-03 17:45:27 +09:00
|
|
|
// Remove the file from the list
|
|
|
|
|
setFiles((prev) => prev.filter((f) => f !== file))
|
|
|
|
|
continue
|
2025-12-11 14:28:02 +09:00
|
|
|
}
|
2026-10-03 17:45:27 +09:00
|
|
|
|
|
|
|
|
setPdfData((prev) => {
|
|
|
|
|
// The file was removed while extracting
|
|
|
|
|
if (!prev.has(file)) return prev
|
|
|
|
|
const next = new Map(prev)
|
|
|
|
|
next.set(file, {
|
|
|
|
|
text,
|
|
|
|
|
charCount: text.length,
|
|
|
|
|
isExtracting: false,
|
|
|
|
|
})
|
|
|
|
|
return next
|
|
|
|
|
})
|
|
|
|
|
} catch (error) {
|
|
|
|
|
console.error("Failed to extract text:", error)
|
|
|
|
|
toast.error(`Failed to read file: ${file.name}`)
|
|
|
|
|
setPdfData((prev) => {
|
|
|
|
|
const next = new Map(prev)
|
|
|
|
|
next.delete(file)
|
|
|
|
|
return next
|
|
|
|
|
})
|
2025-12-11 14:28:02 +09:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return {
|
|
|
|
|
files,
|
|
|
|
|
pdfData,
|
|
|
|
|
handleFileChange,
|
|
|
|
|
setFiles, // Export for external control (e.g., clearing files)
|
|
|
|
|
}
|
|
|
|
|
}
|