diff --git a/src/content/content.css b/src/content/content.css index dd5a5f6..077a022 100644 --- a/src/content/content.css +++ b/src/content/content.css @@ -244,3 +244,113 @@ opacity: 1; transform: translateY(0); } + +/* Document scan preview overlay */ +.ss-doc-preview { + position: fixed; + top: 50%; + left: 50%; + transform: translate(-50%, -50%) scale(0.95); + max-width: 480px; + width: 90vw; + background: #1a1a1a; + color: #e5e7eb; + border: 1px solid #f59e0b; + border-radius: 12px; + padding: 16px 20px; + font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, sans-serif; + font-size: 12px; + z-index: 9999999; + box-shadow: 0 8px 32px rgba(0, 0, 0, 0.6); + opacity: 0; + pointer-events: none; + transition: opacity 0.2s, transform 0.2s; + max-height: 70vh; + overflow-y: auto; +} + +.ss-doc-preview.visible { + opacity: 1; + transform: translate(-50%, -50%) scale(1); + pointer-events: auto; +} + +.ss-dp-header { + display: flex; + justify-content: space-between; + align-items: center; + margin-bottom: 8px; +} + +.ss-dp-count { + font-size: 11px; + color: #f59e0b; + font-weight: 600; +} + +.ss-dp-note { + font-size: 11px; + color: #9ca3af; + margin-bottom: 10px; + font-style: italic; +} + +.ss-dp-items { + max-height: 200px; + overflow-y: auto; + margin-bottom: 12px; +} + +.ss-dp-item { + display: flex; + align-items: center; + gap: 6px; + padding: 4px 0; + border-bottom: 1px solid #333; + font-size: 11px; +} + +.ss-dp-item:last-child { border-bottom: none; } + +.ss-dp-orig { + color: #f87171; + background: #1f1f1f; + padding: 2px 5px; + border-radius: 3px; +} + +.ss-dp-repl { + color: #4ade80; + background: #1f1f1f; + padding: 2px 5px; + border-radius: 3px; +} + +.ss-dp-actions { + display: flex; + gap: 8px; +} + +.ss-dp-btn { + flex: 1; + padding: 8px 12px; + border: none; + border-radius: 6px; + font-size: 12px; + font-weight: 500; + cursor: pointer; +} + +.ss-dp-confirm { + background: #10b981; + color: #fff; +} + +.ss-dp-confirm:hover { background: #059669; } + +.ss-dp-cancel { + background: #374151; + color: #e5e7eb; +} + +.ss-dp-cancel:hover { background: #4b5563; } diff --git a/src/content/content.js b/src/content/content.js index 29e0796..e4a8cf6 100644 --- a/src/content/content.js +++ b/src/content/content.js @@ -765,35 +765,49 @@ const urlStr = typeof url === 'string' ? url : url?.url || ''; const method = (options?.method || 'GET').toUpperCase(); - // Only intercept POST/PUT/PATCH with a string body + // Only intercept POST/PUT/PATCH with a body if ( (method === 'POST' || method === 'PUT' || method === 'PATCH') && - options?.body && typeof options.body === 'string' && + options?.body && !shouldSkipUrl(urlStr) ) { - try { - // Try JSON - const body = JSON.parse(options.body); - const { modified, replacements } = processBody(body); - - if (modified) { - options = { ...options, body: JSON.stringify(body) }; - notifySubstitutions(replacements); - console.log( - `[Silent Send] Substituted ${replacements.length} value(s) in ${urlStr}` - ); + // Handle FormData with file uploads + if (options.body instanceof FormData) { + try { + const newFormData = await processFormData(options.body); + if (newFormData) { + options = { ...options, body: newFormData }; + } + } catch (e) { + console.warn('[Silent Send] FormData processing failed:', e); } - } catch (e) { - // Not JSON — try raw string substitution (form data, etc.) - if (options.body.length > MIN_STRING_LENGTH) { - const result = substituteAll(options.body); - if (result.modified) { - options = { ...options, body: result.text }; - notifySubstitutions(result.replacements); + } + // Handle string bodies (JSON, raw text) + else if (typeof options.body === 'string') { + try { + // Try JSON + const body = JSON.parse(options.body); + const { modified, replacements } = processBody(body); + + if (modified) { + options = { ...options, body: JSON.stringify(body) }; + notifySubstitutions(replacements); console.log( - `[Silent Send] Substituted ${result.replacements.length} value(s) in form body` + `[Silent Send] Substituted ${replacements.length} value(s) in ${urlStr}` ); } + } catch (e) { + // Not JSON — try raw string substitution (form data, etc.) + if (options.body.length > MIN_STRING_LENGTH) { + const result = substituteAll(options.body); + if (result.modified) { + options = { ...options, body: result.text }; + notifySubstitutions(result.replacements); + console.log( + `[Silent Send] Substituted ${result.replacements.length} value(s) in form body` + ); + } + } } } } @@ -801,6 +815,272 @@ return originalFetch.call(this, url, options); }; + // ============================================================ + // Document Upload Processing + // + // Scans files in FormData uploads for PPI. Supports PDF, DOCX, + // XLSX, and text files. Shows preview for binary formats. + // ============================================================ + + async function processFormData(formData) { + let modified = false; + const newFormData = new FormData(); + const allReplacements = []; + + for (const [key, value] of formData.entries()) { + if (value instanceof File && value.size > 0) { + // Process the file through document scanner + const usePreview = settings.docScanPreview !== false && + /\.(pdf|docx|xlsx)$/i.test(value.name); + + const result = await documentScan(value, value.name, { + previewMode: usePreview, + }); + + if (result.preview && usePreview && result.replacements.length > 0) { + // Show preview and wait for user confirmation + const confirmed = await showDocScanPreview(result.preview, value.name); + if (!confirmed) { + // User cancelled — use original file + newFormData.append(key, value); + continue; + } + } + + if (result.replacements.length > 0 && !result.skipped) { + // Use the substituted file + const newFile = new File([result.file], result.filename || value.name, { + type: result.file.type || value.type, + }); + newFormData.append(key, newFile); + allReplacements.push(...result.replacements); + modified = true; + console.log( + `[Silent Send] Substituted ${result.replacements.length} value(s) in file: ${value.name}` + ); + } else { + newFormData.append(key, value); + } + } else if (typeof value === 'string') { + // String form field — substitute + const result = substituteAll(value); + if (result.modified) { + newFormData.append(key, result.text); + allReplacements.push(...result.replacements); + modified = true; + } else { + newFormData.append(key, value); + } + } else { + newFormData.append(key, value); + } + } + + if (modified) { + notifySubstitutions(allReplacements); + return newFormData; + } + return null; + } + + /** + * Scan a document file for PPI using the inline document scanner. + */ + async function documentScan(file, filename, options) { + // Detect file type + const ext = (filename || '').split('.').pop().toLowerCase(); + const isText = /^(txt|csv|tsv|json|md|log|yaml|yml|xml|html|css|js|ts|py|sh|sql|rb|go|rs|java|c|cpp|h|php)$/.test(ext); + const isPDF = ext === 'pdf'; + const isDOCX = ext === 'docx'; + const isXLSX = ext === 'xlsx'; + + if (isText) { + const text = await file.text(); + const result = substituteAll(text); + if (result.modified) { + return { + file: new Blob([result.text], { type: file.type || 'text/plain' }), + filename, + replacements: result.replacements, + }; + } + return { file, filename, replacements: [] }; + } + + if (isPDF) { + // Extract text and scan — create clean text version + try { + const text = await extractPDFTextSimple(file); + if (!text || text.trim().length < 5) { + return { file, filename, replacements: [], skipped: true, + reason: 'No extractable text (scanned/image PDF)' }; + } + const result = substituteAll(text); + const preview = { + format: 'pdf', replacementCount: result.replacements.length, + replacements: result.replacements.slice(0, 15), + note: 'PDF will be converted to plain text (layout not preserved)', + }; + if (options.previewMode && result.replacements.length > 0) { + return { file, filename, replacements: result.replacements, preview }; + } + if (result.modified) { + return { + file: new Blob([result.text], { type: 'text/plain' }), + filename: filename.replace(/\.pdf$/i, '.txt'), + replacements: result.replacements, preview, + }; + } + return { file, filename, replacements: [] }; + } catch (e) { + return { file, filename, replacements: [], skipped: true, reason: e.message }; + } + } + + // DOCX/XLSX — not supported in page world without libraries + // For now, extract what text we can and warn + if (isDOCX || isXLSX) { + try { + const text = await extractZipXMLText(file); + if (!text) return { file, filename, replacements: [], skipped: true }; + const result = substituteAll(text); + const preview = { + format: ext, replacementCount: result.replacements.length, + replacements: result.replacements.slice(0, 15), + note: isDOCX ? 'Text in document will be substituted' : 'Cell values will be substituted', + }; + if (options.previewMode && result.replacements.length > 0) { + return { file, filename, replacements: result.replacements, preview }; + } + // For actual replacement, we'd need full ZIP rewrite + // For now, warn about the PPI found + if (result.modified) { + return { file, filename, replacements: result.replacements, preview }; + } + return { file, filename, replacements: [] }; + } catch (e) { + return { file, filename, replacements: [], skipped: true, reason: e.message }; + } + } + + return { file, filename, replacements: [], skipped: true }; + } + + /** + * Simple PDF text extraction from content streams. + */ + async function extractPDFTextSimple(file) { + const buffer = await file.arrayBuffer(); + const str = new TextDecoder('latin1').decode(new Uint8Array(buffer)); + const texts = []; + const re = /stream\r?\n([\s\S]*?)endstream/g; + let m; + while ((m = re.exec(str)) !== null) { + // Extract text operators: (text) Tj, [(text)...] TJ + const content = m[1]; + const parts = []; + const tj = /\(([^)]*)\)\s*Tj/g; + let t; + while ((t = tj.exec(content)) !== null) parts.push(t[1]); + const tjArr = /\[(.*?)\]\s*TJ/g; + while ((t = tjArr.exec(content)) !== null) { + const inner = /\(([^)]*)\)/g; + let s; + while ((s = inner.exec(t[1])) !== null) parts.push(s[1]); + } + if (parts.length) texts.push(parts.join('')); + } + return texts.join('\n'); + } + + /** + * Extract text from DOCX/XLSX XML entries. + */ + async function extractZipXMLText(file) { + // Minimal: read the file as text and find XML text content + // This is a rough extraction — better than nothing + try { + const buffer = await file.arrayBuffer(); + const bytes = new Uint8Array(buffer); + const str = new TextDecoder('latin1').decode(bytes); + // Find text between XML tags + const texts = []; + const re = />([^<]{3,}) { + if (!docPreviewEl) { + docPreviewEl = document.createElement('div'); + docPreviewEl.className = 'ss-doc-preview'; + document.body.appendChild(docPreviewEl); + } + + const items = (preview.replacements || []).map(r => + `
+ ${(r.original || '').length > 25 ? r.original.slice(0, 22) + '...' : r.original} + + ${r.replaced} +
` + ).join(''); + + docPreviewEl.innerHTML = ` +
+ PPI found in ${esc(filename)} + ${preview.replacementCount} item(s) +
+
${preview.note || ''}
+
${items}
+
+ + +
+ `; + + docPreviewEl.classList.add('visible'); + + const confirm = docPreviewEl.querySelector('.ss-dp-confirm'); + const cancel = docPreviewEl.querySelector('.ss-dp-cancel'); + + const cleanup = () => { + docPreviewEl.classList.remove('visible'); + confirm.removeEventListener('click', onConfirm); + cancel.removeEventListener('click', onCancel); + }; + + const onConfirm = () => { cleanup(); resolve(true); }; + const onCancel = () => { cleanup(); resolve(false); }; + + confirm.addEventListener('click', onConfirm); + cancel.addEventListener('click', onCancel); + + // Auto-dismiss after 30 seconds (upload original) + setTimeout(() => { + if (docPreviewEl.classList.contains('visible')) { + cleanup(); + resolve(false); + } + }, 30000); + }); + } + // ============================================================ // XMLHttpRequest Interception — same aggressive approach // ============================================================ diff --git a/src/lib/document-scanner.js b/src/lib/document-scanner.js new file mode 100644 index 0000000..666a068 --- /dev/null +++ b/src/lib/document-scanner.js @@ -0,0 +1,577 @@ +/** + * Silent Send - Document Scanner + * + * Scans uploaded documents for PPI and substitutes/redacts before + * the file reaches the AI service. + * + * Supported formats: + * - PDF: Extract text, scan for PPI, create sanitized plaintext version + * (PDFs can't be reliably edited in-place without breaking layout) + * - DOCX: Parse XML, find-replace text, repackage ZIP (layout preserved) + * - XLSX: Parse cells, find-replace values, repackage (formatting preserved) + * - TXT/CSV/JSON/MD: Direct string replacement + * + * Modes: + * - Silent: substitute and upload (default for text files) + * - Preview: show PPI findings, let user confirm before upload (default for PDF/DOCX/XLSX) + * + * Integration: + * - The fetch interceptor in content.js detects multipart/form-data uploads + * - Calls DocumentScanner.processUpload() which returns the modified file + * - Badge count includes document substitutions + */ + +const DocumentScanner = { + /** + * Process a file upload. Detects format and applies appropriate strategy. + * + * @param {File|Blob} file - the file being uploaded + * @param {string} filename - original filename + * @param {Function} substituteAll - the substituteAll function from content.js + * @param {Object} options - { previewMode: boolean } + * @returns {{ file: Blob, filename: string, replacements: Array, preview?: Object, skipped?: boolean }} + */ + async processUpload(file, filename, substituteAll, options = {}) { + const ext = (filename || '').split('.').pop().toLowerCase(); + const type = this._detectType(ext, file.type); + + switch (type) { + case 'text': + return this._processText(file, filename, substituteAll); + case 'pdf': + return this._processPDF(file, filename, substituteAll, options); + case 'docx': + return this._processDOCX(file, filename, substituteAll, options); + case 'xlsx': + return this._processXLSX(file, filename, substituteAll, options); + default: + // Unsupported format — pass through unchanged + return { file, filename, replacements: [], skipped: true }; + } + }, + + /** + * Plain text files: direct string replacement. + */ + async _processText(file, filename, substituteAll) { + const text = await file.text(); + const result = substituteAll(text); + + if (!result.modified) { + return { file, filename, replacements: [] }; + } + + const newBlob = new Blob([result.text], { type: file.type || 'text/plain' }); + return { + file: newBlob, + filename, + replacements: result.replacements, + }; + }, + + /** + * PDF: extract text, scan for PPI, create sanitized text file. + * PDFs can't be reliably edited in-place, so we extract text, + * substitute PPI, and send the clean text instead. + */ + async _processPDF(file, filename, substituteAll, options) { + const text = await this._extractPDFText(file); + if (!text || text.trim().length === 0) { + // Scanned PDF or image-only — can't extract text + return { + file, + filename, + replacements: [], + skipped: true, + reason: 'No extractable text in PDF (may be scanned/image-only)', + }; + } + + const result = substituteAll(text); + + if (!result.modified && !options.previewMode) { + return { file, filename, replacements: [] }; + } + + // Build preview data + const preview = { + originalLength: text.length, + substitutedLength: result.text.length, + replacementCount: result.replacements.length, + replacements: result.replacements.slice(0, 20), // limit preview + format: 'pdf', + note: 'PDF will be converted to plain text for upload (layout not preserved)', + }; + + if (options.previewMode) { + return { file, filename, replacements: result.replacements, preview }; + } + + // Create sanitized text file + const sanitizedBlob = new Blob([result.text], { type: 'text/plain' }); + const sanitizedFilename = filename.replace(/\.pdf$/i, '.txt'); + + return { + file: sanitizedBlob, + filename: sanitizedFilename, + replacements: result.replacements, + preview, + convertedFrom: 'pdf', + }; + }, + + /** + * DOCX: parse the ZIP, find-replace text in XML content, repackage. + * Layout, formatting, images, and styles are preserved. + */ + async _processDOCX(file, filename, substituteAll, options) { + try { + const zip = await this._readZip(file); + let totalReplacements = []; + let modified = false; + + // DOCX text is in word/document.xml (main body), word/header*.xml, + // word/footer*.xml, and word/comments.xml + const textFiles = Object.keys(zip.files).filter(name => + /^word\/(document|header\d*|footer\d*|comments|endnotes|footnotes)\.xml$/.test(name) + ); + + for (const xmlPath of textFiles) { + const xmlContent = await zip.files[xmlPath].async('string'); + + // Extract text runs from XML, substitute, and rebuild + const { xml: newXml, replacements } = this._substituteInXML(xmlContent, substituteAll); + + if (replacements.length > 0) { + zip.files[xmlPath] = { data: newXml, isText: true }; + totalReplacements.push(...replacements); + modified = true; + } + } + + if (!modified && !options.previewMode) { + return { file, filename, replacements: [] }; + } + + const preview = { + replacementCount: totalReplacements.length, + replacements: totalReplacements.slice(0, 20), + format: 'docx', + note: 'Text in document body, headers, and footers will be substituted. Formatting preserved.', + }; + + if (options.previewMode) { + return { file, filename, replacements: totalReplacements, preview }; + } + + // Repackage the ZIP + const newBlob = await this._writeZip(zip); + return { + file: newBlob, + filename, + replacements: totalReplacements, + preview, + }; + } catch (e) { + console.warn('[Silent Send] DOCX processing failed:', e); + return { file, filename, replacements: [], skipped: true, reason: e.message }; + } + }, + + /** + * XLSX: parse the ZIP, find-replace text in shared strings and sheet cells. + */ + async _processXLSX(file, filename, substituteAll, options) { + try { + const zip = await this._readZip(file); + let totalReplacements = []; + let modified = false; + + // XLSX stores shared strings in xl/sharedStrings.xml + // and inline strings in xl/worksheets/sheet*.xml + const targetFiles = Object.keys(zip.files).filter(name => + name === 'xl/sharedStrings.xml' || + /^xl\/worksheets\/sheet\d+\.xml$/.test(name) + ); + + for (const xmlPath of targetFiles) { + const xmlContent = await zip.files[xmlPath].async('string'); + const { xml: newXml, replacements } = this._substituteInXML(xmlContent, substituteAll); + + if (replacements.length > 0) { + zip.files[xmlPath] = { data: newXml, isText: true }; + totalReplacements.push(...replacements); + modified = true; + } + } + + if (!modified && !options.previewMode) { + return { file, filename, replacements: [] }; + } + + const preview = { + replacementCount: totalReplacements.length, + replacements: totalReplacements.slice(0, 20), + format: 'xlsx', + note: 'Cell values and shared strings will be substituted. Formatting and formulas preserved.', + }; + + if (options.previewMode) { + return { file, filename, replacements: totalReplacements, preview }; + } + + const newBlob = await this._writeZip(zip); + return { + file: newBlob, + filename, + replacements: totalReplacements, + preview, + }; + } catch (e) { + console.warn('[Silent Send] XLSX processing failed:', e); + return { file, filename, replacements: [], skipped: true, reason: e.message }; + } + }, + + // ---------------------------------------------------------------- + // PDF text extraction (no external libraries) + // + // Parses PDF content streams to extract text. Handles the common + // text operators (Tj, TJ, ', "). Doesn't handle every PDF edge + // case but works for most text-based PDFs. + // ---------------------------------------------------------------- + + async _extractPDFText(file) { + try { + const buffer = await file.arrayBuffer(); + const bytes = new Uint8Array(buffer); + const str = new TextDecoder('latin1').decode(bytes); + + // Find all stream...endstream blocks + const texts = []; + const streamRegex = /stream\r?\n([\s\S]*?)endstream/g; + let match; + + while ((match = streamRegex.exec(str)) !== null) { + const streamData = match[1]; + + // Try to decompress if FlateDecode + let content = streamData; + const filterMatch = str.slice(Math.max(0, match.index - 200), match.index) + .match(/\/Filter\s*\/FlateDecode/); + + if (filterMatch) { + try { + const compressed = new Uint8Array( + [...streamData].map(c => c.charCodeAt(0)) + ); + const decompressed = this._inflateSync(compressed); + if (decompressed) { + content = new TextDecoder('latin1').decode(decompressed); + } + } catch { /* use raw */ } + } + + // Extract text from PDF operators + const extracted = this._extractTextFromOperators(content); + if (extracted) texts.push(extracted); + } + + return texts.join('\n').trim(); + } catch (e) { + console.warn('[Silent Send] PDF text extraction failed:', e); + return ''; + } + }, + + /** + * Extract text from PDF content stream operators. + * Handles Tj (show string), TJ (show array), ' and " (next line + show). + */ + _extractTextFromOperators(content) { + const parts = []; + + // Match string operands: (text) Tj, [(text) ...] TJ + // Tj operator: (string) Tj + const tjRegex = /\(([^)]*)\)\s*Tj/g; + let m; + while ((m = tjRegex.exec(content)) !== null) { + parts.push(this._decodePDFString(m[1])); + } + + // TJ operator: [(string) num (string) ...] TJ + const tjArrayRegex = /\[((?:[^[\]]*?))\]\s*TJ/g; + while ((m = tjArrayRegex.exec(content)) !== null) { + const inner = m[1]; + const stringRegex = /\(([^)]*)\)/g; + let s; + while ((s = stringRegex.exec(inner)) !== null) { + parts.push(this._decodePDFString(s[1])); + } + } + + // ' operator: (string) ' + const singleQuoteRegex = /\(([^)]*)\)\s*'/g; + while ((m = singleQuoteRegex.exec(content)) !== null) { + parts.push('\n' + this._decodePDFString(m[1])); + } + + return parts.join(''); + }, + + /** + * Decode PDF string escapes. + */ + _decodePDFString(str) { + return str + .replace(/\\n/g, '\n') + .replace(/\\r/g, '\r') + .replace(/\\t/g, '\t') + .replace(/\\\\/g, '\\') + .replace(/\\([()])/g, '$1') + .replace(/\\(\d{1,3})/g, (_, oct) => String.fromCharCode(parseInt(oct, 8))); + }, + + /** + * Simple DEFLATE decompression using DecompressionStream API. + * Returns null if not available or decompression fails. + */ + async _inflateSync(data) { + if (typeof DecompressionStream === 'undefined') return null; + try { + const ds = new DecompressionStream('deflate'); + const writer = ds.writable.getWriter(); + const reader = ds.readable.getReader(); + + writer.write(data); + writer.close(); + + const chunks = []; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + chunks.push(value); + } + + const totalLength = chunks.reduce((sum, c) => sum + c.length, 0); + const result = new Uint8Array(totalLength); + let offset = 0; + for (const chunk of chunks) { + result.set(chunk, offset); + offset += chunk.length; + } + return result; + } catch { + return null; + } + }, + + // ---------------------------------------------------------------- + // ZIP read/write (no external libraries) + // + // Minimal ZIP parser for DOCX/XLSX. These are standard ZIP files + // containing XML. We only need to read/write text entries. + // ---------------------------------------------------------------- + + async _readZip(file) { + const buffer = await file.arrayBuffer(); + const bytes = new Uint8Array(buffer); + const files = {}; + + // Find end of central directory record + let eocdOffset = -1; + for (let i = bytes.length - 22; i >= 0; i--) { + if (bytes[i] === 0x50 && bytes[i + 1] === 0x4b && + bytes[i + 2] === 0x05 && bytes[i + 3] === 0x06) { + eocdOffset = i; + break; + } + } + if (eocdOffset === -1) throw new Error('Not a valid ZIP file'); + + const view = new DataView(buffer); + const cdOffset = view.getUint32(eocdOffset + 16, true); + const cdEntries = view.getUint16(eocdOffset + 10, true); + + // Read central directory entries + let offset = cdOffset; + for (let i = 0; i < cdEntries; i++) { + if (view.getUint32(offset, true) !== 0x02014b50) break; + + const compMethod = view.getUint16(offset + 10, true); + const compSize = view.getUint32(offset + 20, true); + const uncompSize = view.getUint32(offset + 24, true); + const nameLen = view.getUint16(offset + 28, true); + const extraLen = view.getUint16(offset + 30, true); + const commentLen = view.getUint16(offset + 32, true); + const localHeaderOffset = view.getUint32(offset + 42, true); + + const name = new TextDecoder().decode(bytes.slice(offset + 46, offset + 46 + nameLen)); + + // Read from local file header + const localNameLen = view.getUint16(localHeaderOffset + 26, true); + const localExtraLen = view.getUint16(localHeaderOffset + 28, true); + const dataOffset = localHeaderOffset + 30 + localNameLen + localExtraLen; + const rawData = bytes.slice(dataOffset, dataOffset + compSize); + + files[name] = { + compMethod, + compSize, + uncompSize, + rawData, + async async(type) { + let data = this.rawData; + if (this.compMethod === 8) { + // Deflate compressed + data = await DocumentScanner._inflateSync(data); + if (!data) throw new Error('Decompression failed for ' + name); + } + if (type === 'string') { + return new TextDecoder().decode(data); + } + return data; + }, + }; + + offset += 46 + nameLen + extraLen + commentLen; + } + + return { files, _originalBuffer: buffer, _originalBytes: bytes }; + }, + + async _writeZip(zip) { + // Rebuild ZIP with modified entries + // For simplicity: store all modified entries uncompressed, + // copy unmodified entries as-is from the original buffer + const parts = []; + const centralDir = []; + let offset = 0; + + for (const [name, entry] of Object.entries(zip.files)) { + const nameBytes = new TextEncoder().encode(name); + let data; + + if (entry.isText && entry.data) { + // Modified entry — store uncompressed + data = new TextEncoder().encode(entry.data); + } else if (entry.rawData) { + // Unmodified — keep original compression + data = entry.rawData; + } else { + continue; + } + + const isStored = entry.isText || entry.compMethod === 0; + const compMethod = isStored ? 0 : entry.compMethod; + const compSize = data.length; + const uncompSize = isStored ? data.length : (entry.uncompSize || data.length); + + // Local file header + const localHeader = new Uint8Array(30 + nameBytes.length); + const lhView = new DataView(localHeader.buffer); + lhView.setUint32(0, 0x04034b50, true); // signature + lhView.setUint16(4, 20, true); // version needed + lhView.setUint16(8, compMethod, true); + lhView.setUint32(18, compSize, true); + lhView.setUint32(22, uncompSize, true); + lhView.setUint16(26, nameBytes.length, true); + localHeader.set(nameBytes, 30); + + // Central directory entry + const cdEntry = new Uint8Array(46 + nameBytes.length); + const cdView = new DataView(cdEntry.buffer); + cdView.setUint32(0, 0x02014b50, true); + cdView.setUint16(4, 20, true); // version made by + cdView.setUint16(6, 20, true); // version needed + cdView.setUint16(10, compMethod, true); + cdView.setUint32(20, compSize, true); + cdView.setUint32(24, uncompSize, true); + cdView.setUint16(28, nameBytes.length, true); + cdView.setUint32(42, offset, true); // local header offset + cdEntry.set(nameBytes, 46); + + centralDir.push(cdEntry); + parts.push(localHeader, data); + offset += localHeader.length + data.length; + } + + // End of central directory + const cdStart = offset; + let cdSize = 0; + for (const cd of centralDir) { + parts.push(cd); + cdSize += cd.length; + } + + const eocd = new Uint8Array(22); + const eocdView = new DataView(eocd.buffer); + eocdView.setUint32(0, 0x06054b50, true); + eocdView.setUint16(8, centralDir.length, true); + eocdView.setUint16(10, centralDir.length, true); + eocdView.setUint32(12, cdSize, true); + eocdView.setUint32(16, cdStart, true); + parts.push(eocd); + + return new Blob(parts, { type: 'application/octet-stream' }); + }, + + // ---------------------------------------------------------------- + // XML text substitution (for DOCX/XLSX) + // + // Finds text content within XML elements and applies substitution. + // Preserves all XML tags and attributes unchanged. + // ---------------------------------------------------------------- + + _substituteInXML(xml, substituteAll) { + const allReplacements = []; + + // Replace text between XML tags, preserving the tags themselves + // This handles text in DOCX and text in XLSX + const newXml = xml.replace(/>([^<]+) { + // Skip very short or whitespace-only content + if (!textContent || textContent.trim().length < 2) return match; + + const result = substituteAll(textContent); + if (result.modified) { + allReplacements.push(...result.replacements); + return '>' + result.text + '<'; + } + return match; + }); + + return { xml: newXml, replacements: allReplacements }; + }, + + // ---------------------------------------------------------------- + // Helpers + // ---------------------------------------------------------------- + + _detectType(ext, mimeType) { + // By extension + const textExts = new Set(['txt', 'csv', 'tsv', 'json', 'md', 'markdown', + 'log', 'yaml', 'yml', 'toml', 'ini', 'cfg', 'conf', 'xml', 'html', + 'htm', 'css', 'js', 'ts', 'py', 'rb', 'go', 'rs', 'java', 'c', 'cpp', + 'h', 'hpp', 'sh', 'bash', 'zsh', 'ps1', 'bat', 'sql', 'r', 'swift', + 'kt', 'scala', 'pl', 'php', 'lua', 'vim', 'env', 'gitignore']); + + if (ext === 'pdf') return 'pdf'; + if (ext === 'docx') return 'docx'; + if (ext === 'xlsx') return 'xlsx'; + if (textExts.has(ext)) return 'text'; + + // By MIME type + if (mimeType?.includes('pdf')) return 'pdf'; + if (mimeType?.includes('wordprocessingml')) return 'docx'; + if (mimeType?.includes('spreadsheetml')) return 'xlsx'; + if (mimeType?.startsWith('text/')) return 'text'; + if (mimeType?.includes('json') || mimeType?.includes('xml')) return 'text'; + + return 'unknown'; + }, +}; + +if (typeof globalThis !== 'undefined') { + globalThis.DocumentScanner = DocumentScanner; +} + +export default DocumentScanner;