import { Database } from 'bun:sqlite'; import { join } from 'node:path'; import { chmodSync } from 'node:fs'; import { DATA_PATH } from '@@/data-path'; import type { EmailSummary } from 'types'; const SCHEMA_TABLES = ` CREATE TABLE IF NOT EXISTS emails ( id TEXT PRIMARY KEY, integration TEXT NOT NULL DEFAULT 'gmail', email_account TEXT NOT NULL DEFAULT '', from_name TEXT, from_address TEXT, from_domain TEXT, to_address TEXT, cc TEXT, subject TEXT, date TEXT, snippet TEXT, html TEXT, text_body TEXT, attachment_count INTEGER DEFAULT 0, read INTEGER DEFAULT 0, deleted INTEGER DEFAULT 0, labels TEXT ); CREATE TABLE IF NOT EXISTS attachments ( id INTEGER PRIMARY KEY AUTOINCREMENT, email_id TEXT REFERENCES emails(id) ON DELETE CASCADE, idx INTEGER, filename TEXT, size INTEGER, content_type TEXT, content TEXT ); CREATE TABLE IF NOT EXISTS sync_meta ( key TEXT PRIMARY KEY, value TEXT ); `; const SCHEMA_INDEXES = ` CREATE INDEX IF NOT EXISTS idx_emails_date ON emails(date); CREATE INDEX IF NOT EXISTS idx_emails_from_domain ON emails(from_domain); CREATE INDEX IF NOT EXISTS idx_emails_from_address ON emails(from_address); CREATE INDEX IF NOT EXISTS idx_emails_integration ON emails(integration); CREATE INDEX IF NOT EXISTS idx_emails_email_account ON emails(email_account); CREATE INDEX IF NOT EXISTS idx_emails_labels ON emails(labels); `; /** Convert label IDs to lowercase comma-separated string for storage */ function labelsToString(labels?: string[]): string | null { if (!labels || labels.length === 0) return null; return labels.map((l) => l.toLowerCase()).join(','); } /** Convert stored comma-separated labels back to array */ function labelsFromString(value: unknown): string[] | undefined { if (typeof value !== 'string' || !value) return undefined; return value.split(','); } function extractAddress(headerValue: string): { name: string; address: string } { const match = headerValue.match(/^"?(.+?)"?\s*<(.+?)>$/); if (match) return { name: match[1]!.trim(), address: match[2]!.toLowerCase() }; const bare = headerValue.trim().toLowerCase(); return { name: '', address: bare }; } function extractDomain(address: string): string { const at = address.lastIndexOf('@'); return at >= 0 ? address.slice(at + 1) : ''; } export function openEmailDb(email: string): Database { const dbPath = join(DATA_PATH, email, 'emails.db'); const db = new Database(dbPath, { create: true }); db.exec('PRAGMA journal_mode = DELETE'); db.exec('PRAGMA foreign_keys = ON'); db.exec(SCHEMA_TABLES); migrate(db); db.exec(SCHEMA_INDEXES); // Try to chmod, but don't crash if permission denied (e.g., file owned by different user) try { chmodSync(dbPath, 0o666); } catch (err) { // File exists with correct permissions, or owned by another user - that's fine } return db; } function migrate(db: Database): void { const cols = db.query('PRAGMA table_info(emails)').all() as Array<{ name: string }>; const colNames = new Set(cols.map((c) => c.name)); if (!colNames.has('deleted')) { db.exec('ALTER TABLE emails ADD COLUMN deleted INTEGER DEFAULT 0'); } if (!colNames.has('integration')) { db.exec("ALTER TABLE emails ADD COLUMN integration TEXT NOT NULL DEFAULT 'gmail'"); } if (!colNames.has('email_account')) { db.exec("ALTER TABLE emails ADD COLUMN email_account TEXT NOT NULL DEFAULT ''"); } if (!colNames.has('labels')) { db.exec('ALTER TABLE emails ADD COLUMN labels TEXT'); } // Ensure sync_meta table exists (for DBs created before it was added to SCHEMA_TABLES) db.exec('CREATE TABLE IF NOT EXISTS sync_meta (key TEXT PRIMARY KEY, value TEXT)'); // Add content column to attachments if missing const attCols = db.query('PRAGMA table_info(attachments)').all() as Array<{ name: string }>; const attColNames = new Set(attCols.map((c) => c.name)); if (!attColNames.has('content')) { db.exec('ALTER TABLE attachments ADD COLUMN content TEXT'); } } type ParsedEmail = { id: string; integration: string; emailAccount: string; fromName: string; fromAddress: string; to: string; cc?: string; subject: string; date: string; snippet: string; html?: string; text?: string; attachments: Array<{ filename: string; size: number; contentType: string; content: string }>; labels?: string[]; }; const upsertEmailStmt = ` INSERT OR REPLACE INTO emails (id, integration, email_account, from_name, from_address, from_domain, to_address, cc, subject, date, snippet, html, text_body, attachment_count, labels) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) `; const deleteAttachmentsStmt = 'DELETE FROM attachments WHERE email_id = ?'; const insertAttachmentStmt = 'INSERT INTO attachments (email_id, idx, filename, size, content_type, content) VALUES (?, ?, ?, ?, ?, ?)'; export function upsertEmail(db: Database, email: ParsedEmail): void { const domain = extractDomain(email.fromAddress); db.exec('BEGIN'); try { db.run(upsertEmailStmt, [ email.id, email.integration, email.emailAccount, email.fromName, email.fromAddress, domain, email.to, email.cc ?? null, email.subject, email.date, email.snippet, email.html ?? null, email.text ?? null, email.attachments.length, labelsToString(email.labels), ]); db.run(deleteAttachmentsStmt, [email.id]); for (let i = 0; i < email.attachments.length; i++) { const att = email.attachments[i]!; db.run(insertAttachmentStmt, [email.id, i, att.filename, att.size, att.contentType, att.content]); } db.exec('COMMIT'); } catch (err) { db.exec('ROLLBACK'); throw err; } } /** Upsert a single email from its raw RFC822 text using fast header parsing. */ type UpsertFromRawEmlParams = { db: Database; id: string; raw: string; integration: string; emailAccount: string; labels?: string[]; }; export function upsertFromRawEml({ db, id, raw, integration, emailAccount, labels }: UpsertFromRawEmlParams): void { const from = extractHeader(raw, 'From'); const { name, address } = extractAddress(from); const to = extractHeader(raw, 'To'); const cc = extractHeader(raw, 'Cc') || null; const subject = extractHeader(raw, 'Subject') || '(no subject)'; const dateStr = extractHeader(raw, 'Date'); const date = dateStr ? new Date(dateStr).toISOString() : new Date(0).toISOString(); const snippet = extractSnippet(raw); const attachments = parseAttachments(raw); const domain = extractDomain(address); const { html, text } = extractBody(raw); db.run(upsertEmailStmt, [ id, integration, emailAccount, name, address, domain, to, cc, subject, date, snippet, html, text, attachments.length, labelsToString(labels), ]); if (attachments.length > 0) { db.run(deleteAttachmentsStmt, [id]); for (let i = 0; i < attachments.length; i++) { const att = attachments[i]!; db.run(insertAttachmentStmt, [id, i, att.filename, att.size, att.contentType, att.content]); } } } /** Convert a db row to an EmailSummary for the API */ export function rowToSummary(row: Record): EmailSummary { const from = row.from_name ? `${row.from_name} <${row.from_address}>` : (row.from_address as string); const labels = labelsFromString(row.labels); return { id: row.id as string, from, to: row.to_address as string, subject: row.subject as string, date: row.date as string, snippet: row.snippet as string, ...(row.attachment_count ? { attachmentCount: row.attachment_count as number } : {}), ...(row.read ? { read: true } : {}), ...(labels ? { labels } : {}), }; } // ── Sync meta helpers ── export function getSyncMeta(db: Database, key: string): string | null { const row = db.query('SELECT value FROM sync_meta WHERE key = ?').get(key) as { value: string } | null; return row?.value ?? null; } export function setSyncMeta(db: Database, key: string, value: string): void { db.run('INSERT OR REPLACE INTO sync_meta (key, value) VALUES (?, ?)', [key, value]); } export function updateEmailLabels(db: Database, id: string, labels: string[]): void { db.run('UPDATE emails SET labels = ? WHERE id = ?', [labelsToString(labels), id]); } // ── Header parsing helpers (same logic as gmail-sync) ── function decodeMimeWords(text: string): string { return text.replace(/=\?([^?]+)\?(B|Q)\?([^?]+)\?=/gi, (_, charset, encoding, encoded) => { try { const normalizedCs = normalizeCharset(charset.toLowerCase()); if (encoding.toUpperCase() === 'B') { const buf = Buffer.from(encoded, 'base64'); return new TextDecoder(normalizedCs, { fatal: false }).decode(buf); } const bytes: number[] = []; for (let i = 0; i < encoded.length; i++) { if (encoded[i] === '_') { bytes.push(0x20); } else if (encoded[i] === '=' && i + 2 < encoded.length) { bytes.push(parseInt(encoded.slice(i + 1, i + 3), 16)); i += 2; } else { bytes.push(encoded.charCodeAt(i)); } } return new TextDecoder(normalizedCs, { fatal: false }).decode(Buffer.from(bytes)); } catch { return encoded; } }); } function extractHeader(raw: string, name: string): string { const match = raw.match(new RegExp(`^${name}:\\s*(.+)$`, 'mi')); return match?.[1]?.trim() ? decodeMimeWords(match[1].trim()) : ''; } /** Extract a header value including folded continuation lines (lines starting with whitespace) */ function extractFullHeader(raw: string, name: string): string { const headerEnd = findHeaderEnd(raw); const headerBlock = headerEnd !== -1 ? raw.slice(0, headerEnd) : raw.slice(0, 4096); const lines = headerBlock.split(/\r?\n/); let result = ''; let capturing = false; for (const line of lines) { if (new RegExp(`^${name}:\\s*`, 'i').test(line)) { result = line.replace(new RegExp(`^${name}:\\s*`, 'i'), ''); capturing = true; } else if (capturing && /^[\t ]/.test(line)) { result += ' ' + line.trim(); } else if (capturing) { break; } } return result.trim(); } function findHeaderEnd(text: string): number { const crlf = text.indexOf('\r\n\r\n'); const lf = text.indexOf('\n\n'); if (crlf !== -1) return crlf + 4; if (lf !== -1) return lf + 2; return -1; } function extractSnippet(raw: string): string { const idx = findHeaderEnd(raw); if (idx === -1) return ''; let body = raw.slice(idx); if (body.trimStart().startsWith('--')) { const afterBoundary = body.slice(body.indexOf('\n') + 1); const partBodyStart = findHeaderEnd(afterBoundary); if (partBodyStart !== -1) body = afterBoundary.slice(partBodyStart); } const nextBoundary = body.indexOf('\n--'); if (nextBoundary !== -1) body = body.slice(0, nextBoundary); return body.replace(/\s+/g, ' ').trim().slice(0, 120); } function decodeQuotedPrintableBytes(text: string): Buffer { const cleaned = text.replace(/=\r?\n/g, ''); const bytes: number[] = []; for (let i = 0; i < cleaned.length; i++) { if (cleaned[i] === '=' && i + 2 < cleaned.length) { const hex = cleaned.slice(i + 1, i + 3); const val = parseInt(hex, 16); if (!isNaN(val)) { bytes.push(val); i += 2; continue; } } bytes.push(cleaned.charCodeAt(i)); } return Buffer.from(bytes); } function extractCharset(contentType: string): string { const match = contentType.match(/charset=["']?([^"';\s]+)/i); return match?.[1]?.toLowerCase() ?? 'utf-8'; } function normalizeCharset(charset: string): string { const map: Record = { 'iso-8859-1': 'latin1', 'iso_8859-1': 'latin1', 'iso-8859-15': 'latin1', 'iso_8859-15': 'latin1', 'windows-1250': 'latin1', 'windows-1251': 'latin1', 'windows-1252': 'latin1', 'windows-1254': 'latin1', 'us-ascii': 'ascii', 'ascii': 'ascii', }; return map[charset] ?? charset; } function decodePartBody(body: string, encoding: string, charset = 'utf-8'): string { const enc = encoding.toLowerCase(); const normalizedCharset = normalizeCharset(charset); if (enc === 'base64') { const buf = Buffer.from(body.replace(/\s/g, ''), 'base64'); return new TextDecoder(normalizedCharset, { fatal: false }).decode(buf); } if (enc === 'quoted-printable') { const buf = decodeQuotedPrintableBytes(body); return new TextDecoder(normalizedCharset, { fatal: false }).decode(buf); } return body; } function extractBody(raw: string): { html: string | null; text: string | null } { const headerEnd = findHeaderEnd(raw); if (headerEnd === -1) return { html: null, text: null }; const topCtRaw = extractFullHeader(raw, 'Content-Type'); const topCt = topCtRaw.toLowerCase(); const topEncoding = extractFullHeader(raw, 'Content-Transfer-Encoding'); // Non-multipart: single body if (!topCt.includes('multipart')) { const body = raw.slice(headerEnd); const charset = extractCharset(topCtRaw); const decoded = decodePartBody(body, topEncoding, charset); if (topCt.includes('text/html')) return { html: decoded, text: null }; return { html: null, text: decoded }; } // Multipart: extract boundary from the raw (case-sensitive) header const boundaryMatch = topCtRaw.match(/boundary=["']?([^"';\s]+)/i); if (!boundaryMatch) return { html: null, text: null }; const boundary = boundaryMatch[1]!; let html: string | null = null; let text: string | null = null; const parts = raw.slice(headerEnd).split(`--${boundary}`); for (const part of parts) { if (part.startsWith('--') || !part.trim()) continue; const partHeaderEnd = findHeaderEnd(part); if (partHeaderEnd === -1) continue; const partCtRaw = extractFullHeader(part, 'Content-Type'); const partCt = partCtRaw.toLowerCase(); const partEnc = extractFullHeader(part, 'Content-Transfer-Encoding'); const partCharset = extractCharset(partCtRaw); const partBody = part.slice(partHeaderEnd); // Recurse into nested multipart (e.g. multipart/alternative inside multipart/mixed) if (partCt.includes('multipart')) { const nested = extractBody(part.trim()); if (nested.html && !html) html = nested.html; if (nested.text && !text) text = nested.text; continue; } if (partCt.includes('text/html') && !html) { html = decodePartBody(partBody, partEnc, partCharset); } else if (partCt.includes('text/plain') && !text) { text = decodePartBody(partBody, partEnc, partCharset); } } return { html, text }; } type AttachmentMeta = { filename: string; size: number; contentType: string; content: string }; function parseAttachments(raw: string): AttachmentMeta[] { const results: AttachmentMeta[] = []; // Match both "attachment" and "inline" dispositions const regex = /^Content-Disposition:\s*(?:attachment|inline)[^\n]*/gim; let match: RegExpExecArray | null; while ((match = regex.exec(raw)) !== null) { const pos = match.index; // Walk backwards to find the start of this MIME part's headers const partStart = raw.lastIndexOf('\n--', pos); const headerBlock = partStart !== -1 ? raw.slice(partStart, pos + 500) : raw.slice(Math.max(0, pos - 500), pos + 500); // Skip inline parts without a filename (e.g. inline text/plain body parts) const hasFilename = /filename/i.test(headerBlock); if (!hasFilename) continue; // Extract filename from Content-Disposition or Content-Type const fnMatch = headerBlock.match(/filename\*?=(?:"([^"]+)"|([^\s;]+))/i); let rawFilename = fnMatch ? (fnMatch[1] ?? fnMatch[2] ?? 'unknown') : 'unknown'; // RFC 5987: filename*=charset''percent-encoded const rfc5987Match = rawFilename.match(/^([^']*)'[^']*'(.+)/); if (rfc5987Match) { const cs = normalizeCharset(rfc5987Match[1]!.toLowerCase() || 'utf-8'); const encoded = rfc5987Match[2]!; const bytes = encoded.replace(/%([0-9A-Fa-f]{2})/g, (_, h: string) => String.fromCharCode(parseInt(h, 16))); rawFilename = new TextDecoder(cs, { fatal: false }).decode(Buffer.from(bytes, 'binary')); } const filename = decodeMimeWords(rawFilename); // Extract content-type const ctMatch = headerBlock.match(/^Content-Type:\s*([^\s;]+)/im); const contentType = ctMatch?.[1] ?? 'application/octet-stream'; // Extract full body content as base64 const partHeaderEnd = findHeaderEnd(raw.slice(pos)); let content = ''; let size = 0; if (partHeaderEnd !== -1) { const bodyStart = pos + partHeaderEnd; const boundaryEnd = raw.indexOf('\n--', bodyStart); const bodyRaw = boundaryEnd !== -1 ? raw.slice(bodyStart, boundaryEnd) : raw.slice(bodyStart); // Detect encoding from part headers const encMatch = headerBlock.match(/^Content-Transfer-Encoding:\s*(\S+)/im); const encoding = encMatch?.[1]?.toLowerCase() ?? 'base64'; if (encoding === 'base64') { content = bodyRaw.replace(/\s/g, ''); } else { // For quoted-printable or 7bit/8bit, re-encode to base64 const buf = encoding === 'quoted-printable' ? decodeQuotedPrintableBytes(bodyRaw) : Buffer.from(bodyRaw); content = buf.toString('base64'); } size = Math.floor(content.length * 3 / 4); } results.push({ filename, size, contentType, content }); } return results; }