import JSZip from "jszip"; const MAX_TEXT_CHARS = 120_000; function clampText(input: string, maxChars = MAX_TEXT_CHARS): string { const trimmed = String(input || "").replace(/\s+\n/g, "\n").trim(); if (!trimmed) return ""; if (trimmed.length <= maxChars) return trimmed; return `${trimmed.slice(0, maxChars)}…`; } function decodeXmlEntities(input: string): string { return input .replace(/</g, "<") .replace(/>/g, ">") .replace(/&/g, "&") .replace(/"/g, '"') .replace(/'/g, "'") .replace(/&#x([0-9a-fA-F]+);/g, (_, hex) => { const code = Number.parseInt(hex, 16); if (!Number.isFinite(code)) return ""; return String.fromCodePoint(code); }) .replace(/&#(\d+);/g, (_, dec) => { const code = Number.parseInt(dec, 10); if (!Number.isFinite(code)) return ""; return String.fromCodePoint(code); }); } function getExtension(fileName: string | null | undefined): string { const raw = String(fileName ?? "").trim().toLowerCase(); const idx = raw.lastIndexOf("."); if (idx === -1) return ""; return raw.slice(idx + 1).replace(/[^a-z0-9]+/g, ""); } function detectKind(args: { mimeType?: string | null; fileName?: string | null }): "pdf" | "docx" | "pptx" | "xlsx" | null { const mime = String(args.mimeType ?? "").toLowerCase().trim(); if (mime === "application/pdf") return "pdf"; if (mime === "application/vnd.openxmlformats-officedocument.wordprocessingml.document") return "docx"; if (mime === "application/vnd.openxmlformats-officedocument.presentationml.presentation") return "pptx"; if (mime === "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet") return "xlsx"; const ext = getExtension(args.fileName ?? null); if (ext === "pdf") return "pdf"; if (ext === "docx") return "docx"; if (ext === "pptx") return "pptx"; if (ext === "xlsx") return "xlsx"; return null; } function extractTextFromXmlByTag(xml: string, tagName: string): string { const out: string[] = []; const re = new RegExp(`<${tagName}[^>]*>([\\s\\S]*?)<\\/${tagName}>`, "g"); let m: RegExpExecArray | null; while ((m = re.exec(xml))) { const raw = m[1] ?? ""; const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim(); if (clean) out.push(clean); } return out.join(" ").trim(); } function extractDocxText(documentXml: string): string { const paras = documentXml.split(/]/g); const out: string[] = []; for (const p of paras) { const line: string[] = []; const re = /]*>([\s\S]*?)<\/w:t>/g; let m: RegExpExecArray | null; while ((m = re.exec(p))) { const raw = m[1] ?? ""; const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim(); if (clean) line.push(clean); } const joined = line.join("").trim(); if (joined) out.push(joined); } return out.join("\n").trim(); } function extractPptxText(slideXmls: string[]): string { const out: string[] = []; for (const xml of slideXmls) { const line = extractTextFromXmlByTag(xml, "a:t"); if (line) out.push(line); } return out.join("\n\n").trim(); } function extractXlsxText(args: { sharedStringsXml: string | null; sheetXmls: string[] }): string { const sharedStrings: string[] = []; if (args.sharedStringsXml) { const re = /]*>([\s\S]*?)<\/t>/g; let m: RegExpExecArray | null; while ((m = re.exec(args.sharedStringsXml))) { const raw = m[1] ?? ""; const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim(); if (clean) sharedStrings.push(clean); } } const out: string[] = []; for (const xml of args.sheetXmls) { const rows = xml.split(/]/g); for (const r of rows) { const cells: string[] = []; // t="s" => sharedStrings index;t="inlineStr" => ...;默认 => number const cellRe = /]*?(?:t="([^"]+)")?[^>]*>([\s\S]*?)<\/c>/g; let cm: RegExpExecArray | null; while ((cm = cellRe.exec(r))) { const t = (cm[1] ?? "").trim(); const body = cm[2] ?? ""; if (t === "inlineStr") { const inline = extractTextFromXmlByTag(body, "t"); if (inline) cells.push(inline); continue; } const vMatch = /([\s\S]*?)<\/v>/.exec(body); if (!vMatch) continue; const rawV = decodeXmlEntities(String(vMatch[1] ?? "")).trim(); if (!rawV) continue; if (t === "s") { const idx = Number.parseInt(rawV, 10); const s = Number.isFinite(idx) ? (sharedStrings[idx] ?? "") : ""; if (s) cells.push(s); } else { cells.push(rawV); } } const line = cells.join(" ").replace(/\s+/g, " ").trim(); if (line) out.push(line); } } return out.join("\n").trim(); } export type AttachmentExtractOk = { ok: true; strategy: string; text: string; meta?: Record; }; export type AttachmentExtractResult = | AttachmentExtractOk | { ok: false; strategy: string; reason: string; meta?: Record }; export async function extractTextFromAttachment(args: { mimeType: string | null; fileName: string | null; bytes: ArrayBuffer; }): Promise { const kind = detectKind({ mimeType: args.mimeType, fileName: args.fileName }); if (!kind) { return { ok: false, strategy: "unsupported", reason: "暂不支持该附件类型" }; } if (kind === "pdf") { try { const mod = await import("pdfjs-dist/legacy/build/pdf.mjs"); const data = new Uint8Array(args.bytes); // eslint-disable-next-line @typescript-eslint/no-explicit-any const loadingTask = (mod as any).getDocument({ data }); // eslint-disable-next-line @typescript-eslint/no-explicit-any const pdf = await (loadingTask as any).promise; const out: string[] = []; const pages = Number(pdf?.numPages ?? 0) || 0; for (let i = 1; i <= pages; i += 1) { // eslint-disable-next-line @typescript-eslint/no-explicit-any const page = await (pdf as any).getPage(i); // eslint-disable-next-line @typescript-eslint/no-explicit-any const content = await (page as any).getTextContent(); // eslint-disable-next-line @typescript-eslint/no-explicit-any const items = Array.isArray((content as any)?.items) ? (content as any).items : []; const line = items .map((it: any) => (typeof it?.str === "string" ? it.str : "")) .join(" ") .replace(/\s+/g, " ") .trim(); if (line) out.push(line); } const text = clampText(out.join("\n\n")); if (!text) { return { ok: false, strategy: "pdfjs", reason: "未提取到可用文本", meta: { pages } }; } return { ok: true, strategy: "pdfjs", text, meta: { pages } }; } catch (err) { const message = err instanceof Error ? err.message : String(err); return { ok: false, strategy: "pdfjs", reason: message }; } } try { const zip = await JSZip.loadAsync(args.bytes); if (kind === "docx") { const file = zip.file("word/document.xml"); const xml = file ? await file.async("string") : ""; const text = clampText(extractDocxText(xml)); if (!text) return { ok: false, strategy: "docx.xml", reason: "未提取到可用文本" }; return { ok: true, strategy: "docx.xml", text }; } if (kind === "pptx") { const slideFiles = Object.keys(zip.files) .filter((p) => /^ppt\/slides\/slide\d+\.xml$/i.test(p)) .sort((a, b) => a.localeCompare(b, undefined, { numeric: true })); const slideXmls: string[] = []; for (const p of slideFiles) { const f = zip.file(p); if (!f) continue; // eslint-disable-next-line no-await-in-loop slideXmls.push(await f.async("string")); } const text = clampText(extractPptxText(slideXmls)); if (!text) return { ok: false, strategy: "pptx.xml", reason: "未提取到可用文本" }; return { ok: true, strategy: "pptx.xml", text, meta: { slides: slideXmls.length } }; } if (kind === "xlsx") { const sharedStringsXml = await (async () => { const f = zip.file("xl/sharedStrings.xml"); if (!f) return null; return await f.async("string"); })(); const sheetFiles = Object.keys(zip.files) .filter((p) => /^xl\/worksheets\/sheet\d+\.xml$/i.test(p)) .sort((a, b) => a.localeCompare(b, undefined, { numeric: true })); const sheetXmls: string[] = []; for (const p of sheetFiles) { const f = zip.file(p); if (!f) continue; // eslint-disable-next-line no-await-in-loop sheetXmls.push(await f.async("string")); } const text = clampText(extractXlsxText({ sharedStringsXml, sheetXmls })); if (!text) return { ok: false, strategy: "xlsx.xml", reason: "未提取到可用文本" }; return { ok: true, strategy: "xlsx.xml", text, meta: { sheets: sheetXmls.length } }; } return { ok: false, strategy: "unsupported", reason: "暂不支持该附件类型" }; } catch (err) { const message = err instanceof Error ? err.message : String(err); return { ok: false, strategy: "zip", reason: message }; } }