251 lines
9.1 KiB
TypeScript
251 lines
9.1 KiB
TypeScript
import JSZip from "jszip";
|
||
|
||
const MAX_TEXT_CHARS = 120_000;
|
||
|
||
function clampText(input: string, maxChars = MAX_TEXT_CHARS): string {
|
||
const trimmed = String(input || "").replace(/\s+\n/g, "\n").trim();
|
||
if (!trimmed) return "";
|
||
if (trimmed.length <= maxChars) return trimmed;
|
||
return `${trimmed.slice(0, maxChars)}…`;
|
||
}
|
||
|
||
function decodeXmlEntities(input: string): string {
|
||
return input
|
||
.replace(/</g, "<")
|
||
.replace(/>/g, ">")
|
||
.replace(/&/g, "&")
|
||
.replace(/"/g, '"')
|
||
.replace(/'/g, "'")
|
||
.replace(/&#x([0-9a-fA-F]+);/g, (_, hex) => {
|
||
const code = Number.parseInt(hex, 16);
|
||
if (!Number.isFinite(code)) return "";
|
||
return String.fromCodePoint(code);
|
||
})
|
||
.replace(/&#(\d+);/g, (_, dec) => {
|
||
const code = Number.parseInt(dec, 10);
|
||
if (!Number.isFinite(code)) return "";
|
||
return String.fromCodePoint(code);
|
||
});
|
||
}
|
||
|
||
function getExtension(fileName: string | null | undefined): string {
|
||
const raw = String(fileName ?? "").trim().toLowerCase();
|
||
const idx = raw.lastIndexOf(".");
|
||
if (idx === -1) return "";
|
||
return raw.slice(idx + 1).replace(/[^a-z0-9]+/g, "");
|
||
}
|
||
|
||
function detectKind(args: { mimeType?: string | null; fileName?: string | null }): "pdf" | "docx" | "pptx" | "xlsx" | null {
|
||
const mime = String(args.mimeType ?? "").toLowerCase().trim();
|
||
if (mime === "application/pdf") return "pdf";
|
||
if (mime === "application/vnd.openxmlformats-officedocument.wordprocessingml.document") return "docx";
|
||
if (mime === "application/vnd.openxmlformats-officedocument.presentationml.presentation") return "pptx";
|
||
if (mime === "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet") return "xlsx";
|
||
|
||
const ext = getExtension(args.fileName ?? null);
|
||
if (ext === "pdf") return "pdf";
|
||
if (ext === "docx") return "docx";
|
||
if (ext === "pptx") return "pptx";
|
||
if (ext === "xlsx") return "xlsx";
|
||
return null;
|
||
}
|
||
|
||
function extractTextFromXmlByTag(xml: string, tagName: string): string {
|
||
const out: string[] = [];
|
||
const re = new RegExp(`<${tagName}[^>]*>([\\s\\S]*?)<\\/${tagName}>`, "g");
|
||
let m: RegExpExecArray | null;
|
||
while ((m = re.exec(xml))) {
|
||
const raw = m[1] ?? "";
|
||
const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim();
|
||
if (clean) out.push(clean);
|
||
}
|
||
return out.join(" ").trim();
|
||
}
|
||
|
||
function extractDocxText(documentXml: string): string {
|
||
const paras = documentXml.split(/<w:p[\s>]/g);
|
||
const out: string[] = [];
|
||
for (const p of paras) {
|
||
const line: string[] = [];
|
||
const re = /<w:t[^>]*>([\s\S]*?)<\/w:t>/g;
|
||
let m: RegExpExecArray | null;
|
||
while ((m = re.exec(p))) {
|
||
const raw = m[1] ?? "";
|
||
const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim();
|
||
if (clean) line.push(clean);
|
||
}
|
||
const joined = line.join("").trim();
|
||
if (joined) out.push(joined);
|
||
}
|
||
return out.join("\n").trim();
|
||
}
|
||
|
||
function extractPptxText(slideXmls: string[]): string {
|
||
const out: string[] = [];
|
||
for (const xml of slideXmls) {
|
||
const line = extractTextFromXmlByTag(xml, "a:t");
|
||
if (line) out.push(line);
|
||
}
|
||
return out.join("\n\n").trim();
|
||
}
|
||
|
||
function extractXlsxText(args: { sharedStringsXml: string | null; sheetXmls: string[] }): string {
|
||
const sharedStrings: string[] = [];
|
||
if (args.sharedStringsXml) {
|
||
const re = /<t[^>]*>([\s\S]*?)<\/t>/g;
|
||
let m: RegExpExecArray | null;
|
||
while ((m = re.exec(args.sharedStringsXml))) {
|
||
const raw = m[1] ?? "";
|
||
const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim();
|
||
if (clean) sharedStrings.push(clean);
|
||
}
|
||
}
|
||
|
||
const out: string[] = [];
|
||
for (const xml of args.sheetXmls) {
|
||
const rows = xml.split(/<row[\s>]/g);
|
||
for (const r of rows) {
|
||
const cells: string[] = [];
|
||
|
||
// t="s" => sharedStrings index;t="inlineStr" => <is><t>...;默认 => <v>number
|
||
const cellRe = /<c\b[^>]*?(?:t="([^"]+)")?[^>]*>([\s\S]*?)<\/c>/g;
|
||
let cm: RegExpExecArray | null;
|
||
while ((cm = cellRe.exec(r))) {
|
||
const t = (cm[1] ?? "").trim();
|
||
const body = cm[2] ?? "";
|
||
if (t === "inlineStr") {
|
||
const inline = extractTextFromXmlByTag(body, "t");
|
||
if (inline) cells.push(inline);
|
||
continue;
|
||
}
|
||
const vMatch = /<v>([\s\S]*?)<\/v>/.exec(body);
|
||
if (!vMatch) continue;
|
||
const rawV = decodeXmlEntities(String(vMatch[1] ?? "")).trim();
|
||
if (!rawV) continue;
|
||
if (t === "s") {
|
||
const idx = Number.parseInt(rawV, 10);
|
||
const s = Number.isFinite(idx) ? (sharedStrings[idx] ?? "") : "";
|
||
if (s) cells.push(s);
|
||
} else {
|
||
cells.push(rawV);
|
||
}
|
||
}
|
||
|
||
const line = cells.join(" ").replace(/\s+/g, " ").trim();
|
||
if (line) out.push(line);
|
||
}
|
||
}
|
||
return out.join("\n").trim();
|
||
}
|
||
|
||
export type AttachmentExtractOk = {
|
||
ok: true;
|
||
strategy: string;
|
||
text: string;
|
||
meta?: Record<string, unknown>;
|
||
};
|
||
export type AttachmentExtractResult =
|
||
| AttachmentExtractOk
|
||
| { ok: false; strategy: string; reason: string; meta?: Record<string, unknown> };
|
||
|
||
export async function extractTextFromAttachment(args: {
|
||
mimeType: string | null;
|
||
fileName: string | null;
|
||
bytes: ArrayBuffer;
|
||
}): Promise<AttachmentExtractResult> {
|
||
const kind = detectKind({ mimeType: args.mimeType, fileName: args.fileName });
|
||
if (!kind) {
|
||
return { ok: false, strategy: "unsupported", reason: "暂不支持该附件类型" };
|
||
}
|
||
|
||
if (kind === "pdf") {
|
||
try {
|
||
const mod = await import("pdfjs-dist/legacy/build/pdf.mjs");
|
||
const data = new Uint8Array(args.bytes);
|
||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||
const loadingTask = (mod as any).getDocument({ data });
|
||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||
const pdf = await (loadingTask as any).promise;
|
||
const out: string[] = [];
|
||
const pages = Number(pdf?.numPages ?? 0) || 0;
|
||
for (let i = 1; i <= pages; i += 1) {
|
||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||
const page = await (pdf as any).getPage(i);
|
||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||
const content = await (page as any).getTextContent();
|
||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||
const items = Array.isArray((content as any)?.items) ? (content as any).items : [];
|
||
const line = items
|
||
.map((it: any) => (typeof it?.str === "string" ? it.str : ""))
|
||
.join(" ")
|
||
.replace(/\s+/g, " ")
|
||
.trim();
|
||
if (line) out.push(line);
|
||
}
|
||
const text = clampText(out.join("\n\n"));
|
||
if (!text) {
|
||
return { ok: false, strategy: "pdfjs", reason: "未提取到可用文本", meta: { pages } };
|
||
}
|
||
return { ok: true, strategy: "pdfjs", text, meta: { pages } };
|
||
} catch (err) {
|
||
const message = err instanceof Error ? err.message : String(err);
|
||
return { ok: false, strategy: "pdfjs", reason: message };
|
||
}
|
||
}
|
||
|
||
try {
|
||
const zip = await JSZip.loadAsync(args.bytes);
|
||
if (kind === "docx") {
|
||
const file = zip.file("word/document.xml");
|
||
const xml = file ? await file.async("string") : "";
|
||
const text = clampText(extractDocxText(xml));
|
||
if (!text) return { ok: false, strategy: "docx.xml", reason: "未提取到可用文本" };
|
||
return { ok: true, strategy: "docx.xml", text };
|
||
}
|
||
|
||
if (kind === "pptx") {
|
||
const slideFiles = Object.keys(zip.files)
|
||
.filter((p) => /^ppt\/slides\/slide\d+\.xml$/i.test(p))
|
||
.sort((a, b) => a.localeCompare(b, undefined, { numeric: true }));
|
||
const slideXmls: string[] = [];
|
||
for (const p of slideFiles) {
|
||
const f = zip.file(p);
|
||
if (!f) continue;
|
||
// eslint-disable-next-line no-await-in-loop
|
||
slideXmls.push(await f.async("string"));
|
||
}
|
||
const text = clampText(extractPptxText(slideXmls));
|
||
if (!text) return { ok: false, strategy: "pptx.xml", reason: "未提取到可用文本" };
|
||
return { ok: true, strategy: "pptx.xml", text, meta: { slides: slideXmls.length } };
|
||
}
|
||
|
||
if (kind === "xlsx") {
|
||
const sharedStringsXml = await (async () => {
|
||
const f = zip.file("xl/sharedStrings.xml");
|
||
if (!f) return null;
|
||
return await f.async("string");
|
||
})();
|
||
|
||
const sheetFiles = Object.keys(zip.files)
|
||
.filter((p) => /^xl\/worksheets\/sheet\d+\.xml$/i.test(p))
|
||
.sort((a, b) => a.localeCompare(b, undefined, { numeric: true }));
|
||
const sheetXmls: string[] = [];
|
||
for (const p of sheetFiles) {
|
||
const f = zip.file(p);
|
||
if (!f) continue;
|
||
// eslint-disable-next-line no-await-in-loop
|
||
sheetXmls.push(await f.async("string"));
|
||
}
|
||
const text = clampText(extractXlsxText({ sharedStringsXml, sheetXmls }));
|
||
if (!text) return { ok: false, strategy: "xlsx.xml", reason: "未提取到可用文本" };
|
||
return { ok: true, strategy: "xlsx.xml", text, meta: { sheets: sheetXmls.length } };
|
||
}
|
||
|
||
return { ok: false, strategy: "unsupported", reason: "暂不支持该附件类型" };
|
||
} catch (err) {
|
||
const message = err instanceof Error ? err.message : String(err);
|
||
return { ok: false, strategy: "zip", reason: message };
|
||
}
|
||
}
|
||
|