0.3.5 共享功能修复
This commit is contained in:
@@ -0,0 +1,250 @@
|
||||
import JSZip from "jszip";
|
||||
|
||||
const MAX_TEXT_CHARS = 120_000;
|
||||
|
||||
function clampText(input: string, maxChars = MAX_TEXT_CHARS): string {
|
||||
const trimmed = String(input || "").replace(/\s+\n/g, "\n").trim();
|
||||
if (!trimmed) return "";
|
||||
if (trimmed.length <= maxChars) return trimmed;
|
||||
return `${trimmed.slice(0, maxChars)}…`;
|
||||
}
|
||||
|
||||
function decodeXmlEntities(input: string): string {
|
||||
return input
|
||||
.replace(/</g, "<")
|
||||
.replace(/>/g, ">")
|
||||
.replace(/&/g, "&")
|
||||
.replace(/"/g, '"')
|
||||
.replace(/'/g, "'")
|
||||
.replace(/&#x([0-9a-fA-F]+);/g, (_, hex) => {
|
||||
const code = Number.parseInt(hex, 16);
|
||||
if (!Number.isFinite(code)) return "";
|
||||
return String.fromCodePoint(code);
|
||||
})
|
||||
.replace(/&#(\d+);/g, (_, dec) => {
|
||||
const code = Number.parseInt(dec, 10);
|
||||
if (!Number.isFinite(code)) return "";
|
||||
return String.fromCodePoint(code);
|
||||
});
|
||||
}
|
||||
|
||||
function getExtension(fileName: string | null | undefined): string {
|
||||
const raw = String(fileName ?? "").trim().toLowerCase();
|
||||
const idx = raw.lastIndexOf(".");
|
||||
if (idx === -1) return "";
|
||||
return raw.slice(idx + 1).replace(/[^a-z0-9]+/g, "");
|
||||
}
|
||||
|
||||
function detectKind(args: { mimeType?: string | null; fileName?: string | null }): "pdf" | "docx" | "pptx" | "xlsx" | null {
|
||||
const mime = String(args.mimeType ?? "").toLowerCase().trim();
|
||||
if (mime === "application/pdf") return "pdf";
|
||||
if (mime === "application/vnd.openxmlformats-officedocument.wordprocessingml.document") return "docx";
|
||||
if (mime === "application/vnd.openxmlformats-officedocument.presentationml.presentation") return "pptx";
|
||||
if (mime === "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet") return "xlsx";
|
||||
|
||||
const ext = getExtension(args.fileName ?? null);
|
||||
if (ext === "pdf") return "pdf";
|
||||
if (ext === "docx") return "docx";
|
||||
if (ext === "pptx") return "pptx";
|
||||
if (ext === "xlsx") return "xlsx";
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractTextFromXmlByTag(xml: string, tagName: string): string {
|
||||
const out: string[] = [];
|
||||
const re = new RegExp(`<${tagName}[^>]*>([\\s\\S]*?)<\\/${tagName}>`, "g");
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = re.exec(xml))) {
|
||||
const raw = m[1] ?? "";
|
||||
const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim();
|
||||
if (clean) out.push(clean);
|
||||
}
|
||||
return out.join(" ").trim();
|
||||
}
|
||||
|
||||
function extractDocxText(documentXml: string): string {
|
||||
const paras = documentXml.split(/<w:p[\s>]/g);
|
||||
const out: string[] = [];
|
||||
for (const p of paras) {
|
||||
const line: string[] = [];
|
||||
const re = /<w:t[^>]*>([\s\S]*?)<\/w:t>/g;
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = re.exec(p))) {
|
||||
const raw = m[1] ?? "";
|
||||
const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim();
|
||||
if (clean) line.push(clean);
|
||||
}
|
||||
const joined = line.join("").trim();
|
||||
if (joined) out.push(joined);
|
||||
}
|
||||
return out.join("\n").trim();
|
||||
}
|
||||
|
||||
function extractPptxText(slideXmls: string[]): string {
|
||||
const out: string[] = [];
|
||||
for (const xml of slideXmls) {
|
||||
const line = extractTextFromXmlByTag(xml, "a:t");
|
||||
if (line) out.push(line);
|
||||
}
|
||||
return out.join("\n\n").trim();
|
||||
}
|
||||
|
||||
function extractXlsxText(args: { sharedStringsXml: string | null; sheetXmls: string[] }): string {
|
||||
const sharedStrings: string[] = [];
|
||||
if (args.sharedStringsXml) {
|
||||
const re = /<t[^>]*>([\s\S]*?)<\/t>/g;
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = re.exec(args.sharedStringsXml))) {
|
||||
const raw = m[1] ?? "";
|
||||
const clean = decodeXmlEntities(raw).replace(/\s+/g, " ").trim();
|
||||
if (clean) sharedStrings.push(clean);
|
||||
}
|
||||
}
|
||||
|
||||
const out: string[] = [];
|
||||
for (const xml of args.sheetXmls) {
|
||||
const rows = xml.split(/<row[\s>]/g);
|
||||
for (const r of rows) {
|
||||
const cells: string[] = [];
|
||||
|
||||
// t="s" => sharedStrings index;t="inlineStr" => <is><t>...;默认 => <v>number
|
||||
const cellRe = /<c\b[^>]*?(?:t="([^"]+)")?[^>]*>([\s\S]*?)<\/c>/g;
|
||||
let cm: RegExpExecArray | null;
|
||||
while ((cm = cellRe.exec(r))) {
|
||||
const t = (cm[1] ?? "").trim();
|
||||
const body = cm[2] ?? "";
|
||||
if (t === "inlineStr") {
|
||||
const inline = extractTextFromXmlByTag(body, "t");
|
||||
if (inline) cells.push(inline);
|
||||
continue;
|
||||
}
|
||||
const vMatch = /<v>([\s\S]*?)<\/v>/.exec(body);
|
||||
if (!vMatch) continue;
|
||||
const rawV = decodeXmlEntities(String(vMatch[1] ?? "")).trim();
|
||||
if (!rawV) continue;
|
||||
if (t === "s") {
|
||||
const idx = Number.parseInt(rawV, 10);
|
||||
const s = Number.isFinite(idx) ? (sharedStrings[idx] ?? "") : "";
|
||||
if (s) cells.push(s);
|
||||
} else {
|
||||
cells.push(rawV);
|
||||
}
|
||||
}
|
||||
|
||||
const line = cells.join(" ").replace(/\s+/g, " ").trim();
|
||||
if (line) out.push(line);
|
||||
}
|
||||
}
|
||||
return out.join("\n").trim();
|
||||
}
|
||||
|
||||
export type AttachmentExtractOk = {
|
||||
ok: true;
|
||||
strategy: string;
|
||||
text: string;
|
||||
meta?: Record<string, unknown>;
|
||||
};
|
||||
export type AttachmentExtractResult =
|
||||
| AttachmentExtractOk
|
||||
| { ok: false; strategy: string; reason: string; meta?: Record<string, unknown> };
|
||||
|
||||
export async function extractTextFromAttachment(args: {
|
||||
mimeType: string | null;
|
||||
fileName: string | null;
|
||||
bytes: ArrayBuffer;
|
||||
}): Promise<AttachmentExtractResult> {
|
||||
const kind = detectKind({ mimeType: args.mimeType, fileName: args.fileName });
|
||||
if (!kind) {
|
||||
return { ok: false, strategy: "unsupported", reason: "暂不支持该附件类型" };
|
||||
}
|
||||
|
||||
if (kind === "pdf") {
|
||||
try {
|
||||
const mod = await import("pdfjs-dist/legacy/build/pdf.mjs");
|
||||
const data = new Uint8Array(args.bytes);
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const loadingTask = (mod as any).getDocument({ data });
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const pdf = await (loadingTask as any).promise;
|
||||
const out: string[] = [];
|
||||
const pages = Number(pdf?.numPages ?? 0) || 0;
|
||||
for (let i = 1; i <= pages; i += 1) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const page = await (pdf as any).getPage(i);
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const content = await (page as any).getTextContent();
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const items = Array.isArray((content as any)?.items) ? (content as any).items : [];
|
||||
const line = items
|
||||
.map((it: any) => (typeof it?.str === "string" ? it.str : ""))
|
||||
.join(" ")
|
||||
.replace(/\s+/g, " ")
|
||||
.trim();
|
||||
if (line) out.push(line);
|
||||
}
|
||||
const text = clampText(out.join("\n\n"));
|
||||
if (!text) {
|
||||
return { ok: false, strategy: "pdfjs", reason: "未提取到可用文本", meta: { pages } };
|
||||
}
|
||||
return { ok: true, strategy: "pdfjs", text, meta: { pages } };
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
return { ok: false, strategy: "pdfjs", reason: message };
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const zip = await JSZip.loadAsync(args.bytes);
|
||||
if (kind === "docx") {
|
||||
const file = zip.file("word/document.xml");
|
||||
const xml = file ? await file.async("string") : "";
|
||||
const text = clampText(extractDocxText(xml));
|
||||
if (!text) return { ok: false, strategy: "docx.xml", reason: "未提取到可用文本" };
|
||||
return { ok: true, strategy: "docx.xml", text };
|
||||
}
|
||||
|
||||
if (kind === "pptx") {
|
||||
const slideFiles = Object.keys(zip.files)
|
||||
.filter((p) => /^ppt\/slides\/slide\d+\.xml$/i.test(p))
|
||||
.sort((a, b) => a.localeCompare(b, undefined, { numeric: true }));
|
||||
const slideXmls: string[] = [];
|
||||
for (const p of slideFiles) {
|
||||
const f = zip.file(p);
|
||||
if (!f) continue;
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
slideXmls.push(await f.async("string"));
|
||||
}
|
||||
const text = clampText(extractPptxText(slideXmls));
|
||||
if (!text) return { ok: false, strategy: "pptx.xml", reason: "未提取到可用文本" };
|
||||
return { ok: true, strategy: "pptx.xml", text, meta: { slides: slideXmls.length } };
|
||||
}
|
||||
|
||||
if (kind === "xlsx") {
|
||||
const sharedStringsXml = await (async () => {
|
||||
const f = zip.file("xl/sharedStrings.xml");
|
||||
if (!f) return null;
|
||||
return await f.async("string");
|
||||
})();
|
||||
|
||||
const sheetFiles = Object.keys(zip.files)
|
||||
.filter((p) => /^xl\/worksheets\/sheet\d+\.xml$/i.test(p))
|
||||
.sort((a, b) => a.localeCompare(b, undefined, { numeric: true }));
|
||||
const sheetXmls: string[] = [];
|
||||
for (const p of sheetFiles) {
|
||||
const f = zip.file(p);
|
||||
if (!f) continue;
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
sheetXmls.push(await f.async("string"));
|
||||
}
|
||||
const text = clampText(extractXlsxText({ sharedStringsXml, sheetXmls }));
|
||||
if (!text) return { ok: false, strategy: "xlsx.xml", reason: "未提取到可用文本" };
|
||||
return { ok: true, strategy: "xlsx.xml", text, meta: { sheets: sheetXmls.length } };
|
||||
}
|
||||
|
||||
return { ok: false, strategy: "unsupported", reason: "暂不支持该附件类型" };
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
return { ok: false, strategy: "zip", reason: message };
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user