← GPT-AI-SLIDE
CODE · 2.2 KB

src/document-reader.js

Workspace snapshot · 09/04 13:45

import * as pdfjs from "pdfjs-dist/legacy/build/pdf.mjs";
import pdfWorker from "pdfjs-dist/legacy/build/pdf.worker.min.mjs?url";
import mammoth from "mammoth/mammoth.browser";
import JSZip from "jszip";
pdfjs.GlobalWorkerOptions.workerSrc = pdfWorker;
const compact = (value) => String(value || "").replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f]/gu, " ").replace(/\s+/gu, " ").trim().slice(0, 12000);
async function readPdf(file) { const pdf = await pdfjs.getDocument({ data: new Uint8Array(await file.arrayBuffer()) }).promise; const pages = []; for (let number = 1; number <= Math.min(pdf.numPages, 40); number += 1) { const page = await pdf.getPage(number); const content = await page.getTextContent(); pages.push(content.items.map((item) => item.str).join(" ")); } return compact(pages.join("\n")); }
async function readPptx(file) { const zip = await JSZip.loadAsync(await file.arrayBuffer()); const names = Object.keys(zip.files).filter((name) => /^ppt\/slides\/slide\d+\.xml$/u.test(name)).sort((a, b) => Number(a.match(/\d+/u)?.[0]) - Number(b.match(/\d+/u)?.[0])); const pages = []; for (const name of names.slice(0, 60)) { const xml = await zip.file(name)?.async("string"); pages.push([...String(xml).matchAll(/<a:t>([\s\S]*?)<\/a:t>/gu)].map((match) => match[1].replaceAll("&amp;", "&").replaceAll("&lt;", "<").replaceAll("&gt;", ">")).join(" ")); } return compact(pages.join("\n")); }
export async function readDocument(file) { if (!file) throw new Error("資料を選択してください"); if (file.size > 15 * 1024 * 1024) throw new Error("資料は15MB以下にしてください"); const ext = file.name.split(".").pop()?.toLowerCase(); let text = ""; if (["txt", "md", "csv"].includes(ext)) text = compact(await file.text()); else if (ext === "pdf") text = await readPdf(file); else if (ext === "docx") text = compact((await mammoth.extractRawText({ arrayBuffer: await file.arrayBuffer() })).value); else if (ext === "pptx") text = await readPptx(file); else throw new Error("PDF・PPTX・DOCX・TXT・MDに対応しています"); if (!text) throw new Error("資料から文章を抽出できませんでした"); return { text, status: `${ext.toUpperCase()}から${text.length.toLocaleString("ja-JP")}文字を読み込みました` }; }