← GPT-AI-LP
CODE · 2.5 KB

src/document-reader.js

Workspace snapshot · 08/31 23:51

import * as pdfjs from "pdfjs-dist/legacy/build/pdf.mjs";
import pdfWorker from "pdfjs-dist/legacy/build/pdf.worker.min.mjs?url";
import mammoth from "mammoth/mammoth.browser";
import JSZip from "jszip";
import { normalizeDocumentText } from "./document-text.js";
pdfjs.GlobalWorkerOptions.workerSrc = pdfWorker;

async function readPdf(file) {
  const pdf = await pdfjs.getDocument({ data: new Uint8Array(await file.arrayBuffer()) }).promise;
  const pages = [];
  for (let number = 1; number <= Math.min(pdf.numPages, 80); number += 1) {
    const page = await pdf.getPage(number);
    const content = await page.getTextContent();
    pages.push(content.items.map((item) => item.str).join(" "));
  }
  return pages.join("\n\n");
}

async function readPptx(file) {
  const zip = await JSZip.loadAsync(await file.arrayBuffer());
  const names = Object.keys(zip.files)
    .filter((name) => /^ppt\/slides\/slide\d+\.xml$/u.test(name))
    .sort((a, b) => Number(a.match(/\d+/u)?.[0]) - Number(b.match(/\d+/u)?.[0]));
  const slides = [];
  for (const name of names.slice(0, 100)) {
    const xml = await zip.file(name)?.async("string");
    slides.push([...String(xml).matchAll(/<a:t>([\s\S]*?)<\/a:t>/gu)]
      .map((match) => match[1].replaceAll("&amp;", "&").replaceAll("&lt;", "<").replaceAll("&gt;", ">"))
      .join(" "));
  }
  return slides.join("\n\n");
}

export async function readDocument(file) {
  if (!file) throw new Error("資料を選択してください");
  if (file.size > 15 * 1024 * 1024) throw new Error("資料は15MB以下にしてください");
  const ext = file.name.split(".").pop()?.toLowerCase();
  let rawText = "";
  if (["txt", "md", "csv"].includes(ext)) rawText = await file.text();
  else if (ext === "pdf") rawText = await readPdf(file);
  else if (ext === "docx") rawText = (await mammoth.extractRawText({ arrayBuffer: await file.arrayBuffer() })).value;
  else if (ext === "pptx") rawText = await readPptx(file);
  else throw new Error("PDF・PPTX・DOCX・TXT・MD・CSVに対応しています");

  const result = normalizeDocumentText(rawText);
  if (!result.text) throw new Error("資料から文章を抽出できませんでした");
  const format = String(ext || "資料").toUpperCase();
  return {
    ...result,
    status: result.truncated
      ? `${format}から${result.originalLength.toLocaleString("ja-JP")}文字を抽出し、生成上限の60,000文字まで読み込みました`
      : `${format}から${result.text.length.toLocaleString("ja-JP")}文字を読み込みました`,
  };
}