You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何修改JavaScript脚本以导出PDF中超链接对应的文本至Excel表格

Hey Damien, I see the issue—your current script pulls the link URLs from the PDF but misses the visible text that's linked. Let's fix that by matching each link's position to the text elements in the PDF, then update your Excel export to include that missing column.

The pdf.js-extract library gives us both the links (with their bounding box coordinates via rect) and all text items in the PDF. We can use those coordinates to find which text sits inside the link's area, which gives us the hyperlink display text.

Modified Full Code

const fs = require("fs");
const path = require("path");
const PDFExtract = require("pdf.js-extract").PDFExtract;
const excelJS = require("exceljs");
const dirname = path.resolve(__dirname);
const pdfExtract = new PDFExtract();
const options = {
  disableCombineTextItems: false,
  normalizeWhitespace: true,
};

// Helper function to check if a text item is inside a link's rectangle
function isTextInLinkRect(textItem, linkRect) {
  // linkRect format: [x1, y1, x2, y2] (coordinates of the link's bounding box)
  const [linkX1, linkY1, linkX2, linkY2] = linkRect;
  // textItem has x, y, width, height
  const textX1 = textItem.x;
  const textY1 = textItem.y;
  const textX2 = textItem.x + textItem.width;
  const textY2 = textItem.y + textItem.height;

  // Check if the text item overlaps sufficiently with the link rect
  return (
    textX1 < linkX2 &&
    textX2 > linkX1 &&
    textY1 < linkY2 &&
    textY2 > linkY1
  );
}

async function exportExcell(linkData, fileName, author) {
  const workbook = new excelJS.Workbook();
  const worksheet = workbook.addWorksheet("Links Extract");

  // Updated columns to match your required fields
  worksheet.columns = [
    { header: "S no.", key: "s_no", width: 10 },
    { header: "来源文件名", key: "fileName", width: 30 },
    { header: "作者", key: "author", width: 30 },
    { header: "超链接文本", key: "linkText", width: 50 },
    { header: "实际链接地址", key: "linkUrl", width: 70 },
  ];

  // Add rows with all required data
  linkData.forEach((item, index) => {
    worksheet.addRow({
      s_no: index + 1,
      fileName: index === 0 ? fileName : "", // Only show filename on first row
      author: index === 0 ? author : "", // Only show author on first row
      linkText: item.linkText,
      linkUrl: item.linkUrl,
    });
  });

  // Make header bold
  worksheet.getRow(1).eachCell((cell) => {
    cell.font = { bold: true };
  });

  try {
    const name = fileName.substring(0, fileName.length - 4);
    const fileWrite = path.join(dirname + `/files/output/${name}_extract_links.xlsx`);
    await workbook.xlsx.writeFile(fileWrite);
    console.log(`Successfully created excel file: ${fileWrite}`);
  } catch (err) {
    console.log(err.message);
  }
}

async function renderPdf(buffer, fileName) {
  try {
    const data = await pdfExtract.extractBuffer(buffer, options);
    
    const linkData = [];
    data.pages.forEach((page) => {
      if (page.links.length > 0) {
        page.links.forEach((link) => {
          // Find all text items that sit inside this link's rectangle
          const matchingTexts = page.items
            .filter(item => item.type === "text")
            .filter(textItem => isTextInLinkRect(textItem, link.rect))
            .map(textItem => textItem.str)
            .join(" "); // Combine multiple text chunks into a single string

          // Avoid duplicate link-text pairs
          if (!linkData.some(item => item.linkUrl === link && item.linkText === matchingTexts)) {
            linkData.push({
              linkText: matchingTexts || "No text found",
              linkUrl: link
            });
          }
        });
      }
    });

    await exportExcell(linkData, fileName, data.meta.info.Author);
  } catch (err) {
    console.log(err);
  }
}

function readFileAsync() {
  return new Promise((resolve, reject) => {
    try {
      fs.readdirSync(path.join(dirname, "/files/uploads"), {
        withFileTypes: true,
      })
      .filter(
        (file) =>
          file.isFile() && 
          file.name.toLowerCase().endsWith(".pdf") && 
          !file.name.includes(".swp")
      )
      .forEach(async (template) => {
        const fileName = template.name;
        const fileToRead = path.join(dirname + `/files/uploads/${fileName}`);
        const buffer = fs.readFileSync(fileToRead); // Fix async flow issue with sync read
        await renderPdf(buffer, fileName);
      });
      resolve({ data: "Success reading files" });
    } catch (e) {
      console.log("Error reading files:", e);
      reject(e);
    }
  });
}

// Execute when user selects files
const onFileSelected = async () => {
  await readFileAsync();
};

onFileSelected();

Key Changes Explained

  1. Added isTextInLinkRect Helper: This function checks if a text item's bounding box overlaps with a link's rectangle, so we can match each link to its visible display text.
  2. Updated Data Structure: Instead of just storing URLs, we now create objects with both linkText (the visible hyperlink text) and linkUrl (the actual address).
  3. Excel Columns Aligned to Your Requirements: Added the missing「超链接文本」column, and renamed columns to match your requested Chinese headers for clarity.
  4. Duplicate Prevention: Added a check to avoid exporting identical link-text pairs multiple times.
  5. Fixed Async Flow: Switched to fs.readFileSync in the file loop to avoid callback hell and ensure proper execution order.

This script will now generate an Excel file with all four columns you need:「来源文件名」「作者」「超链接文本」「实际链接地址」.

内容的提问来源于stack exchange,提问作者Damien

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.27 15:47:40