如何修改JavaScript脚本以导出PDF中超链接对应的文本至Excel表格
Hey Damien, I see the issue—your current script pulls the link URLs from the PDF but misses the visible text that's linked. Let's fix that by matching each link's position to the text elements in the PDF, then update your Excel export to include that missing column.
The pdf.js-extract library gives us both the links (with their bounding box coordinates via rect) and all text items in the PDF. We can use those coordinates to find which text sits inside the link's area, which gives us the hyperlink display text.
Modified Full Code
const fs = require("fs"); const path = require("path"); const PDFExtract = require("pdf.js-extract").PDFExtract; const excelJS = require("exceljs"); const dirname = path.resolve(__dirname); const pdfExtract = new PDFExtract(); const options = { disableCombineTextItems: false, normalizeWhitespace: true, }; // Helper function to check if a text item is inside a link's rectangle function isTextInLinkRect(textItem, linkRect) { // linkRect format: [x1, y1, x2, y2] (coordinates of the link's bounding box) const [linkX1, linkY1, linkX2, linkY2] = linkRect; // textItem has x, y, width, height const textX1 = textItem.x; const textY1 = textItem.y; const textX2 = textItem.x + textItem.width; const textY2 = textItem.y + textItem.height; // Check if the text item overlaps sufficiently with the link rect return ( textX1 < linkX2 && textX2 > linkX1 && textY1 < linkY2 && textY2 > linkY1 ); } async function exportExcell(linkData, fileName, author) { const workbook = new excelJS.Workbook(); const worksheet = workbook.addWorksheet("Links Extract"); // Updated columns to match your required fields worksheet.columns = [ { header: "S no.", key: "s_no", width: 10 }, { header: "来源文件名", key: "fileName", width: 30 }, { header: "作者", key: "author", width: 30 }, { header: "超链接文本", key: "linkText", width: 50 }, { header: "实际链接地址", key: "linkUrl", width: 70 }, ]; // Add rows with all required data linkData.forEach((item, index) => { worksheet.addRow({ s_no: index + 1, fileName: index === 0 ? fileName : "", // Only show filename on first row author: index === 0 ? author : "", // Only show author on first row linkText: item.linkText, linkUrl: item.linkUrl, }); }); // Make header bold worksheet.getRow(1).eachCell((cell) => { cell.font = { bold: true }; }); try { const name = fileName.substring(0, fileName.length - 4); const fileWrite = path.join(dirname + `/files/output/${name}_extract_links.xlsx`); await workbook.xlsx.writeFile(fileWrite); console.log(`Successfully created excel file: ${fileWrite}`); } catch (err) { console.log(err.message); } } async function renderPdf(buffer, fileName) { try { const data = await pdfExtract.extractBuffer(buffer, options); const linkData = []; data.pages.forEach((page) => { if (page.links.length > 0) { page.links.forEach((link) => { // Find all text items that sit inside this link's rectangle const matchingTexts = page.items .filter(item => item.type === "text") .filter(textItem => isTextInLinkRect(textItem, link.rect)) .map(textItem => textItem.str) .join(" "); // Combine multiple text chunks into a single string // Avoid duplicate link-text pairs if (!linkData.some(item => item.linkUrl === link && item.linkText === matchingTexts)) { linkData.push({ linkText: matchingTexts || "No text found", linkUrl: link }); } }); } }); await exportExcell(linkData, fileName, data.meta.info.Author); } catch (err) { console.log(err); } } function readFileAsync() { return new Promise((resolve, reject) => { try { fs.readdirSync(path.join(dirname, "/files/uploads"), { withFileTypes: true, }) .filter( (file) => file.isFile() && file.name.toLowerCase().endsWith(".pdf") && !file.name.includes(".swp") ) .forEach(async (template) => { const fileName = template.name; const fileToRead = path.join(dirname + `/files/uploads/${fileName}`); const buffer = fs.readFileSync(fileToRead); // Fix async flow issue with sync read await renderPdf(buffer, fileName); }); resolve({ data: "Success reading files" }); } catch (e) { console.log("Error reading files:", e); reject(e); } }); } // Execute when user selects files const onFileSelected = async () => { await readFileAsync(); }; onFileSelected();
Key Changes Explained
- Added
isTextInLinkRectHelper: This function checks if a text item's bounding box overlaps with a link's rectangle, so we can match each link to its visible display text. - Updated Data Structure: Instead of just storing URLs, we now create objects with both
linkText(the visible hyperlink text) andlinkUrl(the actual address). - Excel Columns Aligned to Your Requirements: Added the missing「超链接文本」column, and renamed columns to match your requested Chinese headers for clarity.
- Duplicate Prevention: Added a check to avoid exporting identical link-text pairs multiple times.
- Fixed Async Flow: Switched to
fs.readFileSyncin the file loop to avoid callback hell and ensure proper execution order.
This script will now generate an Excel file with all four columns you need:「来源文件名」「作者」「超链接文本」「实际链接地址」.
内容的提问来源于stack exchange,提问作者Damien
相关产品推荐
相关产品推荐

