如何在Node.js中跨操作系统转换文档、图片、文本为PDF?
Got it, let's solve this cross-platform PDF conversion challenge in Node.js. The node-msoffice-pdf tool’s limitation makes total sense—it relies on Microsoft Office, which is only fully supported on Windows (and has quirks on macOS). So we need platform-agnostic tools that work everywhere. Here’s how to handle each file type:
The best cross-platform solution here is LibreOffice/OpenOffice, which has a headless command-line interface we can call from Node.js. It supports most Office formats (docx, xlsx, pptx, etc.) and runs on Windows, macOS, and Linux.
Step 1: Install LibreOffice on all systems
- Windows: Download from the official site, ensure
soffice.exeis added to your system PATH - macOS: Run
brew install libreofficevia Homebrew - Linux (Debian/Ubuntu): Run
sudo apt install libreoffice
Step 2: Node.js Code
const { exec } = require('child_process'); const path = require('path'); function convertOfficeToPdf(inputPath, outputDir) { return new Promise((resolve, reject) => { const resolvedInput = path.resolve(inputPath); const resolvedOutput = path.resolve(outputDir); // Cross-platform LibreOffice command const cmd = `soffice --headless --convert-to pdf --outdir "${resolvedOutput}" "${resolvedInput}"`; exec(cmd, (error, stdout, stderr) => { if (error) { reject(new Error(`Conversion failed: ${stderr || error.message}`)); return; } // LibreOffice uses the input filename with a .pdf extension by default const pdfPath = path.join(resolvedOutput, path.basename(resolvedInput, path.extname(resolvedInput)) + '.pdf'); resolve(pdfPath); }); }); } // Example usage convertOfficeToPdf('./annual-report.docx', './output') .then(pdfPath => console.log(`PDF saved to: ${pdfPath}`)) .catch(err => console.error(err));
Use sharp—a fast, cross-platform image processing library built on libvips. It handles single-image-to-PDF conversion seamlessly.
Step 1: Install Sharp
npm install sharp
Step 2: Node.js Code
const sharp = require('sharp'); const path = require('path'); function convertImageToPdf(inputPath, outputPath) { return sharp(inputPath) .pdf() .toFile(outputPath) .then(() => outputPath) .catch(err => Promise.reject(new Error(`Image conversion failed: ${err.message}`))); } // Example usage convertImageToPdf('./product-photo.png', './product-photo.pdf') .then(pdfPath => console.log(`PDF saved to: ${pdfPath}`)) .catch(err => console.error(err));
Note: For multiple images, convert each to a PDF page then merge them using a library like pdf-lib.
Pure Text (TXT)
Use pdf-lib to create a PDF from scratch and render text directly.
Step 1: Install pdf-lib
npm install pdf-lib
Step 2: Node.js Code
const { PDFDocument, StandardFonts, rgb } = require('pdf-lib'); const fs = require('fs').promises; async function convertTextToPdf(inputPath, outputPath) { const textContent = await fs.readFile(inputPath, 'utf8'); const pdfDoc = await PDFDocument.create(); const page = pdfDoc.addPage([600, 800]); // A4-like dimensions in points // Load a standard font const helveticaFont = await pdfDoc.embedFont(StandardFonts.Helvetica); page.setFont(helveticaFont); page.setFontSize(12); page.setTextColor(rgb(0, 0, 0)); // Render text with line breaks const lines = textContent.split('\n'); let yPosition = 750; // Start near the top of the page const lineHeight = 15; for (const line of lines) { if (yPosition < 50) { // Add new page if we hit the bottom const newPage = pdfDoc.addPage([600, 800]); newPage.setFont(helveticaFont); newPage.setFontSize(12); newPage.setTextColor(rgb(0, 0, 0)); yPosition = 750; } page.drawText(line, { x: 50, y: yPosition }); yPosition -= lineHeight; } const pdfBytes = await pdfDoc.save(); await fs.writeFile(outputPath, pdfBytes); return outputPath; } // Example usage convertTextToPdf('./notes.txt', './notes.pdf') .then(pdfPath => console.log(`PDF saved to: ${pdfPath}`)) .catch(err => console.error(err));
Markdown (MD)
First convert Markdown to HTML with marked, then render the HTML to PDF using Puppeteer (headless Chrome), which ensures consistent styling across platforms.
Step 1: Install Dependencies
npm install marked puppeteer
Step 2: Node.js Code
const marked = require('marked'); const puppeteer = require('puppeteer'); const fs = require('fs').promises; async function convertMarkdownToPdf(inputPath, outputPath) { const mdContent = await fs.readFile(inputPath, 'utf8'); const htmlContent = marked.parse(mdContent); // Wrap HTML in a styled template for better PDF output const fullHtml = ` <!DOCTYPE html> <html> <head> <style> body { font-family: Arial, sans-serif; margin: 2rem; line-height: 1.6; } h1, h2, h3 { color: #2d3748; } code { background: #f7fafc; padding: 0.2rem; border-radius: 0.2rem; } pre { background: #f7fafc; padding: 1rem; border-radius: 0.5rem; overflow-x: auto; } </style> </head> <body>${htmlContent}</body> </html> `; const browser = await puppeteer.launch({ headless: 'new' }); const page = await browser.newPage(); await page.setContent(fullHtml); await page.pdf({ path: outputPath, format: 'A4', margin: { top: '1cm', right: '1cm', bottom: '1cm', left: '1cm' } }); await browser.close(); return outputPath; } // Example usage convertMarkdownToPdf('./readme.md', './readme.pdf') .then(pdfPath => console.log(`PDF saved to: ${pdfPath}`)) .catch(err => console.error(err));
You can wrap all these functions into a single helper to auto-detect file types:
const path = require('path'); async function convertFileToPdf(inputPath, outputPath) { const ext = path.extname(inputPath).toLowerCase(); switch (ext) { case '.docx': case '.doc': case '.xlsx': case '.xls': case '.pptx': case '.ppt': return convertOfficeToPdf(inputPath, path.dirname(outputPath)); case '.jpg': case '.jpeg': case '.png': case '.gif': case '.bmp': return convertImageToPdf(inputPath, outputPath); case '.txt': return convertTextToPdf(inputPath, outputPath); case '.md': return convertMarkdownToPdf(inputPath, outputPath); default: throw new Error(`Unsupported file type: ${ext}`); } }
内容的提问来源于stack exchange,提问作者Hiten

