You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何批量从本地HTML文件提取数据生成对应JSON文件?

批量提取本地HTML文件数据并生成JSON

方案一:静态HTML提取(用Cheerio,轻量快速)

如果你的HTML文件都是静态内容(无JavaScript动态渲染的元素),用Cheerio足够满足需求。它能模拟浏览器DOM操作,API和jQuery类似,学习成本低。

步骤1:准备环境

  1. 下载安装Node.js(官网对应系统版本即可,安装后终端输入node -v能看到版本号即成功)
  2. 在root文件夹下打开终端,执行npm init -y生成package.json文件
  3. 安装依赖:npm install cheerio fs-extra

步骤2:编写提取脚本

在root文件夹下新建extract-data.js文件,内容如下:

const fs = require('fs-extra');
const cheerio = require('cheerio');

// 定义提取数据的函数,对应你浏览器里的getPostData
function getPostData($) {
    const res = {};
    // 用Cheerio语法替代document方法,$即模拟的document
    res.post_title = $('[name="post_title"]').val();
    // 继续添加你需要提取的其他字段,示例:
    // res.post_content = $('.post-content').text().trim();
    return res;
}

async function processAllFiles() {
    // 获取root下的所有子文件夹(1、2、3...)
    const folders = await fs.readdir('./', { withFileTypes: true });
    for (const folder of folders) {
        if (!folder.isDirectory()) continue; // 跳过非文件夹内容
        // 拼接post.html路径
        const htmlPath = `./${folder.name}/post/post.html`;
        // 检查文件是否存在
        if (!(await fs.pathExists(htmlPath))) continue;
        
        // 读取HTML内容
        const htmlContent = await fs.readFile(htmlPath, 'utf8');
        // 加载HTML到Cheerio
        const $ = cheerio.load(htmlContent);
        // 提取数据
        const postData = getPostData($);
        // 拼接JSON文件路径
        const jsonPath = `./${folder.name}/post/post.json`;
        // 写入JSON文件,格式化输出方便阅读
        await fs.writeJson(jsonPath, postData, { spaces: 2 });
        console.log(`已处理:${folder.name}/post/post.json`);
    }
}

// 执行脚本
processAllFiles().catch(err => console.error('处理出错:', err));

步骤3:运行脚本

终端执行node extract-data.js,脚本会自动遍历所有文件夹,提取数据并生成对应的post.json。


方案二:动态HTML提取(用Puppeteer,模拟真实浏览器)

如果HTML里有JavaScript动态渲染的内容(比如页面加载后才生成的元素),Cheerio无法提取这类内容,此时需要用Puppeteer模拟无头浏览器(无可视化窗口的Chrome)处理。

步骤1:准备环境

  1. 同样需先安装Node.js
  2. 在root文件夹下执行npm init -y
  3. 安装依赖:npm install puppeteer fs-extra

步骤2:编写提取脚本

在root文件夹下新建extract-data-dynamic.js文件,内容如下:

const fs = require('fs-extra');
const puppeteer = require('puppeteer');

// 直接复用你浏览器里的getPostData函数,几乎无需修改
const getPostData = () => {
    let res = {};
    var a = document.getElementsByName('post_title')[0]?.value;
    res['post_title'] = a;
    // 继续添加你的其他字段提取逻辑
    return res;
};

async function processAllFiles() {
    // 启动无头浏览器
    const browser = await puppeteer.launch({ headless: 'new' });
    // 获取root下的所有子文件夹
    const folders = await fs.readdir('./', { withFileTypes: true });
    
    for (const folder of folders) {
        if (!folder.isDirectory()) continue;
        const htmlPath = `./${folder.name}/post/post.html`;
        if (!(await fs.pathExists(htmlPath))) continue;

        // 打开新页面
        const page = await browser.newPage();
        // 加载本地HTML文件(转成file协议)
        await page.goto(`file://${process.cwd()}/${htmlPath}`);
        
        // 在浏览器环境执行getPostData函数,获取结果
        const postData = await page.evaluate(getPostData);
        
        // 写入JSON文件
        const jsonPath = `./${folder.name}/post/post.json`;
        await fs.writeJson(jsonPath, postData, { spaces: 2 });
        console.log(`已处理:${folder.name}/post/post.json`);
        
        // 关闭当前页面
        await page.close();
    }
    
    // 关闭浏览器
    await browser.close();
}

processAllFiles().catch(err => console.error('处理出错:', err));

步骤3:运行脚本

终端执行node extract-data-dynamic.js即可,脚本会自动模拟浏览器加载每个HTML,执行提取函数并生成JSON。


内容的提问来源于stack exchange,提问作者Momo Hinamori

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.18 00:14:58