Node.js中如何调整表格解析逻辑实现预期结构化输出
表格解析结构修正方案
现有Node.js文档结构转换代码中,除表格模块外其余解析逻辑正常,但表格输出未将表头与表体集中归类到head和body节点,需修改代码实现预期的结构化格式。
修改后的完整代码
const parsers = new Map(); parsers._get = parsers.get; parsers.get = function (key) { if (this.has(key)) return this._get(key); return () => console.warn(`Parser not implemented for: ${JSON.stringify(key)}`); }; parsers.set("document", parseDocument); parsers.set("paragraph", parseParagraph); parsers.set("text", parseText); parsers.set("table", parseTable); parsers.set("tablerow", parseRow); parsers.set("tablehc", parseHC); parsers.set("tablec", parseTableC); function convert(obj) { return [parsers.get(obj.nodeType)(obj)]; } function parseDocument(obj) { let type = "doc"; let children = []; obj.content.forEach((e) => children.push(parsers.get(e.nodeType)(e))); return { type, children }; } function parseParagraph(obj) { let type = "p"; let children = []; obj.content.forEach((e) => children.push(parsers.get(e.nodeType)(e))); return { type, children }; } function parseText(obj) { const result = {}; result.text = obj.value; obj.marks.forEach((e) => (result[e.type] = true)); return result; } function parseTable(obj) { const type = "table"; const headRows = []; const bodyRows = []; // 区分表头行和表体行 obj.content.forEach(rowNode => { const isHeadRow = rowNode.content.some(cell => cell.nodeType === "tablehc"); const parsedRow = parsers.get(rowNode.nodeType)(rowNode); if (isHeadRow) { headRows.push(parsedRow); } else { bodyRows.push(parsedRow); } }); const children = []; // 生成head节点(如果有表头行) if (headRows.length > 0) { children.push({ type: "head", children: headRows }); } // 生成body节点(如果有表体行) if (bodyRows.length > 0) { children.push({ type: "body", children: bodyRows }); } return { type, children }; } function parseRow(obj) { const type = "tr"; const children = []; obj.content.forEach(cellNode => { children.push(parsers.get(cellNode.nodeType)(cellNode)); }); return { type, children }; } function parseHC(obj) { let type = "th"; let children = []; obj.content.forEach((e) => children.push(parsers.get(e.nodeType)(e))); return { type, children }; } function parseTableC(obj) { let type = "td"; let children = []; obj.content.forEach((e) => children.push(parsers.get(e.nodeType)(e))); return { type, children }; } let result = convert(getSrcData()); console.log("Converted object: ", JSON.stringify(result, null, 2)); function getSrcData() { return { nodeType: "document", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "dummy testing bold", marks: [ { type: "bold", }, ], }, ], }, { nodeType: "table", content: [ { nodeType: "tablerow", content: [ { nodeType: "tablehc", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "hey", marks: [], }, ], }, ], }, { nodeType: "tablehc", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "", marks: [], }, ], }, ], }, ], }, { nodeType: "tablerow", content: [ { nodeType: "tablec", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "code text", marks: [ { type: "code", }, ], }, ], }, ], }, ], }, { nodeType: "tablerow", content: [ { nodeType: "tablec", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "text", marks: [ { type: "bold", }, ], }, ], }, ], }, ], }, ], }, { nodeType: "paragraph", content: [ { nodeType: "text", value: "", marks: [], }, ], }, { nodeType: "table", content: [ { nodeType: "tablerow", content: [ { nodeType: "tablehc", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "hey", marks: [], }, ], }, ], }, { nodeType: "tablehc", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "", marks: [], }, ], }, ], }, { nodeType: "tablehc", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "", marks: [], }, ], }, ], }, ], }, { nodeType: "tablerow", content: [ { nodeType: "tablec", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "code text", marks: [ { type: "code", }, ], }, ], }, ], }, { nodeType: "tablec", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "", marks: [], }, ], }, ], }, ], }, { nodeType: "tablerow", content: [ { nodeType: "tablec", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "bold text", marks: [ { type: "bold", }, ], }, ], }, ], }, { nodeType: "tablec", content: [ { nodeType: "paragraph", content: [ { nodeType: "text", value: "", marks: [], }, ], }, ], }, ], }, ], }, ], }; }
关键修改说明
- 表格顶层分类逻辑:在
parseTable中直接判断每一行是表头行还是表体行,分别收集后统一生成head和body节点,确保所有表头行都集中在head下,表体行集中在body下 - 行解析简化:
parseRow不再处理表头/表体类型判断,只负责生成tr节点并解析内部单元格,逻辑更清晰 - 移除冗余函数:删除了原有的
parseTR函数,因为其功能与parseRow重复,且会导致节点嵌套错误 - 节点类型对齐:严格按照预期输出使用
head和body作为节点类型,替代原有的thead/tbody
内容的提问来源于stack exchange,提问作者Shiv
相关产品推荐
相关产品推荐

