在Node.js环境下如何提取Markdown内容的首个h1标题与第一段正文?
实现方案
Node.js环境下无需调用浏览器API,可通过以下两种方案实现需求:
方案一:使用成熟Markdown解析库(推荐)
该方案可规避自行处理Markdown语法的边缘case问题,稳定性更高,选用无浏览器依赖的marked库实现:
- 安装依赖
npm install marked
- 功能实现代码
const { marked } = require('marked'); function extractMarkdownMeta(markdownContent) { // 将Markdown内容解析为结构化Token列表 const tokens = marked.lexer(markdownContent); let title = ''; let description = ''; let gotTitle = false; let gotDesc = false; for (const token of tokens) { // 匹配第一个h1标题 if (!gotTitle && token.type === 'heading' && token.depth === 1) { title = token.text.trim(); gotTitle = true; continue; } // 匹配标题后的第一个正文段落 if (gotTitle && !gotDesc && token.type === 'paragraph') { description = token.text.trim(); gotDesc = true; break; } } return { title, description }; } // 测试示例 const demoMd = `# What is super? Super is a keyword used to pass props to the upper classes. Let's begin ... .. etc.. `; const result = extractMarkdownMeta(demoMd); console.log('标题:', result.title); // 输出:What is super? console.log('描述:', result.description); // 输出:Super is a keyword used to pass props to the upper classes.
方案二:正则实现(无第三方依赖)
如果不想引入额外依赖,可通过正则匹配实现,适合Markdown格式规范的场景:
function extractMarkdownMeta(markdownContent) { // 匹配第一个h1标题,忽略前置空白行 const titleReg = /^(?:\s*)#\s+(.*?)(?:\r?\n|$)/m; const titleMatch = markdownContent.match(titleReg); const title = titleMatch ? titleMatch[1].trim() : ''; let description = ''; if (title) { // 截取h1标题之后的内容 const contentAfterTitle = markdownContent.slice(titleMatch.index + titleMatch[0].length); // 匹配第一个非特殊Markdown标记开头的段落 const descReg = /^(?:\s*\n)*([^#\n`\-*>.].*?)(?:\s*\n|$)/s; const descMatch = contentAfterTitle.match(descReg); description = descMatch ? descMatch[1].trim() : ''; } return { title, description }; }
内容的提问来源于stack exchange,提问作者Rahul
相关产品推荐
相关产品推荐

