如何修改Google Sheets的importRegex脚本以导入多正则匹配值
修改Google Sheets正则导入脚本以支持多值匹配
原脚本仅能返回第一个匹配结果,要实现提取所有符合正则规则的值,需调整匹配逻辑,收集所有捕获到的内容,具体修改如下:
修改后的完整代码
function importRegex(url, regexInput) { var output = []; var fetchedUrl = UrlFetchApp.fetch(url, {muteHttpExceptions: true}); if (fetchedUrl) { var html = fetchedUrl.getContentText(); if (html.length && regexInput.length) { var regex = new RegExp(regexInput, 'ig'); // 添加全局匹配标志g,保留忽略大小写i let match; // 循环遍历所有匹配项,提取捕获组内容 while ((match = regex.exec(html)) !== null) { if (match[1]) { output.push(unescapeHTML(match[1])); } } } } // 避免请求过于频繁 Utilities.sleep(1000); // 无匹配结果时返回空值,有结果则返回数组(表格自动按行填充) return output.length > 0 ? output : ''; } var htmlEntities = { nbsp: ' ', cent: '¢', pound: '£', yen: '¥', euro: '€', copy: '©', reg: '®', lt: '<', gt: '>', mdash: '–', ndash: '-', quot: '"', amp: '&', apos: '\'' }; function unescapeHTML(str) { return str.replace(/\&([^;]+);/g, function (entity, entityCode) { var match; if (entityCode in htmlEntities) { return htmlEntities[entityCode]; } else if (match = entityCode.match(/^#x([\da-fA-F]+)$/)) { return String.fromCharCode(parseInt(match[1], 16)); } else if (match = entityCode.match(/^#(\d+)$/)) { return String.fromCharCode(~~match[1]); } else { return entity; } }); };
核心修改说明
- 结果存储改为数组:将
output从字符串初始化为数组,用于存储所有匹配结果 - 添加全局匹配标志:创建正则时加入
g标志,实现全局范围内的匹配 - 循环提取所有匹配项:用
RegExp.exec()循环遍历所有匹配结果,提取每个匹配的捕获组内容 - 修正HTML实体转义逻辑:调整
unescapeHTML的正则规则,确保正确解析所有标准HTML实体
使用示例
若要提取页面中所有H2标签的内容,在Google Sheets单元格中调用:
=importRegex("目标页面URL", "<h2[^>]*>(.*?)<\/h2>")
返回的多个结果会自动填充到当前单元格下方的行中。
内容的提问来源于stack exchange,提问作者user3392296
相关产品推荐
相关产品推荐

