网页可见文本替换的JavaScript正则问题及优化需求
问题说明
需要匹配名为mylist的JSON列表中的字符串,将网页所有可见字符串替换为<strong>REPLACED!</strong>,要求脚本兼容现代浏览器且不影响页面加载性能。目前脚本在部分页面(如Google搜索结果)生效,但在Qwant等页面完全无效,求优化方案。
现有脚本
@resource mylist https://example.com/mylist.json (function(){ var mylist = JSON.parse(GM_getResourceText("mylist")); var regexp = new RegExp('\b(' + mylist.join('|') + ')\b(?!\)\))', "gi"); function walk(node) { var child, next; switch ( node.nodeType ) { case 1: case 9: case 11: child = node.firstChild; while ( child ) { next = child.nextSibling; walk(child); child = next; } break; case 3: handleText(node); break; } } function handleText(textNode) { textNode.nodeValue = textNode.nodeValue.replace(regexp, 'REPLACED!'); } walk(document.body); })();
问题分析及优化方案
1. 正则表达式失效问题
原正则的\b是英文单词边界,对带特殊字符、非英文的字符串匹配失效;同时直接拼接列表字符串时未转义正则元字符(如.、*、+),会导致匹配逻辑直接崩溃。
- 优化:给列表中每个字符串做正则转义,改用更灵活的边界匹配逻辑:
// 正则转义工具函数 function escapeRegExp(str) { return str.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); } // 转义后再构建正则 var escapedList = mylist.map(escapeRegExp); var regexp = new RegExp('(?<!\\w)(' + escapedList.join('|') + ')(?!\\w)(?!\\)\\))', "gi");
2. 动态内容未覆盖问题
Qwant这类现代页面会通过JS动态渲染内容,原脚本只在页面初始加载时执行一次,无法处理后续新增的DOM节点。
- 优化:用
MutationObserver监听DOM变化,自动处理新增节点:
// 处理单个节点的逻辑 function processNode(node) { // 跳过脚本、样式等无需处理的节点 const skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'IFRAME']; if (node.nodeType === 1 && skipTags.includes(node.nodeName)) return; if (node.nodeType === 3) { handleText(node); } else if (node.nodeType === 1) { node.childNodes.forEach(processNode); } } // 初始处理现有DOM processNode(document.body); // 监听动态新增内容 const observer = new MutationObserver((mutations) => { mutations.forEach(mutation => { mutation.addedNodes.forEach(processNode); }); }); observer.observe(document.body, { childList: true, subtree: true });
3. 替换内容无法插入HTML标签问题
原脚本直接修改文本节点的nodeValue,无法插入<strong>标签,只能生成纯文本。
- 优化:拆分文本节点,插入HTML元素:
function handleText(textNode) { const text = textNode.nodeValue; if (!regexp.test(text)) return; // 无匹配则直接跳过,提升性能 const parent = textNode.parentNode; let lastIndex = 0; let match; regexp.lastIndex = 0; // 重置正则索引,避免循环异常 while ((match = regexp.exec(text)) !== null) { // 插入匹配前的文本 if (match.index > lastIndex) { parent.insertBefore(document.createTextNode(text.slice(lastIndex, match.index)), textNode); } // 创建并插入strong元素 const strong = document.createElement('strong'); strong.textContent = 'REPLACED!'; parent.insertBefore(strong, textNode); lastIndex = match.index + match[0].length; } // 插入剩余文本 if (lastIndex < text.length) { parent.insertBefore(document.createTextNode(text.slice(lastIndex)), textNode); } // 删除原文本节点 parent.removeChild(textNode); }
4. 性能与执行时机优化
- 在Tampermonkey头部添加
@run-at document-idle,确保脚本在DOM加载完成后执行,避免过早执行导致DOM访问不完整。 - 遍历DOM时跳过无需处理的节点(如脚本、样式标签),减少不必要的遍历操作。
完整优化后脚本
// ==UserScript== // @name String Replacer // @namespace http://tampermonkey.net/ // @version 0.1 // @description Replace target strings with <strong>REPLACED!</strong> // @author You // @match *://*/* // @resource mylist https://example.com/mylist.json // @run-at document-idle // @grant GM_getResourceText // ==/UserScript== (function(){ const mylist = JSON.parse(GM_getResourceText("mylist")); // 正则转义工具函数 function escapeRegExp(str) { return str.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); } const escapedList = mylist.map(escapeRegExp); const regexp = new RegExp('(?<!\\w)(' + escapedList.join('|') + ')(?!\\w)(?!\\)\\))', "gi"); function handleText(textNode) { const text = textNode.nodeValue; if (!regexp.test(text)) return; const parent = textNode.parentNode; let lastIndex = 0; let match; regexp.lastIndex = 0; while ((match = regexp.exec(text)) !== null) { if (match.index > lastIndex) { parent.insertBefore(document.createTextNode(text.slice(lastIndex, match.index)), textNode); } const strong = document.createElement('strong'); strong.textContent = 'REPLACED!'; parent.insertBefore(strong, textNode); lastIndex = match.index + match[0].length; } if (lastIndex < text.length) { parent.insertBefore(document.createTextNode(text.slice(lastIndex)), textNode); } parent.removeChild(textNode); } function processNode(node) { const skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'IFRAME']; if (node.nodeType === 1 && skipTags.includes(node.nodeName)) return; if (node.nodeType === 3) { handleText(node); } else if (node.nodeType === 1) { node.childNodes.forEach(processNode); } } // 初始处理现有DOM processNode(document.body); // 监听动态新增内容 const observer = new MutationObserver((mutations) => { mutations.forEach(mutation => { mutation.addedNodes.forEach(processNode); }); }); observer.observe(document.body, { childList: true, subtree: true }); })();
内容的提问来源于stack exchange,提问作者anon
相关产品推荐
相关产品推荐

