You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于ng2-pdf-viewer与pdf.js的多短语搜索实现问题

解决ng2-pdf-viewer多短语/混合搜索高亮覆盖问题

当前代码的核心问题是每次调用eventBus.dispatch('find')都会触发pdf.js清除已有的高亮,仅保留当前搜索项的高亮,多轮循环调用后只有最后一个搜索项的高亮会显示,之前的全部被覆盖。以下是针对性的解决方案:


步骤1:完善搜索输入解析

先实现一个解析函数,精准拆分带引号的短语和无引号的单个单词:

parseSearchQuery(query: string): Array<{ term: string, isPhrase: boolean }> {
  const result = [];
  // 提取所有带引号的短语
  const phraseRegex = /"([^"]+)"/g;
  let match;
  while ((match = phraseRegex.exec(query)) !== null) {
    result.push({ term: match[1], isPhrase: true });
  }
  // 处理剩余无引号内容,拆分为单个单词
  const remainingText = query.replace(phraseRegex, '').trim();
  if (remainingText) {
    const words = remainingText.split(/\s+/).filter(word => word.trim() !== '');
    words.forEach(word => result.push({ term: word, isPhrase: false }));
  }
  return result;
}

步骤2:自定义搜索与高亮逻辑

替换原有的searchOccurrences函数,新增清除高亮、查找匹配、添加高亮的辅助函数:

searchOccurrences(search = ''): void {
  if (search) {
    this.searchWord = search;
  }
  // 清除之前的所有高亮
  this.clearHighlights();
  
  const searchTerms = this.parseSearchQuery(this.searchWord);
  if (searchTerms.length === 0) return;

  // 重置匹配统计数据
  this.totalMatches = 0;
  this.matches = [];
  this.currentWordIndex = 0;
  this.matchIndex = 0;
  this.currentPageIndex = this.currentPage - 1;

  const pdfDoc = this.pdfViewer.document;
  if (!pdfDoc) return;

  // 遍历PDF所有页面
  for (let pageNum = 1; pageNum <= pdfDoc.numPages; pageNum++) {
    pdfDoc.getPage(pageNum).then(page => {
      return page.getTextContent().then(textContent => {
        const pageView = this.pdfViewer.getPageView(pageNum - 1); // 页面索引从0开始
        const textLayer = pageView.textLayer;
        if (!textLayer) return;

        searchTerms.forEach(termObj => {
          const { term, isPhrase } = termObj;
          const pageMatches = this.findMatchesInTextContent(textContent, term, isPhrase);
          this.totalMatches += pageMatches.length;
          // 收集匹配信息(用于跳页等功能)
          this.matches.push(...pageMatches.map(match => ({ page: pageNum, ...match })));
          // 为当前页面的匹配项添加高亮
          this.addHighlightsToTextLayer(textLayer, pageMatches);
        });
      });
    });
  }

  // 更新首尾匹配页
  this.lastPageMatch = this.getLatestPageOnMatches();
  this.firstPageMatch = this.getFirstPageOnMatches();
}

// 清除所有自定义高亮
clearHighlights(): void {
  document.querySelectorAll('.custom-pdf-highlight').forEach(el => el.remove());
}

// 在PDF文本内容中查找匹配项
findMatchesInTextContent(textContent: any, term: string, isPhrase: boolean): Array<{ startIdx: number, endIdx: number, rects: Array<any> }> {
  const textItems = textContent.items;
  const fullPageText = textItems.map(item => item.str).join('');
  
  // 根据是否为短语,创建对应的正则表达式
  const escapedTerm = term.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
  const regex = isPhrase 
    ? new RegExp(escapedTerm, 'gi') 
    : new RegExp(`\\b${escapedTerm}\\b`, 'gi');

  const matches = [];
  let matchResult;
  while ((matchResult = regex.exec(fullPageText)) !== null) {
    const startPos = matchResult.index;
    const endPos = startPos + matchResult[0].length;

    let currentTextPos = 0;
    let startItemIdx = -1;
    let endItemIdx = -1;
    const rects = [];

    // 定位匹配项对应的文本块索引
    for (let i = 0; i < textItems.length; i++) {
      const item = textItems[i];
      const itemTextLength = item.str.length;
      if (currentTextPos <= startPos && currentTextPos + itemTextLength > startPos) {
        startItemIdx = i;
      }
      if (currentTextPos < endPos && currentTextPos + itemTextLength >= endPos) {
        endItemIdx = i;
        break;
      }
      currentTextPos += itemTextLength;
    }

    if (startItemIdx !== -1 && endItemIdx !== -1) {
      // 收集匹配区域的坐标矩形
      for (let i = startItemIdx; i <= endItemIdx; i++) {
        const item = textItems[i];
        const x = item.transform[4];
        const y = item.transform[5];
        rects.push({
          x,
          y,
          width: item.width,
          height: item.height
        });
      }
      matches.push({ startIdx: startItemIdx, endIdx: endItemIdx, rects });
    }
  }
  return matches;
}

// 向文本层添加高亮元素
addHighlightsToTextLayer(textLayer: any, matches: Array<{ rects: Array<any> }>): void {
  matches.forEach(match => {
    match.rects.forEach(rect => {
      const highlight = document.createElement('div');
      highlight.className = 'custom-pdf-highlight';
      // 调整坐标适配文本层的布局
      highlight.style.position = 'absolute';
      highlight.style.left = `${rect.x}px`;
      highlight.style.top = `${rect.y - rect.height}px`;
      highlight.style.width = `${rect.width}px`;
      highlight.style.height = `${rect.height}px`;
      highlight.style.backgroundColor = 'rgba(255, 255, 0, 0.3)';
      highlight.style.zIndex = '1';
      textLayer.div.appendChild(highlight);
    });
  });
}

步骤3:添加高亮样式

在组件的CSS文件中添加自定义高亮样式(可按需调整):

.custom-pdf-highlight {
  pointer-events: none; /* 避免高亮遮挡文本交互 */
  transition: background-color 0.2s;
}
.custom-pdf-highlight:hover {
  background-color: rgba(255, 200, 0, 0.4) !important;
}

效果说明

改造后四个场景的需求均能满足:

  • 场景1:单个带引号短语会被完整匹配并高亮
  • 场景2:多个带引号短语会各自匹配并同时高亮
  • 场景3:多个无引号单词会匹配所有出现的位置并高亮
  • 场景4:单词和短语会同时被高亮,互不覆盖

内容的提问来源于stack exchange,提问作者César Castro Aroche

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.29 18:32:13