You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何排除HTML标签统计单词/短语位置,替代indexOf()方法

Solutions to Count Word Positions Ignoring HTML Tags

Great question! Dealing with HTML tags while calculating text positions is a common pain point, especially since indexOf() counts tag characters which we need to exclude. Let’s walk through step-by-step solutions for both of your requirements, using your sample string as a reference:

"Lorem ipus lores dolores <strong> mauris eget </strong> lectus tincidunt malesuada"


1. Calculate the Position of a Single Word (Ignoring HTML Tags)

Core Idea

We’ll traverse the original string, skip over any HTML tag content (from < to >), and track two key values:

  • The count of plain text characters (excluding tags)
  • The actual index in the original string

When we find a match for our target word, we’ll return both positions (plain text and original string index).

Example Code (JavaScript)

function findSingleWordPosition(str, targetWord) {
  let inTag = false;
  const wordLength = targetWord.length;
  const strLength = str.length;

  for (let i = 0; i <= strLength - wordLength; i++) {
    // Skip entire HTML tag when encountered
    if (str[i] === '<') {
      inTag = true;
      while (i < strLength && str[i] !== '>') i++;
      inTag = false;
      continue;
    }

    if (!inTag) {
      // Check if current position matches the target word
      let isMatch = true;
      for (let j = 0; j < wordLength; j++) {
        const currentIdx = i + j;
        // Break if we hit a tag mid-match or characters don't align
        if (currentIdx >= strLength || str[currentIdx] === '<' || str[currentIdx] !== targetWord[j]) {
          isMatch = false;
          break;
        }
      }

      if (isMatch) {
        // Calculate plain text position by counting non-tag characters up to this index
        let plainTextPos = 0;
        let tempInTag = false;
        for (let k = 0; k < i; k++) {
          if (str[k] === '<') {
            tempInTag = true;
            while (k < strLength && str[k] !== '>') k++;
            tempInTag = false;
            continue;
          }
          if (!tempInTag) plainTextPos++;
        }

        return {
          plainTextPosition: plainTextPos, // 0-based position in stripped text
          originalStringPosition: i // 0-based index in original string
        };
      }
    }
  }
  return null; // Return null if word not found
}

// Test with your sample string
const sampleStr = "Lorem ipus lores dolores <strong> mauris eget </strong> lectus tincidunt malesuada";
const target = "lectus";
const result = findSingleWordPosition(sampleStr, target);
console.log(result);
// Output: { plainTextPosition: 41, originalStringPosition: 54 }

Explanation

  • The outer loop skips tags entirely, so we never count tag characters.
  • When we find a matching word, we re-traverse the string up to that index to count how many plain text characters came before it (giving us the position in the stripped text).
  • We also return the original string index in case you need to reference the position in the raw input.

2. Count All Positions of a Word/Phrase (Ignoring HTML Tags)

This is an extension of the first solution—instead of returning the first match, we collect all valid matches and their positions.

Example Code (JavaScript)

function findAllPhrasePositions(str, targetPhrase) {
  const matches = [];
  let inTag = false;
  const phraseLength = targetPhrase.length;
  const strLength = str.length;

  for (let i = 0; i <= strLength - phraseLength; i++) {
    if (str[i] === '<') {
      inTag = true;
      while (i < strLength && str[i] !== '>') i++;
      inTag = false;
      continue;
    }

    if (!inTag) {
      let isMatch = true;
      for (let j = 0; j < phraseLength; j++) {
        const currentIdx = i + j;
        if (currentIdx >= strLength || str[currentIdx] === '<' || str[currentIdx] !== targetPhrase[j]) {
          isMatch = false;
          break;
        }
      }

      if (isMatch) {
        // Calculate plain text position
        let plainTextPos = 0;
        let tempInTag = false;
        for (let k = 0; k < i; k++) {
          if (str[k] === '<') {
            tempInTag = true;
            while (k < strLength && str[k] !== '>') k++;
            tempInTag = false;
            continue;
          }
          if (!tempInTag) plainTextPos++;
        }

        matches.push({
          plainTextPosition: plainTextPos,
          originalStringPosition: i
        });

        // Skip past the matched phrase to avoid duplicate matches (adjust if you need overlapping matches)
        i += phraseLength - 1;
      }
    }
  }
  return matches;
}

// Test with a string containing multiple matches
const testStr = "Lorem lectus <em> lectus </em> lectus tincidunt";
const targetPhrase = "lectus";
const allResults = findAllPhrasePositions(testStr, targetPhrase);
console.log(allResults);
// Output: Array of 3 objects, each with positions for each "lectus"

Key Notes

  • To handle overlapping phrases (e.g., finding "aaa" in "aaaa"), remove the line i += phraseLength - 1.
  • For more complex HTML (like nested tags or comments), you could enhance the tag-skipping logic, but this simple implementation works for most common use cases.

内容的提问来源于stack exchange,提问作者Miguel Frias

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.25 07:34:54