You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Tesseract.js实现图片OCR文字高亮功能失败求助

解决Tesseract.js OCR文字标注失败的问题

核心问题分析

  • 调用worker.recognize后仅提取了纯文本text传给result函数,但该函数需要包含words、bbox等位置信息的完整识别结果对象
  • 高亮用的Canvas未与图片尺寸同步,导致绘制的矩形无法正确显示在图片上
  • 图片未加载完成就执行绘制,无法获取正确的图片尺寸
  • Tesseract.js的createWorker参数格式不符合新版API要求

修复后的完整代码

script.js

const langSelect = document.getElementById('langSelect');
const fileInput = document.getElementById('fileInput');
const ocrButton = document.getElementById('ocrButton');
const extractSelect = document.getElementById('extractSelect');
const extractButton = document.getElementById('extractButton');
const imageContainer = document.getElementById('imageContainer');
const textOutput = document.getElementById('textOutput');
const inputOverlay = document.getElementById('highlightCanvas');
const ioctx = inputOverlay.getContext('2d');
let extractedText = '';

fileInput.addEventListener('change', function () {
  const file = fileInput.files[0];
  imageContainer.src = URL.createObjectURL(file);
  // 图片加载完成后同步Canvas尺寸
  imageContainer.onload = function() {
    inputOverlay.width = imageContainer.width;
    inputOverlay.height = imageContainer.height;
    ioctx.clearRect(0, 0, inputOverlay.width, inputOverlay.height);
  };
});

async function performOCR() {
  const lang = langSelect.value;

  if (lang && fileInput.files.length > 0) {
    const file = fileInput.files[0];
    ocrButton.disabled = true;
    ocrButton.classList.add('disabled');
    textOutput.innerHTML = '<div class="loading"><span class="loader animate" aria-label="Processing your request"></span><span>Loading...</span></div>';

    try {
      const { createWorker } = Tesseract;
      // 新版Tesseract.js Worker初始化方式
      const worker = await createWorker({
        workerPath: "./dist/worker.min.js",
        langPath: "./",
        logger: m => console.log(m),
      });
      await worker.loadLanguage(lang);
      await worker.initialize(lang);

      // 获取包含位置信息的完整识别结果
      const { data } = await worker.recognize(file);

      if (data.text.trim()) {
        result(data);
        await worker.terminate();
      } else {
        textOutput.innerText = 'No text found.';
        extractedText = '';
        await worker.terminate();
      }

    } catch (error) {
      textOutput.innerText = 'Error: OCR failed.';
      textOutput.classList.add('error');
      console.error(error);
      extractedText = '';
    }

    ocrButton.disabled = false;
    ocrButton.classList.remove('disabled');
  } else {
    textOutput.innerText = 'Please select a language and an image file.';
  }
}

function extractText() {
  const fileType = extractSelect.value;

  if (fileType && extractedText) {
    const blob = new Blob([extractedText], {
      type: `text/${fileType}`
    });

    const url = URL.createObjectURL(blob);
    const link = document.createElement('a');
    link.href = url;
    link.download = `extracted_text.${fileType}`;
    link.click();
  } else {
    textOutput.innerText = 'Please select a file type and perform OCR first.';
  }
}

function result(res) {
  // 清空Canvas之前的绘制内容
  ioctx.clearRect(0, 0, inputOverlay.width, inputOverlay.height);
  
  console.log('识别结果:', res);
  textOutput.innerText = res.text;
  extractedText = res.text;

  // 遍历识别到的单词,绘制高亮框和基线
  res.words.forEach(function(w) {
    const b = w.bbox;
    ioctx.lineWidth = 2;

    // 绘制红色单词边界框
    ioctx.strokeStyle = 'red';
    ioctx.strokeRect(b.x0, b.y0, b.x1 - b.x0, b.y1 - b.y0);

    // 绘制绿色基线
    ioctx.beginPath();
    ioctx.moveTo(w.baseline.x0, w.baseline.y0);
    ioctx.lineTo(w.baseline.x1, w.baseline.y1);
    ioctx.strokeStyle = 'green';
    ioctx.stroke();
  });
}

index.html

<!DOCTYPE html>
<html lang="en">
<head>
    <meta charset="UTF-8">
    <meta name="viewport" content="width=device-width, initial-scale=1.0">
    <title>OCR Reader</title>
    <script src="./dist/dist-tesseract.min.js"></script>
    <link rel="stylesheet" href="./7.css">
    <link rel="stylesheet" href="./styles.css">
    <style>
        /* 让Canvas覆盖在图片上方 */
        #content > div {
            position: relative;
            display: inline-block;
        }
        #highlightCanvas {
            position: absolute;
            top: 0;
            left: 0;
            pointer-events: none; /* 不干扰图片交互 */
        }
        #imageContainer {
            display: block;
        }
    </style>
</head>
<body>
    <div class="center">
        <select id="langSelect">
            <option value="eng">English</option>
        </select>
        <input type="file" accept="image/png, image/jpeg" id="fileInput">
        <button onclick="performOCR()" id="ocrButton">Perform OCR</button>
        
        <div id="content">
            <div>
                <canvas id="highlightCanvas"></canvas>
                <img id="imageContainer" alt="Uploaded Image">
            </div>
            <div id="textOutput" class="output-box"></div>
            <select id="extractSelect">
                <option value="">Select File Type</option>
                <option value="txt">Text File (.txt)</option>
                <option value="docx">Word Document (.docx)</option>
            </select>
            <button onclick="extractText()" id="extractButton">Extract Text</button>
        </div>
    </div>

    <script src="./dist/worker.min.js"></script>
    <script src="./script.js"></script>
</body>
</html>

关键修复点

  1. 传递完整识别结果:将worker.recognize返回的整个data对象传给result函数,确保能获取到单词位置信息
  2. 同步Canvas尺寸:在图片onload事件中设置Canvas宽高,保证绘制坐标与图片匹配
  3. 调整Canvas布局:通过CSS绝对定位让Canvas覆盖在图片上方,实现高亮效果
  4. 更新Worker初始化:使用新版Tesseract.js的loadLanguage和initialize方法,避免参数错误
  5. 清空Canvas缓存:每次绘制前清空Canvas,避免重复叠加绘制

内容的提问来源于stack exchange,提问作者tabs

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.24 08:51:04