使用Tesseract.js实现图片OCR文字高亮功能失败求助
解决Tesseract.js OCR文字标注失败的问题
核心问题分析
- 调用
worker.recognize后仅提取了纯文本text传给result函数,但该函数需要包含words、bbox等位置信息的完整识别结果对象 - 高亮用的Canvas未与图片尺寸同步,导致绘制的矩形无法正确显示在图片上
- 图片未加载完成就执行绘制,无法获取正确的图片尺寸
- Tesseract.js的
createWorker参数格式不符合新版API要求
修复后的完整代码
script.js
const langSelect = document.getElementById('langSelect'); const fileInput = document.getElementById('fileInput'); const ocrButton = document.getElementById('ocrButton'); const extractSelect = document.getElementById('extractSelect'); const extractButton = document.getElementById('extractButton'); const imageContainer = document.getElementById('imageContainer'); const textOutput = document.getElementById('textOutput'); const inputOverlay = document.getElementById('highlightCanvas'); const ioctx = inputOverlay.getContext('2d'); let extractedText = ''; fileInput.addEventListener('change', function () { const file = fileInput.files[0]; imageContainer.src = URL.createObjectURL(file); // 图片加载完成后同步Canvas尺寸 imageContainer.onload = function() { inputOverlay.width = imageContainer.width; inputOverlay.height = imageContainer.height; ioctx.clearRect(0, 0, inputOverlay.width, inputOverlay.height); }; }); async function performOCR() { const lang = langSelect.value; if (lang && fileInput.files.length > 0) { const file = fileInput.files[0]; ocrButton.disabled = true; ocrButton.classList.add('disabled'); textOutput.innerHTML = '<div class="loading"><span class="loader animate" aria-label="Processing your request"></span><span>Loading...</span></div>'; try { const { createWorker } = Tesseract; // 新版Tesseract.js Worker初始化方式 const worker = await createWorker({ workerPath: "./dist/worker.min.js", langPath: "./", logger: m => console.log(m), }); await worker.loadLanguage(lang); await worker.initialize(lang); // 获取包含位置信息的完整识别结果 const { data } = await worker.recognize(file); if (data.text.trim()) { result(data); await worker.terminate(); } else { textOutput.innerText = 'No text found.'; extractedText = ''; await worker.terminate(); } } catch (error) { textOutput.innerText = 'Error: OCR failed.'; textOutput.classList.add('error'); console.error(error); extractedText = ''; } ocrButton.disabled = false; ocrButton.classList.remove('disabled'); } else { textOutput.innerText = 'Please select a language and an image file.'; } } function extractText() { const fileType = extractSelect.value; if (fileType && extractedText) { const blob = new Blob([extractedText], { type: `text/${fileType}` }); const url = URL.createObjectURL(blob); const link = document.createElement('a'); link.href = url; link.download = `extracted_text.${fileType}`; link.click(); } else { textOutput.innerText = 'Please select a file type and perform OCR first.'; } } function result(res) { // 清空Canvas之前的绘制内容 ioctx.clearRect(0, 0, inputOverlay.width, inputOverlay.height); console.log('识别结果:', res); textOutput.innerText = res.text; extractedText = res.text; // 遍历识别到的单词,绘制高亮框和基线 res.words.forEach(function(w) { const b = w.bbox; ioctx.lineWidth = 2; // 绘制红色单词边界框 ioctx.strokeStyle = 'red'; ioctx.strokeRect(b.x0, b.y0, b.x1 - b.x0, b.y1 - b.y0); // 绘制绿色基线 ioctx.beginPath(); ioctx.moveTo(w.baseline.x0, w.baseline.y0); ioctx.lineTo(w.baseline.x1, w.baseline.y1); ioctx.strokeStyle = 'green'; ioctx.stroke(); }); }
index.html
<!DOCTYPE html> <html lang="en"> <head> <meta charset="UTF-8"> <meta name="viewport" content="width=device-width, initial-scale=1.0"> <title>OCR Reader</title> <script src="./dist/dist-tesseract.min.js"></script> <link rel="stylesheet" href="./7.css"> <link rel="stylesheet" href="./styles.css"> <style> /* 让Canvas覆盖在图片上方 */ #content > div { position: relative; display: inline-block; } #highlightCanvas { position: absolute; top: 0; left: 0; pointer-events: none; /* 不干扰图片交互 */ } #imageContainer { display: block; } </style> </head> <body> <div class="center"> <select id="langSelect"> <option value="eng">English</option> </select> <input type="file" accept="image/png, image/jpeg" id="fileInput"> <button onclick="performOCR()" id="ocrButton">Perform OCR</button> <div id="content"> <div> <canvas id="highlightCanvas"></canvas> <img id="imageContainer" alt="Uploaded Image"> </div> <div id="textOutput" class="output-box"></div> <select id="extractSelect"> <option value="">Select File Type</option> <option value="txt">Text File (.txt)</option> <option value="docx">Word Document (.docx)</option> </select> <button onclick="extractText()" id="extractButton">Extract Text</button> </div> </div> <script src="./dist/worker.min.js"></script> <script src="./script.js"></script> </body> </html>
关键修复点
- 传递完整识别结果:将
worker.recognize返回的整个data对象传给result函数,确保能获取到单词位置信息 - 同步Canvas尺寸:在图片
onload事件中设置Canvas宽高,保证绘制坐标与图片匹配 - 调整Canvas布局:通过CSS绝对定位让Canvas覆盖在图片上方,实现高亮效果
- 更新Worker初始化:使用新版Tesseract.js的
loadLanguage和initialize方法,避免参数错误 - 清空Canvas缓存:每次绘制前清空Canvas,避免重复叠加绘制
内容的提问来源于stack exchange,提问作者tabs
相关产品推荐
相关产品推荐

