Python处理Google Vision API OCR提取行文本与bounding box坐标
完整实现代码
from google.cloud import vision from google.cloud.vision import types import json import io from google.protobuf.json_format import MessageToJson # other code… client = vision.ImageAnnotatorClient() file = "你的图片路径" # 替换为实际图片路径 with io.open(file, 'rb') as image_file: content = image_file.read() image = types.Image(content=content) response = client.document_text_detection(image=image) document = response.full_text_annotation serialized = MessageToJson(document) data = json.loads(serialized) symbols = [] for page in data['pages']: for block in page['blocks']: for paragraph in block['paragraphs']: for word in paragraph['words']: for symbol in word['symbols']: symbols.append(symbol) break_values = ['EOL_SURE_SPACE', 'HYPHEN', 'LINE_BREAK'] space_values = ['UNKNOWN', 'SPACE', 'SURE_SPACE'] # 初始化临时变量缓存当前行数据 current_line = { "content": "", "first_symbol": None, "last_symbol": None } lines = [] for symbol in symbols: # 记录当前行第一个符号用于取左上角坐标 if not current_line["first_symbol"]: current_line["first_symbol"] = symbol # 拼接符号文本 current_line["content"] += symbol["text"] current_line["last_symbol"] = symbol # 处理分隔符逻辑 break_type = None if "property" in symbol and "detectedBreak" in symbol["property"]: break_type = symbol["property"]["detectedBreak"]["type"] if break_type in space_values: current_line["content"] += " " elif break_type in break_values: # 遇到换行,计算当前行的坐标、宽高 first_vertex = current_line["first_symbol"]["boundingBox"]["vertices"][0] last_vertex = current_line["last_symbol"]["boundingBox"]["vertices"][2] line_x = first_vertex["x"] line_y = first_vertex["y"] line_w = last_vertex["x"] - line_x line_h = last_vertex["y"] - line_y # 存入结果列表 lines.append({ "Content": current_line["content"].strip(), "X": line_x, "Y": line_y, "H": line_h, "W": line_w }) # 重置临时变量 current_line = { "content": "", "first_symbol": None, "last_symbol": None } # 处理最后一行没有换行标记的剩余内容 if current_line["content"].strip(): first_vertex = current_line["first_symbol"]["boundingBox"]["vertices"][0] last_vertex = current_line["last_symbol"]["boundingBox"]["vertices"][2] line_x = first_vertex["x"] line_y = first_vertex["y"] line_w = last_vertex["x"] - line_x line_h = last_vertex["y"] - line_y lines.append({ "Content": current_line["content"].strip(), "X": line_x, "Y": line_y, "H": line_h, "W": line_w }) # 输出Markdown格式表格 print("| Content | X | Y | H | W |") print("| --- | --- | --- | --- | --- |") for line in lines: print(f"| {line['Content']} | {line['X']} | {line['Y']} | {line['H']} | {line['W']} |")
逻辑说明
- 用临时变量缓存当前行的文本内容、行首符号、行尾符号,避免频繁读写结果列表
- 遍历每个符号时先拼接文本,再判断符号后的分隔符类型,空格类型直接加空格,换行类型则计算当前行的坐标和宽高存入结果列表,随后重置临时变量
- 遍历结束后额外处理最后一行没有换行标记的剩余内容,避免遗漏
- 最终直接按要求打印Markdown格式的表格,也可根据需求将lines列表导出为CSV、Excel等其他表格格式
内容的提问来源于stack exchange,提问作者paratext
相关产品推荐
相关产品推荐

