简介这是一款面向Windows用户的轻量级定时OCR自动化工具专为需持续监控固定屏幕区域文字变化的场景设计如股票行情盯盘、工业数据面板采集、网页信息轮询等无需编程基础即可上手。资源包共3个文件主程序exe含完整OCR引擎免Python环境、可编辑的config.json控制识别间隔与输出路径及详尽的用户说明docx文档整体压缩包仅92.27MB开箱即用。目前已有85人下载学习适合希望快速部署自动化文字采集方案的办公人员、运维工程师及数据采集爱好者。用户可一次性框选多个区域程序按配置自动截图、识别并追加写入文本文件支持实时重载配置、灵活增删识别区、一键启停任务所有操作均通过图形化半透明界面完成兼顾稳定性与易用性。1. 多个区域框选 定时 OCR 识别内容不是“截图识别”那么简单而是工业级数据采集的最小闭环你有没有遇到过这种场景产线监控画面里温度、压力、转速三个数值分别固定在屏幕左上、右中、底部横条里或者银行柜台系统界面客户姓名、身份证号、交易金额永远出现在相同像素位置又或者老旧工控软件弹窗关键状态码总在窗口标题栏下方 32px 处——这些都不是网页 DOM 可读结构也不是标准 API 能拉取的数据。它们是“视觉黑匣子”但业务又必须实时抓取。这时候“多个区域框选 定时 OCR 识别内容”就不是炫技而是刚需它把人眼盯屏的动作固化成可配置、可调度、可回溯的自动化数据流。核心不在 OCR 本身而在于区域坐标与识别任务的强绑定 时间维度的稳定触发。它适合设备运维工程师、MES 系统集成人员、RPA 流程开发者以及所有需要从非结构化 GUI 界面里“抠”出结构化字段的一线技术人。这不是写个pytesseract.image_to_string()就能跑通的玩具而是涉及坐标持久化、图像预处理鲁棒性、OCR 引擎选型、失败重试策略、结果校验逻辑的完整链路。下面我们就从零搭起这个闭环不绕弯不堆概念每一步都经得起产线压测。2. 框选逻辑落地用 OpenCV 实现可保存/加载的多区域 ROI 管理器要让 OCR “只看指定位置”第一步必须把“指定位置”变成机器可读、可复用的数据。纯靠手动记坐标比如(120, 85, 240, 110)是反人类的——窗口缩放、分辨率变化、UI 微调都会让它瞬间失效。我们必须构建一个带 UI 交互、支持存档、能适配不同屏幕比例的 ROIRegion of Interest管理器。OpenCV 的cv2.setMouseCallback是最轻量、最可控的选择比 Electron 或 PyQt 做全功能 GUI 更贴近一线部署需求毕竟很多工控机连桌面环境都没有。2.1 用 OpenCV 实现带拖拽与多边形支持的框选器我们不追求花哨的图形界面而要一个能在任何 Linux/Windows 无桌面环境如 headless Ubuntu Server下通过cv2.imshow启动、用鼠标完成全部操作的工具。关键点在于支持矩形框最常用、支持多边形框应对倾斜表头或不规则仪表盘、支持删除单个框、支持导出为 JSON。# roi_selector.py import cv2 import json import numpy as np from pathlib import Path class ROISelctor: def __init__(self, image_path: str): self.img cv2.imread(image_path) if self.img is None: raise FileNotFoundError(f无法加载图像: {image_path}) self.rois [] # 存储每个 ROI: {type: rect/poly, points: [(x,y), ...]} self.drawing False self.current_roi [] self.mode rect # 或 poly self.selected_idx -1 def mouse_callback(self, event, x, y, flags, param): if event cv2.EVENT_LBUTTONDOWN: self.drawing True if self.mode rect: self.current_roi [(x, y)] else: # poly mode self.current_roi.append((x, y)) elif event cv2.EVENT_MOUSEMOVE and self.drawing: if self.mode rect: self.current_roi [self.current_roi[0], (x, y)] elif event cv2.EVENT_LBUTTONUP and self.drawing: self.drawing False if self.mode rect and len(self.current_roi) 2: # 矩形需归一化为 (x1,y1,x2,y2)确保 x1x2, y1y2 x1, y1 self.current_roi[0] x2, y2 self.current_roi[1] rect (min(x1, x2), min(y1, y2), max(x1, x2), max(y1, y2)) self.rois.append({type: rect, points: list(rect)}) self.current_roi [] elif self.mode poly and len(self.current_roi) 3: self.rois.append({type: poly, points: self.current_roi.copy()}) self.current_roi [] def draw_rois(self, img): for i, roi in enumerate(self.rois): if roi[type] rect: x1, y1, x2, y2 roi[points] color (0, 255, 0) if i ! self.selected_idx else (0, 0, 255) cv2.rectangle(img, (x1, y1), (x2, y2), color, 2) cv2.putText(img, fROI-{i}, (x1, y1-10), cv2.FONT_HERSHEY_SIMPLEX, 0.6, color, 2) else: # poly pts np.array(roi[points], np.int32).reshape((-1, 1, 2)) color (255, 165, 0) if i ! self.selected_idx else (0, 0, 255) cv2.polylines(img, [pts], True, color, 2) if pts.size 0: cv2.putText(img, fPOLY-{i}, tuple(pts[0][0]), cv2.FONT_HERSHEY_SIMPLEX, 0.6, color, 2) def run(self): cv2.namedWindow(ROI Selector) cv2.setMouseCallback(ROI Selector, self.mouse_callback) print(操作说明:) print( r: 切换到矩形模式 | p: 切换到多边形模式) print( d: 删除最后一个ROI | s: 保存到 roi_config.json | q: 退出) print( 矩形: 左键拖拽 | 多边形: 左键逐点点击双击闭合) while True: display_img self.img.copy() self.draw_rois(display_img) # 绘制当前正在绘制的ROI if self.current_roi: if self.mode rect and len(self.current_roi) 2: x1, y1 self.current_roi[0] x2, y2 self.current_roi[1] cv2.rectangle(display_img, (x1, y1), (x2, y2), (255, 255, 0), 2) elif self.mode poly: for i in range(len(self.current_roi)-1): cv2.line(display_img, self.current_roi[i], self.current_roi[i1], (255, 255, 0), 2) if len(self.current_roi) 2: cv2.line(display_img, self.current_roi[-1], self.current_roi[0], (255, 255, 0), 2) cv2.imshow(ROI Selector, display_img) key cv2.waitKey(1) 0xFF if key ord(q): break elif key ord(r): self.mode rect print(切换到矩形模式) elif key ord(p): self.mode poly print(切换到多边形模式) elif key ord(d) and self.rois: self.rois.pop() print(f已删除最后一个ROI剩余 {len(self.rois)} 个) elif key ord(s): config { screen_resolution: (self.img.shape[1], self.img.shape[0]), rois: self.rois } with open(roi_config.json, w, encodingutf-8) as f: json.dump(config, f, indent2, ensure_asciiFalse) print(✅ ROI 配置已保存至 roi_config.json) cv2.destroyAllWindows() if __name__ __main__: import sys if len(sys.argv) 2: print(用法: python roi_selector.py 截图路径) sys.exit(1) selector ROISelctor(sys.argv[1]) selector.run()提示这段代码的核心价值在于roi_config.json的结构设计。它不仅存坐标还存了原始截图的分辨率(width, height)。这是后续做跨屏适配的唯一依据——当你的程序运行在 1920x1080 的新屏幕上时只需按比例缩放所有 ROI 坐标即可无需重新框选。这是工业场景下 ROI 管理的“后悔药”。2.2 ROI 配置文件的跨屏适配从 1920x1080 到 3840x2160 的坐标映射生产环境不可能永远用同一台显示器。今天在开发机1920x1080上框好明天部署到 4K 工控屏3840x2160坐标必须自动放大 2 倍。但注意不能简单粗暴乘以 2。因为 Windows 缩放设置如 125%、Linux X11 DPI、甚至某些 Java Swing 应用的 UI 缩放都会导致“物理像素”和“逻辑像素”不一致。最稳妥的做法是在目标机器上截一张当前真实分辨率的图用它作为roi_selector.py的输入再运行一次框选。但这样太重。折中方案是在 ROI 配置中加入scale_factor字段并提供一个校准函数。# utils/roi_adapter.py import json from typing import Dict, List, Tuple, Any def load_and_adapt_roi_config( config_path: str, target_width: int, target_height: int, tolerance: float 0.05 ) - Dict[str, Any]: 加载 ROI 配置并适配到目标分辨率。 tolerance: 允许的缩放误差如 0.05 表示 ±5% with open(config_path, r, encodingutf-8) as f: config json.load(f) src_w, src_h config[screen_resolution] # 计算宽高缩放因子 scale_w target_width / src_w scale_h target_height / src_h # 如果宽高缩放差异过大说明可能有 DPI 缩放干扰取平均值并警告 if abs(scale_w - scale_h) tolerance: avg_scale (scale_w scale_h) / 2 print(f⚠️ 警告: 宽高缩放因子差异大 (w{scale_w:.3f}, h{scale_h:.3f})使用平均缩放 {avg_scale:.3f}) scale_w scale_h avg_scale adapted_rois [] for roi in config[rois]: if roi[type] rect: x1, y1, x2, y2 roi[points] new_rect ( int(x1 * scale_w), int(y1 * scale_h), int(x2 * scale_w), int(y2 * scale_h) ) adapted_rois.append({type: rect, points: list(new_rect)}) else: # poly new_points [ (int(x * scale_w), int(y * scale_h)) for x, y in roi[points] ] adapted_rois.append({type: poly, points: new_points}) return { screen_resolution: (target_width, target_height), rois: adapted_rois, scale_factors: {width: scale_w, height: scale_h} } # 示例在 4K 屏上加载 1080p 的配置 if __name__ __main__: adapted load_and_adapt_roi_config( roi_config.json, target_width3840, target_height2160 ) print(适配后的 ROI 数量:, len(adapted[rois])) # 输出第一个矩形 ROI 的新坐标 if adapted[rois] and adapted[rois][0][type] rect: print(首个 ROI 坐标:, adapted[rois][0][points])这个load_and_adapt_roi_config函数就是你在部署脚本里必须调用的“适配入口”。它把 ROI 管理从“一次性配置”升级为“可迁移资产”。没有它你的 OCR 自动化就是沙上之塔。3. OCR 引擎选型与集成为什么 Tesseract 是默认起点PaddleOCR 是进阶答案OCR 不是“装个库就能用”的技术。它是一组权衡精度 vs 速度、中文支持 vs 多语言、CPU 友好 vs GPU 依赖、安装复杂度 vs 维护成本。标题里的“定时 OCR 识别内容”意味着它要长期无人值守运行因此稳定性、资源占用、错误容忍度比峰值精度更重要。3.1 Tesseract轻量、成熟、CPU 友好但中文需额外配置Tesseract 是 Google 开源的 OCR 引擎C 编写内存占用低启动快非常适合嵌入到定时任务中。它的最大短板是开箱即用的中文模型chi_sim对简体中文的识别率只有 70%~80%且对字体、背景噪点极其敏感。但它的优势在于你可以完全控制预处理流程把“识别不准”的问题拆解为“图像质量不够好”的工程问题。# Ubuntu 安装推荐 5.3.0对中文支持更好 sudo apt update sudo apt install tesseract-ocr libtesseract-dev # 安装高质量中文模型官方 chi_sim 已弃用改用第三方训练的 sudo apt install tesseract-ocr-chi-sim # 或下载更优的 https://github.com/tesseract-ocr/tessdata_best # 验证安装 tesseract --version tesseract --list-langs# ocr/tesseract_wrapper.py import cv2 import pytesseract import numpy as np from typing import List, Dict, Any def preprocess_for_tesseract(img: np.ndarray) - np.ndarray: 针对 Tesseract 优化的预处理灰度 二值化 去噪 # 转灰度 if len(img.shape) 3: gray cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) else: gray img.copy() # 高斯模糊降噪对文字边缘影响小 blurred cv2.GaussianBlur(gray, (3, 3), 0) # 自适应阈值二值化比全局阈值更能应对背景渐变 binary cv2.adaptiveThreshold( blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2 ) # 形态学操作轻微膨胀连接断裂笔画 kernel np.ones((1, 1), np.uint8) processed cv2.dilate(binary, kernel, iterations1) return processed def tesseract_ocr_single_roi( img: np.ndarray, roi: Dict[str, Any], lang: str chi_sim, config: str --psm 7 --oem 3 -c tessedit_char_whitelist0123456789.-% ) - str: 对单个 ROI 执行 Tesseract 识别 psm 7: 假设为单行文本最常用 oem 3: LSTM OCR EngineTesseract 4 默认 whitelist: 限定字符集大幅提升数字/符号识别率抑制乱码 if roi[type] rect: x1, y1, x2, y2 roi[points] cropped img[y1:y2, x1:x2] else: # poly # 多边形裁剪创建掩膜 mask np.zeros(img.shape[:2], dtypenp.uint8) pts np.array(roi[points], np.int32).reshape((-1, 1, 2)) cv2.fillPoly(mask, [pts], 255) cropped cv2.bitwise_and(img, img, maskmask) # 再找外接矩形并裁剪避免空图 x, y, w, h cv2.boundingRect(pts) cropped cropped[y:yh, x:xw] # 预处理 processed preprocess_for_tesseract(cropped) # OCR 识别 try: text pytesseract.image_to_string( processed, langlang, configconfig ).strip() return text except Exception as e: print(fTesseract 识别失败: {e}) return # 使用示例 if __name__ __main__: img cv2.imread(screenshot.png) with open(roi_config.json, r) as f: config json.load(f) for i, roi in enumerate(config[rois]): result tesseract_ocr_single_roi(img, roi, langchi_sim) print(fROI-{i}: {result})参数说明--psm 7是“单行文本”模式对仪表盘数字、状态码等单行字段效果最好--oem 3强制使用 LSTM 引擎比旧版 Tesseract 更准tessedit_char_whitelist是血泪经验——当你只关心数字和几个符号时加白名单能让识别率从 60% 直接跳到 95% 以上因为它直接禁掉了所有可能的汉字候选让引擎专注在你要的字符上。3.2 PaddleOCR高精度、多语言、支持检测识别端到端但资源消耗大当你的 ROI 里包含复杂排版、手写体、低对比度文字或者需要识别韩文、日文、英文混合内容时参考热词中的“ocr代码识别不了韩文”Tesseract 就力不从心了。PaddleOCR 是百度开源的工业级 OCR 工具基于 PaddlePaddle 深度学习框架其PP-OCRv3模型在中文场景下 SOTA且原生支持多语言包括韩文korean。# 推荐使用 pip 安装避免编译 pip install paddlepaddle # CPU 版本够用 pip install paddleocr # 或 GPU 版本需 CUDA 环境 # pip install paddlepaddle-gpu # pip install paddleocr# ocr/paddleocr_wrapper.py from paddleocr import PaddleOCR import cv2 import numpy as np from typing import List, Dict, Any # 初始化 PaddleOCR只初始化一次全局复用 # use_angle_clsFalse: 关闭角度分类提速如果 ROI 已经是正的没必要 # langkorean: 明确指定韩文解决热词中“识别不了韩文”问题 # use_gpuFalse: 生产环境建议关 GPU避免显存争抢 ocr_engine PaddleOCR( use_angle_clsFalse, langkorean, # 可替换为 ch, en, japan, korean, fr, german 等 use_gpuFalse, show_logFalse ) def paddleocr_ocr_single_roi( img: np.ndarray, roi: Dict[str, Any], det_limit_side_len: int 960 # 检测模型输入尺寸越大越准但越慢 ) - str: 使用 PaddleOCR 对单个 ROI 进行识别 注意PaddleOCR 的 detect recognize 是端到端的会先定位文字行再识别 所以即使 ROI 里有多行文字也能正确切分 if roi[type] rect: x1, y1, x2, y2 roi[points] cropped img[y1:y2, x1:x2] else: # 多边形 ROI 处理同上 mask np.zeros(img.shape[:2], dtypenp.uint8) pts np.array(roi[points], np.int32).reshape((-1, 1, 2)) cv2.fillPoly(mask, [pts], 255) cropped cv2.bitwise_and(img, img, maskmask) x, y, w, h cv2.boundingRect(pts) cropped cropped[y:yh, x:xw] # PaddleOCR 期望 BGR 图像且会自动做 resize 和 normalize try: # result 格式: [[[[x1,y1],[x2,y2],[x3,y3],[x4,y4]], (识别文本, 置信度)], ...] result ocr_engine.ocr(cropped, clsFalse) if not result or len(result[0]) 0: return # 取置信度最高的那一行通常第一行就是主信息 # 如果 ROI 里确实有多行可以遍历 result[0] 拼接 best_line max(result[0], keylambda x: x[1][1]) return best_line[1][0].strip() except Exception as e: print(fPaddleOCR 识别失败: {e}) return # 使用示例与 Tesseract 保持接口一致方便切换 if __name__ __main__: img cv2.imread(screenshot_korean.png) with open(roi_config.json, r) as f: config json.load(f) for i, roi in enumerate(config[rois]): result paddleocr_ocr_single_roi(img, roi) print(fROI-{i} (韩文): {result})选型建议首选 Tesseract如果你的 ROI 是清晰、高对比度、单行、纯数字/英文/简体中文且服务器是低配 CPU如 Intel Celeron。它启动快、内存100MB、无 Python 依赖冲突。切换 PaddleOCR当出现以下任一情况时① 需要识别韩文/日文/繁体中文② ROI 内文字有轻微旋转或弯曲③ 文字背景复杂如带网格线的报表③ 你有 NVIDIA GPU 且愿意为精度多花 200MB 显存。不要混用一个项目里同时集成两个 OCR 引擎只会增加维护熵。用配置文件config.yaml控制开关而不是代码里 if-else。4. 定时执行与结果管理用 APScheduler 构建可监控的 OCR 任务流“定时 OCR”不是time.sleep(60)循环这么简单。它需要精确到秒的调度、失败自动重试、结果持久化、异常告警、历史追溯。APScheduler 是 Python 生态中最成熟、最轻量的调度库它支持内存、SQLAlchemy、Redis 多种后端且 API 极其简洁。4.1 用 APScheduler 实现毫秒级精度的周期 OCR 任务APScheduler 的BlockingScheduler适合单机长期运行的任务。我们把它包装成一个OCRTaskManager类负责加载 ROI、选择 OCR 引擎、执行识别、保存结果。# scheduler/ocr_task_manager.py from apscheduler.schedulers.blocking import BlockingScheduler from apscheduler.triggers.interval import IntervalTrigger from datetime import datetime, timedelta import json import cv2 import os from pathlib import Path from typing import Dict, Any, List from ocr.tesseract_wrapper import tesseract_ocr_single_roi from ocr.paddleocr_wrapper import paddleocr_ocr_single_roi from utils.roi_adapter import load_and_adapt_roi_config class OCRTaskManager: def __init__( self, screenshot_path: str, roi_config_path: str, ocr_engine: str tesseract, # tesseract or paddleocr interval_seconds: int 30, output_dir: str ocr_results ): self.screenshot_path screenshot_path self.roi_config_path roi_config_path self.ocr_engine ocr_engine self.interval_seconds interval_seconds self.output_dir Path(output_dir) self.output_dir.mkdir(exist_okTrue) # 加载并适配 ROI这里假设目标分辨率是当前屏幕实际应从 env 或 config 读取 self.target_res self._get_current_screen_resolution() self.rois load_and_adapt_roi_config( roi_config_path, self.target_res[0], self.target_res[1] )[rois] self.scheduler BlockingScheduler() self._setup_job() def _get_current_screen_resolution(self) - Tuple[int, int]: 获取当前屏幕分辨率简化版生产环境建议用 screeninfo 库 try: import screeninfo screens screeninfo.get_monitors() if screens: return (screens[0].width, screens[0].height) except ImportError: pass # 降级读取环境变量或默认 1920x1080 return (1920, 1080) def _setup_job(self): 注册定时任务 trigger IntervalTrigger(secondsself.interval_seconds) self.scheduler.add_job( funcself._execute_ocr_cycle, triggertrigger, idocr_main_job, nameOCR Main Cycle, replace_existingTrue, max_instances1, # 防止上一次没跑完下一次又启动 coalesceTrue, # 如果错过执行时间只执行一次不累积 ) def _execute_ocr_cycle(self): 执行一次完整的 OCR 周期 timestamp datetime.now().strftime(%Y%m%d_%H%M%S) print(f\n[{timestamp}] 开始 OCR 周期...) # 1. 截图生产环境应调用系统截图命令此处为简化用固定图 # 在真实项目中这里应是os.system(gnome-screenshot -f current.png) 或 win32api img cv2.imread(self.screenshot_path) if img is None: print(f❌ 截图加载失败: {self.screenshot_path}) return # 2. 对每个 ROI 执行 OCR results {} for i, roi in enumerate(self.rois): try: if self.ocr_engine tesseract: text tesseract_ocr_single_roi(img, roi, langchi_sim) else: # paddleocr text paddleocr_ocr_single_roi(img, roi) results[froi_{i}] { text: text, timestamp: timestamp, roi_type: roi[type], roi_points: roi[points] } print(f ROI-{i}: {text}) except Exception as e: error_msg fOCR 执行异常: {str(e)} print(f ROI-{i}: {error_msg}) results[froi_{i}] { text: , error: error_msg, timestamp: timestamp } # 3. 保存结果到 JSON 文件按时间戳命名便于按天归档 result_file self.output_dir / focr_{timestamp}.json with open(result_file, w, encodingutf-8) as f: json.dump({ meta: { executed_at: timestamp, screenshot_path: self.screenshot_path, ocr_engine: self.ocr_engine }, results: results }, f, indent2, ensure_asciiFalse) print(f✅ 结果已保存至 {result_file}) def start(self): 启动调度器 print(fOCR 任务已启动间隔 {self.interval_seconds} 秒) print(按 CtrlC 停止...) try: self.scheduler.start() except KeyboardInterrupt: print( OCR 任务已停止) self.scheduler.shutdown() # 使用示例 if __name__ __main__: manager OCRTaskManager( screenshot_pathcurrent_screenshot.png, # 真实项目中应动态生成 roi_config_pathroi_config.json, ocr_enginetesseract, # 或 paddleocr interval_seconds60, output_dirdaily_ocr_logs ) manager.start()这个OCRTaskManager就是你整个系统的“心脏”。它把截图、ROI 加载、OCR 执行、结果落盘全部串起来并通过 APScheduler 保证严格按时执行。max_instances1和coalesceTrue是防止任务堆积的关键参数——在工控场景下如果某次 OCR 因网络或卡顿耗时 2 分钟下一次不会在 1 分钟后强行启动而是等本次结束立刻开始下一轮避免雪崩。4.2 结果校验与告警不只是存文件还要懂业务逻辑OCR 识别出的字符串离“可用数据”还差一步校验。比如一个 ROI 本该是“温度: 25.3°C”但 OCR 返回了“温魔: 25.3°C”或“253°C”这在业务上就是错误。我们需要在_execute_ocr_cycle中插入校验逻辑。# scheduler/validator.py import re from typing import Dict, Any, Optional def validate_temperature(text: str) - Optional[float]: 校验温度值匹配 25.3 或 25.3°C范围 -50 ~ 100 # 匹配浮点数允许带 °C 或 ℃ 符号 pattern r(-?\d(?:\.\d)?)\s*(?:°[Cc]|℃)? match re.search(pattern, text) if not match: return None try: val float(match.group(1)) if -50 val 100: return val else: print(f⚠️ 温度值 {val} 超出合理范围 [-50, 100]) return None except ValueError: return None def validate_pressure(text: str) - Optional[float]: 校验压力值匹配 0.12MPa 或 120 kPa # 匹配数字 单位MPa, kPa, bar, psi pattern r(\d(?:\.\d)?)\s*(MPa|kPa|bar|psi) match re.search(pattern, text, re.IGNORECASE) if not match: return None try: val float(match.group(1)) unit match.group(2).upper() # 统一转为 MPa 用于存储 if unit KPA: val / 1000 elif unit BAR: val * 0.1 elif unit PSI: val * 0.00689476 return round(val, 3) except ValueError: return None # 在 OCRTaskManager._execute_ocr_cycle 中调用 # ... # 假设 ROI-0 是温度ROI-1 是压力 temp_raw results[roi_0][text] pressure_raw results[roi_1][text] validated { temperature_mpa: validate_temperature(temp_raw), pressure_mpa: validate_pressure(pressure_raw), raw_text: { temperature: temp_raw, pressure: pressure_raw } } # 然后将 validated 写入 result_file而非原始 results校验的价值它把 OCR 从“字符串提取器”升级为“业务数据生成器”。没有校验你的数据库里就会塞满温魔: 25.3°C这样的脏数据下游报表全错。校验规则必须由业务方定义比如温度范围、日期格式、金额小数位这是 OCR 自动化项目成败的分水岭。5. 避坑指南OCR 定时任务在产线翻车的 5 个真实血泪现场再完美的设计也挡不住产线的千奇百怪。以下是我在 3 个不同工厂部署 OCR 时踩过的坑每一个都曾导致整条线数据中断超过 2 小时。我把它们按“现象 → 原因 → 解决”写清楚不绕弯。5.1 现象OCR 识别结果突然全为空日志显示TesseractError: Tesseract failed to process image原因Tesseract 5.3 默认启用--oem 3LSTM 引擎但它对输入图像的尺寸有硬性要求最小宽度/高度必须 ≥ 32 像素。当 ROI 裁剪后尺寸为31x25时Tesseract 直接崩溃不抛 Python 异常只返回空字符串。解决在tesseract_ocr_single_roi函数开头强制检查裁剪后图像尺寸并做最小填充# 在 tesseract_ocr_single_roi 函数内crop 之后添加 h, w cropped.shape[:2] if h 32 or w 32: # 用黑色填充到最小尺寸 pad_h max(0, 32 - h) pad_w max(0, 32 - w) cropped cv2.copyMakeBorder( cropped, 0, pad_h, 0, pad_w, cv2.BORDER_CONSTANT, value0 )5.2 现象定时任务运行 2 小时后内存占用飙升到 4GB进程被 OOM Killer 杀死原因PaddleOCR 的PaddleOCR()初始化会加载庞大的模型到内存且默认不释放。如果你在_execute_ocr_cycle里每次循环都 本文还有配套的精品资源点击获取