# -*- coding: utf-8 -*- """ 독립 실행 가능한 OCR 스크립트 Google Vision API + HRCenterNet 앙상블 기반 한자 OCR 및 손상 영역 탐지 수정사항: 1. 좌표(X값) 변화를 감지하여 자동으로 열(Column)을 구분하여 출력하는 로직 추가 2. [MASK] 좌표 등 소수점 영역 손실 방지를 위한 Safe Crop(내림/올림) 적용 -> 시각화 뿐만 아니라 JSON 결과 데이터 자체에도 적용하여 소수점 제거 """ import os import sys import json import logging import cv2 import math import numpy as np from pathlib import Path from dotenv import load_dotenv # 현재 스크립트의 디렉토리를 Python 경로에 추가 current_dir = os.path.dirname(os.path.abspath(__file__)) if current_dir not in sys.path: sys.path.insert(0, current_dir) # 환경 변수 로드 load_dotenv() # 로깅 설정 logging.basicConfig( level=logging.INFO, format='%(asctime)s - [%(levelname)s] %(message)s' ) logger = logging.getLogger("DONG_OCR") # OCR 엔진 및 전처리 모듈 import try: from ai_modules.ocr_engine import get_ocr_engine from ai_modules.preprocessor_unified import preprocess_image_unified except ImportError as e: logger.error(f"❌ 모듈 import 실패: {e}") sys.exit(1) def format_ocr_results(raw_results, image_filename): """ OCR 결과를 요청하신 JSON 포맷으로 변환하는 함수 수정: JSON 저장 시에도 Safe Crop(내림/올림)을 적용하여 정수로 변환 """ formatted_list = [] if raw_results is None: raw_results = [] if not raw_results: return {"image": image_filename, "results": []} order_counter = 0 for idx, item in enumerate(raw_results): if not isinstance(item, dict): continue min_x, min_y, max_x, max_y = 0.0, 0.0, 0.0, 0.0 # 1. 이미 'box' 리스트가 있는 경우 if 'box' in item and isinstance(item['box'], list) and len(item['box']) == 4: try: min_x, min_y, max_x, max_y = map(float, item['box']) except: pass # 2. 'box'가 없으면 개별 좌표 키 사용 if min_x == 0 and max_x == 0: mx = item.get('min_x') my = item.get('min_y') Mx = item.get('max_x') My = item.get('max_y') if mx is None: mx = item.get('x', 0) if my is None: my = item.get('y', 0) if Mx is None: Mx = item.get('x2') if Mx is None: width = item.get('width', 0) Mx = mx + width if width > 0 else 0 if My is None: My = item.get('y2') if My is None: height = item.get('height', 0) My = my + height if height > 0 else 0 try: min_x, min_y, max_x, max_y = float(mx), float(my), float(Mx), float(My) except: continue if min_x == 0 and min_y == 0 and max_x == 0 and max_y == 0: width = item.get('width', 0) height = item.get('height', 0) if width > 0 and height > 0: cx, cy = item.get('center_x', width/2), item.get('center_y', height/2) min_x, min_y = cx - width/2, cy - height/2 max_x, max_y = cx + width/2, cy + height/2 else: continue if max_x <= min_x or max_y <= min_y: continue # === [추가됨] JSON 데이터 자체에 Safe Crop 적용 (소수점 제거) === # min 좌표는 내림(floor), max 좌표는 올림(ceil)하여 영역 확보 후 정수 변환 min_x = int(math.floor(min_x)) min_y = int(math.floor(min_y)) max_x = int(math.ceil(max_x)) max_y = int(math.ceil(max_y)) # 음수 좌표 방지 (최소 0) min_x = max(0, min_x) min_y = max(0, min_y) # ========================================================== new_item = { "order": order_counter, "text": item.get('text', ''), "type": item.get('type', 'TEXT'), "box": [min_x, min_y, max_x, max_y], "confidence": float(item.get('confidence', 0.0)), "source": item.get('source', 'Unknown') } formatted_list.append(new_item) order_counter += 1 return {"image": image_filename, "results": formatted_list} def draw_bboxes(image_path, results, output_path): """이미지에 Bounding Box 그리기 (Safe Crop 적용)""" try: img_array = np.fromfile(image_path, np.uint8) img = cv2.imdecode(img_array, cv2.IMREAD_COLOR) if img is None: img = cv2.imread(image_path) if img is None: return box_count = 0 colors = { 'Google': (0, 255, 0), 'Custom': (255, 0, 255), 'MASK1': (255, 0, 0), 'MASK2': (0, 0, 255), 'Default': (0, 255, 255) } for item in results: box = item.get('box', []) if len(box) != 4: continue try: # format_ocr_results에서 이미 정수로 변환되어 오지만, # 안전을 위해 한 번 더 처리 (float로 들어와도 처리 가능하도록 유지) x1 = int(math.floor(float(box[0]))) y1 = int(math.floor(float(box[1]))) x2 = int(math.ceil(float(box[2]))) y2 = int(math.ceil(float(box[3]))) except: continue h, w = img.shape[:2] # 이미지 범위 벗어나지 않게 클리핑 x1 = max(0, min(x1, w-1)) y1 = max(0, min(y1, h-1)) x2 = max(x1+1, min(x2, w)) y2 = max(y1+1, min(y2, h)) text = item.get('text', '') source = item.get('source', '') itype = item.get('type', 'TEXT') if 'MASK1' in itype or '[MASK1]' in text: color = colors['MASK1'] elif 'MASK2' in itype or '[MASK2]' in text: color = colors['MASK2'] elif source in colors: color = colors[source] else: color = colors['Default'] cv2.rectangle(img, (x1, y1), (x2, y2), color, 2) if itype == 'TEXT' and len(text) <= 2: cv2.putText(img, text, (x1, y1-5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, color, 1) elif 'MASK' in itype: label = '[M1]' if itype == 'MASK1' else '[M2]' cv2.putText(img, label, (x1, y1-5), cv2.FONT_HERSHEY_SIMPLEX, 0.4, color, 1) box_count += 1 ext = os.path.splitext(output_path)[1].lower() params = [int(cv2.IMWRITE_JPEG_QUALITY), 95] if ext in ['.jpg', '.jpeg'] else [int(cv2.IMWRITE_PNG_COMPRESSION), 3] result, encoded_img = cv2.imencode(ext, img, params) if result: with open(output_path, mode='wb') as f: encoded_img.tofile(f) logger.info(f"🖼️ B-Box 이미지 저장됨: {output_path} ({box_count}개 박스)") except Exception as e: logger.error(f"❌ 시각화 중 오류: {e}") def run_ocr(image_path, use_preprocessing=True): """OCR 실행, 결과 출력 및 저장""" if not os.path.exists(image_path): logger.error(f"❌ 이미지 없음: {image_path}") return False logger.info(f"🚀 OCR 분석 시작: {image_path}") try: # 1. 전처리 ocr_image_path = image_path preprocess_result = {'success': False} if use_preprocessing: logger.info("📸 이미지 전처리 중...") base_dir = os.path.dirname(os.path.abspath(image_path)) base_name = os.path.splitext(os.path.basename(image_path))[0] swin_path = os.path.join(base_dir, f"{base_name}_swin_temp.jpg") ocr_preprocessed_path = os.path.join(base_dir, f"{base_name}_ocr_temp.png") preprocess_result = preprocess_image_unified( input_path=image_path, output_swin_path=swin_path, output_ocr_path=ocr_preprocessed_path, use_rubbing=True ) if preprocess_result.get('success'): ocr_image_path = ocr_preprocessed_path logger.info(f"✅ 전처리 완료: {ocr_preprocessed_path}") else: logger.warning(f"⚠️ 전처리 실패: {preprocess_result.get('message')}") # 2. 엔진 실행 engine = get_ocr_engine() logger.info("✅ OCR 엔진 로드 완료") try: raw_result = engine.run_ocr(ocr_image_path) except Exception as e: logger.error(f"❌ OCR 실행 예외: {e}") return False if not raw_result: return False is_success = raw_result.get('success', False) if not is_success and 'results' in raw_result and isinstance(raw_result['results'], list): is_success = True if not is_success: logger.error(f"❌ OCR 실패: {raw_result.get('error')}") return False logger.info("\n" + "="*60) logger.info("✅ OCR 분석 완료") # 3. 데이터 포맷팅 formatted_result = format_ocr_results(raw_result.get('results', []), os.path.basename(image_path)) results_list = formatted_result.get('results', []) # 4. [열 구분 출력 로직] 좌표 기반으로 열을 계산하여 출력 logger.info("\n" + "📜 [ 인식된 텍스트 결과 (자동 열 구분) ] " + "-"*25) if not results_list: logger.info(" (결과 없음)") else: columns = [] current_col_text = [] # 첫 번째 글자의 X 중심점 계산 first_box = results_list[0]['box'] prev_cx = (first_box[0] + first_box[2]) / 2 for item in results_list: box = item['box'] curr_cx = (box[0] + box[2]) / 2 # 텍스트 추출 (MASK 처리) text = item.get('text', '') if item.get('type') in ['MASK1', 'MASK2']: text = f"[{item.get('type')}]" # === 열 구분 핵심 로직 === # 이전 글자와 X좌표 중심이 50픽셀 이상 차이나면 새로운 열로 간주 # (일반적으로 세로쓰기에서 줄바꿈 시 X좌표가 크게 변함) if abs(curr_cx - prev_cx) > 50: if current_col_text: columns.append("".join(current_col_text)) current_col_text = [] prev_cx = curr_cx # 새로운 열의 기준으로 갱신 current_col_text.append(text) # 같은 열 내에서는 미세한 X 흔들림이 있을 수 있으므로 prev_cx를 계속 갱신하지 않고 # 해당 열의 '대표' X값을 유지하거나, 혹은 글자마다 갱신할 수 있음. # 여기서는 글자가 비스듬할 수 있으므로 매번 갱신하는 방식을 씀 prev_cx = curr_cx # 마지막 열 추가 if current_col_text: columns.append("".join(current_col_text)) # 출력 for idx, col_text in enumerate(columns, 1): logger.info(f" [열 {idx:02d}] {col_text}") logger.info("-" * 60 + "\n") # 5. 결과 저장 json_path = os.path.splitext(image_path)[0] + "_ocr_result.json" with open(json_path, 'w', encoding='utf-8') as f: json.dump(formatted_result, f, ensure_ascii=False, indent=2) logger.info(f"💾 JSON 결과 저장됨: {json_path}") # 6. 시각화 저장 output_img_path = os.path.splitext(image_path)[0] + "_bbox.jpg" bbox_image_path = ocr_image_path if use_preprocessing and preprocess_result.get('success') else image_path draw_bboxes(bbox_image_path, results_list, output_img_path) # 7. 통계 counts = {'Google':0, 'Custom':0, 'MASK1':0, 'MASK2':0, 'TEXT':0} for r in results_list: if r['source'] in counts: counts[r['source']] += 1 if r['type'] in counts: counts[r['type']] += 1 logger.info("📊 최종 통계") logger.info(f" - 🟢 Google: {counts['Google']}개") logger.info(f" - 🟣 Custom: {counts['Custom']}개") logger.info(f" - 🔵 MASK1: {counts['MASK1']}개") logger.info(f" - 🔴 MASK2: {counts['MASK2']}개") logger.info(f" - 📝 TEXT: {counts['TEXT']}개") logger.info("="*60) return True except Exception as e: logger.error(f"❌ 오류 발생: {e}", exc_info=True) return False def main(): if len(sys.argv) < 2: print("사용법: python dong_ocr.py <이미지>") sys.exit(1) if not os.getenv('OCR_WEIGHTS_BASE_PATH') or not os.getenv('GOOGLE_CREDENTIALS_JSON'): logger.error("❌ 환경변수 미설정") sys.exit(1) if run_ocr(sys.argv[1]): logger.info("✅ 작업 완료!") sys.exit(0) else: logger.error("❌ 작업 실패") sys.exit(1) if __name__ == "__main__": main()