Spaces:
Build error
Build error
| import os | |
| import re | |
| from pathlib import Path | |
| from paddleocr import PaddleOCR | |
| def sort_text(ocr_result, y_threshold=20): | |
| """ | |
| ocr_result: PaddleOCR의 결과 (list of lines) | |
| y_threshold: 같은 줄로 간주할 Y좌표 오차 범위 (픽셀 단위) | |
| """ | |
| final_output = [] | |
| for page in ocr_result: | |
| if not all(k in page for k in('rec_texts', 'rec_scores', 'dt_polys')): | |
| continue | |
| # 1. 데이터 묶기 | |
| lines = [] | |
| for text, score, poly in zip(page['rec_texts'], page['rec_scores'], page['dt_polys']): | |
| # 점수 필터링 | |
| if score < 0.8: continue | |
| # [[x1,y1], [x2,y2], [x3,y3], [x4,y4]] | |
| y = poly[0][1] # 상단 좌측 y값 | |
| x = poly[0][0] # 상단 좌측 x값 | |
| lines.append({'y': y, 'x': x, 'text': text, 'score': score}) | |
| if not lines: continue | |
| # 2. Y좌표 유사도에 따른 행(Row) 그룹화 | |
| # y좌표 기준으로 정렬 | |
| lines.sort(key=lambda x: x['y']) | |
| rows = [] | |
| current_row = [lines[0]] | |
| for i in range(1, len(lines)): | |
| # 현재 텍스트와 이전 텍스트의 Y좌표 차이가 y_threshold 이내면 같은 줄 | |
| if abs(lines[i]['y'] - current_row[-1]['y']) < y_threshold: | |
| current_row.append(lines[i]) | |
| else: | |
| # 줄이 바뀌면 이전 줄을 X좌표 순으로 정렬하여 저장 | |
| current_row.sort(key=lambda x: x['x']) | |
| rows.append(current_row) | |
| current_row = [lines[i]] | |
| # 마지막 줄 처리 | |
| current_row.sort(key=lambda x: x['x']) | |
| rows.append(current_row) | |
| # 3. 텍스트 병합 | |
| text_lines = [] | |
| for row in rows: | |
| row_text = " ".join([item['text'] for item in row]) | |
| # # 같은 줄의 평균을 계산 후 맨 앞에 첨부 | |
| # avg_score = sum([item['score'] for item in row]) / len(row) | |
| # text_lines.append(f"{avg_score:.4f}: {row_text}") | |
| text_lines.append(row_text) | |
| final_output.append("\n".join(text_lines)) | |
| return "\n".join(final_output) | |
| def preprocess_document(file_path): | |
| ocr = PaddleOCR( | |
| lang='korean', | |
| text_det_limit_side_len=2500, | |
| enable_mkldnn=False | |
| ) | |
| input_folder = file_path +'/raw' | |
| output_folder = file_path + '/processed_paddleocr_v4_2500_noscore' | |
| # text 파일 처리 | |
| for txt_file in Path(input_folder).glob('*.txt'): | |
| output_file = Path(output_folder) / f"{txt_file.stem}.txt" | |
| # 파일이 이미 존재하면 건너뛰기 | |
| if output_file.exists(): | |
| print(f'Skip (already exists): {output_file.name}') | |
| continue | |
| with open(txt_file, 'r', encoding='utf-8') as f: | |
| text = f.read() | |
| text = re.sub(r'\n+', '\n', text).strip() | |
| with open(output_file, 'w', encoding='utf-8') as f: | |
| f.write(text) | |
| # jpg 파일 처리 | |
| for jpg_file in Path(input_folder).glob('*'): | |
| # 확장자가 이미지, PDF가 아니면 건너뛰기 | |
| if jpg_file.suffix.lower() not in ['.jpg', '.jpeg', '.png', '.pdf']: continue | |
| output_file = Path(output_folder) / f"{jpg_file.stem}.txt" | |
| # 파일이 이미 존재하면 건너뛰기 | |
| if output_file.exists(): | |
| print(f'Skip (already exists): {output_file.name}') | |
| continue | |
| print(f'Processing file: {jpg_file}') | |
| result = ocr.ocr(str(jpg_file)) | |
| extracted_text = sort_text(result) | |
| print(f'Extracted text: {extracted_text}') | |
| break | |
| with open(output_file, 'w', encoding='utf-8') as f: | |
| f.write(extracted_text) | |
| print(os.system('pwd')) | |
| preprocess_document('data') |