| """ |
| processor.py — 书籍处理:封面提取、格式转换、OCR 重排 |
| """ |
|
|
| import os |
| import io |
| import base64 |
| import subprocess |
| from pathlib import Path |
| from typing import Optional, Tuple |
|
|
| import httpx |
| from PIL import Image |
|
|
| try: |
| import fitz |
| HAS_PYMUPDF = True |
| except ImportError: |
| HAS_PYMUPDF = False |
|
|
| try: |
| import ebooklib |
| from ebooklib import epub as epublib |
| HAS_EBOOKLIB = True |
| except ImportError: |
| HAS_EBOOKLIB = False |
|
|
| try: |
| import pdfplumber |
| HAS_PDFPLUMBER = True |
| except ImportError: |
| HAS_PDFPLUMBER = False |
|
|
| SILICONFLOW_API_KEY = os.environ.get("SILICONFLOW_API_KEY", "") |
| SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1/chat/completions" |
| OCR_MODEL = "PaddlePaddle/PaddleOCR-VL-1.5" |
|
|
|
|
| |
|
|
| def extract_cover_pdf(file_path: str, output_path: str) -> bool: |
| if not HAS_PYMUPDF: |
| return _extract_cover_pdf_poppler(file_path, output_path) |
| try: |
| doc = fitz.open(file_path) |
| page = doc[0] |
| mat = fitz.Matrix(2, 2) |
| pix = page.get_pixmap(matrix=mat) |
| img = _ensure_rgb(Image.frombytes("RGB", [pix.width, pix.height], pix.samples)) |
| w, h = img.size |
| target_h = int(w * 1.5) |
| if h > target_h: |
| img = img.crop((0, 0, w, target_h)) |
| img.thumbnail((400, 600)) |
| img.save(output_path, "JPEG", quality=85) |
| doc.close() |
| return True |
| except Exception as e: |
| print(f"[processor] PDF 封面提取失败: {e}") |
| return False |
|
|
|
|
| def _extract_cover_pdf_poppler(file_path: str, output_path: str) -> bool: |
| try: |
| tmp_prefix = output_path.replace(".jpg", "") |
| result = subprocess.run( |
| ["pdftoppm", "-jpeg", "-r", "150", "-f", "1", "-l", "1", file_path, tmp_prefix], |
| capture_output=True, timeout=30 |
| ) |
| |
| for candidate in [ |
| f"{tmp_prefix}-000001.jpg", |
| f"{tmp_prefix}-1.jpg", |
| f"{tmp_prefix}-1.jpeg", |
| ]: |
| if os.path.exists(candidate): |
| os.rename(candidate, output_path) |
| return True |
| return False |
| except Exception as e: |
| print(f"[processor] poppler 封面提取失败: {e}") |
| return False |
|
|
|
|
| def _ensure_rgb(img: Image.Image) -> Image.Image: |
| if img.mode == "RGBA": |
| bg = Image.new("RGB", img.size, (255, 255, 255)) |
| bg.paste(img, mask=img.split()[3]) |
| return bg |
| return img.convert("RGB") |
|
|
|
|
| def extract_cover_epub(file_path: str, output_path: str) -> bool: |
| if not HAS_EBOOKLIB: |
| return False |
| try: |
| book = epublib.read_epub(file_path) |
| for item in book.get_items(): |
| if item.get_type() == ebooklib.ITEM_COVER: |
| img_data = item.get_content() |
| img = _ensure_rgb(Image.open(io.BytesIO(img_data))) |
| img.thumbnail((400, 600)) |
| img.save(output_path, "JPEG", quality=85) |
| return True |
| for item in book.get_items(): |
| name = item.get_name().lower() |
| if "cover" in name and item.get_type() == ebooklib.ITEM_IMAGE: |
| img_data = item.get_content() |
| img = _ensure_rgb(Image.open(io.BytesIO(img_data))) |
| img.thumbnail((400, 600)) |
| img.save(output_path, "JPEG", quality=85) |
| return True |
| return False |
| except Exception as e: |
| print(f"[processor] EPUB 封面提取失败: {e}") |
| return False |
|
|
|
|
| def extract_cover_image(file_path: str, output_path: str) -> bool: |
| try: |
| img = _ensure_rgb(Image.open(file_path)) |
| img.thumbnail((400, 600)) |
| img.save(output_path, "JPEG", quality=85) |
| return True |
| except Exception as e: |
| print(f"[processor] 图片封面提取失败: {e}") |
| return False |
|
|
|
|
| IMAGE_FMTS = {"jpg", "jpeg", "png", "webp", "gif", "bmp"} |
|
|
|
|
| def extract_cover(file_path: str, fmt: str, output_path: str) -> bool: |
| if fmt == "pdf": |
| return extract_cover_pdf(file_path, output_path) |
| elif fmt in ("epub", "mobi"): |
| return extract_cover_epub(file_path, output_path) |
| elif fmt in IMAGE_FMTS: |
| return extract_cover_image(file_path, output_path) |
| elif fmt in ("azw3", "docx"): |
| epub_tmp = file_path + "_tmp.epub" |
| ok = convert_to_epub(file_path, epub_tmp, fmt) |
| if ok: |
| result = extract_cover_epub(epub_tmp, output_path) |
| if os.path.exists(epub_tmp): |
| os.unlink(epub_tmp) |
| return result |
| return False |
| return False |
|
|
|
|
| |
|
|
| def convert_to_epub(input_path: str, output_path: str, fmt: str = "mobi") -> bool: |
| try: |
| subprocess.run( |
| ["ebook-convert", input_path, output_path, |
| "--output-profile", "tablet", |
| "--no-default-epub-cover"], |
| capture_output=True, timeout=120 |
| ) |
| return os.path.exists(output_path) |
| except subprocess.TimeoutExpired: |
| print(f"[processor] {fmt} 转换超时") |
| return False |
| except Exception as e: |
| print(f"[processor] {fmt} 转换失败: {e}") |
| return False |
|
|
|
|
| |
|
|
| def get_pdf_page_count(file_path: str) -> Optional[int]: |
| if HAS_PYMUPDF: |
| try: |
| doc = fitz.open(file_path) |
| count = len(doc) |
| doc.close() |
| return count |
| except Exception: |
| pass |
| try: |
| result = subprocess.run( |
| ["pdfinfo", file_path], capture_output=True, text=True, timeout=10 |
| ) |
| for line in result.stdout.split("\n"): |
| if line.startswith("Pages:"): |
| return int(line.split(":")[1].strip()) |
| except Exception: |
| pass |
| return None |
|
|
|
|
| |
|
|
| def extract_text_from_pdf(file_path: str, page_num: int = 0) -> Optional[str]: |
| if not HAS_PYMUPDF: |
| return None |
| try: |
| doc = fitz.open(file_path) |
| if page_num >= len(doc): |
| return None |
| page = doc[page_num] |
| text = page.get_text("text") |
| doc.close() |
| return text.strip() if text.strip() else None |
| except Exception as e: |
| print(f"[processor] 文字提取失败: {e}") |
| return None |
|
|
|
|
| def is_scanned_pdf(file_path: str) -> bool: |
| if not HAS_PYMUPDF: |
| return False |
| try: |
| doc = fitz.open(file_path) |
| text_count = 0 |
| check_pages = min(3, len(doc)) |
| for i in range(check_pages): |
| text = doc[i].get_text("text") |
| text_count += len(text.strip()) |
| doc.close() |
| return (text_count / check_pages) < 50 |
| except Exception: |
| return False |
|
|
|
|
| |
|
|
| async def ocr_page_image(image_bytes: bytes) -> str: |
| if not SILICONFLOW_API_KEY: |
| return "" |
| img_b64 = base64.b64encode(image_bytes).decode("utf-8") |
| payload = { |
| "model": OCR_MODEL, |
| "messages": [ |
| { |
| "role": "user", |
| "content": [ |
| {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{img_b64}"}}, |
| {"type": "text", "text": "请识别图片中的所有文字内容,保持原有段落格式,直接输出识别结果,不要添加任何解释。"} |
| ] |
| } |
| ], |
| "max_tokens": 4096, |
| "temperature": 0.1, |
| } |
| headers = { |
| "Authorization": f"Bearer {SILICONFLOW_API_KEY}", |
| "Content-Type": "application/json", |
| } |
| async with httpx.AsyncClient(timeout=60.0) as client: |
| resp = await client.post(SILICONFLOW_BASE_URL, json=payload, headers=headers) |
| resp.raise_for_status() |
| data = resp.json() |
| return data["choices"][0]["message"]["content"].strip() |
|
|
|
|
| async def ocr_pdf_page(file_path: str, page_num: int) -> str: |
| if not HAS_PYMUPDF: |
| return "" |
| try: |
| doc = fitz.open(file_path) |
| if page_num >= len(doc): |
| return "" |
| page = doc[page_num] |
| mat = fitz.Matrix(2, 2) |
| pix = page.get_pixmap(matrix=mat) |
| img_bytes = pix.tobytes("jpeg") |
| doc.close() |
| return await ocr_page_image(img_bytes) |
| except Exception as e: |
| print(f"[processor] OCR 页面失败 page={page_num}: {e}") |
| return "" |
|
|
|
|
| async def process_ocr_full( |
| book_id: str, |
| file_path: str, |
| progress_callback=None, |
| ) -> Optional[str]: |
| if not HAS_PYMUPDF: |
| return None |
| try: |
| src_doc = fitz.open(file_path) |
| total_pages = len(src_doc) |
| src_doc.close() |
|
|
| all_texts = [] |
| for i in range(total_pages): |
| text = await ocr_pdf_page(file_path, i) |
| all_texts.append(text) |
| if progress_callback: |
| await progress_callback(i + 1, total_pages) |
|
|
| ocr_cache_dir = "/tmp/shuku_cache" |
| os.makedirs(ocr_cache_dir, exist_ok=True) |
| output_path = f"{ocr_cache_dir}/{book_id}_ocr.pdf" |
| new_doc = fitz.open() |
| for page_text in all_texts: |
| page = new_doc.new_page(width=595, height=842) |
| page.insert_textbox( |
| fitz.Rect(50, 50, 545, 792), |
| page_text, |
| fontname="china-s", |
| fontsize=12, |
| align=0, |
| ) |
| new_doc.save(output_path) |
| new_doc.close() |
| return output_path |
| except Exception as e: |
| print(f"[processor] OCR 全文处理失败: {e}") |
| return None |
|
|
|
|
| |
|
|
| def extract_text_pdfplumber(file_path: str) -> list[str]: |
| if not HAS_PDFPLUMBER: |
| return [] |
| pages_text = [] |
| try: |
| with pdfplumber.open(file_path) as pdf: |
| for page in pdf.pages: |
| text = page.extract_text() or "" |
| pages_text.append(text.strip()) |
| except Exception as e: |
| print(f"[processor] pdfplumber 提取失败: {e}") |
| return pages_text |
|
|
|
|
| async def process_ocr_full_auto( |
| book_id: str, |
| file_path: str, |
| progress_callback=None, |
| ) -> Optional[str]: |
| pages_text = extract_text_pdfplumber(file_path) |
| total = len(pages_text) |
|
|
| if total > 0: |
| avg_chars = sum(len(t) for t in pages_text) / total |
| else: |
| avg_chars = 0 |
|
|
| if avg_chars >= 50: |
| print(f"[processor] 使用离线重排,平均每页 {avg_chars:.0f} 字") |
| return _build_reflowed_pdf(book_id, pages_text, progress_callback) |
| else: |
| print(f"[processor] 文字稀少(平均 {avg_chars:.0f} 字/页),使用 OCR") |
| return await process_ocr_full(book_id, file_path, progress_callback) |
|
|
|
|
| def _build_reflowed_pdf( |
| book_id: str, |
| pages_text: list[str], |
| progress_callback=None, |
| ) -> Optional[str]: |
| if not HAS_PYMUPDF: |
| return None |
| total = len(pages_text) |
| try: |
| ocr_cache_dir = "/tmp/shuku_cache" |
| os.makedirs(ocr_cache_dir, exist_ok=True) |
| output_path = f"{ocr_cache_dir}/{book_id}_ocr.pdf" |
| new_doc = fitz.open() |
| for i, page_text in enumerate(pages_text): |
| page = new_doc.new_page(width=595, height=842) |
| page.insert_textbox( |
| fitz.Rect(50, 50, 545, 792), |
| page_text, |
| fontname="china-s", |
| fontsize=12, |
| align=0, |
| ) |
| if progress_callback: |
| import asyncio |
| asyncio.ensure_future(progress_callback(i + 1, total)) |
| new_doc.save(output_path) |
| new_doc.close() |
| return output_path |
| except Exception as e: |
| print(f"[processor] 离线重排 PDF 生成失败: {e}") |
| return None |