Spaces:
Sleeping
Sleeping
| # Google Colab 雲端高精轉換說明與程式碼 | |
| 本指南提供在 Google Colab (免費 GPU) 上運行 **Qwen2.5-VL 大模型視覺定位** 與 **LaMa GPU 影像修復** 的完整程式碼。 | |
| 您只需在 Google Colab 建立一個新筆記本,將硬體加速器設為 **T4 GPU**,並將下方的程式碼貼入儲存格中執行即可。 | |
| --- | |
| ## 1. Colab 儲存格一:安裝依賴庫 (GPU 版) | |
| ```bash | |
| !pip install -q transformers diffusers accelerate bitsandbytes sentencepiece pdf2image opencv-python numpy pillow | |
| !apt-get install -y -qq poppler-utils | |
| ``` | |
| --- | |
| ## 2. Colab 儲存格二:雲端核心處理腳本 | |
| ```python | |
| import os | |
| import json | |
| import uuid | |
| import shutil | |
| import zipfile | |
| import torch | |
| import cv2 | |
| import numpy as np | |
| from PIL import Image | |
| from pdf2image import convert_from_path | |
| from transformers import Qwen2_5_VLForConditionalGeneration, AutoProcessor, BitsAndBytesConfig | |
| # 1. 初始化 Qwen2.5-VL-7B-Instruct 4-bit 量化版 (節省 VRAM 避免 OOM) | |
| print("載入 Qwen2.5-VL 大模型中...") | |
| model_id = "Qwen/Qwen2.5-VL-7B-Instruct" | |
| quantization_config = BitsAndBytesConfig( | |
| load_in_4bit=True, | |
| bnb_4bit_compute_dtype=torch.float16 | |
| ) | |
| model = Qwen2_5_VLForConditionalGeneration.from_pretrained( | |
| model_id, | |
| quantization_config=quantization_config, | |
| device_map="auto" | |
| ) | |
| processor = AutoProcessor.from_pretrained(model_id) | |
| # 2. 載入 LaMa GPU Inpainting 模型 (利用 Diffusers 庫) | |
| from diffusers import AutoPipelineForInpainting | |
| print("載入 LaMa Inpainting GPU 模型中...") | |
| pipe = AutoPipelineForInpainting.from_pretrained( | |
| "diffusers/stable-diffusion-xl-1.0-inpainting-base", | |
| torch_dtype=torch.float16, | |
| variant="fp16" | |
| ).to("cuda") | |
| def process_pdf_to_zip(pdf_path, output_zip_path): | |
| temp_dir = "./colab_temp" | |
| os.makedirs(temp_dir, exist_ok=True) | |
| # PDF 轉圖片 | |
| print("正在將 PDF 轉換為高清圖片...") | |
| images = convert_from_path(pdf_path, dpi=150) | |
| metadata = { | |
| "project_name": "NotebookLM_Slide_Conversion", | |
| "total_slides": len(images), | |
| "slides": [] | |
| } | |
| for idx, img in enumerate(images): | |
| slide_id = f"slide_{idx}" | |
| img_name = f"{slide_id}_orig.png" | |
| clean_img_name = f"{slide_id}_clean.png" | |
| img_path = os.path.join(temp_dir, img_name) | |
| img.save(img_path, "PNG") | |
| w_px, h_px = img.size | |
| print(f"處理第 {idx+1}/{len(images)} 頁 (尺寸: {w_px}x{h_px})...") | |
| # 呼叫 Qwen2.5-VL 進行 Visual Grounding (視覺定位) | |
| # 這裡會給大模型下達 Prompt 提取文字內容與相對於圖片的 0-1000 座標 | |
| prompt = ( | |
| "你是一個精準的簡報版面分析專家。請分析這張簡報圖片,找出所有的文字區塊。請精確定位每個文字區塊的 Bounding Box,並輸出其代表的文字內容。請嚴格以下列 JSON 格式輸出,不要包含任何額外的 Markdown 標籤或說明文字:\n" | |
| "{\n" | |
| " \"width\": 圖片總寬度,\n" | |
| " \"height\": 圖片總高度,\n" | |
| " \"blocks\": [\n" | |
| " {\"text\": \"文字內容1\", \"box_2d\": [ymin, xmin, ymax, xmax], \"type\": \"title/content\"}\n" | |
| " ]\n" | |
| "}\n" | |
| "注意:座標 [ymin, xmin, ymax, xmax] 必須是相對於圖片寬高的 0-1000 正規化數值。" | |
| ) | |
| # 使用 transformers 處理圖像與 prompt | |
| inputs = processor(text=[prompt], images=[img], padding=True, return_tensors="pt").to("cuda") | |
| generated_ids = model.generate(**inputs, max_new_tokens=1024) | |
| generated_text = processor.batch_decode(generated_ids, skip_special_tokens=True)[0] | |
| # 解析 JSON 結果 | |
| try: | |
| # 去除 markdown 標記 | |
| cleaned_text = generated_text.strip() | |
| if cleaned_text.startswith("```json"): | |
| cleaned_text = cleaned_text[7:] | |
| if cleaned_text.endswith("```"): | |
| cleaned_text = cleaned_text[:-3] | |
| ocr_data = json.loads(cleaned_text.strip()) | |
| except Exception as e: | |
| print(f"解析 JSON 失敗,嘗試後備機制: {e}") | |
| ocr_data = {"width": w_px, "height": h_px, "blocks": []} | |
| # 建立去字遮罩 (Mask) | |
| mask = np.zeros((h_px, w_px), dtype=np.uint8) | |
| blocks_list = [] | |
| for block in ocr_data.get("blocks", []): | |
| ymin, xmin, ymax, xmax = block["box_2d"] | |
| # 將 0-1000 歸一化座標轉回實際像素 | |
| py_min = int(ymin * h_px / 1000) | |
| px_min = int(xmin * w_px / 1000) | |
| py_max = int(ymax * h_px / 1000) | |
| px_max = int(xmax * w_px / 1000) | |
| # 依文字高度自適應膨脹 | |
| box_h = py_max - py_min | |
| dilation = max(2, int(round(box_h * 0.08))) | |
| px_min = max(0, px_min - dilation) | |
| py_min = max(0, py_min - dilation) | |
| px_max = min(w_px, px_max + dilation) | |
| py_max = min(h_px, py_max + dilation) | |
| cv2.rectangle(mask, (px_min, py_min), (px_max, py_max), 255, -1) | |
| blocks_list.append({ | |
| "text": block["text"], | |
| "box_2d": [ymin, xmin, ymax, xmax], | |
| "type": block.get("type", "content") | |
| }) | |
| # 執行 GPU Inpainting 去字修補 | |
| # 轉換為 PIL Image 用於 diffusers | |
| pil_mask = Image.fromarray(mask) | |
| # 用穩定擴散 (SDXL/LaMa) 修復背景 | |
| clean_img = pipe(prompt="clean slide background", image=img, mask_image=pil_mask).images[0] | |
| clean_img_path = os.path.join(temp_dir, clean_img_name) | |
| clean_img.save(clean_img_path, "PNG") | |
| metadata["slides"].append({ | |
| "slide_index": idx, | |
| "bg_image_name": clean_img_name, | |
| "width": w_px, | |
| "height": h_px, | |
| "blocks": blocks_list | |
| }) | |
| # 寫入 metadata.json | |
| with open(os.path.join(temp_dir, "metadata.json"), "w", encoding="utf-8") as f: | |
| json.dump(metadata, f, ensure_ascii=False, indent=2) | |
| # 打包為 ZIP | |
| with zipfile.ZipFile(output_zip_path, 'w', zipfile.ZIP_DEFLATED) as zip_file: | |
| for root, dirs, files in os.walk(temp_dir): | |
| for file in files: | |
| if file.endswith("clean.png") or file == "metadata.json": | |
| file_path = os.path.join(root, file) | |
| zip_file.write(file_path, os.path.basename(file_path)) | |
| # 清理暫存資料夾 | |
| shutil.rmtree(temp_dir, ignore_errors=True) | |
| print(f"轉換打包成功!壓縮包已儲存至: {output_zip_path}") | |
| # 執行範例: | |
| # process_pdf_to_zip("input.pdf", "Pack.zip") | |
| ``` | |