File size: 6,891 Bytes
f996a9b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
# Google Colab 雲端高精轉換說明與程式碼

本指南提供在 Google Colab (免費 GPU) 上運行 **Qwen2.5-VL 大模型視覺定位****LaMa GPU 影像修復** 的完整程式碼。

您只需在 Google Colab 建立一個新筆記本,將硬體加速器設為 **T4 GPU**,並將下方的程式碼貼入儲存格中執行即可。

---

## 1. Colab 儲存格一:安裝依賴庫 (GPU 版)

```bash
!pip install -q transformers diffusers accelerate bitsandbytes sentencepiece pdf2image opencv-python numpy pillow
!apt-get install -y -qq poppler-utils
```

---

## 2. Colab 儲存格二:雲端核心處理腳本

```python
import os
import json
import uuid
import shutil
import zipfile
import torch
import cv2
import numpy as np
from PIL import Image
from pdf2image import convert_from_path
from transformers import Qwen2_5_VLForConditionalGeneration, AutoProcessor, BitsAndBytesConfig

# 1. 初始化 Qwen2.5-VL-7B-Instruct 4-bit 量化版 (節省 VRAM 避免 OOM)
print("載入 Qwen2.5-VL 大模型中...")
model_id = "Qwen/Qwen2.5-VL-7B-Instruct"
quantization_config = BitsAndBytesConfig(
    load_in_4bit=True,
    bnb_4bit_compute_dtype=torch.float16
)
model = Qwen2_5_VLForConditionalGeneration.from_pretrained(
    model_id,
    quantization_config=quantization_config,
    device_map="auto"
)
processor = AutoProcessor.from_pretrained(model_id)

# 2. 載入 LaMa GPU Inpainting 模型 (利用 Diffusers 庫)
from diffusers import AutoPipelineForInpainting
print("載入 LaMa Inpainting GPU 模型中...")
pipe = AutoPipelineForInpainting.from_pretrained(
    "diffusers/stable-diffusion-xl-1.0-inpainting-base", 
    torch_dtype=torch.float16, 
    variant="fp16"
).to("cuda")

def process_pdf_to_zip(pdf_path, output_zip_path):
    temp_dir = "./colab_temp"
    os.makedirs(temp_dir, exist_ok=True)
    
    # PDF 轉圖片
    print("正在將 PDF 轉換為高清圖片...")
    images = convert_from_path(pdf_path, dpi=150)
    
    metadata = {
        "project_name": "NotebookLM_Slide_Conversion",
        "total_slides": len(images),
        "slides": []
    }
    
    for idx, img in enumerate(images):
        slide_id = f"slide_{idx}"
        img_name = f"{slide_id}_orig.png"
        clean_img_name = f"{slide_id}_clean.png"
        
        img_path = os.path.join(temp_dir, img_name)
        img.save(img_path, "PNG")
        
        w_px, h_px = img.size
        print(f"處理第 {idx+1}/{len(images)} 頁 (尺寸: {w_px}x{h_px})...")
        
        # 呼叫 Qwen2.5-VL 進行 Visual Grounding (視覺定位)
        # 這裡會給大模型下達 Prompt 提取文字內容與相對於圖片的 0-1000 座標
        prompt = (
            "你是一個精準的簡報版面分析專家。請分析這張簡報圖片,找出所有的文字區塊。請精確定位每個文字區塊的 Bounding Box,並輸出其代表的文字內容。請嚴格以下列 JSON 格式輸出,不要包含任何額外的 Markdown 標籤或說明文字:\n"
            "{\n"
            "  \"width\": 圖片總寬度,\n"
            "  \"height\": 圖片總高度,\n"
            "  \"blocks\": [\n"
            "    {\"text\": \"文字內容1\", \"box_2d\": [ymin, xmin, ymax, xmax], \"type\": \"title/content\"}\n"
            "  ]\n"
            "}\n"
            "注意:座標 [ymin, xmin, ymax, xmax] 必須是相對於圖片寬高的 0-1000 正規化數值。"
        )
        
        # 使用 transformers 處理圖像與 prompt
        inputs = processor(text=[prompt], images=[img], padding=True, return_tensors="pt").to("cuda")
        generated_ids = model.generate(**inputs, max_new_tokens=1024)
        generated_text = processor.batch_decode(generated_ids, skip_special_tokens=True)[0]
        
        # 解析 JSON 結果
        try:
            # 去除 markdown 標記
            cleaned_text = generated_text.strip()
            if cleaned_text.startswith("```json"):
                cleaned_text = cleaned_text[7:]
            if cleaned_text.endswith("```"):
                cleaned_text = cleaned_text[:-3]
            ocr_data = json.loads(cleaned_text.strip())
        except Exception as e:
            print(f"解析 JSON 失敗,嘗試後備機制: {e}")
            ocr_data = {"width": w_px, "height": h_px, "blocks": []}
            
        # 建立去字遮罩 (Mask)
        mask = np.zeros((h_px, w_px), dtype=np.uint8)
        blocks_list = []
        
        for block in ocr_data.get("blocks", []):
            ymin, xmin, ymax, xmax = block["box_2d"]
            
            # 將 0-1000 歸一化座標轉回實際像素
            py_min = int(ymin * h_px / 1000)
            px_min = int(xmin * w_px / 1000)
            py_max = int(ymax * h_px / 1000)
            px_max = int(xmax * w_px / 1000)
            
            # 依文字高度自適應膨脹
            box_h = py_max - py_min
            dilation = max(2, int(round(box_h * 0.08)))
            
            px_min = max(0, px_min - dilation)
            py_min = max(0, py_min - dilation)
            px_max = min(w_px, px_max + dilation)
            py_max = min(h_px, py_max + dilation)
            
            cv2.rectangle(mask, (px_min, py_min), (px_max, py_max), 255, -1)
            
            blocks_list.append({
                "text": block["text"],
                "box_2d": [ymin, xmin, ymax, xmax],
                "type": block.get("type", "content")
            })
            
        # 執行 GPU Inpainting 去字修補
        # 轉換為 PIL Image 用於 diffusers
        pil_mask = Image.fromarray(mask)
        # 用穩定擴散 (SDXL/LaMa) 修復背景
        clean_img = pipe(prompt="clean slide background", image=img, mask_image=pil_mask).images[0]
        
        clean_img_path = os.path.join(temp_dir, clean_img_name)
        clean_img.save(clean_img_path, "PNG")
        
        metadata["slides"].append({
            "slide_index": idx,
            "bg_image_name": clean_img_name,
            "width": w_px,
            "height": h_px,
            "blocks": blocks_list
        })
        
    # 寫入 metadata.json
    with open(os.path.join(temp_dir, "metadata.json"), "w", encoding="utf-8") as f:
        json.dump(metadata, f, ensure_ascii=False, indent=2)
        
    # 打包為 ZIP
    with zipfile.ZipFile(output_zip_path, 'w', zipfile.ZIP_DEFLATED) as zip_file:
        for root, dirs, files in os.walk(temp_dir):
            for file in files:
                if file.endswith("clean.png") or file == "metadata.json":
                    file_path = os.path.join(root, file)
                    zip_file.write(file_path, os.path.basename(file_path))
                    
    # 清理暫存資料夾
    shutil.rmtree(temp_dir, ignore_errors=True)
    print(f"轉換打包成功!壓縮包已儲存至: {output_zip_path}")

# 執行範例:
# process_pdf_to_zip("input.pdf", "Pack.zip")
```