Spaces:
Sleeping
Sleeping
File size: 6,891 Bytes
f996a9b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 | # Google Colab 雲端高精轉換說明與程式碼
本指南提供在 Google Colab (免費 GPU) 上運行 **Qwen2.5-VL 大模型視覺定位** 與 **LaMa GPU 影像修復** 的完整程式碼。
您只需在 Google Colab 建立一個新筆記本,將硬體加速器設為 **T4 GPU**,並將下方的程式碼貼入儲存格中執行即可。
---
## 1. Colab 儲存格一:安裝依賴庫 (GPU 版)
```bash
!pip install -q transformers diffusers accelerate bitsandbytes sentencepiece pdf2image opencv-python numpy pillow
!apt-get install -y -qq poppler-utils
```
---
## 2. Colab 儲存格二:雲端核心處理腳本
```python
import os
import json
import uuid
import shutil
import zipfile
import torch
import cv2
import numpy as np
from PIL import Image
from pdf2image import convert_from_path
from transformers import Qwen2_5_VLForConditionalGeneration, AutoProcessor, BitsAndBytesConfig
# 1. 初始化 Qwen2.5-VL-7B-Instruct 4-bit 量化版 (節省 VRAM 避免 OOM)
print("載入 Qwen2.5-VL 大模型中...")
model_id = "Qwen/Qwen2.5-VL-7B-Instruct"
quantization_config = BitsAndBytesConfig(
load_in_4bit=True,
bnb_4bit_compute_dtype=torch.float16
)
model = Qwen2_5_VLForConditionalGeneration.from_pretrained(
model_id,
quantization_config=quantization_config,
device_map="auto"
)
processor = AutoProcessor.from_pretrained(model_id)
# 2. 載入 LaMa GPU Inpainting 模型 (利用 Diffusers 庫)
from diffusers import AutoPipelineForInpainting
print("載入 LaMa Inpainting GPU 模型中...")
pipe = AutoPipelineForInpainting.from_pretrained(
"diffusers/stable-diffusion-xl-1.0-inpainting-base",
torch_dtype=torch.float16,
variant="fp16"
).to("cuda")
def process_pdf_to_zip(pdf_path, output_zip_path):
temp_dir = "./colab_temp"
os.makedirs(temp_dir, exist_ok=True)
# PDF 轉圖片
print("正在將 PDF 轉換為高清圖片...")
images = convert_from_path(pdf_path, dpi=150)
metadata = {
"project_name": "NotebookLM_Slide_Conversion",
"total_slides": len(images),
"slides": []
}
for idx, img in enumerate(images):
slide_id = f"slide_{idx}"
img_name = f"{slide_id}_orig.png"
clean_img_name = f"{slide_id}_clean.png"
img_path = os.path.join(temp_dir, img_name)
img.save(img_path, "PNG")
w_px, h_px = img.size
print(f"處理第 {idx+1}/{len(images)} 頁 (尺寸: {w_px}x{h_px})...")
# 呼叫 Qwen2.5-VL 進行 Visual Grounding (視覺定位)
# 這裡會給大模型下達 Prompt 提取文字內容與相對於圖片的 0-1000 座標
prompt = (
"你是一個精準的簡報版面分析專家。請分析這張簡報圖片,找出所有的文字區塊。請精確定位每個文字區塊的 Bounding Box,並輸出其代表的文字內容。請嚴格以下列 JSON 格式輸出,不要包含任何額外的 Markdown 標籤或說明文字:\n"
"{\n"
" \"width\": 圖片總寬度,\n"
" \"height\": 圖片總高度,\n"
" \"blocks\": [\n"
" {\"text\": \"文字內容1\", \"box_2d\": [ymin, xmin, ymax, xmax], \"type\": \"title/content\"}\n"
" ]\n"
"}\n"
"注意:座標 [ymin, xmin, ymax, xmax] 必須是相對於圖片寬高的 0-1000 正規化數值。"
)
# 使用 transformers 處理圖像與 prompt
inputs = processor(text=[prompt], images=[img], padding=True, return_tensors="pt").to("cuda")
generated_ids = model.generate(**inputs, max_new_tokens=1024)
generated_text = processor.batch_decode(generated_ids, skip_special_tokens=True)[0]
# 解析 JSON 結果
try:
# 去除 markdown 標記
cleaned_text = generated_text.strip()
if cleaned_text.startswith("```json"):
cleaned_text = cleaned_text[7:]
if cleaned_text.endswith("```"):
cleaned_text = cleaned_text[:-3]
ocr_data = json.loads(cleaned_text.strip())
except Exception as e:
print(f"解析 JSON 失敗,嘗試後備機制: {e}")
ocr_data = {"width": w_px, "height": h_px, "blocks": []}
# 建立去字遮罩 (Mask)
mask = np.zeros((h_px, w_px), dtype=np.uint8)
blocks_list = []
for block in ocr_data.get("blocks", []):
ymin, xmin, ymax, xmax = block["box_2d"]
# 將 0-1000 歸一化座標轉回實際像素
py_min = int(ymin * h_px / 1000)
px_min = int(xmin * w_px / 1000)
py_max = int(ymax * h_px / 1000)
px_max = int(xmax * w_px / 1000)
# 依文字高度自適應膨脹
box_h = py_max - py_min
dilation = max(2, int(round(box_h * 0.08)))
px_min = max(0, px_min - dilation)
py_min = max(0, py_min - dilation)
px_max = min(w_px, px_max + dilation)
py_max = min(h_px, py_max + dilation)
cv2.rectangle(mask, (px_min, py_min), (px_max, py_max), 255, -1)
blocks_list.append({
"text": block["text"],
"box_2d": [ymin, xmin, ymax, xmax],
"type": block.get("type", "content")
})
# 執行 GPU Inpainting 去字修補
# 轉換為 PIL Image 用於 diffusers
pil_mask = Image.fromarray(mask)
# 用穩定擴散 (SDXL/LaMa) 修復背景
clean_img = pipe(prompt="clean slide background", image=img, mask_image=pil_mask).images[0]
clean_img_path = os.path.join(temp_dir, clean_img_name)
clean_img.save(clean_img_path, "PNG")
metadata["slides"].append({
"slide_index": idx,
"bg_image_name": clean_img_name,
"width": w_px,
"height": h_px,
"blocks": blocks_list
})
# 寫入 metadata.json
with open(os.path.join(temp_dir, "metadata.json"), "w", encoding="utf-8") as f:
json.dump(metadata, f, ensure_ascii=False, indent=2)
# 打包為 ZIP
with zipfile.ZipFile(output_zip_path, 'w', zipfile.ZIP_DEFLATED) as zip_file:
for root, dirs, files in os.walk(temp_dir):
for file in files:
if file.endswith("clean.png") or file == "metadata.json":
file_path = os.path.join(root, file)
zip_file.write(file_path, os.path.basename(file_path))
# 清理暫存資料夾
shutil.rmtree(temp_dir, ignore_errors=True)
print(f"轉換打包成功!壓縮包已儲存至: {output_zip_path}")
# 執行範例:
# process_pdf_to_zip("input.pdf", "Pack.zip")
```
|