# -*- coding: utf-8 -*- """ 使用方法: python rule_11_vllm.py \ --input_dir "/path/to/your/images" \ --model_path "/path/to/your/Qwen2.5-VL-7B-Instruct" """ import os import re import json import argparse import multiprocessing from pathlib import Path from typing import Dict, List, Optional, Any from tqdm import tqdm from PIL import Image, ImageFile from transformers import AutoProcessor from vllm import LLM, SamplingParams os.environ['VLLM_WORKER_MULTIPROC_METHOD'] = 'spawn' ImageFile.LOAD_TRUNCATED_IMAGES = True IMG_EXTS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".tif", ".tiff"} SYS_PROMPT_TEXT ="""You are a highly critical Senior Art Director and Visual Auditor. Your task is to evaluate "Text Visual Weight & Layout Balance" to prevent visual overcrowding while allowing for artistic typographic choices. INPUT: One image and one natural-language question about text density or layout balance. YOUR TASK: 1. Analyze the visual weight of the text relative to the canvas (Area coverage + Visual heaviness). 2. Apply the "Aesthetic Filter": Distinguish between "Cheap Da Zi Bao" (Violation) and "High-End Artistic Text" (Safe). 3. Determine if the image is a **VIOLATION** (Unsuitable) or **SAFE** (Suitable). 4. Output a JSON object containing a rigorous Chain-of-Thought ("think") and a precise classification label ("answer"). OUTPUT FORMAT: Return EXACTLY two blocks, no extra text: Detailed reasoning steps: 1. Estimate text area coverage -> 2. Assess design quality (Suffocating vs. Artistic) -> 3. Check for product obstruction...{"Answer": "", "Answer type": "Text Visual Weight"} ========================================= CORE PRINCIPLE: BALANCE VS. SUFFOCATION ========================================= - **The Rule:** Marketing text should generally occupy < 25% of the visual weight. - **The Exception:** Large text IS allowed if it is "Concise, Exquisite, and High-End" (Magazine Style). - **The Prohibition:** Large text is FORBIDDEN if it is "Crowded, Aggressive, and Cheap" (Da Zi Bao Style). - **Maximum Text Density:** Regardless of artistic quality, any image containing more than 6 lines of narrative text or 50 words is automatically a VIOLATION (Information Overload). - **Literal Line Counting:** Each line in a bulleted list or paragraph counts as 1 line. A neatly organized list of 10 lines is still a VIOLATION of the 6-line limit. ========================================= STRICT DECISION HIERARCHY (FOLLOW IN ORDER) ========================================= 1. HARD LIMIT CHECK: - Does the image have > 6 lines of text total (including text inside phone/UI)? - If YES -> Label: UNSUITABLE (Reason: Text Density Overload). - Zero UI Exemption: Text inside phone screens or UI mockups is NOT background decoration; it is active text weight. If the phone screen is filled with more than 4-5 lines of content, the entire image is likely UNSUITABLE. 2. VISUAL WEIGHT CHECK: - Does the text (and its background boxes/screens) occupy more than 30% of the canvas? - If YES -> Label: UNSUITABLE (Reason: Excessive Visual Weight). 3. AESTHETIC FILTER (The "Premium" Test): - Is it "Artistic Exception"? ONLY if text is < 3 lines AND elegantly integrated. - Note: A phone screen filled with tiny text is NEVER "Artistic" or "High-End" in an ad context; it is a "Manual Page" (UNSUITABLE). ========================================= CRITERIA FOR 'UNSUITABLE' (VIOLATION / OVERWHELMING) ========================================= 1. **Aggressive "Da Zi Bao" (大字报) Style:** - **Visual Suffocation:** Massive, bold text occupies the central area with zero "breathing room" (negative space). - **Cheap Aesthetic:** It looks like a spam flyer or a shouting warning sign rather than a professional ad. - **Shouting Effect:** The font size is absurdly large relative to the canvas without any artistic justification. 2. **Visual Obstruction & Imbalance:** - **Blocking the Hero:** Text covers the main product, model's face, or key visual storytelling elements. - **Excessive Weight:** The text area visually dominates > 30-40% of the canvas in a messy, cluttered way. 3.**The "Manual/Article" Trap:** - Images that look like an instruction manual page, a reading app screenshot, or a news article are automatically UNSUITABLE. Ads must remain "Visual-First," not "Text-First." ========================================= CRITERIA FOR 'SUITABLE' (SAFE / BALANCED) ========================================= 1. **Standard Good Ratio:** - **Balanced:** Text occupies a reasonable area (roughly < 25% of visual weight). - **Clear Hierarchy:** The Product/Illustration is the HERO; the Text is the SUPPORT. 2. **The "Artistic Exception" (High-End Large Text):** - **Premium Look:** Even if the headline is large, it is concise, elegant, and integrated well with the background. - **Breathing Room:** The layout maintains generous margins and negative space. It feels like a Vogue cover or a movie poster, not a supermarket discount flyer. ========================================= DECISION LOGIC ========================================= - **Unsuitable**: If the text creates a "suffocating" effect, blocks the product, or looks like a cheap, crowded "Da Zi Bao". - **Suitable**: If the text is minimal (<25%), OR if it is large but designed with high artistic quality and ample negative space. """ def collect_images(input_dir: Path) -> List[Dict[str, str]]: if not input_dir.exists(): raise FileNotFoundError(f"Input directory not found: {input_dir}") files = [p for p in input_dir.iterdir() if p.is_file() and p.suffix.lower() in IMG_EXTS] files.sort() print(f"[Info] Found {len(files)} images in {input_dir}") return [{"path": str(p), "filename": p.name} for p in files] def parse_llm_output(text: str) -> Dict[str, Any]: default_res = { "label": "Parse Error", "think": "No reasoning found", "raw": text } if not text: return default_res think_match = re.search(r'(.*?)', text, re.DOTALL) think_content = think_match.group(1).strip() if think_match else "" answer_match = re.search(r'(.*?)', text, re.DOTALL) extracted_label = "Parse Error" if answer_match: json_str = answer_match.group(1).strip() try: data = json.loads(json_str) raw_ans = data.get("Answer", "") if "unsuitable" in raw_ans.lower(): extracted_label = "Unsuitable" elif "suitable" in raw_ans.lower(): extracted_label = "Suitable" else: extracted_label = raw_ans except json.JSONDecodeError: if "Unsuitable" in json_str: extracted_label = "Unsuitable" elif "Suitable" in json_str: extracted_label = "Suitable" else: if "Unsuitable" in text: extracted_label = "Unsuitable" elif "Suitable" in text: extracted_label = "Suitable" return { "label": extracted_label, "think": think_content, "raw": text } def prepare_vllm_inputs(batch_meta: List[Dict], processor) -> List[Dict]: vllm_inputs = [] user_query = "Analyze this image against the design rules and return the JSON decision." for item in batch_meta: img_path = item["path"] try: image_obj = Image.open(img_path).convert("RGB") messages = [ {"role": "system", "content": [{"type": "text", "text": SYS_PROMPT_TEXT}]}, {"role": "user", "content": [ {"type": "image", "image": img_path}, {"type": "text", "text": user_query} ]} ] prompt_text = processor.apply_chat_template( messages, tokenize=False, add_generation_prompt=True ) vllm_inputs.append({ "prompt": prompt_text, "multi_modal_data": {"image": image_obj} }) except Exception as e: print(f"[Warning] Failed to load {img_path}: {e}") vllm_inputs.append(None) return vllm_inputs def main(): parser = argparse.ArgumentParser(description="AI Visual Comfort Auditor") parser.add_argument("--input_dir", type=str, required=True, help="Folder containing images to check") parser.add_argument("--model_path", type=str, required=True, help="Path to local Qwen-VL model") parser.add_argument("--batch_size", type=int, default=512, help="Inference batch size") parser.add_argument("--tp_size", type=int, default=2, help="Tensor Parallel size") args = parser.parse_args() input_path = Path(args.input_dir) meta_data = collect_images(input_path) if not meta_data: print("[Info] No images found. Exiting.") return # --------------------------- # 初始化模型 # --------------------------- print(f"\n[Init] Loading Model: {args.model_path}") llm = LLM( model=args.model_path, tokenizer=args.model_path, trust_remote_code=True, tensor_parallel_size=args.tp_size, gpu_memory_utilization=0.90, max_model_len=8192, enforce_eager=True, limit_mm_per_prompt={"image": 1} ) processor = AutoProcessor.from_pretrained(args.model_path, trust_remote_code=True) # 采样参数 sampling_params = SamplingParams( temperature=0.7, # 稍微降低温度以获得更稳定的分类 max_tokens=1024, top_p=0.9 ) # --------------------------- # 批量推理 # --------------------------- results = [] print(f"\n[Run] Starting Inference on {len(meta_data)} images...") for i in tqdm(range(0, len(meta_data), args.batch_size), desc="Processing Batches"): batch_meta = meta_data[i : i + args.batch_size] batch_inputs = prepare_vllm_inputs(batch_meta, processor) valid_inputs = [inp for inp in batch_inputs if inp is not None] valid_indices = [idx for idx, inp in enumerate(batch_inputs) if inp is not None] if not valid_inputs: continue outputs = llm.generate(valid_inputs, sampling_params=sampling_params, use_tqdm=False) for local_idx, out in enumerate(outputs): original_meta = batch_meta[valid_indices[local_idx]] generated_text = out.outputs[0].text # 解析结果 parsed = parse_llm_output(generated_text) results.append({ "filename": original_meta["filename"], "path": original_meta["path"], "label": parsed["label"], # Suitable / Unsuitable "think": parsed["think"], # 思维链 "raw_output": generated_text }) # --------------------------- # 统计与输出 # --------------------------- total = len(results) unsuitable_count = sum(1 for r in results if r["label"] == "Unsuitable") suitable_count = sum(1 for r in results if r["label"] == "Suitable") error_count = total - unsuitable_count - suitable_count unsuitable_rate = (unsuitable_count / total * 100) if total > 0 else 0 suitable_rate = (suitable_count / total * 100) if total > 0 else 0 print("\n" + "="*60) print(f"AUDIT REPORT FOR: {input_path.name}") print("="*60) print(f"{'Total Images':<25}: {total}") print("-" * 60) print(f"{'UNSUITABLE (Violation)':<25}: {unsuitable_count} ({unsuitable_rate:.2f}%)") print(f"{'SUITABLE (Safe)':<25}: {suitable_count} ({suitable_rate:.2f}%)") print(f"{'Parse Errors':<25}: {error_count}") print("="*60) # 保存结果 output_file = input_path / f"audit_result_{input_path.name}.json" try: with open(output_file, "w", encoding="utf-8") as f: json.dump(results, f, ensure_ascii=False, indent=2) print(f"\n[Done] Detailed JSON report saved to:\n-> {output_file}") except Exception as e: print(f"[Error] Could not save JSON: {e}") if __name__ == "__main__": try: multiprocessing.set_start_method('spawn', force=True) except RuntimeError: pass main()