Q-ReAlign-0.8B / app.py
VOIDER's picture
Update app.py
9fb0e64 verified
Raw
History Blame Contribute Delete
4.36 kB
import time
import torch
import gradio as gr
from PIL import Image
from transformers import AutoModelForImageTextToText, AutoProcessor
CKPT = "q-future/Q-ReAlign-Mini-0.8B"
LEVELS = ["excellent", "good", "fair", "poor", "bad"]
WEIGHTS = [1.0, 0.75, 0.5, 0.25, 0.0]
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"[{time.strftime('%X')}] Loading model on {device}...")
processor = AutoProcessor.from_pretrained(CKPT)
model = AutoModelForImageTextToText.from_pretrained(CKPT, dtype="auto").to(device).eval()
print(f"[{time.strftime('%X')}] Model loaded successfully.")
def score_image(image, task_type):
if image is None:
yield "Error", "Please upload an image."
return
start_time = time.time()
# ЭТАП 1: Оптимизация размера изображения (Критично для CPU)
yield "0.0000", "⏳ Step 1: Resizing image to prevent CPU overload..."
max_size = 768 # Ограничиваем максимальную сторону, сохраняя пропорции
image.thumbnail((max_size, max_size))
print(f"[{time.strftime('%X')}] Image resized to {image.size}")
if task_type == "Quality (IQA)":
prompt = "How would you rate the quality of this image?"
stem = "The quality of the image is"
else:
prompt = "How would you rate the aesthetics of this image?"
stem = "The aesthetics of the image is"
messages = [{"role": "user", "content": [
{"type": "image"},
{"type": "text", "text": prompt},
]}]
# ЭТАП 2: Подготовка токенов
yield "0.0000", "⏳ Step 2: Preprocessing and tokenizing..."
text = processor.apply_chat_template(messages, add_generation_prompt=True) + stem
inputs = processor(
text=[text],
images=[image.convert("RGB")],
return_tensors="pt"
).to(device)
print(f"[{time.strftime('%X')}] Preprocessing done. Input shape: {inputs['input_ids'].shape}")
ids = [processor.tokenizer(" " + w, add_special_tokens=False).input_ids[0] for w in LEVELS]
# ЭТАП 3: Прогон через нейросеть
# Здесь нельзя добавить прогресс-бар, так как это одна атомарная C++ операция PyTorch
yield "0.0000", "⏳ Step 3: Running neural network forward pass (This takes the longest)..."
print(f"[{time.strftime('%X')}] Starting forward pass...")
with torch.no_grad():
outputs = model(**inputs)
print(f"[{time.strftime('%X')}] Forward pass completed!")
# ЭТАП 4: Расчет оценки
yield "0.0000", "⏳ Step 4: Calculating final score..."
probs = outputs.logits[0, -1, ids].softmax(-1)
score = (probs * torch.tensor(WEIGHTS, device=device)).sum().item()
total_time = time.time() - start_time
print(f"[{time.strftime('%X')}] Done! Score: {score:.4f}. Total time: {total_time:.1f}s")
# ИТОГ
yield f"{score:.4f}", f"✅ Done in {total_time:.1f} seconds."
with gr.Blocks(title="Q-ReAlign Mini (0.8B)") as demo:
gr.Markdown("# Q-ReAlign Mini (0.8B) Image Judge")
gr.Markdown(
"Rates the perceptual quality or aesthetics of an image using the Qwen3.5-VL multimodal model. "
"The result is a score ranging from 0 (worst) to 1 (best)."
)
with gr.Row():
with gr.Column():
input_image = gr.Image(type="pil", label="Input Image")
task_dropdown = gr.Dropdown(
choices=["Quality (IQA)", "Aesthetics (IAA)"],
value="Quality (IQA)",
label="Task"
)
submit_btn = gr.Button("Evaluate", variant="primary")
with gr.Column():
output_score = gr.Textbox(label="Final Score", lines=1)
status_box = gr.Textbox(label="Status Log", lines=1, interactive=False)
# Используем outputs в виде списка, чтобы обновлять и оценку, и лог
submit_btn.click(
fn=score_image,
inputs=[input_image, task_dropdown],
outputs=[output_score, status_box]
)
if __name__ == "__main__":
demo.launch()