File size: 5,043 Bytes
f0e1c1f b3b98e8 77ad3e2 b3b98e8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 | # This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load
import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
# Input data files are available in the read-only "../input/" directory
# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory
import os
for dirname, _, filenames in os.walk('/kaggle/input'):
for filename in filenames:
print(os.path.join(dirname, filename))
# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using "Save & Run All"
# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session
get_ipython().getoutput("pip install nltk")
import torch
from transformers import BlipProcessor, BlipForConditionalGeneration
from nltk.translate.bleu_score import sentence_bleu, SmoothingFunction
from tqdm import tqdm
device = "cuda" if torch.cuda.is_available() else "cpu"
model_name = "utkarshpise/blip-ucm-captioning"
processor = BlipProcessor.from_pretrained(model_name)
model = BlipForConditionalGeneration.from_pretrained(model_name)
model.to(device)
model.eval()
print(" Model loaded")
smooth = SmoothingFunction().method1
def evaluate_model(model, loader, processor, device):
model.eval()
total_loss = 0
preds = []
refs = []
with torch.no_grad():
for batch in tqdm(loader, desc="Evaluating"):
batch = {k: v.to(device) for k, v in batch.items()}
# π₯ LM LOSS
outputs = model(**batch)
loss = outputs.loss
total_loss += loss.item()
# π₯ Generate captions
generated_ids = model.generate(
pixel_values=batch["pixel_values"],
max_length=50,
num_beams=5
)
pred = processor.batch_decode(generated_ids, skip_special_tokens=True)
ref = processor.batch_decode(batch["labels"], skip_special_tokens=True)
preds.extend(pred)
refs.extend(refs if False else ref)
avg_loss = total_loss / len(loader)
bleu_scores = []
for p, r in zip(preds, refs):
score = sentence_bleu([r.split()], p.split(), smoothing_function=smooth)
bleu_scores.append(score)
bleu = sum(bleu_scores) / len(bleu_scores)
return avg_loss, bleu
import kagglehub
import os
import json
path = kagglehub.dataset_download("sumanpaul14/ucm-captioning-dataset")
print("Dataset path:", path)
print("Files:", os.listdir(path))
from torch.utils.data import Dataset, DataLoader, random_split
from PIL import Image
import json
import os
BASE_PATH = "/kaggle/input/datasets/sumanpaul14/ucm-captioning-dataset"
DATA_JSON = os.path.join(BASE_PATH, "dataset.json")
IMG_DIR = os.path.join(BASE_PATH, "imgs", "imgs")
# load json
with open(DATA_JSON) as f:
data = json.load(f)
samples = []
for item in data["images"]:
img_path = os.path.join(IMG_DIR, item["filename"])
if os.path.exists(img_path):
for sent in item["sentences"]:
samples.append({
"image": img_path,
"caption": sent["raw"]
})
class UCMCaptionDataset(Dataset):
def __init__(self, samples, processor):
self.samples = samples
self.processor = processor
def __len__(self):
return len(self.samples)
def __getitem__(self, idx):
item = self.samples[idx]
image = Image.open(item["image"]).convert("RGB")
caption = item["caption"]
encoding = self.processor(
images=image,
text=caption,
padding="max_length",
truncation=True,
return_tensors="pt"
)
encoding = {k: v.squeeze(0) for k, v in encoding.items()}
encoding["labels"] = encoding["input_ids"]
return encoding
dataset = UCMCaptionDataset(samples, processor)
train_size = int(0.8 * len(dataset))
test_size = len(dataset) - train_size
_, test_dataset = random_split(dataset, [train_size, test_size])
test_loader = DataLoader(test_dataset, batch_size=8)
loss, bleu = evaluate_model(model, test_loader, processor, device)
print(f"\nLM Loss: {loss}")
print(f" BLEU Score: {bleu}")
import os
print(os.listdir("/kaggle/working"))
from huggingface_hub import login
login()
from huggingface_hub import upload_file
repo_id = "utkarshpise/blip-ucm-captioning"
upload_file(
path_or_fileobj="/kaggle/working/.virtual_documents/__notebook_source__.ipynb",
path_in_repo="inference.py",
repo_id=repo_id,
repo_type="model"
)
print(" Code uploaded!")
Evaluating: 100%|ββββββββββ| 263/263 [09:11<00:00, 2.10s/it]
LM Loss: 0.43680915043834495
BLEU Score: 0.09948045741442776
|