dreamsim-ensemble / configuration_dreamsim.py
bigshanedogg's picture
Upload folder using huggingface_hub
f918a65 verified
Raw
History Blame Contribute Delete
1.99 kB
# DreamSim (HuggingFace format) — unofficial port.
# Copyright (c) 2026 bigshanedogg. Released under the MIT License (see LICENSE).
#
# Derivative of DreamSim (MIT, (c) 2023 Shobhita Sundaram, Netanel Tamir,
# Stephanie Fu, Richard Zhang — https://github.com/ssundaram21/dreamsim).
# Not an official DreamSim release.
"""HF configuration for the DreamSim perceptual-similarity ensemble.
Mirrors the upstream ``dreamsim`` ensemble settings (github ssundaram21/dreamsim,
MIT): three ViT-B/16 backbones (DINO / CLIP / OpenCLIP) whose extracted features are
concatenated and mean/L2-normalized. LoRA is merged into the backbone weights at
conversion time, so this config describes only the resulting architecture.
"""
from transformers import PretrainedConfig
class DreamSimConfig(PretrainedConfig):
model_type = "dreamsim"
def __init__(
self,
dreamsim_type: str = "ensemble",
model_types: str = "dino_vitb16,clip_vitb16,open_clip_vitb16",
feat_types: str = "cls,embedding,embedding",
strides: str = "16,16,16",
img_size: int = 224,
embed_size: int = 1792,
normalize_embeds: bool = True,
lora_merged: bool = True,
**kwargs,
):
# Comma-separated per-backbone specs (same length): the base ViT, which feature
# to extract (cls / embedding), and the patch stride. ``embed_size`` is the
# concatenated feature dim (dino cls 768 + clip embedding 512 + open_clip
# embedding 512 = 1792). ``lora_merged`` records that the shipped weights already
# have the DreamSim LoRA folded in (so no peft is needed at load).
self.dreamsim_type = dreamsim_type
self.model_types = model_types
self.feat_types = feat_types
self.strides = strides
self.img_size = img_size
self.embed_size = embed_size
self.normalize_embeds = normalize_embeds
self.lora_merged = lora_merged
super().__init__(**kwargs)