Whyx-PROmpTea / src /dedup_engine.py
ArtShumov's picture
feat(prod): rewrite pipeline + NoobAI + ensemble tagger (3xWD14+DeepDanbooru-ready) + 1000 artists + negative templates + history ext + scoring config
e6404d0
Raw
History Blame Contribute Delete
2.91 kB
from difflib import SequenceMatcher
from src.synonym_data import _load_groups, _find_synonym_group
def _dedup_exact(tags: list[str]) -> list[str]:
seen = set()
out = []
for t in tags:
k = t.lower().strip()
if k and k not in seen:
seen.add(k)
out.append(t)
return out
def _dedup_synonym_groups(tags: list[str]) -> list[str]:
result = []
for t in tags:
tl = t.lower().strip()
group = _find_synonym_group(t)
if group is None:
result.append(t)
continue
conflict = False
for existing in result:
el = existing.lower().strip()
if el in group and el != tl:
conflict = True
break
if not conflict:
result.append(t)
return result
def _ngram_similarity(a: str, b: str) -> float:
return SequenceMatcher(None, a.lower(), b.lower()).ratio()
def _is_attribute_variant(a: str, b: str) -> bool:
"""Two tags that share a head or a tail but differ in the other part denote
different attributes (e.g. 'blue eyes' vs 'blue hair', 'red dress' vs 'blue
dress') and must both be kept — fuzzy-merging them would drop a distinct
booru attribute."""
aw = a.lower().split()
bw = b.lower().split()
if len(aw) < 2 or len(bw) < 2 or len(aw) != len(bw):
return False
head_same = aw[:-1] == bw[:-1]
tail_same = aw[-1] == bw[-1]
# exactly one side differs -> different attribute, same concept
return head_same != tail_same
def _dedup_fuzzy(tags: list[str], threshold: float = 0.85) -> list[str]:
# Sort by length so a more specific (longer) tag survives its shorter fuzzy
# twin (e.g. "very long flowing red hair" beats "long red hair").
sorted_tags = sorted(tags, key=len, reverse=True)
result = []
for t in sorted_tags:
if len(t) < 4:
result.append(t)
continue
is_dup = False
for existing in result:
if len(existing) < 4:
continue
if _is_attribute_variant(t, existing):
continue
# Cheap upper bound first — a length mismatch alone can decide.
ratio = SequenceMatcher(None, t.lower(), existing.lower()).ratio()
if ratio >= threshold:
is_dup = True
break
if not is_dup:
result.append(t)
return result
def smart_dedup(tags: list[str], model: str = "anima") -> list[str]:
if not tags:
return []
no_exact = _dedup_exact(tags)
no_synonym = _dedup_synonym_groups(no_exact)
threshold = 0.75 if model == "anima" else 0.85
no_fuzzy = _dedup_fuzzy(no_synonym, threshold=threshold)
return no_fuzzy
def reload_groups():
from src.synonym_data import reload_synonym_groups
reload_synonym_groups()
_load_groups()