| |
| """ |
| Generic EFUF data formatter for any relation. |
| |
| Reads all_captions.json + the clean_captions ckpt.json and produces: |
| pos_neg_synthetic_{train,val}.json – per-subsentence pos/neg entries |
| sentences_synthetic_{train,val}.json – whole-caption positive entries |
| dummy_vqa.json – minimal VQA stub |
| |
| Image split (train vs val) is determined by which subdirectory the JPEG was |
| saved to by build_hf_dataset.py (images/train/ or images/val/). |
| |
| Usage: |
| # Single relation |
| cd /data/caotue/multilayer-sae |
| python EFUF/scripts/format_efuf_data.py --relation kitchen_oven |
| |
| # All 4 relations |
| python EFUF/scripts/format_efuf_data.py |
| """ |
|
|
| from __future__ import annotations |
|
|
| import argparse |
| import json |
| import os |
| import re |
| import sys |
|
|
| sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "../..")) |
| from experiment.config.relation_config import RelationConfig, get_relation_config, list_relation_keys |
|
|
| EFUF_DATA = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "data") |
| SUBSENTENCE_SPLITTER = ",.;!?:" |
| NEGATIVE_SCORE = 0.0 |
| POSITIVE_SCORE = 40.0 |
| SENTENCE_MEAN_POS = 40.0 |
| SENTENCE_MIN_POS = 35.0 |
|
|
|
|
| def find_subsentence(text: str, target_words: list[str]) -> tuple[int, int] | None: |
| """Return (start, end) char indices of the subsentence containing a target word.""" |
| for sub in re.split(f"[{SUBSENTENCE_SPLITTER}]+", text): |
| for word in target_words: |
| if re.search(r"\b" + re.escape(word) + r"\b", sub, re.IGNORECASE): |
| start = text.find(sub) |
| return start, start + len(sub) |
| return None |
|
|
|
|
| def mentions_object(caption: str, target_words: list[str]) -> bool: |
| for word in target_words: |
| if re.search(r"\b" + re.escape(word) + r"\b", caption, re.IGNORECASE): |
| return True |
| return False |
|
|
|
|
| def get_image_relpath(image_id: str, img_train_dir: str, img_val_dir: str) -> str | None: |
| if os.path.exists(os.path.join(img_train_dir, f"{image_id}.jpg")): |
| return f"train/{image_id}.jpg" |
| if os.path.exists(os.path.join(img_val_dir, f"{image_id}.jpg")): |
| return f"val/{image_id}.jpg" |
| return None |
|
|
|
|
| def format_relation(relation: str) -> None: |
| rc = get_relation_config(relation) |
| data_dir = os.path.join(EFUF_DATA, relation) |
| captions_path = os.path.join(data_dir, "all_captions.json") |
| ckpt_path = captions_path + ".ckpt.json" |
| img_train_dir = os.path.join(data_dir, "images", "train") |
| img_val_dir = os.path.join(data_dir, "images", "val") |
|
|
| if not os.path.exists(captions_path): |
| print(f"[{relation}] ERROR: {captions_path} not found — run build_hf_dataset.py first") |
| return |
| if not os.path.exists(ckpt_path): |
| print(f"[{relation}] ERROR: {ckpt_path} not found — run clean_captions.py first") |
| return |
|
|
| with open(captions_path) as f: |
| all_captions: list[dict] = json.load(f) |
| with open(ckpt_path) as f: |
| ckpt: dict = json.load(f) |
|
|
| target_words = rc.mention_keywords |
| scene_key = rc.scene_key |
| object_key = rc.object_key |
|
|
| step2_judge: dict[str, str] = ckpt.get("step2_judge_done", {}) |
|
|
| pos_neg_entries: list[dict] = [] |
| sentence_entries: list[dict] = [] |
| stats = {"no_caption": 0, "no_image": 0, "no_mention": 0, "not_hallucinating": 0, "no_subsentence": 0} |
|
|
| for item in all_captions: |
| image_id = item["image_id"] |
| caption: str = item.get("llava_caption", "") |
| scene: int = item.get(scene_key, 0) |
| obj: int = item.get(object_key, 0) |
|
|
| if not caption: |
| stats["no_caption"] += 1 |
| continue |
|
|
| rel_path = get_image_relpath(image_id, img_train_dir, img_val_dir) |
| if rel_path is None: |
| stats["no_image"] += 1 |
| continue |
|
|
| has_mention = mentions_object(caption, target_words) |
|
|
| if obj == 0 and scene == 1: |
| |
| if not has_mention: |
| |
| sentence_entries.append({ |
| "image": rel_path, |
| "sentence": caption, |
| "mean": SENTENCE_MEAN_POS, |
| "min": SENTENCE_MIN_POS, |
| }) |
| continue |
|
|
| |
| if step2_judge.get(image_id) != "YES": |
| stats["not_hallucinating"] += 1 |
| continue |
|
|
| result = find_subsentence(caption, target_words) |
| if result is None: |
| stats["no_subsentence"] += 1 |
| continue |
| start, end = result |
|
|
| pos_neg_entries.append({ |
| "image": rel_path, |
| "sentence": caption[:end].rstrip(), |
| "position": start, |
| "score": NEGATIVE_SCORE, |
| "_type": "negative", |
| }) |
|
|
| elif obj == 1: |
| |
| if not has_mention: |
| |
| sentence_entries.append({ |
| "image": rel_path, |
| "sentence": caption, |
| "mean": SENTENCE_MEAN_POS, |
| "min": SENTENCE_MIN_POS, |
| }) |
| continue |
|
|
| result = find_subsentence(caption, target_words) |
| if result is None: |
| stats["no_subsentence"] += 1 |
| continue |
| start, end = result |
|
|
| pos_neg_entries.append({ |
| "image": rel_path, |
| "sentence": caption[:end].rstrip(), |
| "position": start, |
| "score": POSITIVE_SCORE, |
| "_type": "positive", |
| }) |
| sentence_entries.append({ |
| "image": rel_path, |
| "sentence": caption, |
| "mean": SENTENCE_MEAN_POS, |
| "min": SENTENCE_MIN_POS, |
| }) |
|
|
| else: |
| |
| if scene == 0 and obj == 0: |
| sentence_entries.append({ |
| "image": rel_path, |
| "sentence": caption, |
| "mean": SENTENCE_MEAN_POS, |
| "min": SENTENCE_MIN_POS, |
| }) |
|
|
| neg_count = sum(1 for e in pos_neg_entries if e["_type"] == "negative") |
| pos_count = sum(1 for e in pos_neg_entries if e["_type"] == "positive") |
| print(f"[{relation}] pos_neg: {len(pos_neg_entries)} ({pos_count} pos, {neg_count} neg)") |
| print(f"[{relation}] sentences: {len(sentence_entries)}") |
| print(f"[{relation}] skipped: {stats}") |
|
|
| |
| final_pos_neg = [{k: v for k, v in e.items() if k != "_type"} for e in pos_neg_entries] |
|
|
| def split_by_folder(entries: list[dict]) -> tuple[list[dict], list[dict]]: |
| train = [e for e in entries if e["image"].startswith("train/")] |
| val = [e for e in entries if e["image"].startswith("val/")] |
| return train, val |
|
|
| train_pn, val_pn = split_by_folder(final_pos_neg) |
| train_sent, val_sent = split_by_folder(sentence_entries) |
|
|
| print(f"[{relation}] train pos_neg: {len(train_pn)}, val pos_neg: {len(val_pn)}") |
| print(f"[{relation}] train sent: {len(train_sent)}, val sent: {len(val_sent)}") |
|
|
| for name, data in [ |
| ("pos_neg_synthetic_train", train_pn), |
| ("pos_neg_synthetic_val", val_pn), |
| ("sentences_synthetic_train", train_sent), |
| ("sentences_synthetic_val", val_sent), |
| ("pos_neg_synthetic", final_pos_neg), |
| ("sentences_synthetic", sentence_entries), |
| ]: |
| path = os.path.join(data_dir, f"{name}.json") |
| with open(path, "w") as f: |
| json.dump(data, f, indent=2, ensure_ascii=False) |
|
|
| |
| |
| first_train = next((e for e in pos_neg_entries + sentence_entries if e["image"].startswith("train/")), None) |
| first_rel = first_train["image"] if first_train else (final_pos_neg + sentence_entries)[0]["image"] |
| first_abs = os.path.join(img_train_dir if first_rel.startswith("train/") else img_val_dir, |
| os.path.basename(first_rel)) |
| dummy_vqa = [{ |
| "input": "Describe this image.", |
| "output": all_captions[0].get("llava_caption", "An image.")[:100], |
| "image": first_abs, |
| }] |
| vqa_path = os.path.join(data_dir, "dummy_vqa.json") |
| with open(vqa_path, "w") as f: |
| json.dump(dummy_vqa, f, indent=2) |
|
|
| print(f"[{relation}] Saved all files to {data_dir}") |
|
|
|
|
| def main() -> None: |
| ap = argparse.ArgumentParser(description="Format EFUF training data for any relation") |
| ap.add_argument("--relation", default=None, help="Single relation key; omit for all 4") |
| args = ap.parse_args() |
|
|
| relations = [args.relation] if args.relation else list_relation_keys() |
| for rel in relations: |
| format_relation(rel) |
|
|
| print("\nDone.") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|