File size: 7,552 Bytes
5bfaada
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
#!/usr/bin/env python3
import json
import sys
import re
import argparse
from pathlib import Path
from collections import Counter

# Set of expected categories and their exact counts
EXPECTED_COUNTS = {
    "definicao": 10,
    "conceptual": 10,
    "procedimento": 10,
    "debug": 8,
    "aplicacao": 5,
    "fora_scope": 5,
    "ambigua": 2
}

KNOWN_RANGES = {
    "tacs": {
        "default": (4, 133),
        "variáveis e scope": (7, 9),
        "assincronia: promises": (119, 121),
    },
    "semio": {
        "default": (1, 60),
        "resumo sc": (17, 28),
        "semiótica resumos": (1, 30),
    }
}

def clean_text(text):
    return re.sub(r'\s+', ' ', (text or '').lower().strip())

def validate_dataset(new_dataset_path: Path, old_dataset_path: Path | None, subject: str) -> bool:
    print(f"--- Validating dataset: {new_dataset_path} ---")
    if not new_dataset_path.exists():
        print(f"Error: File {new_dataset_path} does not exist.")
        return False

    try:
        with open(new_dataset_path, "r", encoding="utf-8") as f:
            data = json.load(f)
    except json.JSONDecodeError as e:
        print(f"Error parsing JSON: {e}")
        return False

    if not isinstance(data, list):
        print("Error: Dataset must be a list of items.")
        return False

    # 1. 50 IDs únicos e consecutivos
    if len(data) != 50:
        print(f"Error: Dataset must have exactly 50 items (found {len(data)}).")
        return False

    ids = [item.get("id") for item in data]
    if any(i is None for i in ids):
        print("Error: Some items do not have an ID.")
        return False

    sorted_ids = sorted(ids)
    if sorted_ids != list(range(1, 51)):
        print(f"Error: IDs must be unique and consecutive from 1 to 50. Found: {sorted_ids}")
        return False

    # 2. Distribuição exata
    categories = [item.get("categoria") for item in data]
    counts = Counter(categories)
    for cat, expected in EXPECTED_COUNTS.items():
        if counts[cat] != expected:
            print(f"Error: Category '{cat}' count is {counts[cat]}, but expected {expected}.")
            return False

    # Load old dataset for duplicate/semantic check
    old_data = []
    if old_dataset_path and old_dataset_path.exists():
        with open(old_dataset_path, "r", encoding="utf-8") as f:
            old_data = json.load(f)

    # Pre-parse old questions
    old_questions_clean = [clean_text(item.get("pergunta")) for item in old_data]

    # Valid page limits
    # TACS pages: 1 to 300.
    # Semio pages: 1 to 50, ppt 1 to 20.
    max_page = 300 if subject == "tacs" else 60

    errors = 0
    warnings = 0

    for item in data:
        item_id = item.get("id")
        categoria = item.get("categoria")
        pergunta = item.get("pergunta")
        contexto_esperado = item.get("contexto_esperado")
        must_include = item.get("must_include", [])
        must_not_include = item.get("must_not_include", [])
        expected_sources = item.get("expected_sources", [])
        evaluation_notes = item.get("evaluation_notes", "")

        # 3. Campos obrigatórios
        for field in ["id", "categoria", "pergunta", "contexto_esperado", "must_include", "must_not_include", "expected_sources"]:
            if field not in item:
                print(f"Error [ID {item_id}]: Missing mandatory field '{field}'.")
                errors += 1

        if not pergunta or not pergunta.strip():
            print(f"Error [ID {item_id}]: Empty question.")
            errors += 1

        if not contexto_esperado or not contexto_esperado.strip():
            print(f"Error [ID {item_id}]: Empty expected context.")
            errors += 1

        # 4. Referências existentes e páginas dentro do intervalo
        if categoria != "fora_scope" and not expected_sources:
            print(f"Error [ID {item_id}]: Non-out-of-scope item must have expected_sources.")
            errors += 1

        subject_ranges = KNOWN_RANGES.get(subject, {})
        for src in expected_sources:
            matched_key = "default"
            for key in subject_ranges:
                if key != "default" and key in src.lower():
                    matched_key = key
                    break
            min_p, max_p = subject_ranges[matched_key]
            
            # Extract page numbers using regex
            # e.g., "p. 14", "pp. 76-78", "PowerPoint 11", "página 25"
            page_matches = re.findall(r'\b(?:p\.?|pp\.?|pag\.?|p[aá]gina|powerpoint|slide)\s*[:.]?\s*(\d+)', src, re.IGNORECASE)
            for p_str in page_matches:
                p_num = int(p_str)
                if p_num < min_p or p_num > max_p:
                    print(f"Error [ID {item_id}]: Page number {p_num} in source '{src}' is out of range ({min_p}-{max_p}).")
                    errors += 1

        # 5. Duplicados lexicais / semânticos (jaccard overlap of words)
        p_clean = clean_text(pergunta)
        p_words = set(p_clean.split())
        for old_q in old_questions_clean:
            old_words = set(old_q.split())
            if not p_words or not old_words:
                continue
            intersection = p_words.intersection(old_words)
            union = p_words.union(old_words)
            jaccard = len(intersection) / len(union)
            if jaccard > 0.65:
                print(f"Warning [ID {item_id}]: Potential semantic/lexical duplicate with old dataset (Jaccard={jaccard:.2f}).")
                print(f"  New: {pergunta}")
                print(f"  Old: {old_q}")
                warnings += 1

        # 6. Contradições entre ground truth e must_include
        # Every must_include should be semantically supported in the ground truth
        for must in must_include:
            # Simple check: do we have at least partial match of words?
            must_clean = clean_text(must)
            must_words = [w for w in must_clean.split() if len(w) >= 4]
            # check if at least 50% of the key words of must_include are present in expected context
            matches = sum(1 for w in must_words if w in clean_text(contexto_esperado))
            if must_words and (matches / len(must_words)) < 0.4:
                print(f"Warning [ID {item_id}]: 'must_include' item '{must}' lacks lexical/semantic grounding in 'contexto_esperado'.")
                warnings += 1

        # 7. Referências factualmente suspeitas (e.g. wrong standard JS returns or wrong references)
        if subject == "tacs" and "splice" in p_clean:
            # Ensure return type details of splice are correct (it returns deleted elements, not the original array)
            if "retorna o mesmo array" in clean_text(contexto_esperado) or "retorna o array original" in clean_text(contexto_esperado):
                print(f"Error [ID {item_id}]: Factual error in splice() return description. splice() returns deleted elements.")
                errors += 1

    print(f"Validation finished: {errors} Errors, {warnings} Warnings.")
    return errors == 0

if __name__ == "__main__":
    parser = argparse.ArgumentParser(description="Dataset Validator")
    parser.add_argument("--new", type=Path, required=True, help="Path to new dataset JSON")
    parser.add_argument("--old", type=Path, default=None, help="Path to old dataset JSON (optional)")
    parser.add_argument("--subject", choices=["tacs", "semio"], required=True, help="Subject (tacs/semio)")
    args = parser.parse_args()

    success = validate_dataset(args.new, args.old, args.subject)
    sys.exit(0 if success else 1)