Spaces:
Sleeping
Sleeping
File size: 7,552 Bytes
5bfaada | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 | #!/usr/bin/env python3
import json
import sys
import re
import argparse
from pathlib import Path
from collections import Counter
# Set of expected categories and their exact counts
EXPECTED_COUNTS = {
"definicao": 10,
"conceptual": 10,
"procedimento": 10,
"debug": 8,
"aplicacao": 5,
"fora_scope": 5,
"ambigua": 2
}
KNOWN_RANGES = {
"tacs": {
"default": (4, 133),
"variáveis e scope": (7, 9),
"assincronia: promises": (119, 121),
},
"semio": {
"default": (1, 60),
"resumo sc": (17, 28),
"semiótica resumos": (1, 30),
}
}
def clean_text(text):
return re.sub(r'\s+', ' ', (text or '').lower().strip())
def validate_dataset(new_dataset_path: Path, old_dataset_path: Path | None, subject: str) -> bool:
print(f"--- Validating dataset: {new_dataset_path} ---")
if not new_dataset_path.exists():
print(f"Error: File {new_dataset_path} does not exist.")
return False
try:
with open(new_dataset_path, "r", encoding="utf-8") as f:
data = json.load(f)
except json.JSONDecodeError as e:
print(f"Error parsing JSON: {e}")
return False
if not isinstance(data, list):
print("Error: Dataset must be a list of items.")
return False
# 1. 50 IDs únicos e consecutivos
if len(data) != 50:
print(f"Error: Dataset must have exactly 50 items (found {len(data)}).")
return False
ids = [item.get("id") for item in data]
if any(i is None for i in ids):
print("Error: Some items do not have an ID.")
return False
sorted_ids = sorted(ids)
if sorted_ids != list(range(1, 51)):
print(f"Error: IDs must be unique and consecutive from 1 to 50. Found: {sorted_ids}")
return False
# 2. Distribuição exata
categories = [item.get("categoria") for item in data]
counts = Counter(categories)
for cat, expected in EXPECTED_COUNTS.items():
if counts[cat] != expected:
print(f"Error: Category '{cat}' count is {counts[cat]}, but expected {expected}.")
return False
# Load old dataset for duplicate/semantic check
old_data = []
if old_dataset_path and old_dataset_path.exists():
with open(old_dataset_path, "r", encoding="utf-8") as f:
old_data = json.load(f)
# Pre-parse old questions
old_questions_clean = [clean_text(item.get("pergunta")) for item in old_data]
# Valid page limits
# TACS pages: 1 to 300.
# Semio pages: 1 to 50, ppt 1 to 20.
max_page = 300 if subject == "tacs" else 60
errors = 0
warnings = 0
for item in data:
item_id = item.get("id")
categoria = item.get("categoria")
pergunta = item.get("pergunta")
contexto_esperado = item.get("contexto_esperado")
must_include = item.get("must_include", [])
must_not_include = item.get("must_not_include", [])
expected_sources = item.get("expected_sources", [])
evaluation_notes = item.get("evaluation_notes", "")
# 3. Campos obrigatórios
for field in ["id", "categoria", "pergunta", "contexto_esperado", "must_include", "must_not_include", "expected_sources"]:
if field not in item:
print(f"Error [ID {item_id}]: Missing mandatory field '{field}'.")
errors += 1
if not pergunta or not pergunta.strip():
print(f"Error [ID {item_id}]: Empty question.")
errors += 1
if not contexto_esperado or not contexto_esperado.strip():
print(f"Error [ID {item_id}]: Empty expected context.")
errors += 1
# 4. Referências existentes e páginas dentro do intervalo
if categoria != "fora_scope" and not expected_sources:
print(f"Error [ID {item_id}]: Non-out-of-scope item must have expected_sources.")
errors += 1
subject_ranges = KNOWN_RANGES.get(subject, {})
for src in expected_sources:
matched_key = "default"
for key in subject_ranges:
if key != "default" and key in src.lower():
matched_key = key
break
min_p, max_p = subject_ranges[matched_key]
# Extract page numbers using regex
# e.g., "p. 14", "pp. 76-78", "PowerPoint 11", "página 25"
page_matches = re.findall(r'\b(?:p\.?|pp\.?|pag\.?|p[aá]gina|powerpoint|slide)\s*[:.]?\s*(\d+)', src, re.IGNORECASE)
for p_str in page_matches:
p_num = int(p_str)
if p_num < min_p or p_num > max_p:
print(f"Error [ID {item_id}]: Page number {p_num} in source '{src}' is out of range ({min_p}-{max_p}).")
errors += 1
# 5. Duplicados lexicais / semânticos (jaccard overlap of words)
p_clean = clean_text(pergunta)
p_words = set(p_clean.split())
for old_q in old_questions_clean:
old_words = set(old_q.split())
if not p_words or not old_words:
continue
intersection = p_words.intersection(old_words)
union = p_words.union(old_words)
jaccard = len(intersection) / len(union)
if jaccard > 0.65:
print(f"Warning [ID {item_id}]: Potential semantic/lexical duplicate with old dataset (Jaccard={jaccard:.2f}).")
print(f" New: {pergunta}")
print(f" Old: {old_q}")
warnings += 1
# 6. Contradições entre ground truth e must_include
# Every must_include should be semantically supported in the ground truth
for must in must_include:
# Simple check: do we have at least partial match of words?
must_clean = clean_text(must)
must_words = [w for w in must_clean.split() if len(w) >= 4]
# check if at least 50% of the key words of must_include are present in expected context
matches = sum(1 for w in must_words if w in clean_text(contexto_esperado))
if must_words and (matches / len(must_words)) < 0.4:
print(f"Warning [ID {item_id}]: 'must_include' item '{must}' lacks lexical/semantic grounding in 'contexto_esperado'.")
warnings += 1
# 7. Referências factualmente suspeitas (e.g. wrong standard JS returns or wrong references)
if subject == "tacs" and "splice" in p_clean:
# Ensure return type details of splice are correct (it returns deleted elements, not the original array)
if "retorna o mesmo array" in clean_text(contexto_esperado) or "retorna o array original" in clean_text(contexto_esperado):
print(f"Error [ID {item_id}]: Factual error in splice() return description. splice() returns deleted elements.")
errors += 1
print(f"Validation finished: {errors} Errors, {warnings} Warnings.")
return errors == 0
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Dataset Validator")
parser.add_argument("--new", type=Path, required=True, help="Path to new dataset JSON")
parser.add_argument("--old", type=Path, default=None, help="Path to old dataset JSON (optional)")
parser.add_argument("--subject", choices=["tacs", "semio"], required=True, help="Subject (tacs/semio)")
args = parser.parse_args()
success = validate_dataset(args.new, args.old, args.subject)
sys.exit(0 if success else 1)
|