Download audit_data.py from Future-Labs/minilm-example-router: direct link, hf CLI and curl.
- Browser
- Download file 3.44 kB
-
https://huggingface.co/Future-Labs/minilm-example-router/resolve/main/audit_data.py
- Command line
-
hf download hf://Future-Labs/minilm-example-router/audit_data.py
-
curl -L -o audit_data.py https://huggingface.co/Future-Labs/minilm-example-router/resolve/main/audit_data.py
3.44 kB
| """Audit normalized-text overlap locally before routing evaluation; no model loading.""" | |
| import argparse | |
| import hashlib | |
| import json | |
| import unicodedata | |
| from pathlib import Path | |
| def fingerprint(text): | |
| return hashlib.sha256(' '.join(unicodedata.normalize('NFKC', text).casefold().split()).encode()).hexdigest() | |
| def audit(evaluation, examples): | |
| source = json.loads(Path(examples).read_text(encoding='utf-8')) | |
| if isinstance(source, dict): | |
| if source.get('format') != 'example-router-project-v1': | |
| raise ValueError('Expected an editable project or route-example list') | |
| source = source.get('groups') | |
| if not isinstance(source, list) or not source: | |
| raise ValueError('Expected route groups') | |
| training = {}; labels = set(); training_rows = 0 | |
| for group in source: | |
| if not isinstance(group, dict) or not isinstance(group.get('label'), str) or not group['label'].strip(): | |
| raise ValueError('Each route group needs a nonempty label') | |
| label = group['label'].strip() | |
| if label in labels: | |
| raise ValueError('Route labels must be unique') | |
| labels.add(label) | |
| if not isinstance(group.get('examples'), list) or not group['examples']: | |
| raise ValueError('Each route group needs examples') | |
| for text in group['examples']: | |
| if not isinstance(text, str) or not text.strip() or len(text) > 4000: | |
| raise ValueError('Examples must be nonempty strings of at most 4000 characters') | |
| training.setdefault(fingerprint(text), set()).add(label); training_rows += 1 | |
| # Import is standard-library only and does not load the encoder. | |
| from evaluate import iter_rows | |
| seen = {}; overlap_rows = overlap_label_conflicts = total = 0 | |
| for row in iter_rows(evaluation, labels): | |
| key = fingerprint(row['text']); label = row['label']; total += 1 | |
| seen.setdefault(key, set()).add(label) | |
| if key in training: | |
| overlap_rows += 1 | |
| overlap_label_conflicts += int(any(label != other for other in training[key])) | |
| return {'training_rows': training_rows, 'training_unique_normalized_texts': len(training), | |
| 'training_repeated_rows': training_rows-len(training), | |
| 'training_conflicting_texts': sum(len(v)>1 for v in training.values()), | |
| 'evaluation_rows': total, 'evaluation_unique_normalized_texts': len(seen), | |
| 'evaluation_repeated_rows': total-len(seen), | |
| 'evaluation_conflicting_texts': sum(len(v)>1 for v in seen.values()), | |
| 'overlap_rows': overlap_rows, 'overlap_unique_texts': len(set(seen)&set(training)), | |
| 'overlap_rows_with_label_conflict': overlap_label_conflicts, | |
| 'note': 'Counts only; NFKC, casefold and whitespace normalization. Does not detect paraphrases or guarantee split independence. Review conflicts manually; no rows removed or files changed.'} | |
| def main(): | |
| parser=argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument('examples_json'); parser.add_argument('evaluation_jsonl') | |
| parser.add_argument('--output',required=True); args=parser.parse_args() | |
| result=audit(args.evaluation_jsonl,args.examples_json) | |
| Path(args.output).write_text(json.dumps(result,indent=2)+'\n') | |
| print(f"Audited {result['evaluation_rows']} messages; {result['overlap_rows']} overlap route-building examples.") | |
| if __name__=='__main__':main() | |