Spaces:
Sleeping
Sleeping
File size: 6,743 Bytes
bca5172 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 | #!/usr/bin/env python3
"""
scripts/normalize_sources.py
Normalize or remove local filesystem paths in the cleaned dataset artifacts.
- Reads files from a `final_cleaned_data` directory (or other path you pass).
- Produces normalized outputs with source paths replaced by a compact identifier.
- Writes a `source_map.json` mapping original -> normalized for traceability.
Usage examples (from project root):
# Dry-run: show 10 mappings
python scripts/normalize_sources.py --input-dir final_cleaned_data --dry-run --show 10
# Apply changes and write normalized files
python scripts/normalize_sources.py --input-dir final_cleaned_data --apply
# Use a custom output dir
python scripts/normalize_sources.py --input-dir final_cleaned_data --apply --out-dir final_cleaned_data/normalized
Behavior / heuristics
- If a source contains an .onion hostname, normalized id will be the onion hostname (e.g. 222222222hsoeiok.onion).
- Else, we use the basename of `source_path` (filename).
- For concatenated records with `::original_filename`, the part after `::` is used when present.
- The script preserves the original files by writing new outputs (normalized_*).
"""
from pathlib import Path
import argparse
import json
import csv
import re
from urllib.parse import urlparse
ONION_RE = re.compile(r"([a-z2-7]{16,}\.onion)", flags=re.I)
def extract_identifier(src: str) -> str:
if not src:
return "unknown"
# handle concatenated style: path::inner
if "::" in src:
src = src.split("::", 1)[1]
# try to find onion host
m = ONION_RE.search(src)
if m:
return m.group(1).lower()
# try to parse as URL
if src.startswith("http://") or src.startswith("https://"):
try:
p = urlparse(src)
host = p.netloc
if host:
return host
except Exception:
pass
# fallback: use basename
try:
return Path(src).name
except Exception:
return src.replace('\\', '/').split('/')[-1]
def normalize_jsonl(in_path: Path, out_path: Path, source_map: dict):
out_path.parent.mkdir(parents=True, exist_ok=True)
written = 0
with in_path.open('r', encoding='utf-8') as inf, out_path.open('w', encoding='utf-8') as outf:
for line in inf:
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except Exception:
continue
orig = obj.get('source_path') or obj.get('source') or ''
nid = extract_identifier(orig)
source_map.setdefault(orig, nid)
# update fields
if 'source_path' in obj:
obj['source_path'] = nid
if 'source' in obj:
obj['source'] = nid
# also update site_folder if it looks like a path
if 'site_folder' in obj:
sf = obj.get('site_folder')
if sf and ('\\' in sf or '/' in sf or ':' in sf):
obj['site_folder'] = nid
outf.write(json.dumps(obj, ensure_ascii=False) + '\n')
written += 1
return written
def normalize_csv(in_path: Path, out_path: Path, source_map: dict):
out_path.parent.mkdir(parents=True, exist_ok=True)
written = 0
with in_path.open('r', encoding='utf-8', newline='') as inf:
reader = csv.DictReader(inf)
fieldnames = reader.fieldnames
rows = list(reader)
# update rows
for r in rows:
orig = r.get('source_path') or ''
nid = extract_identifier(orig)
source_map.setdefault(orig, nid)
r['source_path'] = nid
# if there is site_folder field, normalize it too
if 'site_folder' in r and (not r['site_folder'] or ('\\' in r['site_folder'] or '/' in r['site_folder'] or ':' in r['site_folder'])):
r['site_folder'] = nid
# write out
with out_path.open('w', encoding='utf-8', newline='') as outf:
writer = csv.DictWriter(outf, fieldnames=fieldnames)
writer.writeheader()
for r in rows:
writer.writerow(r)
written += 1
return written
def main():
p = argparse.ArgumentParser()
p.add_argument('--input-dir', type=Path, default=Path('final_cleaned_data'))
p.add_argument('--out-dir', type=Path, default=None)
p.add_argument('--apply', action='store_true')
p.add_argument('--show', type=int, default=0, help='Number of mappings to print in dry-run')
p.add_argument('--dry-run', action='store_true')
args = p.parse_args()
inp = args.input_dir
if not inp.exists():
print('Input dir not found:', inp)
return
out_root = Path(args.out_dir) if args.out_dir else inp / 'normalized'
out_root.mkdir(parents=True, exist_ok=True)
source_map = {}
# files to normalize (if present)
jsonl_in = inp / 'cleaned_extracted_text.jsonl'
csv_in = inp / 'cleaned_labeled_dataset.csv'
instr_in = inp / 'cleaned_instruction_tuning.jsonl'
if jsonl_in.exists():
written = normalize_jsonl(jsonl_in, out_root / 'normalized_extracted_text.jsonl', source_map)
print('Processed JSONL:', written)
else:
print('No cleaned_extracted_text.jsonl at', jsonl_in)
if csv_in.exists():
written = normalize_csv(csv_in, out_root / 'normalized_labeled_dataset.csv', source_map)
print('Processed CSV:', written)
else:
print('No cleaned_labeled_dataset.csv at', csv_in)
if instr_in.exists():
written = normalize_jsonl(instr_in, out_root / 'normalized_instruction_tuning.jsonl', source_map)
print('Processed instruction JSONL:', written)
else:
print('No cleaned_instruction_tuning.jsonl at', instr_in)
# write mapping
map_path = out_root / 'source_map.json'
with map_path.open('w', encoding='utf-8') as f:
json.dump(source_map, f, indent=2, ensure_ascii=False)
if args.dry_run or not args.apply:
print('\nDRY RUN - no files overwritten in originals. To write outputs add --apply')
# print sample mappings
cnt = 0
for orig, nid in list(source_map.items())[:args.show or 20]:
print(f'"{orig}" -> "{nid}"')
cnt += 1
print(f'Printed {cnt} mappings. Mapping written to', map_path)
return
# If apply, we've already written normalized files to out_root
print('Wrote normalized outputs to', out_root)
print('Source map written to', map_path)
if __name__ == '__main__':
main()
|