ThreatIntel / scripts /normalize_sources.py
Malek98's picture
Deploy FastAPI backend via automated upload
bca5172 verified
Raw
History Blame Contribute Delete
6.74 kB
#!/usr/bin/env python3
"""
scripts/normalize_sources.py
Normalize or remove local filesystem paths in the cleaned dataset artifacts.
- Reads files from a `final_cleaned_data` directory (or other path you pass).
- Produces normalized outputs with source paths replaced by a compact identifier.
- Writes a `source_map.json` mapping original -> normalized for traceability.
Usage examples (from project root):
# Dry-run: show 10 mappings
python scripts/normalize_sources.py --input-dir final_cleaned_data --dry-run --show 10
# Apply changes and write normalized files
python scripts/normalize_sources.py --input-dir final_cleaned_data --apply
# Use a custom output dir
python scripts/normalize_sources.py --input-dir final_cleaned_data --apply --out-dir final_cleaned_data/normalized
Behavior / heuristics
- If a source contains an .onion hostname, normalized id will be the onion hostname (e.g. 222222222hsoeiok.onion).
- Else, we use the basename of `source_path` (filename).
- For concatenated records with `::original_filename`, the part after `::` is used when present.
- The script preserves the original files by writing new outputs (normalized_*).
"""
from pathlib import Path
import argparse
import json
import csv
import re
from urllib.parse import urlparse
ONION_RE = re.compile(r"([a-z2-7]{16,}\.onion)", flags=re.I)
def extract_identifier(src: str) -> str:
if not src:
return "unknown"
# handle concatenated style: path::inner
if "::" in src:
src = src.split("::", 1)[1]
# try to find onion host
m = ONION_RE.search(src)
if m:
return m.group(1).lower()
# try to parse as URL
if src.startswith("http://") or src.startswith("https://"):
try:
p = urlparse(src)
host = p.netloc
if host:
return host
except Exception:
pass
# fallback: use basename
try:
return Path(src).name
except Exception:
return src.replace('\\', '/').split('/')[-1]
def normalize_jsonl(in_path: Path, out_path: Path, source_map: dict):
out_path.parent.mkdir(parents=True, exist_ok=True)
written = 0
with in_path.open('r', encoding='utf-8') as inf, out_path.open('w', encoding='utf-8') as outf:
for line in inf:
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except Exception:
continue
orig = obj.get('source_path') or obj.get('source') or ''
nid = extract_identifier(orig)
source_map.setdefault(orig, nid)
# update fields
if 'source_path' in obj:
obj['source_path'] = nid
if 'source' in obj:
obj['source'] = nid
# also update site_folder if it looks like a path
if 'site_folder' in obj:
sf = obj.get('site_folder')
if sf and ('\\' in sf or '/' in sf or ':' in sf):
obj['site_folder'] = nid
outf.write(json.dumps(obj, ensure_ascii=False) + '\n')
written += 1
return written
def normalize_csv(in_path: Path, out_path: Path, source_map: dict):
out_path.parent.mkdir(parents=True, exist_ok=True)
written = 0
with in_path.open('r', encoding='utf-8', newline='') as inf:
reader = csv.DictReader(inf)
fieldnames = reader.fieldnames
rows = list(reader)
# update rows
for r in rows:
orig = r.get('source_path') or ''
nid = extract_identifier(orig)
source_map.setdefault(orig, nid)
r['source_path'] = nid
# if there is site_folder field, normalize it too
if 'site_folder' in r and (not r['site_folder'] or ('\\' in r['site_folder'] or '/' in r['site_folder'] or ':' in r['site_folder'])):
r['site_folder'] = nid
# write out
with out_path.open('w', encoding='utf-8', newline='') as outf:
writer = csv.DictWriter(outf, fieldnames=fieldnames)
writer.writeheader()
for r in rows:
writer.writerow(r)
written += 1
return written
def main():
p = argparse.ArgumentParser()
p.add_argument('--input-dir', type=Path, default=Path('final_cleaned_data'))
p.add_argument('--out-dir', type=Path, default=None)
p.add_argument('--apply', action='store_true')
p.add_argument('--show', type=int, default=0, help='Number of mappings to print in dry-run')
p.add_argument('--dry-run', action='store_true')
args = p.parse_args()
inp = args.input_dir
if not inp.exists():
print('Input dir not found:', inp)
return
out_root = Path(args.out_dir) if args.out_dir else inp / 'normalized'
out_root.mkdir(parents=True, exist_ok=True)
source_map = {}
# files to normalize (if present)
jsonl_in = inp / 'cleaned_extracted_text.jsonl'
csv_in = inp / 'cleaned_labeled_dataset.csv'
instr_in = inp / 'cleaned_instruction_tuning.jsonl'
if jsonl_in.exists():
written = normalize_jsonl(jsonl_in, out_root / 'normalized_extracted_text.jsonl', source_map)
print('Processed JSONL:', written)
else:
print('No cleaned_extracted_text.jsonl at', jsonl_in)
if csv_in.exists():
written = normalize_csv(csv_in, out_root / 'normalized_labeled_dataset.csv', source_map)
print('Processed CSV:', written)
else:
print('No cleaned_labeled_dataset.csv at', csv_in)
if instr_in.exists():
written = normalize_jsonl(instr_in, out_root / 'normalized_instruction_tuning.jsonl', source_map)
print('Processed instruction JSONL:', written)
else:
print('No cleaned_instruction_tuning.jsonl at', instr_in)
# write mapping
map_path = out_root / 'source_map.json'
with map_path.open('w', encoding='utf-8') as f:
json.dump(source_map, f, indent=2, ensure_ascii=False)
if args.dry_run or not args.apply:
print('\nDRY RUN - no files overwritten in originals. To write outputs add --apply')
# print sample mappings
cnt = 0
for orig, nid in list(source_map.items())[:args.show or 20]:
print(f'"{orig}" -> "{nid}"')
cnt += 1
print(f'Printed {cnt} mappings. Mapping written to', map_path)
return
# If apply, we've already written normalized files to out_root
print('Wrote normalized outputs to', out_root)
print('Source map written to', map_path)
if __name__ == '__main__':
main()