File size: 6,743 Bytes
bca5172
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
#!/usr/bin/env python3
"""

scripts/normalize_sources.py



Normalize or remove local filesystem paths in the cleaned dataset artifacts.

- Reads files from a `final_cleaned_data` directory (or other path you pass).

- Produces normalized outputs with source paths replaced by a compact identifier.

- Writes a `source_map.json` mapping original -> normalized for traceability.



Usage examples (from project root):



# Dry-run: show 10 mappings

python scripts/normalize_sources.py --input-dir final_cleaned_data --dry-run --show 10



# Apply changes and write normalized files

python scripts/normalize_sources.py --input-dir final_cleaned_data --apply



# Use a custom output dir

python scripts/normalize_sources.py --input-dir final_cleaned_data --apply --out-dir final_cleaned_data/normalized



Behavior / heuristics

- If a source contains an .onion hostname, normalized id will be the onion hostname (e.g. 222222222hsoeiok.onion).

- Else, we use the basename of `source_path` (filename).

- For concatenated records with `::original_filename`, the part after `::` is used when present.

- The script preserves the original files by writing new outputs (normalized_*).

"""

from pathlib import Path
import argparse
import json
import csv
import re
from urllib.parse import urlparse

ONION_RE = re.compile(r"([a-z2-7]{16,}\.onion)", flags=re.I)


def extract_identifier(src: str) -> str:
    if not src:
        return "unknown"
    # handle concatenated style: path::inner
    if "::" in src:
        src = src.split("::", 1)[1]
    # try to find onion host
    m = ONION_RE.search(src)
    if m:
        return m.group(1).lower()
    # try to parse as URL
    if src.startswith("http://") or src.startswith("https://"):
        try:
            p = urlparse(src)
            host = p.netloc
            if host:
                return host
        except Exception:
            pass
    # fallback: use basename
    try:
        return Path(src).name
    except Exception:
        return src.replace('\\', '/').split('/')[-1]


def normalize_jsonl(in_path: Path, out_path: Path, source_map: dict):
    out_path.parent.mkdir(parents=True, exist_ok=True)
    written = 0
    with in_path.open('r', encoding='utf-8') as inf, out_path.open('w', encoding='utf-8') as outf:
        for line in inf:
            line = line.strip()
            if not line:
                continue
            try:
                obj = json.loads(line)
            except Exception:
                continue
            orig = obj.get('source_path') or obj.get('source') or ''
            nid = extract_identifier(orig)
            source_map.setdefault(orig, nid)
            # update fields
            if 'source_path' in obj:
                obj['source_path'] = nid
            if 'source' in obj:
                obj['source'] = nid
            # also update site_folder if it looks like a path
            if 'site_folder' in obj:
                sf = obj.get('site_folder')
                if sf and ('\\' in sf or '/' in sf or ':' in sf):
                    obj['site_folder'] = nid
            outf.write(json.dumps(obj, ensure_ascii=False) + '\n')
            written += 1
    return written


def normalize_csv(in_path: Path, out_path: Path, source_map: dict):
    out_path.parent.mkdir(parents=True, exist_ok=True)
    written = 0
    with in_path.open('r', encoding='utf-8', newline='') as inf:
        reader = csv.DictReader(inf)
        fieldnames = reader.fieldnames
        rows = list(reader)
    # update rows
    for r in rows:
        orig = r.get('source_path') or ''
        nid = extract_identifier(orig)
        source_map.setdefault(orig, nid)
        r['source_path'] = nid
        # if there is site_folder field, normalize it too
        if 'site_folder' in r and (not r['site_folder'] or ('\\' in r['site_folder'] or '/' in r['site_folder'] or ':' in r['site_folder'])):
            r['site_folder'] = nid
    # write out
    with out_path.open('w', encoding='utf-8', newline='') as outf:
        writer = csv.DictWriter(outf, fieldnames=fieldnames)
        writer.writeheader()
        for r in rows:
            writer.writerow(r)
            written += 1
    return written


def main():
    p = argparse.ArgumentParser()
    p.add_argument('--input-dir', type=Path, default=Path('final_cleaned_data'))
    p.add_argument('--out-dir', type=Path, default=None)
    p.add_argument('--apply', action='store_true')
    p.add_argument('--show', type=int, default=0, help='Number of mappings to print in dry-run')
    p.add_argument('--dry-run', action='store_true')
    args = p.parse_args()

    inp = args.input_dir
    if not inp.exists():
        print('Input dir not found:', inp)
        return
    out_root = Path(args.out_dir) if args.out_dir else inp / 'normalized'
    out_root.mkdir(parents=True, exist_ok=True)

    source_map = {}

    # files to normalize (if present)
    jsonl_in = inp / 'cleaned_extracted_text.jsonl'
    csv_in = inp / 'cleaned_labeled_dataset.csv'
    instr_in = inp / 'cleaned_instruction_tuning.jsonl'

    if jsonl_in.exists():
        written = normalize_jsonl(jsonl_in, out_root / 'normalized_extracted_text.jsonl', source_map)
        print('Processed JSONL:', written)
    else:
        print('No cleaned_extracted_text.jsonl at', jsonl_in)

    if csv_in.exists():
        written = normalize_csv(csv_in, out_root / 'normalized_labeled_dataset.csv', source_map)
        print('Processed CSV:', written)
    else:
        print('No cleaned_labeled_dataset.csv at', csv_in)

    if instr_in.exists():
        written = normalize_jsonl(instr_in, out_root / 'normalized_instruction_tuning.jsonl', source_map)
        print('Processed instruction JSONL:', written)
    else:
        print('No cleaned_instruction_tuning.jsonl at', instr_in)

    # write mapping
    map_path = out_root / 'source_map.json'
    with map_path.open('w', encoding='utf-8') as f:
        json.dump(source_map, f, indent=2, ensure_ascii=False)

    if args.dry_run or not args.apply:
        print('\nDRY RUN - no files overwritten in originals. To write outputs add --apply')
        # print sample mappings
        cnt = 0
        for orig, nid in list(source_map.items())[:args.show or 20]:
            print(f'"{orig}" -> "{nid}"')
            cnt += 1
        print(f'Printed {cnt} mappings. Mapping written to', map_path)
        return

    # If apply, we've already written normalized files to out_root
    print('Wrote normalized outputs to', out_root)
    print('Source map written to', map_path)

if __name__ == '__main__':
    main()