File size: 16,162 Bytes
83892b0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
"""

pipeline.py

───────────

Orchestrates the full scraping pipeline end to end.



Flow:

  Step 1 — [optional] Use the SOAP API to discover law IDs by keyword

  Step 2 — For each ID, fetch and parse the HTML page (html_scraper.py)

  Step 3 — Clean the text: fix encoding, diacritics, boilerplate (cleaner.py)

  Step 4 — Save results to data/laws.jsonl and data/articles.jsonl



The script is designed to be safe and resumable:

  - It checks which IDs are already in laws.jsonl and skips them

  - If interrupted, just re-run — it picks up where it left off

  - Failed IDs are saved to data/failed_ids.txt for later retry

  - Delays between requests are randomized to avoid rate limiting



Usage examples:

  # Scrape specific laws by their known IDs (recommended — no API needed)

  python pipeline.py --ids 109567 175630 175538



  # Scrape with a longer delay (safer for large batches)

  python pipeline.py --ids 109567 175630 --delay 4.0



  # Discover IDs via the SOAP API using a keyword, then scrape

  python pipeline.py --keyword "muncă" --max-pages 3



  # Load IDs from a text file (one ID per line)

  python pipeline.py --ids-file my_ids.txt



Output files:

  data/laws.jsonl      — one complete law object per line

  data/articles.jsonl  — one article per line (used by indexer.py)

  data/failed_ids.txt  — IDs that failed, for retry



Install: pip install requests beautifulsoup4 suds-community ftfy tqdm

"""

import json
import time
import random
import argparse
import sys
import io
from pathlib import Path
from tqdm import tqdm

# Force UTF-8 output on Windows (otherwise Romanian characters crash the terminal)
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8", errors="replace")

from html_scraper import scrape_law
from cleaner import clean_law

# ── Configuration ─────────────────────────────────────────────────────────────

DATA_DIR      = Path("data")
LAWS_FILE     = DATA_DIR / "laws.jsonl"
ARTICLES_FILE = DATA_DIR / "articles.jsonl"
FAILED_FILE   = DATA_DIR / "failed_ids.txt"

# The most important Romanian laws for a legal RAG system.
#
# IMPORTANT — how the portal versions laws:
#   Each law has MULTIPLE IDs on legislatie.just.ro. For example, Codul Muncii has:
#     - 41625  → original 2003 form (DetaliiDocumentAfis — old URL pattern)
#     - 75203  → (A) actualizat, updated version
#     - 128646 → (R) republicat 2011, re-numbered articles
#   We want the most recent ACTUALIZAT (A) form — that's the currently-in-force text.
#   These IDs below are the verified, most up-to-date consolidated forms.
#   They were verified by checking the actual URLs in search results.
#
# URL pattern for verification:
#   https://legislatie.just.ro/Public/DetaliiDocument/{ID}
IMPORTANT_LAW_IDS = {
    # Codul Muncii — ID 41627 confirmed from search as current actualizat (A) form
    # The scraper will automatically try both URL patterns
     41627: "Codul Muncii - Legea 53/2003",

    # Codul Civil — confirmed working, scraped 2930 articles successfully
    175630: "Codul Civil - Legea 287/2009",

    # Codul Penal — ID 109855 confirmed from search
    109855: "Codul Penal - Legea 286/2009",

    # Codul de Procedura Civila — ID 140271 confirmed from search (republicat 2015)
    140271: "Codul de Procedura Civila - Legea 134/2010",

    # Codul de Procedura Penala — ID 120609 confirmed from search
    120609: "Codul de Procedura Penala - Legea 135/2010",

    # Legea Societatilor Comerciale
     38533: "Legea Societatilor Comerciale - Legea 31/1990",
}


# ── Helper functions ──────────────────────────────────────────────────────────

def ensure_data_dir():
    """Create the data/ directory if it doesn't already exist."""
    DATA_DIR.mkdir(exist_ok=True)


def load_already_scraped() -> set[int]:
    """

    Read laws.jsonl and return the set of IDs we've already scraped.



    This is how we support resuming an interrupted run:

    if laws.jsonl has 5 laws and we're asked to scrape 10,

    we skip the first 5 and only do the remaining 5.

    """
    if not LAWS_FILE.exists():
        return set()

    scraped = set()
    with open(LAWS_FILE, encoding="utf-8") as f:
        for line in f:
            line = line.strip()
            if not line:
                continue
            try:
                obj = json.loads(line)
                scraped.add(int(obj["id"]))
            except (json.JSONDecodeError, KeyError, ValueError):
                pass  # malformed line — ignore it

    return scraped


def append_jsonl(filepath: Path, obj: dict):
    """

    Write one Python dict as a JSON line at the end of a file.



    JSONL (JSON Lines) format = one JSON object per line.

    It's great for large datasets because you can read one line at a time

    without loading the entire file into memory.

    """
    with open(filepath, "a", encoding="utf-8") as f:
        f.write(json.dumps(obj, ensure_ascii=False) + "\n")


def polite_sleep(base_delay: float):
    """

    Wait for a randomized amount of time between requests.



    Why randomize? Servers detect bots by the regularity of requests.

    A human reading a page waits 2s here, 5s there, 1.5s somewhere else.

    A bot waits exactly 2.0s every single time — that's a dead giveaway.



    We add ±50% random jitter to the base delay:

      base_delay=3.0 → actual wait is between 1.5s and 4.5s

    """
    jitter = random.uniform(-base_delay * 0.5, base_delay * 0.5)
    actual = max(1.0, base_delay + jitter)  # never go below 1 second
    time.sleep(actual)


def get_ids_from_api(keyword: str, max_pages: int) -> list[int]:
    """

    Use the SOAP API to search for law IDs by keyword.



    This requires the suds-community package.

    If it's not installed, we print a helpful message and return empty.

    """
    try:
        from api_client import LegislatieAPIClient
    except ImportError:
        print("ERROR: suds-community is not installed.")
        print("  Run: pip install suds-community")
        print("  Or use --ids to provide IDs directly.")
        return []

    client = LegislatieAPIClient()
    laws   = client.get_all_ids(keyword=keyword, max_pages=max_pages)
    ids    = [int(law.Id) for law in laws if hasattr(law, "Id")]
    print(f"  API found {len(ids)} IDs for keyword '{keyword}'.")
    return ids


# ── Core pipeline ─────────────────────────────────────────────────────────────

def run_pipeline(law_ids: list[int], delay: float = 3.0):
    """

    Main scraping loop.



    For each law ID:

      1. Check if we already scraped it — if yes, skip

      2. Download and parse the HTML page

      3. Clean the text

      4. Append to laws.jsonl (full law) and articles.jsonl (one row per article)

      5. Wait politely before the next request



    Parameters:

      law_ids  — list of integer law IDs to scrape

      delay    — base seconds to wait between requests (default 3.0)

                 Actual wait = delay ± 50% random jitter

                 Recommended: 3.0 for batches up to 20 laws

                              4.0 for larger batches

    """
    ensure_data_dir()
    already_done = load_already_scraped()
    pending      = [id_ for id_ in law_ids if id_ not in already_done]

    print(f"\n{'='*60}")
    print(f"  Laws requested:      {len(law_ids)}")
    print(f"  Already in dataset:  {len(already_done)}")
    print(f"  To scrape now:       {len(pending)}")
    print(f"  Base delay:          {delay}s (±50% jitter)")
    print(f"{'='*60}\n")

    if not pending:
        print("Nothing to do — all requested IDs are already scraped.")
        print(f"Delete {LAWS_FILE} to force a re-scrape.")
        return

    success_count = 0
    failed_ids    = []

    for law_id in tqdm(pending, desc="Scraping", unit="law"):

        # ── Fetch and parse ───────────────────────────────────────────────────
        law = scrape_law(law_id)

        if law is None:
            # scrape_law already printed the reason
            failed_ids.append(law_id)
            polite_sleep(delay)
            continue

        # ── Clean ─────────────────────────────────────────────────────────────
        law = clean_law(law)

        # If cleaning removed ALL articles, something went wrong
        if law["article_count"] == 0:
            print(f"  ⚠ WARNING: ID {law_id} has 0 articles after cleaning. "
                  f"Adding to failed list.")
            failed_ids.append(law_id)
            polite_sleep(delay)
            continue

        # ── Save the full law object ──────────────────────────────────────────
        append_jsonl(LAWS_FILE, {
            "id":            law["id"],
            "title":         law["title"],
            "url":           law["url"],
            "article_count": law["article_count"],
            "articles":      law["articles"],
        })

        # ── Save each article as a separate row for embedding ─────────────────
        # The "chunk" field is what your embedder (indexer.py) will use.
        # It includes the law title + article number so the embedding captures
        # source context, not just the article text in isolation.
        # This improves retrieval accuracy significantly.
        for article in law["articles"]:
            append_jsonl(ARTICLES_FILE, {
                "law_id":         law["id"],
                "law_title":      law["title"],
                "article_number": article["number"],
                "text":           article["text"],
                "chunk": (
                    f"{law['title']}\n"
                    f"{article['number']}\n\n"
                    f"{article['text']}"
                ),
            })

        success_count += 1
        polite_sleep(delay)  # ← be polite to the server

    # ── Summary ───────────────────────────────────────────────────────────────
    print(f"\n{'='*60}")
    print(f"  ✓ Successfully scraped: {success_count} laws")
    print(f"  ✗ Failed:               {len(failed_ids)} laws")

    if failed_ids:
        print(f"\n  Failed IDs: {failed_ids}")
        with open(FAILED_FILE, "w", encoding="utf-8") as f:
            f.write("\n".join(str(i) for i in failed_ids))
        print(f"  Saved to {FAILED_FILE} — retry later with:")
        print(f"    python pipeline.py --ids-file {FAILED_FILE}")

    print(f"\n  Output files:")
    if LAWS_FILE.exists():
        print(f"    {LAWS_FILE}      ({LAWS_FILE.stat().st_size // 1024:,} KB)")
    if ARTICLES_FILE.exists():
        print(f"    {ARTICLES_FILE}  ({ARTICLES_FILE.stat().st_size // 1024:,} KB)")

    print(f"\n  Next step: python indexer.py")
    print(f"{'='*60}\n")


# ── Command-line interface ────────────────────────────────────────────────────

def parse_args():
    parser = argparse.ArgumentParser(
        description=(
            "Scrape Romanian laws from legislatie.just.ro.\n\n"
            "Examples:\n"
            "  python pipeline.py --ids 109567 175630      # Codul Muncii + Codul Civil\n"
            "  python pipeline.py --important              # All 9 important law codes\n"
            "  python pipeline.py --keyword muncă          # Search API by keyword\n"
        ),
        formatter_class=argparse.RawDescriptionHelpFormatter,
    )

    parser.add_argument(
        "--ids",
        nargs="+",
        type=int,
        metavar="ID",
        help="Scrape specific law IDs (skip the API step entirely).",
    )
    parser.add_argument(
        "--important",
        action="store_true",
        help=(
            "Scrape the 9 most important Romanian law codes "
            "(Codul Muncii, Codul Civil, Codul Penal, etc.)."
        ),
    )
    parser.add_argument(
        "--ids-file",
        type=str,
        metavar="FILE",
        help="Path to a text file with one law ID per line.",
    )
    parser.add_argument(
        "--keyword",
        type=str,
        default=None,
        help="Search keyword for the SOAP API (e.g. 'muncă', 'concediu').",
    )
    parser.add_argument(
        "--max-pages",
        type=int,
        default=5,
        help="Max API result pages to fetch (default: 5, ~250 laws per page).",
    )
    parser.add_argument(
        "--delay",
        type=float,
        default=3.0,
        help=(
            "Base seconds to wait between requests (default: 3.0). "
            "Actual wait = delay ± 50%% random jitter. "
            "Use 4.0+ for large batches."
        ),
    )

    return parser.parse_args()


def main():
    args = parse_args()

    # ── Determine which IDs to scrape ─────────────────────────────────────────

    if args.important:
        law_ids = list(IMPORTANT_LAW_IDS.keys())
        print("Scraping the 9 most important Romanian law codes:")
        for law_id, name in IMPORTANT_LAW_IDS.items():
            print(f"  {law_id}{name}")

    elif args.ids:
        law_ids = args.ids
        print(f"Scraping {len(law_ids)} specified IDs: {law_ids}")

    elif args.ids_file:
        path = Path(args.ids_file)
        if not path.exists():
            print(f"ERROR: File not found: {path}")
            sys.exit(1)
        with open(path, encoding="utf-8") as f:
            law_ids = [
                int(line.strip())
                for line in f
                if line.strip().isdigit()
            ]
        print(f"Loaded {len(law_ids)} IDs from {path}.")

    elif args.keyword:
        print(f"Searching SOAP API for keyword: '{args.keyword}'...")
        law_ids = get_ids_from_api(args.keyword, args.max_pages)
        if not law_ids:
            print("No IDs found. Try a different keyword or use --ids directly.")
            sys.exit(1)

    else:
        # No input given — show help and suggest the most useful option
        print("No input specified. Here are your options:\n")
        print("  1. Scrape all important law codes (recommended to start):")
        print("       python pipeline.py --important\n")
        print("  2. Scrape specific IDs:")
        print("       python pipeline.py --ids 109567 175630\n")
        print("  3. Search by keyword:")
        print("       python pipeline.py --keyword muncă\n")
        print("  4. Load IDs from a file:")
        print("       python pipeline.py --ids-file ids.txt\n")
        print("Available law IDs:")
        for law_id, name in IMPORTANT_LAW_IDS.items():
            print(f"  {law_id:>8}  {name}")
        sys.exit(0)

    # ── Run ───────────────────────────────────────────────────────────────────
    run_pipeline(law_ids, delay=args.delay)


if __name__ == "__main__":
    main()