cc-repackage / tests /make_fixtures.py
malteos
Add content_languages filter; vendor index SQL to drop cdx_toolkit dependency
8eaca44 unverified
Raw
History Blame Contribute Delete
3.56 kB
"""Build a toy CC bucket layout for local verification.
Creates, under <root> (default tests/fixtures/cc), the same two subtrees the real
commoncrawl bucket has:
crawl-data/CC-MAIN-2026-25/segments/0001/warc/toy.warc.gz (a few WARC records)
cc-index/table/cc-main/warc/crawl=CC-MAIN-2026-25/subset=warc/part-0.parquet
The parquet index rows carry the exact gzip-member byte ranges of the WARC records,
so `cdxt repackage` can fetch them by range. Three records cover two domains plus a
non-matching control host.
"""
from __future__ import annotations
import os
import sys
from io import BytesIO
import pyarrow as pa
import pyarrow.parquet as pq
from warcio.statusandheaders import StatusAndHeaders
from warcio.warcwriter import WARCWriter
CRAWL = "CC-MAIN-2026-25"
WARC_RELPATH = f"crawl-data/{CRAWL}/segments/0001/warc/toy.warc.gz"
# (url, host, registered_domain, tld, content_languages, payload)
# content_languages mirrors the real index: comma-separated ISO-639-3, most confident
# first — so "deu,eng" must match a filter for either deu or eng.
RECORDS = [
("https://example.com/", "example.com", "example.com", "com", "eng", b"<html>example.com homepage</html>"),
("https://blog.example.org/post", "blog.example.org", "example.org", "org", "deu,eng", b"<html>blog post on example.org with more bytes " + b"x" * 200 + b"</html>"),
("https://www.other.net/", "www.other.net", "other.net", "net", "fra", b"<html>unrelated control host</html>"),
]
def _write_warc(path: str):
os.makedirs(os.path.dirname(path), exist_ok=True)
rows = []
with open(path, "wb") as out:
writer = WARCWriter(out, gzip=True)
writer.write_record(
writer.create_warcinfo_record(os.path.basename(path), {"software": "cc-repackage-fixture"})
)
for url, host, regd, tld, langs, payload in RECORDS:
offset = out.tell()
http_headers = StatusAndHeaders(
"200 OK", [("Content-Type", "text/html")], protocol="HTTP/1.1"
)
rec = writer.create_warc_record(
url, "response", payload=BytesIO(payload), http_headers=http_headers
)
writer.write_record(rec)
length = out.tell() - offset
rows.append(
{
"url": url,
"url_host_name": host,
"url_host_registered_domain": regd,
"url_host_tld": tld,
"warc_filename": WARC_RELPATH,
"warc_record_offset": offset,
"warc_record_length": length,
"content_languages": langs,
"content_mime_type": "text/html",
"fetch_status": 200,
}
)
return rows
def _write_index(root: str, rows):
part_dir = os.path.join(
root, "cc-index", "table", "cc-main", "warc", f"crawl={CRAWL}", "subset=warc"
)
os.makedirs(part_dir, exist_ok=True)
# crawl/subset come from the hive partition path, so they are NOT data columns.
table = pa.Table.from_pylist(rows)
pq.write_table(table, os.path.join(part_dir, "part-0.parquet"))
def build(root: str = "tests/fixtures/cc") -> str:
warc_path = os.path.join(root, WARC_RELPATH)
rows = _write_warc(warc_path)
_write_index(root, rows)
print(f"fixtures built at {root}: {len(rows)} index rows, warc={warc_path}")
return root
if __name__ == "__main__":
build(sys.argv[1] if len(sys.argv) > 1 else "tests/fixtures/cc")