Spaces:
Sleeping
Sleeping
File size: 9,822 Bytes
15eb4b9 8eaca44 15eb4b9 55c930e 15eb4b9 bb9acda 8eaca44 15eb4b9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 | """Central configuration for the CC WARC repackaging Space.
Everything that might need tweaking when the cdx_toolkit branch moves, pricing
changes, or the read backend is switched lives here so the rest of the code stays
declarative.
"""
from __future__ import annotations
import os
# --- cdx_toolkit install -----------------------------------------------------
# Git branch (PR #73) that provides the `repackage` command with pluggable
# range-job sources (cdx / sql{athena,duckdb} / csv). Both the `commoncrawl` and
# `cocrawler` remotes carry this branch at the same commit; `commoncrawl` is the
# current org.
CDXT_GIT = "git+https://github.com/commoncrawl/cdx_toolkit.git@feat/warc-range-sources"
# Only the fetch job runs cdxt; it needs the hf extra for hf:// reads/writes.
CDXT_PIP_SPEC = f"cdx_toolkit[hf] @ {CDXT_GIT}"
# The estimate job's index SQL is vendored in index_query.py, so it needs no
# cdx_toolkit (and no git clone/build of it) — just DuckDB and the bucket API.
# Keeping it off the estimate job's critical path makes the estimate start sooner.
ESTIMATE_PIP_SPECS = ("duckdb", "huggingface_hub")
# --- Common Crawl data on the HF bucket --------------------------------------
CC_BUCKET = "commoncrawl/commoncrawl" # read-only mirror of CC data + index
# DuckDB cannot read hf://buckets/... directly (only hf://datasets|spaces), so the
# columnar index is read from a read-only bucket Volume mounted into the job.
CC_MOUNT = "/cc"
CC_INDEX_SUBPATH = "cc-index/table/cc-main/warc/" # hive-partitioned: crawl=.../subset=warc/*.parquet
# Real fetch jobs read WARC byte ranges from the CC bucket over its CDN-fronted
# resolve URL (benchmark-proven ~1.6k rec/s on cpu-upgrade). cdx_toolkit's range
# reader supports s3:// / hf:// / http(s):// — NOT bare local paths — so the local
# podman simulation serves the fixture over a Range-capable HTTP server instead
# (see CC_PODMAN_WARC_HTTP below).
CC_WARC_CDN_PREFIX = f"hf://buckets/{CC_BUCKET}"
# --- Target output -----------------------------------------------------------
# The user's output bucket is mounted read-write here in both jobs; the range-jobs
# CSV and the output WARC(s) are written under <OUT_MOUNT>/<path>/.
OUT_MOUNT = "/out"
RANGES_CSV_NAME = "ranges.csv"
# --- HF Jobs hardware --------------------------------------------------------
JOB_IMAGE = "python:3.12"
DEFAULT_FLAVOR = "cpu-upgrade"
# USD per hour (https://huggingface.co/docs/huggingface_hub/guides/jobs).
FLAVOR_HOURLY_USD = {
"cpu-basic": 0.01,
"cpu-upgrade": 0.03,
"cpu-performance": 1.90,
"cpu-xl": 1.00,
}
# Conservative sustained fetch throughput (records/second) per flavor for the cost
# estimate. cpu-upgrade + hf-cdn measured ~1637 rec/s in the cdxt-experiment
# benchmark; we round down for a safety margin.
FLAVOR_REC_PER_S = {
"cpu-basic": 450,
"cpu-upgrade": 1500,
"cpu-performance": 4000,
"cpu-xl": 2500,
}
# --- Cost-estimate overheads (from the cdxt-experiment benchmarks) -----------
# Fixed container overhead per job: pip-installing cdx_toolkit etc. (~10s in HF Jobs;
# scheduling time is not billed and is excluded).
CONTAINER_OVERHEAD_S = 10
# Multi-process runs merge the per-process shard WARCs into one <prefix>-NNN series.
# The merge has a fixed setup cost (~35s at 1 GiB, independent of process count) plus a
# size-proportional term (~300 MB/s with the concurrent merge). Single-process jobs do
# not merge. (cdxt-experiment RESULTS §11–12.)
MERGE_BASE_S = 35
MERGE_THROUGHPUT_BYTES_PER_S = 300_000_000
# Default tuning for the fetch job (from the cdxt-experiment benchmarks).
DEFAULT_PARALLEL_READERS = 48
# vCPU count per flavor — the upper bound on fetch worker processes.
FLAVOR_VCPUS = {
"cpu-basic": 2,
"cpu-upgrade": 8,
"cpu-performance": 32,
"cpu-xl": 16,
}
# Records needed to justify an extra worker process. Multi-process mode shards by
# WARC filename and each shard writes its own file, so for small jobs we use fewer
# processes to avoid many tiny/empty output shards.
RECORDS_PER_PROCESS = 20_000
def compute_processes(n_records: int, flavor: str) -> int:
"""Worker processes to use: scale with record count, capped at the flavor's vCPUs."""
import math
vcpus = FLAVOR_VCPUS.get(flavor, FLAVOR_VCPUS[DEFAULT_FLAVOR])
if n_records <= 0:
return 1
return max(1, min(vcpus, math.ceil(n_records / RECORDS_PER_PROCESS)))
# Target output WARC file size in bytes (WARC convention ~1 GiB).
WARC_TARGET_SIZE = 1_000_000_000
# Timeouts (cdx_toolkit accepts strings like "30m"/"6h").
ESTIMATE_TIMEOUT = "30m"
FETCH_TIMEOUT = "6h"
# --- Cost guard --------------------------------------------------------------
# cdx_toolkit auto-approves an index scan bounded to <= this many crawls; beyond
# that it needs --confirm-cost (which we always pass for non-interactive jobs). We
# surface a UI warning at the same threshold.
COST_GUARD_MAX_CRAWLS = 10
# --- Available crawls (fallback) ---------------------------------------------
# Used when the live bucket listing fails. Only crawls that have BOTH WARC data
# (crawl-data/) and a columnar index (cc-index/table/cc-main/warc/) are usable; the
# Refresh button recomputes this intersection live. (CC-MAIN-2026-25 has data but no
# index yet, so it is intentionally omitted.)
FALLBACK_CRAWLS = ["CC-MAIN-2026-25", "CC-MAIN-2026-21", "CC-MAIN-2026-17"]
# --- Content languages -------------------------------------------------------
# The index's `content_languages` column holds the ISO-639-3 code(s) of a document as
# detected by CLD2, comma-separated and most-confident-first (e.g. "fra,eng"). The
# codes below are CLD2's detectable languages — i.e. every value that can appear in
# the column — as (display name, ISO-639-3 code), sorted by name.
CONTENT_LANGUAGES = [
("Afrikaans", "afr"), ("Albanian", "sqi"), ("Amharic", "amh"), ("Arabic", "ara"),
("Armenian", "hye"), ("Azerbaijani", "aze"), ("Basque", "eus"), ("Belarusian", "bel"),
("Bengali", "ben"), ("Bihari", "bih"), ("Bosnian", "bos"), ("Breton", "bre"),
("Bulgarian", "bul"), ("Burmese", "mya"), ("Catalan", "cat"), ("Cherokee", "chr"),
("Chinese", "zho"), ("Croatian", "hrv"), ("Czech", "ces"), ("Danish", "dan"),
("Dhivehi", "div"), ("Dutch", "nld"), ("English", "eng"), ("Esperanto", "epo"),
("Estonian", "est"), ("Faroese", "fao"), ("Finnish", "fin"), ("French", "fra"),
("Frisian", "fry"), ("Galician", "glg"), ("Georgian", "kat"), ("German", "deu"),
("Greek", "ell"), ("Guarani", "grn"), ("Gujarati", "guj"), ("Haitian Creole", "hat"),
("Hausa", "hau"), ("Hawaiian", "haw"), ("Hebrew", "heb"), ("Hindi", "hin"),
("Hmong", "hmn"), ("Hungarian", "hun"), ("Icelandic", "isl"), ("Igbo", "ibo"),
("Indonesian", "ind"), ("Interlingua", "ina"), ("Interlingue", "ile"), ("Irish", "gle"),
("Italian", "ita"), ("Japanese", "jpn"), ("Javanese", "jav"), ("Kannada", "kan"),
("Kazakh", "kaz"), ("Khmer", "khm"), ("Kinyarwanda", "kin"), ("Korean", "kor"),
("Kurdish", "kur"), ("Kyrgyz", "kir"), ("Lao", "lao"), ("Latin", "lat"),
("Latvian", "lav"), ("Lithuanian", "lit"), ("Luxembourgish", "ltz"),
("Macedonian", "mkd"), ("Malagasy", "mlg"), ("Malay", "msa"), ("Malayalam", "mal"),
("Maltese", "mlt"), ("Maori", "mri"), ("Marathi", "mar"), ("Mongolian", "mon"),
("Nepali", "nep"), ("Norwegian", "nor"), ("Nyanja", "nya"), ("Occitan", "oci"),
("Oriya", "ori"), ("Pashto", "pus"), ("Persian", "fas"), ("Polish", "pol"),
("Portuguese", "por"), ("Punjabi", "pan"), ("Quechua", "que"), ("Romanian", "ron"),
("Romansh", "roh"), ("Russian", "rus"), ("Samoan", "smo"), ("Scots Gaelic", "gla"),
("Serbian", "srp"), ("Sesotho", "sot"), ("Shona", "sna"), ("Sindhi", "snd"),
("Sinhala", "sin"), ("Slovak", "slk"), ("Slovenian", "slv"), ("Somali", "som"),
("Spanish", "spa"), ("Sundanese", "sun"), ("Swahili", "swa"), ("Swedish", "swe"),
("Tajik", "tgk"), ("Tamil", "tam"), ("Tatar", "tat"), ("Telugu", "tel"),
("Thai", "tha"), ("Tibetan", "bod"), ("Tigrinya", "tir"), ("Tonga", "ton"),
("Turkish", "tur"), ("Turkmen", "tuk"), ("Ukrainian", "ukr"), ("Urdu", "urd"),
("Uyghur", "uig"), ("Uzbek", "uzb"), ("Vietnamese", "vie"), ("Welsh", "cym"),
("Xhosa", "xho"), ("Yiddish", "yid"), ("Yoruba", "yor"), ("Zulu", "zul"),
]
CONTENT_LANGUAGE_CODES = frozenset(code for _, code in CONTENT_LANGUAGES)
def language_choices() -> list:
"""Gradio (label, value) choices, e.g. ("German (deu)", "deu")."""
return [(f"{name} ({code})", code) for name, code in CONTENT_LANGUAGES]
# --- Executor selection ------------------------------------------------------
# "hfjobs" submits real HF Jobs; "podman" runs the same commands in a local
# container against fixtures for verification.
EXECUTOR = os.environ.get("CC_REPACKAGE_EXECUTOR", "hfjobs")
# For the podman executor: map the abstract bucket mounts to local directories.
PODMAN_CC_DIR = os.environ.get("CC_PODMAN_CC_DIR", "")
PODMAN_OUT_DIR = os.environ.get("CC_PODMAN_OUT_DIR", "")
# Image used by the podman executor (pre-build with cdx_toolkit to skip pip each run).
PODMAN_IMAGE = os.environ.get("CC_PODMAN_IMAGE", JOB_IMAGE)
# If set ("1"), the podman executor skips `pip install` (image already has cdx_toolkit).
PODMAN_SKIP_PIP = os.environ.get("CC_PODMAN_SKIP_PIP", "0") == "1"
# Base HTTP(S) URL the fetch job reads fixture WARCs from in podman mode (a Range-
# capable server serving CC_PODMAN_CC_DIR). From inside the container the host is
# reachable as host.containers.internal.
PODMAN_WARC_HTTP = os.environ.get(
"CC_PODMAN_WARC_HTTP", "http://host.containers.internal:8080"
)
def cc_index_path(mount: str = CC_MOUNT) -> str:
"""Local path to the columnar index under the CC bucket mount."""
return f"{mount.rstrip('/')}/{CC_INDEX_SUBPATH}"
|