"""Audit Assumption 4.2 (kappa >= sqrt(n+3)) across OpenML binary-classification datasets. Definition 4.1 of the paper: for a sorted 1-D dimension x_1<...0) and "successful" when kappa >= sqrt(n+3). Datasets are processed cheapest-first (by cells = features x instances) and results are appended incrementally so the audit can be stopped at any point. """ import json, os, sys, time, warnings import numpy as np import certifi, ssl, urllib.request warnings.filterwarnings("ignore") os.environ["SSL_CERT_FILE"] = certifi.where() os.environ["REQUESTS_CA_BUNDLE"] = certifi.where() ssl._create_default_https_context = lambda *a, **k: ssl.create_default_context(cafile=certifi.where()) os.environ.setdefault("SKLEARN_DATA", os.path.abspath("openml_cache")) from sklearn.datasets import fetch_openml CELL_CAP = int(sys.argv[1]) if len(sys.argv) > 1 else 8_000_000 TIME_BUDGET_S = int(sys.argv[2]) if len(sys.argv) > 2 else 3600 OUT = "openml_audit.jsonl" meta = json.load(open("openml_binary.json"))["data"]["dataset"] rows = [] for x in meta: q = {k["name"]: float(k["value"]) for k in x.get("quality", []) if k.get("value") not in (None, "")} nf, ni = q.get("NumberOfFeatures", 0), q.get("NumberOfInstances", 0) if nf and ni: rows.append({"did": int(x["did"]), "name": x["name"], "nf": nf, "ni": ni, "cells": nf * ni}) rows = [r for r in rows if r["cells"] <= CELL_CAP] # The paper's ~933k audited dimensions live in the high-dimensional datasets, so # take them first; most are wide-but-short and therefore cheap to download. rows.sort(key=lambda r: -r["nf"]) done = set() if os.path.exists(OUT): for line in open(OUT): try: done.add(json.loads(line)["did"]) except Exception: pass start = time.time() out = open(OUT, "a") for i, r in enumerate(rows): if r["did"] in done: continue if time.time() - start > TIME_BUDGET_S: print("time budget reached", flush=True) break try: bunch = fetch_openml(data_id=r["did"], as_frame=True, parser="auto", data_home="openml_cache") df = bunch.data total = valid = success = 0 strict_valid = strict_success = 0 kappas = [] for col in df.columns: s = df[col] v = s.to_numpy() if v.dtype.kind not in "fiu": # categorical / string dimension: still a dimension of the dataset try: v = s.astype("float64").to_numpy() except Exception: total += 1 continue v = v[np.isfinite(v)] total += 1 # (a) strict reading: a dimension is "valid" only if it contains no # repeated value at all. if v.size >= 3: g = np.diff(np.sort(v)) if g.min() > 0: strict_valid += 1 if g.max() / g.min() >= np.sqrt(v.size + 3): strict_success += 1 # (b) de-duplicated reading: repeated values are dropped ("the # theoretical probability of repeated values is zero"), and a # dimension is valid when >=3 distinct values survive. u = np.unique(v) n = u.size if n < 3: continue gaps = np.diff(u) L, U = gaps.min(), gaps.max() valid += 1 kappa = U / L if kappa >= np.sqrt(n + 3): success += 1 else: kappas.append(float(kappa)) rec = {"did": r["did"], "name": r["name"], "n_instances": int(r["ni"]), "total_dims": total, "valid_dims": valid, "success_dims": success, "strict_valid_dims": strict_valid, "strict_success_dims": strict_success, "failing_kappas": kappas[:20]} out.write(json.dumps(rec) + "\n") out.flush() if i % 25 == 0: print(f"[{i}/{len(rows)}] {r['name'][:30]} total={total} valid={valid} ok={success}", flush=True) except Exception as e: out.write(json.dumps({"did": r["did"], "name": r["name"], "error": repr(e)[:200]}) + "\n") out.flush() out.close() print("finished pass", flush=True)