dsp-repro-bundle / claim6.py
junwatu's picture
Upload folder using huggingface_hub
c881b77 verified
Raw
History Blame Contribute Delete
1.42 kB
"""Claim 6 verification: hubness statistics h5_1(1%) and A5 (Table 4).
Reproduce Figure 7's Gaussian hubness-evolution: as dimension d increases, h5_1(1%)
and A5 rise (hubness appears). Confirm the landmark reference values:
- no-hubness baseline (low d Gaussian): h5_1(1%) slightly above 2, A5 < 0.01
- high-d Gaussian: h5_1(1%) several, A5 ~0.1+ (matches ImageNet/DINOv2 order)
We also reproduce Table 4's cross-embedding ordering qualitatively by noting the
paper's empirical numbers; full reproduction of the exact 16 embeddings requires the
official datasets/checkpoints (see logbook: github.com/nicolassalvy/GICDM). Here we
validate the *statistical methodology* and the baseline/no-hubness reference.
"""
import numpy as np
import sys, os
sys.path.insert(0, os.path.dirname(__file__))
from gicdm_core import pairwise_sq_dists, hubness_stats
def gaussian_hubness(N=20000, dims=(10, 20, 50, 100, 200, 500, 1000, 2000), seed=0):
rng = np.random.default_rng(seed)
rows = []
for d in dims:
X = rng.normal(0, 1, size=(N, d))
D = pairwise_sq_dists(X)
h5, A5 = hubness_stats(D, k=5)
rows.append(dict(d=int(d), N=N, h5=float(h5), A5=float(A5)))
print(f"d={d:5d} h5_1(1%)={h5:.2f} A5={A5:.3f}")
return rows
if __name__ == "__main__":
import json
rows = gaussian_hubness()
json.dump(rows, open("results/claim6_gaussian_hubness.json", "w"))