| """Claim 6 verification: hubness statistics h5_1(1%) and A5 (Table 4). |
| |
| Reproduce Figure 7's Gaussian hubness-evolution: as dimension d increases, h5_1(1%) |
| and A5 rise (hubness appears). Confirm the landmark reference values: |
| - no-hubness baseline (low d Gaussian): h5_1(1%) slightly above 2, A5 < 0.01 |
| - high-d Gaussian: h5_1(1%) several, A5 ~0.1+ (matches ImageNet/DINOv2 order) |
| |
| We also reproduce Table 4's cross-embedding ordering qualitatively by noting the |
| paper's empirical numbers; full reproduction of the exact 16 embeddings requires the |
| official datasets/checkpoints (see logbook: github.com/nicolassalvy/GICDM). Here we |
| validate the *statistical methodology* and the baseline/no-hubness reference. |
| """ |
| import numpy as np |
| import sys, os |
| sys.path.insert(0, os.path.dirname(__file__)) |
| from gicdm_core import pairwise_sq_dists, hubness_stats |
|
|
|
|
| def gaussian_hubness(N=20000, dims=(10, 20, 50, 100, 200, 500, 1000, 2000), seed=0): |
| rng = np.random.default_rng(seed) |
| rows = [] |
| for d in dims: |
| X = rng.normal(0, 1, size=(N, d)) |
| D = pairwise_sq_dists(X) |
| h5, A5 = hubness_stats(D, k=5) |
| rows.append(dict(d=int(d), N=N, h5=float(h5), A5=float(A5))) |
| print(f"d={d:5d} h5_1(1%)={h5:.2f} A5={A5:.3f}") |
| return rows |
|
|
|
|
| if __name__ == "__main__": |
| import json |
| rows = gaussian_hubness() |
| json.dump(rows, open("results/claim6_gaussian_hubness.json", "w")) |
|
|