PIN / head_capacity.py
opticalfibre's picture
PIN v5: scripts behind every result
5abe544 verified
Raw
History Blame Contribute Delete
13.9 kB
"""
AT WHAT POINT DOES A READOUT BECOME A MODEL?
A body that never saw a domain could not be made to serve it by adding a
head: an injected member covering unseen classes reached 0.208 against a
0.775 ceiling. That is a safety property if it holds, and quite a strong
one, because it is structural rather than a filter. A filter is a
classifier and can be evaded; a capability absent from the base has no
representation for a head to select.
But it was measured at ONE head size, and a claim of that kind needs a
bound rather than a data point. A head is a readout over features the body
computed -- it can select, weight and recombine, and at a few hundred
values it has almost no room to do anything else. A large enough head is
not a readout at all. Somewhere between the two the guarantee stops
holding, and the question is where.
So this sweeps head capacity against a body trained on classes 0-6 and
frozen, with a member covering 7-9 that the body has never seen, from a
head so small it can barely select to one large enough to be a model in its
own right:
FOLDED LINEAR g row groups, g*classes values. The smallest is a
handful of numbers
FULL LINEAR every hidden unit its own weight
TWO-LAYER a hidden layer of its own before the classes, which is
no longer a readout by any reasonable reading
An in-domain member is swept alongside as a control. If head capacity helps
it and not the unseen one, the limit is the body's features rather than the
head's size, and absence is robust. If the unseen member recovers at some
head size, the guarantee has a bound and a plan tier is a safety parameter
rather than a commercial one.
"""
import numpy as np
import time
import json
from itertools import combinations
try:
import cupy as _cp
_GPU = _cp.cuda.runtime.getDeviceCount() > 0
except Exception:
_GPU = False
xp = _cp if _GPU else np
DT = np.float32
def to_dev(a, dtype=DT):
a = np.asarray(a, dtype=dtype)
return xp.asarray(a) if _GPU else a
def to_host(a):
return _cp.asnumpy(a) if _GPU and isinstance(a, _cp.ndarray) else np.asarray(a)
_MAPS = {}
def tie(shape, K, tag):
key = (tuple(shape), K, tag)
if key not in _MAPS:
import hashlib
h = int(hashlib.md5(str(key).encode()).hexdigest()[:8], 16)
rg = np.random.default_rng(h)
idx = rg.integers(0, K, int(np.prod(shape)))
_MAPS[key] = to_dev(idx, np.int64) if _GPU else idx.astype(np.int64)
return _MAPS[key]
def fold(W, idx, K):
flat = W.reshape(-1)
s = xp.zeros(K, DT); c = xp.zeros(K, DT)
if _GPU:
import cupyx
cupyx.scatter_add(s, idx, flat); cupyx.scatter_add(c, idx, xp.ones_like(flat))
else:
np.add.at(s, idx, flat); np.add.at(c, idx, np.ones_like(flat))
return (s / xp.maximum(c, 1.0))[idx].reshape(W.shape)
def fold_rows(W, groups):
"""Average within row groups: the head's own compression."""
if not groups or groups >= W.shape[0]:
return W
r, c = W.shape
g = groups
while r % g:
g -= 1
m = W.reshape(g, r//g, c).mean(1, keepdims=True)
return xp.broadcast_to(m, (g, r//g, c)).reshape(r, c)
def finit(rg, fan, shape, K=None):
s = np.sqrt(2.0/fan)
if K:
s *= np.sqrt(max(1.0, np.prod(shape)/K))
return rg.normal(0, s, shape)
def train_body(X, Y, masks, D, W, K, epochs, batch, lr, seed):
"""The family, on its own classes. Returns the frozen body."""
rg = np.random.default_rng(seed)
i0 = tie((D, W), K, "body")
W0 = fold(to_dev(finit(rg, D, (D, W), K)), i0, K)
b0 = xp.zeros(W, DT)
S = masks.shape[0]
H = to_dev(finit(rg, W, (S, W, 10)))
B = xp.zeros((S, 10), DT)
P = [W0, b0, H, B]
M = [xp.zeros_like(p) for p in P]; V = [xp.zeros_like(p) for p in P]
n = X.shape[0]; t = 0
for ep in range(epochs):
perm = rg.permutation(n)
for st in range(0, n, batch):
b = perm[st:st+batch]
bi = to_dev(b, np.int64) if _GPU else b
x = X[bi]; y = Y[bi]; mk = masks[:, bi]
z = x @ P[0] + P[1]; a = xp.maximum(z, 0)
lg = xp.einsum('ni,sio->sno', a, P[2]) + P[3][:, None, :]
e = xp.exp(lg - lg.max(-1, keepdims=True))
d = (e/e.sum(-1, keepdims=True) - y[None])*mk[:, :, None]
d = d/max(1, len(b))
da = xp.einsum('sno,sio->ni', d, P[2])
dz = da*(z > 0)
G = [x.T @ dz, dz.sum(0), xp.einsum('ni,sno->sio', a, d), d.sum(1)]
t += 1
for i, (p_, gr) in enumerate(zip(P, G)):
M[i] = 0.9*M[i] + 0.1*gr
V[i] = 0.999*V[i] + 0.001*gr*gr
P[i] = p_ - lr*(M[i]/(1-0.9**t))/(xp.sqrt(V[i]/(1-0.999**t))+1e-8)
P[0] = fold(P[0], i0, K)
return P[0], P[1]
def fit_head(W0, b0, X, Y, mask, W, epochs, batch, lr, seed, groups=None,
hidden=0):
"""Fit ONE member's head against a frozen body.
groups folds the head's rows; hidden gives it a layer of its own, at
which point it is not a readout but a model sitting on frozen
features."""
rg = np.random.default_rng(seed)
n = X.shape[0]
if hidden:
Wh = to_dev(finit(rg, W, (W, hidden)))
bh = xp.zeros(hidden, DT)
Wo = to_dev(finit(rg, hidden, (hidden, 10)))
bo = xp.zeros(10, DT)
P = [Wh, bh, Wo, bo]
else:
Wo = to_dev(finit(rg, W, (W, 10)))
bo = xp.zeros(10, DT)
P = [Wo, bo]
if groups:
P[0] = fold_rows(P[0], groups)
M = [xp.zeros_like(p) for p in P]; V = [xp.zeros_like(p) for p in P]
t = 0
for ep in range(epochs):
perm = rg.permutation(n)
for st in range(0, n, batch):
b = perm[st:st+batch]
bi = to_dev(b, np.int64) if _GPU else b
x = X[bi]; y = Y[bi]; mk = mask[bi]
a = xp.maximum(x @ W0 + b0, 0) # the body is FROZEN
if hidden:
zh = a @ P[0] + P[1]; ah = xp.maximum(zh, 0)
lg = ah @ P[2] + P[3]
else:
lg = a @ P[0] + P[1]
e = xp.exp(lg - lg.max(-1, keepdims=True))
d = (e/e.sum(-1, keepdims=True) - y)*mk[:, None]
d = d/max(1, len(b))
if hidden:
G = [None]*4
G[2] = ah.T @ d; G[3] = d.sum(0)
dh = (d @ P[2].T)*(zh > 0)
G[0] = a.T @ dh; G[1] = dh.sum(0)
else:
G = [a.T @ d, d.sum(0)]
t += 1
for i, (p_, gr) in enumerate(zip(P, G)):
M[i] = 0.9*M[i] + 0.1*gr
V[i] = 0.999*V[i] + 0.001*gr*gr
P[i] = p_ - lr*(M[i]/(1-0.9**t))/(xp.sqrt(V[i]/(1-0.999**t))+1e-8)
if groups and not hidden:
P[0] = fold_rows(P[0], groups)
return P
def score(W0, b0, P, X, y, mask, hidden):
a = xp.maximum(X @ W0 + b0, 0)
if hidden:
lg = xp.maximum(a @ P[0] + P[1], 0) @ P[2] + P[3]
else:
lg = a @ P[0] + P[1]
m = to_host(mask).astype(bool)
return float((to_host(lg.argmax(1))[m] == y[m]).mean())
def head_values(W, groups, hidden):
if hidden:
return W*hidden + hidden + hidden*10 + 10
g = groups or W
while W % g:
g -= 1
return g*10 + 10
def load(name):
if name == "fashion":
from tensorflow import keras
(a, b), (c, d) = keras.datasets.fashion_mnist.load_data()
X = np.concatenate([a, c]).reshape(-1, 784).astype(np.float32)/255.0
y = np.concatenate([b, d]).ravel().astype(np.int64)
else:
from sklearn.datasets import load_digits
dd = load_digits()
X = dd.data.astype(np.float32)/16.0; y = dd.target.astype(np.int64)
return X, y
CFG = dict(dataset="fashion", width=64, K=512, n_train=8000, epochs=40,
batch=128, lr=1e-3, seeds=(0, 1, 2),
family_classes=(0, 1, 2, 3, 4, 5, 6),
unseen_classes=(7, 8, 9))
# from a head that can barely select to one that is a model in its own right
HEADS = [("g=2", 2, 0), ("g=4", 4, 0), ("g=8", 8, 0), ("g=16", 16, 0),
("g=32", 32, 0), ("full linear", None, 0),
("2-layer h=16", None, 16), ("2-layer h=64", None, 64),
("2-layer h=256", None, 256)]
def one_seed(X, y, seed, cfg):
rg = np.random.default_rng(seed)
fam = list(cfg["family_classes"]); uns = list(cfg["unseen_classes"])
D, W = X.shape[1], cfg["width"]
subs = [tuple(sorted(c)) for c in combinations(fam, 4)]
subs = [subs[i] for i in rg.choice(len(subs), 6, replace=False)]
in_dom = tuple(sorted(fam[:4])) # FIXED across seeds
out_dom = tuple(sorted(uns))
def split(cl, n):
i = np.where(np.isin(y, cl))[0]; rg.shuffle(i)
return i[:n], i[n:n+2000]
ftr, _ = split(fam, cfg["n_train"])
itr, ite = split(in_dom, cfg["n_train"]//3)
otr, ote = split(uns, cfg["n_train"]//3)
dev = lambda i: to_dev(X[i])
Y = lambda i: to_dev(np.eye(10, dtype=np.float32)[y[i]])
m1 = lambda i, s: to_dev(np.isin(y[i], s).astype(np.float32))
mS = lambda i, ss: to_dev(np.stack([np.isin(y[i], s)
for s in ss]).astype(np.float32))
W0, b0 = train_body(dev(ftr), Y(ftr), mS(ftr, subs), D, W, cfg["K"],
cfg["epochs"], cfg["batch"], cfg["lr"], seed)
out = {}
for nm, g, h in HEADS:
for tag, tr, te, sub in (("in", itr, ite, in_dom),
("out", otr, ote, out_dom)):
P = fit_head(W0, b0, dev(tr), Y(tr), m1(tr, sub), W,
cfg["epochs"], cfg["batch"], cfg["lr"], seed+1,
groups=g, hidden=h)
out[f"{nm}/{tag}"] = score(W0, b0, P, dev(te), y[te],
m1(te, sub), h)
# ceilings: a whole model of its own for each member
for tag, tr, te, sub in (("in", itr, ite, in_dom),
("out", otr, ote, out_dom)):
Wc, bc = train_body(dev(tr), Y(tr), mS(tr, [sub]), D, W, cfg["K"],
cfg["epochs"], cfg["batch"], cfg["lr"], seed+9)
P = fit_head(Wc, bc, dev(tr), Y(tr), m1(tr, sub), W, cfg["epochs"],
cfg["batch"], cfg["lr"], seed+9)
out[f"ceiling/{tag}"] = score(Wc, bc, P, dev(te), y[te],
m1(te, sub), 0)
return out
def main(**over):
CFG.update(over)
t0 = time.time()
print("=" * 78)
print("AT WHAT POINT DOES A READOUT BECOME A MODEL?")
print("=" * 78)
print(f" backend: {'cupy (GPU)' if _GPU else 'numpy (CPU)'}")
for k, v in CFG.items():
print(f" {k:16s} = {v}")
W = CFG["width"]
print(f"\n the body is trained on classes {CFG['family_classes']} and")
print(f" FROZEN. The unseen member covers {CFG['unseen_classes']}, which")
print(f" it has never seen in any form.\n")
print(f" {'head':>14s} {'values':>8s}")
for nm, g, h in HEADS:
print(f" {nm:>14s} {head_values(W, g, h):8,}")
print("=" * 78, flush=True)
X, y = load(CFG["dataset"])
runs = []
for s in CFG["seeds"]:
runs.append(one_seed(X, y, s, CFG))
print(f" seed {s} done [{time.time()-t0:.0f}s]", flush=True)
json.dump(runs, open("head_capacity.json", "w"), indent=2)
m = {k: float(np.mean([r[k] for r in runs])) for k in runs[0]}
sd = {k: float(np.std([r[k] for r in runs])) for k in runs[0]}
print("\n" + "=" * 78)
print(" READOUT")
print("=" * 78)
ci, co = m["ceiling/in"], m["ceiling/out"]
print(f" ceilings: in-domain {ci:.4f}, unseen {co:.4f}")
print(f" (each member with a body of its own)\n")
print(f" {'head':>14s} {'values':>8s} {'in-domain':>10s} {'vs ceil':>8s}"
f" {'UNSEEN':>8s} {'vs ceil':>8s}")
for nm, g, h in HEADS:
i_, o_ = m[f"{nm}/in"], m[f"{nm}/out"]
print(f" {nm:>14s} {head_values(W, g, h):8,} {i_:10.4f} "
f"{i_-ci:+8.4f} {o_:8.4f} {o_-co:+8.4f}")
outs = np.array([m[f"{nm}/out"] for nm, _, _ in HEADS])
ins = np.array([m[f"{nm}/in"] for nm, _, _ in HEADS])
s_ = max(sd.values())
print(f"\n seed spread (worst) {s_:.4f}")
print(f" the unseen member moves {outs.max()-outs.min():+.4f} across a")
print(f" {head_values(W, None, 256)//head_values(W, 2, 0)}-fold range of"
f" head capacity")
print(f" the in-domain member moves {ins.max()-ins.min():+.4f}")
print()
recovered = [nm for (nm, g, h), o in zip(HEADS, outs) if o > co - 2*s_]
if recovered:
print(f" THE GUARANTEE HAS A BOUND. The unseen member reaches its")
print(f" ceiling once the head holds "
f"{head_values(W, *[x[1:] for x in HEADS if x[0]==recovered[0]][0]):,}"
f" values ({recovered[0]}), so a")
print(f" large enough head recovers a capability the body never had")
print(f" and head size is a safety parameter, not a commercial one.")
elif outs.max() - outs.min() > 4*s_:
print(f" CAPACITY HELPS BUT DOES NOT RESCUE. The unseen member")
print(f" improves with head size and never reaches its ceiling, so")
print(f" absence degrades rather than holds absolutely — a bound")
print(f" exists and is above the range swept here.")
else:
print(f" ABSENCE IS ROBUST. Head capacity moves the unseen member by")
print(f" less than seed noise across the whole range, including")
print(f" heads with a hidden layer of their own. What limits it is")
print(f" the body's features and not the head's size, which is what")
print(f" a structural guarantee would look like.")
print(f"\n total {time.time()-t0:.0f}s; wrote head_capacity.json")
if __name__ == "__main__":
main()