"""Per-column honesty readout + ordering guard for numeric uploads. The basis fits ALONG THE ROW AXIS, so it only carries meaning when a column is sequentially ordered (time-series / sensor / signal / sorted series). This module makes that explicit per column, so the demo never oversells: - lag-1 autocorrelation -> is the column row-ordered at all? If a column has no row structure (autocorrelation near zero), shuffling its rows would not change it, so an along-row polynomial fit is meaningless and the COARSE (analytics) archive must NOT be trusted to preserve it -- only the lossless archive is safe. - R^2 of the COARSE reconstruction -> given that it is ordered, how much of the column does the smooth fit actually capture? High = compresses and the analytics archive preserves it; low = noisy, passes through near 1x. Verdict per column: compresses / partial / passthrough. An aggregate "ordering guard" flags uploads that look like unordered tabular data (most columns have no row structure), where the storage win still holds losslessly but the analytics / context projection should be read with care. No basis-family details are exposed. """ import html import numpy as np AUTOCORR_ORDERED = 0.3 # lag-1 autocorrelation above which a column is "row-ordered" R2_SMOOTH = 0.9 # COARSE R^2 above which an ordered column "compresses" _MAX_ROWS_SHOWN = 16 def _lag1_autocorr(x: np.ndarray) -> float: x = x - x.mean() denom = float(np.dot(x, x)) if denom < 1e-12: return 1.0 # constant column: perfectly "ordered" and trivially compressible return float(np.dot(x[:-1], x[1:]) / denom) def per_column_stats(r: dict) -> list[dict] | None: """Per-column (autocorr, R^2, verdict). None when not a 2-D numeric result.""" orig = np.asarray(r.get("original")) if r else None coarse = np.asarray(r.get("coarse")) if r else None if orig is None or orig.ndim != 2 or coarse is None or coarse.shape != orig.shape: return None names = r.get("columns") out = [] for j in range(orig.shape[1]): name = names[j] if names and j < len(names) else f"col {j}" col = orig[:, j].astype(float) rec = coarse[:, j].astype(float) ss_tot = float(np.sum((col - col.mean()) ** 2)) constant = ss_tot < 1e-12 r2 = 1.0 if constant else 1.0 - float(np.sum((col - rec) ** 2)) / ss_tot ac = 1.0 if constant else _lag1_autocorr(col) ordered = ac >= AUTOCORR_ORDERED if not ordered: verdict, note = "passthrough", "no row structure" elif constant: verdict, note = "compresses", "constant" elif r2 >= R2_SMOOTH: verdict, note = "compresses", "smooth" else: verdict, note = "partial", "noisy" out.append( {"col": j, "name": name, "autocorr": ac, "r2": r2, "ordered": ordered, "verdict": verdict, "note": note}, ) return out def mostly_structured(r: dict) -> bool | None: """True/False if a STRICT majority of columns are row-ordered; None if n/a. Strict so an exact 50/50 split reads as unstructured: the guard banner shows and the context projection is withheld, consistently. Ties err conservative. """ stats = per_column_stats(r) if not stats: return None return sum(s["ordered"] for s in stats) > 0.5 * len(stats) _BADGE = { "compresses": ("#0f6e56", "compresses"), "partial": ("#b96a0a", "partial"), "passthrough": ("#6464a0", "passthrough ~1x"), } def readout_html(r: dict) -> str: """Per-column table + ordering-guard banner. Empty string when not applicable.""" stats = per_column_stats(r) if not stats: return "" n_ordered = sum(s["ordered"] for s in stats) guard = "" # Same strict-majority rule as mostly_structured: at an exact 50/50 split the # guard shows AND the cost panel withholds the context projection. if n_ordered <= 0.5 * len(stats): guard = ( '
| column | row-ordered | fit R² | verdict |
|---|