File size: 6,878 Bytes
919fd68
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
"""Exploration floor for uncertain routing — pages, experts, layers, capabilities.

OPERATOR DOCTRINE (additive-only -- governs every helper below): the 12b
parent/base is a FROZEN HOLLOW VOCAB SUBSTRATE (a tiny lego that just
provides tokenization); the REAL model is the 1T+ ADDITIVE machinery (NoNE
pages, experts, RBO, causal/trauma/MitM/coverage-pressure) that we train.
THIS module widens the exploration aperture ONLY over the ADDITIVE route
bank (rows / pages / experts / layers / capabilities in OUR additive
stack); it never touches/improves/retains the frozen base, and the base is
never a route/floor/verification target here.

OPERATOR DOCTRINE (verbatim, governs every helper below): "the model cannot
constrict to a top k=1 for rows or pages — it needs to be top-16 minimum as a
hard floor at least when uncertain, and able to push higher or smaller as needed
and extend itself... at an uncertain element it should always start with at
least 10% of available layers and experts and capabilities, and then use the
RBO to rotate and extend... exploration is the biggest thing (Bayesian
networks, anti-Thompson, MILT, hill climbing)... consensus based upon the final
emitted experts and knowledge at every stage."

This module is the single source of truth for that floor.  The learned router
width is honoured when the router is confident; when it is uncertain the width
is bounded below by the larger of:

  * **Top-k floor**: at least 16 candidates (rows / pages / experts / layers /
    capabilities in the route bank).
  * **Fraction floor**: at least 10% of the available bank width.

RBO rotation / extension / anti-Thompson / hill-climbing remain the authority
for *which* candidates enter the bank and for pushing *above* the floor — the
floor only sets the minimum starting aperture and never narrows a confident
route.  This is **default-on and provably no-regression for the confident
case**: when ``uncertain`` is false every helper returns the unchanged learned
width.  The ONE allowed hard backstop is the capacity clamp (GPU/row capacity),
which is an anti-OOM guardrail (``# guardrail:not-cap``) not a capability cap.

These helpers intentionally take primitives and never read configuration: every
caller already knows ``available`` (the bank width it just routed against) and
``uncertain`` (its own confidence / route-mass / entropy signal), so the floor
is pure arithmetic with no side effects on the live training graph.
"""

from __future__ import annotations

import math

import torch
from torch import Tensor

EXPLORATION_MIN_TOP_K = 16
EXPLORATION_MIN_AVAILABLE_FRACTION = 0.10
EXPLORATION_UNCERTAINTY_MASS_THRESHOLD = 0.35


def _confidence_floor_active_count(
    available_count: int,
    uncertainty: float | bool,
    *,
    hard_floor: int = EXPLORATION_MIN_TOP_K,
    uncertainty_fraction: float = EXPLORATION_MIN_AVAILABLE_FRACTION,
) -> int:
    """Minimum active count under the operator doctrine.

    Returns ``max(hard_floor, ceil(uncertainty_fraction * available_count))``
    when uncertain, else ``0`` (meaning "no floor — honour the learned width").
    Callers then take ``min(available, max(learned_k, floor))`` so a confident
    route is unchanged and an uncertain route can only widen, never narrow.

    ``uncertainty`` accepts either a bool (confident/exploring) or a float in
    ``[0, 1]`` (a probability / entropy mass); any value ``> 0`` is treated as
    "at least partially uncertain" so a graded signal still engages the floor.
    """

    if available_count <= 0:
        return 0
    uncertain = float(uncertainty) > 0.0 if isinstance(uncertainty, float) else bool(uncertainty)
    if not uncertain:
        return 0
    fraction_k = max(1, math.ceil(available_count * float(uncertainty_fraction)))
    return max(int(hard_floor), fraction_k)


def capacity_clamp_active_count(
    requested: int,
    *,
    capacity: int,
) -> int:
    """Clamp a requested active count to live GPU/row capacity.

    This is the ONE allowed hard backstop. It is an anti-OOM guardrail
    (``# guardrail:not-cap``) — it never narrows below the doctrine floor for
    capability reasons, only because the device literally cannot resident more
    rows/experts this wave. Callers MUST log when this binds.
    """

    return max(0, min(int(requested), int(capacity)))


def exploration_route_width(
    *,
    available: int,
    learned_k: int,
    uncertain: bool,
) -> int:
    """Return route width after applying exploration floors when uncertain.

    Equivalent to ``min(available, max(learned_k, _confidence_floor_active_count(...)))``:
    one source of truth. When confident the learned width is returned unchanged
    (clamped to ``[1, available]``); when uncertain the doctrine floor
    (``max(16, ceil(10% * available))``) widens the aperture, RBO still rotates
    and extends above it.
    """

    if available < 1:
        return 0
    learned = int(learned_k)
    if learned < 1:
        learned = 1
    floor_k = _confidence_floor_active_count(
        available,
        uncertain,
    )
    if floor_k < 1:
        # Confident case: honour the learned width exactly.
        return min(available, learned)
    return min(available, max(learned, floor_k))


def route_uncertainty_from_mass(
    route_mass_t: Tensor,
    *,
    threshold: float = EXPLORATION_UNCERTAINTY_MASS_THRESHOLD,
) -> Tensor:
    """Low retained route mass implies exploration (per batch row)."""

    mass = route_mass_t.detach().reshape(-1).to(dtype=torch.float32)
    finite_mass = torch.where(torch.isfinite(mass), mass, torch.zeros_like(mass))
    return finite_mass.lt(float(threshold))


def batch_route_width_t(
    *,
    available: int,
    learned_k: int,
    uncertain_mask_t: Tensor,
) -> Tensor:
    """Per-row exploration width ``[batch]`` long."""

    mask = uncertain_mask_t.reshape(-1).to(dtype=torch.bool)
    if available < 1:
        return torch.zeros_like(mask, dtype=torch.long)
    available_t = torch.ones_like(mask, dtype=torch.long) * available
    learned_t = (
        torch.ones_like(mask, dtype=torch.long)
        * max(1, min(available, learned_k))
    )
    fractional_floor_t = torch.div(
        available_t + 9,
        10,
        rounding_mode="floor",
    )
    exploration_floor_t = torch.minimum(
        available_t,
        torch.maximum(
            fractional_floor_t,
            torch.ones_like(available_t) * EXPLORATION_MIN_TOP_K,
        ),
    )
    return torch.where(
        mask,
        torch.maximum(learned_t, exploration_floor_t),
        learned_t,
    )


__all__ = [
    "EXPLORATION_MIN_AVAILABLE_FRACTION",
    "EXPLORATION_MIN_TOP_K",
    "EXPLORATION_UNCERTAINTY_MASS_THRESHOLD",
    "_confidence_floor_active_count",
    "batch_route_width_t",
    "capacity_clamp_active_count",
    "exploration_route_width",
    "route_uncertainty_from_mass",
]