animap-gpu / app /adapters /embedding /backbones.py
bluman1's picture
Publish services/inference
4b98524 verified
Raw
History Blame Contribute Delete
17.3 kB
"""Frozen visual embeddings, through onnxruntime.
Directive §3 on DINOv3: "Do not assume fine-tuning is required. Start with frozen
embeddings + nearest-neighbor retrieval." That is what this is — a backbone with
no head, one vector per image, and every capability that wants it (identity,
breed, BCS reference, fecal reference, footpad reference) built as retrieval on
top rather than as a trained classifier.
**Why ONNX and not `transformers`.** ADR 0017 took roughly a gigabyte of torch
out of the serving image and cut start-up from about 28 seconds to under one.
Reaching for `transformers` at serve time hands all of that back for a model
whose forward pass is a fixed graph with no control flow. So torch is a
*build-time* tool: `scripts/export_embedding.py` runs it once on a developer's
machine, and the service ships an `.onnx` that `onnxruntime` — already a
production dependency — loads in about a fifth of a second.
The export puts the pooling and the L2 normalisation inside the graph, so there
is no post-processing convention that can drift between whoever exported the
artefact and whoever serves it. A vector out of this adapter is always unit
length and cosine similarity is always a dot product.
**The artefact is still governed by a card.** This is not a second way to load a
model — `providers.load_card` checksums it and `adapters.licences.gate` checks
what the runtime really loads under, which is the check ADR 0017 added after a
watchdog defeated the card's own licence field.
"""
from __future__ import annotations
import threading
from functools import lru_cache
import numpy as np
from PIL import Image
from app.adapters.base import (
Adapter,
AdapterError,
AdapterSpec,
AdapterUnavailable,
Availability,
MeasuredCost,
Modality,
Placement,
Task,
)
from app.adapters.licences import LicenceRefused, gate
from app.providers import ModelArtefact
_session_lock = threading.Lock()
@lru_cache(maxsize=4)
def _session(artefact_path: str):
"""One session per artefact. Building one costs about as much as an
inference, so a request that rebuilds it doubles its own latency."""
import onnxruntime as ort
return ort.InferenceSession(
artefact_path, providers=["CPUExecutionProvider"]
)
def preprocess(
image: Image.Image,
size: int,
mean: tuple[float, float, float],
std: tuple[float, float, float],
) -> np.ndarray:
"""Resize the short side, centre crop, normalise. RGB, NCHW, float32.
This is torchvision's standard eval transform written out, for the same
reason `detectors/yolox_onnx.py` writes out its own NMS: pulling in a
training framework for one resize is what the ONNX path exists to avoid.
Getting it wrong does not raise — it quietly returns worse vectors — so the
numbers come from the card rather than from a constant here.
"""
rgb = image.convert("RGB")
width, height = rgb.size
scale = size / min(width, height)
resized = rgb.resize(
(max(size, round(width * scale)), max(size, round(height * scale))),
Image.BICUBIC,
)
new_width, new_height = resized.size
left = (new_width - size) // 2
top = (new_height - size) // 2
cropped = resized.crop((left, top, left + size, top + size))
array = np.asarray(cropped, dtype=np.float32) / 255.0
array = (array - np.asarray(mean, dtype=np.float32)) / np.asarray(
std, dtype=np.float32
)
return np.ascontiguousarray(array.transpose(2, 0, 1)[None])
class OnnxEmbeddingAdapter(Adapter):
"""A frozen backbone that turns an image into one unit vector.
Constructed from a `ModelArtefact` that `providers.load_card` has already
checksummed. It reads nothing else and downloads nothing.
"""
def __init__(
self,
artefact: ModelArtefact | None,
spec: AdapterSpec,
*,
input_size: int = 224,
mean: tuple[float, float, float] = (0.485, 0.456, 0.406),
std: tuple[float, float, float] = (0.229, 0.224, 0.225),
dimensions: int = 768,
) -> None:
self.artefact = artefact
self.spec = spec
self.input_size = input_size
self.mean = mean
self.std = std
self.dimensions = dimensions
self._session = None
self._input_name = ""
def availability(self) -> Availability:
if self.artefact is None:
return Availability(
False,
f"No artefact is installed for {self.spec.adapter_id}.",
"Run scripts/install_models.py, which fetches what the "
"committed card names and refuses anything whose checksum "
"does not match.",
)
try:
gate(self.spec.runtime, self.artefact.license)
except LicenceRefused as refusal:
return Availability(
False,
str(refusal),
"Move the capability to a permissively licensed backbone.",
)
if not self.artefact.is_validated:
return Availability(
False,
f"{self.artefact.model_id} has empty validation notes, so "
f"nothing attests that it works.",
"Fill in what was tested, on what data, with what result.",
)
return Availability(True)
def load(self) -> "OnnxEmbeddingAdapter":
availability = self.availability()
if not availability.ready:
raise AdapterUnavailable(availability)
assert self.artefact is not None # availability() proved it
with _session_lock:
session = _session(str(self.artefact.path))
inputs = session.get_inputs()
if len(inputs) != 1:
raise AdapterError(
f"{self.artefact.path.name} takes {len(inputs)} inputs; this "
f"adapter was written for a single image tensor."
)
# Shape-checked rather than trusted, because an export at a different
# resolution produces vectors that are the right length and the wrong
# thing, and nothing downstream would notice.
expected = [3, self.input_size, self.input_size]
actual = list(inputs[0].shape[1:])
if actual != expected:
raise AdapterError(
f"{self.artefact.path.name} takes {actual}, but the card "
f"describes a {expected} input. The artefact and its card "
f"disagree about what was exported."
)
self._session = session
self._input_name = inputs[0].name
return self
def embed(self, image: Image.Image) -> np.ndarray:
if self._session is None:
raise AdapterError(
"embed() called before load(). There is no path to a vector "
"that skipped the artefact check, and this is it refusing."
)
blob = preprocess(image, self.input_size, self.mean, self.std)
vector = self._session.run(None, {self._input_name: blob})[0][0]
if vector.shape[0] != self.dimensions:
raise AdapterError(
f"The graph returned {vector.shape[0]} dimensions; the card "
f"says {self.dimensions}. An index built at one and queried at "
f"the other fails silently, so this fails loudly."
)
return vector.astype(np.float32)
def embed_many(self, images: list[Image.Image]) -> np.ndarray:
"""One row per image. Kept separate because building a retrieval index
is the batch case and a request is the single case, and batching a
request would only add latency."""
return np.stack([self.embed(image) for image in images])
#: §3's first choice. Bespoke Meta licence with a live ambiguity about
#: attribution — see `licences.RUNTIME_LICENCES["dinov3-onnx"]`.
DINOV3_SPEC = AdapterSpec(
adapter_id="dinov3-vits16",
runtime="dinov3-onnx",
tasks=(Task.EMBED,),
modalities=(Modality.IMAGE,),
directive_role=(
"§3 DINOv3 — general visual embeddings, image similarity, retrieval, "
"cattle identity experiments, breed similarity, BCS and fecal and "
"footpad reference retrieval. Frozen, with nearest-neighbour on top."
),
placement=Placement.CPU_SERVICE,
placement_reason=(
"86.6 MB of ONNX, 59 ms a frame and a 470 MB peak single-threaded. It "
"belongs beside the API, and it is small enough that an on-device "
"build is worth investigating — ADR 0002 makes Animap offline-first, "
"and identity is exactly the capability a worker wants in a pen with "
"no signal."
),
measured=MeasuredCost(
hardware="Apple M-series laptop (NOT the target container)",
threads=1,
sample="61 Commons frames, evaluation/dataset.json",
runs=61,
median_seconds=0.059,
peak_rss_mb=470.0,
measured_on="2026-08-21",
),
notes=(
"**Measured at re-identification, which is what it is registered for.** "
"On 169 enrolled cattle from the CC BY 4.0 Zenodo 6324361 muzzle "
"database, five enrolment images each: closed-set top-1 0.977, top-3 "
"0.994, MRR 0.985, against a 0.0059 chance rate. Best of the two "
"servable backbones; the two unservable ones were only run on a "
"30-animal set, where all four saturate. "
"**And the open-set result is the one that shapes the product**: with no "
"threshold it names an unenrolled animal 100% of the time, because every "
"query has a nearest neighbour. The similarity cutoff that admits no "
"impostor accepts only 24.1% of the correct matches — the two "
"distributions overlap badly, enrolled probes median 0.971 against "
"unenrolled median 0.904 with an unenrolled maximum of 0.978. A margin "
"rule does not rescue it. That is why the confirm step in §6.4 is "
"load-bearing rather than decorative. "
"No Nigerian and no zebu animal has been through this; the database is "
"US beef breeds. "
"Exported at 224 px so the comparison against DINOv2-small is "
"like-for-like — timm resolves this checkpoint's native config to 256 px, "
"so these figures understate it slightly. On the older Commons proxy it "
"measured 1.000 species 1-NN and 1.000 Nigerian-cattle 1-NN against a "
"0.357 base rate, better than every other backbone and faster than all "
"but DINOv2."
),
)
#: The same interface over Apache-2.0 weights of the same size and embedding
#: width. Not a downgrade chosen for convenience — §4 asks for the benchmark,
#: and `experiments/cattle_identity/` is where the two are compared.
DINOV2_SPEC = AdapterSpec(
adapter_id="dinov2-small",
runtime="dinov2-onnx",
tasks=(Task.EMBED,),
modalities=(Modality.IMAGE,),
directive_role=(
"§3 DINOv3's role, served from the Apache-2.0 generation. 22.06M "
"parameters against DINOv3 ViT-S/16's 21.60M, and the same 384-wide "
"patch embedding."
),
placement=Placement.CPU_SERVICE,
placement_reason=(
"88.4 MB of ONNX, 81 ms a frame, 390 MB peak. The registered artefact, "
"because it is the one with no licence question attached."
),
measured=MeasuredCost(
hardware="Apple M-series laptop (NOT the target container)",
threads=1,
sample="61 Commons frames, evaluation/dataset.json",
runs=61,
median_seconds=0.081,
peak_rss_mb=390.0,
measured_on="2026-08-21",
),
notes=(
"Re-identification on the same 169 enrolled cattle as DINOv3: closed-set "
"top-1 0.957 against DINOv3's 0.977, top-3 0.986 against 0.994, MRR "
"0.971 against 0.985. Its "
"open-set behaviour is worse in the same shape — 21.3% true accepts at "
"the 1% false-accept point against DINOv3's 24.1%. "
"On the Commons proxy, 0.984 species 1-NN and 0.727 Nigerian-cattle 1-NN "
"against a 0.357 base rate. **Measurably worse than DINOv3 on every "
"figure taken on both arms**, which is what makes DINOv3's licence "
"ambiguity worth somebody's time rather than an academic point: the "
"permissive fallback costs about two points of top-1 on a task where "
"the errors are somebody's cow."
),
)
#: §40.2's head-to-head, and the reason it can exist at all.
#:
#: **This runs and it is not servable, and both halves are deliberate.** The
#: weights are CC-BY-NC-4.0, which a commercial product cannot satisfy at any
#: size; what the founder lifted was the rule that a licence like that stops the
#: model being *measured*. So the artefact is installed, the adapter is built,
#: and `licences.gate` refuses it under the default `enforce` policy and records
#: it under `record`. `describe()` reports `servable: False` either way.
MEGADESCRIPTOR_SPEC = AdapterSpec(
adapter_id="megadescriptor",
runtime="megadescriptor-timm",
tasks=(Task.EMBED,),
modalities=(Modality.IMAGE,),
directive_role=(
"§4 and §40.2 MegaDescriptor — wildlife re-ID embeddings for cattle "
"identity, benchmarked head-to-head against DINOv3."
),
placement=Placement.CPU_SERVICE,
placement_reason=(
"837 MB of ONNX and a measured 1,296 MB peak, which is inside the "
"2,000 MB the CPU worker is judged against but 3.7x DINOv3's peak on "
"the same run. The 0.723 s median is also inside the inline ceiling. "
"It fits; it is simply not worth the room, because it lost the "
"benchmark it was installed to win."
),
measured=MeasuredCost(
hardware="Apple M-series laptop (NOT the target container)",
threads=1,
sample="61 Commons frames, evaluation/dataset.json",
runs=61,
median_seconds=0.723,
peak_rss_mb=1295.8,
measured_on="2026-08-21",
),
notes=(
"**§40.2 answered: DINOv3 wins, and not narrowly.** MegaDescriptor-L-384 "
"measured 0.934 species 1-NN and 0.636 Nigerian-cattle 1-NN against a "
"0.357 base rate; DINOv3 measured 1.000 and 1.000 on the same 61 frames "
"in the same run. It is beaten by Apache-2.0 DINOv2-small on both "
"accuracy figures as well, at roughly 10x the artefact size and 7x the "
"latency. Its nearest-neighbour cosines are much flatter — 0.257 median "
"against DINOv3's 0.672. "
"**None of that measures re-identification**, which is what "
"MegaDescriptor is for: no available image set has the same animal "
"twice, so this says the space is worse *organised* for cattle and "
"geography, not that it cannot tell two White Fulani apart. "
"L-384 was chosen over the smaller variants because §40.2 names it and "
"because 837 MB of ONNX exports cleanly under the 2 GB protobuf limit; "
"T-224, S-224, B-224 and L-224 publish checkpoints of 204, 290, 473 and "
"1,922 MB and none was exported."
),
)
#: The other non-commercial contender, installed for the same reason and to no
#: better end. Its licence problem is quieter than MegaDescriptor's: nothing was
#: granted at all, and silence defaults to all rights reserved.
MIEWID_SPEC = AdapterSpec(
adapter_id="miewid-msv3",
runtime="miewid",
tasks=(Task.EMBED,),
modalities=(Modality.IMAGE,),
directive_role=(
"§4's 'Wildlife ReID embeddings' — the alternative to MegaDescriptor, "
"benchmarked alongside it under §40.2."
),
placement=Placement.CPU_SERVICE,
placement_reason=(
"206 MB of ONNX, 0.213 s a frame, 502 MB peak. Comfortably the cheapest "
"of the two wildlife re-ID models and still 2.9x DINOv3's latency for "
"the worst Nigerian retrieval of the four."
),
measured=MeasuredCost(
hardware="Apple M-series laptop (NOT the target container)",
threads=1,
sample="61 Commons frames, evaluation/dataset.json",
runs=61,
median_seconds=0.213,
peak_rss_mb=502.2,
measured_on="2026-08-21",
),
notes=(
"Measured 0.951 species 1-NN and 0.455 Nigerian-cattle 1-NN against a "
"0.357 base rate — a lift of 1.27x on 11 frames, which is a failure to "
"show anything rather than a measured floor. Last of the four on the "
"figure that matters most for Nigerian farms. "
"**Building this artefact meant running a third party's Python.** The "
"upstream repo ships `modeling_miewid.py` instead of a `transformers` "
"architecture, so `scripts/export_embedding.py` loads it with "
"`trust_remote_code=True`. That is a build-step supply-chain exposure, "
"not a serving one — the service loads a fixed ONNX graph with no "
"Python in it — and the three modules were read before they were run. "
"Preprocessing deviates from the published transform: the model card "
"specifies `Resize((440, 440))` and this pipeline centre-crops, so its "
"figures here may understate it."
),
)