| """Public photo search (spec Β§9, Β§11) β no auth required. |
| |
| A visitor uploads 1-6 photos of their dog and (optionally) a ZIP code and gets back the found/unknown |
| dogs already in the system ranked by photo similarity. This is a read-only convenience query: it |
| embeds each photo transiently (nothing is stored), scores the whole set against the active |
| found/unknown pool (max similarity over every query x candidate image pair β see |
| ``_dog_level_score``), and returns the closest profiles with their photos. It does NOT create a |
| case β that is still the explicit "I lost a dog" / "I found a dog" flow. |
| """ |
| from __future__ import annotations |
|
|
| from fastapi import APIRouter, Depends, File, Form, HTTPException, Request, UploadFile, status |
| from sqlalchemy.orm import Session |
|
|
| from ..db import get_db |
| from ..models.base import SubjectType |
| from ..services import datasets as ds |
| from ..services.geo import get_geo |
| from ..services.images import ImageValidationError, embed_bytes, predict_breeds_bytes |
| from ..services.matching import search_knowns_by_vectors, search_unknowns_by_vectors |
| from .helpers import pictures_for, rate_limit |
|
|
| router = APIRouter(prefix="/search", tags=["search"]) |
|
|
| |
| |
| DEFAULT_SEARCH_RADIUS = 100 |
| |
| MAX_QUERY_IMAGES = 6 |
|
|
|
|
| @router.post("/by-photo") |
| def search_by_photo( |
| request: Request, |
| files: list[UploadFile] = File(...), |
| zip: str | None = Form(None), |
| top_k: int = Form(12), |
| pool: str = Form("found"), |
| db: Session = Depends(get_db), |
| ) -> dict: |
| """Rank a pool against 1-6 uploaded photos of the same dog. Optional ZIP scopes by proximity. |
| |
| ``pool`` selects which side to search: |
| - ``found`` (default): the found/unknown pool β for someone who LOST a dog. |
| - ``lost``: the known/lost pool β for someone who FOUND a dog and wants its owner. |
| """ |
| rate_limit(request, key_prefix="search") |
| pool = pool if pool in ("found", "lost") else "found" |
|
|
| query_vecs = [] |
| for f in files[:MAX_QUERY_IMAGES]: |
| data = f.file.read() |
| try: |
| query_vecs.append(embed_bytes(data)) |
| except ImageValidationError as exc: |
| raise HTTPException(status.HTTP_400_BAD_REQUEST, str(exc)) from exc |
| if not query_vecs: |
| raise HTTPException(status.HTTP_400_BAD_REQUEST, "No image provided") |
|
|
| event_zip = (zip or "").strip() or None |
| radius = DEFAULT_SEARCH_RADIUS if event_zip else -1 |
| top_k = max(1, min(top_k, 50)) |
|
|
| subject_type = SubjectType.unknown if pool == "found" else SubjectType.known |
| search_fn = search_unknowns_by_vectors if pool == "found" else search_knowns_by_vectors |
| ranked, considered = search_fn( |
| db, query_vecs, event_zip=event_zip, radius_miles=radius, top_k=top_k |
| ) |
|
|
| geo = get_geo() |
| results: list[dict] = [] |
| for dog_id, score in ranked: |
| profile = ds.dog_profile(db, subject_type, dog_id) |
| if profile is None: |
| continue |
| photos = [p.model_dump() for p in pictures_for(db, subject_type, dog_id)] |
| distance = ( |
| geo.distance_miles(event_zip, profile.get("zip")) if event_zip else None |
| ) |
| results.append( |
| { |
| "dog": profile, |
| "score": score, |
| "distance_miles": round(distance, 1) if distance is not None else None, |
| "photos": photos, |
| } |
| ) |
|
|
| from ..ml import get_embedder |
|
|
| e = get_embedder() |
| return { |
| "results": results, |
| "model": f"{e.name}/{e.version}", |
| "candidate_count": considered, |
| "zip": event_zip, |
| "radius_miles": radius, |
| "pool": pool, |
| } |
|
|
|
|
| @router.post("/breed") |
| def estimate_breed( |
| request: Request, |
| files: list[UploadFile] = File(...), |
| top_n: int = Form(5), |
| db: Session = Depends(get_db), |
| ) -> dict: |
| """Estimate a dog's breed(s) from 1-6 photos. Returns the top ``top_n`` (1β10) labels. |
| |
| A single photo is a weak signal: the same dog shot from a different angle can flip the top |
| breed entirely. So every supplied photo is classified and the per-label scores are AVERAGED |
| across them (labels outside a photo's top-K count as 0 for that photo, which penalises breeds |
| only one photo agrees on). Read-only and stores nothing; ``db`` is unused but kept in the |
| signature so the route stays consistent with the rest of the router. |
| """ |
| rate_limit(request, key_prefix="search") |
| top_n = max(1, min(top_n, 10)) |
|
|
| totals: dict[str, float] = {} |
| n_images = 0 |
| for f in files[:MAX_QUERY_IMAGES]: |
| try: |
| preds = predict_breeds_bytes(f.file.read()) |
| except ImageValidationError as exc: |
| raise HTTPException(status.HTTP_400_BAD_REQUEST, str(exc)) from exc |
| for label, score in preds: |
| totals[label] = totals.get(label, 0.0) + score |
| n_images += 1 |
| if n_images == 0: |
| raise HTTPException(status.HTTP_400_BAD_REQUEST, "No image provided") |
|
|
| ranked = sorted(((lbl, s / n_images) for lbl, s in totals.items()), key=lambda x: -x[1]) |
|
|
| from ..ml import get_breed_classifier |
|
|
| c = get_breed_classifier() |
| breeds = [{"label": label, "score": score} for label, score in ranked[:top_n]] |
| return {"breeds": breeds, "model": f"{c.name}/{c.version}", "images": n_images} |
|
|