pino-source-code / src /pino /semantics.py
mattbitzesty's picture
v6: ConcentrationAwarePyramidHead + OAV fix + curated descriptors + integration
ad424e4 unverified
Raw
History Blame Contribute Delete
13.9 kB
from __future__ import annotations
import json
import logging
from pathlib import Path
from typing import Any
import numpy as np
logger = logging.getLogger("pino.semantics")
DESCRIPTOR_DIM = 138
# ---------------------------------------------------------------------------
# Vocabulary / annotation loading
# ---------------------------------------------------------------------------
def _project_data_dir() -> Path:
"""Return the project-level data/ directory."""
# src/pino/semantics.py -> src/pino/ -> src/ -> project root
return Path(__file__).resolve().parents[2] / "data"
def load_vocabulary(path: str | Path | None = None) -> list[str]:
"""Load the 138-D pyrfume descriptor vocabulary from disk."""
path = Path(path) if path else _project_data_dir() / "pyrfume_vocabulary.json"
if not path.exists():
raise FileNotFoundError(f"Vocabulary file not found: {path}")
data = json.loads(path.read_text(encoding="utf-8"))
return data["vocabulary"]
def load_descriptor_annotations(path: str | Path | None = None) -> dict[str, list[float]]:
"""Load the CAS/SMILES -> 138-D binary descriptor vector mapping."""
path = Path(path) if path else _project_data_dir() / "pyrfume_descriptors_by_cas.json"
if not path.exists():
raise FileNotFoundError(f"Descriptor annotations not found: {path}")
return json.loads(path.read_text(encoding="utf-8"))
def load_odt_annotations(path: str | Path | None = None) -> dict[str, float]:
"""Load the CAS/SMILES -> ODT (mg/m³) mapping."""
path = Path(path) if path else _project_data_dir() / "pyrfume_odt_by_cas.json"
if not path.exists():
raise FileNotFoundError(f"ODT annotations not found: {path}")
return {k: float(v) for k, v in json.loads(path.read_text(encoding="utf-8")).items()}
# ---------------------------------------------------------------------------
# Per-compound lookup helpers
# ---------------------------------------------------------------------------
def _registry_key(cas: str | None, smiles: str | None) -> str | None:
if cas and not cas.startswith("SMILES:"):
return cas
return smiles
def get_compound_descriptor_vector(
cas: str | None,
smiles: str | None,
descriptor_map: dict[str, list[float]] | None = None,
) -> np.ndarray | None:
"""Return the 138-D binary descriptor vector for a compound, if known."""
if descriptor_map is None:
descriptor_map = load_descriptor_annotations()
key = _registry_key(cas, smiles)
if key is None:
return None
vec = descriptor_map.get(key)
if vec is None:
return None
return np.array(vec, dtype=np.float32)
def get_compound_odt(
cas: str | None,
smiles: str | None,
odt_map: dict[str, float] | None = None,
) -> float | None:
"""Return the ODT (mg/m³) for a compound, if known."""
if odt_map is None:
odt_map = load_odt_annotations()
key = _registry_key(cas, smiles)
if key is None:
return None
return odt_map.get(key)
# ---------------------------------------------------------------------------
# OAV-weighted mixture descriptor targets
# ---------------------------------------------------------------------------
def compute_oav_targets(
composition: list[dict[str, Any]],
concentrations: np.ndarray,
*,
descriptor_map: dict[str, list[float]] | None = None,
odt_map: dict[str, float] | None = None,
eps: float = 1e-12,
) -> np.ndarray:
"""
Compute time-resolved OAV-weighted mixture descriptor vectors.
Args:
composition: list of compound dicts, each with at least 'cas' and/or 'smiles'.
concentrations: (timesteps, n_compounds) array of instantaneous concentrations
in mg/m³.
Returns:
targets: (timesteps, 138) array of mixture descriptor vectors.
"""
timesteps, n_compounds = concentrations.shape
if descriptor_map is None:
descriptor_map = load_descriptor_annotations()
if odt_map is None:
odt_map = load_odt_annotations()
# Build per-compound descriptor vectors and ODTs.
descriptors: list[np.ndarray] = []
odt_values: list[float] = []
for comp in composition:
cas = comp.get("cas")
smiles = comp.get("smiles")
vec = get_compound_descriptor_vector(cas, smiles, descriptor_map=descriptor_map)
odt = get_compound_odt(cas, smiles, odt_map=odt_map)
if vec is None or odt is None:
# Compound lacks annotation; leave it out of the target.
descriptors.append(np.zeros(DESCRIPTOR_DIM, dtype=np.float32))
odt_values.append(1.0) # dummy, will multiply by zero vector
continue
descriptors.append(vec)
odt_values.append(odt)
descriptors = np.vstack(descriptors) # (n_compounds, 138)
odt_values = np.array(odt_values, dtype=np.float32) # (n_compounds,)
# log10(OAV + 1) is used as a positive monotonic weight that equals log-OAV
# for strongly supra-threshold compounds (OAV >> 1) while keeping sub-threshold
# ingredients from collapsing the target to a zero vector.
oav = concentrations / (odt_values[None, :] + eps) # (timesteps, n_compounds)
weights = np.log10(oav + 1.0) # (timesteps, n_compounds)
weighted = weights[:, :, None] * descriptors[None, :, :] # (T, C, 138)
targets = weighted.sum(axis=1) # (T, 138)
# Normalise each timestep vector to unit length (cosine-style target).
norms = np.linalg.norm(targets, axis=1, keepdims=True)
targets = np.divide(targets, norms, out=np.zeros_like(targets), where=norms > eps)
return targets.astype(np.float32)
# ---------------------------------------------------------------------------
# Legacy / compatibility API
# ---------------------------------------------------------------------------
def get_objective_targets(
generation_strategy: str,
timesteps: int,
*,
composition: list[dict[str, Any]] | None = None,
concentrations: np.ndarray | None = None,
seed: int = 2026,
) -> np.ndarray:
"""
Return a (timesteps, 138) objective odor-descriptor target matrix.
If composition and concentrations are provided, the target is a real
OAV-weighted mixture descriptor vector. Otherwise, a deterministic
fallback profile is returned for compatibility.
"""
if composition is not None and concentrations is not None:
return compute_oav_targets(composition, concentrations)
# Fallback: simple genre-agnostic descriptor profile for tests/back-compat.
rng = np.random.default_rng(seed)
tau = np.linspace(0.0, 1.0, timesteps, dtype=np.float32)
targets = np.zeros((timesteps, DESCRIPTOR_DIM), dtype=np.float32)
# Use a few broad descriptors to keep the tensor non-degenerate.
targets[:, 0] = np.exp(-6.0 * tau) # citrus / top decay
targets[:, 1] = 4.0 * tau * np.exp(-4.0 * tau) # floral / heart surge
targets[:, 2] = 0.1 + 0.9 * (1.0 - np.exp(-3.0 * tau)) # woody / base rise
targets[:, 3] = 0.1 + 0.9 * (1.0 - np.exp(-3.0 * tau)) # musk / base rise
targets[:, 4:] = rng.uniform(0.0, 1e-4, size=(timesteps, DESCRIPTOR_DIM - 4)).astype(np.float32)
return np.clip(targets, 0.0, 1.0).astype(np.float32)
# ---------------------------------------------------------------------------
# Weak-label semantic targets (kept for compatibility; derived from strategy)
# ---------------------------------------------------------------------------
GENRE_SEMANTICS: dict[str, dict[str, np.ndarray]] = {
"citrus_cologne": {
"seasonality": np.array([0.30, 0.50, 0.10, 0.10], dtype=np.float32),
"gender_profile": np.array([0.0], dtype=np.float32),
"wearability": np.array([0.85, 0.30, 0.70, 0.40], dtype=np.float32),
},
"fougere": {
"seasonality": np.array([0.35, 0.25, 0.30, 0.10], dtype=np.float32),
"gender_profile": np.array([-0.6], dtype=np.float32),
"wearability": np.array([0.70, 0.60, 0.75, 0.65], dtype=np.float32),
},
"floral_woody": {
"seasonality": np.array([0.40, 0.20, 0.30, 0.10], dtype=np.float32),
"gender_profile": np.array([0.3], dtype=np.float32),
"wearability": np.array([0.60, 0.70, 0.80, 0.70], dtype=np.float32),
},
"amber_oriental": {
"seasonality": np.array([0.10, 0.05, 0.30, 0.55], dtype=np.float32),
"gender_profile": np.array([-0.2], dtype=np.float32),
"wearability": np.array([0.20, 0.95, 0.50, 0.90], dtype=np.float32),
},
"wildcard": {
"seasonality": np.array([0.25, 0.25, 0.25, 0.25], dtype=np.float32),
"gender_profile": np.array([0.0], dtype=np.float32),
"wearability": np.array([0.50, 0.50, 0.50, 0.50], dtype=np.float32),
},
}
def get_semantic_targets(generation_strategy: str) -> dict[str, np.ndarray]:
"""Return the weak-label semantic targets for a given generation strategy."""
return GENRE_SEMANTICS.get(generation_strategy, GENRE_SEMANTICS["wildcard"])
def get_psychometric_targets(generation_strategy: str) -> np.ndarray:
"""Return a 7-D psychometric target vector."""
base = get_semantic_targets(generation_strategy)
season = base["seasonality"]
gender = base["gender_profile"].reshape(-1)
wear = base["wearability"]
wear_compressed = np.array([
(wear[0] + wear[2]) / 2.0,
(wear[1] + wear[3]) / 2.0,
], dtype=np.float32)
return np.concatenate([season, gender, wear_compressed]).astype(np.float32)
def compute_formula_psychometric_targets(
formula: list[dict[str, Any]],
pyramid_targets: np.ndarray,
vocab: list[str] | None = None,
) -> np.ndarray:
"""Derive psychometric targets from actual formula composition.
Instead of using hardcoded genre lookups, compute seasonality, gender,
and wearability from the pyramid descriptor profile.
- Gender: feminine descriptors (floral, rose, sweet, powdery) push negative;
masculine descriptors (woody, leather, tobacco, smoky) push positive.
- Wearability: base-note weight fraction determines day vs night orientation.
- Seasonality: citrus/fresh/green content pushes summer; warm/spicy pushes winter.
Args:
formula: list of {cas, weight_fraction, ...} dicts.
pyramid_targets: (3, 138) array of binary pyramid labels.
vocab: optional vocabulary list for descriptor name lookup.
Returns:
7-D float32 vector: [spring, summer, autumn, winter, gender, day, night]
"""
pyramid = np.asarray(pyramid_targets, dtype=np.float32)
if pyramid.shape != (3, 138):
return get_psychometric_targets("wildcard")
if vocab is None:
try:
vocab = load_vocabulary()
except Exception:
vocab = [str(i) for i in range(138)]
# Build descriptor name → index map
desc_idx = {name: i for i, name in enumerate(vocab) if i < 138}
def _desc_score(indices_of_names: list[str]) -> float:
total = 0.0
for name in indices_of_names:
idx = desc_idx.get(name)
if idx is not None:
total += float(pyramid[2, idx]) * 0.5 + float(pyramid[1, idx]) * 0.3 + float(pyramid[0, idx]) * 0.2
return total
# --- Gender: feminine (negative) vs masculine (positive) ---
feminine = _desc_score(["floral", "rose", "sweet", "powdery", "creamy", "vanilla",
"jasmine", "lily", "waxy", "fruity"])
masculine = _desc_score(["leather", "tobacco", "smoky", "animal", "phenolic",
"earthy", "mossy", "spicy"])
total_g = feminine + masculine
gender = np.clip((masculine - feminine) / max(total_g, 1.0) * 1.5, -0.8, 0.5)
# --- Seasonality ---
summer_descs = _desc_score(["citrus", "fresh", "green", "cooling", "fruity",
"mint", "cucumber", "watery", "ozone"])
winter_descs = _desc_score(["warm", "spicy", "cinnamon", "clove", "incense",
"vanilla", "amber", "balsamic"])
spring_descs = _desc_score(["floral", "rose", "fresh", "green", "lily"])
autumn_descs = _desc_score(["woody", "earthy", "mossy", "dry", "leather"])
season_raw = np.array([spring_descs, summer_descs, autumn_descs, winter_descs],
dtype=np.float32)
season_raw = np.maximum(season_raw, 0.05) # floor
seasonality = season_raw / season_raw.sum()
# --- Wearability ---
top_weight = float(pyramid[0].sum()) / 138.0
base_weight = float(pyramid[2].sum()) / 138.0
# More base notes → night-oriented; more top notes → day-oriented
night_oriented = base_weight / max(top_weight + base_weight, 0.01)
day_wear = np.clip(1.0 - night_oriented, 0.3, 0.9)
night_wear = np.clip(0.3 + night_oriented * 0.6, 0.3, 0.95)
return np.array([
seasonality[0], seasonality[1], seasonality[2], seasonality[3],
gender, day_wear, night_wear
], dtype=np.float32)
def sample_noisy_psychometric_targets(
generation_strategy: str,
*,
seasonality_noise: float = 0.05,
gender_noise: float = 0.1,
wearability_noise: float = 0.05,
rng: np.random.Generator | None = None,
) -> np.ndarray:
"""Add label noise to the 7-D psychometric target vector."""
rng = rng or np.random.default_rng()
base = get_psychometric_targets(generation_strategy)
noisy = base.copy()
noisy[:4] += rng.normal(0, seasonality_noise, size=4).astype(np.float32)
noisy[:4] = np.clip(noisy[:4], 0.0, 1.0)
noisy[:4] = noisy[:4] / noisy[:4].sum() if noisy[:4].sum() > 0 else noisy[:4]
noisy[4] += rng.normal(0, gender_noise, size=1).astype(np.float32)[0]
noisy[4] = np.clip(noisy[4], -1.0, 1.0)
noisy[5:] += rng.normal(0, wearability_noise, size=2).astype(np.float32)
noisy[5:] = np.clip(noisy[5:], 0.0, 1.0)
return noisy