Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
File size: 9,312 Bytes
b05b52c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 | """Evidence-backed coverage contract for flight recommendations.
The comparison engine can calculate an answer exactly *for the price and offers
it is given*. That is different from proving that every seller and every bank
offer has been checked. This module makes that boundary executable: a flight
may only be labelled a verified best offer when the current seller table is
fully mapped and the approved coverage manifest says every in-scope OTA and
issuer has been read successfully within the configured freshness window.
An absent, stale, partial, or manually edited manifest is deliberately a
coverage failure, never an implicit assertion of completeness. The manifest
is produced by dataops/approve.py after a clean crawl and human approval.
"""
from __future__ import annotations
import json
import os
from datetime import date, datetime, timedelta, timezone
from typing import Dict, Iterable, List, Optional
MANIFEST_SCHEMA_VERSION = 1
MANIFEST_PATH = os.getenv(
"FLIGHT_COVERAGE_MANIFEST",
os.path.join(os.path.dirname(__file__), "offers_data", "coverage_manifest.json"),
)
# Flight prices may change minute-to-minute, but a bank/OTA offer verification
# has to be from the same operational day before we make a completeness claim.
MAX_COVERAGE_AGE_HOURS = int(os.getenv("FLIGHT_COVERAGE_MAX_AGE_HOURS", "24"))
_IST = timezone(timedelta(hours=5, minutes=30))
def today_ist() -> str:
return datetime.now(_IST).date().isoformat()
def _parse_day(raw: object) -> Optional[date]:
if not isinstance(raw, str) or not raw:
return None
try:
# A manifest may contain a date or a full ISO timestamp. The date is
# sufficient because dataops runs as a daily operational check.
return date.fromisoformat(raw[:10])
except ValueError:
return None
def _fresh(entry: object, as_of: str, max_age_hours: int) -> bool:
if not isinstance(entry, dict) or entry.get("status") != "verified":
return False
checked = _parse_day(entry.get("checked_at"))
now = _parse_day(as_of)
if checked is None or now is None:
return False
# Date-granularity dataops records cannot use a fractional hour precisely;
# accept only checks made today when configured below a day.
allowed_days = 0 if max_age_hours < 24 else max_age_hours // 24
return checked <= now <= checked + timedelta(days=allowed_days)
def load_manifest(path: Optional[str] = None) -> Dict:
"""Return an untrusted manifest as data. Malformed/missing = no coverage."""
try:
with open(path or MANIFEST_PATH, encoding="utf-8") as f:
value = json.load(f)
return value if isinstance(value, dict) else {}
except (OSError, ValueError, TypeError):
return {}
def flight_channel_keys() -> List[str]:
from channels import FLIGHT_CHANNELS
return [str(ch["key"]) for ch in FLIGHT_CHANNELS]
def supported_issuers() -> List[str]:
"""Issuer labels actually represented by individual card products.
Composite/placeholder labels cannot be verified as a bank and are excluded
from the contractual denominator; each actual card issuer remains in it.
"""
from card_catalogue import all_cards
excluded = {"", "Other bank", "BoB/Federal/SBM"}
return sorted({str(card.issuer) for card in all_cards() if str(card.issuer) not in excluded})
def requirements() -> Dict[str, List[str]]:
return {"channels": flight_channel_keys(), "issuers": supported_issuers()}
def _required_failures(
manifest: Dict, group: str, required: Iterable[str], as_of: str, max_age_hours: int,
) -> List[str]:
records = manifest.get(group) if isinstance(manifest.get(group), dict) else {}
failures: List[str] = []
for key in required:
if not _fresh(records.get(key), as_of, max_age_hours):
failures.append(key)
return failures
def assess_flight_coverage(
sellers: Iterable[Dict],
*,
live: bool,
manifest: Optional[Dict] = None,
as_of: Optional[str] = None,
max_age_hours: int = MAX_COVERAGE_AGE_HOURS,
) -> Dict:
"""Return the machine-readable truthfulness guard for one seller quote.
This assesses the *declared India flight coverage scope* (all supported
issuers and all supported flight OTAs), plus whether the particular live
seller table has an unmapped seller. It never treats a missing seller as a
price match and never lets the recommendation claim completeness.
"""
checked_at = as_of or today_ist()
m = manifest if isinstance(manifest, dict) else load_manifest()
schema_ok = m.get("schema_version") == MANIFEST_SCHEMA_VERSION
req = requirements()
rows = [s for s in sellers if isinstance(s, dict)]
unmapped = sorted({str(s.get("seller") or "Unknown seller") for s in rows
if not s.get("channel_key")})
mapped_keys = sorted({str(s.get("channel_key")) for s in rows if s.get("channel_key")})
unknown_channel_keys = sorted(set(mapped_keys) - set(req["channels"]))
missing_channels = _required_failures(m, "channels", req["channels"], checked_at, max_age_hours)
missing_issuers = _required_failures(m, "issuers", req["issuers"], checked_at, max_age_hours)
reasons: List[str] = []
if not live:
reasons.append("live seller quote unavailable")
if not schema_ok:
reasons.append("approved coverage manifest unavailable or incompatible")
if not rows:
reasons.append("seller table empty")
if unmapped:
reasons.append("one or more live sellers are not mapped to a supported OTA")
if unknown_channel_keys:
reasons.append("seller table contains unknown channel keys")
if missing_channels:
reasons.append("one or more supported OTAs lack a fresh verified crawl")
if missing_issuers:
reasons.append("one or more represented card issuers lack a fresh verified crawl")
complete = not reasons
return {
"scope": "India flight offers: every supported OTA and represented card issuer",
"as_of": checked_at,
"max_age_hours": max_age_hours,
"manifest_schema_ok": schema_ok,
"live_seller_count": len(rows),
"mapped_channel_keys": mapped_keys,
"unmapped_sellers": unmapped,
"unknown_channel_keys": unknown_channel_keys,
"unverified_channels": missing_channels,
"unverified_issuers": missing_issuers,
"complete": complete,
# The UI and API must key any “best/optimal” language off this one
# boolean, rather than interpreting an available comparison as proof.
"recommendation_allowed": complete,
"reasons": reasons,
}
def manifest_from_dataops_report(report: Dict, *, approved_at: Optional[str] = None) -> Dict:
"""Build the manifest dataops may publish after human offer approval.
A clean run proves only sources it actually read. Sources outside the
watchlist remain absent and therefore fail closed in ``assess``. A verified
zero-offer page is legitimate coverage; unread and suspicious pages are
not. This function contains no network access and is deterministic so it
can be unit-tested from a saved report.
"""
stamp = approved_at or str(report.get("generated") or today_ist())[:10]
result: Dict = {
"schema_version": MANIFEST_SCHEMA_VERSION,
"generated_at": stamp,
"scope": "India flight offers: every supported OTA and represented card issuer",
"channels": {},
"issuers": {},
"source_report": {
"generated": report.get("generated"),
"read_ok": bool(report.get("read_ok")),
"pages_unread": int(report.get("pages_unread") or 0),
"suspicious_zeroes": len(report.get("suspicious_zeroes") or []),
},
}
# A globally unclean report must not mint any fresh verification stamps.
clean_run = bool(report.get("read_ok"))
by_channel: Dict[str, List[Dict]] = {}
by_issuer: Dict[str, List[Dict]] = {}
for source in report.get("by_source") or []:
if not isinstance(source, dict):
continue
merchant = str(source.get("merchant_key") or "")
issuer = str(source.get("issuer") or "")
if merchant:
by_channel.setdefault(merchant, []).append(source)
if issuer:
by_issuer.setdefault(issuer, []).append(source)
def record(rows: List[Dict]) -> Dict:
pages_read = sum(int(r.get("pages_read") or 0) for r in rows)
unread = sum(int(r.get("unread_pages") or 0) for r in rows)
suspicious = sum(int(r.get("suspicious_pages") or 0) for r in rows)
ok = clean_run and bool(rows) and pages_read > 0 and unread == 0 and suspicious == 0
return {
"status": "verified" if ok else "unverified",
"checked_at": stamp if ok else None,
"sources": [str(r.get("url") or "") for r in rows],
"pages_read": pages_read,
"unread_pages": unread,
"suspicious_pages": suspicious,
}
for key in flight_channel_keys():
result["channels"][key] = record(by_channel.get(key, []))
for issuer in supported_issuers():
result["issuers"][issuer] = record(by_issuer.get(issuer, []))
return result
|