Spaces:
Running
Running
Mathias Heider
Claude Opus 5.5
Cost notes: gemini-3.5-flash output is 6x input; run #543 re-priced (~$7.50, not $2.02)
8dfd1d1 Download extraction.py from aim4composites/AutonomousAgent: direct link, hf CLI and curl.
- Browser
- Download file 68.7 kB
-
https://huggingface.co/spaces/aim4composites/AutonomousAgent/resolve/main/extraction.py
- Command line
-
hf download hf://spaces/aim4composites/AutonomousAgent/extraction.py
-
curl -L -o extraction.py https://huggingface.co/spaces/aim4composites/AutonomousAgent/resolve/main/extraction.py
68.7 kB
| r""" | |
| extraction.py — single source of truth for the AIM Composites extraction pipeline. | |
| This module consolidates the Gemini prompt/schema and all post-extraction | |
| processing that used to live (and drift) inside ``batch_ingest.py`` and the | |
| Streamlit app repo's ``page_files/categorized/page6.py`` / | |
| ``Backend/Pdf_DataExtraction.py``. | |
| Pipeline shape:: | |
| pdf_bytes | |
| -> extract_from_pdf() # Gemini structured output -> Extraction | |
| -> verify_against_text() # ground every value in the PDF text (Task 1) | |
| -> to_rows() # Extraction -> list[PropertyRow] (Task 2/3/5/6) | |
| -> (caller inserts rows, carrying `status`/`flag_reason`) | |
| Design notes | |
| ------------ | |
| * The model is asked for a *list* of materials, each with structured, | |
| unit-aware property values plus a verbatim ``source_quote`` and ``page``. | |
| * Numbers are parsed into ``value_num`` / ``value_min`` / ``value_max`` / | |
| ``qualifier`` and normalized to a canonical unit (``unit_canonical`` + | |
| ``value_si``) with :mod:`pint`. Plausibility is range-checked *after* | |
| conversion, so a GPa-vs-MPa mix no longer false-flags. | |
| * Nothing is silently dropped: every row carries a ``status`` and, when not | |
| ``ok``, a ``flag_reason``. | |
| Keep ``temperature=0`` and bump :data:`PROMPT_VERSION` on any prompt change. | |
| """ | |
| from __future__ import annotations | |
| import base64 | |
| import dataclasses | |
| import json | |
| import logging | |
| import os | |
| import re | |
| import time | |
| import unicodedata | |
| from typing import Any, Optional | |
| import requests | |
| try: # PyMuPDF — used for text grounding (Task 1) and scanned detection (Task 10) | |
| import fitz # type: ignore | |
| except Exception: # pragma: no cover - import guard | |
| fitz = None # type: ignore | |
| try: | |
| import pint # unit normalization (Task 3) | |
| except Exception: # pragma: no cover - import guard | |
| pint = None # type: ignore | |
| log = logging.getLogger("extraction") | |
| # --------------------------------------------------------------------------- | |
| # Configuration | |
| # --------------------------------------------------------------------------- | |
| # Stable release only — the previous pin ("gemini-2.5-flash-preview-09-2025", | |
| # a *preview*) was shut down by Google on 2026-02-17 and silently killed every | |
| # extraction call. Previews get retired with ~2 weeks notice; stable models | |
| # don't. Overridable via env so a future swap needs no code change. | |
| GEMINI_MODEL = os.environ.get("GEMINI_MODEL", "gemini-3.5-flash") | |
| # Model for the figure (vision) calls: classify + mining. Empty = same as | |
| # GEMINI_MODEL. Classification is an easy task (one call per PDF picking a | |
| # figure kind), so a cheaper model here costs little quality. The agent sets | |
| # both from Run Control each cycle; read them at call time via text_model() / | |
| # vision_model(), never through a name imported at module load. | |
| GEMINI_VISION_MODEL = os.environ.get("GEMINI_VISION_MODEL", "").strip() | |
| PROMPT_VERSION = "2.0" | |
| def text_model() -> str: | |
| """The model the text extraction call uses right now.""" | |
| return (GEMINI_MODEL or "gemini-3.5-flash").strip() | |
| def vision_model() -> str: | |
| """The model the figure classify / mining calls use right now.""" | |
| return (GEMINI_VISION_MODEL or "").strip() or text_model() | |
| # How much the model may "think" per call. Thinking tokens are billed as | |
| # output (6x the input price on gemini-3.5-flash): on 28 Sep 2026 a 25-paper | |
| # cycle cost ~$7.50, $6.69 of it output tokens, ~30k per PDF where the extraction JSON itself | |
| # is ~4k. Values: "dynamic" (the API default: unbounded), "low" / "minimal" / | |
| # "high" (Gemini 3 thinking levels), "off" (budget 0), or an integer token | |
| # budget (Gemini 2.x style). The agent sets this from its config each cycle | |
| # (Run Control); GEMINI_THINKING is the first-boot default and what | |
| # batch_ingest.py uses. If the model rejects the option (400), the call is | |
| # repeated without it and the option is dropped for the rest of the process. | |
| THINKING = os.environ.get("GEMINI_THINKING", "low").strip().lower() or "dynamic" | |
| _THINKING_UNSUPPORTED = False | |
| GEMINI_URL_TEMPLATE = ( | |
| "https://generativelanguage.googleapis.com/v1beta/models/" | |
| "{model}:generateContent?key={key}" | |
| ) | |
| GEMINI_UPLOAD_URL = ( | |
| "https://generativelanguage.googleapis.com/upload/v1beta/files?key={key}" | |
| ) | |
| GEMINI_FILE_STATUS_URL = ( | |
| "https://generativelanguage.googleapis.com/v1beta/{name}?key={key}" | |
| ) | |
| REQUEST_TIMEOUT_S = 300 | |
| # Transient-failure retry policy, mirrored from pdf_crawler.http_get(). The | |
| # Gemini free tier returns 429s under load; a request that fails once is often | |
| # fine moments later. Retry connection errors and 429/5xx, honoring Retry-After. | |
| MAX_RETRIES = 3 | |
| RETRY_STATUS = frozenset({429, 500, 502, 503, 504}) | |
| BACKOFF_BASE = 2.0 # seconds; exponential per attempt | |
| # Inline a PDF up to this size; above it, upload via the Gemini File API so the | |
| # base64-inflated request stays under the generateContent size limit (Task 10). | |
| INLINE_PDF_LIMIT_BYTES = 15 * 1024 * 1024 | |
| INLINE_PAGE_LIMIT = 80 | |
| # Average extractable characters/page below this => treat as scanned/image-only. | |
| SCANNED_TEXT_PER_PAGE = 50 | |
| # --- enums (Task 8) -------------------------------------------------------- | |
| # Aligned with the app repo's page6.py PROPERTY_CATEGORIES. | |
| SECTION_ENUM = [ | |
| "Mechanical", | |
| "Thermal", | |
| "Electrical", | |
| "Physical", | |
| "Optical", | |
| "Rheological", | |
| "Processing", | |
| "Descriptive", | |
| "Composition/Reinforcement", | |
| "Architecture/Structure", | |
| ] | |
| MATERIAL_CLASS_ENUM = ["Polymer", "Fiber", "Composite"] | |
| # --- schema (Tasks 2/3/8) -------------------------------------------------- | |
| _PROPERTY_SCHEMA: dict[str, Any] = { | |
| "type": "OBJECT", | |
| "properties": { | |
| "section": {"type": "STRING", "enum": SECTION_ENUM}, | |
| "property_name": {"type": "STRING"}, | |
| "value_raw": {"type": "STRING"}, | |
| "value_num": {"type": "NUMBER", "nullable": True}, | |
| "value_min": {"type": "NUMBER", "nullable": True}, | |
| "value_max": {"type": "NUMBER", "nullable": True}, | |
| "qualifier": {"type": "STRING"}, | |
| "unit": {"type": "STRING"}, | |
| "test_condition": {"type": "STRING"}, | |
| "comments": {"type": "STRING"}, | |
| "source_quote": {"type": "STRING"}, | |
| "page": {"type": "INTEGER"}, | |
| }, | |
| "required": ["section", "property_name", "value_raw", "unit", "source_quote", "page"], | |
| } | |
| EXTRACTION_SCHEMA: dict[str, Any] = { | |
| "type": "OBJECT", | |
| "properties": { | |
| "materials": { | |
| "type": "ARRAY", | |
| "items": { | |
| "type": "OBJECT", | |
| "properties": { | |
| "material_name": {"type": "STRING"}, | |
| "material_abbreviation": {"type": "STRING"}, | |
| "material_class": {"type": "STRING", "enum": MATERIAL_CLASS_ENUM}, | |
| "trade_grade": {"type": "STRING"}, | |
| "manufacturer": {"type": "STRING"}, | |
| "matrix": {"type": "STRING"}, | |
| "fiber": {"type": "STRING"}, | |
| "fiber_volume_fraction": {"type": "STRING"}, | |
| "properties": {"type": "ARRAY", "items": _PROPERTY_SCHEMA}, | |
| }, | |
| "required": ["material_name", "material_class", "properties"], | |
| }, | |
| } | |
| }, | |
| "required": ["materials"], | |
| } | |
| EXTRACTION_PROMPT = ( | |
| "You are an expert materials scientist. From the attached PDF, extract every " | |
| "distinct material it characterizes as a list under `materials`. A datasheet " | |
| "or paper may describe MORE THAN ONE material (e.g. a neat resin and its " | |
| "composite, or several grades) — return each as its own entry with only its " | |
| "own properties. Do NOT merge two materials into one.\n\n" | |
| "For each material provide:\n" | |
| "- material_name (generic material, e.g. 'isotactic polypropylene')\n" | |
| "- material_abbreviation\n" | |
| "- material_class: one of Polymer | Fiber | Composite. Choose Composite when a " | |
| "matrix is reinforced with fibers (laminate, prepreg, CF/PEEK, glass-filled, a " | |
| "reported fiber volume fraction); Fiber for a bare fiber/yarn/tow datasheet; " | |
| "Polymer otherwise.\n" | |
| "- trade_grade (commercial/trade name; '' if absent)\n" | |
| "- manufacturer (company; '' if absent)\n" | |
| "- matrix, fiber, fiber_volume_fraction (e.g. '55%') — for composites; '' otherwise\n\n" | |
| "Extract ALL numeric and descriptive properties across every category. For " | |
| "each property return:\n" | |
| "- section: one of " + ", ".join(SECTION_ENUM) + "\n" | |
| "- property_name\n" | |
| "- value_raw: the value EXACTLY as printed, including ranges and qualifiers " | |
| "(e.g. '100-120', '≤ -18.0', '~3.5')\n" | |
| "- value_num: the single numeric value, or null if it is a range/non-numeric\n" | |
| "- value_min, value_max: the low/high of a range, else null\n" | |
| "- qualifier: one of '', '<', '<=', '>', '>=', '~', '±'\n" | |
| "- unit: the unit exactly as printed ('' if dimensionless)\n" | |
| "- test_condition (e.g. '23 °C, 50% RH'; '' if none)\n" | |
| "- comments ('' if none)\n" | |
| "- source_quote: the VERBATIM sentence or table cell from the PDF that states " | |
| "this value. Copy it character-for-character; do not paraphrase.\n" | |
| "- page: the 1-based PDF page number the value appears on.\n\n" | |
| "Never invent values. If a value is not in the document, do not include it. " | |
| "Respond ONLY with valid JSON following the schema." | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # Data classes | |
| # --------------------------------------------------------------------------- | |
| class Property: | |
| section: str | |
| property_name: str | |
| value_raw: str | |
| unit: str = "" | |
| value_num: Optional[float] = None | |
| value_min: Optional[float] = None | |
| value_max: Optional[float] = None | |
| qualifier: str = "" | |
| test_condition: str = "" | |
| comments: str = "" | |
| source_quote: str = "" | |
| page: Optional[int] = None | |
| # filled by verify_against_text() / canonicalization: | |
| unit_canonical: str = "" | |
| value_si: Optional[float] = None | |
| status: str = "ok" | |
| flag_reason: str = "" | |
| class Material: | |
| material_name: str | |
| material_abbreviation: str = "" | |
| material_class: str = "" | |
| trade_grade: str = "" | |
| manufacturer: str = "" | |
| matrix: str = "" | |
| fiber: str = "" | |
| fiber_volume_fraction: str = "" | |
| properties: list[Property] = dataclasses.field(default_factory=list) | |
| class Extraction: | |
| materials: list[Material] = dataclasses.field(default_factory=list) | |
| model: str = dataclasses.field(default_factory=text_model) | |
| prompt_version: str = PROMPT_VERSION | |
| doc_status: str = "ok" # "ok" | "scanned_no_text" | "empty_extraction" | |
| # Gemini usage for the text call (0 when no call was made). tokens_out | |
| # includes the model's thinking tokens, which are billed as output; | |
| # tokens_thinking is that thinking share on its own (for cost analysis). | |
| tokens_in: int = 0 | |
| tokens_out: int = 0 | |
| tokens_thinking: int = 0 | |
| def usage_detail(resp: Any) -> tuple[int, int, int]: | |
| """(tokens_in, tokens_out, tokens_thinking) from a generateContent | |
| response's ``usageMetadata``. tokens_out is billed output (candidates + | |
| thinking); tokens_thinking is the thinking part alone. (0, 0, 0) when | |
| absent or unparseable; never raises.""" | |
| try: | |
| data = resp.json() if hasattr(resp, "json") else (resp or {}) | |
| u = (data or {}).get("usageMetadata") or {} | |
| tin = int(u.get("promptTokenCount") or 0) | |
| think = int(u.get("thoughtsTokenCount") or 0) | |
| tout = int(u.get("candidatesTokenCount") or 0) + think | |
| return max(0, tin), max(0, tout), max(0, think) | |
| except Exception: | |
| return 0, 0, 0 | |
| def usage_from_response(resp: Any) -> tuple[int, int]: | |
| """(tokens_in, tokens_out) from a generateContent response's | |
| ``usageMetadata``; (0, 0) when absent or unparseable. Output counts | |
| thinking tokens (``thoughtsTokenCount``) with the candidate tokens, which | |
| is how they are billed. Never raises: cost accounting must not be able | |
| to fail an extraction.""" | |
| try: | |
| data = resp.json() if hasattr(resp, "json") else (resp or {}) | |
| u = (data or {}).get("usageMetadata") or {} | |
| tin = int(u.get("promptTokenCount") or 0) | |
| tout = int(u.get("candidatesTokenCount") or 0) + int(u.get("thoughtsTokenCount") or 0) | |
| return max(0, tin), max(0, tout) | |
| except Exception: | |
| return 0, 0 | |
| class PropertyRow: | |
| """One flattened, fully-resolved row ready for the SQLite mirror. | |
| Carries the legacy columns (`value`, `unit`, `english`, ...) so the CSV | |
| export and page1.py keep working, plus the new structured/provenance | |
| columns added in this phase. | |
| """ | |
| # identity / routing | |
| material_name: str | |
| material_abbreviation: str | |
| material_key: str | |
| material_class: str | |
| # legacy columns (kept populated for backward compat) | |
| section: str | |
| property_name: str | |
| value: str | |
| unit: str | |
| english: str | |
| test_condition: str | |
| comments: str | |
| # composite descriptors promoted to real columns (Task 2) | |
| trade_grade: str = "" | |
| manufacturer: str = "" | |
| matrix: str = "" | |
| fiber: str = "" | |
| fiber_volume_fraction: str = "" | |
| # structured numeric value (Task 3) | |
| value_raw: str = "" | |
| value_num: Optional[float] = None | |
| value_min: Optional[float] = None | |
| value_max: Optional[float] = None | |
| qualifier: str = "" | |
| unit_canonical: str = "" | |
| value_si: Optional[float] = None | |
| # provenance (Tasks 1/6) | |
| source_pdf: str = "" | |
| source_sha1: str = "" | |
| page: Optional[int] = None | |
| source_quote: str = "" | |
| # status (Tasks 1/3/4) | |
| status: str = "ok" | |
| flag_reason: str = "" | |
| # bookkeeping | |
| model: str = dataclasses.field(default_factory=text_model) | |
| prompt_version: str = PROMPT_VERSION | |
| # provenance kind (figure-mining phase): 'text' (grounded in the PDF | |
| # text) or 'figure' (read off a plot/table image — an estimate; see | |
| # figures.py). figure_id points at the harvested PNG. | |
| origin: str = "text" | |
| figure_id: str = "" | |
| # figure-linking phase (figure_links.py, ported from the InDeS mapper): | |
| # for a TEXT row, figure_id may instead point at the harvested figure the | |
| # row's evidence CITES ("see Fig. 3"). The link is grounding evidence, | |
| # never a value — status is untouched. image_url / image are the columns | |
| # the shared Space reads to show a row's plot (S3 ref / PNG bytes). | |
| figure_ref: str = "" # citation as parsed, e.g. "Fig. 3(b)" | |
| figure_link_score: Optional[float] = None # mapper5 score: 0.90-1.00 citation, <0.9 fallback | |
| figure_link_signals: str = "" # "figure_citation, page_confirm" / "page, tokens" | |
| image_url: str = "" | |
| image: bytes = dataclasses.field(default=b"", repr=False, compare=False) | |
| # --------------------------------------------------------------------------- | |
| # Gemini request with retry/backoff (Task 7) | |
| # --------------------------------------------------------------------------- | |
| _KEY_PARAM_RE = re.compile(r"([?&](?:key|api_key|apikey|token|access_token)=)[^&\s'\")\]]+", re.I) | |
| def redact_secrets(text: Any) -> str: | |
| """Strip credential query parameters from error text. requests' HTTPError | |
| and urllib3's connection errors quote the full request URL — for Gemini | |
| that URL carries `?key=<API key>` — so any error message that is logged | |
| or stored has to pass through here first.""" | |
| return _KEY_PARAM_RE.sub(r"\1<redacted>", str(text)) | |
| class GeminiHTTPError(requests.HTTPError): | |
| """HTTPError for a Gemini endpoint whose message names the status and | |
| the API's own error detail (e.g. RESOURCE_EXHAUSTED: "Your project has | |
| exceeded its monthly spending cap") instead of the keyed URL.""" | |
| def __init__(self, message: str, response: requests.Response, status: str = "", | |
| detail: str = "") -> None: | |
| super().__init__(message, response=response) | |
| self.status = status # Gemini "status" field, e.g. RESOURCE_EXHAUSTED | |
| self.detail = detail # Gemini "message" field, human-readable | |
| def quota(self) -> bool: | |
| """True when the API refused for budget/quota reasons (spending cap, | |
| daily quota): a condition that will not clear by retrying now.""" | |
| return (self.status == "RESOURCE_EXHAUSTED" | |
| or "spending cap" in self.detail.lower() | |
| or "quota" in self.detail.lower()) | |
| def _gemini_error_fields(resp: requests.Response) -> tuple[str, str]: | |
| """(status, message) from a Gemini error body, or ('', '') if not JSON.""" | |
| try: | |
| body = resp.json() | |
| except ValueError: | |
| return "", "" | |
| err = (body or {}).get("error") if isinstance(body, dict) else None | |
| if not isinstance(err, dict): | |
| return "", "" | |
| return str(err.get("status") or ""), str(err.get("message") or "")[:300] | |
| def _retry_after(resp: requests.Response) -> Optional[float]: | |
| val = resp.headers.get("Retry-After") | |
| if not val: | |
| return None | |
| try: | |
| # Cap: a hostile/buggy server sending Retry-After: 86400 must not | |
| # stall a cycle for hours (Space anti-hang hardening, commit 3f67f54). | |
| return max(0.0, min(float(val), 300.0)) | |
| except ValueError: | |
| return None | |
| def _request_with_retry( | |
| method: str, | |
| url: str, | |
| *, | |
| what: str = "Gemini", | |
| timeout: int = REQUEST_TIMEOUT_S, | |
| _sleep=time.sleep, | |
| **kw: Any, | |
| ) -> Optional[requests.Response]: | |
| """Issue one HTTP request, retrying 429/5xx and connection errors. | |
| Shared policy for every Gemini endpoint (generateContent, File API start / | |
| upload / status poll): bounded retries with exponential backoff, honoring | |
| a Retry-After header. Returns the 200 Response, or None if retries are | |
| exhausted / the status is non-retryable (after raise_for_status). | |
| """ | |
| for attempt in range(MAX_RETRIES + 1): | |
| try: | |
| resp = requests.request(method, url, timeout=timeout, **kw) | |
| except requests.RequestException as exc: | |
| if attempt < MAX_RETRIES: | |
| delay = BACKOFF_BASE * (2 ** attempt) | |
| log.warning("%s request failed: %s; retry %d/%d in %.1fs", | |
| what, redact_secrets(exc), attempt + 1, MAX_RETRIES, delay) | |
| _sleep(delay) | |
| continue | |
| # Connection errors quote the URL (with the key) — re-raise clean. | |
| raise requests.ConnectionError( | |
| f"{what} request failed: {redact_secrets(exc)}") from None | |
| if resp.status_code == 200: | |
| return resp | |
| status, detail = _gemini_error_fields(resp) | |
| if resp.status_code == 429 and (status == "RESOURCE_EXHAUSTED" | |
| or "spending cap" in detail.lower()): | |
| # A spent budget (monthly spending cap, daily quota) is not a | |
| # momentary rate limit: retrying it in 2/4/8 s only burns time, | |
| # 14 s per PDF, and on 28 Sep 2026 the retries' error text put | |
| # the API key into the public event stream. Fail at once. | |
| log.error("%s -> 429 %s: %s", what, status, detail) | |
| raise GeminiHTTPError(f"429 {status or 'RESOURCE_EXHAUSTED'} from {what}: {detail}", | |
| resp, status, detail) | |
| if resp.status_code in RETRY_STATUS and attempt < MAX_RETRIES: | |
| delay = _retry_after(resp) or BACKOFF_BASE * (2 ** attempt) | |
| log.warning( | |
| "%s -> %s; retry %d/%d in %.1fs", | |
| what, resp.status_code, attempt + 1, MAX_RETRIES, delay, | |
| ) | |
| _sleep(delay) | |
| continue | |
| # Non-retryable (4xx) or retries exhausted: fail loudly, once. This | |
| # raise deliberately sits OUTSIDE the try above — HTTPError is a | |
| # RequestException, and raising it inside used to be caught by the | |
| # connection-error branch and retried, so a 400/401/403 (bad key, | |
| # invalid schema) burned MAX_RETRIES backoffs before surfacing. | |
| # The message names the endpoint and the API's detail, never the | |
| # keyed URL (requests' own raise_for_status() message does). | |
| log.error("%s -> %s: %s", what, resp.status_code, resp.text[:300]) | |
| raise GeminiHTTPError(f"{resp.status_code} {resp.reason or 'error'} from {what}" | |
| + (f": {status} {detail}".rstrip() if (status or detail) else ""), | |
| resp, status, detail) | |
| return None | |
| def thinking_config(setting: Optional[str] = None) -> Optional[dict]: | |
| """The generationConfig.thinkingConfig for the current THINKING setting, | |
| or None for "dynamic" (send nothing: the API default) and after the | |
| model rejected the option once.""" | |
| if _THINKING_UNSUPPORTED: | |
| return None | |
| v = (setting if setting is not None else THINKING or "").strip().lower() | |
| if not v or v == "dynamic": | |
| return None | |
| if v == "off": | |
| return {"thinkingBudget": 0} | |
| if v in ("minimal", "low", "medium", "high"): | |
| return {"thinkingLevel": v} | |
| try: | |
| return {"thinkingBudget": max(0, int(v))} | |
| except ValueError: | |
| log.warning("GEMINI_THINKING=%r not understood; using the API default", v) | |
| return None | |
| def apply_thinking(payload: dict) -> dict: | |
| """Add the thinking option to a generateContent payload (in place).""" | |
| tc = thinking_config() | |
| gc = payload.setdefault("generationConfig", {}) | |
| if tc is None: | |
| gc.pop("thinkingConfig", None) | |
| else: | |
| gc["thinkingConfig"] = tc | |
| return payload | |
| def gemini_request( | |
| url: str, | |
| payload: dict[str, Any], | |
| *, | |
| timeout: int = REQUEST_TIMEOUT_S, | |
| _sleep=time.sleep, | |
| ) -> Optional[requests.Response]: | |
| """POST `payload` to `url`, retrying 429/5xx and connection errors. | |
| Mirrors pdf_crawler.http_get()'s policy: bounded retries with exponential | |
| backoff, honoring a Retry-After header. Returns the 200 Response, or None | |
| if retries are exhausted / the status is non-retryable. | |
| The thinking option (THINKING) is applied here, so every Gemini call in | |
| extraction.py and figures.py gets it. A 400 that names the option | |
| (a model without that field, an out-of-range value) is retried once | |
| without it and disables it for the process: cost tuning must never stop | |
| an extraction. | |
| """ | |
| global _THINKING_UNSUPPORTED | |
| apply_thinking(payload) | |
| try: | |
| return _request_with_retry("POST", url, json=payload, timeout=timeout, _sleep=_sleep) | |
| except GeminiHTTPError as exc: | |
| gc = payload.get("generationConfig") or {} | |
| if (exc.response is not None and exc.response.status_code == 400 | |
| and "thinkingConfig" in gc and "think" in (exc.detail or "").lower()): | |
| log.warning("model rejected thinkingConfig=%s (%s); retrying without it and " | |
| "leaving thinking at the API default from now on", | |
| gc.get("thinkingConfig"), exc.detail[:120]) | |
| _THINKING_UNSUPPORTED = True | |
| gc.pop("thinkingConfig", None) | |
| return _request_with_retry("POST", url, json=payload, timeout=timeout, _sleep=_sleep) | |
| raise | |
| # File API processing wait: exponential poll (1, 2, 4, 8, ... capped at 15 s) | |
| # up to this many seconds. Was a fixed 30 x 1 s, so a large PDF that took | |
| # longer than 30 s to process came back None -> empty_extraction and was | |
| # re-uploaded from scratch on every run. | |
| FILE_ACTIVE_WAIT_S = 180 | |
| FILE_POLL_CAP_S = 15.0 | |
| def _upload_pdf_file( | |
| pdf_bytes: bytes, filename: str, api_key: str, *, _sleep=time.sleep | |
| ) -> Optional[str]: | |
| """Upload a PDF via the Gemini File API (resumable protocol); return file URI. | |
| Used for large/long PDFs (Task 10) where base64 inlining would blow the | |
| generateContent request-size limit. Every hop (start, upload, status poll) | |
| goes through the same retry/backoff policy as generateContent — a single | |
| free-tier 429 used to fail the whole PDF. | |
| """ | |
| start_url = GEMINI_UPLOAD_URL.format(key=api_key) | |
| start = _request_with_retry( | |
| "POST", start_url, what="File API start", | |
| headers={ | |
| "X-Goog-Upload-Protocol": "resumable", | |
| "X-Goog-Upload-Command": "start", | |
| "X-Goog-Upload-Header-Content-Length": str(len(pdf_bytes)), | |
| "X-Goog-Upload-Header-Content-Type": "application/pdf", | |
| "Content-Type": "application/json", | |
| }, | |
| json={"file": {"display_name": filename}}, | |
| _sleep=_sleep, | |
| ) | |
| if start is None: | |
| return None | |
| upload_url = start.headers.get("X-Goog-Upload-URL") | |
| if not upload_url: | |
| log.error("File API did not return an upload URL") | |
| return None | |
| up = _request_with_retry( | |
| "POST", upload_url, what="File API upload", | |
| headers={ | |
| "X-Goog-Upload-Offset": "0", | |
| "X-Goog-Upload-Command": "upload, finalize", | |
| "Content-Length": str(len(pdf_bytes)), | |
| }, | |
| data=pdf_bytes, | |
| _sleep=_sleep, | |
| ) | |
| if up is None: | |
| return None | |
| info = up.json().get("file", {}) | |
| name = info.get("name") | |
| uri = info.get("uri") | |
| state = info.get("state") | |
| # Wait for the file to become ACTIVE before referencing it. | |
| waited = 0.0 | |
| delay = 1.0 | |
| while True: | |
| if state == "ACTIVE": | |
| return uri | |
| if state == "FAILED": | |
| log.error("File API processing failed for %s", filename) | |
| return None | |
| if waited >= FILE_ACTIVE_WAIT_S or not name: | |
| log.error("File API: %s not ACTIVE after %.0fs (state=%s)", | |
| filename, waited, state) | |
| return None | |
| _sleep(delay) | |
| waited += delay | |
| delay = min(delay * 2, FILE_POLL_CAP_S) | |
| poll = _request_with_retry( | |
| "GET", GEMINI_FILE_STATUS_URL.format(name=name, key=api_key), | |
| what="File API status", _sleep=_sleep, | |
| ) | |
| if poll is None: | |
| return None | |
| info = poll.json() | |
| state = info.get("state") | |
| uri = info.get("uri", uri) | |
| def _parse_extraction_json(resp: Any) -> Optional[dict[str, Any]]: | |
| data = resp.json() if hasattr(resp, "json") else (resp or {}) | |
| candidates = data.get("candidates", []) | |
| if not candidates: | |
| return None | |
| parts = candidates[0].get("content", {}).get("parts", []) | |
| for part in parts: | |
| text = (part.get("text") or "").strip() | |
| if text.startswith("{"): | |
| try: | |
| return json.loads(text) | |
| except json.JSONDecodeError: | |
| log.error("Gemini returned non-JSON text") | |
| return None | |
| return None | |
| # --------------------------------------------------------------------------- | |
| # PDF text (grounding + scanned detection) | |
| # --------------------------------------------------------------------------- | |
| def pdf_page_texts(pdf_bytes: bytes) -> list[str]: | |
| """Return per-page extractable text. Empty list if PyMuPDF is unavailable.""" | |
| if fitz is None: | |
| return [] | |
| try: | |
| doc = fitz.open(stream=pdf_bytes, filetype="pdf") | |
| except Exception as exc: # pragma: no cover | |
| log.warning("Could not open PDF for text extraction: %s", exc) | |
| return [] | |
| try: | |
| return [p.get_text() for p in doc] | |
| finally: | |
| doc.close() | |
| def _is_scanned(page_texts: list[str]) -> bool: | |
| if not page_texts: | |
| return False # can't tell without PyMuPDF; don't falsely flag | |
| total = sum(len(t.strip()) for t in page_texts) | |
| return (total / max(1, len(page_texts))) < SCANNED_TEXT_PER_PAGE | |
| # --------------------------------------------------------------------------- | |
| # Extraction entry point | |
| # --------------------------------------------------------------------------- | |
| def extract_from_pdf(pdf_bytes: bytes, filename: str, api_key: str) -> Extraction: | |
| """Extract structured materials from a PDF (Tasks 2/3/8/10). | |
| Returns an Extraction whose ``doc_status`` is ``scanned_no_text`` (and | |
| ``materials`` empty) when the PDF has no extractable text, so the caller | |
| can record the source without fabricating rows. | |
| """ | |
| page_texts = pdf_page_texts(pdf_bytes) | |
| if _is_scanned(page_texts): | |
| log.info("%s looks scanned/image-only; skipping extraction", filename) | |
| return Extraction(doc_status="scanned_no_text") | |
| parts: list[dict[str, Any]] = [{"text": EXTRACTION_PROMPT}] | |
| use_file_api = ( | |
| len(pdf_bytes) > INLINE_PDF_LIMIT_BYTES or len(page_texts) > INLINE_PAGE_LIMIT | |
| ) | |
| if use_file_api: | |
| uri = _upload_pdf_file(pdf_bytes, filename, api_key) | |
| if not uri: | |
| return Extraction(doc_status="empty_extraction") | |
| parts.append({"fileData": {"mimeType": "application/pdf", "fileUri": uri}}) | |
| else: | |
| parts.append( | |
| { | |
| "inlineData": { | |
| "mimeType": "application/pdf", | |
| "data": base64.b64encode(pdf_bytes).decode("utf-8"), | |
| } | |
| } | |
| ) | |
| payload = { | |
| "contents": [{"parts": parts}], | |
| "generationConfig": { | |
| "temperature": 0, | |
| "responseMimeType": "application/json", | |
| "responseSchema": EXTRACTION_SCHEMA, | |
| }, | |
| } | |
| model = text_model() | |
| url = GEMINI_URL_TEMPLATE.format(model=model, key=api_key) | |
| resp = gemini_request(url, payload) | |
| if resp is None: | |
| return Extraction(doc_status="empty_extraction", model=model) | |
| return extraction_from_response(resp, model) | |
| def extraction_from_response(resp: Any, model: Optional[str] = None) -> Extraction: | |
| """Extraction (with usage) from a generateContent response object or its | |
| parsed JSON dict (the Batch API hands back dicts).""" | |
| model = model or text_model() | |
| tokens_in, tokens_out, tokens_thinking = usage_detail(resp) | |
| raw = _parse_extraction_json(resp) | |
| if not raw: | |
| # The call was made and billed even though nothing usable came back. | |
| return Extraction(doc_status="empty_extraction", model=model, | |
| tokens_in=tokens_in, tokens_out=tokens_out, | |
| tokens_thinking=tokens_thinking) | |
| out = _extraction_from_json(raw) | |
| out.model = model | |
| out.tokens_in, out.tokens_out, out.tokens_thinking = tokens_in, tokens_out, tokens_thinking | |
| return out | |
| def _extraction_from_json(raw: dict[str, Any]) -> Extraction: | |
| """Coerce raw model JSON into a typed Extraction (tolerates the old shape).""" | |
| materials_json = raw.get("materials") | |
| if materials_json is None: | |
| # Back-compat: a single-material response with a flat property list. | |
| flat = raw.get("mechanical_properties") or raw.get("properties") or [] | |
| materials_json = [ | |
| { | |
| "material_name": raw.get("material_name", ""), | |
| "material_abbreviation": raw.get("material_abbreviation", ""), | |
| "material_class": raw.get("material_class", ""), | |
| "trade_grade": raw.get("trade_grade", ""), | |
| "manufacturer": raw.get("manufacturer", ""), | |
| "properties": flat, | |
| } | |
| ] | |
| materials: list[Material] = [] | |
| for mj in materials_json or []: | |
| props: list[Property] = [] | |
| for pj in mj.get("properties") or []: | |
| value_raw = (pj.get("value_raw") or pj.get("value") or "").strip() | |
| prop = Property( | |
| section=_norm_section(pj.get("section")), | |
| property_name=(pj.get("property_name") or "").strip() or "Unknown property", | |
| value_raw=value_raw, | |
| unit=(pj.get("unit") or "").strip(), | |
| value_num=_as_float(pj.get("value_num")), | |
| value_min=_as_float(pj.get("value_min")), | |
| value_max=_as_float(pj.get("value_max")), | |
| qualifier=(pj.get("qualifier") or "").strip(), | |
| test_condition=(pj.get("test_condition") or "").strip(), | |
| comments=(pj.get("comments") or "").strip(), | |
| source_quote=(pj.get("source_quote") or "").strip(), | |
| page=_as_int(pj.get("page")), | |
| ) | |
| # Fill structured numeric fields from value_raw if the model omitted them. | |
| _fill_numeric(prop) | |
| props.append(prop) | |
| materials.append( | |
| Material( | |
| material_name=(mj.get("material_name") or "").strip(), | |
| material_abbreviation=(mj.get("material_abbreviation") or "").strip(), | |
| material_class=(mj.get("material_class") or "").strip(), | |
| trade_grade=(mj.get("trade_grade") or "").strip(), | |
| manufacturer=(mj.get("manufacturer") or "").strip(), | |
| matrix=(mj.get("matrix") or "").strip(), | |
| fiber=(mj.get("fiber") or "").strip(), | |
| fiber_volume_fraction=(mj.get("fiber_volume_fraction") or "").strip(), | |
| properties=props, | |
| ) | |
| ) | |
| return Extraction(materials=materials) | |
| def _as_float(v: Any) -> Optional[float]: | |
| if v is None or v == "": | |
| return None | |
| try: | |
| return float(v) | |
| except (TypeError, ValueError): | |
| return None | |
| def _as_int(v: Any) -> Optional[int]: | |
| if v is None or v == "": | |
| return None | |
| try: | |
| return int(v) | |
| except (TypeError, ValueError): | |
| return None | |
| def _norm_section(section: Optional[str]) -> str: | |
| """Normalize free-text section drift to the enum (Task 8).""" | |
| s = (section or "").strip() | |
| if not s: | |
| return "Descriptive" | |
| low = s.lower() | |
| for canon in SECTION_ENUM: | |
| if low == canon.lower(): | |
| return canon | |
| # substring / synonym normalization, e.g. "Mechanical Properties" -> "Mechanical" | |
| synonyms = { | |
| "mechanical": "Mechanical", | |
| "thermal": "Thermal", | |
| "electrical": "Electrical", | |
| "dielectric": "Electrical", | |
| "physical": "Physical", | |
| "optical": "Optical", | |
| "rheolog": "Rheological", | |
| "viscosity": "Rheological", | |
| "processing": "Processing", | |
| "descriptive": "Descriptive", | |
| "composition": "Composition/Reinforcement", | |
| "reinforcement": "Composition/Reinforcement", | |
| "architecture": "Architecture/Structure", | |
| "structure": "Architecture/Structure", | |
| } | |
| for key, canon in synonyms.items(): | |
| if key in low: | |
| return canon | |
| return "Descriptive" | |
| # --------------------------------------------------------------------------- | |
| # Value parsing (Task 3) | |
| # --------------------------------------------------------------------------- | |
| _QUALIFIER_MAP = [ | |
| ("≤", "<="), ("≥", ">="), ("≈", "~"), | |
| ("<=", "<="), (">=", ">="), ("<", "<"), (">", ">"), | |
| ("~", "~"), ("±", "±"), | |
| ] | |
| _NUM_RE = re.compile(r"[-+]?\d{1,3}(?:,\d{3})+(?:\.\d+)?|[-+]?\d*\.?\d+(?:[eE][-+]?\d+)?") | |
| # Plus/minus built from _NUM_RE so both operands accept thousands separators and | |
| # scientific notation ('1,200 ± 100', '1e-3 ± 2e-4') — a plain \d*\.?\d+ operand | |
| # used to anchor after the comma and read '1,200 ± 100' as 200 ± 100. | |
| _PM_RE = re.compile(rf"({_NUM_RE.pattern})\s*(?:±|\+/-|\+-)\s*({_NUM_RE.pattern})") | |
| def _clean_number(tok: str) -> Optional[float]: | |
| try: | |
| return float(tok.replace(",", "")) | |
| except ValueError: | |
| return None | |
| def parse_value_raw(value_raw: str) -> tuple[Optional[float], Optional[float], Optional[float], str]: | |
| """Parse a printed value into (value_num, value_min, value_max, qualifier). | |
| Handles ranges ('100-120', '100 to 120'), qualifiers ('<= -18', '~3.5'), | |
| plus-minus ('5.0 +/- 0.2'), thousands separators and scientific notation. | |
| """ | |
| if not value_raw: | |
| return None, None, None, "" | |
| s = unicodedata.normalize("NFKC", value_raw).strip() | |
| # NFKC does NOT fold the typographic minus (U+2212) or dashes to '-'. PDFs | |
| # typeset negatives and ranges with them, so without this a Tg of '−60 °C' | |
| # parses as +60 and '20−30' loses its range. Same folding as _normalize_text. | |
| s = s.replace("−", "-").replace("–", "-").replace("—", "-") | |
| qualifier = "" | |
| for needle, canon in _QUALIFIER_MAP: | |
| if needle in s: | |
| qualifier = canon | |
| break | |
| # plus/minus -> midpoint value with min/max | |
| pm = _PM_RE.search(s) | |
| if pm: | |
| base = _clean_number(pm.group(1)) | |
| delta = _clean_number(pm.group(2)) | |
| if base is not None and delta is not None: | |
| delta = abs(delta) | |
| return base, base - delta, base + delta, "±" | |
| # Range: two numbers separated by a dash/"to" (all dash variants were folded | |
| # to '-' above). Guard against a leading sign being misread as a separator | |
| # by working on the sign-stripped remainder. | |
| body = s | |
| lead_sign = "" | |
| if body[:1] in "+-": | |
| lead_sign, body = body[0], body[1:] | |
| range_match = re.match( | |
| r"\s*(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d*\.?\d+(?:[eE][-+]?\d+)?)" | |
| r"\s*(?:-|to|\.\.\.|…)\s*" | |
| r"([-+]?\d{1,3}(?:,\d{3})+(?:\.\d+)?|[-+]?\d*\.?\d+(?:[eE][-+]?\d+)?)", | |
| body, | |
| ) | |
| if range_match: | |
| lo = _clean_number(lead_sign + range_match.group(1)) | |
| hi = _clean_number(range_match.group(2)) | |
| if lo is not None and hi is not None: | |
| if lo > hi: | |
| lo, hi = hi, lo | |
| return None, lo, hi, qualifier | |
| nums = _NUM_RE.findall(s) | |
| if not nums: | |
| return None, None, None, qualifier | |
| val = _clean_number(nums[0]) | |
| return val, None, None, qualifier | |
| def _fill_numeric(prop: Property) -> None: | |
| """Backfill value_num/min/max/qualifier from value_raw when model omitted them.""" | |
| has_any = ( | |
| prop.value_num is not None | |
| or prop.value_min is not None | |
| or prop.value_max is not None | |
| ) | |
| if has_any and prop.qualifier: | |
| return | |
| num, lo, hi, qual = parse_value_raw(prop.value_raw) | |
| if prop.value_num is None and prop.value_min is None and prop.value_max is None: | |
| prop.value_num, prop.value_min, prop.value_max = num, lo, hi | |
| if not prop.qualifier: | |
| prop.qualifier = qual | |
| def _representative_value(prop: Property) -> Optional[float]: | |
| if prop.value_num is not None: | |
| return prop.value_num | |
| if prop.value_min is not None and prop.value_max is not None: | |
| return (prop.value_min + prop.value_max) / 2.0 | |
| return prop.value_min if prop.value_min is not None else prop.value_max | |
| # --------------------------------------------------------------------------- | |
| # Unit normalization + plausibility (Task 3) | |
| # --------------------------------------------------------------------------- | |
| if pint is not None: # pragma: no branch | |
| _UREG = pint.UnitRegistry() | |
| _UREG.define("ksi = 1000 * psi") | |
| _UREG.define("Msi = 1000000 * psi") | |
| else: # pragma: no cover | |
| _UREG = None | |
| class _Family: | |
| name: str | |
| keywords: tuple[str, ...] | |
| canonical: str # pint unit string (or display for non-pint families) | |
| display: str # human-facing unit label | |
| lo: float # plausibility low, in `display` units | |
| hi: float # plausibility high, in `display` units | |
| kind: str # "physical" | "temperature" | "raw" | "passthrough" | |
| si_factor: float = 1.0 # for "raw" families: value_si = value_in_display * si_factor | |
| # For "raw" families: accepted printed-unit spellings (normalized via | |
| # _norm_unit_key) -> factor that converts the printed value into `display` | |
| # units. '' may be present to accept a bare number. Anything else is a | |
| # unit_review, never a silent default. Populated below. | |
| unit_map: dict[str, float] = dataclasses.field(default_factory=dict) | |
| # Keyword matching. Multi-word keywords match as substrings ("tensile modulus" | |
| # in "Tensile modulus (0°)"). SHORT tokens (tg, tm, hdt, cte) must match on | |
| # token boundaries — bare substring matching made 'tm' hit "ASTM" (so every | |
| # property citing an ASTM standard became a melting temperature) and 'tg' hit | |
| # "outgassing". Left boundary = not a letter/digit; right boundary = not a | |
| # LETTER (a trailing digit is allowed: 'Tg2', 'CTE1', 'HDT1.8' are how DSC/DMA | |
| # reports and laminate datasheets label second-heating Tg and the two CTE | |
| # regimes). | |
| _SHORT_TOKENS = frozenset({"tg", "tm", "tc", "hdt", "cte", "clte", "cai", "sbs", | |
| "young", "young's", "youngs"}) | |
| _KEYWORD_RE_CACHE: dict[str, "re.Pattern[str]"] = {} | |
| def _keyword_matches(kw: str, name: str) -> bool: | |
| if kw not in _SHORT_TOKENS: | |
| return kw in name | |
| pat = _KEYWORD_RE_CACHE.get(kw) | |
| if pat is None: | |
| pat = re.compile(r"(?<![a-z0-9])" + re.escape(kw) + r"(?![a-z])") | |
| _KEYWORD_RE_CACHE[kw] = pat | |
| return pat.search(name) is not None | |
| # Accepted printed spellings for the raw families. Keys are normalized by | |
| # _norm_unit_key() (lowercased, NFKC, whitespace/µ/degree-sign folded, | |
| # 'x⁻¹'/'x^-1' suffixes rewritten to '/x'). | |
| _CTE_UNIT_MAP: dict[str, float] = { | |
| # -> ppm/°C | |
| "": 1.0, # bare: see _bare_value_factor (ppm vs 1/K by magnitude) | |
| "ppm/c": 1.0, "ppm/k": 1.0, "ppm/degc": 1.0, "ppm/°c": 1.0, "ppm": 1.0, | |
| "um/m/c": 1.0, "um/m/k": 1.0, "um/m/degc": 1.0, "um/m/°c": 1.0, | |
| "um/(m*c)": 1.0, "um/(m*k)": 1.0, "um/(m·c)": 1.0, "um/(m·k)": 1.0, | |
| "um/m·c": 1.0, "um/m·k": 1.0, "um/m°c": 1.0, "um/mk": 1.0, "um/m*c": 1.0, "um/m*k": 1.0, | |
| "um·/m/k": 1.0, "um/m/m/k": 1.0, | |
| "10^-6/c": 1.0, "10^-6/k": 1.0, "10^-6/degc": 1.0, "10^-6/°c": 1.0, | |
| "10-6/c": 1.0, "10-6/k": 1.0, "10-6/°c": 1.0, | |
| "x10^-6/c": 1.0, "x10^-6/k": 1.0, "x10^-6/°c": 1.0, | |
| "x10-6/c": 1.0, "x10-6/k": 1.0, "x10-6/°c": 1.0, | |
| "e-6/c": 1.0, "e-6/k": 1.0, "e-6/°c": 1.0, | |
| "1e-6/c": 1.0, "1e-6/k": 1.0, "1e-6/°c": 1.0, | |
| "uin/in/f": 1.8, "uin/in/°f": 1.8, "uin/in/degf": 1.8, "ppm/f": 1.8, | |
| "ppm/degf": 1.8, "ppm/°f": 1.8, "10^-6/f": 1.8, "10^-6/°f": 1.8, | |
| "10-6/f": 1.8, "10-6/°f": 1.8, "x10^-6/f": 1.8, "x10^-6/°f": 1.8, | |
| "10^-6in/in/f": 1.8, "10^-6in/in/°f": 1.8, "10-6in/in/f": 1.8, "10-6in/in/°f": 1.8, | |
| "1/c": 1e6, "1/k": 1e6, "1/degc": 1e6, "1/°c": 1e6, "/c": 1e6, "/k": 1e6, | |
| "/degc": 1e6, "/°c": 1e6, "m/m/c": 1e6, "m/m/k": 1e6, "m/m/°c": 1e6, | |
| "m/(m*k)": 1e6, "m/(m·k)": 1e6, "mm/mm/c": 1e6, "mm/mm/k": 1e6, "mm/mm/°c": 1e6, | |
| "1/f": 1.8e6, "1/degf": 1.8e6, "1/°f": 1.8e6, "/f": 1.8e6, "/°f": 1.8e6, | |
| "in/in/f": 1.8e6, "in/in/°f": 1.8e6, | |
| } | |
| _ELONGATION_UNIT_MAP: dict[str, float] = { | |
| # -> % | |
| "%": 1.0, "percent": 1.0, "pct": 1.0, | |
| # bare / ratio: see _bare_value_factor — a strain FRACTION (0.024) or a | |
| # percent printed without its unit (2.4), decided by magnitude. | |
| "": 1.0, "-": 1.0, "mm/mm": 100.0, "m/m": 100.0, "in/in": 100.0, | |
| "ratio": 100.0, "strain": 100.0, "fraction": 100.0, | |
| } | |
| _SPECIFIC_GRAVITY_UNIT_MAP: dict[str, float] = { | |
| # dimensionless by definition; tolerate g/cm³ spellings (SG ≈ density in g/cm³) | |
| "": 1.0, "-": 1.0, "g/cm3": 1.0, "g/cm^3": 1.0, "g/cc": 1.0, "g/ml": 1.0, | |
| "g/l": 0.001, "kg/m3": 0.001, "kg/m^3": 0.001, "kg/dm3": 1.0, "kg/dm^3": 1.0, | |
| "kg/l": 1.0, | |
| } | |
| def _bare_value_factor(fam: _Family, rep: float, value_raw: str) -> Optional[float]: | |
| """Factor for a raw-family value printed with NO unit (or a lone dash). | |
| The bare case is genuinely ambiguous, so decide by what the number can be: | |
| * elongation: '%' inside value_raw wins (a datasheet cell "2.4 %" the model | |
| copied whole) -> percent. Otherwise |v| < 0.5 is a strain fraction | |
| (0.024 = 2.4 %; no thermoplastic composite fails below 0.5 % strain); | |
| |v| >= 0.5 is a percent printed without its unit (2.4 -> 2.4 %). The | |
| x100 rule used to apply to every bare number and turned "2.4 %"/'' into | |
| 240 % with status ok. | |
| * cte: |v| < 1e-3 is a 1/K fraction (2.3e-5 -> 23 ppm/°C), else ppm/°C. | |
| * specific gravity: dimensionless, factor 1. | |
| Returns None to say "don't know" -> unit_review. | |
| """ | |
| if fam.name == "elongation": | |
| if "%" in value_raw: | |
| return 1.0 | |
| return 100.0 if abs(rep) < 0.5 else 1.0 | |
| if fam.name == "cte": | |
| return 1e6 if 0 < abs(rep) < 1e-3 else 1.0 | |
| if fam.name == "specific_gravity": | |
| return 1.0 | |
| return None | |
| # Ordered: specific keywords before generic ones (first match wins). | |
| # "passthrough" families exist to *shield* generic keywords: e.g. dielectric, | |
| # impact and tear strength are not pressures, so they must not fall into the | |
| # MPa families below them. They carry no unit check and no plausibility range. | |
| # Compression-after-impact (CAI) is a real MPa strength and must be caught | |
| # BEFORE the impact shield. | |
| PROPERTY_FAMILIES: list[_Family] = [ | |
| _Family("cai_strength", ("compression after impact", "compression-after-impact", | |
| "cai"), | |
| "MPa", "MPa", 0.5, 6000.0, "physical"), | |
| _Family("dielectric_strength", ("dielectric strength", "breakdown strength", | |
| "breakdown voltage"), | |
| "", "", 0.0, 0.0, "passthrough"), | |
| _Family("impact_strength", ("impact strength", "impact resistance", "izod", | |
| "charpy", "impact energy"), | |
| "", "", 0.0, 0.0, "passthrough"), | |
| _Family("tear_strength", ("tear strength", "tear resistance"), | |
| "", "", 0.0, 0.0, "passthrough"), | |
| _Family("peel_strength", ("peel strength", "peel resistance", "bond strength", | |
| "adhesion strength", "adhesive strength", "weld strength"), | |
| "", "", 0.0, 0.0, "passthrough"), | |
| _Family("tensile_modulus", | |
| ("tensile modulus", "modulus of elasticity", "young", "young's", | |
| "youngs", "elastic modulus"), | |
| "GPa", "GPa", 0.01, 1000.0, "physical"), | |
| _Family("flexural_modulus", ("flexural modulus", "bending modulus"), | |
| "GPa", "GPa", 0.01, 800.0, "physical"), | |
| _Family("shear_modulus", ("shear modulus", "modulus of rigidity"), | |
| "GPa", "GPa", 0.005, 500.0, "physical"), | |
| _Family("storage_modulus", ("storage modulus", "compressive modulus", "modulus"), | |
| "GPa", "GPa", 0.001, 1000.0, "physical"), | |
| _Family("tensile_strength", | |
| ("tensile strength", "ultimate tensile", "strength at break", | |
| "yield strength", "tensile stress"), | |
| "MPa", "MPa", 0.5, 10000.0, "physical"), | |
| _Family("flexural_strength", ("flexural strength", "bending strength"), | |
| "MPa", "MPa", 0.5, 4000.0, "physical"), | |
| _Family("compressive_strength", ("compressive strength", "compression strength"), | |
| "MPa", "MPa", 0.5, 6000.0, "physical"), | |
| # Bare "strength" was removed from this family: it captured dielectric / | |
| # impact / tear strength (see passthrough guards above) into an MPa check. | |
| _Family("shear_strength", ("shear strength", "ilss", "interlaminar shear"), | |
| "MPa", "MPa", 0.5, 4000.0, "physical"), | |
| _Family("glass_transition", ("glass transition", "tg"), | |
| "degC", "°C", -150.0, 600.0, "temperature"), | |
| _Family("melting", ("melting", "melt temperature", "tm"), | |
| "degC", "°C", 50.0, 500.0, "temperature"), | |
| _Family("crystallization", ("crystallization",), | |
| "degC", "°C", 0.0, 500.0, "temperature"), | |
| _Family("decomposition", ("decomposition", "degradation temperature"), | |
| "degC", "°C", 100.0, 1200.0, "temperature"), | |
| _Family("hdt", ("heat deflection", "deflection temperature", "hdt", | |
| "heat distortion"), | |
| "degC", "°C", 0.0, 600.0, "temperature"), | |
| _Family("cte", ("thermal expansion", "cte", "clte", "expansion coefficient"), | |
| "ppm/degC", "ppm/°C", -50.0, 500.0, "raw", si_factor=1e-6, | |
| unit_map=_CTE_UNIT_MAP), | |
| # Specific gravity is dimensionless by definition; keep it OUT of the pint | |
| # density family or every SG row is false-flagged missing_unit. | |
| _Family("specific_gravity", ("specific gravity", "relative density"), | |
| "", "", 0.1, 12.0, "raw", si_factor=1.0, | |
| unit_map=_SPECIFIC_GRAVITY_UNIT_MAP), | |
| _Family("density", ("density",), | |
| "g/cm**3", "g/cm³", 0.1, 12.0, "physical"), | |
| _Family("elongation", ("elongation", "strain at break", "strain to failure", | |
| "failure strain", "ultimate strain"), | |
| "%", "%", 0.001, 2000.0, "raw", si_factor=0.01, | |
| unit_map=_ELONGATION_UNIT_MAP), | |
| # Generic composite strengths that are real pressures (MPa). Kept LAST so | |
| # every shield / specific family above wins first; the bare "strength" | |
| # keyword is safe here because dielectric / impact / tear / peel are | |
| # already routed to passthrough families. | |
| _Family("other_strength", | |
| ("short beam", "short-beam", "sbs", "bearing strength", "open hole", | |
| "open-hole", "filled hole", "filled-hole", "notched tensile", | |
| "unnotched", "interlaminar strength", "transverse strength", | |
| "strength"), | |
| "MPa", "MPa", 0.5, 10000.0, "physical"), | |
| ] | |
| def _match_family(prop: Property) -> Optional[_Family]: | |
| name = prop.property_name.lower() | |
| for fam in PROPERTY_FAMILIES: | |
| for kw in fam.keywords: | |
| if _keyword_matches(kw, name): | |
| return fam | |
| return None | |
| # 'X⁻¹' / 'X^-1' / 'X-1' suffix -> '/X' so 'K⁻¹', '10⁻⁶ K⁻¹', 'µm·m⁻¹·K⁻¹' | |
| # collapse onto the '/k' spellings already in the maps. | |
| _INVERSE_SUFFIX_RE = re.compile(r"([a-z°]+)\^?-1") | |
| def _norm_unit_key(unit: str) -> str: | |
| """Normalize a printed unit into a lookup key for the raw-family unit maps.""" | |
| u = unit.strip() | |
| # BEFORE NFKC: it folds 'º' (masculine ordinal, often typed for a degree | |
| # sign) to 'o' and superscripts to plain digits, so handle those first. | |
| u = u.replace("º", "°").replace("˚", "°") | |
| u = u.replace("⁻", "-").replace("⁺", "+") | |
| u = u.translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789")) | |
| u = unicodedata.normalize("NFKC", u).lower() | |
| u = u.replace("µ", "u").replace("μ", "u") | |
| u = u.replace("−", "-").replace("–", "-").replace("—", "-") | |
| u = u.replace(" ", "").replace("(", "").replace(")", "") | |
| u = u.replace("**", "^").replace("×", "x").replace("*10", "x10").replace("e-06", "e-6") | |
| u = u.replace("·", "*") | |
| # 'x⁻¹' style: 'k^-1' -> '/k'; 'um*m-1*k-1' -> 'um/m/k' | |
| u = _INVERSE_SUFFIX_RE.sub(r"/\1", u) | |
| u = u.replace("*/", "/").replace("//", "/") | |
| if u.startswith("*"): | |
| u = u[1:] | |
| return u | |
| def _raw_factor(fam: _Family, unit: str, rep: Optional[float] = None, | |
| value_raw: str = "") -> Optional[float]: | |
| """Factor converting a printed raw-family value into `fam.display` units, or None.""" | |
| key = _norm_unit_key(unit) | |
| if key in ("", "-") and rep is not None: | |
| return _bare_value_factor(fam, rep, value_raw) | |
| if key in fam.unit_map: | |
| return fam.unit_map[key] | |
| # tolerate '°c' vs 'c' vs 'degc' interchangeably | |
| for a, b in (("°c", "c"), ("degc", "c"), ("°f", "f"), ("degf", "f"), | |
| ("°k", "k"), ("degk", "k")): | |
| alt = key.replace(a, b) | |
| if alt in fam.unit_map: | |
| return fam.unit_map[alt] | |
| return None | |
| # A letter followed by 2/3 is a power (cm3, mm2, in2, ft3, kJ/m2 ...) — but NOT | |
| # the exponent marker of scientific notation ('1e3 psi', '10E3 psi'). | |
| _UNIT_POWER_RE = re.compile(r"(?<=[A-Za-z])(?<![0-9.][eE])([23])(?![0-9])") | |
| _SUPERSCRIPT_RE = re.compile(r"([⁰¹²³⁴⁵⁶⁷⁸⁹⁻⁺]+)") | |
| _SUP_TRANS = str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹⁻⁺", "0123456789-+") | |
| def _preprocess_unit(unit: str) -> str: | |
| # Superscripts FIRST: NFKC folds '³' to a plain '3', so '10³ psi' became | |
| # '103 psi' (a silent 10x error on the standard US spelling of ksi) and | |
| # the later '³' replace was dead code. | |
| u = _SUPERSCRIPT_RE.sub(lambda m: "**" + m.group(1).translate(_SUP_TRANS), unit.strip()) | |
| u = unicodedata.normalize("NFKC", u) | |
| u = u.replace("·", "*").replace("−", "-") | |
| u = u.replace("µ", "u").replace("μ", "u") | |
| u = u.replace("^", "**") | |
| u = u.replace("g/cc", "g/cm**3") | |
| # Any letter immediately followed by 2/3 is a power: cm3, mm2, m3, in2, ft3… | |
| # (was a hardcoded list that missed 'mm2' — the standard European MPa | |
| # spelling 'N/mm2' failed to parse and landed in unit_review). | |
| u = _UNIT_POWER_RE.sub(r"**\1", u) | |
| return u | |
| _TEMP_UNITS = { | |
| "": "degC", "c": "degC", "degc": "degC", "°c": "degC", "celsius": "degC", | |
| "k": "kelvin", "kelvin": "kelvin", | |
| "f": "degF", "degf": "degF", "°f": "degF", "fahrenheit": "degF", | |
| } | |
| def canonicalize(prop: Property) -> tuple[str, Optional[float], Optional[str]]: | |
| """Return (unit_canonical, value_si, problem). | |
| `problem` is None on success, or a short reason ("unit_review:...") when the | |
| unit is missing/dimensionally wrong for the property family. | |
| """ | |
| fam = _match_family(prop) | |
| rep = _representative_value(prop) | |
| if fam is None or fam.kind == "passthrough": | |
| # No known family (or a shield family): pass the unit through, no SI | |
| # conversion, no check. | |
| return (prop.unit, None, None) | |
| if rep is None: | |
| return (fam.display, None, None) | |
| if fam.kind == "raw": | |
| # Never apply si_factor blindly: CTE '2.3e-5 1/K' is 23 ppm/°C, not | |
| # 2.3e-11; elongation '0.024' (a strain fraction) is 2.4 %, not 0.024 %. | |
| factor = _raw_factor(fam, prop.unit, rep, prop.value_raw) | |
| if factor is None: | |
| return (fam.display, None, f"unit_review:unexpected_unit:{prop.unit}!~{fam.display}") | |
| return (fam.display, rep * factor * fam.si_factor, None) | |
| if _UREG is None: # pragma: no cover | |
| return (fam.display, None, None) | |
| try: | |
| if fam.kind == "temperature": | |
| key = unicodedata.normalize("NFKC", prop.unit).strip().lower() | |
| src = _TEMP_UNITS.get(key) | |
| if src is None: | |
| return (fam.display, None, f"unit_review:bad_temp_unit:{prop.unit}") | |
| q = _UREG.Quantity(rep, src) | |
| value_si = q.to("kelvin").magnitude | |
| return (fam.display, value_si, None) | |
| # physical (pressure, density, ...) | |
| pre = _preprocess_unit(prop.unit) | |
| if not pre: | |
| return (fam.display, None, "unit_review:missing_unit") | |
| q = rep * _UREG(pre) | |
| target = _UREG(fam.canonical) | |
| if q.dimensionality != target.dimensionality: | |
| return (fam.display, None, | |
| f"unit_review:dim_mismatch:{prop.unit}!~{fam.display}") | |
| value_si = q.to_base_units().magnitude | |
| return (fam.display, value_si, None) | |
| except Exception as exc: # pint parse failure, undefined unit, etc. | |
| return (fam.display, None, f"unit_review:unparseable:{prop.unit}:{exc}") | |
| def _canonical_value(prop: Property, fam: _Family) -> Optional[float]: | |
| """Representative value expressed in the family's `display` unit (for range check).""" | |
| rep = _representative_value(prop) | |
| if rep is None: | |
| return None | |
| if fam.kind == "passthrough": | |
| return None | |
| if fam.kind == "raw": | |
| factor = _raw_factor(fam, prop.unit, rep, prop.value_raw) | |
| return None if factor is None else rep * factor | |
| if _UREG is None: # pragma: no cover | |
| return rep | |
| try: | |
| if fam.kind == "temperature": | |
| key = unicodedata.normalize("NFKC", prop.unit).strip().lower() | |
| src = _TEMP_UNITS.get(key) | |
| if src is None: | |
| return None | |
| return _UREG.Quantity(rep, src).to(fam.canonical).magnitude | |
| pre = _preprocess_unit(prop.unit) | |
| if not pre: | |
| return None | |
| q = rep * _UREG(pre) | |
| if q.dimensionality != _UREG(fam.canonical).dimensionality: | |
| return None | |
| return q.to(fam.canonical).magnitude | |
| except Exception: | |
| return None | |
| def plausibility_problem(prop: Property) -> Optional[str]: | |
| """Range-check the value *after* unit conversion. Returns reason or None.""" | |
| fam = _match_family(prop) | |
| if fam is None or fam.kind == "passthrough": | |
| return None | |
| cval = _canonical_value(prop, fam) | |
| if cval is None: | |
| return None # couldn't convert -> handled as unit_review elsewhere | |
| if not (fam.lo <= cval <= fam.hi): | |
| return f"out_of_range[{fam.lo},{fam.hi}{fam.display}]:{cval:.4g}" | |
| return None | |
| _PLACEHOLDER_VALUES = {"", "n/a", "na", "-", "--", "n.a.", "none", "tbd", "…", "..."} | |
| def _empty_value(prop: Property) -> bool: | |
| # Fold the typographic dashes a table cell is actually printed with ('–', | |
| # '—', '−') before the lookup — a verbatim '–' cell used to pass as a | |
| # value and even ground (its folded '-' matched any hyphen on the page). | |
| v = prop.value_raw.strip().lower() | |
| v = v.replace("–", "-").replace("—", "-").replace("−", "-") | |
| return v in _PLACEHOLDER_VALUES | |
| # --------------------------------------------------------------------------- | |
| # Text grounding (Task 1) | |
| # --------------------------------------------------------------------------- | |
| _SUPERSCRIPT_CHARS_RE = re.compile(r"[⁰¹²³⁴⁵⁶⁷⁸⁹⁺⁻₀₁₂₃₄₅₆₇₈₉]+") | |
| def _normalize_text(s: str) -> str: | |
| # Superscript/subscript digits are footnote markers or degree signs glued | |
| # to a value ('776¹', '0⁰/45⁰/90⁰'); NFKC would fold them into the number | |
| # ('7761'). Replace them with a space BEFORE normalizing. | |
| s = _SUPERSCRIPT_CHARS_RE.sub(" ", s) | |
| s = unicodedata.normalize("NFKC", s) | |
| s = s.replace("–", "-").replace("—", "-").replace("−", "-") | |
| s = re.sub(r"\s+", " ", s) | |
| return s.lower().strip() | |
| _NUMERIC_NEEDLE_RE = re.compile(r"^[-+~<>=≤≥≈±.,\d\s/e]+$", re.IGNORECASE) | |
| def _grounded(needle: str, haystacks: list[str]) -> bool: | |
| """True if `needle` occurs in any (already-normalized) haystack. | |
| Purely numeric needles are matched on **digit boundaries**: a value of "3" | |
| must not be "verified" by the "3" inside "ISO 527-3" or "23 °C", "1.2" | |
| must not match "11.25", and "200" must not match "1,200". Text needles | |
| (source_quote chunks) keep plain substring matching. Haystacks are | |
| expected to be `_normalize_text` output. | |
| """ | |
| n = _normalize_text(needle) | |
| if not n: | |
| return False | |
| if _NUMERIC_NEEDLE_RE.match(n): | |
| # Strip qualifiers/whitespace so "<= -18.0" grounds on "-18.0" (the | |
| # sign is part of the number; the qualifier may be typeset elsewhere). | |
| core = re.sub(r"^[~<>=≤≥≈±\s]+", "", n).strip() | |
| if not core or not re.search(r"\d", core): | |
| return False # '-', '.', 'e', '/' alone can never ground | |
| # Tighten the needle's own range dash ('70 - 75' -> '70-75'); the | |
| # haystack side is handled by the tolerant body pattern below. | |
| core = re.sub(r"(?<=\d) ?- ?(?=\d)", "-", core) | |
| # Left guard: no digit/dot/thousands-comma, and — for an unsigned | |
| # needle — no '-' either, so '3' cannot match the tail of 'ISO 527-3'. | |
| # A signed needle ('-18.0') legitimately starts with the '-'. | |
| left = r"(?<![\d.\-])(?<!\d,)" if core[0] not in "+-" else r"(?<![\d.])(?<!\d,)" | |
| # Right guard: no digit, no '.digit' (11.25), no ',digit' (1,200) — a | |
| # sentence-ending '.' or a list ', ' is a boundary. | |
| right = r"(?!\d|\.\d|,\d)" | |
| # Inside the needle, a range/minus dash may be typeset with spaces on | |
| # the page ('70 – 75', '2818 -0.46'); tolerate optional spaces around | |
| # a digit-dash-digit instead of rewriting the haystack, so a value | |
| # after a placeholder dash ('– 2250') still grounds. | |
| body = re.sub(r"(?<=\d)\\-(?=\d)", r"\\s?-\\s?", re.escape(core)) | |
| pat = re.compile(left + body + right) | |
| return any(pat.search(h) for h in haystacks) | |
| return any(n in h for h in haystacks) | |
| def verify_against_text(extraction: Extraction, page_texts: list[str]) -> Extraction: | |
| """Ground each property's value in the PDF text (Task 1). | |
| For every property: if `value_raw` (or a chunk of `source_quote`) appears on | |
| the cited page (falling back to any page), mark `status='ok'`; otherwise | |
| `status='unverified'`. Also folds in empty-value, unit, and range checks, | |
| using this precedence: | |
| empty_value > unverified > unit_review > out_of_range > ok | |
| """ | |
| if not page_texts: | |
| norm_pages: list[str] = [] | |
| else: | |
| norm_pages = [_normalize_text(t) for t in page_texts] | |
| for material in extraction.materials: | |
| for prop in material.properties: | |
| # always compute canonicalization so rows carry unit_canonical/value_si | |
| unit_canonical, value_si, unit_problem = canonicalize(prop) | |
| prop.unit_canonical = unit_canonical | |
| prop.value_si = value_si | |
| reasons: list[str] = [] | |
| status = "ok" | |
| if _empty_value(prop): | |
| status = "empty_value" | |
| reasons.append("empty_value") | |
| else: | |
| # grounding | |
| if norm_pages: | |
| page_idx = (prop.page - 1) if prop.page else None | |
| page_in_range = page_idx is not None and 0 <= page_idx < len(norm_pages) | |
| cited = [norm_pages[page_idx]] if page_in_range else [] | |
| if page_idx is not None and not page_in_range: | |
| # A cited page that doesn't exist is a wrong page too: | |
| # ground anywhere, but say so. | |
| found = _grounded(prop.value_raw, norm_pages) | |
| if found: | |
| reasons.append("grounded_off_page") | |
| else: | |
| found = _grounded(prop.value_raw, cited or norm_pages) | |
| if not found and cited: | |
| # Fallback: any page. Still grounded (the number IS in | |
| # the PDF) but the cited page was wrong — record it as | |
| # a soft reason so reviewers can see the weaker chain. | |
| found = _grounded(prop.value_raw, norm_pages) | |
| if found: | |
| reasons.append("grounded_off_page") | |
| if not found and prop.source_quote: | |
| chunk = prop.source_quote[:40] | |
| found = _grounded(chunk, norm_pages) | |
| if found: | |
| reasons.append("grounded_via_quote") | |
| if not found: | |
| status = "unverified" | |
| reasons.append("value_not_in_pdf_text") | |
| # unit problem | |
| if status == "ok" and unit_problem: | |
| status = "unit_review" | |
| reasons.append(unit_problem) | |
| # plausibility (only meaningful once unit is sane) | |
| if status == "ok": | |
| rng = plausibility_problem(prop) | |
| if rng: | |
| status = "out_of_range" | |
| reasons.append(rng) | |
| prop.status = status | |
| prop.flag_reason = "; ".join(reasons) | |
| return extraction | |
| # --------------------------------------------------------------------------- | |
| # Classification (Task 5) | |
| # --------------------------------------------------------------------------- | |
| COMPOSITE_KEYWORDS = ( | |
| "composite", "laminate", "reinforced", "prepreg", | |
| "cf/", "gf/", "fiber-reinforced", "fibre-reinforced", | |
| "ud ", "uni-directional", "unidirectional", "woven", | |
| "glass-filled", "glass filled", "carbon-filled", | |
| ) | |
| FIBER_KEYWORDS = ("fiber", "fibre", "yarn", "tow", "roving", "filament") | |
| def classify_material(material: Material) -> str: | |
| """Return 'Polymer' | 'Fiber' | 'Composite' (Task 5). | |
| Primary signal is the model's `material_class` enum; a deterministic | |
| keyword fallback (using sections + Vf presence) only runs when the field is | |
| missing or invalid. The old haystack bug (repeating material_name N times, | |
| never reading section) is gone. | |
| """ | |
| field = (material.material_class or "").strip().capitalize() | |
| if field in MATERIAL_CLASS_ENUM: | |
| return field | |
| # --- deterministic fallback --- | |
| # A reported fiber volume fraction is a strong composite signal. | |
| if material.fiber_volume_fraction.strip(): | |
| return "Composite" | |
| haystack = " ".join( | |
| [ | |
| material.material_name or "", | |
| material.material_abbreviation or "", | |
| material.trade_grade or "", | |
| material.matrix or "", | |
| material.fiber or "", | |
| ] | |
| + [p.section or "" for p in material.properties] | |
| + [p.property_name or "" for p in material.properties] | |
| ).lower() | |
| if any(k in haystack for k in COMPOSITE_KEYWORDS): | |
| return "Composite" | |
| if any(k in haystack for k in FIBER_KEYWORDS): | |
| return "Fiber" | |
| return "Polymer" | |
| def _autoabbr(name: str) -> str: | |
| if not name: | |
| return "UNKNOWN" | |
| parts = [p[0] for p in name.split() if p and p[0].isalpha()] | |
| return "".join(parts).upper() or name[:6].upper() | |
| def material_key(material: Material) -> str: | |
| """Stable material identity (Task 6): normalized name, plus the trade grade. | |
| ``<name>`` when no trade grade is known, else ``<name>|<trade_grade>``. The | |
| grade is part of the identity: a datasheet describing PEEK 150G and PEEK | |
| 450G yields two materials named "PEEK", and without the grade in the key | |
| any property both grades share (density, Tg, ...) dedup-collapsed into | |
| one row and the other was silently dropped. Falls back to the | |
| abbreviation when the name is empty. | |
| """ | |
| name = (material.material_name or "").strip().lower() | |
| name = re.sub(r"\s+", " ", name) | |
| if not name: | |
| name = (material.material_abbreviation or "unknown").strip().lower() | |
| grade = re.sub(r"\s+", " ", (material.trade_grade or "").strip().lower()) | |
| if grade and grade != name and grade not in name: | |
| return f"{name}|{grade}" | |
| return name | |
| # --------------------------------------------------------------------------- | |
| # Flatten to DB rows (Tasks 2/5/6) | |
| # --------------------------------------------------------------------------- | |
| def to_rows(extraction: Extraction, source_pdf: str, source_sha1: str) -> list[PropertyRow]: | |
| """Flatten an Extraction into PropertyRows, iterating materials x properties.""" | |
| rows: list[PropertyRow] = [] | |
| for material in extraction.materials: | |
| mclass = classify_material(material) | |
| mkey = material_key(material) | |
| abbr = material.material_abbreviation or _autoabbr(material.material_name) | |
| for prop in material.properties: | |
| rows.append( | |
| PropertyRow( | |
| material_name=material.material_name, | |
| material_abbreviation=abbr, | |
| material_key=mkey, | |
| material_class=mclass, | |
| section=prop.section, | |
| property_name=prop.property_name, | |
| value=prop.value_raw, # legacy column == value_raw (back-compat) | |
| unit=prop.unit, | |
| english="", # legacy alt-units column; model no longer emits it | |
| test_condition=prop.test_condition, | |
| comments=prop.comments, | |
| trade_grade=material.trade_grade, | |
| manufacturer=material.manufacturer, | |
| matrix=material.matrix, | |
| fiber=material.fiber, | |
| fiber_volume_fraction=material.fiber_volume_fraction, | |
| value_raw=prop.value_raw, | |
| value_num=prop.value_num, | |
| value_min=prop.value_min, | |
| value_max=prop.value_max, | |
| qualifier=prop.qualifier, | |
| unit_canonical=prop.unit_canonical, | |
| value_si=prop.value_si, | |
| source_pdf=source_pdf, | |
| source_sha1=source_sha1, | |
| page=prop.page, | |
| source_quote=prop.source_quote, | |
| status=prop.status, | |
| flag_reason=prop.flag_reason, | |
| model=extraction.model, | |
| prompt_version=extraction.prompt_version, | |
| ) | |
| ) | |
| return rows | |