RICS / backend /config.py
StormShadow308's picture
Add demo documentation and Docker setup for v2 report generation system
aad7814
Raw
History Blame Contribute Delete
19.7 kB
"""Configuration for the template-agnostic v2 backend.
All values are overridable via environment variables (prefix-free) or a ``.env``
file at the repository root. See the package README for the deployment model.
"""
from __future__ import annotations
from functools import lru_cache
from pathlib import Path
from pydantic import Field
from pydantic_settings import BaseSettings, SettingsConfigDict
# backend/config.py -> repo root is one level up from this package.
REPO_ROOT = Path(__file__).resolve().parents[1]
class Settings(BaseSettings):
"""Central v2 settings object."""
model_config = SettingsConfigDict(
env_file=".env",
env_file_encoding="utf-8",
extra="ignore",
)
# ── LLM (OpenAI) ─────────────────────────────────────────────────────────
openai_api_key: str = Field(default="", description="OpenAI API key")
mapping_model: str = Field(
default="gpt-4o-mini",
description="Model used to map surveyor notes onto retrieved master paragraphs.",
)
discovery_model: str = Field(
default="gpt-4o-mini",
description="Model used for master-template schema discovery (JSON mode).",
)
grounding_model: str = Field(
default="gpt-4o-mini",
description="Model used for the PII / grounding audit pass.",
)
openai_request_timeout_seconds: float = Field(
default=60.0,
ge=5.0,
le=600.0,
description="Hard timeout for OpenAI chat and embedding HTTP calls (seconds).",
)
openai_pipeline_timeout_seconds: float = Field(
default=20.0,
ge=5.0,
le=120.0,
description=(
"Strict per-call timeout (seconds) for chat/embeddings inside the "
"report generation pipeline."
),
)
max_concurrent_llm_calls: int = Field(
default=10,
ge=1,
le=100,
description=(
"Maximum concurrent in-flight OpenAI chat completion HTTP requests "
"across parallel section workers."
),
)
openai_rate_limit_max_retries: int = Field(
default=5,
ge=0,
le=20,
description="Retry attempts after OpenAI HTTP 429 rate-limit responses.",
)
openai_rate_limit_backoff_base_seconds: float = Field(
default=1.0,
ge=0.1,
le=60.0,
description="Initial exponential backoff base (seconds) for 429 retries.",
)
openai_rate_limit_backoff_max_seconds: float = Field(
default=60.0,
ge=1.0,
le=300.0,
description="Maximum backoff delay (seconds) between 429 retries.",
)
# ── Embeddings (local MiniLM default; OpenAI optional) ───────────────────
embedding_provider: str = Field(
default="local",
description="Embedding backend: 'local' (sentence-transformers) or 'openai'.",
)
local_embedding_model: str = Field(
default="all-MiniLM-L6-v2",
description="sentence-transformers model used when embedding_provider='local'.",
)
openai_embedding_model: str = Field(
default="text-embedding-3-small",
description="OpenAI embedding model used when embedding_provider='openai'.",
)
# ── PII scrubbing ────────────────────────────────────────────────────────
spacy_model: str = Field(
default="en_core_web_trf",
description=(
"spaCy model used for the NER pass of PII scrubbing. When unavailable, the "
"scrubber falls back to its regex layer only."
),
)
pii_use_spacy: bool = Field(
default=True,
description="Enable the spaCy NER pass on top of the always-on regex pass.",
)
# ── Operator bundle (Master Standard report and paragraphs/) ─────────────
master_template_dir: str = Field(
default="Master Standard report and paragraphs",
description="Folder holding the report template PDF and standard-paragraphs Word file.",
)
report_template_filename: str = Field(
default="SAMPLE LEVEL 3 REPORT NCS.pdf",
description=(
"Report template (PDF): defines section structure, order and ratings. "
"Schema discovery reads this file."
),
)
standard_paragraphs_filename: str = Field(
default="HB-BS STANDARD PARAS v6 Sept 2015.doc",
description=(
"Standard paragraphs (Word): firm-approved boilerplate wording per section. "
"Ingested into the MASTER RAG tier."
),
)
# Backward-compatible alias for the standard-paragraphs file.
master_template_filename: str = Field(
default="HB-BS STANDARD PARAS v6 Sept 2015.doc",
description="Deprecated alias for standard_paragraphs_filename.",
)
master_template_auto_ingest: bool = Field(
default=False,
description=(
"Ingest the operator's shared standard-paragraph master into the MASTER "
"RAG tier on startup. Default False: in the per-tenant model every tenant "
"supplies its own past reports (REFERENCE tier) and generation is sourced "
"exclusively from those at all interference levels. Enable only if a firm "
"wants a shared boilerplate master seeded for the default tenant. The "
"canonical RICS L3 schema is always installed regardless of this flag."
),
)
master_template_prebuilt_schema: str = Field(
default="",
description="Optional path to a prebuilt schema.json; skips discovery when set and present.",
)
master_template_prebuilt_faiss: str = Field(
default="",
description="Optional path to a prebuilt MASTER FAISS dir; skips embedding when set and present.",
)
master_template_upload_enabled: bool = Field(
default=False,
description="Enable the gated admin override routes for replacing the master at runtime.",
)
# ── Optional reference uploads (past completed reports, style only) ────────
reference_auto_ingest_enabled: bool = Field(
default=False,
description=(
"When true, scan reference_auto_ingest_dir for extra past reports. "
"The report-template PDF and standard-paragraphs Word file are never "
"ingested as references."
),
)
reference_auto_ingest_dir: str = Field(
default="",
description="Folder to scan for reference docs; empty means use master_template_dir.",
)
reference_auto_ingest_globs: str = Field(
default="*.pdf,*.docx,*.doc",
description="Comma-separated globs scanned for reference docs (master filename excluded).",
)
# ── RAG / retrieval ──────────────────────────────────────────────────────
data_dir: str = Field(
default_factory=lambda: str(Path.home() / ".rics_v2"),
description="Root directory for per-tenant schema + FAISS artifacts.",
)
default_tenant_id: str = Field(
default="default",
description="Tenant that receives the operator master at startup.",
)
retrieval_top_k: int = Field(default=5, ge=1, le=50)
reference_baseline_top_k: int = Field(
default=4,
ge=1,
le=50,
description="Top REFERENCE-tier blocks retrieved as stylistic baseline scaffolding.",
)
retrieval_section_boost: float = Field(
default=0.25,
ge=0.0,
le=1.0,
description="FAISS score boost when chunk section_id matches the alias-resolved id.",
)
retrieval_lexical_boost: float = Field(
default=0.04,
ge=0.0,
le=0.5,
description="Per shared content-token boost when reranking paragraphs against notes.",
)
note_routing_mode: str = Field(
default="keyword",
description="Note routing: 'keyword' (deterministic regex) or 'rag' (embedding anchors).",
)
note_rag_match_min_score: float = Field(
default=0.72,
ge=0.0,
le=1.0,
description=(
"Minimum cosine similarity for a surveyor note to be mapped onto a "
"canonical section anchor. Below this, the note is surfaced as UNASSIGNED."
),
)
note_rag_ambiguity_margin: float = Field(
default=0.05,
ge=0.0,
le=0.5,
description=(
"Minimum cosine gap between the top two section-anchor matches. When "
"the margin is narrower, the note is treated as ambiguous and UNASSIGNED."
),
)
note_baseline_lexical_min_overlap: float = Field(
default=0.12,
ge=0.0,
le=1.0,
description=(
"Minimum token-overlap ratio between a note and the section baseline "
"to allow in-place mapping without a strong per-note RAG hit."
),
)
paragraph_min_chars: int = Field(default=80, ge=1, le=2000)
paragraph_max_chars: int = Field(default=1200, ge=100, le=4000)
chunk_overlap: int = Field(default=120, ge=0, le=1000)
reference_paragraph_max_chars: int = Field(
default=8000,
ge=500,
le=20000,
description="Max characters per REFERENCE-tier chunk (long-form past report sections).",
)
reference_chunk_overlap: int = Field(
default=1500,
ge=0,
le=5000,
description="Character overlap when splitting oversized REFERENCE sections.",
)
reference_section_complete_enabled: bool = Field(
default=True,
description=(
"Assemble the WHOLE past-report section as the mapping baseline (every "
"chunk for the chosen source+section, in document order) rather than only "
"the top-K semantically nearest chunks. Also lets a section that exists in "
"the index but was missed by similarity search fall back to a metadata "
"fetch before degrading to NOTES_ONLY. Disable to restore top-K-only."
),
)
reference_section_complete_max_chars: int = Field(
default=12000,
ge=1000,
le=40000,
description=(
"Safety cap on an assembled section-complete baseline. Chunks are added "
"in document order until this budget is reached, bounding the mapping "
"prompt while still covering long (30–50 line) past-report sections."
),
)
# ── Generation behaviour ─────────────────────────────────────────────────
default_survey_level: int = Field(
default=3,
ge=1,
le=3,
description="Default RICS survey tier (Level 3 β†’ full in-place narrative edit).",
)
composition_mode: str = Field(
default="in_place_edit",
description="Report composition strategy: style-informed in-place edit on REFERENCE baseline.",
)
max_tokens_mapping: int = Field(default=4096, ge=256, le=16000)
max_tokens_grounding: int = Field(default=1024, ge=256, le=8000)
max_tokens_discovery: int = Field(default=4000, ge=512, le=16000)
notes_expansion_enabled: bool = Field(
default=False,
description="Run the optional notes-expander pass before mapping.",
)
use_llm_paragraph_mapping: bool = Field(
default=True,
description=(
"Style-informed in-place edit on the REFERENCE baseline using the "
"mapping editor prompt. When false, deterministic weave only."
),
)
grounding_enabled: bool = Field(
default=True,
description="Run the grounding/PII audit on each mapped section before assembly.",
)
grounding_alert_threshold: float = Field(
default=0.2,
ge=0.0,
le=1.0,
description=(
"Fraction of sections needing review above which the preview flags "
"manual_review_required."
),
)
ai_transparency_footer_enabled: bool = Field(
default=False,
description="Append an AI transparency footer to generated DOCX when enabled.",
)
template_docx_path: str = Field(
default="",
description="Optional branded DOCX template path for report export.",
)
# ── Report post-processing (final cleanup pass) ───────────────────────────
postprocess_enabled: bool = Field(
default=True,
description="Master switch for the final report post-processing/normalisation pass.",
)
dedup_similarity_threshold: float = Field(
default=0.85,
ge=0.0,
le=1.0,
description="Character-trigram Jaccard threshold for near-duplicate paragraph removal (lower = more aggressive).",
)
extra_header_patterns: list[str] = Field(
default_factory=list,
description="Firm-specific header/artifact regex strings to strip during post-processing (keeps the pipeline firm-agnostic).",
)
postprocess_placeholder_pattern: str = Field(
default=r"\[{1,2}REDACTED_[A-Z]+(?:_\d+)?\]{1,2}",
description="Regex matching redaction tokens for post-processing grammar repair.",
)
postprocess_debug: bool = Field(
default=False,
description="When true, the post-processor prints a before/after preview to stdout.",
)
# ── Property-type retrieval guard ─────────────────────────────────────────
property_guard_enabled: bool = Field(
default=True,
description="Reject retrieved paragraphs whose terminology is incompatible with the property type (e.g. flat content in a house report).",
)
property_blocklist_extra_terms: list[str] = Field(
default_factory=list,
description="Additional firm-specific terms that mark a retrieved paragraph as property-incompatible.",
)
# ── Upload limits (batch + ZIP + notes extract) ───────────────────────────
max_single_upload_bytes: int = Field(
default=50 * 1024 * 1024,
ge=1024 * 1024,
le=200 * 1024 * 1024,
description="Max bytes per uploaded reference file (single or inside ZIP).",
)
max_zip_members: int = Field(
default=40,
ge=1,
le=200,
description="Max non-directory entries allowed inside an uploaded ZIP.",
)
max_zip_uncompressed_bytes: int = Field(
default=200 * 1024 * 1024,
ge=1024 * 1024,
le=1024 * 1024 * 1024,
description="Max total uncompressed size of all files in a ZIP.",
)
max_notes_extract_bytes: int = Field(
default=10 * 1024 * 1024,
ge=64 * 1024,
le=50 * 1024 * 1024,
description="Max bytes for /extract-notes uploads.",
)
# ── Section photos & vision ────────────────────────────────────────────────
max_section_photo_bytes: int = Field(
default=8 * 1024 * 1024,
ge=256 * 1024,
le=50 * 1024 * 1024,
description="Max bytes per uploaded section photo.",
)
max_section_photos_per_section: int = Field(
default=5,
ge=0,
le=50,
description="Max photos stored per report section.",
)
max_section_photos_for_ai: int = Field(
default=2,
ge=0,
le=10,
description="Max photos per section the user may select for AI vision analysis.",
)
section_photo_vision_enabled: bool = Field(
default=True,
description="When true and OpenAI is configured, analyze selected section photos at generation.",
)
vision_model: str = Field(
default="gpt-4o",
description="OpenAI vision-capable model for section photo analysis.",
)
vision_max_tokens: int = Field(
default=1200,
ge=256,
le=4096,
description="Token budget for a section photo vision analysis call.",
)
vision_timeout_seconds: float = Field(
default=90.0,
ge=10.0,
le=300.0,
description="Hard timeout per vision API call (seconds).",
)
vision_max_observations: int = Field(
default=12,
ge=4,
le=40,
description="Max observation lines merged from vision per section.",
)
# ── Auth ─────────────────────────────────────────────────────────────────
jwt_secret: str = Field(
default="change-me-in-production",
description="HMAC secret for signing tenant JWTs.",
)
jwt_algorithm: str = Field(default="HS256")
jwt_expiry_minutes: int = Field(default=60 * 24, ge=5, le=60 * 24 * 30)
admin_token: str = Field(
default="",
description="Static admin token required (in addition to JWT) for /admin routes.",
)
# ── Helpers ──────────────────────────────────────────────────────────────
def resolve_path(self, value: str | Path) -> Path:
"""Resolve a possibly repo-relative path to an absolute path."""
p = Path(value)
return p if p.is_absolute() else (REPO_ROOT / p)
@property
def report_template_path(self) -> Path:
"""Absolute path to the report template (PDF) β€” schema authority."""
return self.resolve_path(self.master_template_dir) / self.report_template_filename
@property
def standard_paragraphs_path(self) -> Path:
"""Absolute path to the standard-paragraphs Word file β€” MASTER RAG authority."""
name = self.standard_paragraphs_filename or self.master_template_filename
return self.resolve_path(self.master_template_dir) / name
@property
def master_template_path(self) -> Path:
"""Backward-compatible alias for :attr:`standard_paragraphs_path`."""
return self.standard_paragraphs_path
@property
def reference_dir_path(self) -> Path:
"""Absolute path to the folder scanned for reference auto-ingest."""
target = self.reference_auto_ingest_dir or self.master_template_dir
return self.resolve_path(target)
@property
def data_dir_path(self) -> Path:
return self.resolve_path(self.data_dir)
@property
def rag_top_k(self) -> int:
"""Spec alias for :attr:`retrieval_top_k`."""
return self.retrieval_top_k
@property
def confidence_threshold(self) -> float:
"""Alias for per-note REFERENCE match gate (CURSOR_REFACTOR parity)."""
return self.note_rag_match_min_score
@property
def branded_template_path(self) -> Path | None:
if not self.template_docx_path.strip():
return None
p = self.resolve_path(self.template_docx_path)
return p if p.is_file() else None
@lru_cache
def get_settings() -> Settings:
"""Cached settings singleton."""
return Settings()
settings = get_settings()