RandomZ / app /services /standard_paragraphs.py
StormShadow308's picture
feat: AI interference levels (minimum|medium|maximum) UI, API, prompts, generation
879e4e0
Raw
History Blame Contribute Delete
13 kB
"""Load per-section standard paragraphs from the firm's Word boilerplate.
Default source (see ``settings.rics_standard_paragraphs_docx``): a single
``HB-BS STANDARD PARAS`` master document under ``Behrang RICS Documents/``,
with headings keyed by RICS section codes (``A``, ``E1``, ``K3``, …).
Supported formats on disk:
* **.docx / .docm** — parsed directly with ``python-docx`` (Open XML).
* **.doc** (legacy Word 97–2003) — converted to a temporary ``.docx`` first,
using the first available tool: **Pandoc**, **LibreOffice** (``soffice``), or
**Microsoft Word** via ``pywin32`` (Windows only, optional install).
When an explicit path is configured and the file exists, **only** that file is
used — no other boilerplate under the knowledge base is merged in.
Used at **0% AI involvement** (deterministic stitch) and prepended to RAG
snippets for the assembly tier (≤12%) so the LLM sees firm text first.
"""
from __future__ import annotations
import fnmatch
import logging
import os
import shutil
import subprocess
import sys
import tempfile
from pathlib import Path
from app.config import settings
from app.services.knowledge_base import iter_knowledge_base_roots
from app.templates.registry import ALL_VALID_SECTION_CODES
logger = logging.getLogger(__name__)
_REPO_ROOT = Path(__file__).resolve().parents[2]
# (signature of discovered files + mtimes) -> merged code -> body
_index_cache: tuple[tuple[tuple[str, int], ...], dict[str, str]] | None = None
# Word VBA: wdFormatXMLDocument — Office Open XML (.docx) without macros
_WD_FORMAT_DOCX = 12
def _codes_for_heading_match() -> tuple[str, ...]:
"""Longest-first so ``E10`` would beat ``E1`` if both existed; RICS uses E1–E9 etc."""
return tuple(sorted(ALL_VALID_SECTION_CODES, key=lambda c: (-len(c), c)))
def _match_section_header(line: str, codes: tuple[str, ...]) -> str | None:
"""If ``line`` opens a section, return its RICS code."""
s = (line or "").strip()
if not s:
return None
upper = s.upper()
for code in codes:
cu = code.upper()
if not upper.startswith(cu):
continue
tail = s[len(code) :].strip()
if not tail:
return code
first = tail[0]
if first in ".:-–—\u2013\u2014)":
return code
if first.isspace():
return code
# "E1Roof" without separator — reject
return None
def _strip_heading_line_from_body(first_body_line: str, code: str) -> str:
"""Remove duplicate title fragment when the heading line was merged into body."""
s = first_body_line.strip()
if s.upper().startswith(code.upper()):
tail = s[len(code) :].strip()
if not tail:
return ""
if tail[0] in ".:-–—\u2013\u2014":
tail = tail[1:].strip()
return tail
return s
def _soffice_executable() -> str | None:
for name in ("soffice", "libreoffice"):
p = shutil.which(name)
if p:
return p
if sys.platform == "win32":
for pf in (
os.environ.get("ProgramFiles", r"C:\Program Files"),
os.environ.get("ProgramFiles(x86)", r"C:\Program Files (x86)"),
):
cand = Path(pf) / "LibreOffice" / "program" / "soffice.exe"
if cand.is_file():
return str(cand)
return None
def _convert_binary_word_to_docx(source: Path) -> tuple[Path | None, Path | None]:
"""Convert legacy ``.doc`` to ``.docx`` in a temp dir.
Returns:
``(path_to_docx, temp_dir)`` — caller must ``shutil.rmtree(temp_dir)``.
``(None, None)`` if conversion failed.
"""
out_dir = Path(tempfile.mkdtemp(prefix="rics_stdpar_"))
out = out_dir / f"{source.stem}.docx"
pandoc = shutil.which("pandoc")
if pandoc:
try:
subprocess.run(
[pandoc, str(source.resolve()), "-o", str(out.resolve())],
check=True,
capture_output=True,
text=True,
timeout=180,
)
if out.is_file() and out.stat().st_size > 0:
logger.info("Standard paragraphs: converted %s via pandoc", source.name)
return out, out_dir
except (subprocess.CalledProcessError, OSError, subprocess.TimeoutExpired) as exc:
logger.debug("pandoc .doc→.docx failed for %s: %s", source, exc)
soffice = _soffice_executable()
if soffice:
try:
subprocess.run(
[
soffice,
"--headless",
"--convert-to",
"docx",
"--outdir",
str(out_dir),
str(source.resolve()),
],
check=True,
capture_output=True,
text=True,
timeout=240,
)
if out.is_file() and out.stat().st_size > 0:
logger.info("Standard paragraphs: converted %s via LibreOffice", source.name)
return out, out_dir
except (subprocess.CalledProcessError, OSError, subprocess.TimeoutExpired) as exc:
logger.debug("LibreOffice .doc→.docx failed for %s: %s", source, exc)
try:
import win32com.client # type: ignore[import-untyped]
except ImportError:
pass
else:
word = None
doc = None
try:
word = win32com.client.Dispatch("Word.Application")
word.Visible = False
word.DisplayAlerts = 0
doc = word.Documents.Open(str(source.resolve()), ReadOnly=True)
doc.SaveAs2(str(out.resolve()), FileFormat=_WD_FORMAT_DOCX)
doc.Close(SaveChanges=False)
doc = None
word.Quit()
word = None
if out.is_file() and out.stat().st_size > 0:
logger.info("Standard paragraphs: converted %s via Microsoft Word (COM)", source.name)
return out, out_dir
except Exception as exc: # noqa: BLE001
logger.debug("Word COM .doc→.docx failed for %s: %s", source, exc)
finally:
if doc is not None:
try:
doc.Close(SaveChanges=False)
except Exception: # noqa: BLE001
pass
if word is not None:
try:
word.Quit()
except Exception: # noqa: BLE001
pass
shutil.rmtree(out_dir, ignore_errors=True)
logger.warning(
"Could not convert legacy Word .doc %s to .docx. Install one of: pandoc, LibreOffice, "
"or on Windows Microsoft Word + ``pip install pywin32``. "
"Alternatively save the master as .docx next to the .doc file.",
source,
)
return None, None
def _parse_openxml_word_doc(path: Path) -> dict[str, str]:
"""Parse a .docx/.docm package into ``section_code -> prose``."""
try:
from docx import Document # type: ignore[import-untyped]
except ImportError:
logger.warning("python-docx missing — cannot load standard paragraphs from %s", path)
return {}
codes = _codes_for_heading_match()
out: dict[str, str] = {}
current: str | None = None
buf: list[str] = []
def flush() -> None:
nonlocal current, buf
if not current:
buf = []
return
parts = [b.strip() for b in buf if b and str(b).strip()]
if parts:
parts[0] = _strip_heading_line_from_body(parts[0], current)
parts = [p for p in parts if p]
text = "\n\n".join(parts).strip()
if text:
out[current] = text
current = None
buf = []
try:
doc = Document(str(path))
except Exception as exc: # noqa: BLE001
logger.warning("Failed to open standard paragraphs Word file %s: %s", path, exc)
return {}
for para in doc.paragraphs:
raw = (para.text or "").strip()
if not raw:
continue
hdr = _match_section_header(raw, codes)
if hdr:
flush()
current = hdr
tail = raw[len(hdr) :].strip()
if tail and tail[0] in ".:-–—\u2013\u2014":
tail = tail[1:].strip()
if tail:
buf.append(tail)
elif current:
buf.append(raw)
flush()
return out
def parse_standard_paragraphs_docx(path: Path) -> dict[str, str]:
"""Parse standard-paragraphs master (``.docx``, ``.docm``, or legacy ``.doc``)."""
suf = path.suffix.lower()
if suf == ".doc":
converted, tmp = _convert_binary_word_to_docx(path)
if converted is None:
return {}
try:
return _parse_openxml_word_doc(converted)
finally:
if tmp is not None:
shutil.rmtree(tmp, ignore_errors=True)
if suf not in (".docx", ".docm"):
logger.warning(
"Unsupported standard-paragraphs format %r — use .docx, .docm, or .doc (%s)",
path.suffix,
path,
)
return {}
return _parse_openxml_word_doc(path)
def _resolve_config_path(raw: str) -> Path | None:
"""Resolve configured path; try alternate extension and repo-root basename."""
raw_s = raw.strip()
if not raw_s:
return None
p = Path(raw_s)
if not p.is_absolute():
p = _REPO_ROOT / p
if p.is_file():
return p
alt = p.with_suffix(".doc") if p.suffix.lower() == ".docx" else p.with_suffix(".docx")
if alt.is_file():
return alt
root_named = _REPO_ROOT / Path(raw_s).name
if root_named.is_file():
return root_named
return None
def _filename_matches_glob(name: str, pattern: str) -> bool:
return fnmatch.fnmatch(name.lower(), pattern.lower().lstrip())
def discover_standard_paragraph_docx_paths() -> list[Path]:
"""Resolve Word sources for standard paragraphs.
* If ``rics_standard_paragraphs_docx`` is set and a matching file exists → return
**only** that path (exclusive; no glob merge under KB roots).
* If set but missing → return empty (do not silently load other boilerplate).
* If unset/empty → discover under ``knowledge_base_dirs`` via
``standard_paragraphs_docx_globs`` (``.docx`` and legacy ``.doc``).
"""
explicit = (settings.rics_standard_paragraphs_docx or "").strip()
if explicit:
rp = _resolve_config_path(explicit)
if rp:
return [rp]
logger.warning(
"rics_standard_paragraphs_docx is set (%r) but file not found — "
"standard paragraphs disabled (no glob fallback)",
explicit,
)
return []
globs = [g.strip() for g in (settings.standard_paragraphs_docx_globs or "").split(",") if g.strip()]
if not globs:
globs = ["*standard*paragraph*.docx"]
paths: list[Path] = []
seen: set[str] = set()
for root in iter_knowledge_base_roots():
if not root.is_dir():
continue
candidates: list[Path] = []
for ext in ("*.docx", "*.docm", "*.doc"):
candidates.extend(sorted(root.rglob(ext)))
for fp in candidates:
name = fp.name
if not any(_filename_matches_glob(name, g) for g in globs):
continue
key = str(fp.resolve()).lower()
if key in seen:
continue
seen.add(key)
paths.append(fp)
return paths
def get_standard_paragraph_index() -> dict[str, str]:
"""Section_code → text from discovered boilerplate Word file(s) (cached by mtime)."""
global _index_cache
files = discover_standard_paragraph_docx_paths()
sig = tuple((str(p.resolve()), int(p.stat().st_mtime_ns)) for p in files if p.exists())
if _index_cache is not None and _index_cache[0] == sig:
return _index_cache[1]
merged: dict[str, str] = {}
for fp in files:
chunk = parse_standard_paragraphs_docx(fp)
if not chunk:
continue
for code, body in chunk.items():
if code not in merged:
merged[code] = body
logger.info(
"Standard paragraphs: loaded %d section(s) from %s",
len(chunk),
fp.name,
)
_index_cache = (sig, merged)
return merged
def get_standard_paragraph_for_section(section_code: str) -> str | None:
"""Return firm standard wording for ``section_code``, or ``None`` if not in the master."""
code = (section_code or "").strip()
if not code:
return None
idx = get_standard_paragraph_index()
body = idx.get(code)
if not body:
return None
return body.strip() or None
def clear_standard_paragraph_cache() -> None:
"""Test hook / reload after replacing the Word file on disk."""
global _index_cache
_index_cache = None