Spaces:
Runtime error
Runtime error
File size: 7,499 Bytes
aad7814 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 | """Split uploaded past-report text into section-tagged paragraphs for REFERENCE RAG."""
from __future__ import annotations
import re
from backend.config import settings
from backend.core.rag_store import TIER_REFERENCE, Chunk
_RICS_HEADING_LINE = re.compile(
r"""
(?m)^\s*
(?:(?:section|part|element|item)\s+)? # optional descriptive label before the code
(?P<code>[A-N]\d{1,2})
\b
(?:[\s:.\-\u2013\u2014]+[^\n]{0,120})? # optional separator + title on the same line
\s*$
""",
re.IGNORECASE | re.VERBOSE,
)
# Page-header/footer furniture that bleeds into PDF text extraction. Stripped from
# section bodies so it never becomes the stored baseline (e.g. a section whose only
# captured body was "Page 24RICS Home Survey - Level 3"). Bare standalone numbers are
# deliberately NOT stripped here — a lone digit may be a rating value, not a page number.
_PAGE_HEADER_FURNITURE_RE = re.compile(
r"(?im)^[ \t]*(?:page[ \t]*\d+[ \t]*)?"
r"rics[ \t]+home[ \t]+survey[ \t]*[-\u2013\u2014][ \t]*level[ \t]+\d.*$"
)
_PAGE_LABEL_RE = re.compile(r"(?im)^[ \t]*page[ \t]*\d+[ \t]*$")
# Survey-report PDFs embed photo index lines and RICS procedural NOTE blocks inside
# section bodies. These are not mappable baseline prose — they bloat the baseline,
# confuse the weave gate, and cause grounding rollbacks. Stripped at ingest only.
_PHOTO_LINE_RE = re.compile(r"(?im)^[ \t]*Photo\s*[-\u2013]?\s*\d+.*$")
_PHOTO_INLINE_RE = re.compile(
r"(?i)\s*Photo\s*[-\u2013]?\s*\d+\s*(?:[^\n.!?]|(?:\n(?![ \t]*Photo\s*[-\u2013]?\s*\d+)))*?[.!?]?"
)
_ELEMENT_TABLE_STUB_RE = re.compile(
r"(?im)^[ \t]*Element\s+no\.?\s*Element\s+name\s*$"
)
# NOTE N: procedural blocks (Building Regs boilerplate etc.) precede real findings.
# Stop at the next line that reads like survey prose, not a caption or header.
_NOTE_BLOCK_RE = re.compile(
r"(?is)"
r"^NOTE\s+\d+\s*:.*?"
r"(?="
r"\n(?:"
r"The|There|We|It|A|An|In|Condition|Main|Our|Your|This|These|Some|No|Where|When|"
r"Whilst|While|During|Following|Upon|After|Before|Internal|External|Ground|Roof|"
r"Wall|Floor|Ceiling|Chimney|Window|Door|Gutter|Rainwater|Damp|Mould|Wood|Timber|"
r"Brick|Stone|Slate|Tile|Pipe|Drain|Electric|Gas|Water|Heating|Boiler|Insulation|"
r"Ventilation|Asbestos|Structural|Surface|Visible|Noted|Observed|Evidence|Appears|"
r"Found|Seen|Installed|Located|Constructed|Finished|Painted|Plastered|Rendered|"
r"Pointed|Bedded|Covering|Structure|Material|Building|Property|Survey|Client|"
r"Surveyor|Defective|Further|Regular|Although|However|Generally|Typically|"
r"Inspection|Inspected|Examined|Checked|Tested|Operated|Opened|Accessed"
r")\b|\Z)"
)
# Collapse runs of blank lines left after stripping.
_MULTI_BLANK_RE = re.compile(r"\n{3,}")
def _strip_page_furniture(text: str) -> str:
"""Remove repeated page header/footer lines bled in by PDF extraction."""
cleaned = _PAGE_HEADER_FURNITURE_RE.sub("", text)
cleaned = _PAGE_LABEL_RE.sub("", cleaned)
return cleaned
def _clean_reference_body(text: str) -> str:
"""Remove non-prose survey-report artifacts from a REFERENCE section body."""
cleaned = _strip_page_furniture(text or "")
cleaned = _NOTE_BLOCK_RE.sub("", cleaned)
cleaned = _ELEMENT_TABLE_STUB_RE.sub("", cleaned)
cleaned = _PHOTO_LINE_RE.sub("", cleaned)
# Trailing / inline photo runs (often concatenated without newlines at section tail).
prev = None
while prev != cleaned:
prev = cleaned
cleaned = _PHOTO_INLINE_RE.sub("", cleaned)
cleaned = _MULTI_BLANK_RE.sub("\n\n", cleaned)
return cleaned.strip()
def _alpha_len(text: str) -> int:
"""Count alphabetic characters — a body-richness proxy for de-duplication."""
return sum(1 for ch in text if ch.isalpha())
def _normalise_code(raw: str) -> str:
return re.sub(r"\s+", "", raw.strip().upper())
def _is_valid_code(code: str, valid_section_ids: set[str] | None) -> bool:
if valid_section_ids is None:
return True
return code in {s.upper() for s in valid_section_ids}
def split_into_section_paragraphs(
text: str,
valid_section_ids: set[str] | None = None,
) -> list[tuple[str, int, str]]:
"""Return ``(section_id, paragraph_index, paragraph_text)`` tuples."""
matches = list(_RICS_HEADING_LINE.finditer(text))
if not matches:
return []
# A RICS code typically appears twice in one report: once as a stub row in the
# ratings summary table (body ≈ "Element no. Element name") and once as the real
# section with prose. Keep only the richest-prose body per code so the table stub
# never shadows the genuine content (both previously shared one chunk_id).
best_body: dict[str, str] = {}
for i, match in enumerate(matches):
code = _normalise_code(match.group("code"))
if not _is_valid_code(code, valid_section_ids):
continue
body_start = match.end()
end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
body = _clean_reference_body(text[body_start:end])
if not body:
continue
if _alpha_len(body) > _alpha_len(best_body.get(code, "")):
best_body[code] = body
if not best_body:
return []
out: list[tuple[str, int, str]] = []
max_chars = settings.reference_paragraph_max_chars
for section_id, body in best_body.items():
if len(body) <= max_chars:
out.append((section_id, 1, body))
continue
paragraphs = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()]
for idx, para in enumerate(paragraphs, start=1):
out.append((section_id, idx, para))
return out
def _chunk_section_body(para_text: str) -> list[str]:
"""Chunk one section body; keep whole block when within REFERENCE max size."""
from backend.core.ingest import _chunk_reference_text
text = (para_text or "").strip()
if not text:
return []
if len(text) <= settings.reference_paragraph_max_chars:
return [text]
return _chunk_reference_text(text)
def build_reference_chunks(
text: str,
*,
source_filename: str,
valid_section_ids: set[str] | None = None,
) -> list[Chunk]:
"""Build REFERENCE chunks with section metadata; prefer long-form section bodies."""
from backend.core.ingest import _chunk_reference_text
tagged = split_into_section_paragraphs(text, valid_section_ids)
if not tagged:
return [
Chunk(
text=_clean_reference_body(t),
tier=TIER_REFERENCE,
is_scrubbed=False,
source_filename=source_filename,
)
for t in _chunk_reference_text(text)
if _clean_reference_body(t).strip()
]
chunks: list[Chunk] = []
for section_id, para_idx, para_text in tagged:
cleaned = _clean_reference_body(para_text)
if not cleaned:
continue
for piece in _chunk_section_body(cleaned):
chunks.append(
Chunk(
text=piece,
section_id=section_id,
tier=TIER_REFERENCE,
is_scrubbed=False,
source_filename=source_filename,
chunk_id=f"{source_filename}:{section_id}:p{para_idx}",
paragraph_index=para_idx,
)
)
return chunks
|