Spaces:
Runtime error
Runtime error
| """Split uploaded past-report text into section-tagged paragraphs for REFERENCE RAG.""" | |
| from __future__ import annotations | |
| import re | |
| from backend.config import settings | |
| from backend.core.rag_store import TIER_REFERENCE, Chunk | |
| _RICS_HEADING_LINE = re.compile( | |
| r""" | |
| (?m)^\s* | |
| (?:(?:section|part|element|item)\s+)? # optional descriptive label before the code | |
| (?P<code>[A-N]\d{1,2}) | |
| \b | |
| (?:[\s:.\-\u2013\u2014]+[^\n]{0,120})? # optional separator + title on the same line | |
| \s*$ | |
| """, | |
| re.IGNORECASE | re.VERBOSE, | |
| ) | |
| # Page-header/footer furniture that bleeds into PDF text extraction. Stripped from | |
| # section bodies so it never becomes the stored baseline (e.g. a section whose only | |
| # captured body was "Page 24RICS Home Survey - Level 3"). Bare standalone numbers are | |
| # deliberately NOT stripped here — a lone digit may be a rating value, not a page number. | |
| _PAGE_HEADER_FURNITURE_RE = re.compile( | |
| r"(?im)^[ \t]*(?:page[ \t]*\d+[ \t]*)?" | |
| r"rics[ \t]+home[ \t]+survey[ \t]*[-\u2013\u2014][ \t]*level[ \t]+\d.*$" | |
| ) | |
| _PAGE_LABEL_RE = re.compile(r"(?im)^[ \t]*page[ \t]*\d+[ \t]*$") | |
| # Survey-report PDFs embed photo index lines and RICS procedural NOTE blocks inside | |
| # section bodies. These are not mappable baseline prose — they bloat the baseline, | |
| # confuse the weave gate, and cause grounding rollbacks. Stripped at ingest only. | |
| _PHOTO_LINE_RE = re.compile(r"(?im)^[ \t]*Photo\s*[-\u2013]?\s*\d+.*$") | |
| _PHOTO_INLINE_RE = re.compile( | |
| r"(?i)\s*Photo\s*[-\u2013]?\s*\d+\s*(?:[^\n.!?]|(?:\n(?![ \t]*Photo\s*[-\u2013]?\s*\d+)))*?[.!?]?" | |
| ) | |
| _ELEMENT_TABLE_STUB_RE = re.compile( | |
| r"(?im)^[ \t]*Element\s+no\.?\s*Element\s+name\s*$" | |
| ) | |
| # NOTE N: procedural blocks (Building Regs boilerplate etc.) precede real findings. | |
| # Stop at the next line that reads like survey prose, not a caption or header. | |
| _NOTE_BLOCK_RE = re.compile( | |
| r"(?is)" | |
| r"^NOTE\s+\d+\s*:.*?" | |
| r"(?=" | |
| r"\n(?:" | |
| r"The|There|We|It|A|An|In|Condition|Main|Our|Your|This|These|Some|No|Where|When|" | |
| r"Whilst|While|During|Following|Upon|After|Before|Internal|External|Ground|Roof|" | |
| r"Wall|Floor|Ceiling|Chimney|Window|Door|Gutter|Rainwater|Damp|Mould|Wood|Timber|" | |
| r"Brick|Stone|Slate|Tile|Pipe|Drain|Electric|Gas|Water|Heating|Boiler|Insulation|" | |
| r"Ventilation|Asbestos|Structural|Surface|Visible|Noted|Observed|Evidence|Appears|" | |
| r"Found|Seen|Installed|Located|Constructed|Finished|Painted|Plastered|Rendered|" | |
| r"Pointed|Bedded|Covering|Structure|Material|Building|Property|Survey|Client|" | |
| r"Surveyor|Defective|Further|Regular|Although|However|Generally|Typically|" | |
| r"Inspection|Inspected|Examined|Checked|Tested|Operated|Opened|Accessed" | |
| r")\b|\Z)" | |
| ) | |
| # Collapse runs of blank lines left after stripping. | |
| _MULTI_BLANK_RE = re.compile(r"\n{3,}") | |
| def _strip_page_furniture(text: str) -> str: | |
| """Remove repeated page header/footer lines bled in by PDF extraction.""" | |
| cleaned = _PAGE_HEADER_FURNITURE_RE.sub("", text) | |
| cleaned = _PAGE_LABEL_RE.sub("", cleaned) | |
| return cleaned | |
| def _clean_reference_body(text: str) -> str: | |
| """Remove non-prose survey-report artifacts from a REFERENCE section body.""" | |
| cleaned = _strip_page_furniture(text or "") | |
| cleaned = _NOTE_BLOCK_RE.sub("", cleaned) | |
| cleaned = _ELEMENT_TABLE_STUB_RE.sub("", cleaned) | |
| cleaned = _PHOTO_LINE_RE.sub("", cleaned) | |
| # Trailing / inline photo runs (often concatenated without newlines at section tail). | |
| prev = None | |
| while prev != cleaned: | |
| prev = cleaned | |
| cleaned = _PHOTO_INLINE_RE.sub("", cleaned) | |
| cleaned = _MULTI_BLANK_RE.sub("\n\n", cleaned) | |
| return cleaned.strip() | |
| def _alpha_len(text: str) -> int: | |
| """Count alphabetic characters — a body-richness proxy for de-duplication.""" | |
| return sum(1 for ch in text if ch.isalpha()) | |
| def _normalise_code(raw: str) -> str: | |
| return re.sub(r"\s+", "", raw.strip().upper()) | |
| def _is_valid_code(code: str, valid_section_ids: set[str] | None) -> bool: | |
| if valid_section_ids is None: | |
| return True | |
| return code in {s.upper() for s in valid_section_ids} | |
| def split_into_section_paragraphs( | |
| text: str, | |
| valid_section_ids: set[str] | None = None, | |
| ) -> list[tuple[str, int, str]]: | |
| """Return ``(section_id, paragraph_index, paragraph_text)`` tuples.""" | |
| matches = list(_RICS_HEADING_LINE.finditer(text)) | |
| if not matches: | |
| return [] | |
| # A RICS code typically appears twice in one report: once as a stub row in the | |
| # ratings summary table (body ≈ "Element no. Element name") and once as the real | |
| # section with prose. Keep only the richest-prose body per code so the table stub | |
| # never shadows the genuine content (both previously shared one chunk_id). | |
| best_body: dict[str, str] = {} | |
| for i, match in enumerate(matches): | |
| code = _normalise_code(match.group("code")) | |
| if not _is_valid_code(code, valid_section_ids): | |
| continue | |
| body_start = match.end() | |
| end = matches[i + 1].start() if i + 1 < len(matches) else len(text) | |
| body = _clean_reference_body(text[body_start:end]) | |
| if not body: | |
| continue | |
| if _alpha_len(body) > _alpha_len(best_body.get(code, "")): | |
| best_body[code] = body | |
| if not best_body: | |
| return [] | |
| out: list[tuple[str, int, str]] = [] | |
| max_chars = settings.reference_paragraph_max_chars | |
| for section_id, body in best_body.items(): | |
| if len(body) <= max_chars: | |
| out.append((section_id, 1, body)) | |
| continue | |
| paragraphs = [p.strip() for p in re.split(r"\n\s*\n", body) if p.strip()] | |
| for idx, para in enumerate(paragraphs, start=1): | |
| out.append((section_id, idx, para)) | |
| return out | |
| def _chunk_section_body(para_text: str) -> list[str]: | |
| """Chunk one section body; keep whole block when within REFERENCE max size.""" | |
| from backend.core.ingest import _chunk_reference_text | |
| text = (para_text or "").strip() | |
| if not text: | |
| return [] | |
| if len(text) <= settings.reference_paragraph_max_chars: | |
| return [text] | |
| return _chunk_reference_text(text) | |
| def build_reference_chunks( | |
| text: str, | |
| *, | |
| source_filename: str, | |
| valid_section_ids: set[str] | None = None, | |
| ) -> list[Chunk]: | |
| """Build REFERENCE chunks with section metadata; prefer long-form section bodies.""" | |
| from backend.core.ingest import _chunk_reference_text | |
| tagged = split_into_section_paragraphs(text, valid_section_ids) | |
| if not tagged: | |
| return [ | |
| Chunk( | |
| text=_clean_reference_body(t), | |
| tier=TIER_REFERENCE, | |
| is_scrubbed=False, | |
| source_filename=source_filename, | |
| ) | |
| for t in _chunk_reference_text(text) | |
| if _clean_reference_body(t).strip() | |
| ] | |
| chunks: list[Chunk] = [] | |
| for section_id, para_idx, para_text in tagged: | |
| cleaned = _clean_reference_body(para_text) | |
| if not cleaned: | |
| continue | |
| for piece in _chunk_section_body(cleaned): | |
| chunks.append( | |
| Chunk( | |
| text=piece, | |
| section_id=section_id, | |
| tier=TIER_REFERENCE, | |
| is_scrubbed=False, | |
| source_filename=source_filename, | |
| chunk_id=f"{source_filename}:{section_id}:p{para_idx}", | |
| paragraph_index=para_idx, | |
| ) | |
| ) | |
| return chunks | |