File size: 8,136 Bytes
a671976
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
"""Schema-ordered DOCX assembly.

Sections with content are emitted in schema order. Empty sections are omitted.
Rating annotations require schema rating system + section has_rating_field + value.
"""

from __future__ import annotations

import io
import logging
from pathlib import Path

from backend.config import settings
from backend.domain.source_attribution import format_reference_source
from backend.models.report import ReportResult
from backend.models.schema import TemplateSchema
from backend.pii import scrubber as pii_scrubber
from backend.storage.photo_layout import SectionPhotoLayout, get_section_photo_layout

logger = logging.getLogger(__name__)

_DEFAULT_RATING_COLOURS = {"1": "00B050", "2": "ED7D31", "3": "C00000", "NI": "808080"}


def _apply_minimal_styles(doc) -> None:
    from docx.shared import Pt

    style = doc.styles["Normal"]
    style.font.name = "Calibri"
    style.font.size = Pt(10)


def _rating_label(schema: TemplateSchema, value: str) -> str:
    fmt = schema.rating_system.format_template
    if fmt and "[VALUE]" in fmt:
        return fmt.replace("[VALUE]", value)
    name = schema.rating_system.name
    if name:
        return f"{name}: {value}"
    return str(value)


def _rating_colour(schema: TemplateSchema, value: str) -> str | None:
    for rv in schema.rating_system.values:
        if rv.value == value and rv.colour:
            return rv.colour.lstrip("#").upper()
    return _DEFAULT_RATING_COLOURS.get(str(value))


def _add_body_paragraph(doc, body: str):
    pii_scrubber.assert_no_pii(body, context="DOCX paragraph")
    return doc.add_paragraph(body)


def _add_reference_attribution(doc, gen) -> None:
    """Footnote listing uploaded past-report sources for this section.

    Gated on ``rag_sources``: only sections attributed during live generation
    carry these display strings. Reports rebuilt from provenance-only payloads
    intentionally omit the internal source footnote.
    """
    if not gen.rag_sources:
        return
    from docx.shared import Pt

    if gen.reference_sources:
        labels = [format_reference_source(src) for src in gen.reference_sources]
    else:
        labels = list(gen.rag_sources)
    text = "Source: " + "; ".join(labels) + "."
    pii_scrubber.assert_no_pii(text, context="DOCX source attribution")
    p = doc.add_paragraph()
    run = p.add_run(text)
    run.italic = True
    run.font.size = Pt(8.5)
    p.paragraph_format.space_before = Pt(4)
    p.paragraph_format.space_after = Pt(6)


def _embed_section_photos(
    doc,
    paths: list[str | Path],
    layout: SectionPhotoLayout,
) -> None:
    """Insert section photos using placement learned from past reports."""
    from docx.shared import Inches, Pt, RGBColor

    clean = [Path(p) for p in paths if p and Path(p).is_file()]
    if not clean:
        return

    cap = doc.add_paragraph()
    cap_r = cap.add_run(layout.caption or "Photographs")
    cap_r.font.bold = True
    cap_r.font.size = Pt(9.5)
    cap_r.font.color.rgb = RGBColor(0x1B, 0x3A, 0x5C)
    cap.paragraph_format.space_after = Pt(4)

    limit = settings.max_section_photos_per_section
    for photo_path in clean[:limit]:
        try:
            doc.add_picture(str(photo_path), width=Inches(layout.width_inches))
        except Exception as exc:  # noqa: BLE001
            logger.debug("Skipping corrupt photo %s: %s", photo_path, exc)
            continue

    spacer = doc.add_paragraph()
    spacer.paragraph_format.space_after = Pt(layout.spacing_after_pt)


def build_docx(
    result: ReportResult,
    schema: TemplateSchema,
    *,
    title: str = "Survey Report",
    include_footer: bool = False,
    template_docx_path: str | None = None,
    section_photo_paths: dict[str, list[str]] | None = None,
) -> bytes:
    from docx import Document
    from docx.shared import RGBColor

    full_text = (
        "\n".join(s.text for s in result.sections) + "\n" + result.unassigned_text
    )
    pii_scrubber.assert_no_pii(full_text, context="DOCX output")

    path = template_docx_path or settings.template_docx_path
    branded = Path(path) if path else None
    if branded and branded.is_file():
        doc = Document(str(branded))
    else:
        doc = Document()
        _apply_minimal_styles(doc)
        doc.add_heading(title, level=0)
        if schema.report_type:
            doc.add_paragraph(schema.report_type)
        if result.property_type:
            doc.add_paragraph(f"Property type: {result.property_type}")
        if result.tenure:
            doc.add_paragraph(f"Tenure: {result.tenure}")
        doc.add_paragraph(
            "[Address, client name, inspection date and surveyor details "
            "to be completed by the surveyor before issue.]"
        ).italic = True
        doc.add_page_break()

    rating_enabled = schema.rating_system.detected
    by_id = {s.section_id: s for s in result.sections}
    photos_by_section = section_photo_paths or {}

    for sec in schema.ordered_sections():
        gen = by_id.get(sec.id)
        if gen is None or gen.status == "empty":
            continue

        heading_text = f"{sec.id}  {sec.title}" if sec.id else sec.title
        heading = doc.add_heading(heading_text, level=min(max(sec.level, 1), 4))

        if rating_enabled and sec.has_rating_field and gen.rating_value:
            label = _rating_label(schema, str(gen.rating_value))
            pii_scrubber.assert_no_pii(label, context="DOCX rating")
            if heading.runs:
                run = heading.runs[0]
                run.text = f"{run.text}  {label}"
                run.bold = True
                hexcol = _rating_colour(schema, str(gen.rating_value))
                if hexcol:
                    run.font.color.rgb = RGBColor.from_string(hexcol)
            else:
                run = heading.add_run(f"  {label}")
                run.bold = True
                hexcol = _rating_colour(schema, str(gen.rating_value))
                if hexcol:
                    run.font.color.rgb = RGBColor.from_string(hexcol)

        layout = get_section_photo_layout(result.tenant_id, sec.id)
        section_paths = (
            photos_by_section.get(sec.id) or photos_by_section.get(sec.id.upper()) or []
        )
        if layout.placement == "after_heading" and section_paths:
            _embed_section_photos(doc, section_paths, layout)

        if gen.status == "NO_RAG_MATCH":
            p = doc.add_paragraph()
            text = gen.text or (
                f'[No template paragraph matched for "{sec.title}". '
                f"Manual entry required.]"
            )
            pii_scrubber.assert_no_pii(text, context="DOCX paragraph")
            p.add_run(text).italic = True
        else:
            body = (gen.text or "").strip()
            if body:
                _add_body_paragraph(doc, body)

        if layout.placement == "after_body" and section_paths:
            _embed_section_photos(doc, section_paths, layout)

        _add_reference_attribution(doc, gen)

    unassigned = by_id.get("UNASSIGNED")
    if unassigned or result.unassigned_text.strip():
        doc.add_page_break()
        doc.add_heading("Unassigned Observations — Manual Review Required", level=2)
        doc.add_paragraph(
            "The following observations from the surveyor's notes could not be "
            "automatically assigned to any template section. Please review and draft manually."
        )
        obs = (
            unassigned.unmatched_observations
            if unassigned and unassigned.unmatched_observations
            else [result.unassigned_text]
        )
        for item in obs:
            if item.strip():
                bullet = f"• {item}"
                pii_scrubber.assert_no_pii(bullet, context="DOCX paragraph")
                doc.add_paragraph(bullet, style="List Bullet")

    show_footer = include_footer or settings.ai_transparency_footer_enabled
    if show_footer:
        doc.add_paragraph(
            "Generated from surveyor notes mapped onto the firm's master template."
        )

    buf = io.BytesIO()
    doc.save(buf)
    return buf.getvalue()