File size: 5,716 Bytes
c8365f5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
"""Conversioni tabellari deterministiche per dati CSV espliciti nel goal.

Il modulo interpreta solo CSV allegati oppure richiesti con ``contenuto esatto:``.
Non apre path arbitrari, non esegue istruzioni contenute nel file e non invoca LLM.
"""
from __future__ import annotations

import csv
import io
import json
import re
from dataclasses import dataclass
from typing import Any

_ATTACHMENT_RE = re.compile(
    r"###\s*📎\s*(?P<name>[^\n`]+?\.csv)\s*\([^\n]*\)\s*```\s*(?P<body>[\s\S]*?)```",
    re.IGNORECASE,
)
# Il target può essere espresso come "file chiamato foo.json" oppure come
# "poi crea foo.json". Il gruppo è limitato a nomi semplici, quindi il parser
# non accetta path traversal o istruzioni aggiuntive.
_TARGET_RE = re.compile(
    r"(?:\b(?:chiamat[oa]|nome|denominat[oa]|come)\s+|\b(?:crea|scrivi)\s+)"
    r"['`\"]?(?P<name>[\w.-]+\.json)\b",
    re.IGNORECASE,
)
_CONVERSION_RE = re.compile(
    r"\b(?:converti|trasforma|conversione|convert|transform)\b[\s\S]{0,240}\b(?:csv|json)\b",
    re.IGNORECASE,
)
_INLINE_CSV_RE = re.compile(
    r"\b(?:crea|scrivi)\s+(?P<name>[\w.-]+\.csv)\s+con\s+contenuto\s+esatto\s*:\s*"
    r"(?P<body>[\s\S]*?)(?=\s*\.\s*(?:poi\s+)?(?:crea|scrivi)\s+[\w.-]+\.json\b|\Z)",
    re.IGNORECASE,
)


@dataclass(frozen=True)
class CsvJsonConversion:
    source_name: str
    target_name: str
    content: str
    row_count: int
    source_content: str
    source_is_inline: bool = False


def _coerce_scalar(value: str) -> Any:
    value = value.strip()
    if re.fullmatch(r"-?(?:0|[1-9]\d*)", value):
        return int(value)
    if re.fullmatch(r"-?(?:0|[1-9]\d*)\.\d+", value):
        return float(value)
    return value


def _csv_body(raw_body: str) -> str:
    lines = raw_body.replace("\r\n", "\n").replace("\r", "\n").split("\n")
    while lines and (not lines[0].strip() or lines[0].lstrip().startswith("## Foglio:")):
        lines.pop(0)
    return "\n".join(lines).strip()


def _parse_csv_rows(csv_body: str) -> list[dict[str, Any]] | None:
    """Legge CSV senza tollerare header/colonne ambigue o righe tronche."""
    try:
        reader = csv.DictReader(io.StringIO(csv_body))
        raw_headers = reader.fieldnames
        if not raw_headers:
            return None
        headers = [str(header or "").strip() for header in raw_headers]
        if any(not header for header in headers) or len(set(headers)) != len(headers):
            return None

        rows: list[dict[str, Any]] = []
        for raw_row in reader:
            # DictReader usa None per colonne in eccesso e per celle mancanti.
            if None in raw_row or any(raw_row.get(header) is None for header in raw_headers):
                return None
            row = {
                headers[index]: _coerce_scalar(raw_row[raw_headers[index]] or "")
                for index in range(len(headers))
            }
            rows.append(row)
        return rows
    except (csv.Error, UnicodeError):
        return None


def validate_csv_json_equivalence(csv_content: str, json_content: str) -> tuple[bool, str]:
    """Verifica che il JSON sia l’array esatto dei record CSV normalizzati.

    La verifica è intenzionalmente stretta: stessa cardinalità, stesso ordine,
    stesse chiavi e stessi valori dopo la coercizione deterministica del CSV.
    """
    expected = _parse_csv_rows(_csv_body(csv_content))
    if expected is None:
        return False, "CSV non valido o ambiguo"
    try:
        actual = json.loads(json_content)
    except (TypeError, json.JSONDecodeError):
        return False, "JSON non valido"
    if not isinstance(actual, list):
        return False, "il JSON deve essere un array"
    if any(not isinstance(record, dict) for record in actual):
        return False, "ogni record JSON deve essere un oggetto"
    if actual != expected:
        return False, "i record JSON non corrispondono esattamente al CSV"
    return True, ""


def _build_conversion(source_name: str, target_name: str, raw_body: str, *, source_is_inline: bool) -> CsvJsonConversion | None:
    csv_body = _csv_body(raw_body)
    rows = _parse_csv_rows(csv_body)
    if rows is None:
        return None
    content = json.dumps(rows, ensure_ascii=False, indent=2) + "\n"
    is_valid, _reason = validate_csv_json_equivalence(csv_body, content)
    if not is_valid:
        # Difesa di coerenza interna: una conversione diretta non può dichiararsi
        # riuscita se il proprio serializzatore non supera il medesimo contratto.
        return None
    return CsvJsonConversion(
        source_name=source_name.strip(),
        target_name=target_name.strip(),
        content=content,
        row_count=len(rows),
        source_content=csv_body + "\n",
        source_is_inline=source_is_inline,
    )


def convert_csv_attachment_to_json(goal: str) -> CsvJsonConversion | None:
    """Converte un CSV allegato o esplicitamente incluso nel goal in JSON.

    Il ritorno è ``None`` quando il goal non definisce una conversione tabellare
    completa: il resto del loop conserva quindi il comportamento esistente.
    """
    if not _CONVERSION_RE.search(goal):
        return None

    target = _TARGET_RE.search(goal)
    if not target:
        return None

    inline = _INLINE_CSV_RE.search(goal)
    if inline:
        return _build_conversion(
            inline.group("name"),
            target.group("name"),
            inline.group("body"),
            source_is_inline=True,
        )

    attachment = _ATTACHMENT_RE.search(goal)
    if not attachment:
        return None
    return _build_conversion(
        attachment.group("name"),
        target.group("name"),
        attachment.group("body"),
        source_is_inline=False,
    )