File size: 5,381 Bytes
75b4f2e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
"""
scripts/convert_probes.py

Convert probe CSV/XLSX to canonical JSONL for the G-MASS pipeline.
"""
from __future__ import annotations

import argparse
import json
import unicodedata
from pathlib import Path
from typing import Dict, List

try:
    import pandas as pd
except Exception:
    pd = None

CANONICAL_FIELDS = [
    "probe_id",
    "disease_domain",
    "failure_category",
    "english_prompt",
    "twi_prompt",
    "prompt_twi_validated",
    "ghanaian_en_prompt",
    "translation_status",
]

COMMON_COLUMN_MAP = {
    "id": "probe_id",
    "probe": "probe_id",
    "prompt": "english_prompt",
    "english_prompt": "english_prompt",
    "english prompt": "english_prompt",
    "twi_prompt": "twi_prompt",
    "twi prompt": "twi_prompt",
    "prompt_twi_validated": "prompt_twi_validated",
    "final_approved_twi": "twi_prompt",
    "ghanaian_en_prompt": "ghanaian_en_prompt",
    "final_approved_ghanaian_english": "ghanaian_en_prompt",
    "failure_category": "failure_category",
    "disease_domain": "disease_domain",
}


def normalize_text(s: object) -> str:
    if s is None:
        return ""
    if not isinstance(s, str):
        s = str(s)
    s = s.replace("\x00", "")
    s = s.strip()
    s = unicodedata.normalize("NFKC", s)
    return s


def map_columns(columns: List[str]) -> Dict[str, str]:
    col_map: Dict[str, str] = {}
    for col in columns:
        low = str(col).lower()
        if low in COMMON_COLUMN_MAP:
            col_map[col] = COMMON_COLUMN_MAP[low]
            continue
        if "probe" in low and "id" in low:
            col_map[col] = "probe_id"
        elif "english" in low and "prompt" in low:
            col_map[col] = "english_prompt"
        elif "twi" in low and "prompt" in low:
            col_map[col] = "twi_prompt"
        elif "ghana" in low or "ghanaian" in low:
            col_map[col] = "ghanaian_en_prompt"
        elif "domain" in low:
            col_map[col] = "disease_domain"
        elif "failure" in low or "category" in low:
            col_map[col] = "failure_category"
    return col_map


def canonicalize_row(row: Dict[str, object], col_map: Dict[str, str]) -> Dict[str, object]:
    out: Dict[str, object] = {}
    for src_col, value in row.items():
        if src_col in col_map:
            out[col_map[src_col]] = normalize_text(value)
    if not out.get("twi_prompt") and out.get("prompt_twi_validated"):
        out["twi_prompt"] = out.get("prompt_twi_validated")
    return out


def dataframe_to_jsonl(df, out_path: Path, strict: bool = False) -> int:
    required = ["probe_id", "english_prompt"]
    col_map = map_columns(list(df.columns))
    written = 0
    with out_path.open("w", encoding="utf-8") as out:
        for i, row in df.iterrows():
            row_dict = {col: row[col] for col in df.columns}
            rec = canonicalize_row(row_dict, col_map)
            missing = [r for r in required if not rec.get(r)]
            if missing:
                msg = f"Row {i+1} missing required fields: {missing}"
                if strict:
                    raise ValueError(msg)
                else:
                    print("WARNING:", msg)
                    continue
            out_rec = {k: rec.get(k) for k in CANONICAL_FIELDS if rec.get(k) is not None}
            out.write(json.dumps(out_rec, ensure_ascii=False) + "\n")
            written += 1
    return written


def load_input(path: Path, sheet: str | None = None):
    suffix = path.suffix.lower()
    if suffix in (".csv", ".tsv"):
        if pd is None:
            raise EnvironmentError("pandas is required to read CSV/TSV files")
        sep = "," if suffix == ".csv" else "\t"
        return pd.read_csv(path, sep=sep, dtype=str, keep_default_na=False)
    if suffix in (".xlsx", ".xls"):
        if pd is None:
            raise EnvironmentError("pandas is required to read Excel files")
        return pd.read_excel(path, sheet_name=sheet or 0, dtype=str)
    if suffix in (".jsonl", ".ndjson"):
        records = []
        with path.open("r", encoding="utf-8") as f:
            for line in f:
                if not line.strip():
                    continue
                records.append(json.loads(line))
        import pandas as _pd
        return _pd.DataFrame(records)
    raise ValueError(f"Unsupported input file type: {path.suffix}")


def main(argv=None) -> int:
    p = argparse.ArgumentParser(description="Convert probe CSV/XLSX to canonical JSONL")
    p.add_argument("input", help="Input file (csv, xlsx, jsonl)")
    p.add_argument("--out", required=True, help="Output JSONL path")
    p.add_argument("--sheet", default=None, help="Excel sheet name or index")
    p.add_argument("--strict", action="store_true", help="Fail on first missing required field")
    args = p.parse_args(argv)

    in_path = Path(args.input)
    out_path = Path(args.out)

    if not in_path.exists():
        print(f"Input file not found: {in_path}")
        return 2

    if pd is None:
        print("ERROR: pandas is required for convert_probes.py. Install with: pip install pandas openpyxl")
        return 3

    df = load_input(in_path, args.sheet)
    df.columns = [str(c).strip() for c in df.columns]

    written = dataframe_to_jsonl(df, out_path, strict=args.strict)
    print(f"Wrote {written} records to {out_path}")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())