File size: 6,760 Bytes
5952424
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
"""Documentary papyri (DDbDP via papyri_clean.jsonl) loader — segments, splits, planes.

Mirrors iphi.py's interface exactly (same record dict keys, phi_id := TM number) so the
finetune/eval stack is reusable. Split rule (same as the 10-fold split's fixed
documentary rule): TM last digit 3 -> test, 4 -> val, else train — fold-0's pretraining
excluded TM-3/4 papyri and excised their literary echoes, so both eval splits are
genuinely unseen by the fold-0 torso.

Lacunae: '-' runs (lost chars of known length) AND U+2026 '…' (unknown length); each
record is split at either into segments of continuous known text.
"""
from __future__ import annotations

import json, os, re, sys
from pathlib import Path

import numpy as np

from data.normalize import ALPHABET, Stats, normalize_record  # noqa: F401
try:                    # imported as a package module (python -m insc.data.papyri)
    from insc.data.meta_vocab import UNK_REGION, UNK_CENTURY
except ImportError:     # imported with insc/data on sys.path (how the trainers load it)
    from meta_vocab import UNK_REGION, UNK_CENTURY

JSONL = Path(os.path.expandvars("$AGD_DATA/data/papyri_clean.jsonl"))
GAP_RE = re.compile(r"-+|…+")
_DASH_RE = re.compile(r"-+")
_ELLIPSIS_RE = re.compile(r"…+")

MASK, UNK_BND, UNK_DIA, UNK_PUNCT = 24, 3, 48, 6
ELLIPSIS_STAND_IN_MIN, ELLIPSIS_STAND_IN_MAX = 20, 30  # '…' has no known length -- per-run
                                                        # random stand-in width, never supervised
                                                        # regardless (same as a real '-' run)


def split_of(tm):
    s = str(tm).strip()
    if not s or not s[-1].isdigit():
        return "train"
    test_d = os.environ.get("INSC_TEST_DIGIT", "3")
    val_d = os.environ.get("INSC_VAL_DIGIT", "4")
    return {val_d: "val", test_d: "test"}.get(s[-1], "train")


def text_to_full_planes(text, rng, stats=None):
    """Raw text (with '-' AND '…' damage runs) -> (chars, boundary, dia, punct,
    is_real_lacuna) arrays, WHOLE text, damage kept in place as MASK+unknown positions
    rather than split away. Mirrors iphi.py's text_to_full_planes() exactly, generalized
    to papyri's second lacuna convention: '-' runs have a KNOWN length (count the dashes,
    stonecutter/scribe spacing); '…' runs have an UNKNOWN length -- there's no real count
    to use, so each '…' occurrence gets its own random stand-in width in
    [ELLIPSIS_STAND_IN_MIN, ELLIPSIS_STAND_IN_MAX]. Either way the run is marked
    is_real_lacuna=True and is NEVER a supervision target downstream (no ground truth
    exists for either convention, unlike a synthetically-masked span over known text)."""
    st = stats if stats is not None else Stats()
    parts = GAP_RE.split(text)
    gaps = GAP_RE.findall(text)
    chars_l, bnd_l, dia_l, cap_l, punct_l, real_l = [], [], [], [], [], []
    for i, seg in enumerate(parts):
        if seg.strip():
            nr = normalize_record(seg, st, with_punct=True)
            if nr is None:
                return None
            c, b, d, cp, p = nr
            chars_l.append(c); bnd_l.append(b); dia_l.append(d); cap_l.append(cp); punct_l.append(p)
            real_l.append(np.zeros(len(c), dtype=bool))
        if i < len(gaps):
            g = gaps[i]
            n = len(g) if g[0] == "-" else int(rng.integers(ELLIPSIS_STAND_IN_MIN,
                                                            ELLIPSIS_STAND_IN_MAX + 1))
            chars_l.append(np.full(n, MASK, np.int64))
            bnd_l.append(np.full(n, UNK_BND, np.int64))
            dia_l.append(np.full(n, UNK_DIA, np.int64))
            cap_l.append(np.zeros(n, np.int64))
            punct_l.append(np.full(n, UNK_PUNCT, np.int64))
            real_l.append(np.ones(n, dtype=bool))
    if not chars_l:
        return None
    return (np.concatenate(chars_l), np.concatenate(bnd_l), np.concatenate(dia_l),
            np.concatenate(cap_l), np.concatenate(punct_l), np.concatenate(real_l))


def load_whole_full(split=None, min_len=32, max_len=None, field="text", max_records=None,
                    seed=0):
    """Whole papyri, FULL planes (chars/boundary/dia/punct) + is_real_lacuna, damage kept
    in place as MASK+unknown positions instead of split away -- mirrors iphi.py's
    load_whole_full(). See text_to_full_planes() for the '-' vs '…' handling."""
    out = []
    st = Stats()
    rng = np.random.default_rng(seed)
    for line in JSONL.open(encoding="utf-8"):
        r = json.loads(line)
        sp = split_of(r.get("TM"))
        if split and sp != split:
            continue
        text = r.get(field) or ""
        if not text:
            continue
        planes = text_to_full_planes(text, rng, st)
        if planes is None:
            continue
        chars, boundary, dia, cap, punct, is_real_lacuna = planes
        if len(chars) < min_len or (max_len and len(chars) > max_len):
            continue
        out.append(dict(chars=chars, boundary=boundary, dia=dia, cap=cap, punct=punct,
                        is_real_lacuna=is_real_lacuna,
                        phi_id=r.get("TM"), seg=0, split=sp,
                        region=None, tpq=None, taq=None,
                        region_id=UNK_REGION, century_id=UNK_CENTURY))
        if max_records and len(out) >= max_records:
            return out
    return out


def load(split=None, min_len=32, field="text", max_records=None):
    """Yield dicts: chars/boundary/dia/cap/punct planes + phi_id (=TM)/seg/split.
    One dict per continuous SEGMENT (split at '-' runs and '…')."""
    out = []
    st = Stats()
    for line in JSONL.open(encoding="utf-8"):
        r = json.loads(line)
        sp = split_of(r.get("TM"))
        if split and sp != split:
            continue
        text = r.get(field) or ""
        for seg_i, seg in enumerate(GAP_RE.split(text)):
            if len(seg.strip()) < min_len:
                continue
            nr = normalize_record(seg, st, with_punct=True)
            if nr is None or len(nr[0]) < min_len:
                continue
            chars, boundary, dia, cap, punct = nr
            out.append(dict(
                chars=chars, boundary=boundary, dia=dia, cap=cap, punct=punct,
                phi_id=r.get("TM"), seg=seg_i, split=sp,
                region=None, tpq=None, taq=None,
                # papyri_clean.jsonl carries no date/place metadata (TM/file/text only) --
                # always UNK. Papyrus records therefore ride the region/century embedding
                # at its "unknown" row; only I.PHI (iphi.py) records are actually primed.
                region_id=UNK_REGION, century_id=UNK_CENTURY))
            if max_records and len(out) >= max_records:
                return out
    return out