File size: 15,820 Bytes
2874635
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
#!/usr/bin/env python3
"""
llm.py — GPT transliteration for the HF Space.

Port of app/services/transliteration_services.py with the FastAPI stack
removed: pydantic-settings -> env vars (HF Space secrets are env vars, so
this is strictly simpler than shipping a .env), pydantic BaseModel ->
dataclasses (the three models carry no validators, so this is lossless).

SYSTEM_PROMPT, segment_into_chunks and build_egyptologist_prompt are copied
VERBATIM — including the h() blank->'unknown' normalization and the
matched-vs-[UNRESOLVED cartouche prompt split. Keep the '[UNRESOLVED' prefix
wording in sync with the backend if either side changes.

Contract preserved: this module NEVER raises on an LLM failure. No API key or
any OpenAI error returns a TransliterationOut with `error` set.
"""

from __future__ import annotations

import json
import logging
import os
from dataclasses import dataclass, field
from typing import Optional

from openai import OpenAI

logger = logging.getLogger('sphinxeyes.transliteration')

OPENAI_API_KEY       = os.getenv('OPENAI_API_KEY', '')
OPENAI_MODEL         = os.getenv('OPENAI_MODEL', 'gpt-4o')
TRANSLIT_MAX_SIGNS   = int(os.getenv('TRANSLIT_MAX_SIGNS', '20'))
TRANSLIT_TEMPERATURE = float(os.getenv('TRANSLIT_TEMPERATURE', '0.1'))


# Controlled vocabularies — MUST match app/schemas/transliterations.py.
# 'unknown' is always first = the default.
PERIODS = ['unknown', 'old_kingdom', 'first_intermediate', 'middle_kingdom',
           'second_intermediate', 'new_kingdom', 'third_intermediate',
           'late_period', 'ptolemaic', 'roman']
TEXT_TYPES = ['unknown', 'stela', 'temple_wall', 'tomb_wall', 'papyrus',
              'sarcophagus', 'obelisk', 'statue', 'offering_table',
              'pyramid_texts']
SUPPORTS = ['unknown', 'limestone', 'sandstone', 'granite', 'papyrus', 'wood',
            'plaster', 'faience', 'metal']
LOCATION_TYPES = ['unknown', 'pyramid', 'temple', 'tomb', 'museum', 'open_site']


@dataclass
class TextContext:
    period        : str = 'unknown'
    text_type     : str = 'unknown'
    support       : str = 'unknown'
    location_type : str = 'unknown'
    site          : str = 'unknown'
    dynasty       : str = 'unknown'
    kings_reign   : str = 'unknown'


@dataclass
class ChunkTransliteration:
    chunk_index      : int
    gardiner_codes   : list[str]
    transliteration  : str
    english_gloss    : str
    linguistic_notes : str = ''
    confidence       : str = 'LOW'
    period_note      : str = ''
    is_cartouche     : bool = False


@dataclass
class TransliterationOut:
    chunks               : list[ChunkTransliteration] = field(default_factory=list)
    full_transliteration : str = ''
    full_translation     : str = ''
    model                : str = OPENAI_MODEL
    n_chunks             : int = 0
    error                : Optional[str] = None


# Chunking (port of segement_into_chunck.segment_into_chunks)

def segment_into_chunks(
    codes           : list[str],
    confidences     : list[float],
    boundary_hints  : list[int],
    max_signs       : int = 20,
) -> list[list[tuple[str, float]]]:
    """
    Split the ordered (code, conf) sequence into chunks of at most
    `max_signs`, cutting preferentially at line boundaries so a chunk
    never mixes signs from different physical lines.
    """
    pairs = list(zip(codes, confidences))
    bounds = sorted(b for b in boundary_hints if 0 < b <= len(pairs))
    if not bounds or bounds[-1] != len(pairs):
        bounds.append(len(pairs))

    chunks: list[list[tuple[str, float]]] = []
    start = 0
    for b in bounds:
        line = pairs[start:b]
        # split an over-long line at max_signs
        for i in range(0, len(line), max_signs):
            piece = line[i:i + max_signs]
            if piece:
                chunks.append(piece)
        start = b
    return chunks


# Prompt (port of prompt2GPT.build_egyptologist_prompt)

# The role/methodology half of the prompt. Static -> sent as the system
# message (also lets OpenAI cache it across the per-chunk calls).
SYSTEM_PROMPT = """You are an expert Egyptologist and philologist specializing in Middle Egyptian \
(the classical language of Dynasties XI-XVIII, also used ceremonially long after). You read \
hieroglyphic inscriptions from Gardiner sign codes and produce scholarly transliterations and \
English glosses.

HOW THE INPUT WAS PRODUCED (important for judging its reliability):
The Gardiner sequence comes from a computer-vision pipeline, not a human copyist:
1. A YOLO detector recognizes individual signs on the photograph (each has a confidence score).
2. A spatial algorithm reconstructs reading order (quadrat stacking, line breaks).
3. A lexicon-based corrector (Viterbi over a dictionary trie + bigram model) may have already
   substituted some low-confidence detections.
4. Cartouches are matched against a verified royal-name lexicon (Beckerath) — when a royal name
   is given, it is MORE reliable than the raw sign codes around it.
Consequences you must handle:
- Signs may be MISCLASSIFIED as visually similar signs. Frequent confusions of this detector:
  G1 (vulture) <-> G5 (falcon) <-> G39 (duck); N5 (sun disc) <-> N33 (pellet) <-> Aa1 (placenta)
  <-> O49 (town); X1 (bread loaf) <-> X8 (conical loaf); S29 (folded cloth) <-> O34 (door bolt);
  W19 (milk jug) <-> W14 (water jar); M23 (sedge) <-> M22 (rush); D21 (mouth) <-> D4 (eye);
  Y1 <-> Y2 (papyrus rolls); Z1 <-> Z4 (strokes). When a LOW or MED confidence sign yields
  nonsense but one of its confusion partners yields coherent Middle Egyptian, prefer the partner
  and say so in the notes.
- A sign may be MISSING (detector miss) — small phonetic complements and determinatives are the
  usual casualties. You may posit an omitted complement when the reading obviously requires it.
- 'Unknown' tokens are undetected signs: treat them as lacunae, transliterate as [...].
- Reading order is usually right but not guaranteed within a quadrat; minor transpositions
  (e.g. honorific transposition of nTr / nsw / ra) should be restored silently.

METHOD — work like a philologist, not a code mapper:
1. Segment the sign string into words: identify uniliterals, biliterals, triliterals, phonetic
   complements (do not transliterate a complement twice), determinatives (classify, never
   pronounce), and logograms.
2. Look for the high-frequency formulae of monumental texts and let them anchor the reading:
   htp-di-nsw (offering formula), sA ra (son of Ra), nb tAwy (lord of the Two Lands),
   nTr nfr (the good god), di anx (given life), mAa-xrw (true of voice), anx wDA snb,
   nswt-bity (dual king), Dt / nHH (forever), epithets of deities and royal titulary.
3. If a royal cartouche is identified, use it as the chronological and thematic anchor: titles
   and epithets adjacent to a cartouche almost always belong to the standard titulary sequence.
4. Use the archaeological context: a temple wall favours royal/divine formulae; a stela favours
   the offering formula and filiation (X sA Y, mAat-xrw); pyramid texts favour Old Kingdom
   spellings; a papyrus may be literary or administrative.
5. Commit to ONE most-probable reading. Note real alternatives briefly instead of hedging.

OUTPUT — JSON only, no preamble, exactly these keys:
{
  "transliteration": "Unified Leiden conventions (aA not aleph-glyph fallback; use . for suffixes, = for clitics, [...] for lacunae, ( ) for restored signs)",
  "english_gloss": "~ one plain-English sentence, functional gloss for non-specialists",
  "linguistic_notes": "max 2 sentences: key ambiguity, any confusion-pair substitution you made, notable grammar",
  "confidence": "HIGH|MEDIUM|LOW",
  "period_note": "one short remark tying the reading to the stated period/reign, or '' if context was unknown"
}
Answer in ENGLISH only."""


def build_egyptologist_prompt(
    codes            : list[str],
    confidences      : list[float],
    ctx              : TextContext,
    direction        : str,
    layout           : str,
    cartouche_names  : list[str],
    previous_context : str | None,
    chunk_info       : str,
) -> str:
    """Build the per-chunk USER message (the system message is static)."""
    signs = ' — '.join(
        f'{c}({"HIGH" if s >= 0.80 else "MED" if s >= 0.50 else "LOW"}:{s:.2f})'
        for c, s in zip(codes, confidences)
    )

    def h(v: str) -> str:                     # human-readable context value
        # blank/whitespace (client sent '' after the user erased a field)
        # must read as 'unknown', not an empty line the LLM could misread
        return (v or '').strip().replace('_', ' ') or 'unknown'

    # Matched names are authoritative; UNRESOLVED entries (match refused)
    # carry raw interior signs only — never present those as verified.
    matched    = [n for n in cartouche_names if not n.startswith('[UNRESOLVED')]
    unresolved = [n for n in cartouche_names if n.startswith('[UNRESOLVED')]
    parts = []
    if matched:
        parts.append(
            'Royal cartouches identified in this scene (lexicon-verified — '
            'treat as authoritative, more reliable than raw sign codes):\n  '
            + '\n  '.join(matched))
    if unresolved:
        parts.append(
            'Cartouches detected but NOT resolved to a known royal name — '
            'read their raw interior signs yourself (a royal name or epithet '
            'is likely; do not invent a specific king):\n  '
            + '\n  '.join(unresolved))
    cartouche_block = '\n- '.join(parts) if parts else 'Contains cartouche: no'
    previous_block = (
        f'\nPREVIOUS SEGMENTS of the same inscription (already transliterated '
        f'— keep names, epithets and topic consistent with them):\n'
        f'{previous_context}\n' if previous_context else ''
    )

    return f"""DETECTED SIGN SEQUENCE ({chunk_info}; each segment is one physical line of text; Gardiner codes with detector confidence):
{signs}

ARCHAEOLOGICAL CONTEXT (fields marked 'unknown' were not supplied by the user — do not invent them, but exploit every field that IS given):
- Period: {h(ctx.period)}
- Dynasty: {h(ctx.dynasty)}
- King's reign: {h(ctx.kings_reign)}
- Text type: {h(ctx.text_type)}
- Physical support: {h(ctx.support)}
- Location type: {h(ctx.location_type)}
- Site: {h(ctx.site)}
- Reading direction: {direction}
- Layout: {layout}
- {cartouche_block}
{previous_block}
Transliterate and gloss this segment. JSON only."""


class TransliterationService:
    """GPT-4o transliteration. Degrades gracefully — never raises."""

    def __init__(self) -> None:
        self._client = OpenAI(api_key=OPENAI_API_KEY) if OPENAI_API_KEY else None

    @property
    def enabled(self) -> bool:
        return self._client is not None

    def transliterate(
        self,
        raw       : dict,          # SphinxPipeline.run() output
        ctx       : TextContext,
    ) -> TransliterationOut:
        outer = raw['outer']
        codes = outer['correction']['flat_corrected_seq']
        confs = [slot[0][1] if slot else 0.0 for slot in outer['slots']]
        # corrected seq and slots are index-aligned; guard anyway
        if len(confs) != len(codes):
            confs = (confs + [0.0] * len(codes))[:len(codes)]

        cartouche_names = []
        for c in raw['cartouches']:
            if c.get('translit'):
                cartouche_names.append(
                    f"{c['translit']}{c['english']} (interior: {' '.join(c['spelling'] or [])})"
                )
            elif c.get('n_members', 0) > 0:
                raw_codes = ' '.join(
                    slot[0][0] for slot in c.get('slots', []) if slot
                )
                if raw_codes:
                    cartouche_names.append(
                        f"[UNRESOLVED cartouche — raw signs detected but no confident "
                        f"royal-name match: {raw_codes}]"
                    )
            else:
                cartouche_names.append(f"[UNRESOLVED cartouche — no signs detected]")

        return self.transliterate_sequence(
            codes, confs, outer['boundary_hints'], cartouche_names,
            direction=raw['direction'], layout=raw['layout'], ctx=ctx,
        )

    def transliterate_sequence(
        self,
        codes           : list[str],
        confidences     : list[float],
        boundary_hints  : list[int],
        cartouche_names : list[str],
        *,
        direction       : str,
        layout          : str,
        ctx             : TextContext,
    ) -> TransliterationOut:
        if not self.enabled:
            return TransliterationOut(
                chunks=[], full_transliteration='', full_translation='',
                model=OPENAI_MODEL, n_chunks=0,
                error='OPENAI_API_KEY not configured — transliteration disabled.',
            )

        chunks = segment_into_chunks(
            codes, confidences, boundary_hints, max_signs=TRANSLIT_MAX_SIGNS,
        )

        results: list[ChunkTransliteration] = []
        for i, chunk in enumerate(chunks):
            c_codes = [c for c, _ in chunk]
            c_confs = [s for _, s in chunk]
            previous = ' | '.join(r.transliteration for r in results) or None

            prompt = build_egyptologist_prompt(
                c_codes, c_confs, ctx,
                direction=direction, layout=layout,
                cartouche_names=cartouche_names,
                previous_context=previous,
                chunk_info=f'Segment {i + 1} of {len(chunks)}',
            )
            try:
                response = self._client.chat.completions.create(
                    model           = OPENAI_MODEL,
                    messages        = [
                        {'role': 'system', 'content': SYSTEM_PROMPT},
                        {'role': 'user',   'content': prompt},
                    ],
                    temperature     = TRANSLIT_TEMPERATURE,
                    response_format = {'type': 'json_object'},
                )
                parsed = json.loads(response.choices[0].message.content)
            except Exception as e:
                logger.exception(f'LLM transliteration failed on chunk {i}')
                return TransliterationOut(
                    chunks=results,
                    full_transliteration=' | '.join(r.transliteration for r in results),
                    full_translation=' '.join(r.english_gloss for r in results),
                    model=OPENAI_MODEL, n_chunks=len(chunks),
                    error=f'LLM call failed on segment {i + 1}/{len(chunks)}: {e}',
                )

            results.append(ChunkTransliteration(
                chunk_index      = i,
                gardiner_codes   = c_codes,
                transliteration  = str(parsed.get('transliteration', '')),
                english_gloss    = str(parsed.get('english_gloss', '')),
                linguistic_notes = str(parsed.get('linguistic_notes', '')),
                confidence       = parsed.get('confidence', 'LOW')
                                   if parsed.get('confidence') in ('HIGH', 'MEDIUM', 'LOW')
                                   else 'LOW',
                period_note      = str(parsed.get('period_note', '')),
            ))

        return TransliterationOut(
            chunks               = results,
            full_transliteration = ' | '.join(r.transliteration for r in results),
            full_translation     = ' '.join(r.english_gloss for r in results),
            model                = OPENAI_MODEL,
            n_chunks             = len(results),
        )


# One client for the process lifetime — the FastAPI version built a new
# OpenAI client per request, which is wrong for a long-lived Gradio app.
SERVICE = TransliterationService()