Spaces:
Running on Zero
Running on Zero
File size: 15,820 Bytes
2874635 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 | #!/usr/bin/env python3
"""
llm.py — GPT transliteration for the HF Space.
Port of app/services/transliteration_services.py with the FastAPI stack
removed: pydantic-settings -> env vars (HF Space secrets are env vars, so
this is strictly simpler than shipping a .env), pydantic BaseModel ->
dataclasses (the three models carry no validators, so this is lossless).
SYSTEM_PROMPT, segment_into_chunks and build_egyptologist_prompt are copied
VERBATIM — including the h() blank->'unknown' normalization and the
matched-vs-[UNRESOLVED cartouche prompt split. Keep the '[UNRESOLVED' prefix
wording in sync with the backend if either side changes.
Contract preserved: this module NEVER raises on an LLM failure. No API key or
any OpenAI error returns a TransliterationOut with `error` set.
"""
from __future__ import annotations
import json
import logging
import os
from dataclasses import dataclass, field
from typing import Optional
from openai import OpenAI
logger = logging.getLogger('sphinxeyes.transliteration')
OPENAI_API_KEY = os.getenv('OPENAI_API_KEY', '')
OPENAI_MODEL = os.getenv('OPENAI_MODEL', 'gpt-4o')
TRANSLIT_MAX_SIGNS = int(os.getenv('TRANSLIT_MAX_SIGNS', '20'))
TRANSLIT_TEMPERATURE = float(os.getenv('TRANSLIT_TEMPERATURE', '0.1'))
# Controlled vocabularies — MUST match app/schemas/transliterations.py.
# 'unknown' is always first = the default.
PERIODS = ['unknown', 'old_kingdom', 'first_intermediate', 'middle_kingdom',
'second_intermediate', 'new_kingdom', 'third_intermediate',
'late_period', 'ptolemaic', 'roman']
TEXT_TYPES = ['unknown', 'stela', 'temple_wall', 'tomb_wall', 'papyrus',
'sarcophagus', 'obelisk', 'statue', 'offering_table',
'pyramid_texts']
SUPPORTS = ['unknown', 'limestone', 'sandstone', 'granite', 'papyrus', 'wood',
'plaster', 'faience', 'metal']
LOCATION_TYPES = ['unknown', 'pyramid', 'temple', 'tomb', 'museum', 'open_site']
@dataclass
class TextContext:
period : str = 'unknown'
text_type : str = 'unknown'
support : str = 'unknown'
location_type : str = 'unknown'
site : str = 'unknown'
dynasty : str = 'unknown'
kings_reign : str = 'unknown'
@dataclass
class ChunkTransliteration:
chunk_index : int
gardiner_codes : list[str]
transliteration : str
english_gloss : str
linguistic_notes : str = ''
confidence : str = 'LOW'
period_note : str = ''
is_cartouche : bool = False
@dataclass
class TransliterationOut:
chunks : list[ChunkTransliteration] = field(default_factory=list)
full_transliteration : str = ''
full_translation : str = ''
model : str = OPENAI_MODEL
n_chunks : int = 0
error : Optional[str] = None
# Chunking (port of segement_into_chunck.segment_into_chunks)
def segment_into_chunks(
codes : list[str],
confidences : list[float],
boundary_hints : list[int],
max_signs : int = 20,
) -> list[list[tuple[str, float]]]:
"""
Split the ordered (code, conf) sequence into chunks of at most
`max_signs`, cutting preferentially at line boundaries so a chunk
never mixes signs from different physical lines.
"""
pairs = list(zip(codes, confidences))
bounds = sorted(b for b in boundary_hints if 0 < b <= len(pairs))
if not bounds or bounds[-1] != len(pairs):
bounds.append(len(pairs))
chunks: list[list[tuple[str, float]]] = []
start = 0
for b in bounds:
line = pairs[start:b]
# split an over-long line at max_signs
for i in range(0, len(line), max_signs):
piece = line[i:i + max_signs]
if piece:
chunks.append(piece)
start = b
return chunks
# Prompt (port of prompt2GPT.build_egyptologist_prompt)
# The role/methodology half of the prompt. Static -> sent as the system
# message (also lets OpenAI cache it across the per-chunk calls).
SYSTEM_PROMPT = """You are an expert Egyptologist and philologist specializing in Middle Egyptian \
(the classical language of Dynasties XI-XVIII, also used ceremonially long after). You read \
hieroglyphic inscriptions from Gardiner sign codes and produce scholarly transliterations and \
English glosses.
HOW THE INPUT WAS PRODUCED (important for judging its reliability):
The Gardiner sequence comes from a computer-vision pipeline, not a human copyist:
1. A YOLO detector recognizes individual signs on the photograph (each has a confidence score).
2. A spatial algorithm reconstructs reading order (quadrat stacking, line breaks).
3. A lexicon-based corrector (Viterbi over a dictionary trie + bigram model) may have already
substituted some low-confidence detections.
4. Cartouches are matched against a verified royal-name lexicon (Beckerath) — when a royal name
is given, it is MORE reliable than the raw sign codes around it.
Consequences you must handle:
- Signs may be MISCLASSIFIED as visually similar signs. Frequent confusions of this detector:
G1 (vulture) <-> G5 (falcon) <-> G39 (duck); N5 (sun disc) <-> N33 (pellet) <-> Aa1 (placenta)
<-> O49 (town); X1 (bread loaf) <-> X8 (conical loaf); S29 (folded cloth) <-> O34 (door bolt);
W19 (milk jug) <-> W14 (water jar); M23 (sedge) <-> M22 (rush); D21 (mouth) <-> D4 (eye);
Y1 <-> Y2 (papyrus rolls); Z1 <-> Z4 (strokes). When a LOW or MED confidence sign yields
nonsense but one of its confusion partners yields coherent Middle Egyptian, prefer the partner
and say so in the notes.
- A sign may be MISSING (detector miss) — small phonetic complements and determinatives are the
usual casualties. You may posit an omitted complement when the reading obviously requires it.
- 'Unknown' tokens are undetected signs: treat them as lacunae, transliterate as [...].
- Reading order is usually right but not guaranteed within a quadrat; minor transpositions
(e.g. honorific transposition of nTr / nsw / ra) should be restored silently.
METHOD — work like a philologist, not a code mapper:
1. Segment the sign string into words: identify uniliterals, biliterals, triliterals, phonetic
complements (do not transliterate a complement twice), determinatives (classify, never
pronounce), and logograms.
2. Look for the high-frequency formulae of monumental texts and let them anchor the reading:
htp-di-nsw (offering formula), sA ra (son of Ra), nb tAwy (lord of the Two Lands),
nTr nfr (the good god), di anx (given life), mAa-xrw (true of voice), anx wDA snb,
nswt-bity (dual king), Dt / nHH (forever), epithets of deities and royal titulary.
3. If a royal cartouche is identified, use it as the chronological and thematic anchor: titles
and epithets adjacent to a cartouche almost always belong to the standard titulary sequence.
4. Use the archaeological context: a temple wall favours royal/divine formulae; a stela favours
the offering formula and filiation (X sA Y, mAat-xrw); pyramid texts favour Old Kingdom
spellings; a papyrus may be literary or administrative.
5. Commit to ONE most-probable reading. Note real alternatives briefly instead of hedging.
OUTPUT — JSON only, no preamble, exactly these keys:
{
"transliteration": "Unified Leiden conventions (aA not aleph-glyph fallback; use . for suffixes, = for clitics, [...] for lacunae, ( ) for restored signs)",
"english_gloss": "~ one plain-English sentence, functional gloss for non-specialists",
"linguistic_notes": "max 2 sentences: key ambiguity, any confusion-pair substitution you made, notable grammar",
"confidence": "HIGH|MEDIUM|LOW",
"period_note": "one short remark tying the reading to the stated period/reign, or '' if context was unknown"
}
Answer in ENGLISH only."""
def build_egyptologist_prompt(
codes : list[str],
confidences : list[float],
ctx : TextContext,
direction : str,
layout : str,
cartouche_names : list[str],
previous_context : str | None,
chunk_info : str,
) -> str:
"""Build the per-chunk USER message (the system message is static)."""
signs = ' — '.join(
f'{c}({"HIGH" if s >= 0.80 else "MED" if s >= 0.50 else "LOW"}:{s:.2f})'
for c, s in zip(codes, confidences)
)
def h(v: str) -> str: # human-readable context value
# blank/whitespace (client sent '' after the user erased a field)
# must read as 'unknown', not an empty line the LLM could misread
return (v or '').strip().replace('_', ' ') or 'unknown'
# Matched names are authoritative; UNRESOLVED entries (match refused)
# carry raw interior signs only — never present those as verified.
matched = [n for n in cartouche_names if not n.startswith('[UNRESOLVED')]
unresolved = [n for n in cartouche_names if n.startswith('[UNRESOLVED')]
parts = []
if matched:
parts.append(
'Royal cartouches identified in this scene (lexicon-verified — '
'treat as authoritative, more reliable than raw sign codes):\n '
+ '\n '.join(matched))
if unresolved:
parts.append(
'Cartouches detected but NOT resolved to a known royal name — '
'read their raw interior signs yourself (a royal name or epithet '
'is likely; do not invent a specific king):\n '
+ '\n '.join(unresolved))
cartouche_block = '\n- '.join(parts) if parts else 'Contains cartouche: no'
previous_block = (
f'\nPREVIOUS SEGMENTS of the same inscription (already transliterated '
f'— keep names, epithets and topic consistent with them):\n'
f'{previous_context}\n' if previous_context else ''
)
return f"""DETECTED SIGN SEQUENCE ({chunk_info}; each segment is one physical line of text; Gardiner codes with detector confidence):
{signs}
ARCHAEOLOGICAL CONTEXT (fields marked 'unknown' were not supplied by the user — do not invent them, but exploit every field that IS given):
- Period: {h(ctx.period)}
- Dynasty: {h(ctx.dynasty)}
- King's reign: {h(ctx.kings_reign)}
- Text type: {h(ctx.text_type)}
- Physical support: {h(ctx.support)}
- Location type: {h(ctx.location_type)}
- Site: {h(ctx.site)}
- Reading direction: {direction}
- Layout: {layout}
- {cartouche_block}
{previous_block}
Transliterate and gloss this segment. JSON only."""
class TransliterationService:
"""GPT-4o transliteration. Degrades gracefully — never raises."""
def __init__(self) -> None:
self._client = OpenAI(api_key=OPENAI_API_KEY) if OPENAI_API_KEY else None
@property
def enabled(self) -> bool:
return self._client is not None
def transliterate(
self,
raw : dict, # SphinxPipeline.run() output
ctx : TextContext,
) -> TransliterationOut:
outer = raw['outer']
codes = outer['correction']['flat_corrected_seq']
confs = [slot[0][1] if slot else 0.0 for slot in outer['slots']]
# corrected seq and slots are index-aligned; guard anyway
if len(confs) != len(codes):
confs = (confs + [0.0] * len(codes))[:len(codes)]
cartouche_names = []
for c in raw['cartouches']:
if c.get('translit'):
cartouche_names.append(
f"{c['translit']} — {c['english']} (interior: {' '.join(c['spelling'] or [])})"
)
elif c.get('n_members', 0) > 0:
raw_codes = ' '.join(
slot[0][0] for slot in c.get('slots', []) if slot
)
if raw_codes:
cartouche_names.append(
f"[UNRESOLVED cartouche — raw signs detected but no confident "
f"royal-name match: {raw_codes}]"
)
else:
cartouche_names.append(f"[UNRESOLVED cartouche — no signs detected]")
return self.transliterate_sequence(
codes, confs, outer['boundary_hints'], cartouche_names,
direction=raw['direction'], layout=raw['layout'], ctx=ctx,
)
def transliterate_sequence(
self,
codes : list[str],
confidences : list[float],
boundary_hints : list[int],
cartouche_names : list[str],
*,
direction : str,
layout : str,
ctx : TextContext,
) -> TransliterationOut:
if not self.enabled:
return TransliterationOut(
chunks=[], full_transliteration='', full_translation='',
model=OPENAI_MODEL, n_chunks=0,
error='OPENAI_API_KEY not configured — transliteration disabled.',
)
chunks = segment_into_chunks(
codes, confidences, boundary_hints, max_signs=TRANSLIT_MAX_SIGNS,
)
results: list[ChunkTransliteration] = []
for i, chunk in enumerate(chunks):
c_codes = [c for c, _ in chunk]
c_confs = [s for _, s in chunk]
previous = ' | '.join(r.transliteration for r in results) or None
prompt = build_egyptologist_prompt(
c_codes, c_confs, ctx,
direction=direction, layout=layout,
cartouche_names=cartouche_names,
previous_context=previous,
chunk_info=f'Segment {i + 1} of {len(chunks)}',
)
try:
response = self._client.chat.completions.create(
model = OPENAI_MODEL,
messages = [
{'role': 'system', 'content': SYSTEM_PROMPT},
{'role': 'user', 'content': prompt},
],
temperature = TRANSLIT_TEMPERATURE,
response_format = {'type': 'json_object'},
)
parsed = json.loads(response.choices[0].message.content)
except Exception as e:
logger.exception(f'LLM transliteration failed on chunk {i}')
return TransliterationOut(
chunks=results,
full_transliteration=' | '.join(r.transliteration for r in results),
full_translation=' '.join(r.english_gloss for r in results),
model=OPENAI_MODEL, n_chunks=len(chunks),
error=f'LLM call failed on segment {i + 1}/{len(chunks)}: {e}',
)
results.append(ChunkTransliteration(
chunk_index = i,
gardiner_codes = c_codes,
transliteration = str(parsed.get('transliteration', '')),
english_gloss = str(parsed.get('english_gloss', '')),
linguistic_notes = str(parsed.get('linguistic_notes', '')),
confidence = parsed.get('confidence', 'LOW')
if parsed.get('confidence') in ('HIGH', 'MEDIUM', 'LOW')
else 'LOW',
period_note = str(parsed.get('period_note', '')),
))
return TransliterationOut(
chunks = results,
full_transliteration = ' | '.join(r.transliteration for r in results),
full_translation = ' '.join(r.english_gloss for r in results),
model = OPENAI_MODEL,
n_chunks = len(results),
)
# One client for the process lifetime — the FastAPI version built a new
# OpenAI client per request, which is wrong for a long-lived Gradio app.
SERVICE = TransliterationService()
|