DocDoeAI / scripts /build_humanism_bilingual_progressive.py
asnannp's picture
deploy: sync backend to Space root (learn-lesson HF cache fix)
3bcdb36
Raw
History Blame Contribute Delete
18.7 kB
from __future__ import annotations
import argparse
import json
import os
import re
import sys
import time
from pathlib import Path
from dotenv import load_dotenv
ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT))
from backend.scripts.generate_source_backed_science_blueprints import call_groq, call_openrouter
SECTION_TITLES = {
"opening": ("Chapter opening", "അധ്യായത്തിന്റെ തുടക്കം"),
"renaissance-background": ("Renaissance background", "റെനൈസൻസിന്റെ പശ്ചാത്തലം"),
"why-italy": ("Why Italy?", "എന്തുകൊണ്ട് ഇറ്റലി?"),
"crusades-and-trade": ("Crusades and trade", "കുരിശുയുദ്ധങ്ങളും വ്യാപാരവും"),
"black-death": ("Black Death", "ബ്ലാക്ക് ഡെത്ത്"),
"patronage-and-manuscripts": ("Patronage and manuscripts", "സംരക്ഷണവും കൈയെഴുത്തുപ്രതികളും"),
"humanism": ("Humanism", "മാനവികത"),
"renaissance-art": ("Renaissance art", "റെനൈസൻസ് കല"),
"literature-politics-printing": ("Literature, politics and printing", "സാഹിത്യം, രാഷ്ട്രീയം, അച്ചടി"),
"historiography": ("Historiography", "ചരിത്രരചന"),
"renaissance-science": ("Renaissance science", "റെനൈസൻസ് ശാസ്ത്രം"),
"reformation": ("Reformation", "മതനവീകരണം"),
"counter-reformation": ("Counter-Reformation", "പ്രതി മതനവീകരണം"),
"master-recap": ("Exam recap", "പരീക്ഷാ ആവർത്തനം"),
}
NAME_REPLACEMENTS = {
"Renaissance": "റെനൈസൻസ്",
"Humanism": "മാനവികത",
"Italy": "ഇറ്റലി",
"Venice": "വെനീസ്",
"Milan": "മിലാൻ",
"Genoa": "ജെനോവ",
"Florence": "ഫ്ലോറൻസ്",
"Rome": "റോം",
"Medici": "മെഡിച്ചി",
"Constantinople": "കോൺസ്റ്റാന്റിനോപ്പിൾ",
"Roger Bacon": "റോജർ ബേക്കൺ",
"Leonardo da Vinci": "ലിയോനാർഡോ ഡാ വിഞ്ചി",
"Michelangelo": "മൈക്കലാഞ്ചലോ",
"Machiavelli": "മക്കിയവെല്ലി",
"Copernicus": "കോപ്പർനിക്കസ്",
"Galileo": "ഗലീലിയോ",
"Kepler": "കെപ്ലർ",
"Newton": "ന്യൂട്ടൺ",
"Martin Luther": "മാർട്ടിൻ ലൂഥർ",
"Gutenberg": "ഗുട്ടൻബർഗ്",
"Erasmus": "ഇറാസ്മസ്",
"Petrarch": "പെട്രാർക്ക്",
"Dante": "ഡാന്റെ",
"Boccaccio": "ബൊക്കാച്ചിയോ",
"Raphael": "റാഫേൽ",
"Donatello": "ഡൊണാറ്റെല്ലോ",
"Brunelleschi": "ബ്രൂണെല്ലെസ്കി",
"Paracelsus": "പാരസെൽസസ്",
"Vesalius": "വെസാലിയസ്",
"Ignatius Loyola": "ഇഗ്നേഷ്യസ് ലൊയോള",
}
def compact_scene(scene: dict) -> dict:
visual = scene.get("visual") or {}
return {
"id": scene["id"],
"section": scene["section"],
"title": scene["title"],
"point": scene["point"],
"narration": scene["narration"],
"visual_kind": visual.get("kind", "teaching_board"),
"items": visual.get("items") or [],
"left": visual.get("left") or [],
"right": visual.get("right") or [],
"source_pages": scene.get("sourcePages") or [],
}
def prompt_for(batch: list[dict]) -> str:
return f"""
You are translating a verified Kerala SCERT Class 10 Social Science history lesson
for a bilingual DocDoe teaching video. Preserve the exact factual meaning, names,
dates, cause-effect relationships, and exam value. Do not add facts.
For each input scene return:
- id unchanged
- title_ml: short natural Malayalam title
- display_text_ml: natural Malayalam version of the point, suitable as the main on-screen line
- tts_text_ml: natural teacher-style Malayalam in Malayalam script, 18 to 34 words,
easy for a weak Class 10 student; keep foreign names in Malayalam phonetic spelling
- items_ml, left_ml, right_ml: Malayalam translations matching the input arrays exactly
- caption_cues_ml: 2 to 4 complete semantic clauses copied from tts_text_ml in order;
normally 5 to 10 spoken words, never end with a comma, maximum two lines when displayed
- caption_cues_en: semantic segmentation of the exact English narration in order;
each cue normally 6 to 11 words, never end with a comma, never split a proper name,
and do not paraphrase or omit words
Malayalam pronunciation rules:
Renaissance=റെനൈസൻസ്, Italy=ഇറ്റലി, Venice=വെനീസ്, Milan=മിലാൻ,
Genoa=ജെനോവ, Florence=ഫ്ലോറൻസ്, Rome=റോം, Medici=മെഡിച്ചി,
Constantinople=കോൺസ്റ്റാന്റിനോപ്പിൾ, Humanism=മാനവികത,
Roger Bacon=റോജർ ബേക്കൺ, Leonardo da Vinci=ലിയോനാർഡോ ഡാ വിഞ്ചി,
Michelangelo=മൈക്കലാഞ്ചലോ, Machiavelli=മക്കിയവെല്ലി,
Copernicus=കോപ്പർനിക്കസ്, Galileo=ഗലീലിയോ, Kepler=കെപ്ലർ,
Newton=ന്യൂട്ടൺ, Martin Luther=മാർട്ടിൻ ലൂഥർ.
Use Malayalam script for ordinary explanation. Official English terms must not
replace Malayalam prose. Return valid JSON only:
{{"scenes":[{{"id":"...", "title_ml":"...", "display_text_ml":"...",
"tts_text_ml":"...", "items_ml":[], "left_ml":[], "right_ml":[],
"caption_cues_ml":[], "caption_cues_en":[]}}]}}
INPUT:
{json.dumps(batch, ensure_ascii=False)}
""".strip()
def validate(source: list[dict], translated: dict) -> list[dict]:
output = translated.get("scenes")
if not isinstance(output, list) or len(output) != len(source):
raise ValueError("Provider did not return one translated scene per input scene.")
by_id = {scene.get("id"): scene for scene in output}
validated: list[dict] = []
for original in source:
item = by_id.get(original["id"])
if not item:
raise ValueError(f"Missing translated scene {original['id']}.")
for key in ("title_ml", "display_text_ml", "tts_text_ml"):
if not str(item.get(key, "")).strip():
raise ValueError(f"{original['id']} is missing {key}.")
latin = re.findall(r"\b[A-Za-z]{3,}\b", item["tts_text_ml"])
if len(latin) > 1:
raise ValueError(f"{original['id']} contains Latin prose in Malayalam TTS: {latin}")
for key, source_key in (("items_ml", "items"), ("left_ml", "left"), ("right_ml", "right")):
if len(item.get(key) or []) != len(original[source_key]):
raise ValueError(f"{original['id']} has mismatched {key}.")
for key in ("caption_cues_ml", "caption_cues_en"):
cues = item.get(key)
if not isinstance(cues, list) or not cues:
raise ValueError(f"{original['id']} is missing {key}.")
if any(str(cue).rstrip().endswith(",") for cue in cues):
raise ValueError(f"{original['id']} has a comma-ending cue in {key}.")
validated.append(item)
return validated
def semantic_cues(text: str, maximum_words: int) -> list[str]:
sentences = [
sentence.strip()
for sentence in re.split(r"(?<=[.!?।])\s+", text.strip())
if sentence.strip()
]
cues: list[str] = []
conjunctions = {
"and", "but", "while", "because", "therefore", "then", "which",
"എന്നാൽ", "അതുകൊണ്ട്", "കൂടാതെ", "എന്നും", "അതിനാൽ", "അപ്പോൾ",
}
for sentence in sentences:
words = sentence.split()
while len(words) > maximum_words:
candidates = [
index for index in range(5, min(maximum_words + 1, len(words)))
if words[index].strip(",:;").lower() in conjunctions
or words[index - 1].endswith((",", ";", ":"))
]
split_at = candidates[-1] if candidates else maximum_words
cue = " ".join(words[:split_at]).rstrip(",;:")
cues.append(cue + ("." if cue[-1:] not in ".!?" else ""))
words = words[split_at:]
if words:
cue = " ".join(words).rstrip(",;:")
if cues and len(words) < 4 and len(cues[-1].split()) + len(words) <= maximum_words:
cues[-1] = f"{cues[-1].rstrip('.')} {cue}"
else:
cues.append(cue)
return [cue for cue in cues if cue]
def remove_latin_prose(text: str) -> str:
output = text
for source, target in sorted(NAME_REPLACEMENTS.items(), key=lambda item: -len(item[0])):
output = re.sub(re.escape(source), target, output, flags=re.IGNORECASE)
return output
def build_with_nllb(scenes: list[dict], model_name: str) -> list[dict]:
import torch
from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
tokenizer = AutoTokenizer.from_pretrained(model_name, src_lang="eng_Latn")
model = AutoModelForSeq2SeqLM.from_pretrained(
model_name,
torch_dtype=torch.float16 if torch.cuda.is_available() else torch.float32,
)
device = "cuda" if torch.cuda.is_available() else "cpu"
model.to(device)
model.eval()
target_id = tokenizer.convert_tokens_to_ids("mal_Mlym")
requests: list[tuple[str, str, int | None]] = []
for scene in scenes:
requests.extend([
(scene["id"], "title_ml", None),
(scene["id"], "display_text_ml", None),
(scene["id"], "tts_text_ml", None),
])
for key, values in (("items_ml", scene["items"]), ("left_ml", scene["left"]), ("right_ml", scene["right"])):
for index, _ in enumerate(values):
requests.append((scene["id"], key, index))
text_lookup: list[str] = []
by_id = {scene["id"]: scene for scene in scenes}
for scene_id, field, index in requests:
scene = by_id[scene_id]
if field == "title_ml":
text_lookup.append(scene["title"])
elif field == "display_text_ml":
text_lookup.append(scene["point"])
elif field == "tts_text_ml":
text_lookup.append(scene["narration"])
else:
source_key = field.removesuffix("_ml")
text_lookup.append(str(scene[source_key][index]))
translations: list[str] = []
for start in range(0, len(text_lookup), 8):
batch = text_lookup[start : start + 8]
encoded = tokenizer(batch, return_tensors="pt", padding=True, truncation=True, max_length=512).to(device)
with torch.inference_mode():
generated = model.generate(
**encoded,
forced_bos_token_id=target_id,
max_new_tokens=256,
num_beams=4,
)
translations.extend(tokenizer.batch_decode(generated, skip_special_tokens=True))
print(json.dumps({"event": "nllb_translation_progress", "complete": len(translations), "total": len(text_lookup)}), flush=True)
results: dict[str, dict] = {
scene["id"]: {
"id": scene["id"],
"items_ml": [""] * len(scene["items"]),
"left_ml": [""] * len(scene["left"]),
"right_ml": [""] * len(scene["right"]),
}
for scene in scenes
}
for request, translated in zip(requests, translations):
scene_id, field, index = request
value = remove_latin_prose(translated.strip())
if index is None:
results[scene_id][field] = value
else:
results[scene_id][field][index] = value
output: list[dict] = []
for scene in scenes:
item = results[scene["id"]]
item["caption_cues_en"] = semantic_cues(scene["narration"], 11)
item["caption_cues_ml"] = semantic_cues(item["tts_text_ml"], 9)
output.append(item)
del model
if torch.cuda.is_available():
torch.cuda.empty_cache()
return output
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--source", default="data/teaching/history/humanism-production.json")
parser.add_argument(
"--output",
default="data/teaching/social-science/production/humanism-progressive-bilingual-v2.json",
)
parser.add_argument("--model", default="google/gemini-2.5-flash")
parser.add_argument("--provider", choices=("openrouter", "groq", "nllb"), default="openrouter")
parser.add_argument("--batch-size", type=int, default=8)
parser.add_argument("--resume", action="store_true")
args = parser.parse_args()
load_dotenv(ROOT / ".env")
key_name = "GROQ_API_KEY" if args.provider == "groq" else "OPENROUTER_API_KEY"
api_key = os.getenv(key_name, "").strip() if args.provider != "nllb" else ""
if args.provider != "nllb" and not api_key:
raise SystemExit(f"{key_name} is not configured.")
source_path = ROOT / args.source
output_path = ROOT / args.output
source = json.loads(source_path.read_text(encoding="utf-8"))
scenes = [compact_scene(scene) for scene in source["scenes"]]
completed: dict[str, dict] = {}
if args.resume and output_path.exists():
previous = json.loads(output_path.read_text(encoding="utf-8"))
completed = {unit["id"]: unit for unit in previous.get("units", [])}
output_path.parent.mkdir(parents=True, exist_ok=True)
if args.provider == "nllb":
translated = build_with_nllb(scenes, args.model)
for original, item in zip(scenes, translated):
section_en, section_ml = SECTION_TITLES[original["section"]]
completed[original["id"]] = {
**original,
"section_title_en": section_en,
"section_title_ml": section_ml,
"display_text_en": original["point"],
"display_text_ml": item["display_text_ml"],
"title_en": original["title"],
"title_ml": item["title_ml"],
"tts_text_en": original["narration"],
"tts_text_ml": item["tts_text_ml"],
"items_ml": item["items_ml"],
"left_ml": item["left_ml"],
"right_ml": item["right_ml"],
"caption_cues_en": item["caption_cues_en"],
"caption_cues_ml": item["caption_cues_ml"],
"pauseAfterSeconds": 0.38,
"retrievalPauseSeconds": 3.2 if "Retrieval" in original["title"] and "answer" not in original["title"].lower() else 0,
}
for start in range(0, len(scenes), args.batch_size):
if args.provider == "nllb":
break
batch = [scene for scene in scenes[start : start + args.batch_size] if scene["id"] not in completed]
if not batch:
continue
error: Exception | None = None
for attempt in range(1, 4):
try:
result = (
call_groq(prompt_for(batch), api_key, args.model)
if args.provider == "groq"
else call_openrouter(prompt_for(batch), api_key, args.model)
)
translated = validate(batch, result)
for original, item in zip(batch, translated):
section_en, section_ml = SECTION_TITLES[original["section"]]
completed[original["id"]] = {
**original,
"section_title_en": section_en,
"section_title_ml": section_ml,
"display_text_en": original["point"],
"display_text_ml": item["display_text_ml"],
"title_en": original["title"],
"title_ml": item["title_ml"],
"tts_text_en": original["narration"],
"tts_text_ml": item["tts_text_ml"],
"items_ml": item["items_ml"],
"left_ml": item["left_ml"],
"right_ml": item["right_ml"],
"caption_cues_en": item["caption_cues_en"],
"caption_cues_ml": item["caption_cues_ml"],
"pauseAfterSeconds": 0.38,
"retrievalPauseSeconds": 3.2 if "Retrieval" in original["title"] and "answer" not in original["title"].lower() else 0,
}
print(json.dumps({"event": "translation_batch_complete", "start": start + 1, "count": len(batch)}), flush=True)
error = None
break
except Exception as exc:
error = exc
print(json.dumps({"event": "translation_batch_retry", "start": start + 1, "attempt": attempt, "error": str(exc)}), flush=True)
time.sleep(attempt * 2)
if error:
raise error
ordered = [completed[scene["id"]] for scene in scenes if scene["id"] in completed]
output_path.write_text(json.dumps({
"schemaVersion": 2,
"lessonId": "hist-part1-ch01-humanism-progressive-v2",
"chapterId": "hist-part1-ch01-humanism",
"title": "Humanism",
"title_ml": "മാനവികത",
"subject": "Social Science I",
"classLevel": "Kerala SSLC Class 10",
"source": args.source,
"units": ordered,
}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
ordered = [completed[scene["id"]] for scene in scenes if scene["id"] in completed]
output_path.write_text(json.dumps({
"schemaVersion": 2,
"lessonId": "hist-part1-ch01-humanism-progressive-v2",
"chapterId": "hist-part1-ch01-humanism",
"title": "Humanism",
"title_ml": "മാനവികത",
"subject": "Social Science I",
"classLevel": "Kerala SSLC Class 10",
"source": args.source,
"units": ordered,
}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({"event": "bilingual_chapter_ready", "path": str(output_path), "units": len(completed)}))
return 0
if __name__ == "__main__":
raise SystemExit(main())