Spaces:
Running
Running
| """ | |
| Osage Language API — FastAPI backend for RAG-powered Osage queries. | |
| Provides: | |
| /query — Full RAG: retrieve context + LLM generation | |
| /search — Lightweight dictionary/grammar search (no LLM) | |
| /health — Health check | |
| Usage: | |
| conda activate osage | |
| uvicorn app.api:app --host 0.0.0.0 --port 8000 | |
| uvicorn app.api:app --host 0.0.0.0 --port 8000 --reload # dev mode | |
| # With LoRA adapter: | |
| OSAGE_ADAPTER=adapters uvicorn app.api:app --port 8000 | |
| """ | |
| import os | |
| import sys | |
| import time | |
| from pathlib import Path | |
| from contextlib import asynccontextmanager | |
| from fastapi import FastAPI, Request, Response | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from fastapi.staticfiles import StaticFiles | |
| from pydantic import BaseModel, Field | |
| BASE = Path(__file__).resolve().parent.parent | |
| sys.path.insert(0, str(BASE)) | |
| # ── Models ──────────────────────────────────────────────────────────────────── | |
| class QueryRequest(BaseModel): | |
| question: str = Field(..., description="Question about the Osage language") | |
| n_results: int = Field(12, description="Number of context chunks to retrieve") | |
| class QueryResponse(BaseModel): | |
| response: str | |
| sources: list[dict] | |
| elapsed: float | |
| class SearchRequest(BaseModel): | |
| query: str = Field(..., description="Search term (Osage or English)") | |
| n_results: int = Field(20, description="Number of results to return") | |
| class SearchResponse(BaseModel): | |
| results: list[dict] | |
| count: int | |
| class TranslateRequest(BaseModel): | |
| text: str = Field(..., description="Text to translate (English or Osage script)") | |
| max_results: int = Field(20, description="Maximum number of matches to return") | |
| class TranslateResponse(BaseModel): | |
| tier: int | |
| query: str | |
| direction: str | |
| matches: list[dict] | |
| count: int | |
| elapsed_ms: float | |
| class ReportPayload(BaseModel): | |
| # Everything optional so clients with partial state can still submit. | |
| # Two field-name schemes are accepted (frontend's flag() emits the second | |
| # set; legacy callers / scripts use the first): | |
| # input/expected/got/tab AND query/osage|apa|english/source | |
| # The endpoint normalizes them when writing to reports.jsonl. | |
| model_config = {"extra": "allow"} # tolerate any extra free-form fields | |
| input: str = Field("", description="User's original query (legacy)") | |
| expected: str = Field("", description="User's correction (legacy)") | |
| got: str = Field("", description="What the app returned (legacy)") | |
| note: str = Field("", description="Free-text comment") | |
| tab: str = Field("", description="Active tab (legacy: translate/teacher/...)") | |
| direction: str = Field("", description="e2o / o2e / other") | |
| ua: str = Field("", description="User-agent string for debugging") | |
| # Accept ts as int OR ISO string OR float; we only store it as-is. | |
| ts: str | int | float = Field(0, description="Client timestamp (ISO or ms epoch)") | |
| # New-schema fields used by translate.js OsageReports.flag(): | |
| query: str = Field("", description="User's original query (new)") | |
| osage: str = Field("", description="Osage-script form returned") | |
| apa: str = Field("", description="APA form returned") | |
| english: str = Field("", description="English form returned (or teacher answer)") | |
| source: str = Field("", description="translate / teacher / ai-teacher") | |
| context: str = Field("", description="Free-form context (e.g. 'Tab: Teacher')") | |
| # ── App Setup ───────────────────────────────────────────────────────────────── | |
| generator = None | |
| translator = None | |
| teacher = None | |
| ai_llm = None # Sandboxed LLM for AI Teacher tab only | |
| async def lifespan(app: FastAPI): | |
| """Load the translator (fast, no GPU) and optionally the generator (needs GPU).""" | |
| global generator, translator, teacher, ai_llm | |
| # Tier 1: always load — fast dictionary lookup, no GPU | |
| from app.translate import OsageTranslator | |
| print("Loading Tier 1 translator...") | |
| translator = OsageTranslator() | |
| print(f"Translator ready! ({translator.entry_count} entries)") | |
| try: | |
| from app.teacher import OsageTeacher | |
| teacher = OsageTeacher() | |
| print("Teacher ready!") | |
| except Exception as e: | |
| print(f"Teacher not loaded (non-fatal): {e}") | |
| teacher = None | |
| # AI Teacher LLM: separate from main teacher, uses HF Inference API | |
| try: | |
| from app.llm_wrapper import LLMWrapper | |
| ai_llm = LLMWrapper.create(backend="auto") | |
| if ai_llm.is_available: | |
| print(f"AI Teacher LLM ready: {ai_llm.backend_name}") | |
| else: | |
| print("AI Teacher LLM: no HF_TOKEN — AI Teacher tab will show T1 answers only") | |
| ai_llm = None | |
| except Exception as e: | |
| print(f"AI Teacher LLM not loaded (non-fatal): {e}") | |
| ai_llm = None | |
| # Tier 2: load LLM generator only if not in lightweight mode | |
| if not os.environ.get("OSAGE_LIGHTWEIGHT"): | |
| try: | |
| from rag.generator import OsageGenerator | |
| adapter = os.environ.get("OSAGE_ADAPTER") | |
| if adapter and not Path(adapter).is_absolute(): | |
| adapter = str(BASE / adapter) | |
| print(f"Loading Osage generator (adapter={adapter})...") | |
| generator = OsageGenerator(adapter_path=adapter) | |
| print("Generator ready!") | |
| except Exception as e: | |
| print(f"Generator not loaded (Tier 2 unavailable): {e}") | |
| else: | |
| print("Lightweight mode — Tier 2 (LLM) disabled") | |
| yield | |
| app = FastAPI( | |
| title="Osage Language API", | |
| description="RAG-powered API for Osage (Wazhazhe ie) language queries", | |
| version="0.1.0", | |
| lifespan=lifespan, | |
| ) | |
| # CORS — allow the static site and localhost dev servers | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=["*"], | |
| allow_methods=["GET", "POST"], | |
| allow_headers=["*"], | |
| ) | |
| # Serve static files at /static | |
| static_dir = BASE / "app" / "static" | |
| if static_dir.exists(): | |
| app.mount("/static", StaticFiles(directory=str(static_dir), html=True), name="static") | |
| # Serve local audio clips at /audio (dictionary / Sonny / Osage app). | |
| # Used by Phase C play buttons. URL format: /audio/sonny/<file>.mp3. | |
| audio_dir = BASE / "data" / "audio" | |
| if audio_dir.exists(): | |
| app.mount("/audio", StaticFiles(directory=str(audio_dir)), name="audio") | |
| # ── Routes ──────────────────────────────────────────────────────────────────── | |
| async def root(): | |
| """Redirect root to static UI.""" | |
| from fastapi.responses import RedirectResponse | |
| return RedirectResponse(url="/static/index.html") | |
| async def health(): | |
| return {"status": "ok", "generator_loaded": generator is not None, | |
| "teacher_loaded": teacher is not None, | |
| "ai_llm": ai_llm.backend_name if ai_llm else "none"} | |
| # Per-session teacher instances for follow-up context | |
| _teacher_sessions: dict = {} | |
| def _get_teacher(session_id: str | None): | |
| """Get or create a teacher instance for this session.""" | |
| if not session_id or teacher is None: | |
| return teacher | |
| if session_id not in _teacher_sessions: | |
| # Reuse the translator from the global teacher, just track context separately | |
| from app.teacher import OsageTeacher | |
| t = OsageTeacher.__new__(OsageTeacher) | |
| t.__dict__.update(teacher.__dict__) # share loaded data | |
| t._last_topic = None | |
| t._last_verb = None | |
| t._last_query = None | |
| t._last_answer = None | |
| _teacher_sessions[session_id] = t | |
| # Evict old sessions (keep last 50) | |
| if len(_teacher_sessions) > 50: | |
| oldest = list(_teacher_sessions.keys())[0] | |
| del _teacher_sessions[oldest] | |
| return _teacher_sessions[session_id] | |
| async def ask(query: str, session: str | None = None): | |
| """Full teacher pipeline: translation, grammar, error check, drills, conjugation. | |
| This is the primary student-facing endpoint. Routes queries to the | |
| appropriate handler based on intent detection. | |
| Pass ?session=ID for follow-up context across requests. | |
| """ | |
| t = _get_teacher(session) | |
| if t is None: | |
| return {"error": "Teacher not loaded"} | |
| result = t.ask(query) | |
| # Add disambiguation alternatives for ambiguous lookups | |
| # When confidence is low/medium/proposed OR answer has "Also:", show alternatives | |
| if result.get("confidence") in ("low", "medium", "proposed", "none"): | |
| try: | |
| tr = t.translator.translate(query, max_results=5) | |
| matches = tr.get("matches", []) | |
| if len(matches) >= 2: | |
| seen = set() | |
| alts = [] | |
| for m in matches[:6]: | |
| eng = m.get("english", "").split(",")[0].strip()[:40] | |
| apa = m.get("apa", "") | |
| key = eng.lower() | |
| if key not in seen and eng: | |
| seen.add(key) | |
| alts.append({"apa": apa, "english": eng, | |
| "osage": m.get("osage_script", "")}) | |
| if len(alts) >= 2: | |
| result["alternatives"] = alts | |
| except Exception: | |
| pass | |
| return result | |
| async def ask_ai(query: str, session: str | None = None): | |
| """AI-enhanced teacher: T1 translation + LLM explanation. | |
| Always attempts LLM enrichment. Falls back to T1-only if LLM unavailable. | |
| This is the endpoint for the sandboxed AI Teacher tab. | |
| The LLM (ai_llm) is completely separate from the main Teacher pipeline. | |
| """ | |
| t = _get_teacher(session) | |
| if t is None: | |
| return {"error": "Teacher not loaded"} | |
| # Step 1: Get T1 answer (always fast, always available) | |
| t1_result = t.ask(query) | |
| t1_answer = t1_result.get("answer", "") | |
| # Step 2: Try LLM enrichment — but SKIP for routes where T1 is already complete | |
| # Drills, conjugation, marker explanations, verified lookups don't benefit from LLM | |
| t1_route = t1_result.get("route", "") | |
| t1_conf = t1_result.get("confidence", "") | |
| skip_llm_routes = {"drill", "conjugation", "verb_class_explanation", | |
| "marker_comparison", "marker_explanation", "examples", | |
| "topic_lookup", "topic_more", "error_check", | |
| "how_do_you_say", "meta"} | |
| skip_llm = (t1_route in skip_llm_routes or | |
| t1_conf == "proposed" or # constrained generator output — unreliable | |
| (t1_conf == "verified" and t1_route in ("tier1", "tier1+grammar"))) | |
| if ai_llm is not None and not skip_llm: | |
| try: | |
| from app.llm_wrapper import format_llm_prompt | |
| from rag.t1_context import get_t1_data, format_t1_context, route_query | |
| # Build T1 context for the LLM | |
| t1_data = get_t1_data(t.translator, query) | |
| t1_context = format_t1_context(query, t1_data) | |
| # Grammar hits if relevant | |
| grammar_text = "" | |
| route = route_query(query) | |
| if route != "tier1": | |
| grammar_hits = t.grammar.search(query, limit=3) | |
| if grammar_hits: | |
| grammar_text = t.grammar.format_for_prompt(grammar_hits, max_chars=2000) | |
| is_grammar = route != "tier1" | |
| system, user_msg = format_llm_prompt( | |
| query, t1_context, grammar_text, | |
| is_grammar=is_grammar, t1_answer=t1_answer, | |
| ) | |
| llm_response = ai_llm.generate(system, user_msg, max_tokens=512) | |
| if llm_response: | |
| return { | |
| "answer": llm_response, | |
| "tier": "tier1+tier2", | |
| "confidence": t1_result.get("confidence", "medium"), | |
| "route": "ai_teacher", | |
| "t1_answer": t1_answer, | |
| "t1_matches": t1_result.get("t1_matches", 0), | |
| "grammar_hits": t1_result.get("grammar_hits", 0), | |
| "elapsed_ms": t1_result.get("elapsed_ms", 0), | |
| "llm_backend": ai_llm.backend_name, | |
| } | |
| except Exception: | |
| pass # LLM failed — fall through to T1 | |
| # Fallback: return T1 answer with note | |
| t1_result["tier"] = t1_result.get("tier", "tier1") | |
| if ai_llm is None: | |
| t1_result["llm_note"] = "AI not available — showing verified T1 answer" | |
| return t1_result | |
| async def ask_post(req: QueryRequest): | |
| """POST version of /ask for longer queries.""" | |
| if teacher is None: | |
| return {"error": "Teacher not loaded"} | |
| result = teacher.ask(req.question) | |
| return result | |
| async def query(req: QueryRequest): | |
| """ | |
| Full RAG query: retrieve context from ChromaDB, generate response with LLM. | |
| This is the main endpoint for the "Ask the Teacher" feature. | |
| """ | |
| if generator is None: | |
| return QueryResponse(response="Generator not loaded", sources=[], elapsed=0) | |
| result = generator.query(req.question, n_results=req.n_results) | |
| return QueryResponse( | |
| response=result["response"], | |
| sources=result["sources"], | |
| elapsed=result["elapsed"], | |
| ) | |
| async def search(req: SearchRequest): | |
| """ | |
| Lightweight search across dictionary, grammar, and examples. | |
| No LLM generation — just retrieval results. | |
| """ | |
| if generator is None: | |
| return SearchResponse(results=[], count=0) | |
| results = generator.search(req.query, n_results=req.n_results) | |
| return SearchResponse(results=results, count=len(results)) | |
| async def search_get(query: str, n: int = 20): | |
| """GET endpoint for simple search.""" | |
| if generator is None: | |
| return {"results": [], "count": 0} | |
| results = generator.search(query, n_results=n) | |
| return {"results": results, "count": len(results)} | |
| async def translate(req: TranslateRequest): | |
| """ | |
| Tier 1: Fast dictionary lookup translation. No LLM needed. | |
| Handles English→Osage and Osage→English automatically. | |
| """ | |
| if translator is None: | |
| return TranslateResponse(tier=1, query=req.text, direction="unknown", | |
| matches=[], count=0, elapsed_ms=0) | |
| result = translator.translate(req.text, max_results=req.max_results) | |
| return TranslateResponse(**{k: result[k] for k in | |
| ["tier", "query", "direction", "matches", "count", "elapsed_ms"]}) | |
| async def translate_get(text: str, n: int = 20): | |
| """GET endpoint for quick translation lookup.""" | |
| if translator is None: | |
| return {"tier": 1, "query": text, "matches": [], "count": 0, "elapsed_ms": 0} | |
| return translator.translate(text, max_results=n) | |
| async def gloss(text: str): | |
| """Word-by-word interlinear gloss of an Osage sentence.""" | |
| if translator is None: | |
| return {"glosses": [], "elapsed_ms": 0} | |
| return translator.gloss(text) | |
| async def analyze(text: str): | |
| """Morphological analysis of an Osage word or sentence.""" | |
| if translator is None: | |
| return {"input": text, "analyses": [], "elapsed_ms": 0} | |
| return translator.analyze(text) | |
| async def conjugate(text: str): | |
| """Show all known conjugated forms for a verb.""" | |
| if translator is None: | |
| return {"verb": text, "forms": [], "count": 0, "elapsed_ms": 0} | |
| return translator.conjugation_table(text) | |
| async def find_sentence(text: str, n: int = 10): | |
| """Find closest Osage sentence pair for English input.""" | |
| if translator is None: | |
| return {"matches": [], "count": 0, "elapsed_ms": 0} | |
| return translator.find_sentence(text, max_results=n) | |
| async def drill_new( | |
| topic: str = "", | |
| difficulty: str = "beginner", | |
| type: str = "", | |
| exclude: str = "", | |
| prefer: str = "", | |
| chapter: int = 0, | |
| ): | |
| """Generate one practice exercise as structured JSON. | |
| Query params: | |
| topic — 'greetings', 'animals', 'family', 'body', 'food', 'verbs' | |
| difficulty — 'beginner' | 'intermediate' | 'advanced' | |
| type — 'translate_e2o' | 'translate_o2e' | 'conjugate' | | |
| 'sentence_e2o'; empty = pick from difficulty level | |
| exclude — comma-separated answer_english values recently seen | |
| prefer — comma-separated English words to PREFER (spaced | |
| repetition: weak words due for review) | |
| chapter — Pratt chapter id (1-23). Restricts vocab to entries | |
| whose English overlaps that chapter's examples. | |
| Falls back to unfiltered vocab if <5 entries match. | |
| """ | |
| if translator is None: | |
| return {"error": "Translator not loaded"} | |
| from app.drill_mode import DrillEngine | |
| drill = DrillEngine(translator) | |
| excluded = {e.strip().lower() for e in (exclude or "").split(",") if e.strip()} | |
| preferred = [e.strip().lower() for e in (prefer or "").split(",") if e.strip()] | |
| chapter_id = chapter or None | |
| ex = None | |
| # If a preferred list is given, try to construct a drill directly | |
| # from the first preferred word's dictionary entry (bypasses random | |
| # sampling which rarely hits specific words). | |
| if preferred: | |
| drill._get_vocab() # ensure vocab_cache populated | |
| # Respect chapter filter on preferred: only keep prefer words | |
| # that also live in the chapter vocab when chapter is set. | |
| pool = drill._chapter_filtered_vocab(chapter_id) if chapter_id else drill._vocab_cache | |
| eng_index = {} | |
| for entry in pool: | |
| eng_index.setdefault(entry["english"].lower().strip(), entry) | |
| for pref in preferred: | |
| if pref in excluded: | |
| continue | |
| entry = eng_index.get(pref) | |
| if not entry: | |
| continue | |
| # Build a translate_e2o exercise directly from the matched entry | |
| ex = { | |
| "type": "translate_e2o", | |
| "prompt": f"How do you say \"{entry['english']}\" in Osage?", | |
| "answer_apa": entry.get("apa", ""), | |
| "answer_osage": entry.get("osage_script", ""), | |
| "answer_english": entry["english"], | |
| "hint": f"This word is from the {entry.get('source', 'dictionary')}.", | |
| "difficulty": difficulty, | |
| "topic": topic or None, | |
| "chapter": chapter_id, | |
| "spaced_repetition": True, | |
| } | |
| break | |
| # Random fallback (when no prefer match): up to 12 retries to avoid excluded | |
| if ex is None: | |
| best_fallback = None | |
| for attempt in range(12): | |
| cand = drill.generate( | |
| exercise_type=(type or None), | |
| difficulty=difficulty, | |
| topic=(topic or None), | |
| chapter=chapter_id, | |
| ) | |
| if not cand: | |
| continue | |
| ans_eng = (cand.get("answer_english") or "").lower().strip() | |
| if ans_eng in excluded: | |
| continue | |
| if best_fallback is None: | |
| best_fallback = cand | |
| break | |
| ex = best_fallback | |
| # Audio URL for the answer (if any) | |
| if ex: | |
| url = translator.find_audio( | |
| osage=ex.get("answer_osage", ""), | |
| english=ex.get("answer_english", ""), | |
| ) | |
| if url: | |
| ex["audio_url"] = url | |
| return ex or {"error": "drill generation failed"} | |
| async def drill_topics(): | |
| """List the drill topic categories the server knows about.""" | |
| if translator is None: | |
| return {"topics": []} | |
| from app.drill_mode import DrillEngine | |
| drill = DrillEngine(translator) | |
| return {"topics": list(drill.TOPICS.keys()) + ["verbs"]} | |
| async def full_translate(text: str, n: int = 10): | |
| """ | |
| Full translation pipeline: teacher-enhanced lookup + glossing + analysis. | |
| Routes through teacher pipeline for 99.9% accuracy, with raw translator | |
| data included for display (gloss, conjugation, analysis). | |
| """ | |
| if translator is None: | |
| return {"error": "Translator not loaded"} | |
| from app.translate import _is_osage_script, _is_apa_text, _normalize_english | |
| is_osage = _is_osage_script(text) | |
| is_apa = not is_osage and _is_apa_text(text) | |
| word_count = len(text.strip().split()) | |
| # Get teacher answer (enhanced pipeline) if available | |
| teacher_answer = None | |
| if teacher is not None: | |
| try: | |
| tr = teacher.ask(text) | |
| teacher_answer = tr.get("answer", "") | |
| except Exception: | |
| pass | |
| result = { | |
| "query": text, | |
| "is_osage": is_osage, | |
| "is_apa": is_apa, | |
| } | |
| if is_apa: | |
| # APA input: route through translate() which handles APA detection | |
| result["lookup"] = translator.translate(text, max_results=n) | |
| # Also convert to Osage script for gloss/analysis | |
| try: | |
| from scripts.orthography_converter import apa_to_osage_script | |
| osage_words = [] | |
| for w in text.split(): | |
| try: | |
| osage_words.append(apa_to_osage_script(w)) | |
| except Exception: | |
| osage_words.append(w) | |
| osage_text = " ".join(osage_words) | |
| if _is_osage_script(osage_text): | |
| result["gloss"] = translator.gloss(osage_text) | |
| result["analysis"] = translator.analyze(osage_text) | |
| except Exception: | |
| pass | |
| elif is_osage: | |
| # For Osage input: lookup + gloss + analyze | |
| result["lookup"] = translator.translate_fuzzy(text, max_results=n) | |
| result["gloss"] = translator.gloss(text) | |
| result["analysis"] = translator.analyze(text) | |
| elif word_count == 1: | |
| # Single English word: lookup + conjugation | |
| result["lookup"] = translator.translate_fuzzy(text, max_results=n) | |
| matches = result["lookup"].get("matches", []) | |
| if matches: | |
| top_osage = matches[0].get("osage_script", "") | |
| if top_osage: | |
| conj = translator.conjugation_table(top_osage) | |
| if conj.get("count", 0) > 1: | |
| result["conjugation"] = conj | |
| else: | |
| # Multi-word English: try E2O composition FIRST, then sentence match | |
| e2o = translator.compose_to_osage(text) | |
| result["sentences"] = translator.find_sentence(text, max_results=5) | |
| if e2o: | |
| # E2O composition succeeded — use it as primary result | |
| result["lookup"] = { | |
| "matches": [{ | |
| "osage_script": e2o["osage_script"], | |
| "apa": e2o["apa"], | |
| "english": e2o["english"], | |
| "source": e2o["source"], | |
| "match_type": e2o["match_type"], | |
| "score": e2o["score"], | |
| "explain": e2o.get("explain"), | |
| }], | |
| "count": 1, | |
| "direction": "english_to_osage", | |
| } | |
| else: | |
| # Fall back to sentence match or fuzzy lookup | |
| sent_matches = result["sentences"].get("matches", []) | |
| if sent_matches and sent_matches[0].get("similarity", 0) >= 0.4: | |
| best = sent_matches[0] | |
| result["lookup"] = { | |
| "matches": [{ | |
| "osage_script": best["osage_script"], | |
| "apa": best["apa"], | |
| "english": best["english"], | |
| "source": best["source"], | |
| "match_type": "sentence", | |
| "score": 0.0, | |
| }], | |
| "count": 1, | |
| "direction": "english_to_osage", | |
| } | |
| else: | |
| result["lookup"] = translator.translate_fuzzy(text, max_results=n) | |
| # Word-by-word decomposition: look up each content word | |
| stop_words = {"i", "a", "an", "the", "is", "am", "are", "was", "were", | |
| "to", "of", "in", "for", "on", "with", "at", "by", "and", | |
| "or", "but", "not", "do", "does", "did", "will", "would", | |
| "can", "could", "my", "your", "his", "her", "it", "this", | |
| "that", "some", "very", "really", "just", "also", "too"} | |
| words = [w.strip(".,!?") for w in text.lower().split()] | |
| content_words = [w for w in words if w and w not in stop_words] | |
| word_results = [] | |
| for word in content_words: | |
| lookup = translator.translate(word, max_results=3) | |
| matches = lookup.get("matches", []) | |
| # Also try stemmed forms | |
| if not matches: | |
| lookup = translator.translate_fuzzy(word, max_results=3) | |
| matches = lookup.get("matches", []) | |
| if matches: | |
| best = matches[0] | |
| word_results.append({ | |
| "english": word, | |
| "osage_script": best["osage_script"], | |
| "apa": best["apa"], | |
| "osage_english": best["english"][:40], | |
| "found": True, | |
| }) | |
| else: | |
| word_results.append({ | |
| "english": word, | |
| "osage_script": "", | |
| "apa": "", | |
| "osage_english": "", | |
| "found": False, | |
| }) | |
| result["word_breakdown"] = word_results | |
| # Phase C: attach audio URLs to any returned matches so the UI can | |
| # render a play button for dictionary-backed pronunciations. | |
| if result.get("lookup") and result["lookup"].get("matches"): | |
| for m in result["lookup"]["matches"]: | |
| if not m.get("audio_url"): | |
| url = translator.find_audio( | |
| osage=m.get("osage_script", ""), | |
| english=m.get("english", ""), | |
| ) | |
| if url: | |
| m["audio_url"] = url | |
| # Include teacher-enhanced answer if available (richer than raw translator) | |
| if teacher_answer: | |
| result["teacher_answer"] = teacher_answer | |
| return result | |
| # In-process rate-limit state for /report. IP → (window_start_ts, count). | |
| # Simple fixed-window counter; resets each minute. Not shared across workers. | |
| _REPORT_RATE: dict[str, tuple[float, int]] = {} | |
| _REPORT_RATE_LIMIT = 10 # reports per IP per minute | |
| async def submit_report(payload: ReportPayload, request: Request): | |
| """Append a user bug report to data/feedback/reports.jsonl. | |
| Client-side reports stored in LocalStorage (keyed REPORT_KEY) can also | |
| be POSTed here when the user opts into "Send automatically". Rate | |
| limited to 10/min/IP; returns {ok, queued} or 429. | |
| """ | |
| client_ip = (request.client.host if request.client else "unknown") or "unknown" | |
| now = time.time() | |
| win_start, count = _REPORT_RATE.get(client_ip, (now, 0)) | |
| if now - win_start > 60: | |
| win_start, count = now, 0 | |
| if count >= _REPORT_RATE_LIMIT: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=429, detail="rate_limited") | |
| _REPORT_RATE[client_ip] = (win_start, count + 1) | |
| fb_dir = Path(__file__).resolve().parent.parent / "data" / "feedback" | |
| fb_dir.mkdir(parents=True, exist_ok=True) | |
| fb_path = fb_dir / "reports.jsonl" | |
| # Coalesce the two field-name schemes. Frontend (translate.js flag()) sends | |
| # query/osage/apa/english/source; legacy clients send input/got/tab. | |
| user_query = (payload.query or payload.input or "")[:2000] | |
| user_expected = (payload.expected or "")[:2000] | |
| app_returned = (payload.got or payload.english or payload.osage or payload.apa or "")[:2000] | |
| active_tab = (payload.source or payload.tab or "")[:64] | |
| record = { | |
| "server_ts": int(now * 1000), | |
| "client_ts": payload.ts, | |
| "input": user_query, | |
| "expected": user_expected, | |
| "got": app_returned, | |
| "note": payload.note[:2000], | |
| "tab": active_tab, | |
| "direction": payload.direction[:16], | |
| "osage": payload.osage[:2000], | |
| "apa": payload.apa[:2000], | |
| "context": payload.context[:512], | |
| "ua": payload.ua[:512], | |
| "ip_hash": str(hash(client_ip) & 0xFFFFFFFF), | |
| } | |
| import json as _json | |
| with open(fb_path, "a", encoding="utf-8") as f: | |
| f.write(_json.dumps(record, ensure_ascii=False) + "\n") | |
| return {"ok": True, "queued": True} | |
| async def export_reports(request: Request): | |
| """Stream reports.jsonl. Guarded by REPORTS_EXPORT_TOKEN env var. | |
| Enables `scripts/review_feedback.py --fetch <url>` to pull reports | |
| from a deployed instance (HF Space). Returns 404 when the token | |
| env is unset (feature disabled) or the token doesn't match. | |
| """ | |
| expected = os.environ.get("REPORTS_EXPORT_TOKEN", "") | |
| if not expected: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=404) | |
| supplied = request.headers.get("x-reports-token") or request.query_params.get("token", "") | |
| if supplied != expected: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=404) | |
| fb_path = Path(__file__).resolve().parent.parent / "data" / "feedback" / "reports.jsonl" | |
| if not fb_path.exists(): | |
| return Response(content="", media_type="application/x-ndjson") | |
| return Response(content=fb_path.read_bytes(), media_type="application/x-ndjson") | |
| # ── Speaker Review (read-only, first cut) ────────────────────────────────── | |
| # | |
| # Surfaces pending items from the three review queues (Phase 2 grammar | |
| # content, LF1932 cedilla candidates, user reports) in a format the speaker | |
| # review UI at /static/review.html can render. Read-only: no write-back yet. | |
| # | |
| # Token-guarded with REVIEW_TOKEN env var so the endpoint can ship in the | |
| # sovereign image without exposing the queue publicly when unset, mirrors | |
| # the /report/export pattern. | |
| def _check_review_token(request: Request) -> None: | |
| expected = os.environ.get("REVIEW_TOKEN", "") | |
| # When REVIEW_TOKEN is not set, allow access on localhost only (dev/sovereign mode). | |
| # Production HF Space must set the token to expose this endpoint. | |
| if not expected: | |
| client_host = (request.client.host if request.client else "") | |
| if client_host not in ("127.0.0.1", "localhost", "::1", ""): | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=404) | |
| return | |
| supplied = request.headers.get("x-review-token") or request.query_params.get("token", "") | |
| if supplied != expected: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=404) | |
| async def review_pending(request: Request, category: str = "grammar", limit: int = 200): | |
| """Return pending items for speaker review. | |
| category=grammar → pending Phase 2 content items (any reviewed != approved/pratt_sourced/revised) | |
| category=lf → uncorrected LF1932 entries with a `g` (cedilla candidates) | |
| category=reports → flagged user reports awaiting triage | |
| """ | |
| import json as _json | |
| _check_review_token(request) | |
| base_path = Path(__file__).resolve().parent.parent | |
| out = {"category": category, "items": []} | |
| if category == "grammar": | |
| files = [ | |
| "seed_templates.json", | |
| "verb_paradigms_generated.json", | |
| "grammar_explanations_generated.json", | |
| "error_corrections_generated.json", | |
| "error_corrections_advanced.json", | |
| "conjugation_drills_generated.json", | |
| "sentence_breakdowns_generated.json", | |
| "translation_exercises_generated.json", | |
| "vocabulary_exercises_generated.json", | |
| "common_qa_generated.json", | |
| ] | |
| approved_set = {"approved", "pratt_sourced", "revised"} | |
| for fname in files: | |
| fpath = base_path / "data" / "grammar_content" / fname | |
| if not fpath.exists(): | |
| continue | |
| for it in _json.loads(fpath.read_text()): | |
| if it.get("reviewed") not in approved_set: | |
| out["items"].append({"_file": fname, **it}) | |
| if len(out["items"]) >= limit: | |
| break | |
| if len(out["items"]) >= limit: | |
| break | |
| elif category == "lf": | |
| # Filter out malformed extractions: some LF entries have English | |
| # definition text leaked into the headword field (e.g., "a'-ba my | |
| # stooping shoulders." or "hiu'dse a gorilla; a big"). Hide them | |
| # from the speaker review queue — they need a separate cleanup pass. | |
| _LF_JUNK_TOKENS = { | |
| "my", "your", "his", "her", "the", "with", "from", "and", "of", | |
| "by", "to", "is", "as", "on", "in", "at", "for", "or", "an", "a", | |
| "stooping", "shoulders", "backward", "forward", "down", "up", | |
| "near", "toward", "moving", "sitting", "lying", "standing", | |
| "sleeping", "hidden", "appearing", "expression", "action", | |
| "having", "plenty", "this", "that", "gorilla", "big", "small", | |
| "many", "few", "all", "some", "made", "make", "given", "give", | |
| "side", "person", "people", "thing", "things", "take", "took", | |
| "hold", "before", "after", "while", "since", "during", "between", | |
| } | |
| def _lf_is_junk(h: str) -> bool: | |
| if not h.strip(): | |
| return True | |
| # All-uppercase headwords are section headers (e.g., "OSAGE-ENGLISH"), | |
| # not real entries. Real LF headwords are lowercase or capitalized | |
| # only on the first letter (proper names like "Wa-ko"-da"). | |
| stripped = h.replace("-", "").replace("'", "").replace(" ", "") | |
| if stripped and stripped == stripped.upper() and any(c.isalpha() for c in stripped): | |
| return True | |
| tokens = h.split() | |
| if len(tokens) < 2: | |
| return False | |
| for tok in tokens: | |
| clean = tok.strip(".,;:?!\"'()[]{}").lower() | |
| if clean in _LF_JUNK_TOKENS: | |
| return True | |
| return False | |
| lf_path = base_path / "data" / "processed" / "la_flesche_1932_dictionary.jsonl" | |
| if lf_path.exists(): | |
| with open(lf_path) as f: | |
| for i, line in enumerate(f): | |
| if not line.strip(): | |
| continue | |
| e = _json.loads(line) | |
| h = e.get("headword_lf", "") | |
| if "g" not in h.lower() or "gth" in h.lower(): | |
| continue | |
| if e.get("headword_lf_corrected"): | |
| continue | |
| if _lf_is_junk(h): | |
| continue | |
| out["items"].append({ | |
| "lf_index": i, | |
| "headword_lf": h, | |
| "page_pdf": e.get("page_pdf", 0), | |
| "page_book": e.get("page_book"), | |
| "definitions_lf": e.get("definitions_lf", []), | |
| }) | |
| if len(out["items"]) >= limit: | |
| break | |
| elif category == "reports": | |
| fb_path = base_path / "data" / "feedback" / "reports.jsonl" | |
| if fb_path.exists(): | |
| with open(fb_path) as f: | |
| for line in f: | |
| if not line.strip(): | |
| continue | |
| out["items"].append(_json.loads(line)) | |
| if len(out["items"]) >= limit: | |
| break | |
| else: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail=f"unknown category: {category!r}") | |
| out["count"] = len(out["items"]) | |
| return out | |
| class ReviewDecision(BaseModel): | |
| category: str = Field(..., description="grammar | lf | reports") | |
| item_id: str = Field(..., description="grammar item id, or `lf_<index>`, or report id/timestamp") | |
| decision: str = Field(..., description="approve|reject|revise (grammar) | cedilla|real_g|skip (lf) | fixed|broken|dismiss (reports)") | |
| reviewer: str = Field("", description="Speaker initials or name") | |
| notes: str = Field("", description="Optional reviewer notes / revision request / cedilla positions") | |
| corrected_headword: str = Field("", description="LF only: the corrected headword string with ç placed where it belongs") | |
| async def review_decision(request: Request, body: ReviewDecision): | |
| """Record a speaker's review decision. | |
| All decisions are appended to data/feedback/review_decisions.jsonl as an | |
| immutable audit log. Some decisions also update canonical sources: | |
| grammar: approve/reject → rewrite the item's `reviewed` field in its source JSON | |
| lf: cedilla → append to lf_visual_corrections.jsonl with speaker_review source | |
| reports: fixed/broken → append to user_reports_{fixed,broken}.jsonl | |
| """ | |
| import json as _json | |
| import datetime as _dt | |
| _check_review_token(request) | |
| base_path = Path(__file__).resolve().parent.parent | |
| fb_dir = base_path / "data" / "feedback" | |
| fb_dir.mkdir(exist_ok=True) | |
| audit_path = fb_dir / "review_decisions.jsonl" | |
| record = { | |
| "ts": _dt.datetime.utcnow().isoformat(timespec="seconds") + "Z", | |
| "category": body.category, | |
| "item_id": body.item_id, | |
| "decision": body.decision, | |
| "reviewer": body.reviewer.strip() or "anonymous", | |
| "notes": body.notes, | |
| } | |
| if body.corrected_headword: | |
| record["corrected_headword"] = body.corrected_headword | |
| # Always append to audit log (idempotent enough — duplicate decisions just re-log). | |
| with open(audit_path, "a", encoding="utf-8") as f: | |
| f.write(_json.dumps(record, ensure_ascii=False) + "\n") | |
| applied = {"audit": True} | |
| if body.category == "grammar": | |
| # Update the item's reviewed field in its source JSON | |
| files = [ | |
| "seed_templates.json", | |
| "verb_paradigms_generated.json", | |
| "grammar_explanations_generated.json", | |
| "error_corrections_generated.json", | |
| "error_corrections_advanced.json", | |
| "conjugation_drills_generated.json", | |
| "sentence_breakdowns_generated.json", | |
| "translation_exercises_generated.json", | |
| "vocabulary_exercises_generated.json", | |
| "common_qa_generated.json", | |
| ] | |
| new_status = {"approve": "approved", "reject": "rejected", "revise": "pending"}.get(body.decision) | |
| if new_status is None: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail=f"unknown grammar decision: {body.decision!r}") | |
| found = False | |
| for fname in files: | |
| fpath = base_path / "data" / "grammar_content" / fname | |
| if not fpath.exists(): | |
| continue | |
| items = _json.loads(fpath.read_text()) | |
| changed = False | |
| for it in items: | |
| if it.get("id") == body.item_id: | |
| it["reviewed"] = new_status | |
| if record["reviewer"]: | |
| it["reviewer"] = record["reviewer"] | |
| it["reviewed_at"] = record["ts"][:10] | |
| if body.notes: | |
| it["review_notes"] = body.notes | |
| changed = True | |
| found = True | |
| break | |
| if changed: | |
| fpath.write_text(_json.dumps(items, ensure_ascii=False, indent=2) + "\n") | |
| applied["file"] = fname | |
| break | |
| applied["item_found"] = found | |
| elif body.category == "lf": | |
| # Parse lf_<index> id | |
| if not body.item_id.startswith("lf_"): | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail="lf decisions need item_id like 'lf_<index>'") | |
| try: | |
| lf_idx = int(body.item_id[3:]) | |
| except ValueError: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail="lf_<index> must be an integer") | |
| # For 'cedilla' decision, append to lf_visual_corrections.jsonl | |
| if body.decision == "cedilla": | |
| if not body.corrected_headword: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail="cedilla decision requires corrected_headword") | |
| # Find original headword for the index | |
| lf_path = base_path / "data" / "processed" / "la_flesche_1932_dictionary.jsonl" | |
| orig_h = None | |
| page_pdf = 0 | |
| with open(lf_path) as f: | |
| for i, line in enumerate(f): | |
| if i == lf_idx: | |
| e = _json.loads(line) | |
| orig_h = e.get("headword_lf", "") | |
| page_pdf = e.get("page_pdf", 0) | |
| break | |
| if orig_h is None: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail=f"lf_index {lf_idx} not found") | |
| positions = [j for j, (a, b) in enumerate(zip(orig_h, body.corrected_headword)) if a != b] | |
| corr_path = base_path / "data" / "processed" / "lf_visual_corrections.jsonl" | |
| row = { | |
| "lf_index": lf_idx, | |
| "lf_headword": orig_h, | |
| "lf_page_pdf": page_pdf, | |
| "corrected_headword": body.corrected_headword, | |
| "positions_changed": positions, | |
| "correction_source": "speaker_review", | |
| "confidence": "high", | |
| "reviewer": record["reviewer"], | |
| "reviewed_at": record["ts"][:10], | |
| "rule": f"speaker review — {body.notes}" if body.notes else "speaker review", | |
| } | |
| with open(corr_path, "a", encoding="utf-8") as f: | |
| f.write(_json.dumps(row, ensure_ascii=False) + "\n") | |
| applied["lf_correction_appended"] = True | |
| elif body.decision in ("real_g", "skip"): | |
| # Just log — no data change. Future runs of correction-mining can read the | |
| # audit log to skip these indices. | |
| pass | |
| else: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail=f"unknown lf decision: {body.decision!r}") | |
| elif body.category == "reports": | |
| # Move report to fixed/broken/dismiss | |
| reports_path = fb_dir / "reports.jsonl" | |
| eval_dir = base_path / "data" / "eval" | |
| target_path = None | |
| if body.decision == "fixed": | |
| target_path = eval_dir / "user_reports_fixed.jsonl" | |
| elif body.decision == "broken": | |
| target_path = eval_dir / "user_reports_broken.jsonl" | |
| elif body.decision != "dismiss": | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail=f"unknown reports decision: {body.decision!r}") | |
| # Read all reports, find the matching one (by timestamp / id), move it. | |
| if reports_path.exists(): | |
| kept, moved = [], [] | |
| with open(reports_path) as f: | |
| for line in f: | |
| if not line.strip(): | |
| continue | |
| rec = _json.loads(line) | |
| if str(rec.get("id", "")) == body.item_id or str(rec.get("timestamp", "")) == body.item_id: | |
| moved.append(rec) | |
| else: | |
| kept.append(rec) | |
| with open(reports_path, "w") as f: | |
| for rec in kept: | |
| f.write(_json.dumps(rec, ensure_ascii=False) + "\n") | |
| if target_path and moved: | |
| with open(target_path, "a") as f: | |
| for rec in moved: | |
| rec.setdefault("triaged_by", record["reviewer"]) | |
| rec.setdefault("triaged_at", record["ts"][:10]) | |
| if body.notes: | |
| rec["triage_notes"] = body.notes | |
| f.write(_json.dumps(rec, ensure_ascii=False) + "\n") | |
| applied["reports_moved"] = len(moved) | |
| else: | |
| from fastapi import HTTPException | |
| raise HTTPException(status_code=400, detail=f"unknown category: {body.category!r}") | |
| return {"ok": True, "applied": applied, "audit_record": record} | |