"""AI feature phases only (no Redis, job queue, Temporal, or other backend scale). Phase 1 — Core RAG generation (standard pipeline) Phase 2 — Agentic inspector, vision, optional LLM validators / ingest sanitiser Phase 3 — AI latency & retrieval quality (async LLM, speculation, prompt cache, hybrid RAG) """ from __future__ import annotations from typing import Any from app.agentic.runtime_status import inspector_public_status, is_openai_inspector_live from app.agentic.speculative_executor import speculative_execution_active from app.config import effective_openai_api_key, settings from app.db.database import ( is_sqlite_database, multi_section_parallel_enabled, parallel_section_writes_safe, ) from app.llm.prompt_cache import prompt_caching_active from app.retrieval.semantic_cache import is_semantic_cache_active def _openai_configured() -> bool: return bool(effective_openai_api_key()) def ai_phase1_status() -> dict[str, Any]: """Embeddings, retrieval, standard generate / proofread / enhance.""" backend = (settings.vectorstore_backend or "faiss").strip().lower() return { "name": "core_rag_generation", "openai_required": True, "openai_configured": _openai_configured(), "live": _openai_configured(), "capabilities": { "generate_proofread_enhance": _openai_configured(), "notes_expansion_llm": _openai_configured(), "style_analyzer_llm": _openai_configured(), "postprocess_verify_llm": _openai_configured(), "embeddings": { "prefer_local": bool(settings.prefer_local_embeddings), "local_model": settings.local_embedding_model, "openai_model": settings.embedding_model if not settings.prefer_local_embeddings else None, }, "vector_retrieval": { "backend": backend, "hierarchical_rag": bool(settings.hierarchical_rag_enabled), "hybrid_retrieval": bool(settings.enable_hybrid_retrieval) and backend == "qdrant", "semantic_cache": is_semantic_cache_active(), }, "personalised_style_rag": bool(settings.personalised_style_rag_enabled), "rag_upload_sanitisation": bool(settings.enable_rag_upload_sanitisation), "rag_sanitisation_regex_only": bool(settings.enable_rag_upload_sanitisation) and not bool(getattr(settings, "rag_sanitisation_use_llm", False)), "notes_only_generation": bool(settings.notes_only_generation), "interference_levels": True, "llm_section_validator": bool(settings.llm_section_validator_enabled), }, "flags": { "chat_model": (settings.chat_model or "").strip() or None, "prefer_local_embeddings": bool(settings.prefer_local_embeddings), "vectorstore_backend": backend, "personalised_style_rag_enabled": bool(settings.personalised_style_rag_enabled), "enable_rag_upload_sanitisation": bool(settings.enable_rag_upload_sanitisation), "notes_only_generation": bool(settings.notes_only_generation), }, } def ai_phase2_status() -> dict[str, Any]: """Inspector tool loop, agentic generate, section photo vision.""" key_ok = _openai_configured() vision_on = bool(settings.section_photo_vision_enabled) notes_only = bool(settings.notes_only_generation) agentic_on_notes = bool(settings.agentic_inspector_when_notes_only) post_generate_inspector = (not notes_only or agentic_on_notes) and is_openai_inspector_live() return { "name": "agentic_inspector_and_vision", "openai_required": True, "openai_configured": key_ok, "live": key_ok and (is_openai_inspector_live() or vision_on), "capabilities": { "inspector_tool_agent": bool(settings.inspector_tool_agent), "inspector_live": is_openai_inspector_live(), "post_generate_uses_inspector": post_generate_inspector, "agentic_generate_endpoint": key_ok and is_openai_inspector_live(), "section_photo_vision": key_ok and vision_on, "section_photo_analyze_on_upload": bool( getattr(settings, "section_photo_analyze_on_upload", False) ), "rag_sanitisation_llm_ingest": bool( settings.enable_rag_upload_sanitisation and getattr(settings, "rag_sanitisation_use_llm", False) ), "llm_section_validator": bool(settings.llm_section_validator_enabled), }, "flags": { "inspector_tool_agent": bool(settings.inspector_tool_agent), "inspector_body_model": (settings.inspector_body_model or "").strip() or None, "section_photo_vision_model": ( (getattr(settings, "section_photo_vision_model", None) or "").strip() or (settings.chat_model or "").strip() or "gpt-4o" ), "agentic_inspector_when_notes_only": agentic_on_notes, "primary_generate_pipeline": (settings.primary_generate_pipeline or "agentic") .strip() .lower(), "production_ai_profile": bool(getattr(settings, "production_ai_profile", False)), }, "rics_inspector": inspector_public_status(), } def ai_phase3_status() -> dict[str, Any]: """Faster / cheaper OpenAI usage and stronger retrieval — not deployment queues.""" async_on = bool(settings.enable_async_pipeline) backend = (settings.vectorstore_backend or "faiss").strip().lower() sla_s = int(getattr(settings, "generation_sla_seconds", 600)) return { "name": "ai_latency_and_retrieval_quality", "openai_required": True, "openai_configured": _openai_configured(), "live": _openai_configured() and async_on, "generation_sla_seconds": sla_s, "generation_timeout_seconds": int(getattr(settings, "generation_timeout_seconds", 720)), "capabilities": { "async_llm_pipeline": async_on, "parallel_multi_section": multi_section_parallel_enabled(), "speculative_inspector_tools": speculative_execution_active(), "openai_prompt_cache": prompt_caching_active(), "hybrid_retrieval_active": bool(settings.enable_hybrid_retrieval) and backend == "qdrant", "semantic_retrieval_cache_active": is_semantic_cache_active(), }, "flags": { "enable_async_pipeline": async_on, "enable_speculative_executor": bool(settings.enable_speculative_executor), "enable_prompt_caching": bool(settings.enable_prompt_caching), "max_concurrent_llm_calls": int(settings.max_concurrent_llm_calls), "vectorstore_backend": backend, "enable_hybrid_retrieval": bool(settings.enable_hybrid_retrieval), "semantic_cache_enabled": bool(settings.semantic_cache_enabled), }, } def collect_ai_phases() -> dict[str, Any]: return { "phase1": ai_phase1_status(), "phase2": ai_phase2_status(), "phase3": ai_phase3_status(), } def collect_ai_phase_warnings() -> list[str]: """Misconfigurations that weaken AI behaviour (excludes Redis/Temporal/job queue).""" warnings: list[str] = [] key_ok = _openai_configured() if not key_ok: warnings.append( "OPENAI_API_KEY unset: all LLM phases (generation, inspector, vision) are off." ) if key_ok and not settings.section_photo_vision_enabled: warnings.append( "SECTION_PHOTO_VISION_ENABLED=false: photos are stored but not analysed by vision at generate." ) if key_ok and not settings.inspector_tool_agent: warnings.append( "INSPECTOR_TOOL_AGENT=false: agentic paths use the legacy fixed pipeline, not tool-calling." ) if ( key_ok and settings.notes_only_generation and not settings.agentic_inspector_when_notes_only ): warnings.append( "Phase 2 inactive on POST /generate: NOTES_ONLY_GENERATION=true and " "AGENTIC_INSPECTOR_WHEN_NOTES_ONLY=false. Use agentic/generate or enable " "AGENTIC_INSPECTOR_WHEN_NOTES_ONLY." ) if key_ok and settings.enable_rag_upload_sanitisation and getattr( settings, "rag_sanitisation_use_llm", False ): warnings.append( "RAG_SANITISATION_USE_LLM=true: ingest competes with generation/vision quota; " "use regex-only in production (auto-off on HF via production_ai_profile)." ) if not settings.enable_rag_upload_sanitisation and settings.personalised_style_rag_enabled: warnings.append( "ENABLE_RAG_UPLOAD_SANITISATION=false with personalised RAG: uploads indexed without PII redaction." ) if settings.enable_speculative_executor and not speculative_execution_active(): warnings.append( "Phase 3: ENABLE_SPECULATIVE_EXECUTOR=true but speculation inactive " "(requires ENABLE_ASYNC_PIPELINE=true and Phase 2 inspector path)." ) if settings.enable_prompt_caching and not prompt_caching_active(): warnings.append( "Phase 3: ENABLE_PROMPT_CACHING=true requires ENABLE_ASYNC_PIPELINE=true." ) if settings.enable_async_pipeline and not key_ok: warnings.append("Phase 3: ENABLE_ASYNC_PIPELINE=true has no effect without OPENAI_API_KEY.") backend = (settings.vectorstore_backend or "faiss").strip().lower() if settings.enable_hybrid_retrieval and backend != "qdrant": warnings.append( "Phase 3: ENABLE_HYBRID_RETRIEVAL=true needs VECTORSTORE_BACKEND=qdrant." ) if settings.semantic_cache_enabled and not is_semantic_cache_active(): warnings.append( "Phase 3: SEMANTIC_CACHE_ENABLED=true needs VECTORSTORE_BACKEND=qdrant." ) sla_s = int(getattr(settings, "generation_sla_seconds", 600)) if key_ok and not multi_section_parallel_enabled(): warnings.append( f"Full-report SLA ({sla_s}s): ENABLE_ASYNC_PIPELINE=true and parallel sections " "are required; sequential generation often exceeds 10 minutes." ) elif key_ok and is_sqlite_database() and not parallel_section_writes_safe(): warnings.append( f"Full-report SLA ({sla_s}s): set ALLOW_SQLITE_PARALLEL_SECTIONS=true or use PostgreSQL." ) if ( settings.enable_speculative_executor and settings.enable_async_pipeline and settings.notes_only_generation and not settings.agentic_inspector_when_notes_only ): warnings.append( "Phase 3 speculation enabled but POST /generate skips the inspector (Phase 2 off on that path)." ) return warnings