Spaces:

VEDAGI1
/

Medica_DecisionSupportAI

Sleeping

App Files Files Community

Rajan Sharma commited on Sep 21

Commit

f6cdc91

verified ·

1 Parent(s): ed780b7

Update app.py

Browse files

Files changed (1) hide show

app.py +87 -83

app.py CHANGED Viewed

@@ -47,13 +47,12 @@ from huggingface_hub import login
 from safety import safety_filter, refusal_reply
 from retriever import init_retriever, retrieve_context
-from decision_math import compute_operational_numbers   # fixed import name
 from prompt_templates import build_system_preamble
 from upload_ingest import extract_text_from_files
 from session_rag import SessionRAG
-from mdsi_analysis import capacity_projection, cost_estimate, outcomes_summary
-# NEW: dynamic data plumbing
 from data_registry import DataRegistry
 from schema_mapper import map_concepts, build_phase1_questions
 from auto_metrics import build_data_findings_markdown
@@ -68,23 +67,21 @@ USE_HOSTED_COHERE = bool(COHERE_API_KEY and _HAS_COHERE)
 # Larger output budget for Phase 2
 MAX_NEW_TOKENS = int(os.getenv("MAX_NEW_TOKENS", "2048"))
-# ---------- System Master (Phase 2) ----------
 SYSTEM_MASTER = """
 SYSTEM ROLE
-You are ClarityOps, a medical analytics system that interacts only via this chat.
 Absolute rules:
 - Use ONLY information provided in this conversation (scenario text + uploaded files + user answers).
 - Never invent data. If something required is missing after clarifications, write the literal token: INSUFFICIENT_DATA.
-- Produce clear calculations (show multipliers and totals), follow medical units, and keep privacy safeguards (aggregate; suppress cohorts <10).
-Formatting hard rules for Phase 2:
-- Start with the header: “Structured Analysis”
-- Follow this section order:
-  1. Prioritization
-  2. Capacity
-  3. Cost
-  4. Clinical Benefits
-  5. ClarityOps Top 3 Recommendations
-- End with a brief “Provenance” mapping outputs to scenario text, uploaded files, and answers.
 """.strip()
 # ---------- Helpers ----------
@@ -122,10 +119,21 @@ def _sanitize_text(s: str) -> str:
     return re2.sub(r'[\p{C}--[\n\t]]+', '', s)
 def is_scenario_triggered(text: str, uploaded_files_paths) -> bool:
     t = (text or "").lower()
-    has_keyword = "scenario" in t
     has_files = bool(uploaded_files_paths)
-    return has_keyword or has_files
 # ---------- Cohere first ----------
 def cohere_chat(message, history):
@@ -206,39 +214,20 @@ def local_generate(model, tokenizer, input_ids, max_new_tokens=MAX_NEW_TOKENS):
 # ---------- Snapshot & retrieval ----------
 def _load_snapshot(path=SNAPSHOT_PATH):
     try:
         with open(path, "r", encoding="utf-8") as f:
             return json.load(f)
     except Exception:
-        return {
-            "timestamp": None, "beds_total": 400, "staffed_ratio": 1.0, "occupied_pct": 0.97,
-            "ed_census": 62, "ed_admits_waiting": 19, "avg_ed_wait_hours": 8,
-            "discharge_ready_today": 11, "discharge_barriers": {"allied_health": 7, "placement": 4},
-            "rn_shortfall": {"med_ward_A": 1, "med_ward_B": 1},
-            "forecast_admits_next_24h": {"respiratory": 14, "other": 9},
-            "isolation_needs_waiting": {"contact": 3, "airborne": 1}, "telemetry_needed_waiting": 5
-        }
 init_retriever()
 _session_rag = SessionRAG()
-# ---------- Executive pre-compute (MDSi block) ----------
-def _mdsi_block():
-    base_capacity = capacity_projection(18, 48, 6)
-    cons_capacity = capacity_projection(12, 48, 6)
-    opt_capacity = capacity_projection(24, 48, 6)
-    cost_1200 = cost_estimate(1200, 74.0, 75000.0)
-    outcomes = outcomes_summary()
-    return json.dumps({
-        "capacity_projection": {"conservative": cons_capacity, "base": base_capacity, "optimistic": opt_capacity},
-        "cost_for_1200": cost_1200,
-        "outcomes_summary": outcomes
-    }, indent=2)
 # NEW: session-scoped data registry
 _data_registry = DataRegistry()
-# ---------- Core chat logic (auto scenario, dynamic Phase 1) ----------
 def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answers=False):
     try:
         log_event("user_message", None, {"sizes": {"chars": len(user_msg or "")}})
@@ -249,10 +238,10 @@ def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answe
             return history + [(user_msg, ans)], awaiting_answers
         if is_identity_query(safe_in, history):
-            ans = "I am ClarityOps, your strategic decision making AI partner."
             return history + [(user_msg, ans)], awaiting_answers
-        # 1) Ingest uploads into RAG AND DataRegistry (files alone can trigger Scenario Mode)
         artifacts = []
         if uploaded_files_paths:
             ing = extract_text_from_files(uploaded_files_paths)
@@ -269,7 +258,7 @@ def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answe
                 "chunks": len(chunks), "artifacts": len(artifacts), "tables": len(_data_registry.names())
             })
-        # quick helper
         if re.search(r"\b(columns?|headers?)\b", (safe_in or "").lower()):
             cols = _session_rag.get_latest_csv_columns()
             if cols:
@@ -302,12 +291,12 @@ def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answe
             })
             return history + [(user_msg, safe_out)], awaiting_answers
-        # ---------- Scenario Mode ----------
         # 3) Build dynamic concept mapping from scenario + data
         mapping = map_concepts(safe_in, _data_registry)
         if not awaiting_answers:
-            # PHASE 1: ask only for missing/ambiguous
             phase1 = build_phase1_questions(scenario_text=safe_in, registry=_data_registry, mapping=mapping)
             phase1 = _sanitize_text(phase1)
             log_event("assistant_reply", None, {
@@ -318,56 +307,52 @@ def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answe
             })
             return history + [(user_msg, phase1)], True
-        # PHASE 2: compute data findings in Python, then let LLM write the narrative
         data_findings_md, missing_keys = build_data_findings_markdown(_data_registry, mapping)
-        # If critical missing items remain, surface INSUFFICIENT_DATA context to the model + ask for the rest
-        insuff_note = ""
         if missing_keys:
-            insuff_note = (
-                "\n\nUncomputable (still missing columns/defs): "
                 + ", ".join(sorted(set(missing_keys)))
-                + ". If any of these are essential to the requested outputs, write INSUFFICIENT_DATA where appropriate."
             )
-        # Preamble context (snapshot + policy)
-        session_snips = "\n---\n".join(_session_rag.retrieve(
-            "diabetes screening Indigenous Métis mobile program cost throughput outcomes logistics",
-            k=6
-        ))
         snapshot = _load_snapshot()
-        policy_context = retrieve_context(
-            "mobile diabetes screening Indigenous community outreach cultural safety data governance outcomes"
-        )
-        computed = compute_operational_numbers(snapshot)
-        user_lower = (safe_in or "").lower()
-        mdsi_extra = ""
-        if any(k in user_lower for k in ["diabetes", "mdsi", "mobile screening"]):
-            mdsi_extra = _mdsi_block()
-        # Build artifact + table summary for the prompt
         registry_summary = _data_registry.summarize_for_prompt()
-        artifact_block = "Uploaded Data Files (tables):\n" + registry_summary
         scenario_block = safe_in if len((safe_in or "")) > 0 else ""
         system_preamble = build_system_preamble(
             snapshot=snapshot,
             policy_context=policy_context,
-            computed_numbers=computed,
-            scenario_text=scenario_block + f"\n\n{artifact_block}\n\n{data_findings_md}" + (f"\n\nExecutive Pre-Computed Blocks:\n{mdsi_extra}" if mdsi_extra else "") + insuff_note,
             session_snips=session_snips
         )
         directive = (
-            "\n\n[INSTRUCTION TO MODEL]\n"
-            "Produce **Phase 2** now: begin with 'Structured Analysis' and follow the exact section order "
-            "(Prioritization, Capacity, Cost, Clinical Benefits, ClarityOps Top 3 Recommendations). "
-            "Use the **Python-computed tables** in the context as ground truth; when something is truly missing, write INSUFFICIENT_DATA. "
-            "Show calculations, units, and add a brief Provenance.\n"
         )
-        augmented_user = SYSTEM_MASTER + "\n\n" + system_preamble + "\n\nUser scenario & answers:\n" + safe_in + directive
         out = cohere_chat(augmented_user, history)
         if not out:
@@ -402,10 +387,28 @@ def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answe
             pass
         return history + [(user_msg, err)], awaiting_answers
 # ---------- Theme & CSS ----------
 theme = gr.themes.Soft(primary_hue="teal", neutral_hue="slate", radius_size=gr.themes.sizes.radius_lg)
 custom_css = """
-:root { --brand-bg: #0f172a; --brand-accent: #0d9488; --brand-text: #0f172a; --brand-text-light: #ffffff; }  /* bg same as chat for integrated look */
 html, body, .gradio-container { height: 100vh; }
 .gradio-container { background: var(--brand-bg); display: flex; flex-direction: column; }
@@ -433,33 +436,33 @@ textarea, input, .gr-input { border-radius: 12px !important; }
 # ---------- UI ----------
 with gr.Blocks(theme=theme, css=custom_css, analytics_enabled=False) as demo:
-    # --- HERO (initial Google-like screen) ---
     with gr.Column(elem_id="hero-wrap", visible=True) as hero_wrap:
         with gr.Column(elem_id="hero"):
-            gr.HTML("<h2>What can I assist with?</h2>")
             with gr.Row(elem_classes="search-row"):
                 hero_msg = gr.Textbox(
-                    placeholder="Ask anything (type 'scenario' and/or attach files for Scenario Mode)…",
                     show_label=False,
                     lines=1,
                     elem_classes="hero-box"
                 )
                 hero_send = gr.Button("➤", scale=0, elem_id="hero-send")
-            gr.Markdown('<div class="hint">Scenario Mode triggers when you type the word <b>scenario</b> or upload files. Phase&nbsp;1 asks dynamic clarifications; Phase&nbsp;2 returns a structured analysis.</div>')
     # --- MAIN APP (hidden until first message) ---
     with gr.Column(elem_id="chat-container", visible=False) as app_wrap:
         chat = gr.Chatbot(label="", show_label=False, height="80vh")
         with gr.Row():
             uploads = gr.Files(
-                label="Upload docs/images (PDF, DOCX, CSV, PNG, JPG)",
                 file_types=["file"], file_count="multiple", height=68
             )
         with gr.Row(elem_id="chat-input-row"):
             msg = gr.Textbox(
                 label="",
                 show_label=False,
-                placeholder="Continue here. Paste scenario details (include the word 'scenario' to trigger), add files above.",
                 scale=10,
                 elem_id="chat-msg",
                 lines=1,
@@ -529,8 +532,9 @@ with gr.Blocks(theme=theme, css=custom_css, analytics_enabled=False) as demo:
                concurrency_limit=2, queue=True)
     def _on_clear():
-        # Also clear the in-memory data registry for a fresh scenario
         _data_registry.clear()
         return (
             [], "", [], False,
             gr.update(visible=True),
@@ -542,7 +546,7 @@ with gr.Blocks(theme=theme, css=custom_css, analytics_enabled=False) as demo:
 if __name__ == "__main__":
     port = int(os.environ.get("PORT", "7860"))
-    demo.launch(server_name="0.0.0.0", server_port=port, show_api=False, max_threads=8)

 from safety import safety_filter, refusal_reply
 from retriever import init_retriever, retrieve_context
+from decision_math import compute_operational_numbers
 from prompt_templates import build_system_preamble
 from upload_ingest import extract_text_from_files
 from session_rag import SessionRAG
+# NEW: dynamic data analysis framework
 from data_registry import DataRegistry
 from schema_mapper import map_concepts, build_phase1_questions
 from auto_metrics import build_data_findings_markdown
 # Larger output budget for Phase 2
 MAX_NEW_TOKENS = int(os.getenv("MAX_NEW_TOKENS", "2048"))
+# ---------- Generic System Prompt ----------
 SYSTEM_MASTER = """
 SYSTEM ROLE
+You are an AI analytical system that provides data-driven insights for any scenario.
 Absolute rules:
 - Use ONLY information provided in this conversation (scenario text + uploaded files + user answers).
 - Never invent data. If something required is missing after clarifications, write the literal token: INSUFFICIENT_DATA.
+- Provide clear analysis with calculations, evidence, and reasoning.
+- Maintain privacy safeguards (aggregate data; suppress small cohorts <10).
+- Adapt your analysis approach to the specific scenario and data provided.
+Formatting rules for structured analysis:
+- Start with the header: "Structured Analysis"
+- Organize analysis into logical sections based on the scenario requirements
+- End with concrete recommendations and a brief "Provenance" mapping outputs to scenario text, uploaded files, and answers.
 """.strip()
 # ---------- Helpers ----------
     return re2.sub(r'[\p{C}--[\n\t]]+', '', s)
 def is_scenario_triggered(text: str, uploaded_files_paths) -> bool:
+    """Detect if this should be treated as a scenario analysis request."""
     t = (text or "").lower()
+    # Scenario keywords
+    scenario_keywords = [
+        "scenario", "analysis", "analyze", "assess", "evaluate", "recommendation",
+        "strategy", "plan", "solution", "decision", "priority", "allocate", "resource"
+    ]
+    has_keyword = any(keyword in t for keyword in scenario_keywords)
     has_files = bool(uploaded_files_paths)
+    # If files are uploaded, assume scenario mode
+    # If certain analytical keywords are present, assume scenario mode
+    return has_files or has_keyword
 # ---------- Cohere first ----------
 def cohere_chat(message, history):
 # ---------- Snapshot & retrieval ----------
 def _load_snapshot(path=SNAPSHOT_PATH):
+    """Load operational snapshot if available."""
     try:
         with open(path, "r", encoding="utf-8") as f:
             return json.load(f)
     except Exception:
+        return {}  # Return empty dict if no snapshot available
 init_retriever()
 _session_rag = SessionRAG()
 # NEW: session-scoped data registry
 _data_registry = DataRegistry()
+# ---------- Core chat logic (generic scenario handling) ----------
 def clarityops_reply(user_msg, history, tz, uploaded_files_paths, awaiting_answers=False):
     try:
         log_event("user_message", None, {"sizes": {"chars": len(user_msg or "")}})
             return history + [(user_msg, ans)], awaiting_answers
         if is_identity_query(safe_in, history):
+            ans = "I am an AI analytical system designed to help you analyze scenarios and make data-driven decisions."
             return history + [(user_msg, ans)], awaiting_answers
+        # 1) Ingest uploads into RAG AND DataRegistry
         artifacts = []
         if uploaded_files_paths:
             ing = extract_text_from_files(uploaded_files_paths)
                 "chunks": len(chunks), "artifacts": len(artifacts), "tables": len(_data_registry.names())
             })
+        # Quick helper for column inspection
         if re.search(r"\b(columns?|headers?)\b", (safe_in or "").lower()):
             cols = _session_rag.get_latest_csv_columns()
             if cols:
             })
             return history + [(user_msg, safe_out)], awaiting_answers
+        # ---------- Generic Scenario Analysis Mode ----------
         # 3) Build dynamic concept mapping from scenario + data
         mapping = map_concepts(safe_in, _data_registry)
         if not awaiting_answers:
+            # PHASE 1: ask for missing/ambiguous information
             phase1 = build_phase1_questions(scenario_text=safe_in, registry=_data_registry, mapping=mapping)
             phase1 = _sanitize_text(phase1)
             log_event("assistant_reply", None, {
             })
             return history + [(user_msg, phase1)], True
+        # PHASE 2: compute data analysis and generate structured response
         data_findings_md, missing_keys = build_data_findings_markdown(_data_registry, mapping)
+        # Build context for analysis
+        insufficient_data_note = ""
         if missing_keys:
+            insufficient_data_note = (
+                "\n\nData limitations: Missing or uncomputable: "
                 + ", ".join(sorted(set(missing_keys)))
+                + ". Where these are essential to analysis, write INSUFFICIENT_DATA."
             )
+        # Get relevant context from uploaded documents
+        # Extract key terms from scenario to improve retrieval
+        scenario_terms = _extract_key_terms_from_scenario(safe_in)
+        session_snips = "\n---\n".join(_session_rag.retrieve(scenario_terms, k=6))
+        # Load any available operational data
         snapshot = _load_snapshot()
+        computed_numbers = compute_operational_numbers(snapshot) if snapshot else {}
+        # Get general policy/context if available
+        policy_context = retrieve_context(scenario_terms)
+        # Build comprehensive data summary for analysis
         registry_summary = _data_registry.summarize_for_prompt()
+        artifact_block = "Uploaded Data Files:\n" + registry_summary if registry_summary else "No data files uploaded."
         scenario_block = safe_in if len((safe_in or "")) > 0 else ""
         system_preamble = build_system_preamble(
             snapshot=snapshot,
             policy_context=policy_context,
+            computed_numbers=computed_numbers,
+            scenario_text=scenario_block + f"\n\n{artifact_block}\n\n{data_findings_md}" + insufficient_data_note,
             session_snips=session_snips
         )
         directive = (
+            "\n\n[ANALYSIS INSTRUCTION]\n"
+            "Provide a structured analysis appropriate to this scenario. Begin with 'Structured Analysis' and "
+            "organize your response into logical sections based on what the scenario requires. Use the data "
+            "provided as ground truth. When information is missing, write INSUFFICIENT_DATA. Show your reasoning "
+            "and calculations. End with concrete recommendations and a brief Provenance section.\n"
         )
+        augmented_user = SYSTEM_MASTER + "\n\n" + system_preamble + "\n\nScenario and context:\n" + safe_in + directive
         out = cohere_chat(augmented_user, history)
         if not out:
             pass
         return history + [(user_msg, err)], awaiting_answers
+def _extract_key_terms_from_scenario(scenario_text: str) -> str:
+    """Extract key terms from scenario text for better context retrieval."""
+    if not scenario_text:
+        return ""
+    # Simple extraction of important words (remove common stop words)
+    stop_words = {
+        'the', 'and', 'or', 'but', 'in', 'on', 'at', 'to', 'for', 'of', 'with', 'by',
+        'is', 'are', 'was', 'were', 'be', 'been', 'have', 'has', 'had', 'do', 'does', 'did',
+        'a', 'an', 'this', 'that', 'these', 'those', 'i', 'you', 'he', 'she', 'it', 'we', 'they'
+    }
+    words = re.findall(r'\b[a-zA-Z]{3,}\b', scenario_text.lower())
+    key_terms = [word for word in words if word not in stop_words]
+    # Return first 10-15 key terms
+    return ' '.join(key_terms[:15])
 # ---------- Theme & CSS ----------
 theme = gr.themes.Soft(primary_hue="teal", neutral_hue="slate", radius_size=gr.themes.sizes.radius_lg)
 custom_css = """
+:root { --brand-bg: #0f172a; --brand-accent: #0d9488; --brand-text: #0f172a; --brand-text-light: #ffffff; }
 html, body, .gradio-container { height: 100vh; }
 .gradio-container { background: var(--brand-bg); display: flex; flex-direction: column; }
 # ---------- UI ----------
 with gr.Blocks(theme=theme, css=custom_css, analytics_enabled=False) as demo:
+    # --- HERO (initial screen) ---
     with gr.Column(elem_id="hero-wrap", visible=True) as hero_wrap:
         with gr.Column(elem_id="hero"):
+            gr.HTML("<h2>What scenario can I help you analyze?</h2>")
             with gr.Row(elem_classes="search-row"):
                 hero_msg = gr.Textbox(
+                    placeholder="Describe your scenario or ask any question (upload files for data analysis)…",
                     show_label=False,
                     lines=1,
                     elem_classes="hero-box"
                 )
                 hero_send = gr.Button("➤", scale=0, elem_id="hero-send")
+            gr.Markdown('<div class="hint">Upload files and describe your scenario for comprehensive analysis. The system will ask clarifying questions, then provide structured insights.</div>')
     # --- MAIN APP (hidden until first message) ---
     with gr.Column(elem_id="chat-container", visible=False) as app_wrap:
         chat = gr.Chatbot(label="", show_label=False, height="80vh")
         with gr.Row():
             uploads = gr.Files(
+                label="Upload data files (PDF, DOCX, CSV, PNG, JPG)",
                 file_types=["file"], file_count="multiple", height=68
             )
         with gr.Row(elem_id="chat-input-row"):
             msg = gr.Textbox(
                 label="",
                 show_label=False,
+                placeholder="Continue the conversation. Provide additional details or answer clarifying questions.",
                 scale=10,
                 elem_id="chat-msg",
                 lines=1,
                concurrency_limit=2, queue=True)
     def _on_clear():
+        # Clear the in-memory data registry for a fresh scenario
         _data_registry.clear()
+        _session_rag.clear()  # Also clear RAG session if available
         return (
             [], "", [], False,
             gr.update(visible=True),
 if __name__ == "__main__":
     port = int(os.environ.get("PORT", "7860"))
+    demo.launch(server_name="0.0.0.0", server_port=port, show_api=False, max_threads=40)ds=8)