File size: 17,905 Bytes
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
64ce412
6b6e83f
 
 
64ce412
 
 
 
 
 
 
 
 
 
 
 
 
6b6e83f
 
 
 
03dfdfb
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
64ce412
 
 
 
 
 
 
 
 
6b6e83f
 
 
64ce412
6b6e83f
64ce412
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ac9a774
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6b6e83f
 
 
 
 
 
 
 
 
 
 
ac9a774
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
927d2e8
 
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2a8c836
927d2e8
 
2a8c836
 
 
 
 
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
64ce412
6b6e83f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
"""
web_app.py  β€”  Complaint Auto-Routing System Β· Gradio Web Interface
────────────────────────────────────────────────────────────────

Run:
    python app/web_app.py
    # β†’ opens at http://localhost:7860

Install Gradio:
    pip install gradio

No external API keys required.
"""

import os
import sys
import glob
import json
import textwrap

# ─── Proactive Windows FFmpeg PATH Discovery ────────────────────
if sys.platform == "win32":
    # Standard winget installation directory
    winget_packages = os.path.expandvars(r"%LOCALAPPDATA%\Microsoft\WinGet\Packages")
    if os.path.exists(winget_packages):
        # Find any Gyan.FFmpeg bin folder
        ffmpeg_bins = glob.glob(os.path.join(winget_packages, "Gyan.FFmpeg*", "**", "bin"), recursive=True)
        if ffmpeg_bins:
            ffmpeg_path = ffmpeg_bins[0]
            if ffmpeg_path not in os.environ["PATH"]:
                os.environ["PATH"] = ffmpeg_path + os.pathsep + os.environ["PATH"]
                print(f"[Startup] Automatically resolved system FFmpeg path at: {ffmpeg_path}")

sys.path.insert(0, os.path.dirname(os.path.dirname(__file__)))

from inference.engine import ComplaintRoutingEngine, SAVE_DIR

# ─── Automatic Offline Model Training & Setup ───────────────────
def ensure_models_trained():
    required_files = [
        "embedding_engine.pkl",
        "officer_classifier.pkl",
        "priority_classifier.pkl",
        "eta_regressor.pkl",
        "label_encoders.pkl",
        "vector_store.pkl"
    ]
    all_exist = all(os.path.exists(os.path.join(SAVE_DIR, f)) for f in required_files)
    if not all_exist:
        print("[Startup] Missing trained models. Initiating automatic data generation and training...")
        
        # 1. Generate data if missing
        data_path = os.path.join(os.path.dirname(os.path.dirname(__file__)), "data", "synthetic_complaints.csv")
        if not os.path.exists(data_path):
            print("[Startup] Generating synthetic complaints dataset...")
            from data.generate_data import generate_complaints
            df = generate_complaints(n_per_officer=100)
            os.makedirs(os.path.dirname(data_path), exist_ok=True)
            df.to_csv(data_path, index=False)
            print(f"[Startup] Generated {len(df)} complaints.")
            
        # 2. Train models
        print("[Startup] Training models offline...")
        from models.train import main as train_main
        train_main()
        print("[Startup] Model training complete.")

ensure_models_trained()
engine = ComplaintRoutingEngine().load(SAVE_DIR)

# Priority colours (HTML)
PRIORITY_BADGE = {
    "High":   '<span style="display:inline-block;white-space:nowrap;min-width:70px;text-align:center;background:#ef4444;color:#fff;padding:4px 10px;border-radius:4px;font-weight:700;font-size:12px">High</span>',
    "Medium": '<span style="display:inline-block;white-space:nowrap;min-width:70px;text-align:center;background:#f59e0b;color:#fff;padding:4px 10px;border-radius:4px;font-weight:700;font-size:12px">Med</span>',
    "Low":    '<span style="display:inline-block;white-space:nowrap;min-width:70px;text-align:center;background:#22c55e;color:#fff;padding:4px 10px;border-radius:4px;font-weight:700;font-size:12px">Low</span>',
}

DEPT_ICON = {
    "Infrastructure & Roads":    "πŸ›£οΈ",
    "Water & Sanitation":        "πŸ’§",
    "Electricity & Utilities":   "⚑",
    "Public Safety & Security":  "πŸ›‘οΈ",
    "Health & Environment":      "🌿",
    "Land & Property":           "🏠",
    "Transport & Traffic":       "🚌",
    "Administrative Services":   "πŸ“‹",
}


def build_output_html(result: dict) -> str:
    o   = result["officer"]
    p   = result["priority"]
    eta = result["eta_days"]
    sim = result.get("similar_complaints", [])
    icon = DEPT_ICON.get(o["department"], "πŸ›οΈ")
    badge = PRIORITY_BADGE.get(p["level"], p["level"])

    # ── Similar complaints table
    sim_rows = ""
    for s in sim:
        snip = textwrap.shorten(s["text_snippet"].replace("…", ""), width=80)
        sb   = PRIORITY_BADGE.get(s["priority"], s["priority"])
        sim_rows += f"""
        <tr>
          <td style="padding:8px 8px;font-size:12px;color:#475569">{s['complaint_id']}</td>
          <td style="padding:8px 8px;font-size:12px;color:#1e293b;line-height:1.4">{snip}</td>
          <td style="padding:8px 8px;text-align:center">{sb}</td>
          <td style="padding:8px 8px;text-align:center;font-size:12px;color:#1e293b;font-weight:600">{s['eta_days']}d</td>
          <td style="padding:8px 8px;text-align:center;font-size:12px;color:#6366f1;font-weight:600">{s['similarity_score']:.3f}</td>
        </tr>"""

    # Check for audio/video transcription text
    transcription_section = ""
    if result.get("source_text"):
        transcription_section = f"""
  <div style="background:#f8fafc;border:1px solid #e2e8f0;border-radius:10px;padding:14px;margin-bottom:16px">
    <div style="font-size:11px;color:#64748b;font-weight:600;text-transform:uppercase;letter-spacing:.5px;margin-bottom:6px">Transcribed Text</div>
    <div style="font-size:13px;color:#1e293b;line-height:1.5;font-style:italic">"{result['source_text']}"</div>
  </div>"""

    html = f"""
<div style="font-family:Inter,system-ui,sans-serif;max-width:700px">

  {transcription_section}

  <div style="display:grid;grid-template-columns:1fr 1fr 1fr;gap:12px;margin-bottom:16px">
    <div style="background:#f0f9ff;border:1px solid #bae6fd;border-radius:10px;padding:14px">
      <div style="font-size:11px;color:#0369a1;font-weight:600;text-transform:uppercase;letter-spacing:.5px">Assigned Officer</div>
      <div style="font-size:20px;margin:4px 0"></div>
      <div style="font-weight:700;font-size:15px;color:#0c4a6e">{o['name']}</div>
      <div style="font-size:12px;color:#0369a1;margin-top:2px">{o['department']}</div>
      <div style="font-size:11px;color:#94a3b8;margin-top:4px">{o['id']}  Β·  {o['confidence']}% conf.</div>
    </div>

    <div style="background:#fefce8;border:1px solid #fde68a;border-radius:10px;padding:14px">
      <div style="font-size:11px;color:#92400e;font-weight:600;text-transform:uppercase;letter-spacing:.5px">Priority</div>
      <div style="font-size:20px;margin:4px 0"></div>
      <div style="margin-top:4px">{badge}</div>
      <div style="font-size:11px;color:#94a3b8;margin-top:6px">{p['confidence']}% confidence</div>
    </div>

    <div style="background:#f0fdf4;border:1px solid #bbf7d0;border-radius:10px;padding:14px">
      <div style="font-size:11px;color:#166534;font-weight:600;text-transform:uppercase;letter-spacing:.5px">Est. Resolution</div>
      <div style="font-size:20px;margin:4px 0"></div>
      <div style="font-weight:700;font-size:22px;color:#14532d">{eta}</div>
      <div style="font-size:12px;color:#166534">day(s)</div>
    </div>
  </div>

  <div style="background:#fafafa;border:1px solid #e2e8f0;border-radius:10px;padding:14px">
    <div style="font-size:12px;font-weight:700;color:#1e293b;margin-bottom:8px">Similar Past Complaints (Top {len(sim)})</div>
    <table style="width:100%;border-collapse:collapse">
      <thead>
        <tr style="border-bottom:1px solid #e2e8f0">
          <th style="text-align:left;font-size:11px;color:#475569;padding:6px 8px;width:70px">ID</th>
          <th style="text-align:left;font-size:11px;color:#475569;padding:6px 8px">Snippet</th>
          <th style="font-size:11px;color:#475569;padding:6px 8px;width:105px;text-align:center">Priority</th>
          <th style="font-size:11px;color:#475569;padding:6px 8px;width:60px;text-align:center">ETA</th>
          <th style="font-size:11px;color:#475569;padding:6px 8px;width:60px;text-align:center">Score</th>
        </tr>
      </thead>
      <tbody>{sim_rows}</tbody>
    </table>
  </div>

</div>
"""
    return html


def route_text_complaint(text: str, top_k: int) -> tuple:
    if not text.strip():
        return "<p style='color:red'>Please enter complaint text.</p>", ""
    result = engine.predict(text.strip(), top_k_similar=int(top_k))
    html   = build_output_html(result)
    raw    = json.dumps(result, indent=2, ensure_ascii=False)
    return html, raw


def route_audio_complaint(audio_file, top_k: int) -> tuple:
    if audio_file is None:
        return "<p style='color:red'>Please upload an audio file.</p>", ""
    try:
        result = engine.process(audio_path=audio_file, top_k=int(top_k))
    except ImportError as e:
        return f"<p style='color:orange;font-weight:600'>Dependency Error: {e}</p>", ""
    except Exception as e:
        err_msg = str(e)
        if "ffmpeg" in err_msg.lower() or "winerror 2" in err_msg.lower():
            return (
                "<div style='background:#fffbeb;border:1px solid #fef3c7;padding:16px;border-radius:8px;color:#b45309;line-height:1.5'>"
                "<strong>System Configuration Error: FFmpeg not detected!</strong><br/>"
                "Whisper requires FFmpeg to process and decode audio uploads.<br/><br/>"
                "<strong>To fix this on Windows:</strong><br/>"
                "1. Open PowerShell as Administrator and run: <code>winget install Gyan.FFmpeg</code><br/>"
                "2. <strong>Crucial:</strong> Close and restart your IDE (VS Code), terminal, or command prompt so Windows reloads the new PATH system variable.<br/>"
                "3. Restart the web app and try again."
                "</div>",
                f"Error details: {err_msg}"
            )
        return f"<div style='color:red;padding:12px;border:1px solid #fecaca;background:#fef2f2;border-radius:8px;line-height:1.5'><strong>Error processing audio:</strong> {err_msg}</div>", f"Error details: {err_msg}"
    html = build_output_html(result)
    raw  = json.dumps(result, indent=2, ensure_ascii=False)
    return html, raw


def route_video_complaint(video_file, top_k: int) -> tuple:
    if video_file is None:
        return "<p style='color:red'>Please upload a video file.</p>", ""
    try:
        result = engine.process(video_path=video_file, top_k=int(top_k))
    except ImportError as e:
        return f"<p style='color:orange;font-weight:600'>Dependency Error: {e}</p>", ""
    except Exception as e:
        err_msg = str(e)
        if "ffmpeg" in err_msg.lower() or "winerror 2" in err_msg.lower():
            return (
                "<div style='background:#fffbeb;border:1px solid #fef3c7;padding:16px;border-radius:8px;color:#b45309;line-height:1.5'>"
                "<strong>System Configuration Error: FFmpeg not detected!</strong><br/>"
                "Whisper requires FFmpeg to extract audio from video uploads.<br/><br/>"
                "<strong>To fix this on Windows:</strong><br/>"
                "1. Open PowerShell as Administrator and run: <code>winget install Gyan.FFmpeg</code><br/>"
                "2. <strong>Crucial:</strong> Close and restart your IDE (VS Code), terminal, or command prompt so Windows reloads the new PATH system variable.<br/>"
                "3. Restart the web app and try again."
                "</div>",
                f"Error details: {err_msg}"
            )
        return f"<div style='color:red;padding:12px;border:1px solid #fecaca;background:#fef2f2;border-radius:8px;line-height:1.5'><strong>Error processing video:</strong> {err_msg}</div>", f"Error details: {err_msg}"
    html = build_output_html(result)
    raw  = json.dumps(result, indent=2, ensure_ascii=False)
    return html, raw


def create_app():
    try:
        import gradio as gr
    except ImportError:
        raise ImportError("pip install gradio  # then retry")

    EXAMPLE_COMPLAINTS = [
        ["There is a massive pothole on Brigade Road near the hospital causing accidents. URGENT! People are in immediate danger.", 5],
        ["My ration card application has been pending for 45 days. Not urgent, but the matter needs attention when convenient.", 5],
        ["Sewage water overflowing near Central Park. This is a serious problem affecting daily life. Requesting action at the earliest.", 5],
        ["This is a very serious problem. A live electric wire has fallen near the school. People are in immediate danger. Urgent action needed!", 5],
        ["Bus route 42 from East Colony has been suspended for 10 days without notice. Multiple families are affected.", 5],
        ["Illegal construction is happening on government land near the Railway Station. Not urgent but needs attention.", 5],
    ]

    with gr.Blocks(
        title="Complaint Auto-Routing System",
        theme=gr.themes.Soft(),
        css="""
        * { font-family: 'Inter', 'Segoe UI', system-ui, sans-serif !important; }
        .gradio-container { max-width: 900px !important; margin: 0 auto; }
        footer { display: none; }
        """,
    ) as demo:

        gr.Markdown("""
# Complaint Auto-Routing System

AI/ML system that automatically routes complaints to the right officer, predicts priority and resolution time, and retrieves similar past complaints β€” **fully offline, no external APIs**.
        """)

        with gr.Tabs():

            # ── Text Tab
            with gr.Tab("Text Complaint"):
                with gr.Row():
                    text_input = gr.Textbox(
                        label="Complaint Text",
                        placeholder="Describe your complaint here in English…",
                        lines=4,
                    )
                    top_k_text = gr.Slider(1, 10, value=5, step=1, label="Similar complaints")
                text_btn    = gr.Button("Route Complaint", variant="primary")
                text_output = gr.HTML(label="Routing Result")

                with gr.Accordion("Show raw JSON", open=False):
                    text_json = gr.Code(label="JSON", language="json")

                gr.Markdown("<br>πŸ’‘ **Tip:** Click on any of the examples below to instantly test the routing system!")

                gr.Examples(
                    examples=EXAMPLE_COMPLAINTS,
                    inputs=[text_input, top_k_text],
                )

                text_btn.click(
                    fn=route_text_complaint,
                    inputs=[text_input, top_k_text],
                    outputs=[text_output, text_json],
                )

            # ── Audio Tab
            with gr.Tab("Audio Complaint"):
                gr.Markdown("""
Upload an audio recording of the complaint in English.
**Requires:** `pip install openai-whisper` (local model, English)
                """)
                audio_input = gr.Audio(type="filepath", label="Audio File (.wav, .mp3, .m4a)")
                top_k_audio = gr.Slider(1, 10, value=5, step=1, label="Similar complaints")
                audio_btn   = gr.Button("Transcribe & Route", variant="primary")
                audio_out   = gr.HTML(label="Result")
                audio_json  = gr.Code(label="JSON", language="json")

                gr.Markdown("<br>πŸ’‘ **Tip:** Click the sample audio file below to test the transcription and routing without needing your own file!")

                gr.Examples(
                    examples=[["data/sample_audio.mp3", 5]],
                    inputs=[audio_input, top_k_audio],
                )

                audio_btn.click(
                    fn=route_audio_complaint,
                    inputs=[audio_input, top_k_audio],
                    outputs=[audio_out, audio_json],
                )

            # ── Video Tab
            with gr.Tab("Video Complaint"):
                gr.Markdown("""
Upload a video of the complainant speaking.
**Requires:** `pip install openai-whisper` + `ffmpeg` installed system-wide.
                """)
                video_input = gr.Video(label="Video File (.mp4, .mkv, .avi)")
                top_k_video = gr.Slider(1, 10, value=5, step=1, label="Similar complaints")
                video_btn   = gr.Button("Extract Audio & Route", variant="primary")
                video_out   = gr.HTML(label="Result")
                video_json  = gr.Code(label="JSON", language="json")
                video_btn.click(
                    fn=route_video_complaint,
                    inputs=[video_input, top_k_video],
                    outputs=[video_out, video_json],
                )

            # ── About Tab
            with gr.Tab("About"):
                gr.Markdown("""
## Architecture

| Component | Model | Notes |
|-----------|-------|-------|
| **Embeddings** | TF-IDF + SVD (256-dim) | Offline baseline; swap for `all-MiniLM-L6-v2` (English SentenceTransformer) |
| **Officer Routing** | SVM (RBF kernel) | 8-class, probability calibrated |
| **Priority** | Random Forest | High / Medium / Low |
| **ETA Prediction** | Gradient Boosting Regressor | MAE β‰ˆ 5–8 days |
| **Similarity Search** | Cosine over NumPy matrix | FAISS drop-in available |
| **Audio/Video** | Whisper (local) | English, fully offline |

## Officers
| ID | Name | Department |
|----|------|-----------|
| OFF001 | Rahul Sharma | Infrastructure & Roads |
| OFF002 | Priya Mehta | Water & Sanitation |
| OFF003 | Amit Verma | Electricity & Utilities |
| OFF004 | Sunita Patel | Public Safety & Security |
| OFF005 | Vijay Kumar | Health & Environment |
| OFF006 | Anjali Singh | Land & Property |
| OFF007 | Ravi Nair | Transport & Traffic |
| OFF008 | Meena Reddy | Administrative Services |

## No External APIs
All inference happens locally. Models are trained from scratch on synthetic data.
For production, replace synthetic data with real complaint records.
                """)

    return demo


if __name__ == "__main__":
    app = create_app()
    app.launch(server_name="0.0.0.0", server_port=7860, share=False)