File size: 26,781 Bytes
7388f6f
ba1f020
7388f6f
bc001cf
ba1f020
 
 
7388f6f
 
 
ba1f020
7388f6f
 
ba1f020
 
7388f6f
 
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
 
ba1f020
 
 
 
 
 
 
 
 
 
 
 
7388f6f
ba1f020
7388f6f
 
 
 
d42cb0e
7388f6f
 
 
 
 
ba1f020
7388f6f
 
 
ba1f020
7388f6f
 
 
ba1f020
7388f6f
 
 
ba1f020
7388f6f
 
 
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d42cb0e
ba1f020
ce4c24f
d42cb0e
ba1f020
 
d42cb0e
ba1f020
 
 
 
 
 
 
 
 
 
d42cb0e
 
ce4c24f
7388f6f
ba1f020
7388f6f
ba1f020
d42cb0e
7388f6f
ba1f020
 
 
7388f6f
ba1f020
 
 
 
 
7388f6f
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
ba1f020
7388f6f
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
 
 
 
 
 
 
 
ba1f020
 
 
 
 
 
7388f6f
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
 
ba1f020
 
 
 
7388f6f
 
d42cb0e
7388f6f
 
ba1f020
7388f6f
 
ba1f020
 
 
d42cb0e
7388f6f
ba1f020
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7388f6f
 
 
 
ce4c24f
ba1f020
 
e93bd9c
ba1f020
ce4c24f
ba1f020
7388f6f
ba1f020
 
 
 
ce4c24f
ba1f020
ce4c24f
 
ba1f020
 
 
 
7388f6f
ba1f020
 
 
ce4c24f
 
 
 
 
ba1f020
 
 
 
 
 
3bb3c2e
ba1f020
ce4c24f
 
 
 
 
 
d42cb0e
 
ba1f020
d42cb0e
ba1f020
7388f6f
ba1f020
 
d42cb0e
7388f6f
ce4c24f
7388f6f
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
# ===========================
# ACC FAQ Chatbot – Final One-Paste Colab
# ===========================

# ---------- Imports ----------
import os, re, json, csv, random, datetime, zipfile, xml.etree.ElementTree as ET
from typing import List, Dict, Tuple
import numpy as np
import gradio as gr

import docx  # python-docx
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
from sklearn.decomposition import TruncatedSVD
from sklearn.preprocessing import normalize as sk_normalize

import nltk
from nltk.corpus import wordnet as wn, stopwords
from nltk.stem import PorterStemmer

from rapidfuzz import process, fuzz
from sentence_transformers import SentenceTransformer
import faiss

# ---------- NLTK (safe if cached) ----------
try: _ = wn.all_synsets()
except LookupError:
    nltk.download("wordnet"); nltk.download("omw-1.4")
try: _ = stopwords.words("english")
except LookupError:
    nltk.download("stopwords")

# ---------- Config ----------
SUMMARY_SENTENCES = 2                 # default TL;DR length
UNMATCHED_LOG_PATH = "unmatched_queries_log.csv"
SVD_COMPONENTS = 300                  # 200–400 typically good
RANDOM_SEED = 13
np.random.seed(RANDOM_SEED); random.seed(RANDOM_SEED)
EMBEDDINGS_ENABLED = True             # MiniLM + FAISS layer

# ---------- YOUR SITE LINKS (EDIT THESE) ----------
SITE_LINKS = {
    "home":   "https://your-site.example/",         # main / homepage
    "apply":  "https://your-site.example/apply",
    "grants": "https://your-site.example/apply",
    "donate": "https://your-site.example/donate",
    "about":  "https://your-site.example/about",
    "contact":"https://your-site.example/contact",
    "faq":    "https://your-site.example/faq",
}

# ---------- Small talk ----------
SMALL_TALK = {
    "greetings": [" hi ", " hello ", " hey ", " salaam ", " salam ", " assalam "],
    "how_are_you": [" how are you", " how are u", " how r u", " how’s it going", " hows it going"],
    "goodbye": [" bye", " goodbye", " see you", " take care"],
    "thanks": [" thanks", " thank you", " thx", " shukran", " jazakallah", " jazak allah", " jazakallahu khair"]
}
SMALL_TALK_RESPONSES = {
    "greetings": [
        "👋 Hello! How can I help you today?",
        "🤗 Hi there! What would you like to know?",
        "🙌 Hey! I’m your ACC assistant — ask me anything.",
    ],
    "how_are_you": [
        "😊 I’m doing great, thanks for asking! How can I support you today?",
        "🌟 Always ready to help with ACC info — what would you like to know?",
    ],
    "goodbye": [
        "👋 Goodbye! Wishing you the best.",
        "🤲 Take care! I’m here whenever you need me.",
    ],
    "thanks": [
        "💝 You’re most welcome!",
        "🤗 Glad I could help.",
    ],
}

# ---------- (Optional) Quick keyword map for ultra-common snippets ----------
KEYWORD_MAP = {
    "donation": "💝 You can donate via Credit/Debit Card, Zelle (donate@acceducate.org), Bank Transfer, or by mailing checks.",
    "zelle": "📲 Donate via Zelle using donate@acceducate.org.",
    "check": "✉️ Mail checks to: 7750 N MacArthur Blvd, Ste 120-282, Irving, TX 75063.",
    "zakat": "🌙 Zakat is one of Islam’s five pillars. ACC ensures zakat goes to eligible students.",
    "riba": "❌ Riba means interest, prohibited in Islam. ACC provides riba-free loans.",
    "interest": "❌ Interest (riba) is haram. ACC ensures loans are interest-free.",
    "loan": "💵 ACC offers interest-free loans to students.",
    "minimum loan": "🔢 Minimum loan: $1,000.",
    "maximum loan": "🔢 Maximum loan: $10,000.",
    "repay": "📅 Repayment begins 6 months after graduation with flexible monthly installments.",
    "about acc": "🌟 ACC is a nonprofit (founded 2013) providing riba-free loans.",
    "what is acc": "🌟 ACC is a nonprofit (founded 2013) providing riba-free loans.",
    "scam": "✅ ACC is a registered 501(c)(3) nonprofit. It is legitimate and transparent.",
    "email": "📧 adviser@acceducate.org.",
    "mail": "📧 adviser@acceducate.org.",
    "address": "📍 Office: 955 W John Carpenter Fwy #100, Irving, TX 75039.",
    "facebook": "🔗 https://www.facebook.com/acontinuouscharity",
    "instagram": "📸 https://www.instagram.com/acontinuouscharity",
    "youtube": "▶️ https://www.youtube.com/channel/UCkq8wAvAqt54yzjzL_QIq3w",
    "twitter": "🐦 https://x.com/accnational",
}

# ---------- Donation / Repayment context cues ----------
DONATION_CUES = [" donate"," donation"," give"," giving"," zakat"," sadaqah"," sadaqa"," appeal"," campaign"," fundraiser"," payment options"," apple pay"," card"," zelle"," bank transfer"," check"," pay to donate"," ways to give"]
REPAYMENT_CUES = [" repay"," repayment"," installments"," monthly payment"," pay back"," due"," loan payment"," when do i pay"," when to repay"]

# ---------- Navigation cues (includes Home & “shop” redirects) ----------
NAV_CUES = [
    ("home",   ["home", "homepage", "main page", "start page", "website home",
                "what is this website about", "what does this website do", "what is this site",
                "buy coffee", "buy clothes", "do you sell products", "store", "shop", "shopping", "purchase", "products", "where is the shop"]),
    ("apply",  ["apply online","application portal","start application","apply now","grant application","grant page","take me to the grant","grant"]),
    ("donate", ["donate now","donation page","donate online","give now","take me to donate","payment page"]),
    ("about",  ["about us","about acc","learn about acc","organization info"]),
    ("contact",["contact","contact us","reach you","email you","support email"]),
    ("faq",    ["faq","faqs","help center","help desk"]),
]

# ---------- Built-in Add-on entries (loan-side + website redirects) ----------
HOME_URL  = SITE_LINKS.get("home","")
APPLY_URL = SITE_LINKS.get("apply","")

ADDON_ENTRIES = [
    # Loan-side
    ("When do I start repayment?", "Repayment begins 6 months after graduation. Payments are made in manageable monthly installments."),
    ("When does repayment start?", "Repayment begins 6 months after graduation. Payments are made in manageable monthly installments."),
    ("Is there a grace period?", "Yes. There is a 6-month grace period after graduation before repayment begins."),
    ("How much are monthly installments?", "Installments are structured to be manageable; exact amounts depend on the total borrowed and agreed schedule."),
    ("Can I pay early or extra?", "Yes. You may make extra or early payments. Early repayment reduces your outstanding balance sooner."),
    ("How do I make repayments?", "Follow the repayment instructions provided by ACC. If you have any issues, please contact adviser@acceducate.org."),
    ("Payment options to repay a loan", "Repayments follow the instructions provided by ACC and are made in monthly installments. Contact adviser@acceducate.org if you need help."),
    ("What happens if I can’t pay?", "If you experience a hardship, contact adviser@acceducate.org to discuss available options per ACC policy."),
    ("Who can apply for a loan?", "Students pursuing accredited education who meet ACC’s criteria may apply. Documentation and an interview are typically required."),
    ("Am I eligible for ACC loan?", "Eligibility depends on ACC’s criteria for students pursuing accredited education. Applicants provide documentation and complete an interview."),
    ("How do I apply for a loan?", "Complete the application on the ACC website, submit required documents, and complete the interview process."),
    ("Where can I apply for a loan?", f"You can start your application online here: {APPLY_URL}" if APPLY_URL else "You can start your application on the ACC website."),
    ("What is the minimum loan amount?", "The minimum loan amount is $1,000."),
    ("What is the maximum loan amount?", "The maximum loan amount is $10,000."),
    ("Is there any interest?", "No. ACC provides riba-free (interest-free) loans."),
    ("Are ACC loans halal?", "Yes. ACC loans are riba-free (interest-free)."),
    ("How long does the application review take?", "Processing times can vary. You’ll be contacted by ACC as your application is reviewed and next steps are scheduled."),
    ("Are there application deadlines?", "Application timelines may vary. Please check the application page for current information."),
    # Website redirects to Home (with blurb)
    ("What is this website about?", f"ACC (A Continuous Charity) is a 501(c)(3) nonprofit founded in 2013 that provides riba-free (interest-free) loans to students. Learn more on our homepage: {HOME_URL}"),
    ("What does this website do?", f"ACC supports students with interest-free (riba-free) educational loans funded by donations. Visit our homepage for an overview: {HOME_URL}"),
    ("What is this site?", f"This is ACC’s official site for information, donations, and student support. Start here: {HOME_URL}"),
    ("Can we buy coffee here?", f"No — this is not an e-commerce site. ACC is a nonprofit aiding students with riba-free loans. Please see our homepage: {HOME_URL}"),
    ("Can we buy clothes here?", f"No — this is not an online store. ACC is a nonprofit supporting students with riba-free loans. Learn more: {HOME_URL}"),
    ("Do you sell products?", f"No — ACC is a nonprofit organization and does not operate an online store. See our homepage for mission and programs: {HOME_URL}"),
    ("Where is the shop?", f"ACC does not have an online shop. To learn about our mission and programs, please visit the homepage: {HOME_URL}"),
]

# ---------- Loaders ----------


def load_csv(path: str) -> List[Dict[str, str]]:
    import pandas as pd
    qas = []
    df = pd.read_csv(path)
    # normalize headers
    cols = {c.lower(): c for c in df.columns}
    qcol = cols.get("question") or cols.get("q") or cols.get("title")
    acol = cols.get("answer")   or cols.get("a") or cols.get("content") or cols.get("text")
    if not qcol or not acol:
        raise ValueError("CSV must have 'question' and 'answer' columns.")
    for q, a in df[[qcol, acol]].itertuples(index=False):
        if isinstance(q, str) and isinstance(a, str) and q.strip() and a.strip():
            qas.append({"question": q.strip(), "answer": a.strip()})
    return qas



def load_file(path: str) -> List[Dict[str, str]]:
    pl = path.lower()
    if pl.endswith(".csv"):  return load_csv(path)
    raise ValueError("Unsupported file format. Use .docx, .csv, or .json")

# ---------- Text utils ----------
STOP = set(stopwords.words("english"))
STEM = PorterStemmer()

def normalize_basic(text: str) -> str:
    t = " " + (text or "").lower() + " "
    t = re.sub(r"[^a-z0-9 ]+", " ", t)
    t = re.sub(r"\s+", " ", t).strip()
    return t

def tokenize_and_stem(text: str) -> List[str]:
    t = normalize_basic(text)
    tokens = [w for w in t.split() if w not in STOP]
    return [STEM.stem(w) for w in tokens]

def normalize_for_index(text: str) -> str:
    rep = {
        r"\bmin\b": "minimum",
        r"\bmax\b": "maximum",
        r"\bdon\b": "donation",
        r"\bfund\b": "donation",
        r"\bmoney\b": "donation",
        r"\brecurring\b": "monthly",
        r"\bacc\b": "a continuous charity",
        r"\bmail\b": "email",
        r"\bzelle\b": "zelle",
        r"\briba\b": "riba",
    }
    t = " " + (text or "").lower() + " "
    for k, v in rep.items():
        t = re.sub(k, f" {v} ", t)
    tokens = tokenize_and_stem(t)
    return " ".join(tokens)

def split_sentences(text: str) -> List[str]:
    parts = re.split(r'(?<=[.!?])\s+', (text or "").strip())
    return [p.strip() for p in parts if p.strip()]

def short_summary(answer: str, n: int = SUMMARY_SENTENCES) -> str:
    sents = split_sentences(answer)
    if len(sents) <= n:
        return (answer or "").strip()
    return " ".join(sents[:n]).strip()

# ---------- Intent / Nav helpers ----------
def _detect_nav_target(text: str):
    q = " " + (text or "").lower().strip() + " "
    for key, triggers in NAV_CUES:
        for t in triggers:
            if f" {t} " in q or q.strip() == t:
                return key
    if " take me to " in q:
        for key in SITE_LINKS:
            if key in q: return key
    return None

# ---------- Chatbot ----------
class FAQChatbot:
    def __init__(self, qa_list: List[Dict[str, str]]):
        assert qa_list, "QA list is empty."
        # De-dupe and clean
        seen = set(); uniq = []
        for qa in qa_list:
            q = (qa.get("question") or "").strip()
            a = (qa.get("answer") or "").strip()
            if not q or not a: continue
            if q.lower() in seen: continue
            seen.add(q.lower()); uniq.append({"question": q, "answer": a})
        # Add-on merge (runtime; only if not present)
        for q_add, a_add in ADDON_ENTRIES:
            if q_add.strip().lower() not in seen:
                uniq.append({"question": q_add.strip(), "answer": a_add.strip()})
                seen.add(q_add.strip().lower())

        self.qa_list   = uniq
        self.questions = [x["question"] for x in uniq]
        self.answers   = [x["answer"]   for x in uniq]

        # Precompute normalized forms
        self.norm_qs_basic = [normalize_basic(q) for q in self.questions]
        self.norm_qs_idx   = [normalize_for_index(q) for q in self.questions]
        self.q_tokens      = [tokenize_and_stem(q) for q in self.questions]

        # Vectorizers & matrices
        self.vec_word = TfidfVectorizer(analyzer="word", ngram_range=(1, 2), min_df=2, sublinear_tf=True)
        self.vec_char = TfidfVectorizer(analyzer="char", ngram_range=(3, 5), min_df=2)
        self.mat_word = self.vec_word.fit_transform(self.norm_qs_idx)
        self.mat_char = self.vec_char.fit_transform(self.norm_qs_idx)

        # Semantic layer (LSA)
        max_comp = max(2, min(SVD_COMPONENTS, self.mat_word.shape[1] - 1))
        self.svd = TruncatedSVD(n_components=max_comp)
        self.mat_sem = sk_normalize(self.svd.fit_transform(self.mat_word))

        # Embeddings + FAISS
        self._faiss = None; self._emb_qs = None; self._enc = None
        if EMBEDDINGS_ENABLED and len(self.questions) > 0:
            self._enc = SentenceTransformer("sentence-transformers/all-MiniLM-L6-v2")
            self._emb_qs = self._enc.encode(self.questions, normalize_embeddings=True).astype("float32")
            d = self._emb_qs.shape[1]
            self._faiss = faiss.IndexFlatIP(d)
            self._faiss.add(self._emb_qs)

        # UI voice
        self.response_styles = [
            "🙌 Of course, let me explain:",
            "👍 Absolutely, here’s the info:",
            "🌟 Great question! Here’s what I found:",
            "🤝 Glad you asked! Here’s what I know:",
            "📘 Here’s a quick summary:",
        ]

        # Keyword override
        self.keyword_map = KEYWORD_MAP.copy()
        self._kw_keys = list(self.keyword_map.keys())
        self._keyword_matcher = lambda q: (lambda m: (m[0], m[1] / 100.0) if m else ("", 0.0))(
            process.extractOne(q, self._kw_keys, scorer=fuzz.token_set_ratio)
        )

        # Trigger index for common features
        self.trigger_index = self._build_trigger_index(top_k=300)
        self._last_intent = None
        self._log_buf = []

    def _build_trigger_index(self, top_k=300) -> Dict[str, int]:
        vec = TfidfVectorizer(analyzer="word", ngram_range=(1, 2), min_df=5)
        X = vec.fit_transform(self.norm_qs_idx)
        if X.shape[1] == 0:
            return {}
        vocab = np.array(vec.get_feature_names_out())
        scores = np.asarray(X.sum(axis=0)).ravel()
        top_idx = scores.argsort()[::-1][:top_k]
        Xd = X.toarray()
        triggers = {}
        for i in top_idx:
            feat = vocab[i]
            col = Xd[:, i]
            best_q = int(col.argmax())
            triggers[feat] = best_q
        return triggers

    def _exact_or_near_match(self, qn_basic: str) -> Tuple[int, float]:
        try:
            idx = self.norm_qs_basic.index(qn_basic)
            return idx, 1.0
        except ValueError:
            pass
        q_tokens = tokenize_and_stem(qn_basic)
        best_idx, best_score = -1, 0.0
        for i, toks in enumerate(self.q_tokens):
            s = len(set(q_tokens) & set(toks)) / max(1, len(set(q_tokens) | set(toks)))
            if s > best_score:
                best_idx, best_score = i, s
        return best_idx, best_score

    def _keyword_override(self, raw_q: str) -> Tuple[str, float]:
        return self._keyword_matcher(raw_q)

    def _trigger_lookup(self, qn_idx: str) -> Tuple[int, float]:
        tokens = set(qn_idx.split())
        cand = [self.trigger_index[t] for t in tokens if t in self.trigger_index]
        if not cand:
            return -1, 0.0
        best_idx = max(set(cand), key=cand.count)
        conf = min(0.55, 0.25 + 0.05 * len(cand))
        return best_idx, conf

    def _semantic_scores(self, qn_idx: str):
        q_w = self.vec_word.transform([qn_idx])
        q_sem = sk_normalize(self.svd.transform(q_w))
        sims_sem = cosine_similarity(q_sem, self.mat_sem)[0]
        return sims_sem

    def _semantic_topk(self, query: str, k: int = 5):
        if self._faiss is None:
            return np.array([], dtype=int), np.array([], dtype=float)
        qv = self._enc.encode([query], normalize_embeddings=True).astype("float32")
        sims, idxs = self._faiss.search(qv, min(k, len(self.questions)))
        return idxs[0], sims[0]

    def _log_unmatched(self, raw_q: str, suggestions: List[str], scores: List[float]):
        self._log_buf.append([
            datetime.datetime.utcnow().isoformat(), raw_q,
            suggestions[0] if len(suggestions) > 0 else "", f"{scores[0]:.4f}" if len(scores) > 0 else "",
            suggestions[1] if len(suggestions) > 1 else "", f"{scores[1]:.4f}" if len(scores) > 1 else "",
            suggestions[2] if len(suggestions) > 2 else "", f"{scores[2]:.4f}" if len(scores) > 2 else "",
        ])
        if len(self._log_buf) >= 25:
            self._flush_log()

    def _flush_log(self):
        exists = os.path.exists(UNMATCHED_LOG_PATH)
        with open(UNMATCHED_LOG_PATH, "a", encoding="utf-8", newline="") as f:
            w = csv.writer(f)
            if not exists:
                w.writerow(["timestamp","query","suggestion_1","score_1","suggestion_2","score_2","suggestion_3","score_3"])
            w.writerows(self._log_buf)
        self._log_buf.clear()

    def _context_flags(self, qlow: str) -> Tuple[bool, bool]:
        is_donation = any(c in qlow for c in DONATION_CUES)
        is_repayment = any(c in qlow for c in REPAYMENT_CUES)
        if is_donation and not is_repayment:
            self._last_intent = "donation"
        elif is_repayment and not is_donation:
            self._last_intent = "repayment"
        return is_donation, is_repayment

    def _render(self, ans: str):
        intro = random.choice(self.response_styles)
        sents = split_sentences(ans)
        if len(ans) > 180 and len(sents) > SUMMARY_SENTENCES:
            tl = short_summary(ans, SUMMARY_SENTENCES)
            if tl.strip().lower() != sents[0].strip().lower():
                return f"{intro}\n**TL;DR:** {tl}\n\n👉 {ans}"
        return f"{intro}\n👉 {ans}"

    def get_answer(self, query: str) -> str:
        raw_q = (query or "").strip()
        if not raw_q:
            return "⚠️ Please type a question so I can assist you better."
        qlow = " " + raw_q.lower() + " "

        # small talk
        for cat, trigs in SMALL_TALK.items():
            if any(t in qlow for t in trigs):
                return random.choice(SMALL_TALK_RESPONSES[cat])

        # identity
        if any(p in qlow for p in [" who are you", " what are you", " tell me about yourself"]):
            return random.choice([
                "🤖 I’m ACC’s virtual assistant, here to answer your questions about donations, loans, and more.",
                "📘 I’m an AI chatbot trained on ACC’s FAQs to guide you with accurate answers.",
                "🌟 I’m your online assistant for A Continuous Charity, here to help you understand ACC.",
            ])

        # navigation (incl. home & ecommerce)
        nav = _detect_nav_target(raw_q)
        if nav and nav in SITE_LINKS and SITE_LINKS[nav]:
            label = {
                "home":"Home","apply":"Apply Online","grants":"Grant Application",
                "donate":"Donate Now","about":"About Us","contact":"Contact Us","faq":"FAQ"
            }.get(nav, nav.title())
            intro = random.choice(self.response_styles)
            if nav == "home":
                blurb = "🌟 ACC (A Continuous Charity) is a 501(c)(3) nonprofit founded in 2013 that provides riba-free (interest-free) loans to students."
                return f"{intro}\n{blurb}\n👉 **[{label}]({SITE_LINKS[nav]})**"
            return f"{intro}\n👉 **[{label}]({SITE_LINKS[nav]})**"

        # credibility/safety
        if any(k in qlow for k in [" scam", " fraud", " fake", " legit", " legitimate", " trust ", " real "]):
            return self._render("✅ ACC is a registered **501(c)(3)** nonprofit. It is legitimate and transparent.")

        # topic-ish fallback: payment options ambiguity
        is_donation, is_repayment = self._context_flags(qlow)
        if " payment option" in qlow or " pay " in qlow or qlow.strip() == "payment options?":
            if is_donation or self._last_intent == "donation":
                return self._render(
                    "💝 **Ways to donate to ACC**\n\n• Credit/Debit Card (online)\n• Zelle: donate@acceducate.org\n• Bank Transfer (contact us)\n• Check by Mail: 7750 N MacArthur Blvd, Ste 120-282, Irving, TX 75063\n\nℹ️ Apple Pay may appear on supported devices/browsers."
                )
            if is_repayment or self._last_intent == "repayment":
                return self._render("📅 **Repayment** begins **6 months after graduation** with monthly installments.")

        # retrieval pipeline
        qn_basic = normalize_basic(raw_q)
        qn_idx   = normalize_for_index(raw_q)

        # exact / near-exact
        near_idx, near_score = self._exact_or_near_match(qn_basic)
        if near_score >= 0.90:
            return self._render(self.answers[near_idx])

        # keyword quick win
        key, key_score = self._keyword_override(qlow)
        if key_score >= 0.60 and key in self.keyword_map:
            return self._render(self.keyword_map[key])

        # trigger lookup
        trig_idx, trig_conf = self._trigger_lookup(qn_idx)
        if trig_idx >= 0 and trig_conf >= 0.45:
            return self._render(self.answers[trig_idx])

        # TF-IDF + char + LSA
        q_w = self.vec_word.transform([qn_idx])
        q_c = self.vec_char.transform([qn_idx])
        sims_word = cosine_similarity(q_w, self.mat_word)[0]
        sims_char = cosine_similarity(q_c, self.mat_char)[0]
        sims_sem  = self._semantic_scores(qn_idx)
        sims_blend_lsa = 0.50 * sims_sem + 0.30 * sims_word + 0.20 * sims_char

        # Embeddings + FAISS blend
        final_scores = sims_blend_lsa.copy()
        if EMBEDDINGS_ENABLED:
            idx_e, sc_e = self._semantic_topk(raw_q, k=max(5, min(10, len(self.questions))))
            emb = np.zeros_like(final_scores)
            if len(idx_e) > 0:
                emb[idx_e] = sc_e
            final_scores = 0.55 * emb + 0.30 * sims_sem + 0.10 * sims_word + 0.05 * sims_char

        best_idx = int(np.argmax(final_scores))
        best_score = float(final_scores[best_idx])

        # decisions
        if near_score >= 0.75:
            return self._render(self.answers[near_idx])
        if best_score >= (0.40 if EMBEDDINGS_ENABLED else 0.36):
            return self._render(self.answers[best_idx])

        # suggestions + log
        if best_score >= (0.22 if EMBEDDINGS_ENABLED else 0.18):
            top = np.argsort(final_scores)[::-1][:3]
            suggestions = [self.questions[i] for i in top]
            scores = [float(final_scores[i]) for i in top]
            self._log_unmatched(raw_q, suggestions, scores)
            msg = "🤔 I’m not completely sure — did you mean one of these?\n\n"
            msg += "\n".join([f"🔹 {s}" for s in suggestions])
            msg += "\n\nYou can click one of these or rephrase your question."
            return msg

        self._log_unmatched(raw_q, [], [])
        return "😕 I couldn’t find anything close. Could you rephrase your question or add a bit more detail?"

# ---------- Build bot from path ----------
def build_bot_from_path(path: str) -> FAQChatbot:
    qas = load_file(path)
    return FAQChatbot(qas)


# ---------- Default dataset candidates ----------
CANDIDATES = [
    "dataset.csv",  # fixed typo ✅
]

DEFAULT_DATA = next((p for p in CANDIDATES if os.path.exists(p)), None)

bot = None
if DEFAULT_DATA:
    try:
        bot = build_bot_from_path(DEFAULT_DATA)
        print(f"✅ Loaded dataset: {DEFAULT_DATA} (QAs: {len(bot.qa_list)})")
    except Exception as e:
        print("❌ Default load failed:", e)



# ---------- Gradio UI ----------
with gr.Blocks() as demo:
    gr.Markdown("## 🤖 ACC Online Assistant")

    summary_n = gr.Slider(1, 5, value=SUMMARY_SENTENCES, step=1, label="TL;DR sentences")
    status    = gr.Markdown(value=f"✅ Loaded default dataset ({len(bot.qa_list)} QAs)" if bot else "❌ Failed to load dataset.")
    chatbot   = gr.Chatbot(
        height=520,
        show_label=False,
        value=[["🤖 Assistant", "👋 Hello! I’m your ACC Online Assistant. Ask me anything."]]
    )
    msg       = gr.Textbox(label="💬 Ask me anything about ACC", placeholder="Type your question here...")
    clear     = gr.Button("Clear")

    _bot = {"inst": bot}

    # Respond to user query
    def respond(message, chat_history, n_sentences=None):
        global SUMMARY_SENTENCES
        SUMMARY_SENTENCES = int(n_sentences)
        if _bot["inst"] is None:
            return "", chat_history + [["🤖 Assistant", "⚠️ No dataset loaded."]]
        reply = _bot["inst"].get_answer(message)
        chat_history.append(["🙂 You", message])
        chat_history.append(["🤖 Assistant", reply])
        return "", chat_history

    # Clear chat
    def do_clear():
        return "", [["🤖 Assistant", "👋 Hello! I’m your ACC Online Assistant. Ask me anything."]]

    # Input bindings
    msg.submit(respond, [msg, chatbot, summary_n], [msg, chatbot])
    clear.click(do_clear, [], [msg, chatbot], queue=False)

# ---------- Launch ----------
demo.launch(server_name="0.0.0.0", server_port=7860)