File size: 9,643 Bytes
6a2ebe8
195ff64
6a2ebe8
195ff64
 
6a2ebe8
195ff64
 
 
 
6a2ebe8
195ff64
 
 
 
 
6a2ebe8
 
965a09d
5f75ccf
195ff64
 
 
 
965a09d
 
6a2ebe8
 
195ff64
6a2ebe8
 
195ff64
6a2ebe8
195ff64
 
6a2ebe8
195ff64
 
6a2ebe8
195ff64
 
 
 
 
6a2ebe8
195ff64
 
 
6a2ebe8
195ff64
5f75ccf
6a2ebe8
195ff64
 
 
 
 
6a2ebe8
5f75ccf
 
 
 
 
 
195ff64
 
 
 
 
 
6a2ebe8
195ff64
 
 
 
 
 
6a2ebe8
195ff64
 
6a2ebe8
195ff64
 
 
 
 
 
 
6a2ebe8
195ff64
6a2ebe8
195ff64
 
 
6a2ebe8
 
195ff64
 
 
 
5f75ccf
195ff64
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5f75ccf
195ff64
5f75ccf
195ff64
 
6a2ebe8
 
 
195ff64
 
 
5f75ccf
195ff64
 
 
 
 
 
 
 
5f75ccf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
195ff64
6a2ebe8
195ff64
 
 
 
 
5f75ccf
 
 
 
195ff64
6a2ebe8
 
3bc0c14
 
 
 
 
 
 
 
 
6a2ebe8
195ff64
6a2ebe8
 
195ff64
6a2ebe8
195ff64
 
 
 
 
5f75ccf
 
 
195ff64
6a2ebe8
 
 
 
 
 
 
 
 
5f75ccf
6a2ebe8
 
195ff64
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5f75ccf
 
 
 
3bc0c14
 
 
 
 
 
 
6a2ebe8
 
 
e5b31ad
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
"""
CodeSearch β€” Gradio UI

Side-by-side comparison of four retrieval modes over CodeSearchNet Python:
    BM25 Β· Dense (MiniLM) Β· Hybrid (RRF) Β· Hybrid + cross-encoder rerank

Boot is NON-INDEXING by design:
    - Dense vectors live in Qdrant, built offline via scripts/index_corpus.py.
    - BM25 loads a prebuilt slim index (.cache/bm25-slim) in full-corpus mode.
    - The app refuses to embed the corpus at boot (see the guard below).

Modes:
    SMOKE_TEST_SIZE=-1  β†’ full corpus; BM25 from the slim index. (HF Space secret)
    SMOKE_TEST_SIZE=100 β†’ local pipeline check; BM25 built in-memory. Dense/hybrid
                          modes return full-Qdrant ids not in the small corpus, so
                          only BM25 is meaningful in smoke mode.
"""

import os
import random
import sys
import time
import traceback

sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "src"))

import gradio as gr

from codesearch.config import SMOKE_TEST_SIZE, TOP_K
from codesearch.data import load_codesearch
from codesearch.retrievers.bm25 import BM25Retriever
from codesearch.retrievers.bm25_index import BM25Index
from codesearch.retrievers.dense import DenseRetriever
from codesearch.retrievers.hybrid import HybridRetriever
from codesearch.retrievers.reranker import CrossEncoderReranker

_HERE = os.path.dirname(os.path.abspath(__file__))
BM25_SLIM_DIR = os.path.join(_HERE, ".cache", "bm25-slim")

# Candidate pool for hybrid fusion and reranking in the UI. Small on purpose:
# reranking is inference-bound (~20 pairs/query on CPU), and K=20 keeps
# hybrid+rerank under the 3s latency budget. Eval uses K=100 for Recall; the UI
# trades a little recall for responsiveness. See M3.2 design notes.
UI_POOL = 20

# ---------------------------------------------------------------------------
# Boot: load corpus + wire retrievers once. No indexing happens here.
# ---------------------------------------------------------------------------

print("=== CodeSearch Boot ===")
corpus, queries = load_codesearch()  # queries = eval set (test docstrings w/ known gold)

# Two id-keyed views over the corpus:
#   corpus_by_id  β€” for rendering result cards (dense/hybrid hits carry id+score only)
#   corpus_lookup β€” reranker candidate text (code_tokens, already space-joined)
corpus_by_id = {d["id"]: d for d in corpus}
corpus_lookup = {d["id"]: d["code_tokens"] for d in corpus}

# Ground truth for the eval queries: each test docstring maps to the id of the
# function it documents. Lets the UI badge the "correct" hit and its rank β€” but
# only for these known queries; a free-form query has no gold.
gold_by_query = {q["query"]: q["relevant_id"] for q in queries}
eval_query_texts = list(gold_by_query.keys())

if SMOKE_TEST_SIZE == -1:
    print(f"Loading slim BM25 index from {BM25_SLIM_DIR} ...")
    bm25 = BM25Retriever(BM25Index.load_index_only(BM25_SLIM_DIR, corpus))
else:
    print(f"Smoke mode (n={SMOKE_TEST_SIZE}): building in-memory BM25 index...")
    bm25 = BM25Retriever(corpus)

dense = DenseRetriever(recreate_collection=False)
if not dense.collection_exists_and_populated():
    raise RuntimeError(
        "Qdrant collection is empty. Build it offline with scripts/index_corpus.py β€” "
        "the app will not embed the full corpus at boot."
    )

hybrid = HybridRetriever(bm25, dense, pool_size=UI_POOL)
reranker = CrossEncoderReranker(hybrid, corpus_lookup, pool_size=UI_POOL)

RETRIEVERS = {
    "BM25": bm25,
    "Dense (MiniLM)": dense,
    "Hybrid (RRF)": hybrid,
    "Hybrid + Rerank": reranker,
}
MODES = list(RETRIEVERS.keys())

print(f"Ready. corpus={len(corpus):,} docs Β· modes={MODES}")

# ---------------------------------------------------------------------------
# Search + formatting
# ---------------------------------------------------------------------------


def _fmt_latency(dt: float) -> str:
    return f"⏱ **{dt:.2f} s**" if dt >= 1.0 else f"⏱ **{dt * 1000:.0f} ms**"


def _format(hits: list[dict], mode: str, gold_id: str | None = None) -> str:
    if not hits:
        return "_No results._"

    cards = []
    for i, h in enumerate(hits, 1):
        doc = corpus_by_id.get(h["id"])
        if doc is None:
            cards.append(
                f"**#{i}** β€” `{h['id']}` not in local corpus "
                f"(expected in smoke mode; full corpus on the Space)."
            )
            continue

        docstring = (doc.get("docstring") or "").strip()[:200]
        code = (doc.get("code") or "").strip()[:400]
        url = doc.get("url", "")
        score = h.get("score", 0.0)

        # Provenance line: shows WHY a doc ranked where it did, per mode.
        prov = f"score **{score:.4f}**"
        if h.get("bm25_rank") is not None or h.get("dense_rank") is not None:
            prov += f" Β· bm25 #{h.get('bm25_rank')} Β· dense #{h.get('dense_rank')}"
        if "pre_rerank_score" in h:
            prov += f" Β· pre-rerank {h['pre_rerank_score']:.4f}"

        gold = " βœ… **GOLD**" if gold_id is not None and h["id"] == gold_id else ""
        cards.append(
            f"**#{i}**{gold} β€” {prov}\n\n"
            f"{docstring}{'…' if docstring else ''}\n\n"
            f"```python\n{code}{'…' if code else ''}\n```\n"
            + (f"[source]({url})\n" if url else "")
            + "\n---"
        )
    return "\n".join(cards)


def _run(mode: str, query: str, gold_id: str | None) -> tuple[str, str]:
    retr = RETRIEVERS[mode]
    t0 = time.perf_counter()
    try:
        hits = retr.retrieve(query, top_k=TOP_K)
    except Exception as e:  # one mode failing must not take down the search
        traceback.print_exc()
        return f"⚠️ **{mode} failed:** `{type(e).__name__}: {e}`", "β€”"
    dt = time.perf_counter() - t0

    # Gold verdict header β€” only for known eval queries (free-form has no gold).
    verdict = ""
    if gold_id is not None:
        rank = next((i for i, h in enumerate(hits, 1) if h["id"] == gold_id), None)
        verdict = (
            f"### βœ… Gold answer at rank #{rank}\n\n"
            if rank is not None
            else f"### βœ— Gold answer not in top-{TOP_K}\n\n"
        )
    return verdict + _format(hits, mode, gold_id), _fmt_latency(dt)


def pick_random() -> str:
    """Fill the query box with a random eval query (one whose gold we know)."""
    return random.choice(eval_query_texts) if eval_query_texts else ""


def search(query: str, left_mode: str, right_mode: str):
    """Run both columns and return (left_md, left_latency, right_md, right_latency)."""
    if not query.strip():
        msg = "_Enter a query above._"
        return msg, "", msg, ""
    # Known only for eval queries (exact match); free-form typing β†’ gold_id None.
    gold_id = gold_by_query.get(query) or gold_by_query.get(query.strip())
    left_md, left_lat = _run(left_mode, query, gold_id)
    right_md, right_lat = _run(right_mode, query, gold_id)
    return left_md, left_lat, right_md, right_lat


def update_column(query: str, mode: str) -> tuple[str, str]:
    """Re-run a single column β€” wired to each retriever dropdown's change so
    switching the retriever refreshes that column without re-running the other."""
    if not query.strip():
        return "_Enter a query above._", ""
    gold_id = gold_by_query.get(query) or gold_by_query.get(query.strip())
    return _run(mode, query, gold_id)


# ---------------------------------------------------------------------------
# UI β€” selectable two-column compare
# ---------------------------------------------------------------------------

with gr.Blocks(title="CodeSearch") as demo:
    gr.Markdown(
        f"""
# πŸ” CodeSearch
Natural-language code search over **CodeSearchNet Python** ({len(corpus):,} functions).
Pick a retriever per column and compare β€” watch where lexical (BM25), semantic
(dense), fusion (RRF), and reranking disagree.

Hit **🎲 Random eval query** to load a real benchmark query with a known answer β€”
the **βœ… GOLD** badge marks the correct function, and each column reports the rank it landed at.
"""
    )

    with gr.Row():
        query_box = gr.Textbox(
            placeholder="e.g. parse a JSON file and return a dict",
            label="Query",
            scale=4,
        )
        search_btn = gr.Button("Search", variant="primary", scale=1)
        random_btn = gr.Button("🎲 Random eval query", scale=1)

    with gr.Row():
        with gr.Column():
            left_mode = gr.Dropdown(MODES, value="BM25", label="Left retriever")
            left_lat = gr.Markdown()
            left_out = gr.Markdown()
        with gr.Column():
            right_mode = gr.Dropdown(
                MODES, value="Hybrid + Rerank", label="Right retriever"
            )
            right_lat = gr.Markdown()
            right_out = gr.Markdown()

    inputs = [query_box, left_mode, right_mode]
    outputs = [left_out, left_lat, right_out, right_lat]
    search_btn.click(fn=search, inputs=inputs, outputs=outputs)
    query_box.submit(fn=search, inputs=inputs, outputs=outputs)
    # Random: fill the box with a known-gold eval query, then run the comparison.
    random_btn.click(fn=pick_random, outputs=query_box).then(
        fn=search, inputs=inputs, outputs=outputs
    )
    # Changing a retriever re-runs only its own column (keeps the other as-is).
    left_mode.change(
        fn=update_column, inputs=[query_box, left_mode], outputs=[left_out, left_lat]
    )
    right_mode.change(
        fn=update_column, inputs=[query_box, right_mode], outputs=[right_out, right_lat]
    )


if __name__ == "__main__":
    demo.launch(theme=gr.themes.Monochrome())