File size: 8,813 Bytes
b30f068
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
#!/usr/bin/env python3
"""
Offline CDMS index builder — the reproducible source of truth.

Ingests every PDF in data/pdfs/cdms/, extracts + chunks + BATCH-embeds them,
writes the SQLite metadata DB (data/cdms_metadata.db) and the Qdrant vector
index, then exports a portable Qdrant snapshot (data/qdrant_snapshots/) that is
committed and restored into the served Docker Qdrant at deploy.

This replaces the ad-hoc reprocess_pdfs.py. Unlike the old script it:
  * points at the real CDMS folder (data/pdfs/cdms, not data/pdfs),
  * injects the filename->cdms.net URL manifest so citations keep the source link,
  * validates that PDFs == documents == distinct Qdrant docs and points == chunks,
  * exports a server-format snapshot for the two-process Docker deploy.

Usage:
    # Build into a running Docker Qdrant (recommended) and export a snapshot:
    docker run -d -p 6333:6333 -v "$PWD/data/qdrant_snapshots:/qdrant/snapshots" qdrant/qdrant
    python scripts/build_index.py --force

    # Build into the on-disk local store instead (single-process only, no snapshot):
    python scripts/build_index.py --force --local

Flags:
    --force   reprocess every PDF from scratch (REQUIRED for the first real build,
              because the shipped index is stale/partial: 26 of 53 PDFs indexed).
    --local   build into the on-disk Qdrant (QDRANT_FORCE_LOCAL) instead of Docker.
              No portable snapshot is produced in this mode.
"""

import argparse
import json
import sqlite3
import sys
from pathlib import Path

PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))


def load_url_manifest(manifest_path: Path) -> dict:
    """Return {filename: cdms.net URL}; empty (with a warning) if absent."""
    if not manifest_path.exists():
        print(f"⚠️  No URL manifest at {manifest_path}")
        print("   Citations will fall back to filename + page (no source link).")
        return {}
    try:
        data = json.loads(manifest_path.read_text())
        print(f"🔗 Loaded {len(data)} URL mapping(s) from {manifest_path.name}")
        return data
    except Exception as e:
        print(f"⚠️  Could not read URL manifest ({e}); continuing without URLs.")
        return {}


def build(force: bool, local: bool):
    import os

    # A build must never trip the served-mode fail-fast.
    os.environ.pop("QDRANT_REQUIRE_DOCKER", None)
    if local:
        os.environ["QDRANT_FORCE_LOCAL"] = "1"

    from src.config.paths import PDF_DIR, URL_MANIFEST_PATH
    from src.cdms.document_loader import DocumentLoader

    if not PDF_DIR.exists():
        sys.exit(f"❌ PDF folder not found: {PDF_DIR}")

    pdfs = sorted(PDF_DIR.glob("*.pdf"))
    if not pdfs:
        sys.exit(f"❌ No PDFs in {PDF_DIR}")

    urls = load_url_manifest(URL_MANIFEST_PATH)
    missing_urls = [p.name for p in pdfs if p.name not in urls]

    loader = DocumentLoader(pdf_folder=str(PDF_DIR))
    if loader.embedding_service is None:
        sys.exit("❌ OpenAI embeddings unavailable (check OPENAI_API_KEY in .env).")
    if loader.vector_store is None:
        sys.exit("❌ Qdrant vector store unavailable.")

    print(f"\n📚 Building index from {len(pdfs)} PDF(s) in {PDF_DIR}")
    print(f"   Mode: {'on-disk (local)' if local else 'Docker Qdrant'} | force={force}")
    print("-" * 70)

    results = []
    for pdf in pdfs:
        res = loader.load_pdf(str(pdf), force_reprocess=force, pdf_url=urls.get(pdf.name))
        results.append((pdf.name, res))
        if res.get("skipped"):
            print(f"⏭️  {pdf.name}: already processed (use --force to rebuild)")
        elif res.get("success"):
            print(f"✅ {pdf.name}: {res.get('chunks_stored', 0)} chunks, "
                  f"{res.get('embeddings_generated', 0)} embeddings")
        else:
            print(f"❌ {pdf.name}: {res.get('error', 'failed')}")

    if missing_urls:
        print(f"\n⚠️  {len(missing_urls)} PDF(s) have no URL in the manifest "
              f"(citations degrade to filename+page): {', '.join(missing_urls[:8])}"
              f"{' ...' if len(missing_urls) > 8 else ''}")

    return loader


def validate(loader) -> bool:
    """Assert PDFs == docs == distinct-Qdrant-docs and points == chunks."""
    from src.config.paths import PDF_DIR, DB_PATH
    from src.cdms.product_catalog import ProductCatalog

    print("\n" + "=" * 70)
    print("VALIDATION")
    print("=" * 70)

    n_pdfs = len(list(PDF_DIR.glob("*.pdf")))

    conn = sqlite3.connect(str(DB_PATH))
    n_docs = conn.execute("SELECT COUNT(*) FROM documents WHERE processed=1").fetchone()[0]
    n_chunks = conn.execute("SELECT COUNT(*) FROM document_chunks").fetchone()[0]
    per_pdf = conn.execute(
        """SELECT d.filename, COUNT(dc.id)
           FROM documents d LEFT JOIN document_chunks dc ON dc.document_id = d.id
           GROUP BY d.id ORDER BY d.filename"""
    ).fetchall()
    conn.close()

    zero_chunk = [fn for fn, n in per_pdf if n == 0]

    info = loader.vector_store.get_collection_info()
    points = info.get("cdms_documents", {}).get("points_count", None)

    print(f"  PDFs on disk .............. {n_pdfs}")
    print(f"  Documents (processed) ..... {n_docs}")
    print(f"  Chunks (SQLite) ........... {n_chunks}")
    print(f"  Points (Qdrant) ........... {points}")

    ok = True
    if n_docs != n_pdfs:
        print(f"  ⚠️  documents ({n_docs}) != PDFs ({n_pdfs}) — run with --force.")
        ok = False
    if points is not None and points != n_chunks:
        print(f"  ❌ Qdrant points ({points}) != SQLite chunks ({n_chunks}).")
        ok = False

    if zero_chunk:
        # Known-empty (corrupted / scanned) PDFs are expected, not failures.
        expected_empty = {"ACQUIT.pdf", "asenra_g.pdf"}
        unexpected = [z for z in zero_chunk if z not in expected_empty]
        print(f"  ℹ️  0-chunk PDFs ({len(zero_chunk)}): {', '.join(zero_chunk)}")
        print(f"     (expected-empty: {', '.join(sorted(expected_empty))} — corrupted/scanned)")
        if unexpected:
            print(f"  ⚠️  Unexpected 0-chunk PDFs (investigate): {', '.join(unexpected)}")

    products = sorted(ProductCatalog().available_products())
    print(f"\n  🟢 Answerable products ({len(products)}):")
    print("     " + ", ".join(products))

    print("\n  " + ("✅ VALIDATION PASSED" if ok else "❌ VALIDATION FAILED"))
    return ok


def export_snapshot(loader) -> bool:
    """Create + download a portable Qdrant snapshot into data/qdrant_snapshots/."""
    from src.config.paths import SNAPSHOT_DIR

    store = loader.vector_store
    if not store.using_docker:
        print("\n⚠️  Snapshot export skipped: build ran against the on-disk store, "
              "not Docker.\n   The server-format snapshot needs a Docker Qdrant. "
              "For a single-process deploy you may commit data/qdrant_local instead.")
        return False

    print("\n" + "=" * 70)
    print("SNAPSHOT EXPORT")
    print("=" * 70)
    try:
        snap = store.client.create_snapshot(collection_name="cdms_documents")
        name = snap.name
        print(f"  Created snapshot on server: {name}")

        import requests
        url = f"http://{store.host}:{store.port}/collections/cdms_documents/snapshots/{name}"
        resp = requests.get(url, timeout=120)
        resp.raise_for_status()

        SNAPSHOT_DIR.mkdir(parents=True, exist_ok=True)
        out = SNAPSHOT_DIR / "cdms_documents.snapshot"
        out.write_bytes(resp.content)
        mb = len(resp.content) / (1024 * 1024)
        print(f"  📦 Saved {mb:.1f} MB -> {out}")
        print("  Commit this file (Git LFS). Restore it at deploy with "
              "scripts/restore_snapshot.py.")
        return True
    except Exception as e:
        print(f"  ⚠️  Snapshot export failed ({e}).")
        print("     If the Qdrant container was started with "
              "'-v \"$PWD/data/qdrant_snapshots:/qdrant/snapshots\"', the .snapshot "
              "file is already on disk under data/qdrant_snapshots/.")
        return False


def main():
    ap = argparse.ArgumentParser(description="Build the committed CDMS index.")
    ap.add_argument("--force", action="store_true",
                    help="reprocess every PDF (required for the first real build)")
    ap.add_argument("--local", action="store_true",
                    help="build into the on-disk Qdrant instead of Docker (no snapshot)")
    ap.add_argument("--no-snapshot", action="store_true", help="skip snapshot export")
    args = ap.parse_args()

    loader = build(force=args.force, local=args.local)
    valid = validate(loader)
    if not args.local and not args.no_snapshot:
        export_snapshot(loader)

    sys.exit(0 if valid else 1)


if __name__ == "__main__":
    main()