File size: 1,933 Bytes
53bdcd2
 
dbabef2
 
53bdcd2
dbabef2
53bdcd2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
from collections import Counter

from db.parsers.bns.cleaner import BNSTextCleaner
from db.parsers.bns.bns_parser import BNSParser

from db.parsers.bns.chunker import (
    LegalChunker,
    chunks_to_dicts
)

from ingest import (
    LegalIngestionPipeline
)

# =====================================================
# LOAD BNS
# =====================================================

with open(
    "../../pdfs/bns.txt",
    "r",
    encoding="utf8"
) as f:

    text = f.read()

# =====================================================
# CLEAN
# =====================================================

cleaner = BNSTextCleaner()

text = cleaner.clean(
    text
)

# =====================================================
# PARSE
# =====================================================

parser = BNSParser()

document = parser.parse(
    text
)

# =====================================================
# CHUNK
# =====================================================

chunker = LegalChunker()

chunks = chunks_to_dicts(
    chunker.chunk_document(
        document
    )
)

# =====================================================
# STATS
# =====================================================

print("\n========== CHUNK STATS ==========\n")

print(
    f"Total Chunks: {len(chunks)}"
)

levels = Counter(
    chunk["level"]
    for chunk in chunks
)

for level, count in sorted(
    levels.items()
):
    print(
        f"{level}: {count}"
    )

# =====================================================
# INGEST
# =====================================================

pipeline = (
    LegalIngestionPipeline()
)

# Optional:
# wipe collection before indexing

try:

    pipeline.store.client.delete_collection(
        collection_name="bns"
    )

    print(
        "\nDeleted existing collection."
    )

except Exception:
    pass

pipeline.ingest(
    chunks=chunks,
    batch_size=64
)

print(
    "\nBNS successfully indexed."
)