File size: 1,771 Bytes
53bdcd2 dbabef2 53bdcd2 dbabef2 53bdcd2 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 | from tqdm import tqdm
from qdrant_client.models import (
PointStruct
)
from db.parsers.bns.embedder_temp import (
LegalEmbedder
)
from db.parsers.bns.qdrant_store import (
QdrantStore
)
class LegalIngestionPipeline:
def __init__(self):
self.embedder = (
LegalEmbedder()
)
self.store = (
QdrantStore(
collection_name="bns"
)
)
def ingest(
self,
chunks,
batch_size=64
):
dimension = (
self.embedder
.model
.get_sentence_embedding_dimension()
)
self.store.create_collection(
dimension
)
point_id = 1
for start in tqdm(
range(
0,
len(chunks),
batch_size
)
):
batch = chunks[
start:
start + batch_size
]
texts = [
chunk[
"enriched_text"
]
for chunk in batch
]
embeddings = (
self.embedder.embed(
texts
)
)
points = []
for chunk, embedding in zip(
batch,
embeddings
):
points.append(
PointStruct(
id=point_id,
vector=
embedding.tolist(),
payload=
chunk
)
)
point_id += 1
self.store.upsert(
points
) |