Spaces:
Sleeping
Sleeping
| import re | |
| import os | |
| import duckdb | |
| import pandas as pd | |
| from .embeddings import model | |
| def split_into_sentences(text): | |
| text = re.sub(r'\s+', ' ', text).strip() | |
| fragments = text.split('. ') | |
| return [f.strip() for f in fragments if f.strip()] | |
| def init_document_db(csv_path="data/case-teaching-cabot.csv"): | |
| db_path = "document.db" | |
| if os.path.exists(db_path): | |
| os.remove(db_path) | |
| con = duckdb.connect(db_path) | |
| con.execute("CREATE SEQUENCE document_seq START 1;") | |
| con.execute("CREATE SEQUENCE sentence_seq START 1;") | |
| con.execute(""" | |
| CREATE TABLE document ( | |
| document_id INTEGER PRIMARY KEY DEFAULT nextval('document_seq'), | |
| content TEXT | |
| ); | |
| """) | |
| con.execute(""" | |
| CREATE TABLE sentence ( | |
| sentence_id INTEGER PRIMARY KEY DEFAULT nextval('sentence_seq'), | |
| document_id INTEGER REFERENCES document(document_id), | |
| content TEXT | |
| ); | |
| """) | |
| con.execute(""" | |
| CREATE TABLE sentence_embedding ( | |
| sentence_id INTEGER REFERENCES sentence(sentence_id), | |
| embedding DOUBLE[] | |
| ); | |
| """) | |
| df = pd.read_csv(csv_path) | |
| for _, row in df.iterrows(): | |
| doc_text = str(row['document']) | |
| document_id = con.execute( | |
| "INSERT INTO document (content) VALUES (?) RETURNING document_id", | |
| [doc_text] | |
| ).fetchone()[0] | |
| for sentence_text in split_into_sentences(doc_text): | |
| con.execute( | |
| "INSERT INTO sentence (document_id, content) VALUES (?, ?)", | |
| [document_id, sentence_text] | |
| ) | |
| sentences = con.execute("SELECT sentence_id, content FROM sentence").fetchall() | |
| for sentence_id, content in sentences: | |
| embedding = model.encode(content) | |
| con.execute( | |
| "INSERT INTO sentence_embedding VALUES (?, ?)", | |
| [sentence_id, embedding.tolist()] | |
| ) | |
| return con | |