singhankur01 commited on
Commit
5b9ca00
·
verified ·
1 Parent(s): 3a09536

Update utils/DocsLoader.py

Browse files
Files changed (1) hide show
  1. utils/DocsLoader.py +13 -20
utils/DocsLoader.py CHANGED
@@ -1,20 +1,21 @@
1
  import os
2
  import tempfile
3
  import requests
4
- import re
5
  from fastapi import HTTPException
 
6
  from langchain_community.document_loaders import PyMuPDFLoader, Docx2txtLoader
7
  from langchain.text_splitter import RecursiveCharacterTextSplitter
8
  from langchain.schema import Document
9
 
 
10
  def load_and_chunk(url: str) -> list[Document]:
11
  resp = requests.get(url)
12
  if resp.status_code != 200:
13
- raise HTTPException(400, "Document download failed")
14
 
15
  content_type = resp.headers.get("Content-Type", "").lower()
16
  url_lower = url.lower()
17
- text = ""
18
 
19
  if "application/pdf" in content_type or ".pdf" in url_lower:
20
  with tempfile.NamedTemporaryFile(delete=False, suffix=".pdf") as tmp:
@@ -22,37 +23,29 @@ def load_and_chunk(url: str) -> list[Document]:
22
  tmp_path = tmp.name
23
  try:
24
  loader = PyMuPDFLoader(tmp_path)
25
- # Extract full text first for better chunking
26
- pages = loader.load()
27
- text = "\n".join([p.page_content for p in pages])
28
  finally:
29
  os.remove(tmp_path)
30
 
31
- elif ("application/vnd.openxmlformats-officedocument.wordprocessingml.document" in content_type
32
- or ".docx" in url_lower):
 
 
33
  with tempfile.NamedTemporaryFile(delete=False, suffix=".docx") as tmp:
34
  tmp.write(resp.content)
35
  tmp_path = tmp.name
36
  try:
37
  loader = Docx2txtLoader(tmp_path)
38
- docs = loader.load()
39
- text = "\n".join([d.page_content for d in docs])
40
  finally:
41
  os.remove(tmp_path)
42
 
43
  elif "text/plain" in content_type or ".txt" in url_lower:
44
  text = resp.content.decode("utf-8", errors="ignore")
 
45
 
46
  else:
47
  raise HTTPException(400, f"Unsupported document type: {content_type}")
48
 
49
- # Clean and normalize text
50
- text = re.sub(r'\s+', ' ', text).strip()
51
-
52
- # Semantic chunking
53
- splitter = RecursiveCharacterTextSplitter(
54
- chunk_size=1000,
55
- chunk_overlap=150,
56
- separators=["\n\n", "\n", ". ", "! ", "? ", " ", ""]
57
- )
58
- return splitter.split_text(text)
 
1
  import os
2
  import tempfile
3
  import requests
4
+
5
  from fastapi import HTTPException
6
+ # from langchain_community.document_loaders import PyPDFLoader, Docx2txtLoader
7
  from langchain_community.document_loaders import PyMuPDFLoader, Docx2txtLoader
8
  from langchain.text_splitter import RecursiveCharacterTextSplitter
9
  from langchain.schema import Document
10
 
11
+
12
  def load_and_chunk(url: str) -> list[Document]:
13
  resp = requests.get(url)
14
  if resp.status_code != 200:
15
+ raise HTTPException(400, "Could not download document")
16
 
17
  content_type = resp.headers.get("Content-Type", "").lower()
18
  url_lower = url.lower()
 
19
 
20
  if "application/pdf" in content_type or ".pdf" in url_lower:
21
  with tempfile.NamedTemporaryFile(delete=False, suffix=".pdf") as tmp:
 
23
  tmp_path = tmp.name
24
  try:
25
  loader = PyMuPDFLoader(tmp_path)
26
+ docs = loader.load_and_split()
 
 
27
  finally:
28
  os.remove(tmp_path)
29
 
30
+ elif (
31
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document" in content_type
32
+ or ".docx" in url_lower
33
+ ):
34
  with tempfile.NamedTemporaryFile(delete=False, suffix=".docx") as tmp:
35
  tmp.write(resp.content)
36
  tmp_path = tmp.name
37
  try:
38
  loader = Docx2txtLoader(tmp_path)
39
+ docs = loader.load_and_split()
 
40
  finally:
41
  os.remove(tmp_path)
42
 
43
  elif "text/plain" in content_type or ".txt" in url_lower:
44
  text = resp.content.decode("utf-8", errors="ignore")
45
+ docs = [Document(page_content=text)]
46
 
47
  else:
48
  raise HTTPException(400, f"Unsupported document type: {content_type}")
49
 
50
+ splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=150)
51
+ return splitter.split_documents(docs)