PDFSearch / app.py
Pranav4datasc's picture
Upload 3 files
d7e954c verified
Raw
History Blame Contribute Delete
3.48 kB
import streamlit as st
from pypdf import PdfReader
from langchain.text_splitter import RecursiveCharacterTextSplitter
import os
from langchain_google_genai import GoogleGenerativeAIEmbeddings
from langchain_google_genai import ChatGoogleGenerativeAI
import google.generativeai as genai
from langchain_community.vectorstores import FAISS
from langchain.chains.question_answering import load_qa_chain
from langchain.prompts import PromptTemplate
from dotenv import load_dotenv
load_dotenv()
genai.configure(api_key=os.getenv("GOOGLE_API_KEY"))
print('Here am I ...1')
def get_pdf_text(pdf_docs):
text =""
print('Here am I ...1.1')
for pdf in pdf_docs:
print('Here am I ...1.2')
pdf_reader=PdfReader(pdf)
print('Here am I ...1.3')
for page in pdf_reader.pages:
print('for pages in...text+=')
text+=page.extract_text()
print('Here am I ...2')
return text
def get_text_chunks(text):
print('Here am I ...3')
text_splitter=RecursiveCharacterTextSplitter(chunk_size=10000, chunk_overlap=1000)
chunks=text_splitter.split_text(text)
return chunks
def get_vector_store(text_chunks):
embeddings=GoogleGenerativeAIEmbeddings(model="models/embeddings-001")
vector_stores=FAISS.from_texts(text_chunks,embedding=embeddings)
vector_stores.save_local("faiss_index")
def get_conversational_chain():
Prompt_template="""
Answer the question as detailed as possible from the provided context, make sure to provide all the details.
if the answer is not in the provided context just say "answer is not available in the context",dont provide the wrong answer.\n\n_
Context:\n{context}?\n
Question:\n{question}\n
Answer:
"""
# print('Here am I ...4')
model=ChatGoogleGenerativeAI(model="gemini-pro",temperature=0.3)
PromptTemplate(template=Prompt_template,input_variables=["context","question"])
chain=load_qa_chain(model,chain_type="stuff",prompt=prompt)
return chain
def user_input(user_question):
embeddings = GoogleGenerativeAIEmbeddings(model="models/embeddings-001")
new_db = FAISS.load_local("faiss_index",embeddings)
docs = new_db.similarity_search(user_question)
chain = get_conversational_chain()
response = chain(
{"input_documents":docs,"question":user_question},
return_only_outputs=True)
print(response)
st.write("Reply:", response["Output_text"])
# print('Here am I ...just bef main')
def main():
# print('Here am I ...inside main')
st.set_page_config("Chat with multiple PDF")
st.header("Chat with multiple PDF using Gemini AI")
print('Here am I ...5')
user_question = st.text_input("Ask a question from a PDF files")
if user_question:
user_input(user_question)
print('Here am I ...6')
with st.sidebar:
st.title("Menu:")
pdf_docs = st.file_uploader("Upload your PDF files and click on the submit button")
if st.button("Submit and Process"):
with st.spinner("Processing..."):
print('Just before get pdf text call...')
raw_text = get_pdf_text(pdf_docs)
print('Just bef get text chunks cal...')
text_chunks = get_text_chunks(raw_text)
get_vector_store(text_chunks)
st.success("Done")
st.print("In main st.print")
#print('Here am I ...the end')
if __name__ == "__main__":
main()