import streamlit as st import random, string import os import os.path from os import listdir from os.path import isfile, join import requests import PyPDF2 from langchain.text_splitter import RecursiveCharacterTextSplitter from pinecone import Pinecone, ServerlessSpec from groq import Groq from sentence_transformers import SentenceTransformer # Access the variables GROQ_API_KEY = st.secrets["GROQ_API_KEY"] PINECONE_API_KEY = st.secrets["PINECONE_API_KEY"] COHERE_API_KEY = st.secrets["COHERE_API_KEY"] # Initialize Groq client client = Groq(api_key = GROQ_API_KEY) # Initialize Pinecone pc = Pinecone(api_key = PINECONE_API_KEY) # Create or connect to an existing index index = pc.Index("sample1") if 'upload_state' not in st.session_state: st.session_state.upload_state = '' if 'chat_list' not in st.session_state: st.session_state.chat_list = [] em_model = SentenceTransformer("all-MiniLM-L6-v2") def get_query_embdedding(embed): query_embedding = em_model.encode([embed]).tolist() return query_embedding st.title('Create Summary From Pdf File') uploaded_files = st.file_uploader("Choose a PDF file", accept_multiple_files=True, type=['pdf']) # if uploaded_file is not None: # bytes_data = uploaded_file.getvalue() # data = uploaded_file.getvalue().decode('utf-8', 'ignore').splitlines() # st.session_state["preview"] = '' # for i in range(0, min(5, len(data))): # st.session_state["preview"] += data[i] # preview = st.text_area("PDF Preview", "", height=150, key="preview") # upload_state = st.text_area("Upload State", "", key="upload_state") pdf_text = '' ns = "ns_"+''.join(random.choice(string.ascii_uppercase + string.ascii_lowercase + string.digits) for _ in range(7)) def upload(ns): for uploaded_file in uploaded_files: if uploaded_file is None: st.session_state.upload_state = "Upload a file first!" else: pdf = PyPDF2.PdfReader(uploaded_file) pdf_text = "" for page in pdf.pages: pdf_text += page.extract_text() st.session_state.upload_state = "Saved successfully!" if pdf_text : # Split the text into chunks text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200) texts = text_splitter.split_text(pdf_text) # # Embedding the chunks r1 = em_model.encode(texts) # # Upsert the embeddings into the index for i in range(len(texts)): index.upsert([((str(i), r1[i], {"text": texts[i]}))], ns) st.write(st.session_state.upload_state) st.button("Upload file to Process", on_click=upload(ns), key="process_but") # if st.button("Upload file to Process"): # upload(ns) # Read the PDF file # pdf = PyPDF2.PdfReader("pdf_files/gandhi.pdf") # pdf_text = "" # for page in pdf.pages: # pdf_text += page.extract_text() # Define the query query = st.chat_input("Enter Your Summarize Query?") # query = "Who is Bhagat singh?" docs = '' if query: # Get the query embedding question_embedding = get_query_embdedding(query) # Query the Pinecone index query_result = index.query(namespace = ns, vector=question_embedding, top_k=5, include_metadata=True) # Extract metadata from query result docs = {x["metadata"]['text']: i for i, x in enumerate(query_result["matches"])} # print (docs) # Create a template for the summary Template = f"Based on the following context: {docs} generate a precise summary related to the question: {query}" # print(Template) # Generate the summary chat_completion = client.chat.completions.create( messages=[ { "role": "user", "content": Template, } ], model="llama3-70b-8192", ) # Print the summary response = chat_completion.choices[0].message.content # print(response) result = {"ques":query, "ans":response} st.session_state.chat_list.append(result) for c_list in st.session_state.chat_list: with st.chat_message("user"): st.write(c_list["ques"]) with st.chat_message("machine"): st.write(c_list["ans"]) # if docs: # with st.chat_message("chatbot"): # st.write(docs)