import streamlit as st import os import json from langchain.chat_models import ChatOpenAI from langchain.embeddings import HuggingFaceEmbeddings from langchain.chains import RetrievalQA from langchain.vectorstores import FAISS from langchain.prompts.prompt import PromptTemplate from langchain.embeddings import HuggingFaceEmbeddings import json import hashlib import pdf2bib # Initialize the storage directories if they don't exist model_name = "intfloat/e5-large-v2" model_kwargs = {'device': 'cpu'} encode_kwargs = {'normalize_embeddings': False} embeddings = HuggingFaceEmbeddings( model_name=model_name, model_kwargs=model_kwargs, encode_kwargs=encode_kwargs ) USERS_DB_FILE = 'users_db.json' def load_users(): if os.path.exists(USERS_DB_FILE): with open(USERS_DB_FILE, 'r') as file: return json.load(file) return {} # Helper function to save users to file def save_users(users_db): with open(USERS_DB_FILE, 'w') as file: json.dump(users_db, file, indent=4) # Helper function to hash passwords def hash_password(password): return hashlib.sha256(password.encode()).hexdigest() # Function to check if a user exists def user_exists(username): users_db = load_users() return username in users_db # Sign-up function def sign_up_user(username, password): users_db = load_users() if username in users_db: return False, "Username already exists." else: assets_dir = os.path.join('user_assets', f'user_{username}_assets') users_db[username] = { "username": username, "password": hash_password(password), "assets_dir": assets_dir } save_users(users_db) # Create user's asset directory if not os.path.exists(assets_dir): os.makedirs(assets_dir) return True, "User created successfully." # Load existing metadata from JSON file def load_metadata(metadata_dir): metadata_path = os.path.join(metadata_dir, 'metadata.json') if os.path.exists(metadata_path): with open(metadata_path, 'r') as f: return json.load(f) return {} def save_metadata(metadata, metadata_dir): metadata_path = os.path.join(metadata_dir, 'metadata.json') with open(metadata_path, 'w') as f: json.dump(metadata, f, indent=4) #Add a new entry to the metadata def add_metadata(file_path, title, author, year,journal, publisher, DOI,page,volume,issue): metadata = load_metadata(st.session_state['metadata_dir']) metadata_dir = st.session_state['metadata_dir'] metadata[file_path] = { 'title': title, 'author': author, 'year': year, 'journal': journal, 'publisher': publisher, 'DOI': DOI, 'page':page, "volume":volume, "issue":issue } save_metadata(metadata, metadata_dir) def edit_metadata_of_selected_pdf(metadata): st.title('PDF Metadata Editor') # Use a select box for users to choose a PDF file_names = list(metadata.keys()) selected_file = st.selectbox('Select a PDF file to edit metadata', file_names) metadata_dir = st.session_state['metadata_dir'] # When a file is selected, show its metadata in editable form fields if selected_file: data = metadata[selected_file] st.subheader('Edit Metadata for Selected PDF:') # Use columns to arrange the text inputs col1, col2 = st.columns(2) with st.form(key=f'edit_form_{selected_file}'): with col1: edited_title = st.text_input('Title', value=data['title']) edited_author = st.text_input('Author', value=data['author']) edited_year = st.text_input('Published Year', value=data['year'], max_chars=4) edited_volume = st.text_input('volume', value=data['volume']) with col2: edited_journal = st.text_input('journal', value=data['journal']) edited_publisher = st.text_input('Publisher', value=data['publisher']) edited_DOI = st.text_input('DOI', value=data['DOI']) edited_page = st.text_input('Page', value=data['page']) edited_issue = st.text_input('issue', value=data['issue']) # Submit button for the form submit_button = st.form_submit_button(label='Update Metadata') if submit_button: # Update the metadata dictionary with the new values metadata[selected_file] = { 'title': edited_title, 'author': edited_author, 'year': edited_year, 'journal': edited_journal, 'publisher': edited_publisher, 'DOI': edited_DOI, 'page': edited_page, 'volume':edited_volume, 'issue':edited_issue } save_metadata(metadata,metadata_dir) # Make sure this function properly handles the saving st.success('Metadata updated successfully!') # ... [The previous code sections remain unchanged] ... # Function to delete selected PDF and its metadata def delete_pdf_and_metadata(metadata): # Use a select box for users to choose a PDF to delete metadata_dir = st.session_state['metadata_dir'] #st.write(list(metadata.keys())) file_names = list(metadata.keys()) if not file_names: st.write("No PDFs available to delete.") return selected_file_to_delete = st.selectbox('Select a PDF file to delete', file_names) st.write(selected_file_to_delete) # Button to delete the PDF and its metadata if st.button(f"Delete '{selected_file_to_delete}' and its metadata"): # Delete the PDF file os.remove(selected_file_to_delete) # Delete the metadata entry del metadata[selected_file_to_delete] save_metadata(metadata, metadata_dir) # Update the display #st.experimental_rerun() def generate_response(input_text,openai_api_key): db = FAISS.load_local(st.session_state['embed_dir'], embeddings) docs = db.similarity_search(input_text,k=5) json1 = json.dumps(docs[0].metadata) json2 = json.dumps(docs[1].metadata) json3 = json.dumps(docs[2].metadata) json4 = json.dumps(docs[3].metadata) json5 = json.dumps(docs[4].metadata) QA_TEMPLATE = """ provide an academic answer within 100 words based on the context :" {context} Question: {question} use metadata in the "metadata" html block to create APA style references and list them in a reference section: {source1},{source2},{source3},{source4},{source5} make sure to add references format the answer in markdown format """ QA_PROMPT = PromptTemplate(input_variables=["question", "context"], partial_variables={"source1":json1, "source2":json2, "source3":json3,"source4":json4,"source5":json5}, template=QA_TEMPLATE, ) llm = ChatOpenAI( model_name="gpt-3.5-turbo", temperature=0.05, max_tokens=1500, openai_api_key=openai_api_key ) # Define retriever retriever = db.as_retriever(search_type="mmr",search_kwargs={"k": 5}) qa_chain = RetrievalQA.from_chain_type(llm, retriever=retriever, chain_type="stuff", #"stuff", "map_reduce","refine", "map_rerank" return_source_documents=True, verbose=True, chain_type_kwargs={"prompt": QA_PROMPT} ) return qa_chain({'query': input_text}) #Dummy user database def hash_password(password): return hashlib.sha256(password.encode()).hexdigest() # Simple user database structure users_db = load_users() # Login function def login_user(username, password): if username in users_db and users_db[username]['password'] == hash_password(password): return True return False # Initialize user assets directory def init_user_assets(username): assets_dir = os.path.join(users_db[username]['assets_dir']) pdf_dir = os.path.join(assets_dir, 'pdfs') metadata_dir = os.path.join(assets_dir, 'metadata') embed_dir = os.path.join(assets_dir, 'embed') if not os.path.exists(pdf_dir): os.makedirs(pdf_dir) if not os.path.exists(metadata_dir): os.makedirs(metadata_dir) if not os.path.exists(embed_dir): os.makedirs(embed_dir) return assets_dir, pdf_dir, metadata_dir, embed_dir def all_author_names(data): author_full_names = [] authors_string =[] if not data: authors_string='' # Iterate over each author in the author list else: for author in data["author"]: # Combine given and family names full_name = f"{author['family']},{author['given']}" author_full_names.append(full_name) # Join the list of author names into a single string separated by commas authors_string = "; ".join(author_full_names) return authors_string # Main App def main(): st.title('LLM powered paper management system') # New sidebar options if 'logged_in' not in st.session_state or not st.session_state['logged_in']: with st.sidebar: tab1, tab2 = st.tabs(["Login", "Sign-up"]) with tab1: username = st.text_input("Username") password = st.text_input("Password", type='password') if st.sidebar.button("Login"): if login_user(username, password): st.session_state['logged_in'] = True st.session_state['username'] = username # Initialize user assets upon login assets_dir, pdf_dir, metadata_dir, embed_dir = init_user_assets(username) st.session_state['assets_dir'] = assets_dir st.session_state['pdf_dir'] = pdf_dir st.session_state['metadata_dir'] = metadata_dir st.session_state['embed_dir'] = embed_dir st.success(f"Logged in as {username}") else: st.error("Incorrect username or password") with tab2: new_username = st.text_input("Choose a username", key="signup_username") new_password = st.text_input("Choose a password", type='password', key="signup_password") confirm_password = st.text_input("Confirm password", type='password', key="confirm_password") if st.button("Sign up"): if new_password == confirm_password: success, message = sign_up_user(new_username, new_password) if success: st.success(message) #st.session_state['logged_in'] = True #st.session_state['username'] = new_username st.experimental_rerun() else: st.error(message) else: st.error("Passwords do not match.") if st.session_state.get('logged_in'): # Continue with the rest of the app tab1, tab2, tab3,tab4 = st.tabs(["Upload pdfs", "Manage metadata", "Embed pdfs","Retrive pdfs"]) # PDF file upload with tab1: uploaded_files = st.file_uploader("Choose PDF files", accept_multiple_files=True, type='pdf') for uploaded_file in uploaded_files: if uploaded_file is not None: # Save file file_path = os.path.join(st.session_state['pdf_dir'], uploaded_file.name) with open(file_path, "wb") as f: f.write(uploaded_file.getbuffer()) pdfextractdata = pdf2bib.pdf2bib(file_path) #st.write(pdfextractdata) pdfextractdata_metadata = {} if pdfextractdata['metadata'] is None else pdfextractdata.get('metadata', {}) #st.write(pdfextractdata_metadata) #st.write(pdfextractdata['metadata']) # Collect metadata with st.form(key=uploaded_file.name): st.write("Metadata for:", uploaded_file.name) col1, col2, col3 = st.columns(3) with col1: title = st.text_input('Title', key=f'title_{uploaded_file.name}', value=pdfextractdata_metadata.get('title', '')) year = st.text_input('Published Year', key=f'year_{uploaded_file.name}', value=pdfextractdata_metadata.get('year', '')) volume = st.text_input('Volume', key=f'volume_{uploaded_file.name}', value=pdfextractdata_metadata.get('volume', '')) with col2: author = st.text_input('Author', key=f'author_{uploaded_file.name}', value=all_author_names(pdfextractdata_metadata)) journal = st.text_input('Journal', key=f'journal_{uploaded_file.name}', value=pdfextractdata_metadata.get('journal', '')) issue = st.text_input('Issue', key=f'issue_{uploaded_file.name}', value=pdfextractdata_metadata.get('issue', '')) with col3: publisher = st.text_input('Publisher', key=f'publisher_{uploaded_file.name}', value=pdfextractdata_metadata.get('publisher', '')) DOI = st.text_input('DOI', key=f'DOI_{uploaded_file.name}', value=pdfextractdata_metadata.get('doi', '')) page = st.text_input('Pages', key=f'page_{uploaded_file.name}', value=pdfextractdata_metadata.get('page', '')) submitted = st.form_submit_button('Save Metadata') if submitted: add_metadata(file_path, title, author, year, journal,publisher, DOI,page,volume,issue) st.success('Metadata saved!') with tab2: edit_metadata_of_selected_pdf(load_metadata(st.session_state['metadata_dir'])) delete_pdf_and_metadata(load_metadata(st.session_state['metadata_dir'])) with tab3: submittedDB = st.button('embed pdfs') if submittedDB: from langchain.document_loaders import PyPDFDirectoryLoader from langchain.text_splitter import RecursiveCharacterTextSplitter from langchain.docstore.document import Document text_splitter = RecursiveCharacterTextSplitter( # Set a really small chunk size, just to show. chunk_size = 1000, chunk_overlap = 0, length_function = len, ) loader = PyPDFDirectoryLoader(st.session_state['pdf_dir']) docs = loader.load_and_split(text_splitter=text_splitter) st.write(docs[0]) for doc in docs: source_file = doc.metadata['source'] metadata = load_metadata(st.session_state['metadata_dir']) # Check if the source_file is in the metadata dictionary if source_file in metadata: # Extract the corresponding metadata corresponding_metadata = metadata[source_file] # Update the page_content metadata with the corresponding details doc.metadata.update({ "title": corresponding_metadata["title"], "author": corresponding_metadata["author"], "year": corresponding_metadata["year"], "journal": corresponding_metadata["journal"], "publisher": corresponding_metadata["publisher"], "DOI": corresponding_metadata["DOI"], "page": corresponding_metadata["page"], }) st.write(docs[0]) db = FAISS.from_documents(docs, embeddings) db.save_local(st.session_state['embed_dir']) with tab4: # load IPCC vector database with st.sidebar: openai_api_key = st.text_input("OpenAI API Key", type="password") "[Get an OpenAI API key](https://platform.openai.com/account/api-keys)" with st.form("my_form"): text = st.text_area("Enter text:", "") submitted = st.form_submit_button("Submit") if not openai_api_key: st.info("Please add your OpenAI API key to continue.") elif submitted: result = generate_response(text, openai_api_key) st.markdown(result["result"]) #st.markdown(result["source_documents"]) # Here you would implement the logic for user sign up # ... # Function calls to hash the password (just for demonstration, remove in production) # make_hashes("admin") if __name__ == '__main__': main()