SummaryModel / app.py
akhilsu's picture
Update app.py
a3cddef verified
Raw
History Blame Contribute Delete
3.8 kB
import streamlit as st
from sentence_transformers import SentenceTransformer, util
import PyPDF2
from PIL import Image
import pytesseract
import speech_recognition as sr
import docx
import tempfile
import os
# Load embedding model
model = SentenceTransformer('all-MiniLM-L6-v2')
# Database to store file content and embeddings
file_db = []
# Function to extract text from a PDF
def extract_text_from_pdf(pdf_file):
pdf_reader = PyPDF2.PdfReader(pdf_file)
text = ""
for page in pdf_reader.pages:
text += page.extract_text()
return text
# Function to extract text from an image
def extract_text_from_image(image_file):
image = Image.open(image_file)
text = pytesseract.image_to_string(image)
return text
# Function to extract text from a video (audio)
def extract_text_from_video(video_file):
r = sr.Recognizer()
with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as temp_audio:
temp_audio_path = temp_audio.name
os.system(f"ffmpeg -i {video_file.name} -vn -ar 16000 -ac 1 -f wav {temp_audio_path}")
with sr.AudioFile(temp_audio_path) as source:
audio_data = r.record(source)
os.remove(temp_audio_path)
return r.recognize_google(audio_data)
# Function to extract text from a TXT file
def extract_text_from_txt(txt_file):
return txt_file.read().decode("utf-8")
# Function to extract text from a Word document
def extract_text_from_word(docx_file):
doc = docx.Document(docx_file)
text = "\n".join([paragraph.text for paragraph in doc.paragraphs])
return text
# Function to process files and generate embeddings
def process_file(file):
file_type = file.type.split('/')[1]
content = ""
try:
if file.type == 'application/pdf': # PDF files
content = extract_text_from_pdf(file)
elif file.type in ['image/png', 'image/jpeg']: # Images
content = extract_text_from_image(file)
elif file.type == 'video/mp4': # Videos
content = extract_text_from_video(file)
elif file.type == 'text/plain': # TXT files
content = extract_text_from_txt(file)
elif file.name.endswith('.docx'): # Word documents
content = extract_text_from_word(file)
except Exception as e:
st.error(f"Error processing file {file.name}: {e}")
if content:
embedding = model.encode(content, convert_to_tensor=True)
file_db.append({"name": file.name, "content": content, "embedding": embedding})
st.success(f"File '{file.name}' processed successfully!")
else:
st.warning(f"Could not extract text from '{file.name}'")
# Streamlit UI
st.title("AI-based Text Search for Files")
st.write("Upload documents, images, or videos, and perform AI-powered text search.")
# File upload section
uploaded_files = st.file_uploader(
"Upload your files",
type=['pdf', 'png', 'jpg', 'jpeg', 'mp4', 'txt', 'docx'],
accept_multiple_files=True
)
if uploaded_files:
for file in uploaded_files:
process_file(file)
# Search section
query = st.text_input("Enter your search query:")
if query:
query_embedding = model.encode(query, convert_to_tensor=True)
results = []
for file in file_db:
similarity = util.pytorch_cos_sim(query_embedding, file['embedding'])[0][0].item()
results.append({"name": file['name'], "similarity": similarity, "content": file['content']})
# Sort results by similarity
results = sorted(results, key=lambda x: x['similarity'], reverse=True)
st.write("Search Results:")
for result in results[:5]: # Display top 5 results
st.write(f"**{result['name']}** (Similarity: {result['similarity']:.2f})")
st.write(result['content'][:500] + "...") # Show a preview