Spaces:
Sleeping
Sleeping
| import streamlit as st | |
| import PyPDF2 | |
| from PIL import Image, ExifTags | |
| import pytesseract | |
| import hashlib | |
| import cv2 | |
| import numpy as np | |
| import io | |
| # Set up Streamlit app title | |
| st.title("DocuVerify - Advanced Document Authentication") | |
| # File uploader | |
| uploaded_file = st.file_uploader("Upload a PDF or Image file", type=["pdf", "jpg", "jpeg", "png", "gif"]) | |
| def compute_hash(file): | |
| hasher = hashlib.sha256() | |
| file.seek(0) | |
| hasher.update(file.read()) | |
| return hasher.hexdigest() | |
| def extract_text_from_image(image): | |
| return pytesseract.image_to_string(image) | |
| def analyze_pdf(file): | |
| try: | |
| reader = PyPDF2.PdfReader(file) | |
| info = reader.metadata | |
| text = "" | |
| for page in reader.pages: | |
| text += page.extract_text() or "" | |
| st.subheader("PDF Metadata") | |
| if info: | |
| for key, value in info.items(): | |
| st.write(f"{key}: {value}") | |
| else: | |
| st.write("No metadata found.") | |
| st.subheader("Extracted Text from PDF") | |
| st.text_area("Extracted Text", text[:10000]) # Display first 10000 chars | |
| file_hash = compute_hash(file) | |
| st.write(f"File Integrity Hash (SHA-256): {file_hash}") | |
| if info and info.get('/Producer') and info.get('/CreationDate'): | |
| return "Legit" | |
| else: | |
| return "Forged (Missing key metadata)" | |
| except Exception as e: | |
| st.write(f"Error analyzing PDF: {str(e)}") | |
| return "Forged (Error reading PDF)" | |
| def analyze_image(file): | |
| try: | |
| img = Image.open(file) | |
| exif_data = img._getexif() | |
| st.subheader("Image Metadata") | |
| if exif_data: | |
| for tag, value in exif_data.items(): | |
| tag_name = ExifTags.TAGS.get(tag, tag) | |
| st.write(f"{tag_name}: {value}") | |
| else: | |
| st.write("No EXIF metadata found.") | |
| text = extract_text_from_image(img) | |
| st.subheader("Extracted Text from Image") | |
| st.text_area("Extracted Text", text[:1000]) | |
| file_hash = compute_hash(file) | |
| st.write(f"File Integrity Hash (SHA-256): {file_hash}") | |
| img_cv = cv2.cvtColor(np.array(img), cv2.COLOR_RGB2BGR) | |
| gray = cv2.cvtColor(img_cv, cv2.COLOR_BGR2GRAY) | |
| edges = cv2.Canny(gray, 50, 150) | |
| st.subheader("Edge Detection Preview") | |
| st.image(edges, caption="Edges Detected", use_column_width=True, channels="GRAY") | |
| if exif_data: | |
| return "Legit" | |
| else: | |
| return "Forged (No EXIF metadata)" | |
| except Exception as e: | |
| st.write(f"Error analyzing image: {str(e)}") | |
| return "Forged (Error processing image)" | |
| # Analyze uploaded file | |
| if uploaded_file is not None: | |
| file_extension = uploaded_file.name.split(".")[-1].lower() | |
| if file_extension == "pdf": | |
| st.write("Analyzing PDF file...") | |
| status = analyze_pdf(uploaded_file) | |
| elif file_extension in ["jpg", "jpeg", "png", "gif"]: | |
| st.write("Analyzing Image file...") | |
| status = analyze_image(uploaded_file) | |
| else: | |
| st.write("Unsupported file format.") | |
| status = "Unsupported Format" | |
| st.subheader("Final Status: ") | |
| st.write(f"**{status}**") | |
| else: | |
| st.write("Please upload a PDF or Image file to analyze.") | |