Spaces:
Sleeping
Sleeping
| import os | |
| import io | |
| from flask import Flask, render_template, request, jsonify, send_file | |
| from werkzeug.utils import secure_filename | |
| import PyPDF2 | |
| from io import BytesIO | |
| from reportlab.pdfgen import canvas | |
| from reportlab.lib.pagesizes import letter | |
| import nltk | |
| from nltk.tokenize import sent_tokenize, word_tokenize | |
| from nltk.corpus import stopwords | |
| from collections import defaultdict | |
| import heapq | |
| import tempfile | |
| import json | |
| # ============================== | |
| # FIXED NLTK DOWNLOAD SECTION | |
| # ============================== | |
| def download_nltk_resources(): | |
| resources = ['punkt', 'punkt_tab', 'stopwords'] | |
| for resource in resources: | |
| try: | |
| nltk.data.find(f'tokenizers/{resource}') | |
| except LookupError: | |
| try: | |
| nltk.data.find(f'corpora/{resource}') | |
| except LookupError: | |
| nltk.download(resource) | |
| download_nltk_resources() | |
| # ============================== | |
| app = Flask(__name__) | |
| app.config['MAX_CONTENT_LENGTH'] = 16 * 1024 * 1024 # 16MB max file size | |
| app.config['UPLOAD_FOLDER'] = tempfile.gettempdir() | |
| app.config['SECRET_KEY'] = 'your-secret-key-here' | |
| ALLOWED_EXTENSIONS = {'pdf'} | |
| def allowed_file(filename): | |
| return '.' in filename and filename.rsplit('.', 1)[1].lower() in ALLOWED_EXTENSIONS | |
| def extract_text_from_pdf(pdf_file): | |
| """Extract text from uploaded PDF file""" | |
| pdf_reader = PyPDF2.PdfReader(pdf_file) | |
| text = "" | |
| for page in pdf_reader.pages: | |
| page_text = page.extract_text() | |
| if page_text: | |
| text += page_text + "\n" | |
| return text | |
| def summarize_text(text, num_sentences=5): | |
| """Summarize text using frequency-based scoring""" | |
| sentences = sent_tokenize(text) | |
| if len(sentences) <= num_sentences: | |
| return text | |
| stop_words = set(stopwords.words('english')) | |
| words = word_tokenize(text.lower()) | |
| words = [word for word in words if word.isalnum() and word not in stop_words] | |
| word_freq = defaultdict(int) | |
| for word in words: | |
| word_freq[word] += 1 | |
| sentence_scores = defaultdict(int) | |
| for i, sentence in enumerate(sentences): | |
| sentence_words = word_tokenize(sentence.lower()) | |
| for word in sentence_words: | |
| if word in word_freq: | |
| sentence_scores[i] += word_freq[word] | |
| top_sentences = heapq.nlargest(num_sentences, sentence_scores, key=sentence_scores.get) | |
| top_sentences.sort() | |
| summary = [sentences[i] for i in top_sentences] | |
| return ' '.join(summary) | |
| def analyze_pdf_statistics(text): | |
| """Analyze PDF and return statistics""" | |
| sentences = sent_tokenize(text) | |
| words = word_tokenize(text) | |
| characters = len(text) | |
| return { | |
| 'page_count': text.count('\f') + 1, | |
| 'sentence_count': len(sentences), | |
| 'word_count': len(words), | |
| 'character_count': characters, | |
| 'avg_sentence_length': round(len(words) / len(sentences), 2) if sentences else 0, | |
| 'avg_word_length': round(sum(len(word) for word in words) / len(words), 2) if words else 0 | |
| } | |
| def create_summary_pdf(summary, filename="summary.pdf"): | |
| """Create a downloadable PDF from summary""" | |
| buffer = BytesIO() | |
| pdf = canvas.Canvas(buffer, pagesize=letter) | |
| pdf.setTitle("PDF Summary") | |
| pdf.setFont("Helvetica-Bold", 16) | |
| pdf.drawString(72, 750, "PDF Document Summary") | |
| pdf.line(72, 745, 540, 745) | |
| pdf.setFont("Helvetica", 12) | |
| y_position = 720 | |
| max_width = 500 | |
| words = summary.split() | |
| line = [] | |
| for word in words: | |
| line.append(word) | |
| test_line = ' '.join(line) | |
| if pdf.stringWidth(test_line, "Helvetica", 12) > max_width: | |
| line.pop() | |
| pdf.drawString(72, y_position, ' '.join(line)) | |
| y_position -= 20 | |
| line = [word] | |
| if y_position < 50: | |
| pdf.showPage() | |
| pdf.setFont("Helvetica", 12) | |
| y_position = 750 | |
| if line: | |
| pdf.drawString(72, y_position, ' '.join(line)) | |
| pdf.save() | |
| buffer.seek(0) | |
| return buffer | |
| def index(): | |
| return render_template('index.html') | |
| def upload_pdf(): | |
| if 'pdf_file' not in request.files: | |
| return jsonify({'error': 'No file uploaded'}), 400 | |
| file = request.files['pdf_file'] | |
| if file.filename == '': | |
| return jsonify({'error': 'No file selected'}), 400 | |
| if not allowed_file(file.filename): | |
| return jsonify({'error': 'Only PDF files are allowed'}), 400 | |
| try: | |
| text = extract_text_from_pdf(file) | |
| if not text.strip(): | |
| return jsonify({'error': 'Could not extract text from PDF. The PDF might be scanned or image-based.'}), 400 | |
| summary_ratio = float(request.form.get('summary_ratio', 0.3)) | |
| total_sentences = len(sent_tokenize(text)) | |
| num_sentences = max(3, int(total_sentences * summary_ratio)) | |
| summary = summarize_text(text, num_sentences) | |
| stats = analyze_pdf_statistics(text) | |
| original_words = len(word_tokenize(text)) | |
| summary_words = len(word_tokenize(summary)) | |
| compression_ratio = ( | |
| ((original_words - summary_words) / original_words) * 100 | |
| if original_words > 0 else 0 | |
| ) | |
| return jsonify({ | |
| 'success': True, | |
| 'summary': summary, | |
| 'statistics': stats, | |
| 'compression_ratio': round(compression_ratio, 2), | |
| 'summary_length': summary_words, | |
| 'filename': secure_filename(file.filename) | |
| }) | |
| except Exception as e: | |
| return jsonify({'error': f'Error processing PDF: {str(e)}'}), 500 | |
| def download_summary(): | |
| data = request.json | |
| summary = data.get('summary', '') | |
| filename = data.get('filename', 'summary.pdf') | |
| if not summary: | |
| return jsonify({'error': 'No summary to download'}), 400 | |
| try: | |
| pdf_buffer = create_summary_pdf(summary, filename) | |
| return send_file( | |
| pdf_buffer, | |
| as_attachment=True, | |
| download_name=filename.replace('.pdf', '_summary.pdf'), | |
| mimetype='application/pdf' | |
| ) | |
| except Exception as e: | |
| return jsonify({'error': f'Error creating PDF: {str(e)}'}), 500 | |
| def export_summary(): | |
| data = request.json | |
| summary = data.get('summary', '') | |
| if not summary: | |
| return jsonify({'error': 'No summary to export'}), 400 | |
| return send_file( | |
| BytesIO(summary.encode()), | |
| as_attachment=True, | |
| download_name='summary.txt', | |
| mimetype='text/plain' | |
| ) | |
| if __name__ == '__main__': | |
| os.makedirs('uploads', exist_ok=True) | |
| os.makedirs('templates', exist_ok=True) | |
| os.makedirs('static', exist_ok=True) | |
| print("Starting PDF Summarizer Server...") | |
| print("Open your browser and navigate to: http://localhost:5000") | |
| app.run(debug=True, host='0.0.0.0', port = 7860) |