Spaces:
Sleeping
Sleeping
| from fastapi import FastAPI, UploadFile, File, Form, HTTPException, BackgroundTasks | |
| from fastapi.responses import FileResponse, JSONResponse | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from typing import List, Optional, Dict, Any | |
| import subprocess | |
| import os | |
| import uuid | |
| import shutil | |
| import asyncio | |
| import logging | |
| from datetime import datetime, timedelta | |
| from pathlib import Path | |
| import secrets | |
| import time | |
| import zipfile | |
| import io | |
| from PyPDF2 import PdfReader, PdfWriter | |
| from PIL import Image | |
| # ===================================================== | |
| # CONFIGURATION | |
| # ===================================================== | |
| # File limits | |
| MAX_FILE_SIZE = 50 * 1024 * 1024 # 50MB | |
| MAX_FILES_PER_REQUEST = 10 | |
| # All supported image formats | |
| IMAGE_EXTENSIONS = { | |
| 'jpg', 'jpeg', 'png', 'webp', 'avif', 'heic', 'heif', 'gif', 'bmp', | |
| 'tiff', 'tif', 'svg', 'ico', 'psd', 'psb', 'ai', 'eps', 'raw', 'dng', | |
| 'cr2', 'cr3', 'nef', 'nrw', 'arw', 'sr2', 'srf', 'srw', 'orf', 'rw2', | |
| 'raf', 'rwl', 'pef', 'dcr', 'kdc', 'erf', 'mef', 'mos', 'mrw', 'x3f', | |
| '3fr', 'iiq', 'bay', 'cap', 'dcm', 'dicm', 'fits', 'hdr', 'exr', | |
| 'jp2', 'j2k', 'jpf', 'jpm', 'mj2', 'jxr', 'hdp', 'wdp', 'tga', | |
| 'icb', 'vda', 'vst', 'pcx', 'pic', 'pict', 'pct', 'pnm', 'pbm', | |
| 'pgm', 'ppm', 'pam', 'xbm', 'xpm', 'dds', 'ktx', 'astc', 'bpg', | |
| 'flif', 'qoi', 'cur', 'ani', 'wbmp', 'sgi', 'rgb', 'rgba', 'bw', | |
| 'cin', 'sun', 'ras', 'emf', 'wmf', 'cgm', 'odg', 'apng', 'mng' | |
| } | |
| # Document extensions | |
| DOCUMENT_EXTENSIONS = {'pdf', 'doc', 'docx', 'odt', 'rtf', 'txt', 'html', 'ppt', 'pptx', 'xls', 'xlsx'} | |
| # All allowed extensions | |
| ALLOWED_EXTENSIONS = DOCUMENT_EXTENSIONS | IMAGE_EXTENSIONS | |
| # Output formats (all image formats + common document formats) | |
| ALLOWED_OUTPUT_FORMATS = { | |
| 'pdf', 'docx', 'odt', 'rtf', 'txt', 'html', | |
| 'jpg', 'jpeg', 'png', 'webp', 'gif', 'bmp', 'tiff', 'tif', | |
| 'ico', 'jp2', 'j2k', 'pcx', 'tga', 'ppm', 'pgm', 'pbm', 'pnm' | |
| } | |
| # Directories | |
| TEMP_DIR = "temp" | |
| PROGRESS_DIR = "progress" | |
| LOG_DIR = "logs" | |
| # Retention | |
| FILE_RETENTION_HOURS = 24 # Keep files for 24 hours | |
| CLEANUP_INTERVAL_HOURS = 1 # Run cleanup every hour | |
| PROGRESS_RETENTION_HOURS = 48 # Keep progress data for 48 hours | |
| # Create directories | |
| os.makedirs(TEMP_DIR, exist_ok=True) | |
| os.makedirs(PROGRESS_DIR, exist_ok=True) | |
| os.makedirs(LOG_DIR, exist_ok=True) | |
| # ===================================================== | |
| # LOGGING SETUP | |
| # ===================================================== | |
| # Configure logging with both file and console output | |
| logging.basicConfig( | |
| level=logging.INFO, | |
| format='%(asctime)s - %(name)s - %(levelname)s - %(message)s', | |
| handlers=[ | |
| logging.FileHandler(os.path.join(LOG_DIR, f"app_{datetime.now().strftime('%Y%m%d')}.log")), | |
| logging.StreamHandler() | |
| ] | |
| ) | |
| logger = logging.getLogger(__name__) | |
| # Separate error log | |
| error_handler = logging.FileHandler(os.path.join(LOG_DIR, f"errors_{datetime.now().strftime('%Y%m%d')}.log")) | |
| error_handler.setLevel(logging.ERROR) | |
| error_formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s') | |
| error_handler.setFormatter(error_formatter) | |
| logger.addHandler(error_handler) | |
| # ===================================================== | |
| # PROGRESS STORE | |
| # ===================================================== | |
| progress_store: Dict[str, Dict[str, Any]] = {} | |
| class ProgressTracker: | |
| def __init__(self, job_id: str): | |
| self.job_id = job_id | |
| self.progress = 0 | |
| self.status = "pending" | |
| self.message = "Job started" | |
| self.result_file = None | |
| self.result_files = [] | |
| self.error = None | |
| self.start_time = time.time() | |
| self.is_bulk = False | |
| def update(self, progress: int, message: str = None): | |
| self.progress = progress | |
| if message: | |
| self.message = message | |
| self.save() | |
| logger.info(f"Job {self.job_id} - Progress: {progress}% - {message}") | |
| def complete(self, result_file: str, result_files: List[str] = None): | |
| self.progress = 100 | |
| self.status = "completed" | |
| self.message = "Conversion completed successfully" | |
| self.result_file = result_file | |
| if result_files: | |
| self.result_files = result_files | |
| duration = time.time() - self.start_time | |
| self.save() | |
| logger.info(f"Job {self.job_id} - Completed in {duration:.2f}s - Output: {result_file}") | |
| def complete_bulk(self, result_files: List[str], zip_file: str): | |
| self.progress = 100 | |
| self.status = "completed" | |
| self.message = "Bulk conversion completed successfully" | |
| self.result_file = zip_file | |
| self.result_files = result_files | |
| self.is_bulk = True | |
| duration = time.time() - self.start_time | |
| self.save() | |
| logger.info(f"Job {self.job_id} - Bulk completed in {duration:.2f}s - {len(result_files)} files") | |
| def fail(self, error: str): | |
| self.status = "failed" | |
| self.message = error | |
| self.error = error | |
| duration = time.time() - self.start_time | |
| self.save() | |
| logger.error(f"Job {self.job_id} - Failed after {duration:.2f}s - Error: {error}") | |
| def save(self): | |
| progress_store[self.job_id] = { | |
| "progress": self.progress, | |
| "status": self.status, | |
| "message": self.message, | |
| "result_file": self.result_file, | |
| "result_files": self.result_files, | |
| "is_bulk": self.is_bulk, | |
| "error": self.error, | |
| "timestamp": datetime.now().isoformat(), | |
| "duration_seconds": round(time.time() - self.start_time, 2) | |
| } | |
| # ===================================================== | |
| # FILE SIZE & SECURITY HELPERS | |
| # ===================================================== | |
| def get_safe_filename(original_filename: str) -> tuple[str, str]: | |
| """Generate safe filename with extension validation""" | |
| # Extract extension securely | |
| clean_name = Path(original_filename).name | |
| ext = clean_name.split(".")[-1].lower() if "." in clean_name else "" | |
| # Validate extension | |
| if ext not in ALLOWED_EXTENSIONS: | |
| logger.warning(f"Invalid extension rejected: {ext} from {original_filename}") | |
| raise HTTPException(400, f"File type {ext} not allowed. Allowed: {', '.join(ALLOWED_EXTENSIONS)}") | |
| # Generate secure random filename | |
| random_name = secrets.token_urlsafe(16) | |
| safe_filename = f"{random_name}.{ext}" | |
| return safe_filename, ext | |
| def get_output_filename(original_filename: str, output_format: str) -> str: | |
| """Generate output filename preserving original name""" | |
| # Remove extension from original | |
| base_name = Path(original_filename).stem | |
| # Clean the filename (remove special characters) | |
| clean_name = "".join(c for c in base_name if c.isalnum() or c in " ._-") | |
| # Add output extension | |
| return f"{clean_name}.{output_format.lower()}" | |
| async def save_file_with_limits(upload_file: UploadFile, max_size: int = MAX_FILE_SIZE) -> tuple[str, str, int]: | |
| """Save uploaded file with size limits and security checks""" | |
| # Read file content | |
| content = await upload_file.read() | |
| file_size = len(content) | |
| # Check file size | |
| if file_size > max_size: | |
| logger.warning(f"File too large: {file_size} bytes from {upload_file.filename}") | |
| raise HTTPException( | |
| 400, | |
| f"File too large. Max size: {max_size // 1024 // 1024}MB, Your file: {file_size // 1024 // 1024}MB" | |
| ) | |
| # Generate safe filename | |
| safe_filename, ext = get_safe_filename(upload_file.filename) | |
| filepath = os.path.join(TEMP_DIR, safe_filename) | |
| # Save file | |
| with open(filepath, "wb") as buffer: | |
| buffer.write(content) | |
| logger.info(f"File saved: {safe_filename} (Original: {upload_file.filename}, Size: {file_size} bytes)") | |
| return filepath, upload_file.filename, file_size | |
| def get_file_size_mb(filepath: str) -> float: | |
| """Get file size in MB""" | |
| return os.path.getsize(filepath) / (1024 * 1024) | |
| def get_image_format(ext: str) -> str: | |
| """Map file extension to PIL format""" | |
| format_map = { | |
| 'jpg': 'JPEG', 'jpeg': 'JPEG', 'png': 'PNG', 'gif': 'GIF', | |
| 'bmp': 'BMP', 'tiff': 'TIFF', 'tif': 'TIFF', 'webp': 'WEBP', | |
| 'ico': 'ICO', 'pcx': 'PCX', 'tga': 'TGA', 'ppm': 'PPM', | |
| 'pgm': 'PGM', 'pbm': 'PBM', 'pnm': 'PNM', 'jp2': 'JPEG2000', | |
| 'j2k': 'JPEG2000', 'exr': 'EXR' | |
| } | |
| return format_map.get(ext.lower(), ext.upper()) | |
| # ===================================================== | |
| # CONVERSION HELPERS | |
| # ===================================================== | |
| def libreoffice_convert(input_file: str, output_format: str, output_filename: str = None) -> Optional[str]: | |
| """Convert file using LibreOffice with timeout""" | |
| try: | |
| logger.info(f"Starting LibreOffice conversion: {input_file} -> {output_format}") | |
| # Run conversion with timeout | |
| result = subprocess.run( | |
| [ | |
| "libreoffice", | |
| "--headless", | |
| "--convert-to", | |
| output_format, | |
| input_file, | |
| "--outdir", | |
| TEMP_DIR, | |
| ], | |
| check=True, | |
| capture_output=True, | |
| text=True, | |
| timeout=300 # 5 minute timeout | |
| ) | |
| base_name = os.path.splitext(os.path.basename(input_file))[0] | |
| temp_output = os.path.join(TEMP_DIR, f"{base_name}.{output_format}") | |
| if os.path.exists(temp_output): | |
| # If output_filename is provided, rename the file | |
| if output_filename: | |
| output_file = os.path.join(TEMP_DIR, output_filename) | |
| os.rename(temp_output, output_file) | |
| logger.info(f"LibreOffice conversion successful: {output_file} ({get_file_size_mb(output_file):.2f}MB)") | |
| return output_file | |
| else: | |
| logger.info(f"LibreOffice conversion successful: {temp_output} ({get_file_size_mb(temp_output):.2f}MB)") | |
| return temp_output | |
| else: | |
| logger.error(f"LibreOffice output file not found: {temp_output}") | |
| return None | |
| except subprocess.TimeoutExpired: | |
| logger.error(f"LibreOffice conversion timeout after 300s: {input_file}") | |
| return None | |
| except subprocess.CalledProcessError as e: | |
| logger.error(f"LibreOffice error (code {e.returncode}): {e.stderr}") | |
| return None | |
| except Exception as e: | |
| logger.error(f"LibreOffice conversion exception: {str(e)}", exc_info=True) | |
| return None | |
| def convert_image_to_image(input_path: str, output_path: str, output_format: str) -> bool: | |
| """Convert between image formats using PIL""" | |
| try: | |
| logger.info(f"Converting image: {input_path} -> {output_format}") | |
| # Open the image | |
| image = Image.open(input_path) | |
| # Convert RGBA to RGB for JPEG | |
| if output_format.lower() in ['jpg', 'jpeg'] and image.mode == 'RGBA': | |
| # Create a white background | |
| background = Image.new('RGB', image.size, (255, 255, 255)) | |
| background.paste(image, mask=image.split()[3]) # Use alpha channel as mask | |
| image = background | |
| elif output_format.lower() in ['jpg', 'jpeg'] and image.mode != 'RGB': | |
| image = image.convert('RGB') | |
| # Handle GIF optimization | |
| if output_format.lower() == 'gif': | |
| image.save(output_path, 'GIF', optimize=True) | |
| logger.info(f"Image converted successfully: {output_path} ({get_file_size_mb(output_path):.2f}MB)") | |
| return True | |
| # Save with appropriate format | |
| save_format = get_image_format(output_format) | |
| # Special handling for some formats | |
| if output_format.lower() in ['ico']: | |
| # ICO needs specific sizes | |
| image.save(output_path, format='ICO', sizes=[(image.width, image.height)]) | |
| else: | |
| # Default save | |
| image.save(output_path, format=save_format, quality=95, optimize=True) | |
| logger.info(f"Image converted successfully: {output_path} ({get_file_size_mb(output_path):.2f}MB)") | |
| return True | |
| except Exception as e: | |
| logger.error(f"Image conversion failed: {str(e)}", exc_info=True) | |
| return False | |
| def convert_image_to_pdf(input_path: str, output_path: str) -> bool: | |
| """Convert single image to PDF""" | |
| try: | |
| logger.info(f"Converting image to PDF: {input_path}") | |
| image = Image.open(input_path).convert("RGB") | |
| image.save(output_path, "PDF") | |
| logger.info(f"Image to PDF successful: {output_path} ({get_file_size_mb(output_path):.2f}MB)") | |
| return True | |
| except Exception as e: | |
| logger.error(f"Image to PDF conversion failed: {str(e)}", exc_info=True) | |
| return False | |
| def convert_pdf_to_image(input_path: str, output_path: str, output_format: str = 'jpeg', page_num: int = 0) -> bool: | |
| """Convert PDF to image""" | |
| try: | |
| from pdf2image import convert_from_path | |
| logger.info(f"Converting PDF to image: {input_path}") | |
| pages = convert_from_path(input_path, timeout=60) | |
| if pages and page_num < len(pages): | |
| image = pages[page_num] | |
| # Handle format-specific conversions | |
| if output_format.lower() in ['jpg', 'jpeg'] and image.mode == 'RGBA': | |
| background = Image.new('RGB', image.size, (255, 255, 255)) | |
| background.paste(image, mask=image.split()[3]) | |
| image = background | |
| elif output_format.lower() in ['jpg', 'jpeg'] and image.mode != 'RGB': | |
| image = image.convert('RGB') | |
| save_format = get_image_format(output_format) | |
| image.save(output_path, format=save_format, quality=95) | |
| logger.info(f"PDF to image successful: {output_path} ({get_file_size_mb(output_path):.2f}MB)") | |
| return True | |
| logger.error(f"PDF has no pages or page {page_num} not found. Total pages: {len(pages)}") | |
| return False | |
| except Exception as e: | |
| logger.error(f"PDF to image conversion failed: {str(e)}", exc_info=True) | |
| return False | |
| def merge_pdf_files(file_paths: List[str], output_path: str) -> bool: | |
| """Merge multiple PDF files""" | |
| try: | |
| logger.info(f"Merging {len(file_paths)} PDF files") | |
| writer = PdfWriter() | |
| for file_path in file_paths: | |
| reader = PdfReader(file_path) | |
| for page in reader.pages: | |
| writer.add_page(page) | |
| with open(output_path, "wb") as f: | |
| writer.write(f) | |
| logger.info(f"PDF merge successful: {output_path} ({get_file_size_mb(output_path):.2f}MB)") | |
| return True | |
| except Exception as e: | |
| logger.error(f"PDF merge failed: {str(e)}", exc_info=True) | |
| return False | |
| def split_pdf_file(input_path: str, output_path: str, page_numbers: List[int]) -> bool: | |
| """Split PDF by page numbers""" | |
| try: | |
| logger.info(f"Splitting PDF: {input_path} - Pages: {page_numbers}") | |
| reader = PdfReader(input_path) | |
| writer = PdfWriter() | |
| for page_num in page_numbers: | |
| if 1 <= page_num <= len(reader.pages): | |
| writer.add_page(reader.pages[page_num - 1]) | |
| else: | |
| logger.warning(f"Page {page_num} not found in PDF (total pages: {len(reader.pages)})") | |
| with open(output_path, "wb") as f: | |
| writer.write(f) | |
| logger.info(f"PDF split successful: {output_path} ({get_file_size_mb(output_path):.2f}MB)") | |
| return True | |
| except Exception as e: | |
| logger.error(f"PDF split failed: {str(e)}", exc_info=True) | |
| return False | |
| def is_image_format(extension: str) -> bool: | |
| """Check if extension is an image format""" | |
| return extension.lower() in IMAGE_EXTENSIONS | |
| def create_zip_file(files: List[str], zip_path: str) -> bool: | |
| """Create a zip file from a list of files""" | |
| try: | |
| with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf: | |
| for file_path in files: | |
| if os.path.exists(file_path): | |
| arcname = os.path.basename(file_path) | |
| zipf.write(file_path, arcname) | |
| logger.info(f"Created zip file: {zip_path} with {len(files)} files") | |
| return True | |
| except Exception as e: | |
| logger.error(f"Failed to create zip file: {str(e)}", exc_info=True) | |
| return False | |
| # ===================================================== | |
| # BACKGROUND JOB PROCESSING | |
| # ===================================================== | |
| def process_conversion(job_id: str, input_path: str, original_filename: str, output_format: str, options: dict = None): | |
| """Background task for conversion with progress tracking""" | |
| tracker = ProgressTracker(job_id) | |
| try: | |
| # Validate output format | |
| if output_format.lower() not in ALLOWED_OUTPUT_FORMATS: | |
| raise Exception(f"Unsupported output format: {output_format}. Allowed: {', '.join(ALLOWED_OUTPUT_FORMATS)}") | |
| # Step 1: Validate input | |
| tracker.update(10, "Validating input file") | |
| if not os.path.exists(input_path): | |
| raise Exception("Input file not found") | |
| input_size_mb = get_file_size_mb(input_path) | |
| logger.info(f"Processing job {job_id}: {original_filename} ({input_size_mb:.2f}MB) -> {output_format}") | |
| # Step 2: Check format and process accordingly | |
| tracker.update(20, f"Preparing conversion to {output_format.upper()}") | |
| input_ext = original_filename.split(".")[-1].lower() | |
| output_file = None | |
| # Generate output filename preserving original name | |
| output_filename = get_output_filename(original_filename, output_format) | |
| # Determine conversion type | |
| is_input_image = is_image_format(input_ext) | |
| is_output_image = is_image_format(output_format.lower()) | |
| is_output_pdf = output_format.lower() == 'pdf' | |
| is_input_pdf = input_ext == 'pdf' | |
| # Handle different conversion types | |
| if is_input_image and is_output_pdf: | |
| # Image to PDF | |
| tracker.update(30, "Converting image to PDF") | |
| output_file = os.path.join(TEMP_DIR, output_filename) | |
| success = convert_image_to_pdf(input_path, output_file) | |
| if not success: | |
| raise Exception("Failed to convert image to PDF") | |
| tracker.update(80, "Image converted successfully") | |
| elif is_input_pdf and is_output_image: | |
| # PDF to Image | |
| tracker.update(30, f"Converting PDF to {output_format.upper()}") | |
| output_file = os.path.join(TEMP_DIR, output_filename) | |
| success = convert_pdf_to_image(input_path, output_file, output_format) | |
| if not success: | |
| raise Exception(f"Failed to convert PDF to {output_format}") | |
| tracker.update(80, "PDF converted to image successfully") | |
| elif is_input_image and is_output_image: | |
| # Image to Image | |
| tracker.update(30, f"Converting image to {output_format.upper()}") | |
| output_file = os.path.join(TEMP_DIR, output_filename) | |
| success = convert_image_to_image(input_path, output_file, output_format) | |
| if not success: | |
| raise Exception(f"Failed to convert image to {output_format}") | |
| tracker.update(80, "Image converted successfully") | |
| elif output_format.lower() == "pdf" and options and options.get("merge"): | |
| # Merge PDFs | |
| tracker.update(30, "Merging PDF files") | |
| output_file = os.path.join(TEMP_DIR, output_filename) | |
| files_to_merge = options.get("files", [input_path]) | |
| success = merge_pdf_files(files_to_merge, output_file) | |
| if not success: | |
| raise Exception("Failed to merge PDF files") | |
| tracker.update(80, "PDFs merged successfully") | |
| elif output_format.lower() == "pdf" and options and options.get("split"): | |
| # Split PDF | |
| tracker.update(30, "Splitting PDF file") | |
| output_file = os.path.join(TEMP_DIR, output_filename) | |
| pages = options.get("pages", [1]) | |
| success = split_pdf_file(input_path, output_file, pages) | |
| if not success: | |
| raise Exception("Failed to split PDF file") | |
| tracker.update(80, "PDF split successfully") | |
| else: | |
| # Use LibreOffice for document conversions | |
| tracker.update(40, "Starting LibreOffice conversion") | |
| output_file = libreoffice_convert(input_path, output_format.lower(), output_filename) | |
| if not output_file: | |
| raise Exception(f"LibreOffice conversion failed for {output_format}") | |
| tracker.update(70, "LibreOffice conversion completed") | |
| # Step 3: Verify output | |
| tracker.update(90, "Verifying output file") | |
| if not output_file or not os.path.exists(output_file): | |
| raise Exception("Output file was not created") | |
| output_size_mb = get_file_size_mb(output_file) | |
| logger.info(f"Job {job_id} output size: {output_size_mb:.2f}MB") | |
| # Step 4: Complete | |
| tracker.complete(output_file) | |
| except Exception as e: | |
| logger.error(f"Job {job_id} failed: {str(e)}", exc_info=True) | |
| tracker.fail(str(e)) | |
| finally: | |
| # Clean up input file | |
| try: | |
| if os.path.exists(input_path): | |
| os.remove(input_path) | |
| logger.debug(f"Cleaned up input file: {input_path}") | |
| except Exception as e: | |
| logger.warning(f"Failed to cleanup input file {input_path}: {e}") | |
| def process_bulk_conversion(job_id: str, input_files: List[tuple], output_format: str): | |
| """Background task for bulk conversion with progress tracking""" | |
| tracker = ProgressTracker(job_id) | |
| tracker.is_bulk = True | |
| converted_files = [] | |
| try: | |
| # Validate output format | |
| if output_format.lower() not in ALLOWED_OUTPUT_FORMATS: | |
| raise Exception(f"Unsupported output format: {output_format}. Allowed: {', '.join(ALLOWED_OUTPUT_FORMATS)}") | |
| total_files = len(input_files) | |
| logger.info(f"Processing bulk job {job_id}: {total_files} files -> {output_format}") | |
| for idx, (input_path, original_filename) in enumerate(input_files): | |
| progress = 10 + int((idx / total_files) * 70) | |
| tracker.update(progress, f"Converting file {idx + 1}/{total_files}: {original_filename}") | |
| try: | |
| # Get output filename preserving original name | |
| output_filename = get_output_filename(original_filename, output_format) | |
| output_path = os.path.join(TEMP_DIR, f"{job_id}_{idx}_{output_filename}") | |
| input_ext = original_filename.split(".")[-1].lower() | |
| is_input_image = is_image_format(input_ext) | |
| is_output_image = is_image_format(output_format.lower()) | |
| is_output_pdf = output_format.lower() == 'pdf' | |
| is_input_pdf = input_ext == 'pdf' | |
| success = False | |
| if is_input_image and is_output_pdf: | |
| # Image to PDF | |
| success = convert_image_to_pdf(input_path, output_path) | |
| elif is_input_pdf and is_output_image: | |
| # PDF to Image | |
| success = convert_pdf_to_image(input_path, output_path, output_format) | |
| elif is_input_image and is_output_image: | |
| # Image to Image | |
| success = convert_image_to_image(input_path, output_path, output_format) | |
| else: | |
| # Use LibreOffice for documents | |
| output_file = libreoffice_convert(input_path, output_format.lower(), output_filename) | |
| if output_file and os.path.exists(output_file): | |
| # Rename to unique name for bulk | |
| unique_path = os.path.join(TEMP_DIR, f"{job_id}_{idx}_{output_filename}") | |
| os.rename(output_file, unique_path) | |
| output_path = unique_path | |
| success = True | |
| if success and os.path.exists(output_path): | |
| converted_files.append(output_path) | |
| logger.info(f"Bulk conversion {idx+1}/{total_files}: {original_filename} -> {output_filename}") | |
| else: | |
| logger.warning(f"Bulk conversion failed for: {original_filename}") | |
| except Exception as e: | |
| logger.error(f"Error converting {original_filename}: {str(e)}") | |
| # Continue with other files | |
| tracker.update(90, "Creating zip archive") | |
| # Create zip file | |
| zip_filename = f"converted_files_{job_id}.zip" | |
| zip_path = os.path.join(TEMP_DIR, zip_filename) | |
| if converted_files: | |
| success = create_zip_file(converted_files, zip_path) | |
| if not success: | |
| raise Exception("Failed to create zip file") | |
| tracker.complete_bulk(converted_files, zip_path) | |
| else: | |
| raise Exception("No files were successfully converted") | |
| except Exception as e: | |
| logger.error(f"Bulk job {job_id} failed: {str(e)}", exc_info=True) | |
| tracker.fail(str(e)) | |
| finally: | |
| # Clean up input files | |
| for input_path, _ in input_files: | |
| try: | |
| if os.path.exists(input_path): | |
| os.remove(input_path) | |
| except: | |
| pass | |
| # ===================================================== | |
| # AUTOMATIC CLEANUP | |
| # ===================================================== | |
| async def cleanup_old_files(): | |
| """Remove files older than retention period""" | |
| while True: | |
| try: | |
| await asyncio.sleep(CLEANUP_INTERVAL_HOURS * 3600) | |
| cutoff_time = datetime.now() - timedelta(hours=FILE_RETENTION_HOURS) | |
| deleted_count = 0 | |
| freed_space = 0 | |
| logger.info(f"Starting cleanup - Removing files older than {FILE_RETENTION_HOURS} hours") | |
| # Clean temp directory | |
| for filename in os.listdir(TEMP_DIR): | |
| filepath = os.path.join(TEMP_DIR, filename) | |
| try: | |
| file_mtime = datetime.fromtimestamp(os.path.getmtime(filepath)) | |
| if file_mtime < cutoff_time: | |
| file_size = os.path.getsize(filepath) | |
| os.remove(filepath) | |
| deleted_count += 1 | |
| freed_space += file_size | |
| logger.debug(f"Deleted old file: {filename} ({file_size / (1024*1024):.2f}MB)") | |
| except Exception as e: | |
| logger.warning(f"Failed to delete {filename}: {e}") | |
| # Clean up old progress entries | |
| expired_jobs = [] | |
| for job_id, job_data in progress_store.items(): | |
| if 'timestamp' in job_data: | |
| job_time = datetime.fromisoformat(job_data['timestamp']) | |
| if datetime.now() - job_time > timedelta(hours=PROGRESS_RETENTION_HOURS): | |
| expired_jobs.append(job_id) | |
| for job_id in expired_jobs: | |
| # Also delete result file if it exists | |
| result_file = progress_store[job_id].get('result_file') | |
| if result_file and os.path.exists(result_file): | |
| try: | |
| os.remove(result_file) | |
| logger.debug(f"Deleted result file for expired job: {result_file}") | |
| except: | |
| pass | |
| # Delete individual result files | |
| result_files = progress_store[job_id].get('result_files', []) | |
| for file_path in result_files: | |
| if os.path.exists(file_path): | |
| try: | |
| os.remove(file_path) | |
| except: | |
| pass | |
| del progress_store[job_id] | |
| logger.info(f"Cleanup completed - Deleted {deleted_count} files, freed {freed_space / (1024*1024):.2f}MB, removed {len(expired_jobs)} expired progress entries") | |
| except Exception as e: | |
| logger.error(f"Cleanup task failed: {str(e)}", exc_info=True) | |
| # ===================================================== | |
| # CREATE FASTAPI APP | |
| # ===================================================== | |
| app = FastAPI( | |
| title="Universal Document Converter API", | |
| description="Production-ready document conversion API with file size limits, health checks, and automatic cleanup", | |
| version="2.0.0" | |
| ) | |
| # Add CORS middleware | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=["*"], # Configure appropriately for production | |
| allow_credentials=True, | |
| allow_methods=["GET", "POST", "DELETE"], | |
| allow_headers=["*"], | |
| ) | |
| # ===================================================== | |
| # API ENDPOINTS | |
| # ===================================================== | |
| async def home(): | |
| """API information endpoint""" | |
| return { | |
| "service": "Universal Document Converter API", | |
| "version": "2.0.0", | |
| "status": "running", | |
| "timestamp": datetime.now().isoformat(), | |
| "limits": { | |
| "max_file_size_mb": MAX_FILE_SIZE // (1024 * 1024), | |
| "max_files_per_request": MAX_FILES_PER_REQUEST, | |
| "allowed_extensions": list(ALLOWED_EXTENSIONS), | |
| "output_formats": list(ALLOWED_OUTPUT_FORMATS), | |
| "file_retention_hours": FILE_RETENTION_HOURS | |
| }, | |
| "features": { | |
| "bulk_conversion": True, | |
| "preserve_filenames": True, | |
| "zip_download": True | |
| }, | |
| "endpoints": { | |
| "/convert": "POST - Convert single file with progress tracking", | |
| "/convert/bulk": "POST - Convert multiple files with progress tracking", | |
| "/progress/{job_id}": "GET - Check conversion progress", | |
| "/download/{job_id}": "GET - Download converted file(s)", | |
| "/health": "GET - Health check endpoint", | |
| "/stats": "GET - System statistics", | |
| "/cleanup/{job_id}": "DELETE - Cleanup job files" | |
| } | |
| } | |
| async def health_check(): | |
| """Production health check endpoint""" | |
| checks = { | |
| "libreoffice": False, | |
| "temp_dir": os.path.exists(TEMP_DIR), | |
| "disk_space": True, | |
| "memory_status": "ok" | |
| } | |
| status_code = 200 | |
| # Check LibreOffice availability | |
| try: | |
| result = subprocess.run( | |
| ["libreoffice", "--headless", "--version"], | |
| capture_output=True, | |
| text=True, | |
| timeout=10 | |
| ) | |
| checks["libreoffice"] = result.returncode == 0 | |
| if checks["libreoffice"]: | |
| logger.debug("LibreOffice health check passed") | |
| else: | |
| logger.warning("LibreOffice health check failed") | |
| except Exception as e: | |
| logger.warning(f"LibreOffice health check error: {e}") | |
| checks["libreoffice"] = False | |
| # Check disk space | |
| try: | |
| import shutil | |
| usage = shutil.disk_usage(TEMP_DIR) | |
| free_gb = usage.free / (1024**3) | |
| total_gb = usage.total / (1024**3) | |
| if usage.free < 100 * 1024 * 1024: # Less than 100MB free | |
| checks["disk_space"] = False | |
| checks["disk_warning"] = f"Only {free_gb:.1f}GB free of {total_gb:.1f}GB total" | |
| status_code = 503 | |
| logger.warning(f"Low disk space: {free_gb:.1f}GB free") | |
| else: | |
| logger.debug(f"Disk space OK: {free_gb:.1f}GB free of {total_gb:.1f}GB total") | |
| except Exception as e: | |
| logger.error(f"Disk space check failed: {e}") | |
| # Check progress store size | |
| if len(progress_store) > 1000: | |
| checks["memory_status"] = "high_load" | |
| logger.warning(f"High number of active jobs: {len(progress_store)}") | |
| # Determine overall status | |
| if all([checks["libreoffice"], checks["temp_dir"], checks["disk_space"]]): | |
| status = "healthy" | |
| elif checks["libreoffice"] is False: | |
| status = "degraded" | |
| status_code = 503 | |
| else: | |
| status = "unhealthy" | |
| status_code = 503 | |
| logger.info(f"Health check: {status} (code {status_code})") | |
| return JSONResponse( | |
| status_code=status_code, | |
| content={ | |
| "status": status, | |
| "timestamp": datetime.now().isoformat(), | |
| "checks": checks, | |
| "stats": { | |
| "active_jobs": len(progress_store), | |
| "temp_files": len(os.listdir(TEMP_DIR)), | |
| "file_retention_hours": FILE_RETENTION_HOURS | |
| } | |
| } | |
| ) | |
| async def convert_file( | |
| background_tasks: BackgroundTasks, | |
| output_format: str = Form(...), | |
| file: Optional[UploadFile] = File(None), | |
| merge_files: Optional[List[UploadFile]] = File(None), | |
| split_pages: Optional[str] = Form(None) | |
| ): | |
| """ | |
| Convert documents with file size limits and progress tracking | |
| Limits: | |
| - Max file size: 50MB | |
| - Allowed extensions: All common document and image formats | |
| Returns job_id for progress tracking | |
| """ | |
| job_id = str(uuid.uuid4()) | |
| options = {} | |
| try: | |
| # Validate output format | |
| if output_format.lower() not in ALLOWED_OUTPUT_FORMATS: | |
| raise HTTPException( | |
| 400, | |
| f"Unsupported output format: {output_format}. Allowed: {', '.join(ALLOWED_OUTPUT_FORMATS)}" | |
| ) | |
| # Handle merge operation | |
| if merge_files and output_format.lower() == "pdf": | |
| if len(merge_files) > MAX_FILES_PER_REQUEST: | |
| raise HTTPException(400, f"Too many files. Max {MAX_FILES_PER_REQUEST} files per request") | |
| options["merge"] = True | |
| options["files"] = [] | |
| for upload_file in merge_files: | |
| if not upload_file.filename.lower().endswith('.pdf'): | |
| raise HTTPException(400, f"File {upload_file.filename} is not a PDF") | |
| path, original, size = await save_file_with_limits(upload_file) | |
| options["files"].append(path) | |
| if not options["files"]: | |
| raise HTTPException(400, "No valid PDF files provided for merge") | |
| input_path = options["files"][0] | |
| original_filename = f"merged_{len(options['files'])}_files.pdf" | |
| # Handle split operation | |
| elif split_pages and output_format.lower() == "pdf": | |
| if not file: | |
| raise HTTPException(400, "File required for split operation") | |
| options["split"] = True | |
| try: | |
| pages = [int(p.strip()) for p in split_pages.split(",")] | |
| if not pages: | |
| raise ValueError("No pages specified") | |
| options["pages"] = pages | |
| except ValueError: | |
| raise HTTPException(400, "Invalid page numbers format. Use comma-separated numbers (e.g., '1,3,5')") | |
| input_path, original_filename, file_size = await save_file_with_limits(file) | |
| # Handle single file conversion | |
| elif file: | |
| input_path, original_filename, file_size = await save_file_with_limits(file) | |
| else: | |
| raise HTTPException(400, "No file provided") | |
| # Initialize progress | |
| progress_store[job_id] = { | |
| "progress": 0, | |
| "status": "pending", | |
| "message": "Job queued", | |
| "result_file": None, | |
| "result_files": [], | |
| "is_bulk": False, | |
| "error": None, | |
| "timestamp": datetime.now().isoformat() | |
| } | |
| # Start background task | |
| background_tasks.add_task( | |
| process_conversion, | |
| job_id, | |
| input_path, | |
| original_filename, | |
| output_format, | |
| options | |
| ) | |
| logger.info(f"Job {job_id} queued: {original_filename} -> {output_format}") | |
| return JSONResponse({ | |
| "job_id": job_id, | |
| "status": "processing", | |
| "message": "Conversion started", | |
| "check_progress": f"/progress/{job_id}", | |
| "estimated_time": "5-30 seconds depending on file size" | |
| }) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"Job {job_id} initialization failed: {str(e)}", exc_info=True) | |
| raise HTTPException(500, f"Failed to process request: {str(e)}") | |
| async def convert_bulk_files( | |
| background_tasks: BackgroundTasks, | |
| output_format: str = Form(...), | |
| files: List[UploadFile] = File(...) | |
| ): | |
| """ | |
| Convert multiple files with progress tracking | |
| Limits: | |
| - Max file size per file: 50MB | |
| - Max files: 10 | |
| - Allowed extensions: All common document and image formats | |
| Returns job_id for progress tracking | |
| """ | |
| job_id = str(uuid.uuid4()) | |
| try: | |
| # Validate output format | |
| if output_format.lower() not in ALLOWED_OUTPUT_FORMATS: | |
| raise HTTPException( | |
| 400, | |
| f"Unsupported output format: {output_format}. Allowed: {', '.join(ALLOWED_OUTPUT_FORMATS)}" | |
| ) | |
| # Check number of files | |
| if len(files) > MAX_FILES_PER_REQUEST: | |
| raise HTTPException(400, f"Too many files. Max {MAX_FILES_PER_REQUEST} files per request") | |
| if len(files) == 0: | |
| raise HTTPException(400, "No files provided") | |
| # Save all files | |
| saved_files = [] | |
| for upload_file in files: | |
| path, original, size = await save_file_with_limits(upload_file) | |
| saved_files.append((path, original)) | |
| # Initialize progress | |
| progress_store[job_id] = { | |
| "progress": 0, | |
| "status": "pending", | |
| "message": "Bulk job queued", | |
| "result_file": None, | |
| "result_files": [], | |
| "is_bulk": True, | |
| "error": None, | |
| "timestamp": datetime.now().isoformat() | |
| } | |
| # Start background task | |
| background_tasks.add_task( | |
| process_bulk_conversion, | |
| job_id, | |
| saved_files, | |
| output_format | |
| ) | |
| logger.info(f"Bulk job {job_id} queued: {len(saved_files)} files -> {output_format}") | |
| return JSONResponse({ | |
| "job_id": job_id, | |
| "status": "processing", | |
| "message": f"Bulk conversion started for {len(saved_files)} files", | |
| "check_progress": f"/progress/{job_id}", | |
| "estimated_time": f"{len(saved_files) * 5}-{len(saved_files) * 30} seconds depending on file sizes" | |
| }) | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| logger.error(f"Bulk job {job_id} initialization failed: {str(e)}", exc_info=True) | |
| raise HTTPException(500, f"Failed to process request: {str(e)}") | |
| async def get_progress(job_id: str): | |
| """Check conversion progress""" | |
| if job_id not in progress_store: | |
| raise HTTPException(404, f"Job {job_id} not found") | |
| progress_data = progress_store[job_id] | |
| response = { | |
| "job_id": job_id, | |
| "progress": progress_data["progress"], | |
| "status": progress_data["status"], | |
| "message": progress_data["message"], | |
| "completed": progress_data["progress"] == 100, | |
| "is_bulk": progress_data.get("is_bulk", False), | |
| "error": progress_data.get("error"), | |
| "timestamp": progress_data["timestamp"], | |
| "duration_seconds": progress_data.get("duration_seconds", 0) | |
| } | |
| if progress_data["status"] == "completed" and progress_data.get("result_file"): | |
| response["download_url"] = f"/download/{job_id}" | |
| if progress_data.get("is_bulk") and progress_data.get("result_files"): | |
| response["files_converted"] = len(progress_data["result_files"]) | |
| return JSONResponse(response) | |
| async def download_file(job_id: str): | |
| """Download converted file(s)""" | |
| if job_id not in progress_store: | |
| raise HTTPException(404, f"Job {job_id} not found") | |
| job_data = progress_store[job_id] | |
| if job_data["status"] != "completed": | |
| raise HTTPException(400, f"File not ready. Status: {job_data['status']}, Progress: {job_data['progress']}%") | |
| if not job_data.get("result_file") or not os.path.exists(job_data["result_file"]): | |
| raise HTTPException(404, "Output file not found or has been cleaned up") | |
| filename = os.path.basename(job_data["result_file"]) | |
| file_size_mb = get_file_size_mb(job_data["result_file"]) | |
| logger.info(f"Downloading {filename} ({file_size_mb:.2f}MB) for job {job_id}") | |
| # If it's a zip file for bulk conversion | |
| if job_data.get("is_bulk") and filename.endswith('.zip'): | |
| return FileResponse( | |
| job_data["result_file"], | |
| filename=filename, | |
| media_type="application/zip" | |
| ) | |
| return FileResponse( | |
| job_data["result_file"], | |
| filename=filename, | |
| media_type="application/octet-stream" | |
| ) | |
| async def get_stats(): | |
| """Get system statistics""" | |
| total_temp_files = len(os.listdir(TEMP_DIR)) | |
| total_temp_size = sum(os.path.getsize(os.path.join(TEMP_DIR, f)) for f in os.listdir(TEMP_DIR) if os.path.isfile(os.path.join(TEMP_DIR, f))) | |
| completed_jobs = sum(1 for job in progress_store.values() if job["status"] == "completed") | |
| failed_jobs = sum(1 for job in progress_store.values() if job["status"] == "failed") | |
| pending_jobs = sum(1 for job in progress_store.values() if job["status"] == "pending") | |
| bulk_jobs = sum(1 for job in progress_store.values() if job.get("is_bulk", False)) | |
| return { | |
| "timestamp": datetime.now().isoformat(), | |
| "jobs": { | |
| "total": len(progress_store), | |
| "completed": completed_jobs, | |
| "failed": failed_jobs, | |
| "pending": pending_jobs, | |
| "bulk": bulk_jobs | |
| }, | |
| "storage": { | |
| "temp_files": total_temp_files, | |
| "temp_size_mb": round(total_temp_size / (1024 * 1024), 2), | |
| "file_retention_hours": FILE_RETENTION_HOURS | |
| }, | |
| "limits": { | |
| "max_file_size_mb": MAX_FILE_SIZE // (1024 * 1024), | |
| "max_files_per_request": MAX_FILES_PER_REQUEST, | |
| "allowed_extensions": len(ALLOWED_EXTENSIONS) | |
| } | |
| } | |
| async def cleanup_job(job_id: str): | |
| """Manually clean up job files""" | |
| if job_id not in progress_store: | |
| raise HTTPException(404, f"Job {job_id} not found") | |
| job_data = progress_store[job_id] | |
| # Clean up main result file | |
| result_file = job_data.get("result_file") | |
| if result_file and os.path.exists(result_file): | |
| try: | |
| os.remove(result_file) | |
| logger.info(f"Manually cleaned up result file for job {job_id}: {result_file}") | |
| except Exception as e: | |
| logger.warning(f"Failed to cleanup result file for job {job_id}: {e}") | |
| # Clean up individual result files | |
| result_files = job_data.get("result_files", []) | |
| for file_path in result_files: | |
| if os.path.exists(file_path): | |
| try: | |
| os.remove(file_path) | |
| logger.info(f"Manually cleaned up individual file for job {job_id}: {file_path}") | |
| except Exception as e: | |
| logger.warning(f"Failed to cleanup individual file for job {job_id}: {e}") | |
| del progress_store[job_id] | |
| return JSONResponse({"message": f"Job {job_id} cleaned up successfully"}) | |
| # ===================================================== | |
| # STARTUP EVENT | |
| # ===================================================== | |
| async def startup_event(): | |
| """Initialize production environment""" | |
| logger.info("=" * 60) | |
| logger.info("Starting Universal Document Converter API") | |
| logger.info(f"Version: 2.0.0") | |
| logger.info(f"Configuration:") | |
| logger.info(f" - Max file size: {MAX_FILE_SIZE // (1024 * 1024)}MB") | |
| logger.info(f" - Max files per bulk: {MAX_FILES_PER_REQUEST}") | |
| logger.info(f" - File retention: {FILE_RETENTION_HOURS} hours") | |
| logger.info(f" - Cleanup interval: {CLEANUP_INTERVAL_HOURS} hours") | |
| logger.info(f" - Allowed extensions: {len(ALLOWED_EXTENSIONS)} types") | |
| logger.info(f" - Temp directory: {TEMP_DIR}") | |
| logger.info(f" - Log directory: {LOG_DIR}") | |
| logger.info(f"Features:") | |
| logger.info(f" - Bulk conversion: Enabled") | |
| logger.info(f" - Preserve filenames: Enabled") | |
| logger.info(f" - ZIP download: Enabled") | |
| # Verify LibreOffice availability | |
| try: | |
| result = subprocess.run(["libreoffice", "--headless", "--version"], capture_output=True, text=True) | |
| if result.returncode == 0: | |
| logger.info(f"LibreOffice available: {result.stdout.strip()}") | |
| else: | |
| logger.error("LibreOffice not available - conversions will fail!") | |
| except Exception as e: | |
| logger.error(f"Failed to check LibreOffice: {e}") | |
| # Start background cleanup task | |
| asyncio.create_task(cleanup_old_files()) | |
| logger.info("Background cleanup task started") | |
| logger.info("API startup complete - Ready to accept requests") | |
| logger.info("=" * 60) | |
| # ===================================================== | |
| # MAIN (for development) | |
| # ===================================================== | |
| if __name__ == "__main__": | |
| import uvicorn | |
| uvicorn.run( | |
| "app:app", | |
| host="0.0.0.0", | |
| port=8000, | |
| reload=True, | |
| log_level="info" | |
| ) |