Spaces:
Runtime error
Runtime error
| import pdfplumber | |
| import pandas as pd | |
| import re | |
| import os | |
| from typing import List, Dict, Tuple, Any | |
| class PDFSectionExtractor: | |
| def __init__(self, pdf_path: str): | |
| """Initialize with path to PDF file.""" | |
| self.pdf_path = pdf_path | |
| self.tables = [] | |
| self.table_names = {} | |
| self.table_content_markers = set() | |
| self.table_content_lines = set() | |
| self.table_headers = set() | |
| self.document_name = None # Set to None to identify when it's missing | |
| self.document_id = None | |
| def extract_document_info_from_pdf(self) -> Tuple[str, str]: | |
| """Extract document name and ID from the first page of PDF.""" | |
| with pdfplumber.open(self.pdf_path) as pdf: | |
| if not pdf.pages: | |
| return "", "" | |
| first_page_text = pdf.pages[0].extract_text() | |
| if not first_page_text: | |
| return "", "" | |
| doc_name_pattern = r'[dD]ocument\s*[nN]ame:\s*(.*?)(?:\n|$)' | |
| doc_id_pattern = r'[dD]ocument\s*[iI][dD]:\s*(.*?)(?:\n|$)' | |
| doc_name_match = re.search(doc_name_pattern, first_page_text) | |
| doc_id_match = re.search(doc_id_pattern, first_page_text) | |
| print(f" extract document info function{doc_name_match}") | |
| doc_name = doc_name_match.group(1).strip() if doc_name_match else "" | |
| doc_id = doc_id_match.group(1).strip() if doc_id_match else "" | |
| # print(f"Document name from pdf inside function{doc_name},{doc_id}") | |
| return doc_name, doc_id | |
| def get_document_info(self) -> Tuple[str, str]: | |
| """ | |
| Get document info from the PDF content. | |
| If missing, fall back to filename for document ID only. | |
| """ | |
| if self.document_name is None or self.document_id is None: | |
| # Try extracting from PDF | |
| doc_name, doc_id = self.extract_document_info_from_pdf() | |
| self.document_name = doc_name | |
| self.document_id = doc_id | |
| # print(f"Entering if one,{self.document_name}") | |
| # Fallback only if document ID or name is missing | |
| if not self.document_name or not self.document_id: | |
| doc_name = os.path.basename(self.pdf_path) | |
| doc_id_match = re.search(r'\[\d{2}\s+\d{2}\s+\d{2}\]\s+(\d+)', doc_name) | |
| if not self.document_id: | |
| self.document_id = doc_id_match.group(1) if doc_id_match else "" | |
| if not self.document_name: | |
| self.document_name = doc_name | |
| # print(f"final{self.document_name},{self.document_id}") | |
| return self.document_name, self.document_id | |
| def is_level_three_section(self, section_number: str) -> bool: | |
| """Check if the section number is a level three section (e.g., 2.1.1).""" | |
| return len(section_number.split('.')) == 3 | |
| def find_table_names(self, text: str) -> List[Dict[str, str]]: | |
| """Extract table names from text.""" | |
| table_pattern = r'\*\*Table\s+(\d+):\s+([^*]+)\*\*' | |
| return [(match.group(1), match.group(2).strip()) | |
| for match in re.finditer(table_pattern, text)] | |
| def store_table_headers(self, table: List[List[str]]): | |
| """Store table headers.""" | |
| if table and table[0]: | |
| header_row = table[0] | |
| for header in header_row: | |
| if header: | |
| header_text = self.fix_table_cell_spacing(header) | |
| if header_text: | |
| self.table_headers.add(header_text) | |
| def store_table_content(self, table: List[List[str]]): | |
| """Store all content from table cells, including headers.""" | |
| if not table: | |
| return | |
| # Store headers separately | |
| self.store_table_headers(table) | |
| # Store all cell content | |
| for row in table: | |
| for cell in row: | |
| if cell: | |
| cell_text = self.fix_table_cell_spacing(cell) | |
| if cell_text: | |
| self.table_content_lines.add(cell_text) | |
| def fix_table_cell_spacing(self, cell_text: str) -> str: | |
| """Fix spacing issues within table cells.""" | |
| if not isinstance(cell_text, str): | |
| return str(cell_text) | |
| text = str(cell_text).strip() | |
| text = re.sub(r',(?=\S)', ', ', text) | |
| text = re.sub(r'(\d)([a-zA-Z])', r'\1 \2', text) | |
| text = re.sub(r'([a-zA-Z])(\d)', r'\1 \2', text) | |
| text = re.sub(r'([a-z])([A-Z])', r'\1 \2', text) | |
| text = re.sub(r'\s+', ' ', text) | |
| return text.strip() | |
| def contains_table_content(self, line: str) -> bool: | |
| """Check if a line contains any stored table content or headers.""" | |
| line = self.fix_table_cell_spacing(line) | |
| return (any(table_line in line for table_line in self.table_content_lines) or | |
| any(header in line for header in self.table_headers)) | |
| def is_table_content(self, line: str) -> bool: | |
| """Identify if a line is part of a table.""" | |
| if any(marker in line for marker in ['|', '+', '─', '│', '┌', '┐', '└', '┘', '├', '┤', '┬', '┴']): | |
| return True | |
| if re.search(r'\S+\s{2,}\S+', line): | |
| return True | |
| if re.match(r'^[\s\-+|=_]{3,}$', line): | |
| return True | |
| return False | |
| def filter_table_content(self, content: str) -> str: | |
| """Remove lines that contain table content.""" | |
| lines = content.split('\n') | |
| filtered_lines = [] | |
| for line in lines: | |
| line = line.strip() | |
| if line and not self.contains_table_content(line): | |
| filtered_lines.append(line) | |
| return '\n'.join(filtered_lines) | |
| def process_table(self, table: List[List[str]], table_num: int) -> pd.DataFrame: | |
| """Process table data with proper spacing and ensure unique column names.""" | |
| if not table: | |
| return pd.DataFrame() | |
| # Store all table content including headers | |
| self.store_table_content(table) | |
| # Create DataFrame with default column names | |
| df = pd.DataFrame(table) | |
| if df.empty: | |
| return df | |
| # Get the header row (first row) | |
| header = df.iloc[0].tolist() | |
| # Create unique column names | |
| unique_cols = [] | |
| col_count = {} | |
| for col in header: | |
| col = str(col) if col else "Unnamed" | |
| if col in col_count: | |
| col_count[col] += 1 | |
| unique_cols.append(f"{col}_{col_count[col]}") | |
| else: | |
| col_count[col] = 0 | |
| unique_cols.append(col) | |
| # Set the unique column names | |
| df.columns = unique_cols | |
| # Remove the header row since we used it for column names | |
| df = df.iloc[1:].reset_index(drop=True) | |
| # Fix spacing in cells | |
| for col in df.columns: | |
| df[col] = df[col].astype(str).apply(self.fix_table_cell_spacing) | |
| # Add metadata | |
| df.insert(0, 'Document_Name', self.document_name) | |
| df.insert(1, 'Document_ID', self.document_id) | |
| df.insert(2, 'Table_Number', table_num) | |
| return df | |
| def extract_tables(self) -> List[pd.DataFrame]: | |
| """Extract tables from PDF and return a list of DataFrames.""" | |
| tables = [] | |
| table_counter = 1 | |
| # Ensure we have document info before processing tables | |
| self.get_document_info() | |
| with pdfplumber.open(self.pdf_path) as pdf: | |
| for page in pdf.pages: | |
| detected_tables = page.extract_tables() | |
| for table in detected_tables: | |
| if table: | |
| df = self.process_table(table, table_counter) | |
| if not df.empty: | |
| tables.append(df) | |
| table_counter += 1 | |
| return tables | |
| def extract_sections(self) -> List[Dict[str, str]]: | |
| """Extract sections from PDF with content, excluding tables.""" | |
| sections = [] | |
| current_section = None | |
| current_content = [] | |
| section_pattern = r'^(\d+\.(?:\d+)?(?:\.\d+)?)\s+(.+)$' | |
| # Ensure we have document info before processing sections | |
| doc_name, doc_id = self.get_document_info() | |
| table_counter = 1 | |
| with pdfplumber.open(self.pdf_path) as pdf: | |
| for page in pdf.pages: | |
| text = page.extract_text(x_tolerance=1) | |
| if not text: | |
| continue | |
| lines = text.split('\n') | |
| table_found = False | |
| for line in lines: | |
| line = line.strip() | |
| match = re.match(section_pattern, line) | |
| if match: | |
| if current_section: | |
| content_text = self.process_content(current_content) | |
| content_text = self.filter_table_content(content_text) | |
| sections.append(self.create_section_dict( | |
| doc_name, doc_id, current_section, content_text)) | |
| current_section = (match.group(1), match.group(2)) | |
| current_content = [] | |
| elif current_section: | |
| if self.is_table_content(line): | |
| if not table_found: | |
| current_content.append(f"[Table {table_counter} is available in Table_{table_counter} sheet]") | |
| table_counter += 1 | |
| table_found = True | |
| else: | |
| current_content.append(line) | |
| table_found = False | |
| if current_section: | |
| content_text = self.process_content(current_content) | |
| content_text = self.filter_table_content(content_text) | |
| sections.append(self.create_section_dict( | |
| doc_name, doc_id, current_section, content_text)) | |
| return sections | |
| def process_content(self, content_lines: List[str]) -> str: | |
| """Process content lines and remove duplicate table references.""" | |
| processed_lines = [] | |
| last_line_was_table_ref = False | |
| for line in content_lines: # Fixed: using content_lines parameter | |
| line = line.strip() | |
| if '[Table' in line: | |
| if not last_line_was_table_ref: | |
| processed_lines.append(line) | |
| last_line_was_table_ref = True | |
| else: | |
| processed_lines.append(line) | |
| last_line_was_table_ref = False | |
| return '\n'.join(processed_lines) | |
| def create_section_dict(self, doc_name: str, doc_id: str, | |
| section_tuple: Tuple[str, str], content: str) -> Dict[str, str]: | |
| """Create a dictionary for section data.""" | |
| section_number, section_name = section_tuple | |
| if self.is_level_three_section(section_number): | |
| return { | |
| 'document_name': doc_name, | |
| 'document_id': doc_id, | |
| 'section_number': section_number, | |
| 'section_name': '', | |
| 'content': f"{section_name}\n{content}" | |
| } | |
| else: | |
| return { | |
| 'document_name': doc_name, | |
| 'document_id': doc_id, | |
| 'section_number': section_number, | |
| 'section_name': section_name, | |
| 'content': content | |
| } | |
| def convert_pdf_to_excel(pdf_path: str, excel_path: str): | |
| """Convert PDF with sections and tables to Excel file with separate sheets for each table.""" | |
| try: | |
| extractor = PDFSectionExtractor(pdf_path) | |
| # Extract and process tables | |
| tables = extractor.extract_tables() | |
| # Extract sections | |
| sections = extractor.extract_sections() | |
| df_sections = pd.DataFrame(sections) | |
| # Write to Excel file | |
| with pd.ExcelWriter(excel_path, engine='openpyxl') as writer: | |
| # Write sections sheet | |
| df_sections.to_excel(writer, index=False, sheet_name='Sections') | |
| # Write each table to its own sheet | |
| for i, table_df in enumerate(tables, 1): | |
| sheet_name = f'Table_{i}' | |
| table_df.to_excel(writer, index=False, sheet_name=sheet_name) | |
| print(f"Successfully saved PDF data to {excel_path}") | |
| print(f"Created {len(tables)} table sheets") | |
| except Exception as e: | |
| print(f"Error occurred: {e}") | |
| raise | |
| if __name__ == "__main__": | |
| pdf_path = "[00 12 10] 26251-100-3DR-S04-00001_002.docx.pdf" | |
| excel_path = "format.xlsx" | |
| convert_pdf_to_excel(pdf_path, excel_path) |