Spaces:
Sleeping
Sleeping
| import os | |
| import json | |
| from huggingface_hub import InferenceClient | |
| from schema import ExtractedDocument | |
| from document_parser import extract_text_from_file | |
| def process_document(file_path: str, original_filename: str) -> dict: | |
| raw_text = extract_text_from_file(file_path, original_filename) | |
| # Connect to the Qwen 2.5 72B model | |
| client = InferenceClient("Qwen/Qwen2.5-72B-Instruct", token=os.environ.get("HF_TOKEN")) | |
| system_prompt = f"""You are a strict, enterprise-grade data extraction AI. | |
| Read the text and extract the data perfectly. | |
| CRITICAL RULES: | |
| 1. Output ONLY valid JSON. No markdown blocks. | |
| 2. You MUST use the exact categories provided in the schema. Do not invent new document categories. | |
| 3. Be exhaustive. If a field exists in the schema (like tax_id or utility_usage) and the data exists in the text, you MUST extract it. | |
| SCHEMA: | |
| {ExtractedDocument.model_json_schema()} | |
| """ | |
| messages = [ | |
| {"role": "system", "content": system_prompt}, | |
| {"role": "user", "content": f"Analyze and extract this document:\n\n{raw_text}"} | |
| ] | |
| try: | |
| response = client.chat_completion( | |
| messages=messages, | |
| max_tokens=1024, | |
| temperature=0.1 # Keep it low for strict data extraction | |
| ) | |
| # Clean the output in case Qwen adds stray markdown ticks | |
| raw_ai_output = response.choices[0].message.content.strip() | |
| if raw_ai_output.startswith("```json"): | |
| raw_ai_output = raw_ai_output[7:-3].strip() | |
| elif raw_ai_output.startswith("```"): | |
| raw_ai_output = raw_ai_output[3:-3].strip() | |
| parsed_json = json.loads(raw_ai_output) | |
| validated_data = ExtractedDocument(**parsed_json) | |
| return validated_data.model_dump() | |
| except Exception as e: | |
| raise RuntimeError(f"Data Extraction Failed: {str(e)}") |