""" Post-processing utilities for OCR outputs Strips reasoning/thinking tokens and normalizes Text / Markdown / JSON formats. """ import re import json from typing import Dict, Any, Tuple, Optional def clean_reasoning_and_special_tokens(text: str) -> str: """ Strips internal model reasoning/thinking tokens like ..., ..., <|thought|>...<|/thought|>, etc., leaving only the final answer. """ if not isinstance(text, str): return str(text) # Remove standard thinking tags (multiline, dotall) text = re.sub(r".*?", "", text, flags=re.DOTALL | re.IGNORECASE) text = re.sub(r".*?", "", text, flags=re.DOTALL | re.IGNORECASE) text = re.sub(r"<\|thought\|>.*?<\|/thought\|>", "", text, flags=re.DOTALL | re.IGNORECASE) text = re.sub(r"\[THINKING\].*?\[/THINKING\]", "", text, flags=re.DOTALL | re.IGNORECASE) text = re.sub(r"<\|reasoning\|>.*?<\|/reasoning\|>", "", text, flags=re.DOTALL | re.IGNORECASE) # Remove dangling open tags if any text = re.sub(r".*", "", text, flags=re.DOTALL | re.IGNORECASE) # Strip common end of generation or special system tokens special_tokens = [ "<|im_end|>", "<|endoftext|>", "<|eot_id|>", "<|end_of_text|>", "<|output|>", "<|input|>", "", "" ] for token in special_tokens: text = text.replace(token, "") return text.strip() def extract_json_from_text(text: str) -> Optional[Dict[str, Any]]: """ Attempts to extract and parse JSON object/array from a model output string. """ if not text: return None cleaned = clean_reasoning_and_special_tokens(text) # Direct json load check try: return json.loads(cleaned) except Exception: pass # Look for ```json ... ``` codeblocks json_block = re.search(r"```(?:json)?\s*([\s\S]*?)\s*```", cleaned) if json_block: try: return json.loads(json_block.group(1).strip()) except Exception: pass # Look for { ... } or [ ... ] brace_match = re.search(r"(\{[\s\S]*\}|\[[\s\S]*\])", cleaned) if brace_match: try: return json.loads(brace_match.group(1).strip()) except Exception: pass return None def format_ocr_output(raw_output: Any, default_type: str = "markdown") -> Dict[str, Any]: """ Normalizes any OCR model raw response into standardized dictionary: { 'text': str, 'markdown': str, 'json': Optional[Any], 'output_type': 'markdown' | 'text' | 'json' } """ if isinstance(raw_output, dict): if "text" in raw_output or "markdown" in raw_output or "json" in raw_output: text_val = clean_reasoning_and_special_tokens(raw_output.get("text", "")) md_val = clean_reasoning_and_special_tokens(raw_output.get("markdown", text_val)) json_val = raw_output.get("json", None) if json_val is None and md_val: json_val = extract_json_from_text(md_val) return { "text": text_val, "markdown": md_val, "json": json_val, "output_type": raw_output.get("output_type", default_type) } else: # The raw output is a json object itself json_str = json.dumps(raw_output, indent=2, ensure_ascii=False) return { "text": json_str, "markdown": f"```json\n{json_str}\n```", "json": raw_output, "output_type": "json" } if isinstance(raw_output, (list, tuple)): # List of items or detections json_str = json.dumps(raw_output, indent=2, ensure_ascii=False) lines = [] for item in raw_output: if isinstance(item, str): lines.append(item) elif isinstance(item, dict) and "text" in item: lines.append(str(item["text"])) else: lines.append(str(item)) text_content = "\n".join(lines) return { "text": text_content, "markdown": text_content, "json": raw_output, "output_type": default_type } # Otherwise raw_output is a string cleaned_text = clean_reasoning_and_special_tokens(str(raw_output)) parsed_json = extract_json_from_text(cleaned_text) output_type = default_type if parsed_json is not None and default_type == "json": output_type = "json" return { "text": cleaned_text, "markdown": cleaned_text, "json": parsed_json, "output_type": output_type }