Image-to-Text
Transformers
Joblib
Persian
English
document-ai
ocr
invoice
persian
enterprise
aria-ai
Instructions to use alirezaaminzadeh/docflow-invoice-parser-fa with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use alirezaaminzadeh/docflow-invoice-parser-fa with Transformers:
# Use a pipeline as a high-level helper # Warning: Pipeline type "image-to-text" is no longer supported in transformers v5. # You must load the model directly (see below) or downgrade to v4.x with: # 'pip install "transformers<5.0.0' from transformers import pipeline pipe = pipeline("image-to-text", model="alirezaaminzadeh/docflow-invoice-parser-fa")# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("alirezaaminzadeh/docflow-invoice-parser-fa", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| """OCR engine abstraction with EasyOCR backend for Persian/English invoices.""" | |
| from __future__ import annotations | |
| import logging | |
| from io import BytesIO | |
| from pathlib import Path | |
| from typing import TYPE_CHECKING | |
| import numpy as np | |
| from PIL import Image | |
| if TYPE_CHECKING: | |
| from numpy.typing import NDArray | |
| logger = logging.getLogger(__name__) | |
| _reader = None | |
| def _get_reader(): | |
| global _reader | |
| if _reader is None: | |
| import easyocr | |
| logger.info("Initializing EasyOCR reader (fa + en)...") | |
| _reader = easyocr.Reader(["fa", "en"], gpu=False, verbose=False) | |
| return _reader | |
| def load_image(source: str | Path | bytes | Image.Image) -> Image.Image: | |
| if isinstance(source, Image.Image): | |
| return source.convert("RGB") | |
| if isinstance(source, bytes): | |
| return Image.open(BytesIO(source)).convert("RGB") | |
| path = Path(source) | |
| if path.suffix.lower() == ".pdf": | |
| from pdf2image import convert_from_path | |
| pages = convert_from_path(str(path), dpi=200, first_page=1, last_page=1) | |
| return pages[0].convert("RGB") | |
| return Image.open(path).convert("RGB") | |
| def image_to_array(image: Image.Image) -> "NDArray[np.uint8]": | |
| return np.array(image) | |
| def extract_text(image: Image.Image) -> tuple[str, list[dict]]: | |
| """Run OCR and return full text plus structured bounding-box results.""" | |
| reader = _get_reader() | |
| arr = image_to_array(image) | |
| results = reader.readtext(arr, detail=1, paragraph=False) | |
| lines: list[str] = [] | |
| structured: list[dict] = [] | |
| for bbox, text, confidence in results: | |
| cleaned = text.strip() | |
| if not cleaned: | |
| continue | |
| lines.append(cleaned) | |
| structured.append( | |
| { | |
| "text": cleaned, | |
| "confidence": float(confidence), | |
| "bbox": [[float(p[0]), float(p[1])] for p in bbox], | |
| } | |
| ) | |
| full_text = "\n".join(lines) | |
| return full_text, structured | |