Image-to-Text
PEFT
Safetensors
Portuguese
English
vision-language
table-extraction
scientific-figures
markdown-table
qwen2.5-vl
lora
icdar-metric-loss
Instructions to use lucasoc/sci-image-models with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use lucasoc/sci-image-models with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2.5-VL-3B-Instruct") model = PeftModel.from_pretrained(base_model, "lucasoc/sci-image-models") - Notebooks
- Google Colab
- Kaggle
| # Standalone inference example using the Hugging Face adapter | |
| import torch | |
| from PIL import Image | |
| from transformers import AutoProcessor, Qwen2_5_VLForConditionalGeneration, BitsAndBytesConfig | |
| from peft import PeftModel | |
| from qwen_vl_utils import process_vision_info | |
| import sys | |
| def main(image_path: str, adapter_id: str = "."): | |
| model_id = "Qwen/Qwen2.5-VL-3B-Instruct" | |
| print(f"Loading processor for {adapter_id}...") | |
| processor = AutoProcessor.from_pretrained(adapter_id) | |
| print("Loading base model in 4-bit...") | |
| bnb_config = BitsAndBytesConfig( | |
| load_in_4bit=True, | |
| bnb_4bit_quant_type="nf4", | |
| bnb_4bit_compute_dtype=torch.float16, | |
| bnb_4bit_use_double_quant=True, | |
| ) | |
| base_model = Qwen2_5_VLForConditionalGeneration.from_pretrained( | |
| model_id, | |
| quantization_config=bnb_config, | |
| torch_dtype=torch.float16, | |
| device_map="auto" | |
| ) | |
| print("Loading PEFT LoRA adapter...") | |
| model = PeftModel.from_pretrained(base_model, adapter_id) | |
| model.eval() | |
| image = Image.open(image_path).convert("RGB") | |
| messages = [ | |
| { | |
| "role": "user", | |
| "content": [ | |
| {"type": "image", "image": image}, | |
| {"type": "text", "text": "Extract the plotted quantitative data into a clean Markdown table with column headers."} | |
| ] | |
| } | |
| ] | |
| text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) | |
| image_inputs, _ = process_vision_info(messages) | |
| inputs = processor(text=[text], images=image_inputs, padding=True, return_tensors="pt").to("cuda") | |
| with torch.inference_mode(): | |
| generated_ids = model.generate(**inputs, max_new_tokens=1024, temperature=0.0) | |
| generated_ids_trimmed = [ | |
| out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids) | |
| ] | |
| output_text = processor.batch_decode( | |
| generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False | |
| )[0] | |
| print("\n=== Extracted Markdown Table ===\n") | |
| print(output_text) | |
| if __name__ == "__main__": | |
| img = sys.argv[1] if len(sys.argv) > 1 else "figure.png" | |
| main(img) | |