Image-Text-to-Text
Transformers
Safetensors
nemotron_parse
feature-extraction
VLM
OCR
Parse
conversational
custom_code
Instructions to use sassoftware/NVIDIA-Nemotron-Parse-v1.2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use sassoftware/NVIDIA-Nemotron-Parse-v1.2 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="sassoftware/NVIDIA-Nemotron-Parse-v1.2", trust_remote_code=True) messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("sassoftware/NVIDIA-Nemotron-Parse-v1.2", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use sassoftware/NVIDIA-Nemotron-Parse-v1.2 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "sassoftware/NVIDIA-Nemotron-Parse-v1.2" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "sassoftware/NVIDIA-Nemotron-Parse-v1.2", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/sassoftware/NVIDIA-Nemotron-Parse-v1.2
- SGLang
How to use sassoftware/NVIDIA-Nemotron-Parse-v1.2 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "sassoftware/NVIDIA-Nemotron-Parse-v1.2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "sassoftware/NVIDIA-Nemotron-Parse-v1.2", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "sassoftware/NVIDIA-Nemotron-Parse-v1.2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "sassoftware/NVIDIA-Nemotron-Parse-v1.2", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use sassoftware/NVIDIA-Nemotron-Parse-v1.2 with Docker Model Runner:
docker model run hf.co/sassoftware/NVIDIA-Nemotron-Parse-v1.2
File size: 4,073 Bytes
0f80974 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 | import re
from latex2html import (
convert_html_table_to_csv,
convert_html_table_to_json,
convert_html_tables_to_markdown,
latex_table_to_html,
)
def extract_classes_bboxes(text: str):
_re_extract_class_bbox = re.compile(r'<x_(\d+(?:\.\d+)?)><y_(\d+(?:\.\d+)?)>(.*?)<x_(\d+(?:\.\d+)?)><y_(\d+(?:\.\d+)?)><class_([^>]+)>', re.DOTALL)
classes = []
bboxes = []
texts = []
for m in _re_extract_class_bbox.finditer(text):
x1, y1, text, x2, y2, cls = m.groups()
classes.append(cls)
bboxes.append((float(x1), float(y1), float(x2), float(y2)))
texts.append(text)
# TODO: Remove when fixed
classes = [
"Formula" if cls == "Inline-formula" else cls for cls in classes
]
assert "Page-number" not in classes
return classes, bboxes, texts
def transform_bbox_to_original(bbox, original_width, original_height, target_w=1664, target_h=2048):
# Replicate exact resize logic
aspect_ratio = original_width / original_height
new_height = original_height
new_width = original_width
if original_height > target_h:
new_height = target_h
new_width = int(new_height * aspect_ratio)
if new_width > target_w:
new_width = target_w
new_height = int(new_width / aspect_ratio)
resized_width = new_width
resized_height = new_height
# Calculate padding
pad_left = (target_w - resized_width) // 2
pad_top = (target_h - resized_height) // 2
# # Transform: use the ACTUAL resized dimensions, not the scale
# # X coords
left = ((bbox[0] * target_w) - pad_left) * original_width / resized_width
right = ((bbox[2] * target_w) - pad_left) * original_width / resized_width
# # Y coords - using original_height / resized_height directly
top = ((bbox[1] * target_h) - pad_top) * original_height / resized_height
bottom = ((bbox[3] * target_h) - pad_top) * original_height / resized_height
return left, top, right, bottom
def postprocess_text(text, cls = 'Text', text_format='markdown', table_format='latex', blank_text_in_figures=False):
assert text_format in ['markdown', 'plain'], 'Unknown text format. Supported: markdown | plain'
assert table_format in ['latex', 'HTML', 'markdown', 'json', 'csv'], \
'Unknown table format. Supported: latex | HTML | markdown | json | csv'
if cls != 'Table':
if text_format == 'plain':
text = convert_mmd_to_plain_text_ours(text)
elif table_format == 'HTML':
text = latex_table_to_html(text)
elif table_format == 'markdown':
text = convert_html_tables_to_markdown(latex_table_to_html(text))
elif table_format == 'json':
text = convert_html_table_to_json(latex_table_to_html(text))
elif table_format == 'csv':
text = convert_html_table_to_csv(latex_table_to_html(text))
if blank_text_in_figures and cls == 'Picture':
text = ''
return text
def remove_nemotron_formatting(text):
text = text.replace('<tbc>', '')
text = text.replace('\\<|unk|\\>', '')
text = text.replace('\\unknown', '')
return text
def convert_mmd_to_plain_text_ours(mmd_text):
mmd_text = re.sub(r'<sup>(.*?)</sup>', r'^{\\1}', mmd_text, flags=re.DOTALL)
mmd_text = re.sub(r'<sub>(.*?)</sub>', r'_{\\1}', mmd_text, flags=re.DOTALL)
mmd_text = mmd_text.replace('<br>', '\n')
# Remove headers (e.g., ##)
mmd_text = re.sub(r'#+\s', '', mmd_text)
# Remove bold (e.g., **)
mmd_text = re.sub(r'\*\*(.*?)\*\*', r'\1', mmd_text)
#mmd_text = mmd_text.replace("**","")
# Remove italic (e.g., *)
mmd_text = re.sub(r'\*(.*?)\*', r'\1', mmd_text)
# Remove emphasized text formatting (e.g., _)
mmd_text = re.sub(r'(?<!\w)_([^_]+)_', r'\1', mmd_text)
# Remove formulas inside paragraphs (e.g., \(R_{ij}(P^{a})=0\))
#mmd_text = re.sub(r'\\\((.*?)\\\)', '', mmd_text)
# Remove asterisk in lists
#mmd_text = re.sub(r'^\*\s', '', mmd_text, flags=re.MULTILINE)
return mmd_text.strip()
|