Spaces:
Sleeping
Sleeping
File size: 7,515 Bytes
6235f5a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 | import cv2
import pytesseract
import streamlit as st
import av
import numpy as np
from streamlit_webrtc import webrtc_streamer, VideoProcessorBase
class OCRProcessor(VideoProcessorBase):
def __init__(self):
# Quick knobs we’ll tweak from the UI
self.conf_threshold = 25
self.box_color_bgr = (0, 255, 0)
self.text_color_bgr = (0, 255, 0)
self.box_thickness = 1
self.text_thickness = 1
def recv(self, frame):
img = frame.to_ndarray(format="bgr24")
# Step 1: simplify colors → grayscale helps OCR focus on shapes
gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
# Step 2: separate text/background using Otsu threshold
thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]
try:
# Step 3: run OCR and get word boxes + confidences
data = pytesseract.image_to_data(thresh, output_type=pytesseract.Output.DICT)
# Step 4: draw only what we trust
for i in range(len(data['text'])):
# Keep boxes with decent confidence and non-empty text
if int(data['conf'][i]) > self.conf_threshold and data['text'][i].strip():
x, y, w, h = data['left'][i], data['top'][i], data['width'][i], data['height'][i]
# Box shows where Tesseract thinks the word is
cv2.rectangle(img, (x, y), (x + w, y + h), self.box_color_bgr, self.box_thickness)
# Label it so we see what was read
cv2.putText(img, data['text'][i], (x, y-10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, self.text_color_bgr, self.text_thickness)
except Exception as e:
st.error(f"OCR Error: {str(e)}")
return av.VideoFrame.from_ndarray(img, format="bgr24")
def _hex_to_bgr(hex_color):
# Streamlit gives hex; OpenCV expects BGR tuples
hex_color = hex_color.lstrip('#')
if len(hex_color) == 6:
r = int(hex_color[0:2], 16)
g = int(hex_color[2:4], 16)
b = int(hex_color[4:6], 16)
return (b, g, r)
return (0, 255, 0)
def run_ocr_on_bgr(image_bgr, conf_threshold=25, box_color_bgr=(0, 255, 0), text_color_bgr=(0, 255, 0), box_thickness=1, text_thickness=1):
# Same pipeline for still images
gray = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2GRAY)
thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]
annotated = image_bgr.copy()
texts = []
try:
data = pytesseract.image_to_data(thresh, output_type=pytesseract.Output.DICT)
for i in range(len(data['text'])):
text = data['text'][i]
conf_str = data['conf'][i]
try:
conf = int(conf_str)
except Exception:
conf = -1
# Quick sanity checks: enough confidence and not empty
if conf > conf_threshold and text.strip():
x, y, w, h = data['left'][i], data['top'][i], data['width'][i], data['height'][i]
cv2.rectangle(annotated, (x, y), (x + w, y + h), box_color_bgr, box_thickness)
cv2.putText(annotated, text, (x, y - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, text_color_bgr, text_thickness)
texts.append(text)
except Exception as e:
st.error(f"OCR Error: {str(e)}")
return annotated, texts
def read_settings_from_state():
# Pull current UI settings so everything stays in sync
conf_threshold = st.session_state["conf_threshold"]
box_color_bgr = _hex_to_bgr(st.session_state["box_color"])
text_color_bgr = _hex_to_bgr(st.session_state["text_color"])
box_thickness = st.session_state["box_thickness"]
text_thickness = st.session_state["text_thickness"]
return conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness
def decode_image_from_uploader(file_like):
# Turn the uploaded file into an OpenCV image
file_bytes = np.frombuffer(file_like.getvalue(), np.uint8)
return cv2.imdecode(file_bytes, cv2.IMREAD_COLOR)
def display_annotated_and_text(annotated_bgr, texts, caption):
# Show what OCR saw and list the tokens found
st.image(cv2.cvtColor(annotated_bgr, cv2.COLOR_BGR2RGB), caption=caption)
if texts:
st.subheader("Detected Text:")
for t in texts:
st.write(f"• {t}")
else:
st.info("No text detected.")
st.title("Real-time OCR - Computer Vision")
st.write("Pick a mode below and let’s read some text together.")
# Sidebar controls
st.sidebar.header("OCR Settings")
# Defaults
_DEFAULTS = {
"conf_threshold": 25,
"box_color": "#00FF00",
"text_color": "#00FF00",
"box_thickness": 1,
"text_thickness": 1,
}
# Initialize session state
for k, v in _DEFAULTS.items():
if k not in st.session_state:
st.session_state[k] = v
# Reset button
if st.sidebar.button("Reset to defaults"):
for k, v in _DEFAULTS.items():
st.session_state[k] = v
st.rerun()
# Widgets bound to session state keys
st.sidebar.slider("Confidence threshold", min_value=0, max_value=100, step=1, key="conf_threshold")
st.sidebar.color_picker("Bounding box color", key="box_color")
st.sidebar.color_picker("Text color", key="text_color")
st.sidebar.slider("Box thickness", min_value=1, max_value=5, key="box_thickness")
st.sidebar.slider("Text thickness", min_value=1, max_value=3, key="text_thickness")
# Read current settings
conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness = read_settings_from_state()
mode = st.radio("Mode", ("Real-time", "Take Photo", "Upload Photo"), horizontal=True)
if mode == "Real-time":
st.write("We’ll use your webcam and draw boxes live.")
ctx = webrtc_streamer(
key="ocr-camera",
video_processor_factory=OCRProcessor,
rtc_configuration={
"iceServers": [{"urls": ["stun:stun.l.google.com:19302"]}]
}
)
if ctx and ctx.video_processor:
# Pass current UI choices into the processor
ctx.video_processor.conf_threshold = conf_threshold
ctx.video_processor.box_color_bgr = box_color_bgr
ctx.video_processor.text_color_bgr = text_color_bgr
ctx.video_processor.box_thickness = box_thickness
ctx.video_processor.text_thickness = text_thickness
elif mode == "Take Photo":
img_file = st.camera_input("Take a photo")
if img_file is not None:
image_bgr = decode_image_from_uploader(img_file)
if image_bgr is not None:
# One-click OCR on your snapshot
annotated, texts = run_ocr_on_bgr(image_bgr, conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness)
display_annotated_and_text(annotated, texts, caption="Annotated photo")
else:
st.error("Could not decode image.")
elif mode == "Upload Photo":
uploaded = st.file_uploader("Upload an image", type=["png", "jpg", "jpeg"])
if uploaded is not None:
image_bgr = decode_image_from_uploader(uploaded)
if image_bgr is not None:
# Run the same pipeline for your file
annotated, texts = run_ocr_on_bgr(image_bgr, conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness)
display_annotated_and_text(annotated, texts, caption="Annotated upload")
else:
st.error("Could not decode image.")
|