File size: 7,515 Bytes
6235f5a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
import cv2
import pytesseract
import streamlit as st
import av
import numpy as np
from streamlit_webrtc import webrtc_streamer, VideoProcessorBase

class OCRProcessor(VideoProcessorBase):
    def __init__(self):
        # Quick knobs we’ll tweak from the UI
        self.conf_threshold = 25
        self.box_color_bgr = (0, 255, 0)
        self.text_color_bgr = (0, 255, 0)
        self.box_thickness = 1
        self.text_thickness = 1
    def recv(self, frame):
        img = frame.to_ndarray(format="bgr24")
        
        # Step 1: simplify colors → grayscale helps OCR focus on shapes
        gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
        
        # Step 2: separate text/background using Otsu threshold
        thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]
        
        try:
            # Step 3: run OCR and get word boxes + confidences
            data = pytesseract.image_to_data(thresh, output_type=pytesseract.Output.DICT)
            
            # Step 4: draw only what we trust
            for i in range(len(data['text'])):
                # Keep boxes with decent confidence and non-empty text
                if int(data['conf'][i]) > self.conf_threshold and data['text'][i].strip():
                    x, y, w, h = data['left'][i], data['top'][i], data['width'][i], data['height'][i]
                    
                    # Box shows where Tesseract thinks the word is
                    cv2.rectangle(img, (x, y), (x + w, y + h), self.box_color_bgr, self.box_thickness)
                    
                    # Label it so we see what was read
                    cv2.putText(img, data['text'][i], (x, y-10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, self.text_color_bgr, self.text_thickness)
            
        except Exception as e:
            st.error(f"OCR Error: {str(e)}")
        
        return av.VideoFrame.from_ndarray(img, format="bgr24")

def _hex_to_bgr(hex_color):
    # Streamlit gives hex; OpenCV expects BGR tuples
    hex_color = hex_color.lstrip('#')
    if len(hex_color) == 6:
        r = int(hex_color[0:2], 16)
        g = int(hex_color[2:4], 16)
        b = int(hex_color[4:6], 16)
        return (b, g, r)
    return (0, 255, 0)

def run_ocr_on_bgr(image_bgr, conf_threshold=25, box_color_bgr=(0, 255, 0), text_color_bgr=(0, 255, 0), box_thickness=1, text_thickness=1):
    # Same pipeline for still images
    gray = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2GRAY)
    thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]
    annotated = image_bgr.copy()
    texts = []
    try:
        data = pytesseract.image_to_data(thresh, output_type=pytesseract.Output.DICT)
        for i in range(len(data['text'])):
            text = data['text'][i]
            conf_str = data['conf'][i]
            try:
                conf = int(conf_str)
            except Exception:
                conf = -1
            # Quick sanity checks: enough confidence and not empty
            if conf > conf_threshold and text.strip():
                x, y, w, h = data['left'][i], data['top'][i], data['width'][i], data['height'][i]
                cv2.rectangle(annotated, (x, y), (x + w, y + h), box_color_bgr, box_thickness)
                cv2.putText(annotated, text, (x, y - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.5, text_color_bgr, text_thickness)
                texts.append(text)
    except Exception as e:
        st.error(f"OCR Error: {str(e)}")
    return annotated, texts

def read_settings_from_state():
    # Pull current UI settings so everything stays in sync
    conf_threshold = st.session_state["conf_threshold"]
    box_color_bgr = _hex_to_bgr(st.session_state["box_color"])
    text_color_bgr = _hex_to_bgr(st.session_state["text_color"])
    box_thickness = st.session_state["box_thickness"]
    text_thickness = st.session_state["text_thickness"]
    return conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness

def decode_image_from_uploader(file_like):
    # Turn the uploaded file into an OpenCV image
    file_bytes = np.frombuffer(file_like.getvalue(), np.uint8)
    return cv2.imdecode(file_bytes, cv2.IMREAD_COLOR)

def display_annotated_and_text(annotated_bgr, texts, caption):
    # Show what OCR saw and list the tokens found
    st.image(cv2.cvtColor(annotated_bgr, cv2.COLOR_BGR2RGB), caption=caption)
    if texts:
        st.subheader("Detected Text:")
        for t in texts:
            st.write(f"• {t}")
    else:
        st.info("No text detected.")

st.title("Real-time OCR - Computer Vision")
st.write("Pick a mode below and let’s read some text together.")

# Sidebar controls
st.sidebar.header("OCR Settings")

# Defaults
_DEFAULTS = {
    "conf_threshold": 25,
    "box_color": "#00FF00",
    "text_color": "#00FF00",
    "box_thickness": 1,
    "text_thickness": 1,
}

# Initialize session state
for k, v in _DEFAULTS.items():
    if k not in st.session_state:
        st.session_state[k] = v

# Reset button
if st.sidebar.button("Reset to defaults"):
    for k, v in _DEFAULTS.items():
        st.session_state[k] = v
    st.rerun()

# Widgets bound to session state keys
st.sidebar.slider("Confidence threshold", min_value=0, max_value=100, step=1, key="conf_threshold")
st.sidebar.color_picker("Bounding box color", key="box_color")
st.sidebar.color_picker("Text color", key="text_color")
st.sidebar.slider("Box thickness", min_value=1, max_value=5, key="box_thickness")
st.sidebar.slider("Text thickness", min_value=1, max_value=3, key="text_thickness")

# Read current settings
conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness = read_settings_from_state()

mode = st.radio("Mode", ("Real-time", "Take Photo", "Upload Photo"), horizontal=True)

if mode == "Real-time":
    st.write("We’ll use your webcam and draw boxes live.")
    ctx = webrtc_streamer(
        key="ocr-camera",
        video_processor_factory=OCRProcessor,
        rtc_configuration={
            "iceServers": [{"urls": ["stun:stun.l.google.com:19302"]}]
        }
    )
    if ctx and ctx.video_processor:
        # Pass current UI choices into the processor
        ctx.video_processor.conf_threshold = conf_threshold
        ctx.video_processor.box_color_bgr = box_color_bgr
        ctx.video_processor.text_color_bgr = text_color_bgr
        ctx.video_processor.box_thickness = box_thickness
        ctx.video_processor.text_thickness = text_thickness
elif mode == "Take Photo":
    img_file = st.camera_input("Take a photo")
    if img_file is not None:
        image_bgr = decode_image_from_uploader(img_file)
        if image_bgr is not None:
            # One-click OCR on your snapshot
            annotated, texts = run_ocr_on_bgr(image_bgr, conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness)
            display_annotated_and_text(annotated, texts, caption="Annotated photo")
        else:
            st.error("Could not decode image.")
elif mode == "Upload Photo":
    uploaded = st.file_uploader("Upload an image", type=["png", "jpg", "jpeg"])
    if uploaded is not None:
        image_bgr = decode_image_from_uploader(uploaded)
        if image_bgr is not None:
            # Run the same pipeline for your file
            annotated, texts = run_ocr_on_bgr(image_bgr, conf_threshold, box_color_bgr, text_color_bgr, box_thickness, text_thickness)
            display_annotated_and_text(annotated, texts, caption="Annotated upload")
        else:
            st.error("Could not decode image.")