import streamlit as st from openai import OpenAI from PIL import Image import io import base64 st.set_page_config(page_title="Intelligent OCR & Insight", layout="centered", page_icon="🖼️") st.title("🖼️ Intelligent Image Text Extraction") st.write("Extract, translate, or summarize text using **meta-llama/llama-3.2-11b-vision-instruct**") # --- 1. AKILLI API ANAHTARI YÖNETİMİ --- DEFAULT_API_KEY = "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" try: hf_api_key = st.secrets.get("OPENROUTER_API_KEY", "") except Exception: hf_api_key = "" # Sol menü (Sidebar) Yapılandırması with st.sidebar: st.header("🔑 API Configuration") user_api_key = st.text_input( "Custom OpenRouter API Key (Optional)", type="password", help="The system uses a built-in key by default. If the system key reaches its limit, please provide your own key here." ) if user_api_key: st.success("🎯 Using your custom API key.") else: st.info("⚡ Using system default API key.") st.markdown("---") st.header("⚙️ Task Settings") # Seçenekleri ham LLaMA halüsinasyonlarını engelleyecek şekilde optimize ettik task_type = st.selectbox( "Choose Action", ["Direct Text Extraction", "Extract & Translate to Turkish", "Extract & Summarize"] ) active_api_key = user_api_key if user_api_key else (hf_api_key if hf_api_key else DEFAULT_API_KEY) # --- KRİTİK DÜZELTME: LLaMA için kesin sınırlandırılmış promptlar --- prompt_dict = { "Direct Text Extraction": ( "You are an expert OCR system. Task: Analyze the image and extract all visible text. " "Rules: Return ONLY the raw extracted text exactly as it appears in the image. Do not include any explanations, " "do not introduce code snippets, do not add introductory phrases like 'Here is the text', and do not output random characters." ), "Extract & Translate to Turkish": ( "You are an expert translator. Task: Read the text within the image and translate it line-by-line into fluent Turkish. " "Rules: Output ONLY the direct Turkish translation of the document text. Do not summarize it, do not write a description about the letter, " "and do not add conversational filler." ), "Extract & Summarize": ( "You are an advanced text analytics system. Task: Extract the text from the image and provide a highly structured, " "concise summary using bullet points. Rules: Output ONLY the summary. Do not include meta-commentary or technical jargon." ) } # --- 2. GÖRSEL YÜKLEME VE BASE64 --- uploaded_file = st.file_uploader("Upload an image", type=["png", "jpg", "jpeg"]) def encode_image_to_base64(pil_image): buffered = io.BytesIO() pil_image.save(buffered, format="PNG") return base64.b64encode(buffered.getvalue()).decode('utf-8') if uploaded_file is not None: image = Image.open(uploaded_file) st.image(image, caption="Uploaded Image", use_container_width=True) if st.button("🔍 Process Image"): with st.spinner("LLaMA Vision is analyzing the image..."): try: client = OpenAI(base_url="https://openrouter.ai/api/v1", api_key=active_api_key) base64_image = encode_image_to_base64(image) response = client.chat.completions.create( model="meta-llama/llama-3.2-11b-vision-instruct", messages=[ { "role": "user", "content": [ {"type": "text", "text": prompt_dict[task_type]}, { "type": "image_url", "image_url": {"url": f"data:image/png;base64,{base64_image}"} } ] } ], temperature=0.1 # Düşük sıcaklık: Modelin uydurmasını/halüsinasyon görmesini engeller, netliği artırır! ) extracted_text = response.choices[0].message.content.strip() st.subheader("📄 Analysis Result") st.text_area("Result Output", extracted_text, height=300) st.download_button( label="📥 Download Result as .txt", data=extracted_text, file_name="llama_vision_output.txt", mime="text/plain" ) except Exception as e: st.error("❌ An error occurred during processing.") st.warning( "⚠️ **Quota or Connection Limit Reached:** The built-in shared API key might have run out of credits. " "To continue using the app without interruption, please generate a free/paid key on **OpenRouter.ai** " "and paste it into the **🔑 API Configuration** section in the left sidebar." ) with st.expander("Show technical error details"): st.code(str(e)) # --- SAYFA ALTI (FOOTER) --- st.markdown("---") st.markdown(""" ### 🚀 Real-World Applications & Use Cases This vision-language system leveraging **LLaMA 3.2 Vision** goes beyond traditional OCR by understanding the context, layout, and meaning of the text within images. * **📂 Document Digitization:** Auto-extract tabular data, invoices, and forms directly into structured text. * **🏥 Healthcare Admin:** Convert scanned doctor prescriptions and reports into digital medical data. * **🛒 Retail & Logistics:** Process invoices or scan warehouse package labels automatically. * **🎓 Academic Research:** Convert handwritten study notes or whiteboard layouts into clean text. * **🏭Intelligent Industrial Data Pipelines & Smart Logistics * **Context-Aware Quality Control:** Integrates object detection with semantic extraction. For instance, after a vision model identifies an item (e.g., specific produce, automotive part, or electronic kit), this pipeline parses labels to extract critical context like **expiration dates**, **serial numbers**, or **country of origin**. * **End-to-End Supply Chain Traceability:** Automates compliance checks and inventory logging at shipping hubs without human intervention, reducing friction in dynamic warehouse environments. """)