study-buddy / app /agents /senses_agent.py
GitHub Actions
deploy d092bea3608b7a29952f16357fda39b7a29e399b
2e818da
Raw
History Blame Contribute Delete
4.38 kB
"""Multimodal vision agent -> analyses cropped PDF regions.
Uses Cerebras vision-capable model (gemma-4-31b).
Images must be base64-encoded PNG data URIs -> hosted URLs not supported.
"""
from __future__ import annotations
from typing import Literal
from pydantic import BaseModel
from app.agents.cerebras_client import CerebrasClient
VISION_MODEL_ID = "gemma-4-31b"
class InsightPayload(BaseModel):
summary: str
observations: list[str]
suggested_questions: list[str]
class RegionDescription(BaseModel):
type: Literal["figure", "plot", "diagram", "table", "equation", "other"]
caption: str # one-line description of what the region shows
extracted_content: str # LaTeX for equations, markdown table for tables, else a short description
wiki_term: str = "" # 1-2 word search term for Wiki/Deep Dive flows
class SensesAgent:
def __init__(self) -> None:
self._client = CerebrasClient()
def analyze_visual_context(
self,
image_base64: str,
region_text: str,
anchor_label: str,
familiarity: str,
document_id: str = "",
) -> InsightPayload:
"""Analyse a cropped PDF region image + surrounding text.
Returns structured observations and self-quiz questions.
"""
messages = [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {"url": f"data:image/png;base64,{image_base64}"},
},
{
"type": "text",
"text": (
f"The student highlighted this region from their study material.\n"
f"Concept: '{anchor_label}'\n"
f"Surrounding text: {region_text[:500]}\n"
f"Familiarity level: {familiarity}\n\n"
"Describe what you see in the image, state key observations for "
"learning, and provide 2-3 self-quiz questions the student can use."
),
},
],
}
]
payload = self._client.structured_complete(
messages, InsightPayload, model=VISION_MODEL_ID, max_completion_tokens=4096
)
return payload
def describe_region(self, image_base64: str, type_hint: str = "") -> RegionDescription:
"""Read a cropped page region and return its type, caption, and extracted content.
For a formula → LaTeX; for a table → GitHub-markdown table; otherwise a
short description. Used to make figures/tables/formulas chat- and wiki-ready.
"""
messages = [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {"url": f"data:image/png;base64,{image_base64}"},
},
{
"type": "text",
"text": (
"This is a cropped region from an academic paper"
+ (f" (likely a {type_hint})" if type_hint else "")
+ ".\n"
"1. Classify its type.\n"
"2. Give a one-line caption of what it shows.\n"
"3. extracted_content: if it is a mathematical formula, output ONLY "
"valid LaTeX. If it is a table, output it as a GitHub-markdown table. "
"Otherwise give a 1-2 sentence factual description. "
"4. wiki_term: a 1-2 word noun phrase that would make a good Wikipedia "
"search term for this region. Prefer the main concept over a sentence. "
"If there is no sensible search term, leave it blank.\n"
"Transcribe only what is visible -> do not invent data."
),
},
],
}
]
return self._client.structured_complete(
messages, RegionDescription, model=VISION_MODEL_ID, max_completion_tokens=4096
)