FOOD_VISION_V2 / utils.py
esssyjr's picture
Update utils.py
4b165c2 verified
Raw
History Blame Contribute Delete
3.14 kB
from huggingface_hub import InferenceClient
from PIL import Image
import io
import base64
import json
def detect_food_or_ingredient(image: Image.Image, client: InferenceClient, api_key: str) -> dict:
"""Detect food or ingredient using a vision-capable LLM (e.g., LLaVA-13B)."""
# Convert image to base64 for LLM processing
buffered = io.BytesIO()
image.save(buffered, format="JPEG")
img_base64 = base64.b64encode(buffered.getvalue()).decode("utf-8")
prompt = """
You are an expert in Nigerian cuisine. Analyze the provided image and identify whether it depicts a prepared food dish or a raw ingredient. Return a JSON response with:
- name: The name of the food or ingredient (e.g., 'jollof rice', 'yam')
- type: Either 'food' or 'ingredient'
Focus on Nigerian foods and ingredients (e.g., egusi soup, pounded yam, tomatoes, okra).
"""
# Use chat completion for vision models like LLaVA
response = client.chat_completion(
messages=[
{
"role": "user",
"content": [
{"type": "text", "text": prompt},
{"type": "image_url", "image_url": f"data:image/jpeg;base64,{img_base64}"}
]
}
],
max_tokens=100
)
result = json.loads(response.choices[0].message.content)
return result
def get_nutritional_info(item: str, client: InferenceClient) -> str:
"""Get nutritional info and preparation methods using Nebius LLM."""
prompt = f"""
You are a nutrition expert specializing in Nigerian cuisine. Provide detailed information about {item}, including:
- Health benefits
- Preparation methods
- Diabetic-friendly options
- Ingredients breakdown
Format the response in markdown with clear sections.
"""
response = client.text_generation(prompt, max_new_tokens=500)
return response
def suggest_recipe(ingredient: str, client: InferenceClient) -> dict:
"""Suggest a recipe based on an ingredient using Nebius LLM."""
prompt = f"""
You are a chef specializing in Nigerian cuisine. Given the ingredient {ingredient}, suggest a recipe that includes:
- Dish name
- Additional ingredients needed
- Step-by-step preparation instructions
Format the response in markdown with clear sections.
"""
response = client.text_generation(prompt, max_new_tokens=500)
return {"dish_name": f"{ingredient}-based dish", "recipe": response}
def generate_dish_image(dish_name: str, api_key: str) -> Image.Image:
"""Generate an image of the prepared dish using Nebius Flux.1."""
client = InferenceClient(model="black-forest-labs/FLUX.1-schnell", token=api_key)
image_bytes = client.text_to_image(f"A beautifully prepared Nigerian dish: {dish_name}")
return Image.open(io.BytesIO(image_bytes))
def speech_to_text(audio_path: str) -> str:
"""Convert audio to text using Hugging Face Whisper model."""
client = InferenceClient(model="openai/whisper-large-v3")
with open(audio_path, "rb") as audio_file:
text = client.audio_to_text(audio_file)
return text