Final_Assignment / tools.py
srilearns's picture
Changed to GROQ
52ef1d1 verified
Raw
History Blame Contribute Delete
3.6 kB
import pandas as pd
from smolagents import tool
import litellm
from dotenv import load_dotenv
import os
import whisper
from pathlib import Path
@tool
def image_interpretation(file:str) ->str:
"""
Returns a high quality textual description of a provided image
Args:
file: A .png or .jpg image provided by the user
Returns:
Text describing the scene
Example:
query 1: dog.png
output 1: A golden retriever sitting on the beach at mid-day
query 2: scene.jpg
output 2: A landscape picture of sunrise at a mountain range
"""
load_dotenv()
key = os.getenv("GROQ_API_KEY")
prompt = f"""
Open the image at {file} and analyze it and understand it.
Generate a high quality textual description of what you see, it has to be detailed
and accurate, don't make up anything you don't see in the image.
Return ONLY the textual description.
"""
message = [{'role':'user', 'content':prompt}]
response = litellm.completion(model="groq/llama-3.3-70b-versatile", messages=message, api_key=key)
description = response.choices[0].message.content
return description
@tool
def create_transcripts(file:str) -> str:
"""
Returns high quality transcripts of audio files
Args:
file: Users .mp3 file containing audio
Returns:
Text file (transcript of conversation)
Example:
file: hello_world.mp3
output: "Hello world, I am a computer"
"""
file_path = Path(file).resolve()
model = whisper.load_model("base")
result = model.transcribe(str(file_path))
return result["text"]
@tool
def file_handling(file:str) -> pd.DataFrame:
"""
Returns excel sheets and csv files converted into a pandas Dataframe
Args:
file: The file path for the
Outputs:
The user's original content but as a pandas dataframe
"""
extension = file.split('.')[-1]
if extension == 'xlsx':
df = pd.read_excel(f"{file}")
elif extension == 'csv':
df = pd.read_csv(f"{file}")
else:
df = "Sorry, can't open any other files"
return df
@tool
def fix_input(query:str) -> str:
"""
Handles incomplete, missing or jumbled user input
Args:
query: User query that can potentially contain formatting errors like missing errors,
spelling mistakes, jumbled word letter, etc.
Returns:
The corrected user input, which can be interpreted properly in natural language
Example:
query 1: What is th capirsl o Franc
output 1: What is the capital of France?
query 2: I am planning a road trip with my family of 4. Which should I rent?
output 2: I am planning a road trip with my family of 4. Which car should I rent?
"""
load_dotenv()
key = os.getenv("GROQ_API_KEY")
prompt = f""" Is the user's question: {query} in the correct natural language format, is it clear?
Can you understand it as it is? If yes, then return the same query. If not, try (i) correcting
spelling mistakes (ii) filling in missing letters / words (iii) re-arranging jumbled words / word orders
(iv) reversing the string (v) any other suitable string and sentence operations.
Finally return ONLY the corrected string.
"""
message = [{"role": "user", "content": prompt}]
response = litellm.completion(model="groq/llama-3.3-70b-versatile", messages=message, api_key=key)
corrected_query = response.choices[0].message.content
return corrected_query