Final_Assignment

Sleeping

App Files Files Community

layonsan commited on Aug 10, 2025

Commit

f58978f

1 Parent(s): 81917a3

Draft code for agents

Browse files

Files changed (8) hide show

.gitignore +203 -0
.python-version +1 -0
agent.py +208 -0
app.py +10 -3
main.py +6 -0
pyproject.toml +24 -0
utility.py +24 -0
uv.lock +0 -0

.gitignore ADDED Viewed

	@@ -0,0 +1,203 @@

+# Byte-compiled / optimized / DLL files
+__pycache__/
+*.py[codz]
+*$py.class
+# C extensions
+*.so
+# Distribution / packaging
+.Python
+build/
+develop-eggs/
+dist/
+downloads/
+eggs/
+.eggs/
+lib/
+lib64/
+parts/
+sdist/
+var/
+wheels/
+share/python-wheels/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+# PyInstaller
+#  Usually these files are written by a python script from a template
+#  before PyInstaller builds the exe, so as to inject date/other infos into it.
+*.manifest
+*.spec
+# Installer logs
+pip-log.txt
+pip-delete-this-directory.txt
+# Unit test / coverage reports
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+.cache
+nosetests.xml
+coverage.xml
+*.cover
+*.py.cover
+.hypothesis/
+.pytest_cache/
+cover/
+# Translations
+*.mo
+*.pot
+# Django stuff:
+*.log
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+# Flask stuff:
+instance/
+.webassets-cache
+# Scrapy stuff:
+.scrapy
+# Sphinx documentation
+docs/_build/
+# PyBuilder
+.pybuilder/
+target/
+# Jupyter Notebook
+.ipynb_checkpoints
+# IPython
+profile_default/
+ipython_config.py
+# pyenv
+#   For a library or package, you might want to ignore these files since the code is
+#   intended to run in multiple environments; otherwise, check them in:
+# .python-version
+# pipenv
+#   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
+#   However, in case of collaboration, if having platform-specific dependencies or dependencies
+#   having no cross-platform support, pipenv may install dependencies that don't work, or not
+#   install all needed dependencies.
+#Pipfile.lock
+# UV
+#   Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
+#   This is especially recommended for binary packages to ensure reproducibility, and is more
+#   commonly ignored for libraries.
+#uv.lock
+# poetry
+#   Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
+#   This is especially recommended for binary packages to ensure reproducibility, and is more
+#   commonly ignored for libraries.
+#   https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
+#poetry.lock
+#poetry.toml
+# pdm
+#   Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
+#   pdm recommends including project-wide configuration in pdm.toml, but excluding .pdm-python.
+#   https://pdm-project.org/en/latest/usage/project/#working-with-version-control
+#pdm.lock
+#pdm.toml
+.pdm-python
+.pdm-build/
+# pixi
+#   Similar to Pipfile.lock, it is generally recommended to include pixi.lock in version control.
+#pixi.lock
+#   Pixi creates a virtual environment in the .pixi directory, just like venv module creates one
+#   in the .venv directory. It is recommended not to include this directory in version control.
+.pixi
+# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
+__pypackages__/
+# Celery stuff
+celerybeat-schedule
+celerybeat.pid
+# SageMath parsed files
+*.sage.py
+# Environments
+.env
+.envrc
+.venv
+env/
+venv/
+ENV/
+env.bak/
+venv.bak/
+# Spyder project settings
+.spyderproject
+.spyproject
+# Rope project settings
+.ropeproject
+# mkdocs documentation
+/site
+# mypy
+.mypy_cache/
+.dmypy.json
+dmypy.json
+# Pyre type checker
+.pyre/
+# pytype static type analyzer
+.pytype/
+# Cython debug symbols
+cython_debug/
+# PyCharm
+#  JetBrains specific template is maintained in a separate JetBrains.gitignore that can
+#  be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
+#  and can be added to the global gitignore or merged into this file.  For a more nuclear
+#  option (not recommended) you can uncomment the following to ignore the entire idea folder.
+#.idea/
+# Abstra
+# Abstra is an AI-powered process automation framework.
+# Ignore directories containing user credentials, local state, and settings.
+# Learn more at https://abstra.io/docs
+.abstra/
+# Visual Studio Code
+#  Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
+#  that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
+#  and can be added to the global gitignore or merged into this file. However, if you prefer,
+#  you could uncomment the following to ignore the entire vscode folder
+# .vscode/
+# Ruff stuff:
+.ruff_cache/
+# PyPI configuration file
+.pypirc
+# Marimo
+marimo/_static/
+marimo/_lsp/
+__marimo__/
+# Streamlit
+.streamlit/secrets.toml

.python-version ADDED Viewed

	@@ -0,0 +1 @@


1	+ 3.10

agent.py ADDED Viewed

	@@ -0,0 +1,208 @@

+"""LangGraph Agent"""
+import os
+from dotenv import load_dotenv
+from langchain_core.tools import tool
+from langchain_tavily import TavilySearch
+from langchain_community.document_loaders import ArxivLoader, WikipediaLoader
+from langchain_core.messages import AIMessage
+from langgraph.graph import StateGraph, MessagesState
+from langchain_google_genai import ChatGoogleGenerativeAI
+from langchain_groq import ChatGroq
+from langchain_huggingface import ChatHuggingFace, HuggingFaceEndpoint, HuggingFaceEmbeddings
+from langchain_community.vectorstores import SupabaseVectorStore
+from langchain.tools.retriever import create_retriever_tool
+from supabase.client import Client, create_client
+@tool
+def multiply(a: int, b: int) -> int:
+    """Multiply two numbers.
+    Args:
+        a: first int
+        b: second int
+    """
+    return a * b
+@tool
+def add(a: int, b: int) -> int:
+    """Add two numbers.
+    Args:
+        a: first int
+        b: second int
+    """
+    return a + b
+@tool
+def subtract(a: int, b: int) -> int:
+    """Subtract two numbers.
+    Args:
+        a: first int
+        b: second int
+    """
+    return a - b
+@tool
+def divide(a: int, b: int) -> int:
+    """Divide two numbers.
+    Args:
+        a: first int
+        b: second int
+    """
+    if b == 0:
+        raise ValueError("Cannot divide by zero.")
+    return a / b
+@tool
+def modulus(a: int, b: int) -> int:
+    """Get the modulus of two numbers.
+    Args:
+        a: first int
+        b: second int
+    """
+    return a % b
+@tool
+def web_search(query: str) -> str:
+    """Search the web for a query.
+    Args:
+        query: The search query string.
+    Returns:
+        The search results as a string.
+    """
+    raw_result = TavilySearch(max_results=3).invoke(query)
+    search_results = raw_result.get("results", [])
+    formatted_search_results = "\n\n---\n\n".join(
+        [
+            f'<Document source="{res.get("url")}" page=""/>\n{res.get("content", "")}\n</Document>'
+            for res in search_results
+        ])
+    return {"web_results": formatted_search_results}
+@tool
+def arxiv_search(query: str) -> str:
+    """Search Arxiv for a query and return maximum 3 result.
+    Args:
+        query: The search query."""
+    loader = ArxivLoader(query=query, load_max_docs=3).load()
+    docs = loader.load()
+    formatted_list = []
+    for doc in docs:
+        if "id" in doc:
+            arxiv_id = doc["id"]
+            source = f"https://arxiv.org/abs/{arxiv_id}"
+            formatted = f'<Document Source="{source}" page="{doc.metadata.get("page", "")}"/>\n{doc.page_content[:1000]}\n</Document>'
+            formatted_list.append(formatted)
+    formatted_search_docs = "\n\n---\n\n".join(formatted_list)
+    return {"arxiv_results": formatted_search_docs}
+@tool
+def wiki_search(query: str) -> str:
+    """Search Wikipedia for a query and return maximum 3 result.
+    Args:
+        query: The search query."""
+    loader = WikipediaLoader(query=query, load_max_docs=3)
+    docs = loader.load()
+    formatted_docs = "\n\n---\n\n".join(
+        f'<Document Source="{doc.metadata.get("source", "")}" page="{doc.metadata.get("page", "")}"/>\n{doc.page_content[:1000]}\n</Document>'
+        for doc in docs
+    )
+    return {"wiki_results": formatted_docs}
+tools = [
+    multiply,
+    add,
+    subtract,
+    divide,
+    modulus,
+    wiki_search,
+    web_search,
+    arxiv_search,
+]
+# Build retriever
+embeddings = HuggingFaceEmbeddings(model_name="sentence-transformers/all-mpnet-base-v2") #  dim=768
+supabase: Client = create_client(
+    os.environ.get("SUPABASE_URL"),
+    os.environ.get("SUPABASE_SERVICE_KEY"))
+vector_store = SupabaseVectorStore(
+    client=supabase,
+    embedding= embeddings,
+    table_name="documents",
+    query_name="match_documents_langchain",
+)
+create_retriever_tool = create_retriever_tool(
+    retriever=vector_store.as_retriever(),
+    name="Question Search",
+    description="A tool to retrieve similar questions from a vector store.",
+)
+# Build graph function
+def build_graph(provider: str = "google"):
+    """Build the graph"""
+    # Load environment variables from .env file
+    if provider == "google":
+        # Google Gemini
+        llm = ChatGoogleGenerativeAI(model="gemini-2.0-flash", temperature=0)
+    elif provider == "groq":
+        # Groq https://console.groq.com/docs/models
+        llm = ChatGroq(model="qwen-qwq-32b", temperature=0) # optional : qwen-qwq-32b gemma2-9b-it
+    elif provider == "huggingface":
+        # TODO: Add huggingface endpoint
+        llm = ChatHuggingFace(
+            llm=HuggingFaceEndpoint(
+                url="https://api-inference.huggingface.co/models/Meta-DeepLearning/llama-2-7b-chat-hf",
+                temperature=0,
+            ),
+        )
+    else:
+        raise ValueError("Invalid provider. Choose 'google', 'groq' or 'huggingface'.")
+    # Bind tools to LLM
+    llm_with_tools = llm.bind_tools(tools)
+    def retriever(state: MessagesState):
+        query = state["messages"][-1].content
+        similar_doc = vector_store.similarity_search(query, k=1)[0]
+        content = similar_doc.page_content
+        if "Final answer :" in content:
+            answer = content.split("Final answer :")[-1].strip()
+        else:
+            answer = content.strip()
+        return {"messages": [AIMessage(content=answer)]}
+    builder = StateGraph(MessagesState)
+    builder.add_node("retriever", retriever)
+    # Retriever start and end points
+    builder.set_entry_point("retriever")
+    builder.set_finish_point("retriever")
+    # Compile graph
+    return builder.compile()
+if __name__ == "__main__":
+    # Example usage
+    print("testing agent tools")
+    print(web_search("LangGraph Agent"))  # Outputs search results as a string

app.py CHANGED Viewed

@@ -4,6 +4,10 @@ import requests
 import inspect
 import pandas as pd
 # (Keep Constants as is)
 # --- Constants ---
 DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
@@ -13,11 +17,14 @@ DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
 class BasicAgent:
     def __init__(self):
         print("BasicAgent initialized.")
     def __call__(self, question: str) -> str:
         print(f"Agent received question (first 50 chars): {question[:50]}...")
-        fixed_answer = "This is a default answer."
-        print(f"Agent returning fixed answer: {fixed_answer}")
-        return fixed_answer
 def run_and_submit_all( profile: gr.OAuthProfile | None):
     """

 import inspect
 import pandas as pd
+from langchain_core.messages import HumanMessage
+from agent import build_graph
 # (Keep Constants as is)
 # --- Constants ---
 DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
 class BasicAgent:
     def __init__(self):
         print("BasicAgent initialized.")
+        self.graph = build_graph()
     def __call__(self, question: str) -> str:
         print(f"Agent received question (first 50 chars): {question[:50]}...")
+        messages = [HumanMessage(content=question)]
+        result = self.graph.invoke({"messages": messages})
+        answer = result['messages'][-1].content
+        return answer
 def run_and_submit_all( profile: gr.OAuthProfile | None):
     """

main.py ADDED Viewed

	@@ -0,0 +1,6 @@

+def main():
+    print("Hello from final-assignment!")
+if __name__ == "__main__":
+    main()

pyproject.toml ADDED Viewed

	@@ -0,0 +1,24 @@

+[project]
+name = "final-assignment"
+version = "0.1.0"
+description = "Add your description here"
+readme = "README.md"
+requires-python = ">=3.10"
+dependencies = [
+    "arxiv>=2.2.0",
+    "gradio[oauth]>=5.36.2",
+    "ipykernel>=6.30.0",
+    "langchain>=0.3.27",
+    "langchain-community>=0.3.27",
+    "langchain-google-genai>=2.1.9",
+    "langchain-groq>=0.3.7",
+    "langchain-huggingface>=0.3.1",
+    "langchain-tavily>=0.2.11",
+    "langgraph>=0.6.2",
+    "pymupdf>=1.26.3",
+    "python-dotenv>=1.1.1",
+    "requests>=2.32.4",
+    "ruff>=0.12.1",
+    "supabase>=2.18.0",
+    "wikipedia>=1.4.0",
+]

utility.py ADDED Viewed

	@@ -0,0 +1,24 @@

+import json
+def read_jsonl_file(file_path):
+    """
+    Reads a .jsonl file and returns a list of Python dictionaries,
+    where each dictionary represents a JSON object from a line in the file.
+    """
+    data = []
+    try:
+        with open(file_path, 'r', encoding='utf-8') as f:
+            for line in f:
+                # Strip whitespace and check if the line is not empty
+                stripped_line = line.strip()
+                if stripped_line:
+                    try:
+                        json_object = json.loads(stripped_line)
+                        data.append(json_object)
+                    except json.JSONDecodeError as e:
+                        print(f"Error decoding JSON on line: {stripped_line}. Error: {e}")
+    except FileNotFoundError:
+        print(f"Error: The file '{file_path}' was not found.")
+    except Exception as e:
+        print(f"An unexpected error occurred: {e}")
+    return data

uv.lock ADDED Viewed

The diff for this file is too large to render. See raw diff