{ "cells": [ { "cell_type": "code", "execution_count": 1, "id": "e1d5b361-1136-479d-b9a7-0fe69c9e949e", "metadata": {}, "outputs": [], "source": [ "import os\n", "import dotenv\n", "\n", "from dotenv import load_dotenv, find_dotenv\n", "_ = load_dotenv(find_dotenv()) # read local .env file\n", "\n", "dotenv.load_dotenv()\n", "hf_api_key = os.environ['HF_API_KEY']" ] }, { "cell_type": "code", "execution_count": 2, "id": "9844aadb-ca4e-4af9-8b4e-8d2e7c68c141", "metadata": {}, "outputs": [], "source": [ "# Helper function\n", "import requests, json\n", "\n", "#Summarization endpoint\n", "def get_summary(inputs, parameters=None,ENDPOINT_URL=os.environ['HF_API_SUMMARY_BASE']): \n", " headers = {\n", " \"Authorization\": f\"Bearer {hf_api_key}\",\n", " \"Content-Type\": \"application/json\"\n", " }\n", " data = { \"inputs\": inputs }\n", " if parameters is not None:\n", " data.update({\"parameters\": parameters})\n", " response = requests.request(\"POST\",\n", " ENDPOINT_URL, headers=headers,\n", " data=json.dumps(data)\n", " )\n", " \n", " return json.loads(response.content.decode(\"utf-8\"))" ] }, { "cell_type": "code", "execution_count": 3, "id": "8db16da1-edbf-479e-956f-e182b6efebc5", "metadata": {}, "outputs": [], "source": [ "from langchain.chat_models import ChatOpenAI\n", "from langchain.document_loaders import WebBaseLoader\n", "from langchain.chains.summarize import load_summarize_chain\n" ] }, { "cell_type": "code", "execution_count": 4, "id": "97d04666-3245-46c0-913a-be4e459b2224", "metadata": {}, "outputs": [], "source": [ "from langchain.tools import Tool\n", "from langchain.utilities import GoogleSearchAPIWrapper\n", "\n", "search = GoogleSearchAPIWrapper()\n", "\n", "def top5_results(query):\n", " return search.results(query, 5)\n", "\n", "\n", "tool5 = Tool(\n", " name=\"Google Search Snippets\",\n", " description=\"Search Google for recent results.\",\n", " func=top5_results,\n", ")\n", "\n", "tool1 = Tool(\n", " name=\"Google Search\",\n", " description=\"Search Google for recent results.\",\n", " func=search.run,\n", ")" ] }, { "cell_type": "code", "execution_count": 5, "id": "07d9710c-b770-48ec-9df6-a22014f28f39", "metadata": {}, "outputs": [], "source": [ "from langchain.document_loaders import WebBaseLoader" ] }, { "cell_type": "code", "execution_count": 64, "id": "ff430a0f-4d9d-4abb-b28c-2e1e7210e0c5", "metadata": {}, "outputs": [], "source": [ "def get1Result(query):\n", " return tool1.run(query)\n", " \n", "def getLinks(query):\n", " results = tool5.run(query)\n", " #This will return Title, Link and Snippet however for now we are interested in only getting the link\n", " links = []\n", " for item in results:\n", " links.append(item['link']) \n", " return links\n", " \n", "def getContent(query):\n", " summaries = []\n", " links = getLinks(query)\n", " #for now just taking the first website and generating the summary of it.\n", " loader = WebBaseLoader(links)\n", " docs = loader.load()\n", " \n", " # Split the document_list into individual lists\n", " all_documents = [[document] for document in docs]\n", " llm = ChatOpenAI(temperature=0, model_name=\"gpt-3.5-turbo-16k\")\n", " chain = load_summarize_chain(llm, chain_type=\"stuff\")\n", " \n", " # Now, each element in individual_lists is a list containing one Document object\n", " for individual_docs in all_documents:\n", " #reducing the content limit so that it doesn't exceeds openAI context window\n", " individual_docs[0].page_content = individual_docs[0].page_content[:5000]\n", " summary = chain.run(individual_docs)\n", " summaries.append(summary) \n", " \n", " return summaries, links\n", " \n", "def get5Results(query):\n", " output_text = \"\"\n", " i = 1\n", " summaries, links = getContent(query)\n", " for summary, link in zip(summaries, links):\n", " output_text += f\"{i}. Summary: {summary} \\n\\nLink: {link} \\n\\n\"\n", " i += 1\n", " return output_text\n", "\n", "def AIGoogleSearch(query):\n", " #output = get1Result(query)\n", " output = get5Results(query)\n", " return output" ] }, { "cell_type": "code", "execution_count": 67, "id": "c051bce0-0067-4362-948f-c64b2adb1b1c", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Running on local URL: http://127.0.0.1:7863\n", "Running on public URL: https://63c6dd458928e0fa28.gradio.live\n", "\n", "This share link expires in 72 hours. For free permanent hosting and GPU upgrades, run `gradio deploy` from Terminal to deploy to Spaces (https://huggingface.co/spaces)\n" ] }, { "data": { "text/html": [ "
" ], "text/plain": [ "" ] }, "metadata": {}, "output_type": "display_data" }, { "data": { "text/plain": [] }, "execution_count": 67, "metadata": {}, "output_type": "execute_result" } ], "source": [ "import gradio as gr\n", "\n", "with gr.Blocks() as demo:\n", " gr.Markdown(\"Welcome to AI Search\")\n", " with gr.Column():\n", " inp = gr.Textbox(label=\"Search Query\")\n", " out = gr.Textbox(label=\"Result\")\n", " btn = gr.Button(\"Run\")\n", " btn.click(fn=AIGoogleSearch, inputs=inp, outputs=out)\n", "\n", "gr.close_all()\n", "\n", "demo.launch(share=True)" ] }, { "cell_type": "code", "execution_count": null, "id": "1ef468f8-1a5d-4cfe-a2cc-cdb694816599", "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "Python 3 (ipykernel)", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.11.4" } }, "nbformat": 4, "nbformat_minor": 5 }