DeepSearch / source /aisearch_v1.py
sk1502's picture
Code update
0fb6b90
Raw
History Blame Contribute Delete
11 kB
import requests
# from groq import Groq
from bs4 import BeautifulSoup
import os
import re
from dotenv import load_dotenv
# from ollama import chat
from google import genai
from google.genai import types
import asyncio
from openai import AsyncAzureOpenAI
import urllib3
# import bs4
import fitz
from pydantic import BaseModel
import json
from langchain_community.document_loaders import WebBaseLoader
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
load_dotenv()
# GROQ_APIKEY = os.environ.get('GROQ_APIKEY')
GOOGS_APIKEY = os.environ.get('GOOGS_APIKEY')
SEARCH_ENGINE_ID = os.environ.get('SEARCH_ENGINE_ID')
GEMINI_APIKEY = os.environ.get('GEMINI_APIKEY')
az_client = AsyncAzureOpenAI(
api_key=os.environ.get('AZURE_OPENAI_API_KEY'),
api_version="2024-12-01-preview", # Use the appropriate API version
azure_endpoint=os.environ.get('AZURE_OPENAI_ENDPOINT')
)
az_client_in = AsyncAzureOpenAI(
api_key=os.environ.get('AZURE_OPENAI_API_KEY_IN'),
api_version="2024-12-01-preview", # Use the appropriate API version
azure_endpoint=os.environ.get('AZURE_OPENAI_ENDPOINT_IN')
)
class ReportResponse(BaseModel):
title : str
justification : str
summary : str
link : str
# class ListReport(BaseModel)
# client = genai.Client(api_key=GEMINI_APIKEY)
gemini_client = genai.Client(api_key=GEMINI_APIKEY)
class aiSearch:
def __init__(self):
# self.query = query
# self.client = Groq(
# api_key=GROQ_APIKEY,
# )
self.research_prompt = """
<role>
You are AI agent, who is specialized in research and analysis of Indian Judgments.
</role>
<input format>
[
{
"title" : title of the url,
"link" : url link,
"snippet" : sinnept from link
"text" : text from url webpage
},
{
"title" : title of the url,
"link" : url link,
"snippet" : sinnept from link
"text" : text from url webpage
}
]
</input format>
<instructions>
1. Analyze each link and it's text to find relevant judgement to the user query .
2. Once analyzed extract judgement names and text of the judgement
3. In the response provide judgement name, one-two line justifications on why judgement is relevant to the query and text of the judgement in the webpage
4. If there is no relevant judgement return empty output
5. Think step by step before the response and only use the context for response.
6. Only shortlist Indian court based judgements.
7. Return the url link of the webpage used as the source in the output.
</instructions>
<output format>
[
{
"title" : Judgement title,
"justification" : Justification on why judgement is relevant,
"summary" : Summary of the Judgement,
"link" : url link provided in input
},
{
"title" : Judgement title,
"justification" : Justification on why judgement is relevant,
"summary" : Summary of the Judgement,
"link" : url link provided in input
}
]
</output format>
"""
self.sp_final_report = """
<role>
You are AI agent, who is specialized in making formatted report.
</role>
<input format>
[{
"title" : Judgement title,
"justification" : Justification on why judgement is relevant,
"summary" : Summary of the Judgement,
"link" : url link of webpage
}]
</input format>
<instructions>
1. Convert the input provided into a report.
2. Do not include the report if it says no relevant judgement found and justifications says document not relevant
3. Return the clickable url links as well in the output from the input
4. Do not include any extra text on warnings about URL
5. Only shortlist Indian court based judgements
</instructions>
<output format>
1. **Judgement name**
Justification: reaosn to shortlist the judgement
Summary: Summary of the judgement
Source: (URL_link)
2. **Judgement name**
Justification: reaosn to shortlist the judgement
Summary: Summary of the judgement
Source: (URL_link)
</output format>"""
async def gpt4omini(self,query):
chat_completion = await az_client_in.chat.completions.create(
messages=[
{
"role" : "system",
"content" : self.sp_final_report
},
{
"role": "user",
"content": query,
}
],
stream=True,
model="gpt-4o-mini",
temperature=0,
# reasoning_effort='low',
# response_format=ReportResponse
)
return chat_completion
async def o3_mini(self,query):
chat_completion = await az_client.beta.chat.completions.parse(
messages=[
{
"role" : "system",
"content" : self.research_prompt
},
{
"role": "user",
"content": query,
}
],
model="o3-mini",
# temperature=0,
# reasoning_effort='low',
response_format=ReportResponse
)
return chat_completion.choices[0].message.content
async def gemini(self,prompt):
# try:
response = await gemini_client.aio.models.generate_content(
model='gemini-1.5-flash',
contents = prompt,
config=types.GenerateContentConfig(
system_instruction=self.research_prompt,
max_output_tokens= 4096,
temperature= 0.1
)
)
return response.text
# except Exception as e:
# print(e)
# return ''
def google_custom_search(self,query):
"""
Make a request to Google Custom Search API
Args:
query (str): Search query
api_key (str): Your Google API key
cx (str): Your Programmable Search Engine ID
Returns:
dict: JSON response from the API
"""
base_url = "https://www.googleapis.com/customsearch/v1"
en_query = f"{query}, Indian Judgements"
params = {
'q': en_query,
'key': GOOGS_APIKEY,
'cx': SEARCH_ENGINE_ID,
'cr':'countryIN',
'gl': 'in',
}
try:
response = requests.get(base_url, params=params,verify=False)
response.raise_for_status()
return response.json()
except requests.exceptions.RequestException as e:
print(f"An error occurred: {e}")
return None
def extract_text_from_html(self,url):
try:
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
'Accept-Language': 'en-US,en;q=0.5'
}
response = requests.get(url, headers=headers, verify=False)
soup = BeautifulSoup(response.text, 'html.parser')
for element in soup(['script', 'style']):
element.decompose()
text_chunks = []
content_tags = ['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'li', 'div']
for tag in soup.find_all(content_tags):
text = tag.get_text(strip=True)
if len(text) < 3:
continue
if tag.name in ['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'div']:
text = f"\n{text}\n"
elif tag.name == 'li':
text = f"• {text}\n"
text = re.sub(r'\s+', ' ', text).strip()
if text:
text_chunks.append(text)
text = ' '.join(text_chunks)
text = re.sub(r'\s+', ' ', text)
text = re.sub(r'\n\s*\n', '\n\n', text)
except Exception as e:
print(e)
text = ''
return text.strip()
async def langchain_extracter(self,url):
try:
loader = WebBaseLoader(web_paths=[url],verify_ssl=False,continue_on_failure = True)
docs = []
async for doc in loader.alazy_load():
docs.append(doc)
return docs[0].page_content
except Exception as ex:
print(f"Error occured while requesting url : {url} exception : {ex}")
def pdfreader(self,url):
ext_txt = ''
try:
response = requests.get(url)
# Open the PDF with PyMuPDF
with fitz.open(stream=response.content, filetype="pdf") as doc:
ext_txt = ''
for page_num, page in enumerate(doc, start=1):
text = page.get_text()
ext_txt+=text
# print(f"--- Page {page_num} ---\n{text}\n")
except:
print(url)
return ext_txt
async def run_search(self,query):
self.results = self.google_custom_search(query)
self.url_items = []
for item in self.results['items']:
if len(self.url_items) < 12:
temp = {
"title" : item["title"],
"link" : item["link"],
"snippet" : item["snippet"]
}
self.url_items.append(temp)
for item in self.url_items:
if item['link'].endswith(".pdf"):
item['text'] = self.pdfreader(item['link'])
else:
item['text'] = await self.langchain_extracter(item['link'])
batch_size = 3
batches = [self.url_items[i:i+batch_size] for i in range(0, len(self.url_items), batch_size)]
tasks = []
for batch in batches:
context_prompt = f"""
<context>
{batch}
</context>
<user query>
{query}
</user query>
"""
tasks.append(self.gemini(context_prompt))
# response = await self.gemini(context_promt)
batch_responses = await asyncio.gather(*tasks)
reports = [resp for resp in batch_responses]
print(reports)
stream_responses = await self.gpt4omini(str(reports))
return stream_responses