narrativenet-api / app /api /endpoints.py
rohan21005's picture
Initial launch of Narrative intelligence platformm
647ef8d
Raw
History Blame Contribute Delete
6.58 kB
from fastapi import APIRouter, HTTPException, UploadFile, File
import pypdf
import io
from pydantic import BaseModel
from app.services.ml_pipeline import ml_pipeline
from app.services.db_service import db_service
import random
import requests
router = APIRouter()
class ArticleRequest(BaseModel):
text: str
title: str = "Uploaded Article"
@router.post("/analyze/article")
def analyze_article(request: ArticleRequest):
if not request.text:
raise HTTPException(status_code=400, detail="Text is required")
try:
# Run ML analysis
analysis = ml_pipeline.analyze_article(request.text)
article_data = {
"title": request.title,
"text_snippet": request.text[:100] + "...",
**analysis
}
# Save to db
db_service.save_analysis(article_data)
# PyMongo mutates the dict and adds an ObjectId under '_id'. Pop it before returning!
article_data.pop('_id', None)
return article_data
except Exception as e:
import traceback
traceback.print_exc()
raise HTTPException(status_code=500, detail=f"Analysis failed: {str(e)}")
@router.post("/analyze/pdf")
async def analyze_pdf(file: UploadFile = File(...)):
if not file.filename.lower().endswith(".pdf"):
raise HTTPException(status_code=400, detail="Only PDF files are supported")
try:
content = await file.read()
import fitz
import numpy as np
doc = fitz.open(stream=content, filetype="pdf")
extracted_text = ""
for page in doc:
extracted_text += page.get_text() + "\n"
if len(extracted_text.strip()) < 50:
import easyocr
from PIL import Image
img_texts = []
reader = easyocr.Reader(['en'], gpu=False)
for i in range(min(3, len(doc))):
page = doc[i]
pix = page.get_pixmap()
img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples)
result = reader.readtext(np.array(img))
img_texts.extend([res[1] for res in result])
extracted_text = " ".join(img_texts)
if len(extracted_text.strip()) < 50:
raise HTTPException(
status_code=400,
detail="Could not extract enough text from the PDF even with OCR. Please ensure the document is clear."
)
# Limit text for analysis
text_to_analyze = extracted_text[:5000] # Process first 5k chars for speed
analysis = ml_pipeline.analyze_article(text_to_analyze)
article_data = {
"title": file.filename,
"text_snippet": text_to_analyze[:100] + "...",
**analysis
}
db_service.save_analysis(article_data)
# Pop PyMongo ObjectId before JSON serialization
article_data.pop('_id', None)
return article_data
except Exception as e:
import traceback
traceback.print_exc()
if hasattr(e, 'status_code'):
raise e
raise HTTPException(status_code=500, detail=f"PDF extraction failed: {str(e)}")
@router.get("/graph/narratives")
def get_narratives_graph():
return db_service.get_graph_data()
import time
_trending_cache = {"time": 0, "data": None}
@router.get("/news/trending")
def get_trending_news():
global _trending_cache
if time.time() - _trending_cache["time"] < 600 and _trending_cache["data"]:
return _trending_cache["data"]
api_key = "6065e35156724b269f88dcaa2dab2777"
url = f"https://newsapi.org/v2/top-headlines?country=us&apiKey={api_key}"
try:
response = requests.get(url, timeout=5)
data = response.json()
articles = data.get("articles", [])[:3] # Process max 3 articles to avoid CPU hang
trending_topics = []
cluster_counts = {}
highest_fake_prob = 0
most_emotional_topic = "General News"
for art in articles:
text = str(art.get('title', '')) + " " + str(art.get('description', ''))
if len(text.strip()) < 10:
continue
analysis = ml_pipeline.analyze_article(text)
topic = analysis.get("topic", "General")
fake_prob = analysis.get("fake_prob", 0.1)
emotion = analysis.get("emotion", "neutral")
cluster = analysis.get("cluster", "General")
cluster_counts[cluster] = cluster_counts.get(cluster, 0) + 1
if fake_prob > highest_fake_prob:
highest_fake_prob = fake_prob
most_emotional_topic = topic
# Avoid massive duplicate topics
if topic not in [t['name'] for t in trending_topics]:
trending_topics.append({
"name": topic[:30] + ("..." if len(topic) > 30 else ""),
"fake_prob": fake_prob,
"emotion": emotion
})
# Sort clusters by frequency
growing_clusters = sorted(cluster_counts.keys(), key=lambda k: cluster_counts[k], reverse=True)[:3]
insights = f"Real-time scan resolved {len(trending_topics)} top entities. The cluster '{growing_clusters[0] if growing_clusters else 'General'}' is currently expanding heavily. Maximum deception probability detected in '{most_emotional_topic}'."
# fallback if API returned empty
if not trending_topics:
raise Exception("No topics extracted")
result = {
"trending_topics": trending_topics[:5],
"growing_clusters": growing_clusters,
"insights": insights
}
_trending_cache["time"] = time.time()
_trending_cache["data"] = result
return result
except Exception as e:
import traceback
traceback.print_exc()
print(f"News API Error details: {e}")
# Default fallback
return {
"trending_topics": [
{"name": "API Connection Disrupted", "fake_prob": 0.5, "emotion": "sadness"}
],
"growing_clusters": ["System"],
"insights": f"News stream offline or analyzing failed: {e}. Reverting to static analytics."
}