| """PYQ Pattern Analysis Service. |
| |
| Analyses extracted PYQ questions for patterns: |
| - Repeated topics across years |
| - Marks distribution |
| - Answer type distribution |
| - Chapter-wise and topic-wise question counts |
| - Years seen |
| |
| Rules: |
| - Every pattern must be backed by actual question IDs. |
| - If fewer than 2 years exist, say "not enough data for pattern analysis." |
| - Do not use words like "sure shot" or fake high-yield. |
| - Use "seen in uploaded papers" instead. |
| """ |
| from __future__ import annotations |
|
|
| from collections import Counter |
| from dataclasses import dataclass, field |
| from typing import Any |
|
|
| from sqlalchemy import select |
| from sqlalchemy.orm import Session |
|
|
| from app.models.previous_question import PreviousQuestion |
| from app.models.previous_paper import PreviousPaper |
|
|
|
|
| @dataclass |
| class PatternEvidence: |
| question_id: str |
| question_text: str |
| year: int | None |
| marks: int | None |
| chapter: str |
| topic: str |
| paper_id: str |
| paper_title: str |
| source_url: str | None |
|
|
|
|
| @dataclass |
| class TopicPattern: |
| topic: str |
| chapter: str |
| count: int |
| years_seen: list[int] |
| question_ids: list[str] |
| evidence: list[PatternEvidence] |
|
|
|
|
| @dataclass |
| class MarksDistribution: |
| marks: int |
| count: int |
| percentage: float |
|
|
|
|
| @dataclass |
| class AnswerTypeDistribution: |
| answer_type: str |
| count: int |
| percentage: float |
|
|
|
|
| @dataclass |
| class ChapterWiseCount: |
| chapter: str |
| total_questions: int |
| total_marks: int |
| topics: list[str] |
|
|
|
|
| @dataclass |
| class PatternAnalysisResult: |
| repeated_topics: list[TopicPattern] = field(default_factory=list) |
| marks_distribution: list[MarksDistribution] = field(default_factory=list) |
| answer_type_distribution: list[AnswerTypeDistribution] = field(default_factory=list) |
| chapter_wise_count: list[ChapterWiseCount] = field(default_factory=list) |
| topic_wise_count: list[TopicPattern] = field(default_factory=list) |
| years_seen: list[int] = field(default_factory=list) |
| total_questions: int = 0 |
| total_papers: int = 0 |
| evidence_list: list[PatternEvidence] = field(default_factory=list) |
| analysis_note: str = "" |
| confidence: float = 0.0 |
|
|
|
|
| _MIN_YEARS_FOR_PATTERN = 2 |
|
|
|
|
| def analyze_pyq_patterns( |
| db: Session, |
| user_id: str, |
| *, |
| subject: str = "", |
| board: str = "", |
| class_level: str = "", |
| chapter: str = "", |
| topic: str = "", |
| year_range: tuple[int | None, int | None] | None = None, |
| paper_ids: list[str] | None = None, |
| ) -> PatternAnalysisResult: |
| """Analyze PYQ patterns from uploaded questions in the database. |
| |
| All patterns are backed by actual question IDs from the database. |
| """ |
| result = PatternAnalysisResult() |
|
|
| paper_query = ( |
| select(PreviousPaper) |
| .where(PreviousPaper.user_id == user_id) |
| .where(PreviousPaper.verification_status == "verified") |
| ) |
| if paper_ids: |
| paper_query = paper_query.where(PreviousPaper.id.in_(paper_ids)) |
| if subject: |
| paper_query = paper_query.where(PreviousPaper.subject == subject) |
| if board: |
| paper_query = paper_query.where(PreviousPaper.board == board) |
| if class_level: |
| paper_query = paper_query.where(PreviousPaper.class_level == class_level) |
|
|
| papers = list(db.scalars(paper_query).all()) |
|
|
| if year_range: |
| min_year, max_year = year_range |
| papers = [ |
| paper for paper in papers |
| if (not min_year or not paper.year or paper.year >= min_year) |
| and (not max_year or not paper.year or paper.year <= max_year) |
| ] |
|
|
| result.total_papers = len(papers) |
|
|
| questions: list[PreviousQuestion] = [] |
| for paper in papers: |
| paper_questions = list(paper.questions) |
| for q in paper_questions: |
| |
| if chapter and q.chapter and chapter.lower() not in q.chapter.lower(): |
| continue |
| if topic and q.topic and topic.lower() not in q.topic.lower(): |
| continue |
| if year_range: |
| min_year, max_year = year_range |
| if min_year and q.year and q.year < min_year: |
| continue |
| if max_year and q.year and q.year > max_year: |
| continue |
| questions.append(q) |
|
|
| result.total_questions = len(questions) |
|
|
| if not questions: |
| result.analysis_note = ( |
| "No uploaded PYQ questions found for the given filters. " |
| "Upload previous year question papers to unlock pattern analysis." |
| ) |
| return result |
|
|
| |
| evidence_map: dict[str, PatternEvidence] = {} |
| for q in questions: |
| evidence = PatternEvidence( |
| question_id=q.id, |
| question_text=q.question_text[:200], |
| year=q.year, |
| marks=q.marks, |
| chapter=q.chapter or "", |
| topic=q.topic or "", |
| paper_id=q.previous_paper.id, |
| paper_title=q.previous_paper.title, |
| source_url=q.previous_paper.source_url, |
| ) |
| evidence_map[q.id] = evidence |
| result.evidence_list.append(evidence) |
|
|
| |
| all_years = sorted({q.year for q in questions if q.year is not None}) |
| result.years_seen = all_years |
|
|
| |
| if len(all_years) < _MIN_YEARS_FOR_PATTERN: |
| result.analysis_note = ( |
| f"Only {len(all_years)} year(s) of data available. " |
| "Upload papers from at least 2 different years for meaningful pattern analysis." |
| ) |
| result.confidence = 0.3 |
| |
| _compute_basic_stats(questions, evidence_map, result) |
| return result |
|
|
| |
| result.confidence = min(0.5 + len(all_years) * 0.05, 1.0) |
|
|
| |
| topic_year_map: dict[str, dict[int, list[str]]] = {} |
| for q in questions: |
| |
| if q.year is None: |
| continue |
| topic_key = q.topic or q.chapter or "Unknown" |
| if topic_key not in topic_year_map: |
| topic_year_map[topic_key] = {} |
| year = q.year |
| if year not in topic_year_map[topic_key]: |
| topic_year_map[topic_key][year] = [] |
| topic_year_map[topic_key][year].append(q.id) |
|
|
| for topic_key, year_map in topic_year_map.items(): |
| if len(year_map) >= 2: |
| years = sorted([y for y in year_map if y > 0]) |
| q_ids = [] |
| evidence_items = [] |
| for y, ids in year_map.items(): |
| q_ids.extend(ids) |
| evidence_items.extend([evidence_map[qid] for qid in ids if qid in evidence_map]) |
|
|
| |
| chapter_val = "" |
| for q in questions: |
| if (q.topic or q.chapter or "") == topic_key: |
| chapter_val = q.chapter or "" |
| break |
|
|
| result.repeated_topics.append(TopicPattern( |
| topic=topic_key, |
| chapter=chapter_val, |
| count=len(q_ids), |
| years_seen=years, |
| question_ids=q_ids, |
| evidence=evidence_items[:5], |
| )) |
|
|
| |
| result.repeated_topics.sort(key=lambda t: t.count, reverse=True) |
|
|
| _compute_basic_stats(questions, evidence_map, result) |
|
|
| if not result.analysis_note: |
| result.analysis_note = ( |
| f"Pattern analysis based on {result.total_questions} questions " |
| f"across {len(all_years)} years ({', '.join(str(y) for y in all_years)}). " |
| "All patterns are backed by uploaded question paper evidence." |
| ) |
|
|
| return result |
|
|
|
|
| def _compute_basic_stats( |
| questions: list[PreviousQuestion], |
| evidence_map: dict[str, PatternEvidence], |
| result: PatternAnalysisResult, |
| ) -> None: |
| """Compute marks distribution, answer type distribution, and chapter-wise counts.""" |
| |
| marks_counter: Counter[int] = Counter() |
| for q in questions: |
| if q.marks is not None: |
| marks_counter[q.marks] += 1 |
|
|
| total = len(questions) or 1 |
| result.marks_distribution = [ |
| MarksDistribution( |
| marks=marks, |
| count=count, |
| percentage=round(count / total * 100, 1), |
| ) |
| for marks, count in sorted(marks_counter.items()) |
| ] |
|
|
| |
| type_counter: Counter[str] = Counter() |
| for q in questions: |
| |
| at = q.answer_type or q.question_type or "unknown" |
| type_counter[at] += 1 |
|
|
| result.answer_type_distribution = [ |
| AnswerTypeDistribution( |
| answer_type=at, |
| count=count, |
| percentage=round(count / total * 100, 1), |
| ) |
| for at, count in type_counter.most_common() |
| ] |
|
|
| |
| chapter_map: dict[str, dict[str, Any]] = {} |
| for q in questions: |
| ch = q.chapter or "Unknown" |
| if ch not in chapter_map: |
| chapter_map[ch] = {"count": 0, "marks": 0, "topics": set()} |
| chapter_map[ch]["count"] += 1 |
| chapter_map[ch]["marks"] += q.marks or 0 |
| if q.topic: |
| chapter_map[ch]["topics"].add(q.topic) |
|
|
| result.chapter_wise_count = [ |
| ChapterWiseCount( |
| chapter=ch, |
| total_questions=data["count"], |
| total_marks=data["marks"], |
| topics=sorted(data["topics"]), |
| ) |
| for ch, data in sorted(chapter_map.items(), key=lambda x: x[1]["count"], reverse=True) |
| ] |
|
|
| |
| topic_map: dict[str, dict[str, Any]] = {} |
| for q in questions: |
| t = q.topic or q.chapter or "Unknown" |
| if t not in topic_map: |
| topic_map[t] = {"count": 0, "years": set(), "ids": [], "chapter": q.chapter or ""} |
| topic_map[t]["count"] += 1 |
| if q.year: |
| topic_map[t]["years"].add(q.year) |
| topic_map[t]["ids"].append(q.id) |
|
|
| result.topic_wise_count = [ |
| TopicPattern( |
| topic=t, |
| chapter=data["chapter"], |
| count=data["count"], |
| years_seen=sorted(data["years"]), |
| question_ids=data["ids"], |
| evidence=[evidence_map[qid] for qid in data["ids"][:5] if qid in evidence_map], |
| ) |
| for t, data in sorted(topic_map.items(), key=lambda x: x[1]["count"], reverse=True) |
| ] |
|
|