| """PYQ (Previous Year Question) extraction service. |
| |
| Extracts structured question data from uploaded exam papers. |
| Uses deterministic regex patterns first, then optional AI enhancement. |
| Never hallucinates metadata — missing fields are marked as such. |
| """ |
| from __future__ import annotations |
|
|
| import re |
| from dataclasses import dataclass, field |
| from typing import Any |
|
|
|
|
| @dataclass |
| class ExtractedPYQQuestion: |
| question_number: str = "" |
| question_text: str = "" |
| marks: int | None = None |
| year: int | None = None |
| subject: str = "" |
| chapter: str = "" |
| topic: str = "" |
| answer_type: str = "" |
| formula_needed: bool = False |
| diagram_needed: bool = False |
| extracted_answer_if_available: str = "" |
| confidence: float = 0.0 |
| missing_metadata: list[str] = field(default_factory=list) |
|
|
|
|
| @dataclass |
| class PYQExtractionResult: |
| questions: list[ExtractedPYQQuestion] = field(default_factory=list) |
| extraction_confidence: float = 0.0 |
| missing_metadata: list[str] = field(default_factory=list) |
| warnings: list[str] = field(default_factory=list) |
|
|
|
|
| _YEAR_PATTERN = re.compile(r"\b(?:19|20)\d{2}\b") |
| _QUESTION_NUM_PATTERN = re.compile( |
| r"^(?:Q\.?\s*)?(\d+[.)]\s*)", |
| re.MULTILINE, |
| ) |
| _MARKS_PATTERN = re.compile( |
| r"(\d+)\s*(?:marks?|m(?:ks?)?)", |
| re.IGNORECASE, |
| ) |
| _SECTION_PATTERN = re.compile( |
| r"\bsection\s*[a-c]\b", |
| re.IGNORECASE, |
| ) |
|
|
| _ANSWER_TYPE_KEYWORDS: dict[str, list[str]] = { |
| "proof": ["prove", "show that", "demonstrate", "verify"], |
| "derivation": ["derive", "derivation", "obtain an expression"], |
| "numerical": ["calculate", "compute", "find the value", "numerical", "solve"], |
| "short_answer": ["define", "state", "list", "name", "mention", "what is"], |
| "essay": ["explain in detail", "discuss", "describe in detail", "elaborate"], |
| "diagram": ["draw", "sketch", "label", "diagram"], |
| "MCQ": ["a)", "b)", "c)", "d)", "choose the correct", "select"], |
| } |
|
|
| _FORMULA_HINTS = re.compile( |
| r"[=+\-*/^]|sin|cos|tan|log|ln|sqrt|int|d/dx|Σ|∫|Δ", |
| re.IGNORECASE, |
| ) |
|
|
| _DIAGRAM_HINTS = re.compile( |
| r"\b(diagram|figure|graph|plot|sketch|draw|circuit|ray\s*diagram|free\s*body)\b", |
| re.IGNORECASE, |
| ) |
|
|
|
|
| def extract_pyq_from_text( |
| text: str, |
| *, |
| source_metadata: dict[str, Any] | None = None, |
| year_hint: int | None = None, |
| subject_hint: str = "", |
| class_level_hint: str = "", |
| board_hint: str = "", |
| ) -> PYQExtractionResult: |
| """Extract PYQ questions from uploaded paper text. |
| |
| Uses deterministic regex patterns. Does NOT call AI — pure rules-based. |
| Missing metadata is reported, never guessed. |
| """ |
| source_metadata = source_metadata or {} |
| result = PYQExtractionResult() |
|
|
| if not text or not text.strip(): |
| result.warnings.append("Empty text provided — no questions extracted.") |
| return result |
|
|
| |
| detected_year = year_hint |
| if not detected_year: |
| year_matches = _YEAR_PATTERN.findall(text) |
| if year_matches: |
| candidate_years = [int(y) for y in year_matches if 2000 <= int(y) <= 2030] |
| if candidate_years: |
| detected_year = max(set(candidate_years), key=candidate_years.count) |
|
|
| |
| detected_subject = subject_hint or source_metadata.get("subject", "") |
|
|
| |
| |
| blocks = _split_into_question_blocks(text) |
|
|
| if not blocks: |
| result.warnings.append( |
| "Could not identify individual question blocks. " |
| "Ensure the paper has numbered questions (e.g., '1.', 'Q1.')." |
| ) |
| return result |
|
|
| for block in blocks: |
| q = _extract_single_question( |
| block, |
| year=detected_year, |
| subject=detected_subject, |
| class_level=class_level_hint, |
| board=board_hint, |
| ) |
| if q.question_text: |
| result.questions.append(q) |
|
|
| |
| if result.questions: |
| confidences = [q.confidence for q in result.questions] |
| result.extraction_confidence = sum(confidences) / len(confidences) if confidences else 0.0 |
| else: |
| result.extraction_confidence = 0.0 |
|
|
| |
| missing = [] |
| if not detected_year: |
| missing.append("year") |
| if not detected_subject: |
| missing.append("subject") |
| if not class_level_hint: |
| missing.append("class_level") |
| if not board_hint: |
| missing.append("board") |
| result.missing_metadata = missing |
|
|
| if missing: |
| result.warnings.append( |
| f"Missing metadata: {', '.join(missing)}. " |
| "Provide these to improve extraction accuracy." |
| ) |
|
|
| return result |
|
|
|
|
| def _split_into_question_blocks(text: str) -> list[str]: |
| """Split exam paper text into individual question blocks.""" |
| |
| text = text.replace("\r\n", "\n").replace(";", ";\n") |
| lines = text.split("\n") |
| blocks: list[str] = [] |
| current_block: list[str] = [] |
| in_question = False |
|
|
| for line in lines: |
| stripped = line.strip() |
| if not stripped: |
| if current_block: |
| blocks.append("\n".join(current_block)) |
| current_block = [] |
| in_question = False |
| continue |
|
|
| |
| is_new_question = False |
| |
| if re.match(r"^(?:Q\.?\s*)?\d+[.)]\s", stripped): |
| is_new_question = True |
| |
| elif re.match(r"^(?:question|q)\s*\d+", stripped, re.IGNORECASE): |
| is_new_question = True |
| elif re.match(r"^section\s*[a-c]", stripped, re.IGNORECASE): |
| is_new_question = True |
|
|
| if is_new_question: |
| if current_block: |
| blocks.append("\n".join(current_block)) |
| current_block = [stripped] |
| in_question = True |
| elif in_question: |
| current_block.append(stripped) |
| elif not blocks: |
| |
| inline_match = re.search(r"\b(?:Q\.?\s*)?(\d+)[.)]\s+(.+?)(?:\.\s|\.\s*$|$)", stripped) |
| if inline_match: |
| |
| preamble = stripped[:inline_match.start()].strip() |
| q_text = stripped[inline_match.start():].strip() |
| if preamble: |
| |
| pass |
| blocks.append(q_text) |
|
|
| if current_block: |
| blocks.append("\n".join(current_block)) |
|
|
| return blocks |
|
|
|
|
| def _extract_single_question( |
| block: str, |
| *, |
| year: int | None, |
| subject: str, |
| class_level: str, |
| board: str, |
| ) -> ExtractedPYQQuestion: |
| """Extract structured data from a single question block.""" |
| q = ExtractedPYQQuestion() |
| q.year = year |
| q.subject = subject |
|
|
| |
| num_match = re.match(r"^(?:Q\.?\s*)?(\d+)[.)]\s*", block) |
| if num_match: |
| q.question_number = num_match.group(1) |
| block = block[num_match.end():] |
|
|
| |
| marks_match = _MARKS_PATTERN.search(block) |
| if marks_match: |
| q.marks = int(marks_match.group(1)) |
|
|
| |
| q.question_text = re.sub(r"\(?\s*\d+\s*marks?\s*\)?", "", block, flags=re.IGNORECASE).strip() |
|
|
| |
| q.answer_type = _detect_answer_type(block) |
|
|
| |
| q.formula_needed = bool(_FORMULA_HINTS.search(block)) |
|
|
| |
| q.diagram_needed = bool(_DIAGRAM_HINTS.search(block)) |
|
|
| |
| score = 0.0 |
| if q.question_text: |
| score += 0.3 |
| if q.question_number: |
| score += 0.15 |
| if q.marks is not None: |
| score += 0.15 |
| if q.year is not None: |
| score += 0.15 |
| if q.subject: |
| score += 0.1 |
| if q.answer_type and q.answer_type != "unknown": |
| score += 0.1 |
| q.confidence = min(score, 1.0) |
|
|
| |
| if not q.year: |
| q.missing_metadata.append("year") |
| if not q.subject: |
| q.missing_metadata.append("subject") |
|
|
| return q |
|
|
|
|
| def _detect_answer_type(text: str) -> str: |
| """Detect the answer type from question text.""" |
| text_lower = text.lower() |
| scores: dict[str, int] = {} |
| for answer_type, keywords in _ANSWER_TYPE_KEYWORDS.items(): |
| hits = sum(1 for kw in keywords if kw in text_lower) |
| if hits: |
| scores[answer_type] = hits |
|
|
| if not scores: |
| |
| if re.search(r"\b\d+\s*marks?\b", text, re.IGNORECASE): |
| marks_match = re.search(r"(\d+)\s*marks?", text, re.IGNORECASE) |
| if marks_match: |
| marks = int(marks_match.group(1)) |
| if marks <= 2: |
| return "short_answer" |
| if marks <= 5: |
| return "essay" |
| return "unknown" |
|
|
| return max(scores, key=scores.get) |
|
|