Spaces:
Sleeping
Sleeping
File size: 5,498 Bytes
87112c5 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 | import os
import re
import json
import logging
logger = logging.getLogger(__name__)
def parse_restructured_text(text_block):
"""
Parse Gemini's restructured text into structured MCQs and Theory questions.
Args:
text_block: Raw text output from Gemini
Returns:
tuple: (mcqs_list, theories_list)
"""
mcqs = []
theories = []
# Split into chunks by double newlines
chunks = [c.strip() for c in text_block.split("\n\n") if c.strip()]
for chunk in chunks:
# Try to parse as MCQ
if "MCQ" in chunk or "Stem:" in chunk:
mcq = parse_mcq_chunk(chunk)
if mcq:
mcqs.append(mcq)
# Try to parse as Theory question
elif "Theory" in chunk or ("Question:" in chunk and "Answer:" in chunk):
theory = parse_theory_chunk(chunk)
if theory:
theories.append(theory)
logger.info(f"Parsed {len(mcqs)} MCQs and {len(theories)} theory questions")
return mcqs, theories
def parse_mcq_chunk(chunk):
"""
Parse a single MCQ chunk into structured format.
Expected format:
MCQ [number]
Stem: [question]
Key: [correct answer]
Distractors:
- [distractor 1]
- [distractor 2]
- [distractor 3]
Returns:
dict or None
"""
try:
# Extract stem
stem_match = re.search(r"Stem:\s*(.+?)(?=\nKey:|\nDistractors:|$)", chunk, re.DOTALL)
if not stem_match:
logger.warning(f"No stem found in MCQ chunk: {chunk[:100]}")
return None
stem = stem_match.group(1).strip()
# Extract key (correct answer)
key_match = re.search(r"Key:\s*(.+?)(?=\nDistractors:|\n-|$)", chunk, re.DOTALL)
if not key_match:
logger.warning(f"No key found in MCQ chunk: {chunk[:100]}")
return None
key = key_match.group(1).strip()
# Extract distractors
distractor_pattern = r"-\s*(.+?)(?=\n-|\n\n|$)"
distractors = re.findall(distractor_pattern, chunk, re.DOTALL)
distractors = [d.strip() for d in distractors if d.strip()]
if len(distractors) < 3:
logger.warning(f"Only {len(distractors)} distractors found, need 3")
# Pad with generic distractors if needed
while len(distractors) < 3:
distractors.append("None of the above")
return {
"stem": stem,
"key": key,
"distractors": distractors[:3] # Ensure exactly 3
}
except Exception as e:
logger.error(f"Error parsing MCQ chunk: {e}")
return None
def parse_theory_chunk(chunk):
"""
Parse a single Theory question chunk into structured format.
Expected format:
Theory [number]
Question: [question text]
Answer: [answer text]
Returns:
dict or None
"""
try:
# Extract question
question_match = re.search(r"Question:\s*(.+?)(?=\nAnswer:|$)", chunk, re.DOTALL)
if not question_match:
logger.warning(f"No question found in theory chunk: {chunk[:100]}")
return None
question = question_match.group(1).strip()
# Extract answer
answer_match = re.search(r"Answer:\s*(.+)$", chunk, re.DOTALL)
if not answer_match:
logger.warning(f"No answer found in theory chunk: {chunk[:100]}")
return None
answer = answer_match.group(1).strip()
return {
"question": question,
"answer": answer
}
except Exception as e:
logger.error(f"Error parsing theory chunk: {e}")
return None
def load_studykit_dataset():
"""
Load the StudyKit dataset JSON file.
Returns:
list: List of study kit entries
Raises:
FileNotFoundError: If dataset file doesn't exist
"""
# Determine path relative to this file
here = os.path.dirname(os.path.abspath(__file__))
# Try multiple possible paths
possible_paths = [
os.path.join(here, "..", "..", "dataset", "studykit_questions_dataset.json"),
os.path.join(here, "..", "dataset", "studykit_questions_dataset.json"),
os.path.join(here, "dataset", "studykit_questions_dataset.json"),
]
for dataset_path in possible_paths:
dataset_path = os.path.normpath(dataset_path)
if os.path.exists(dataset_path):
try:
with open(dataset_path, "r", encoding="utf-8") as f:
data = json.load(f)
logger.info(f"Loaded dataset from: {dataset_path}")
return data
except json.JSONDecodeError as e:
logger.error(f"Invalid JSON in dataset: {e}")
return []
raise FileNotFoundError(
f"Cannot find studykit_questions_dataset.json in any of: {possible_paths}"
)
def fetch_internet_content(topic):
"""
Placeholder for fetching content from the internet.
You should implement this based on your requirements.
Args:
topic: Topic to search for
Returns:
str: Content text or empty string
"""
logger.warning(f"fetch_internet_content called for topic: {topic}")
logger.warning("This function is not implemented. Returning empty content.")
# TODO: Implement web scraping or API calls to fetch content
return "" |