flask_api / question_api /utils.py
vikkyblacq's picture
first commit
87112c5
Raw
History Blame Contribute Delete
5.5 kB
import os
import re
import json
import logging
logger = logging.getLogger(__name__)
def parse_restructured_text(text_block):
"""
Parse Gemini's restructured text into structured MCQs and Theory questions.
Args:
text_block: Raw text output from Gemini
Returns:
tuple: (mcqs_list, theories_list)
"""
mcqs = []
theories = []
# Split into chunks by double newlines
chunks = [c.strip() for c in text_block.split("\n\n") if c.strip()]
for chunk in chunks:
# Try to parse as MCQ
if "MCQ" in chunk or "Stem:" in chunk:
mcq = parse_mcq_chunk(chunk)
if mcq:
mcqs.append(mcq)
# Try to parse as Theory question
elif "Theory" in chunk or ("Question:" in chunk and "Answer:" in chunk):
theory = parse_theory_chunk(chunk)
if theory:
theories.append(theory)
logger.info(f"Parsed {len(mcqs)} MCQs and {len(theories)} theory questions")
return mcqs, theories
def parse_mcq_chunk(chunk):
"""
Parse a single MCQ chunk into structured format.
Expected format:
MCQ [number]
Stem: [question]
Key: [correct answer]
Distractors:
- [distractor 1]
- [distractor 2]
- [distractor 3]
Returns:
dict or None
"""
try:
# Extract stem
stem_match = re.search(r"Stem:\s*(.+?)(?=\nKey:|\nDistractors:|$)", chunk, re.DOTALL)
if not stem_match:
logger.warning(f"No stem found in MCQ chunk: {chunk[:100]}")
return None
stem = stem_match.group(1).strip()
# Extract key (correct answer)
key_match = re.search(r"Key:\s*(.+?)(?=\nDistractors:|\n-|$)", chunk, re.DOTALL)
if not key_match:
logger.warning(f"No key found in MCQ chunk: {chunk[:100]}")
return None
key = key_match.group(1).strip()
# Extract distractors
distractor_pattern = r"-\s*(.+?)(?=\n-|\n\n|$)"
distractors = re.findall(distractor_pattern, chunk, re.DOTALL)
distractors = [d.strip() for d in distractors if d.strip()]
if len(distractors) < 3:
logger.warning(f"Only {len(distractors)} distractors found, need 3")
# Pad with generic distractors if needed
while len(distractors) < 3:
distractors.append("None of the above")
return {
"stem": stem,
"key": key,
"distractors": distractors[:3] # Ensure exactly 3
}
except Exception as e:
logger.error(f"Error parsing MCQ chunk: {e}")
return None
def parse_theory_chunk(chunk):
"""
Parse a single Theory question chunk into structured format.
Expected format:
Theory [number]
Question: [question text]
Answer: [answer text]
Returns:
dict or None
"""
try:
# Extract question
question_match = re.search(r"Question:\s*(.+?)(?=\nAnswer:|$)", chunk, re.DOTALL)
if not question_match:
logger.warning(f"No question found in theory chunk: {chunk[:100]}")
return None
question = question_match.group(1).strip()
# Extract answer
answer_match = re.search(r"Answer:\s*(.+)$", chunk, re.DOTALL)
if not answer_match:
logger.warning(f"No answer found in theory chunk: {chunk[:100]}")
return None
answer = answer_match.group(1).strip()
return {
"question": question,
"answer": answer
}
except Exception as e:
logger.error(f"Error parsing theory chunk: {e}")
return None
def load_studykit_dataset():
"""
Load the StudyKit dataset JSON file.
Returns:
list: List of study kit entries
Raises:
FileNotFoundError: If dataset file doesn't exist
"""
# Determine path relative to this file
here = os.path.dirname(os.path.abspath(__file__))
# Try multiple possible paths
possible_paths = [
os.path.join(here, "..", "..", "dataset", "studykit_questions_dataset.json"),
os.path.join(here, "..", "dataset", "studykit_questions_dataset.json"),
os.path.join(here, "dataset", "studykit_questions_dataset.json"),
]
for dataset_path in possible_paths:
dataset_path = os.path.normpath(dataset_path)
if os.path.exists(dataset_path):
try:
with open(dataset_path, "r", encoding="utf-8") as f:
data = json.load(f)
logger.info(f"Loaded dataset from: {dataset_path}")
return data
except json.JSONDecodeError as e:
logger.error(f"Invalid JSON in dataset: {e}")
return []
raise FileNotFoundError(
f"Cannot find studykit_questions_dataset.json in any of: {possible_paths}"
)
def fetch_internet_content(topic):
"""
Placeholder for fetching content from the internet.
You should implement this based on your requirements.
Args:
topic: Topic to search for
Returns:
str: Content text or empty string
"""
logger.warning(f"fetch_internet_content called for topic: {topic}")
logger.warning("This function is not implemented. Returning empty content.")
# TODO: Implement web scraping or API calls to fetch content
return ""