import os import re import json import logging logger = logging.getLogger(__name__) def parse_restructured_text(text_block): """ Parse Gemini's restructured text into structured MCQs and Theory questions. Args: text_block: Raw text output from Gemini Returns: tuple: (mcqs_list, theories_list) """ mcqs = [] theories = [] # Split into chunks by double newlines chunks = [c.strip() for c in text_block.split("\n\n") if c.strip()] for chunk in chunks: # Try to parse as MCQ if "MCQ" in chunk or "Stem:" in chunk: mcq = parse_mcq_chunk(chunk) if mcq: mcqs.append(mcq) # Try to parse as Theory question elif "Theory" in chunk or ("Question:" in chunk and "Answer:" in chunk): theory = parse_theory_chunk(chunk) if theory: theories.append(theory) logger.info(f"Parsed {len(mcqs)} MCQs and {len(theories)} theory questions") return mcqs, theories def parse_mcq_chunk(chunk): """ Parse a single MCQ chunk into structured format. Expected format: MCQ [number] Stem: [question] Key: [correct answer] Distractors: - [distractor 1] - [distractor 2] - [distractor 3] Returns: dict or None """ try: # Extract stem stem_match = re.search(r"Stem:\s*(.+?)(?=\nKey:|\nDistractors:|$)", chunk, re.DOTALL) if not stem_match: logger.warning(f"No stem found in MCQ chunk: {chunk[:100]}") return None stem = stem_match.group(1).strip() # Extract key (correct answer) key_match = re.search(r"Key:\s*(.+?)(?=\nDistractors:|\n-|$)", chunk, re.DOTALL) if not key_match: logger.warning(f"No key found in MCQ chunk: {chunk[:100]}") return None key = key_match.group(1).strip() # Extract distractors distractor_pattern = r"-\s*(.+?)(?=\n-|\n\n|$)" distractors = re.findall(distractor_pattern, chunk, re.DOTALL) distractors = [d.strip() for d in distractors if d.strip()] if len(distractors) < 3: logger.warning(f"Only {len(distractors)} distractors found, need 3") # Pad with generic distractors if needed while len(distractors) < 3: distractors.append("None of the above") return { "stem": stem, "key": key, "distractors": distractors[:3] # Ensure exactly 3 } except Exception as e: logger.error(f"Error parsing MCQ chunk: {e}") return None def parse_theory_chunk(chunk): """ Parse a single Theory question chunk into structured format. Expected format: Theory [number] Question: [question text] Answer: [answer text] Returns: dict or None """ try: # Extract question question_match = re.search(r"Question:\s*(.+?)(?=\nAnswer:|$)", chunk, re.DOTALL) if not question_match: logger.warning(f"No question found in theory chunk: {chunk[:100]}") return None question = question_match.group(1).strip() # Extract answer answer_match = re.search(r"Answer:\s*(.+)$", chunk, re.DOTALL) if not answer_match: logger.warning(f"No answer found in theory chunk: {chunk[:100]}") return None answer = answer_match.group(1).strip() return { "question": question, "answer": answer } except Exception as e: logger.error(f"Error parsing theory chunk: {e}") return None def load_studykit_dataset(): """ Load the StudyKit dataset JSON file. Returns: list: List of study kit entries Raises: FileNotFoundError: If dataset file doesn't exist """ # Determine path relative to this file here = os.path.dirname(os.path.abspath(__file__)) # Try multiple possible paths possible_paths = [ os.path.join(here, "..", "..", "dataset", "studykit_questions_dataset.json"), os.path.join(here, "..", "dataset", "studykit_questions_dataset.json"), os.path.join(here, "dataset", "studykit_questions_dataset.json"), ] for dataset_path in possible_paths: dataset_path = os.path.normpath(dataset_path) if os.path.exists(dataset_path): try: with open(dataset_path, "r", encoding="utf-8") as f: data = json.load(f) logger.info(f"Loaded dataset from: {dataset_path}") return data except json.JSONDecodeError as e: logger.error(f"Invalid JSON in dataset: {e}") return [] raise FileNotFoundError( f"Cannot find studykit_questions_dataset.json in any of: {possible_paths}" ) def fetch_internet_content(topic): """ Placeholder for fetching content from the internet. You should implement this based on your requirements. Args: topic: Topic to search for Returns: str: Content text or empty string """ logger.warning(f"fetch_internet_content called for topic: {topic}") logger.warning("This function is not implemented. Returning empty content.") # TODO: Implement web scraping or API calls to fetch content return ""