Spaces:
Sleeping
Sleeping
| import os | |
| import re | |
| import json | |
| import logging | |
| logger = logging.getLogger(__name__) | |
| def parse_restructured_text(text_block): | |
| """ | |
| Parse Gemini's restructured text into structured MCQs and Theory questions. | |
| Args: | |
| text_block: Raw text output from Gemini | |
| Returns: | |
| tuple: (mcqs_list, theories_list) | |
| """ | |
| mcqs = [] | |
| theories = [] | |
| # Split into chunks by double newlines | |
| chunks = [c.strip() for c in text_block.split("\n\n") if c.strip()] | |
| for chunk in chunks: | |
| # Try to parse as MCQ | |
| if "MCQ" in chunk or "Stem:" in chunk: | |
| mcq = parse_mcq_chunk(chunk) | |
| if mcq: | |
| mcqs.append(mcq) | |
| # Try to parse as Theory question | |
| elif "Theory" in chunk or ("Question:" in chunk and "Answer:" in chunk): | |
| theory = parse_theory_chunk(chunk) | |
| if theory: | |
| theories.append(theory) | |
| logger.info(f"Parsed {len(mcqs)} MCQs and {len(theories)} theory questions") | |
| return mcqs, theories | |
| def parse_mcq_chunk(chunk): | |
| """ | |
| Parse a single MCQ chunk into structured format. | |
| Expected format: | |
| MCQ [number] | |
| Stem: [question] | |
| Key: [correct answer] | |
| Distractors: | |
| - [distractor 1] | |
| - [distractor 2] | |
| - [distractor 3] | |
| Returns: | |
| dict or None | |
| """ | |
| try: | |
| # Extract stem | |
| stem_match = re.search(r"Stem:\s*(.+?)(?=\nKey:|\nDistractors:|$)", chunk, re.DOTALL) | |
| if not stem_match: | |
| logger.warning(f"No stem found in MCQ chunk: {chunk[:100]}") | |
| return None | |
| stem = stem_match.group(1).strip() | |
| # Extract key (correct answer) | |
| key_match = re.search(r"Key:\s*(.+?)(?=\nDistractors:|\n-|$)", chunk, re.DOTALL) | |
| if not key_match: | |
| logger.warning(f"No key found in MCQ chunk: {chunk[:100]}") | |
| return None | |
| key = key_match.group(1).strip() | |
| # Extract distractors | |
| distractor_pattern = r"-\s*(.+?)(?=\n-|\n\n|$)" | |
| distractors = re.findall(distractor_pattern, chunk, re.DOTALL) | |
| distractors = [d.strip() for d in distractors if d.strip()] | |
| if len(distractors) < 3: | |
| logger.warning(f"Only {len(distractors)} distractors found, need 3") | |
| # Pad with generic distractors if needed | |
| while len(distractors) < 3: | |
| distractors.append("None of the above") | |
| return { | |
| "stem": stem, | |
| "key": key, | |
| "distractors": distractors[:3] # Ensure exactly 3 | |
| } | |
| except Exception as e: | |
| logger.error(f"Error parsing MCQ chunk: {e}") | |
| return None | |
| def parse_theory_chunk(chunk): | |
| """ | |
| Parse a single Theory question chunk into structured format. | |
| Expected format: | |
| Theory [number] | |
| Question: [question text] | |
| Answer: [answer text] | |
| Returns: | |
| dict or None | |
| """ | |
| try: | |
| # Extract question | |
| question_match = re.search(r"Question:\s*(.+?)(?=\nAnswer:|$)", chunk, re.DOTALL) | |
| if not question_match: | |
| logger.warning(f"No question found in theory chunk: {chunk[:100]}") | |
| return None | |
| question = question_match.group(1).strip() | |
| # Extract answer | |
| answer_match = re.search(r"Answer:\s*(.+)$", chunk, re.DOTALL) | |
| if not answer_match: | |
| logger.warning(f"No answer found in theory chunk: {chunk[:100]}") | |
| return None | |
| answer = answer_match.group(1).strip() | |
| return { | |
| "question": question, | |
| "answer": answer | |
| } | |
| except Exception as e: | |
| logger.error(f"Error parsing theory chunk: {e}") | |
| return None | |
| def load_studykit_dataset(): | |
| """ | |
| Load the StudyKit dataset JSON file. | |
| Returns: | |
| list: List of study kit entries | |
| Raises: | |
| FileNotFoundError: If dataset file doesn't exist | |
| """ | |
| # Determine path relative to this file | |
| here = os.path.dirname(os.path.abspath(__file__)) | |
| # Try multiple possible paths | |
| possible_paths = [ | |
| os.path.join(here, "..", "..", "dataset", "studykit_questions_dataset.json"), | |
| os.path.join(here, "..", "dataset", "studykit_questions_dataset.json"), | |
| os.path.join(here, "dataset", "studykit_questions_dataset.json"), | |
| ] | |
| for dataset_path in possible_paths: | |
| dataset_path = os.path.normpath(dataset_path) | |
| if os.path.exists(dataset_path): | |
| try: | |
| with open(dataset_path, "r", encoding="utf-8") as f: | |
| data = json.load(f) | |
| logger.info(f"Loaded dataset from: {dataset_path}") | |
| return data | |
| except json.JSONDecodeError as e: | |
| logger.error(f"Invalid JSON in dataset: {e}") | |
| return [] | |
| raise FileNotFoundError( | |
| f"Cannot find studykit_questions_dataset.json in any of: {possible_paths}" | |
| ) | |
| def fetch_internet_content(topic): | |
| """ | |
| Placeholder for fetching content from the internet. | |
| You should implement this based on your requirements. | |
| Args: | |
| topic: Topic to search for | |
| Returns: | |
| str: Content text or empty string | |
| """ | |
| logger.warning(f"fetch_internet_content called for topic: {topic}") | |
| logger.warning("This function is not implemented. Returning empty content.") | |
| # TODO: Implement web scraping or API calls to fetch content | |
| return "" |