File size: 5,498 Bytes
87112c5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
import os
import re
import json
import logging

logger = logging.getLogger(__name__)


def parse_restructured_text(text_block):
    """
    Parse Gemini's restructured text into structured MCQs and Theory questions.
    
    Args:
        text_block: Raw text output from Gemini
    
    Returns:
        tuple: (mcqs_list, theories_list)
    """
    mcqs = []
    theories = []

    # Split into chunks by double newlines
    chunks = [c.strip() for c in text_block.split("\n\n") if c.strip()]

    for chunk in chunks:
        # Try to parse as MCQ
        if "MCQ" in chunk or "Stem:" in chunk:
            mcq = parse_mcq_chunk(chunk)
            if mcq:
                mcqs.append(mcq)
        
        # Try to parse as Theory question
        elif "Theory" in chunk or ("Question:" in chunk and "Answer:" in chunk):
            theory = parse_theory_chunk(chunk)
            if theory:
                theories.append(theory)

    logger.info(f"Parsed {len(mcqs)} MCQs and {len(theories)} theory questions")
    return mcqs, theories


def parse_mcq_chunk(chunk):
    """
    Parse a single MCQ chunk into structured format.
    
    Expected format:
        MCQ [number]
        Stem: [question]
        Key: [correct answer]
        Distractors:
        - [distractor 1]
        - [distractor 2]
        - [distractor 3]
    
    Returns:
        dict or None
    """
    try:
        # Extract stem
        stem_match = re.search(r"Stem:\s*(.+?)(?=\nKey:|\nDistractors:|$)", chunk, re.DOTALL)
        if not stem_match:
            logger.warning(f"No stem found in MCQ chunk: {chunk[:100]}")
            return None
        stem = stem_match.group(1).strip()

        # Extract key (correct answer)
        key_match = re.search(r"Key:\s*(.+?)(?=\nDistractors:|\n-|$)", chunk, re.DOTALL)
        if not key_match:
            logger.warning(f"No key found in MCQ chunk: {chunk[:100]}")
            return None
        key = key_match.group(1).strip()

        # Extract distractors
        distractor_pattern = r"-\s*(.+?)(?=\n-|\n\n|$)"
        distractors = re.findall(distractor_pattern, chunk, re.DOTALL)
        distractors = [d.strip() for d in distractors if d.strip()]

        if len(distractors) < 3:
            logger.warning(f"Only {len(distractors)} distractors found, need 3")
            # Pad with generic distractors if needed
            while len(distractors) < 3:
                distractors.append("None of the above")

        return {
            "stem": stem,
            "key": key,
            "distractors": distractors[:3]  # Ensure exactly 3
        }

    except Exception as e:
        logger.error(f"Error parsing MCQ chunk: {e}")
        return None


def parse_theory_chunk(chunk):
    """
    Parse a single Theory question chunk into structured format.
    
    Expected format:
        Theory [number]
        Question: [question text]
        Answer: [answer text]
    
    Returns:
        dict or None
    """
    try:
        # Extract question
        question_match = re.search(r"Question:\s*(.+?)(?=\nAnswer:|$)", chunk, re.DOTALL)
        if not question_match:
            logger.warning(f"No question found in theory chunk: {chunk[:100]}")
            return None
        question = question_match.group(1).strip()

        # Extract answer
        answer_match = re.search(r"Answer:\s*(.+)$", chunk, re.DOTALL)
        if not answer_match:
            logger.warning(f"No answer found in theory chunk: {chunk[:100]}")
            return None
        answer = answer_match.group(1).strip()

        return {
            "question": question,
            "answer": answer
        }

    except Exception as e:
        logger.error(f"Error parsing theory chunk: {e}")
        return None


def load_studykit_dataset():
    """
    Load the StudyKit dataset JSON file.
    
    Returns:
        list: List of study kit entries
    
    Raises:
        FileNotFoundError: If dataset file doesn't exist
    """
    # Determine path relative to this file
    here = os.path.dirname(os.path.abspath(__file__))
    
    # Try multiple possible paths
    possible_paths = [
        os.path.join(here, "..", "..", "dataset", "studykit_questions_dataset.json"),
        os.path.join(here, "..", "dataset", "studykit_questions_dataset.json"),
        os.path.join(here, "dataset", "studykit_questions_dataset.json"),
    ]
    
    for dataset_path in possible_paths:
        dataset_path = os.path.normpath(dataset_path)
        if os.path.exists(dataset_path):
            try:
                with open(dataset_path, "r", encoding="utf-8") as f:
                    data = json.load(f)
                logger.info(f"Loaded dataset from: {dataset_path}")
                return data
            except json.JSONDecodeError as e:
                logger.error(f"Invalid JSON in dataset: {e}")
                return []
    
    raise FileNotFoundError(
        f"Cannot find studykit_questions_dataset.json in any of: {possible_paths}"
    )


def fetch_internet_content(topic):
    """
    Placeholder for fetching content from the internet.
    You should implement this based on your requirements.
    
    Args:
        topic: Topic to search for
    
    Returns:
        str: Content text or empty string
    """
    logger.warning(f"fetch_internet_content called for topic: {topic}")
    logger.warning("This function is not implemented. Returning empty content.")
    # TODO: Implement web scraping or API calls to fetch content
    return ""