""" Quiz Module - Question Answering & Model Evaluation Fitur: - Automatic quiz generation dari dataset - Manual quiz creation - Answer validation dengan multiple metrics - Performance tracking - Difficulty levels """ import os import json import time import uuid import random from typing import List, Dict, Optional, Tuple, Any from pathlib import Path from dataclasses import dataclass, field, asdict from datetime import datetime from enum import Enum from rich.table import Table from rich import box from rich.prompt import Prompt, Confirm from rich.panel import Panel from config import DATA_DIR, EXPORT_DIR from utils import console, Theme, debug_logger, error_logger class DifficultyLevel(Enum): EASY = 1 MEDIUM = 2 HARD = 3 EXPERT = 4 class QuestionType(Enum): MULTIPLE_CHOICE = "multiple_choice" TRUE_FALSE = "true_false" SHORT_ANSWER = "short_answer" LONG_ANSWER = "long_answer" FILL_BLANK = "fill_blank" @dataclass class Question: """Struktur question""" id: str text: str question_type: QuestionType difficulty: DifficultyLevel correct_answer: str options: List[str] = field(default_factory=list) # Untuk multiple choice keywords: List[str] = field(default_factory=list) # Keywords untuk validasi hints: List[str] = field(default_factory=list) explanation: str = "" source: str = "manual" # manual, auto_generated, dataset created_at: str = field(default_factory=lambda: datetime.now().isoformat()) def to_dict(self) -> Dict[str, Any]: return { 'id': self.id, 'text': self.text, 'question_type': self.question_type.value, 'difficulty': self.difficulty.value, 'correct_answer': self.correct_answer, 'options': self.options, 'keywords': self.keywords, 'hints': self.hints, 'explanation': self.explanation, 'source': self.source, 'created_at': self.created_at, } @dataclass class QuizResult: """Hasil quiz""" quiz_id: str question_id: str user_answer: str correct_answer: str score: float # 0.0 - 1.0 is_correct: bool time_taken: float # seconds answered_at: str = field(default_factory=lambda: datetime.now().isoformat()) def to_dict(self) -> Dict[str, Any]: return asdict(self) class AnswerValidator: """Validate answers dengan berbagai metode""" def __init__(self): self.similarity_threshold = 0.7 def validate_multiple_choice(self, user_answer: str, correct_answer: str, options: List[str]) -> Tuple[bool, float]: """Validate multiple choice answer""" is_correct = user_answer.strip().lower() == correct_answer.strip().lower() score = 1.0 if is_correct else 0.0 return is_correct, score def validate_true_false(self, user_answer: str, correct_answer: str) -> Tuple[bool, float]: """Validate true/false answer""" user_answer = user_answer.strip().lower() correct_answer = correct_answer.strip().lower() true_variants = {'true', 'benar', 'ya', 'yes', 't', 'b', '1'} false_variants = {'false', 'salah', 'tidak', 'no', 'f', 's', '0'} user_val = user_answer in true_variants or user_answer not in false_variants correct_val = correct_answer in true_variants or correct_answer not in false_variants is_correct = user_val == correct_val score = 1.0 if is_correct else 0.0 return is_correct, score def validate_short_answer(self, user_answer: str, correct_answer: str, keywords: List[str] = None) -> Tuple[bool, float]: """Validate short answer dengan keyword matching""" user_lower = user_answer.strip().lower() correct_lower = correct_answer.strip().lower() # Exact match if user_lower == correct_lower: return True, 1.0 # Partial match dengan similarity score = self._calculate_similarity(user_lower, correct_lower) # Keyword matching if keywords: keywords_found = sum(1 for kw in keywords if kw.lower() in user_lower) keyword_score = keywords_found / len(keywords) if keywords else 0 score = max(score, keyword_score) is_correct = score >= self.similarity_threshold return is_correct, score def validate_long_answer(self, user_answer: str, correct_answer: str, keywords: List[str] = None) -> Tuple[bool, float]: """Validate long answer dengan semantic similarity""" user_lower = user_answer.strip().lower() score = 0.0 # Keyword matching if keywords: keywords_found = sum(1 for kw in keywords if kw.lower() in user_lower) score = keywords_found / len(keywords) if keywords else 0 else: # Use word overlap as fallback user_words = set(user_lower.split()) correct_words = set(correct_answer.lower().split()) overlap = len(user_words & correct_words) total = len(user_words | correct_words) score = overlap / total if total > 0 else 0 is_correct = score >= self.similarity_threshold return is_correct, score def validate_fill_blank(self, user_answer: str, correct_answer: str, keywords: List[str] = None) -> Tuple[bool, float]: """Validate fill-the-blank answer""" # Similar to short answer return self.validate_short_answer(user_answer, correct_answer, keywords) def validate(self, question: Question, user_answer: str) -> Tuple[bool, float]: """Main validation method""" try: if question.question_type == QuestionType.MULTIPLE_CHOICE: return self.validate_multiple_choice(user_answer, question.correct_answer, question.options) elif question.question_type == QuestionType.TRUE_FALSE: return self.validate_true_false(user_answer, question.correct_answer) elif question.question_type == QuestionType.SHORT_ANSWER: return self.validate_short_answer(user_answer, question.correct_answer, question.keywords) elif question.question_type == QuestionType.LONG_ANSWER: return self.validate_long_answer(user_answer, question.correct_answer, question.keywords) elif question.question_type == QuestionType.FILL_BLANK: return self.validate_fill_blank(user_answer, question.correct_answer, question.keywords) except Exception as e: error_logger.error(f"Validation error: {e}") return False, 0.0 def _calculate_similarity(self, str1: str, str2: str) -> float: """Calculate simple string similarity""" from difflib import SequenceMatcher return SequenceMatcher(None, str1, str2).ratio() class QuizManager: """Manage quizzes dan track results""" def __init__(self, quiz_dir: str = None): self.quiz_dir = Path(quiz_dir or os.path.join(DATA_DIR, "quizzes")) self.quiz_dir.mkdir(exist_ok=True, parents=True) self.questions: Dict[str, Question] = {} self.quizzes: Dict[str, List[str]] = {} # quiz_id -> list of question_ids self.results: List[QuizResult] = [] self.validator = AnswerValidator() self._load_questions() self._load_results() def add_question(self, question: Question) -> None: """Add question""" if question is None: return self.questions[question.id] = question self._save_questions() def add_questions(self, questions: List[Question]) -> None: """Add multiple questions (skip None, dedupe by ID)""" added = 0 for q in questions or []: if q is None: continue self.questions[q.id] = q added += 1 if added > 0: self._save_questions() def create_quiz(self, quiz_name: str, question_ids: List[str]) -> Optional[str]: """Create quiz dari selected questions""" if not question_ids: console.print(Theme.warning("Tidak ada pertanyaan untuk dibuat quiz")) return None quiz_id = f"quiz_{int(time.time())}_{uuid.uuid4().hex[:6]}" self.quizzes[quiz_id] = list(question_ids) console.print( Theme.success( f"Quiz created: {quiz_id} dengan {len(question_ids)} pertanyaan" ) ) return quiz_id def create_auto_quiz( self, num_questions: int = 5, difficulty: DifficultyLevel = None, quiz_name: str = None, ) -> Optional[str]: """Auto create quiz berdasarkan criteria. Return None kalau tidak bisa.""" # --- GUARD 1: tidak ada pertanyaan tersedia --- if not self.questions: console.print(Theme.warning("Belum ada pertanyaan tersedia")) return None # --- GUARD 2: num_questions tidak valid --- if num_questions is None or num_questions <= 0: console.print(Theme.warning(f"Jumlah soal tidak valid: {num_questions}")) return None questions = list(self.questions.values()) # Filter by difficulty if specified if difficulty: questions = [q for q in questions if q.difficulty == difficulty] if not questions: console.print( Theme.warning( f"Tidak ada pertanyaan dengan difficulty {difficulty}" ) ) return None # Random sample (aman kalau num_questions > len(questions)) n = min(num_questions, len(questions)) selected = random.sample(questions, n) selected_ids = [q.id for q in selected] return self.create_quiz( quiz_name or f"Auto Quiz {len(self.quizzes) + 1}", selected_ids ) def start_quiz(self, quiz_id: str) -> None: """Interactive quiz session""" # --- GUARD 1: quiz_id tidak ada --- if quiz_id is None: console.print(Theme.warning("Quiz dibatalkan")) return if quiz_id not in self.quizzes: console.print(Theme.error(f"Quiz {quiz_id} tidak ditemukan")) return question_ids = self.quizzes[quiz_id] total_questions = len(question_ids) # --- GUARD 2: quiz kosong (cegah ZeroDivisionError) --- if total_questions == 0: console.print(Theme.error("Quiz ini tidak memiliki pertanyaan")) return total_score = 0.0 answered_questions = 0 console.print(Panel( f"[cyan]Quiz: {quiz_id}[/cyan]\n" f"[white]Total Questions: {total_questions}[/white]", style="cyan" )) for idx, qid in enumerate(question_ids, 1): # --- GUARD 3: qid tidak ada di self.questions --- if qid not in self.questions: console.print( Theme.warning(f" [skip] Pertanyaan {qid} tidak ditemukan") ) continue question = self.questions[qid] console.print(f"\n[cyan]Question {idx}/{total_questions}[/cyan]") console.print(f"[yellow]{question.text}[/yellow]") # Display options for multiple choice if question.question_type == QuestionType.MULTIPLE_CHOICE: for i, opt in enumerate(question.options, 1): console.print(f" [{i}] {opt}") # Display hints if available if question.hints: if Confirm.ask("[dim]Lihat hint?[/dim]", default=False): for hint in question.hints: console.print(f"[yellow]💡 {hint}[/yellow]") start_time = time.time() user_answer = Prompt.ask("[green]Your answer[/green]") time_taken = time.time() - start_time # Validate answer is_correct, score = self.validator.validate(question, user_answer) total_score += score answered_questions += 1 # Display result if is_correct: console.print(Theme.success(" ✓ Benar!")) else: console.print( Theme.error(f" ✗ Salah. Jawaban: {question.correct_answer}") ) if question.explanation: console.print(f"[dim]{question.explanation}[/dim]") # Save result result = QuizResult( quiz_id=quiz_id, question_id=qid, user_answer=user_answer, correct_answer=question.correct_answer, score=score, is_correct=is_correct, time_taken=time_taken, ) self.results.append(result) # --- GUARD 4: tidak ada pertanyaan yang benar-benar dijawab --- if answered_questions == 0: console.print( Theme.error("Tidak ada pertanyaan yang berhasil dijawab") ) self._save_results() return # Summary (pakai answered_questions, bukan total_questions) final_score = (total_score / answered_questions) * 100 correct_count = sum( 1 for r in self.results if r.quiz_id == quiz_id and r.is_correct ) console.print(Panel( f"[green]Quiz Completed![/green]\n" f"[cyan]Score: {final_score:.1f}%[/cyan]\n" f"[white]Correct: {correct_count}/{answered_questions}[/white]", style="green" )) self._save_results() def generate_questions_from_qa_pairs( self, qa_pairs: List[Tuple[str, str]] ) -> List[Question]: """Generate questions dari Q&A pairs (skip yang kosong/terlalu pendek)""" questions = [] skipped = 0 run_id = uuid.uuid4().hex[:8] for idx, pair in enumerate(qa_pairs or []): try: question_text, answer = pair except (TypeError, ValueError): skipped += 1 continue question_text = (question_text or "").strip() answer = (answer or "").strip() # Skip kalau terlalu pendek if len(question_text) < 3 or len(answer) < 2: skipped += 1 continue keywords = [w for w in answer.split() if len(w) > 3][:15] question = Question( id=f"auto_{run_id}_{idx}", text=question_text, question_type=QuestionType.SHORT_ANSWER, difficulty=DifficultyLevel.MEDIUM, correct_answer=answer, keywords=keywords, source="auto_generated", ) questions.append(question) if skipped > 0: debug_logger.debug( f"generate_questions_from_qa_pairs: skipped {skipped} invalid pairs" ) return questions def get_statistics(self, quiz_id: str = None) -> Dict[str, Any]: """Get quiz statistics""" if quiz_id: results = [r for r in self.results if r.quiz_id == quiz_id] else: results = self.results if not results: return {'error': 'No results found'} total = len(results) correct = sum(1 for r in results if r.is_correct) avg_score = sum(r.score for r in results) / total if total > 0 else 0 avg_time = sum(r.time_taken for r in results) / total if total > 0 else 0 return { 'total_attempts': total, 'correct_answers': correct, 'accuracy': (correct / total * 100) if total > 0 else 0, 'average_score': avg_score * 100, 'average_time_seconds': avg_time, 'total_time_seconds': sum(r.time_taken for r in results), } def display_results(self, quiz_id: str = None) -> None: """Display quiz results""" stats = self.get_statistics(quiz_id) if 'error' in stats: console.print(Theme.warning(stats['error'])) return table = Table(title="Quiz Statistics", box=box.ROUNDED) table.add_column("Metric", style="cyan") table.add_column("Value", style="green") table.add_row("Total Attempts", str(stats['total_attempts'])) table.add_row( "Correct Answers", f"{stats['correct_answers']}/{stats['total_attempts']}", ) table.add_row("Accuracy", f"{stats['accuracy']:.1f}%") table.add_row("Average Score", f"{stats['average_score']:.1f}%") table.add_row( "Average Time per Question", f"{stats['average_time_seconds']:.1f}s", ) table.add_row("Total Time", f"{stats['total_time_seconds']:.0f}s") console.print(table) # ------------------------------------------------------------------ # Persistence # ------------------------------------------------------------------ def _save_questions(self) -> None: """Save questions to disk""" try: questions_file = self.quiz_dir / "questions.json" data = {qid: q.to_dict() for qid, q in self.questions.items()} with open(questions_file, 'w', encoding='utf-8') as f: json.dump(data, f, ensure_ascii=False, indent=2) except Exception as e: error_logger.error(f"Questions save error: {e}") def _load_questions(self) -> None: """Load questions from disk (toleran terhadap data lama)""" try: questions_file = self.quiz_dir / "questions.json" if not questions_file.exists(): return with open(questions_file, 'r', encoding='utf-8') as f: data = json.load(f) loaded = 0 skipped = 0 for qid, q_data in data.items(): try: # Toleran: kalau question_type / difficulty tidak dikenal, # fallback ke default supaya file lama tetap bisa dibaca. try: q_type = QuestionType(q_data.get('question_type', 'short_answer')) except ValueError: q_type = QuestionType.SHORT_ANSWER try: q_diff = DifficultyLevel(q_data.get('difficulty', 2)) except (ValueError, TypeError): q_diff = DifficultyLevel.MEDIUM question = Question( id=q_data.get('id', qid), text=q_data.get('text', ''), question_type=q_type, difficulty=q_diff, correct_answer=q_data.get('correct_answer', ''), options=q_data.get('options', []) or [], keywords=q_data.get('keywords', []) or [], hints=q_data.get('hints', []) or [], explanation=q_data.get('explanation', ''), source=q_data.get('source', 'manual'), created_at=q_data.get('created_at', ''), ) self.questions[question.id] = question loaded += 1 except Exception as inner_e: skipped += 1 debug_logger.debug( f"Skipping malformed question {qid}: {inner_e}" ) if loaded > 0: console.print( Theme.success(f"Loaded {loaded} questions from disk") ) if skipped > 0: console.print( Theme.warning(f"Skipped {skipped} malformed questions") ) except Exception as e: error_logger.error(f"Questions load error: {e}") def _save_results(self) -> None: """Save results to disk""" try: results_file = self.quiz_dir / "results.json" with open(results_file, 'w', encoding='utf-8') as f: json.dump( [r.to_dict() for r in self.results], f, ensure_ascii=False, indent=2, default=str, ) except Exception as e: error_logger.error(f"Results save error: {e}") def _load_results(self) -> None: """Load results from disk""" try: results_file = self.quiz_dir / "results.json" if not results_file.exists(): return with open(results_file, 'r', encoding='utf-8') as f: data = json.load(f) for r_data in data: try: result = QuizResult( quiz_id=r_data.get('quiz_id', ''), question_id=r_data.get('question_id', ''), user_answer=r_data.get('user_answer', ''), correct_answer=r_data.get('correct_answer', ''), score=float(r_data.get('score', 0.0)), is_correct=bool(r_data.get('is_correct', False)), time_taken=float(r_data.get('time_taken', 0.0)), answered_at=r_data.get('answered_at', ''), ) self.results.append(result) except Exception as inner_e: debug_logger.debug(f"Skipping malformed result: {inner_e}") except Exception as e: debug_logger.debug(f"Results load error: {e}")