FannyFa-Model-V1 / quiz_module.py
FannyFa's picture
Upload 16 files
3eecd6b verified
Raw History Blame Contribute Delete
21.9 kB
"""
Quiz Module - Question Answering & Model Evaluation
Fitur:
- Automatic quiz generation dari dataset
- Manual quiz creation
- Answer validation dengan multiple metrics
- Performance tracking
- Difficulty levels
"""
import os
import json
import time
import uuid
import random
from typing import List, Dict, Optional, Tuple, Any
from pathlib import Path
from dataclasses import dataclass, field, asdict
from datetime import datetime
from enum import Enum
from rich.table import Table
from rich import box
from rich.prompt import Prompt, Confirm
from rich.panel import Panel
from config import DATA_DIR, EXPORT_DIR
from utils import console, Theme, debug_logger, error_logger
class DifficultyLevel(Enum):
EASY = 1
MEDIUM = 2
HARD = 3
EXPERT = 4
class QuestionType(Enum):
MULTIPLE_CHOICE = "multiple_choice"
TRUE_FALSE = "true_false"
SHORT_ANSWER = "short_answer"
LONG_ANSWER = "long_answer"
FILL_BLANK = "fill_blank"
@dataclass
class Question:
"""Struktur question"""
id: str
text: str
question_type: QuestionType
difficulty: DifficultyLevel
correct_answer: str
options: List[str] = field(default_factory=list) # Untuk multiple choice
keywords: List[str] = field(default_factory=list) # Keywords untuk validasi
hints: List[str] = field(default_factory=list)
explanation: str = ""
source: str = "manual" # manual, auto_generated, dataset
created_at: str = field(default_factory=lambda: datetime.now().isoformat())
def to_dict(self) -> Dict[str, Any]:
return {
'id': self.id,
'text': self.text,
'question_type': self.question_type.value,
'difficulty': self.difficulty.value,
'correct_answer': self.correct_answer,
'options': self.options,
'keywords': self.keywords,
'hints': self.hints,
'explanation': self.explanation,
'source': self.source,
'created_at': self.created_at,
}
@dataclass
class QuizResult:
"""Hasil quiz"""
quiz_id: str
question_id: str
user_answer: str
correct_answer: str
score: float # 0.0 - 1.0
is_correct: bool
time_taken: float # seconds
answered_at: str = field(default_factory=lambda: datetime.now().isoformat())
def to_dict(self) -> Dict[str, Any]:
return asdict(self)
class AnswerValidator:
"""Validate answers dengan berbagai metode"""
def __init__(self):
self.similarity_threshold = 0.7
def validate_multiple_choice(self, user_answer: str, correct_answer: str, options: List[str]) -> Tuple[bool, float]:
"""Validate multiple choice answer"""
is_correct = user_answer.strip().lower() == correct_answer.strip().lower()
score = 1.0 if is_correct else 0.0
return is_correct, score
def validate_true_false(self, user_answer: str, correct_answer: str) -> Tuple[bool, float]:
"""Validate true/false answer"""
user_answer = user_answer.strip().lower()
correct_answer = correct_answer.strip().lower()
true_variants = {'true', 'benar', 'ya', 'yes', 't', 'b', '1'}
false_variants = {'false', 'salah', 'tidak', 'no', 'f', 's', '0'}
user_val = user_answer in true_variants or user_answer not in false_variants
correct_val = correct_answer in true_variants or correct_answer not in false_variants
is_correct = user_val == correct_val
score = 1.0 if is_correct else 0.0
return is_correct, score
def validate_short_answer(self, user_answer: str, correct_answer: str, keywords: List[str] = None) -> Tuple[bool, float]:
"""Validate short answer dengan keyword matching"""
user_lower = user_answer.strip().lower()
correct_lower = correct_answer.strip().lower()
# Exact match
if user_lower == correct_lower:
return True, 1.0
# Partial match dengan similarity
score = self._calculate_similarity(user_lower, correct_lower)
# Keyword matching
if keywords:
keywords_found = sum(1 for kw in keywords if kw.lower() in user_lower)
keyword_score = keywords_found / len(keywords) if keywords else 0
score = max(score, keyword_score)
is_correct = score >= self.similarity_threshold
return is_correct, score
def validate_long_answer(self, user_answer: str, correct_answer: str, keywords: List[str] = None) -> Tuple[bool, float]:
"""Validate long answer dengan semantic similarity"""
user_lower = user_answer.strip().lower()
score = 0.0
# Keyword matching
if keywords:
keywords_found = sum(1 for kw in keywords if kw.lower() in user_lower)
score = keywords_found / len(keywords) if keywords else 0
else:
# Use word overlap as fallback
user_words = set(user_lower.split())
correct_words = set(correct_answer.lower().split())
overlap = len(user_words & correct_words)
total = len(user_words | correct_words)
score = overlap / total if total > 0 else 0
is_correct = score >= self.similarity_threshold
return is_correct, score
def validate_fill_blank(self, user_answer: str, correct_answer: str, keywords: List[str] = None) -> Tuple[bool, float]:
"""Validate fill-the-blank answer"""
# Similar to short answer
return self.validate_short_answer(user_answer, correct_answer, keywords)
def validate(self, question: Question, user_answer: str) -> Tuple[bool, float]:
"""Main validation method"""
try:
if question.question_type == QuestionType.MULTIPLE_CHOICE:
return self.validate_multiple_choice(user_answer, question.correct_answer, question.options)
elif question.question_type == QuestionType.TRUE_FALSE:
return self.validate_true_false(user_answer, question.correct_answer)
elif question.question_type == QuestionType.SHORT_ANSWER:
return self.validate_short_answer(user_answer, question.correct_answer, question.keywords)
elif question.question_type == QuestionType.LONG_ANSWER:
return self.validate_long_answer(user_answer, question.correct_answer, question.keywords)
elif question.question_type == QuestionType.FILL_BLANK:
return self.validate_fill_blank(user_answer, question.correct_answer, question.keywords)
except Exception as e:
error_logger.error(f"Validation error: {e}")
return False, 0.0
def _calculate_similarity(self, str1: str, str2: str) -> float:
"""Calculate simple string similarity"""
from difflib import SequenceMatcher
return SequenceMatcher(None, str1, str2).ratio()
class QuizManager:
"""Manage quizzes dan track results"""
def __init__(self, quiz_dir: str = None):
self.quiz_dir = Path(quiz_dir or os.path.join(DATA_DIR, "quizzes"))
self.quiz_dir.mkdir(exist_ok=True, parents=True)
self.questions: Dict[str, Question] = {}
self.quizzes: Dict[str, List[str]] = {} # quiz_id -> list of question_ids
self.results: List[QuizResult] = []
self.validator = AnswerValidator()
self._load_questions()
self._load_results()
def add_question(self, question: Question) -> None:
"""Add question"""
if question is None:
return
self.questions[question.id] = question
self._save_questions()
def add_questions(self, questions: List[Question]) -> None:
"""Add multiple questions (skip None, dedupe by ID)"""
added = 0
for q in questions or []:
if q is None:
continue
self.questions[q.id] = q
added += 1
if added > 0:
self._save_questions()
def create_quiz(self, quiz_name: str, question_ids: List[str]) -> Optional[str]:
"""Create quiz dari selected questions"""
if not question_ids:
console.print(Theme.warning("Tidak ada pertanyaan untuk dibuat quiz"))
return None
quiz_id = f"quiz_{int(time.time())}_{uuid.uuid4().hex[:6]}"
self.quizzes[quiz_id] = list(question_ids)
console.print(
Theme.success(
f"Quiz created: {quiz_id} dengan {len(question_ids)} pertanyaan"
)
)
return quiz_id
def create_auto_quiz(
self,
num_questions: int = 5,
difficulty: DifficultyLevel = None,
quiz_name: str = None,
) -> Optional[str]:
"""Auto create quiz berdasarkan criteria. Return None kalau tidak bisa."""
# --- GUARD 1: tidak ada pertanyaan tersedia ---
if not self.questions:
console.print(Theme.warning("Belum ada pertanyaan tersedia"))
return None
# --- GUARD 2: num_questions tidak valid ---
if num_questions is None or num_questions <= 0:
console.print(Theme.warning(f"Jumlah soal tidak valid: {num_questions}"))
return None
questions = list(self.questions.values())
# Filter by difficulty if specified
if difficulty:
questions = [q for q in questions if q.difficulty == difficulty]
if not questions:
console.print(
Theme.warning(
f"Tidak ada pertanyaan dengan difficulty {difficulty}"
)
)
return None
# Random sample (aman kalau num_questions > len(questions))
n = min(num_questions, len(questions))
selected = random.sample(questions, n)
selected_ids = [q.id for q in selected]
return self.create_quiz(
quiz_name or f"Auto Quiz {len(self.quizzes) + 1}", selected_ids
)
def start_quiz(self, quiz_id: str) -> None:
"""Interactive quiz session"""
# --- GUARD 1: quiz_id tidak ada ---
if quiz_id is None:
console.print(Theme.warning("Quiz dibatalkan"))
return
if quiz_id not in self.quizzes:
console.print(Theme.error(f"Quiz {quiz_id} tidak ditemukan"))
return
question_ids = self.quizzes[quiz_id]
total_questions = len(question_ids)
# --- GUARD 2: quiz kosong (cegah ZeroDivisionError) ---
if total_questions == 0:
console.print(Theme.error("Quiz ini tidak memiliki pertanyaan"))
return
total_score = 0.0
answered_questions = 0
console.print(Panel(
f"[cyan]Quiz: {quiz_id}[/cyan]\n"
f"[white]Total Questions: {total_questions}[/white]",
style="cyan"
))
for idx, qid in enumerate(question_ids, 1):
# --- GUARD 3: qid tidak ada di self.questions ---
if qid not in self.questions:
console.print(
Theme.warning(f" [skip] Pertanyaan {qid} tidak ditemukan")
)
continue
question = self.questions[qid]
console.print(f"\n[cyan]Question {idx}/{total_questions}[/cyan]")
console.print(f"[yellow]{question.text}[/yellow]")
# Display options for multiple choice
if question.question_type == QuestionType.MULTIPLE_CHOICE:
for i, opt in enumerate(question.options, 1):
console.print(f" [{i}] {opt}")
# Display hints if available
if question.hints:
if Confirm.ask("[dim]Lihat hint?[/dim]", default=False):
for hint in question.hints:
console.print(f"[yellow]💡 {hint}[/yellow]")
start_time = time.time()
user_answer = Prompt.ask("[green]Your answer[/green]")
time_taken = time.time() - start_time
# Validate answer
is_correct, score = self.validator.validate(question, user_answer)
total_score += score
answered_questions += 1
# Display result
if is_correct:
console.print(Theme.success(" ✓ Benar!"))
else:
console.print(
Theme.error(f" ✗ Salah. Jawaban: {question.correct_answer}")
)
if question.explanation:
console.print(f"[dim]{question.explanation}[/dim]")
# Save result
result = QuizResult(
quiz_id=quiz_id,
question_id=qid,
user_answer=user_answer,
correct_answer=question.correct_answer,
score=score,
is_correct=is_correct,
time_taken=time_taken,
)
self.results.append(result)
# --- GUARD 4: tidak ada pertanyaan yang benar-benar dijawab ---
if answered_questions == 0:
console.print(
Theme.error("Tidak ada pertanyaan yang berhasil dijawab")
)
self._save_results()
return
# Summary (pakai answered_questions, bukan total_questions)
final_score = (total_score / answered_questions) * 100
correct_count = sum(
1
for r in self.results
if r.quiz_id == quiz_id and r.is_correct
)
console.print(Panel(
f"[green]Quiz Completed![/green]\n"
f"[cyan]Score: {final_score:.1f}%[/cyan]\n"
f"[white]Correct: {correct_count}/{answered_questions}[/white]",
style="green"
))
self._save_results()
def generate_questions_from_qa_pairs(
self, qa_pairs: List[Tuple[str, str]]
) -> List[Question]:
"""Generate questions dari Q&A pairs (skip yang kosong/terlalu pendek)"""
questions = []
skipped = 0
run_id = uuid.uuid4().hex[:8]
for idx, pair in enumerate(qa_pairs or []):
try:
question_text, answer = pair
except (TypeError, ValueError):
skipped += 1
continue
question_text = (question_text or "").strip()
answer = (answer or "").strip()
# Skip kalau terlalu pendek
if len(question_text) < 3 or len(answer) < 2:
skipped += 1
continue
keywords = [w for w in answer.split() if len(w) > 3][:15]
question = Question(
id=f"auto_{run_id}_{idx}",
text=question_text,
question_type=QuestionType.SHORT_ANSWER,
difficulty=DifficultyLevel.MEDIUM,
correct_answer=answer,
keywords=keywords,
source="auto_generated",
)
questions.append(question)
if skipped > 0:
debug_logger.debug(
f"generate_questions_from_qa_pairs: skipped {skipped} invalid pairs"
)
return questions
def get_statistics(self, quiz_id: str = None) -> Dict[str, Any]:
"""Get quiz statistics"""
if quiz_id:
results = [r for r in self.results if r.quiz_id == quiz_id]
else:
results = self.results
if not results:
return {'error': 'No results found'}
total = len(results)
correct = sum(1 for r in results if r.is_correct)
avg_score = sum(r.score for r in results) / total if total > 0 else 0
avg_time = sum(r.time_taken for r in results) / total if total > 0 else 0
return {
'total_attempts': total,
'correct_answers': correct,
'accuracy': (correct / total * 100) if total > 0 else 0,
'average_score': avg_score * 100,
'average_time_seconds': avg_time,
'total_time_seconds': sum(r.time_taken for r in results),
}
def display_results(self, quiz_id: str = None) -> None:
"""Display quiz results"""
stats = self.get_statistics(quiz_id)
if 'error' in stats:
console.print(Theme.warning(stats['error']))
return
table = Table(title="Quiz Statistics", box=box.ROUNDED)
table.add_column("Metric", style="cyan")
table.add_column("Value", style="green")
table.add_row("Total Attempts", str(stats['total_attempts']))
table.add_row(
"Correct Answers",
f"{stats['correct_answers']}/{stats['total_attempts']}",
)
table.add_row("Accuracy", f"{stats['accuracy']:.1f}%")
table.add_row("Average Score", f"{stats['average_score']:.1f}%")
table.add_row(
"Average Time per Question",
f"{stats['average_time_seconds']:.1f}s",
)
table.add_row("Total Time", f"{stats['total_time_seconds']:.0f}s")
console.print(table)
# ------------------------------------------------------------------
# Persistence
# ------------------------------------------------------------------
def _save_questions(self) -> None:
"""Save questions to disk"""
try:
questions_file = self.quiz_dir / "questions.json"
data = {qid: q.to_dict() for qid, q in self.questions.items()}
with open(questions_file, 'w', encoding='utf-8') as f:
json.dump(data, f, ensure_ascii=False, indent=2)
except Exception as e:
error_logger.error(f"Questions save error: {e}")
def _load_questions(self) -> None:
"""Load questions from disk (toleran terhadap data lama)"""
try:
questions_file = self.quiz_dir / "questions.json"
if not questions_file.exists():
return
with open(questions_file, 'r', encoding='utf-8') as f:
data = json.load(f)
loaded = 0
skipped = 0
for qid, q_data in data.items():
try:
# Toleran: kalau question_type / difficulty tidak dikenal,
# fallback ke default supaya file lama tetap bisa dibaca.
try:
q_type = QuestionType(q_data.get('question_type', 'short_answer'))
except ValueError:
q_type = QuestionType.SHORT_ANSWER
try:
q_diff = DifficultyLevel(q_data.get('difficulty', 2))
except (ValueError, TypeError):
q_diff = DifficultyLevel.MEDIUM
question = Question(
id=q_data.get('id', qid),
text=q_data.get('text', ''),
question_type=q_type,
difficulty=q_diff,
correct_answer=q_data.get('correct_answer', ''),
options=q_data.get('options', []) or [],
keywords=q_data.get('keywords', []) or [],
hints=q_data.get('hints', []) or [],
explanation=q_data.get('explanation', ''),
source=q_data.get('source', 'manual'),
created_at=q_data.get('created_at', ''),
)
self.questions[question.id] = question
loaded += 1
except Exception as inner_e:
skipped += 1
debug_logger.debug(
f"Skipping malformed question {qid}: {inner_e}"
)
if loaded > 0:
console.print(
Theme.success(f"Loaded {loaded} questions from disk")
)
if skipped > 0:
console.print(
Theme.warning(f"Skipped {skipped} malformed questions")
)
except Exception as e:
error_logger.error(f"Questions load error: {e}")
def _save_results(self) -> None:
"""Save results to disk"""
try:
results_file = self.quiz_dir / "results.json"
with open(results_file, 'w', encoding='utf-8') as f:
json.dump(
[r.to_dict() for r in self.results],
f,
ensure_ascii=False,
indent=2,
default=str,
)
except Exception as e:
error_logger.error(f"Results save error: {e}")
def _load_results(self) -> None:
"""Load results from disk"""
try:
results_file = self.quiz_dir / "results.json"
if not results_file.exists():
return
with open(results_file, 'r', encoding='utf-8') as f:
data = json.load(f)
for r_data in data:
try:
result = QuizResult(
quiz_id=r_data.get('quiz_id', ''),
question_id=r_data.get('question_id', ''),
user_answer=r_data.get('user_answer', ''),
correct_answer=r_data.get('correct_answer', ''),
score=float(r_data.get('score', 0.0)),
is_correct=bool(r_data.get('is_correct', False)),
time_taken=float(r_data.get('time_taken', 0.0)),
answered_at=r_data.get('answered_at', ''),
)
self.results.append(result)
except Exception as inner_e:
debug_logger.debug(f"Skipping malformed result: {inner_e}")
except Exception as e:
debug_logger.debug(f"Results load error: {e}")