from __future__ import annotations from dataclasses import dataclass import os from pathlib import Path import tempfile def number(name: str, default: int, minimum: int = 1) -> int: try: value = int(os.getenv(name, str(default))) except ValueError as exc: raise ValueError(f"{name} must be an integer.") from exc if value < minimum: raise ValueError(f"{name} must be >= {minimum}.") return value @dataclass(frozen=True) class Settings: app_title: str = os.getenv("APP_TITLE", "Document Exam Trainer") embedding_model: str = os.getenv("EMBEDDING_MODEL_ID", "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2") question_model: str = os.getenv("QUESTION_MODEL_ID", "Qwen/Qwen3-4B") grading_model: str = os.getenv("GRADING_MODEL_ID", "Qwen/Qwen3-14B") speech_model: str = os.getenv("SPEECH_MODEL_ID", "small") demo: bool = os.getenv("DEMO_MODE", "0") == "1" root: Path = Path(os.getenv("TRAINER_DATA_DIR", str(Path(tempfile.gettempdir()) / "document-exam-trainer-private"))).resolve() max_files: int = number("MAX_DOCUMENTS", 60) max_file_bytes: int = number("MAX_FILE_MB", 60) * 1024 * 1024 max_total_bytes: int = number("MAX_CORPUS_MB", 160) * 1024 * 1024 max_expanded_bytes: int = number("MAX_EXPANDED_MB", 240) * 1024 * 1024 max_units: int = number("MAX_DOCUMENT_UNITS", 1200) max_chunks: int = number("MAX_CHUNKS", 2200) max_text_chars: int = number("MAX_TEXT_CHARS", 4_000_000) chunk_chars: int = number("CHUNK_CHARS", 1800) max_audio_seconds: int = number("MAX_AUDIO_SECONDS", 300) max_audio_bytes: int = number("MAX_AUDIO_MB", 60) * 1024 * 1024 max_answer_chars: int = number("MAX_ANSWER_CHARS", 18000) session_ttl: int = number("SESSION_TTL_SECONDS", 7200) max_sessions: int = number("MAX_SESSIONS", 40) question_seconds: int = number("QUESTION_GPU_DURATION_SECONDS", 45) grading_seconds: int = number("GRADING_GPU_DURATION_SECONDS", 60) hint_seconds: int = number("HINT_GPU_DURATION_SECONDS", 25) max_prompt_tokens: int = number("MAX_PROMPT_TOKENS", 8500) SETTINGS = Settings()