import os # Ensure CUDA 12 libraries are findable for faster-whisper/CTranslate2 _cuda12_path = "/usr/local/lib/ollama/cuda_v12" if os.path.isdir(_cuda12_path): cur = os.environ.get("LD_LIBRARY_PATH", "") if _cuda12_path not in cur: os.environ["LD_LIBRARY_PATH"] = f"{_cuda12_path}:{cur}" if cur else _cuda12_path # get_device comes from the stt package (already a dependency) rather than # jebin_lib: importing jebin_lib pulls in hf_bucket_client, which imports # huggingface_hub.sync_bucket - removed from huggingface_hub - so a single # broken transitive import would take down server startup. try: from jebin_lib.utils import get_device except ImportError: from stt.common import get_device class Config: PORT = int(os.environ.get('PORT', 7860)) UPLOAD_FOLDER = 'uploads' TEMP_DIR = 'temp_dir' DATABASE_FILE = 'audio_captions.db' ALLOWED_EXTENSIONS = {'wav', 'mp3', 'flac', 'ogg', 'm4a', 'aac', 'mp4', 'mkv', 'avi', 'mov'} CWD = "./" PYTHON_PATH = "stt-transcribe" STT_MODEL_NAME = "parakeet" # Parakeet is English-only, so non-English audio and X->English translation # are routed to a whisper engine instead. STT_MULTILINGUAL_MODEL_NAME = os.environ.get('STT_MULTILINGUAL_MODEL', 'fasterwhispher') # faster-whisper checkpoint size used for those jobs; "base" translates # Hindi poorly, so default to something usable. STT_WHISPER_MODEL = os.environ.get('STT_WHISPER_MODEL', 'small') POLL_INTERVAL = 3 settings = Config() os.makedirs(settings.UPLOAD_FOLDER, exist_ok=True) os.makedirs(settings.TEMP_DIR, exist_ok=True)