File size: 3,376 Bytes
b2931f4
 
 
 
 
 
 
 
 
 
 
 
 
a8bc5d0
 
 
 
 
 
 
 
b2931f4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
# src/finrag/config.py
from pydantic_settings import BaseSettings, SettingsConfigDict
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[3]
# config.py β†’ finrag/ β†’ src/ β†’ backend/ β†’ REPO_ROOT

class Settings(BaseSettings):
    # Both provider keys are optional: the active one is decided by
    # `llm_provider`, and we validate presence lazily at call time so a
    # Gemini-only setup doesn't need an Anthropic key (and vice versa).
    anthropic_api_key: str | None = None
    gemini_api_key: str | None = None
    cohere_api_key: str
    # Vector store location. Two mutually exclusive modes:
    #   β€’ qdrant_path set  β†’ EMBEDDED Qdrant (in-process, on-disk at this path).
    #     The public deploy uses this: the vector store is baked into the Docker
    #     image alongside DuckDB + BM25, so there is no external cluster to idle-
    #     wipe and no QDRANT_URL/KEY secret. Relative paths resolve from REPO_ROOT.
    #   β€’ qdrant_path empty β†’ REMOTE Qdrant at qdrant_url (local docker-compose in
    #     dev, or a managed cluster). The original behaviour, unchanged.
    qdrant_path: str | None = None
    qdrant_url: str = "http://localhost:6333"
    qdrant_api_key: str | None = None
    duckdb_path: str = "./data/duckdb/finrag.duckdb"
    llm_mode: str = "cloud"
    # Which backend the finrag.llm dispatchers use: "anthropic" | "gemini" | "local".
    llm_provider: str = "anthropic"
    # Anthropic model. Default = the eval-validated Sonnet; the public demo deploy
    # overrides this to Haiku (CLAUDE_MODEL=claude-haiku-4-5-20251001) to cut the
    # per-question cost ~10x behind the guardrails below.
    claude_model: str = "claude-sonnet-4-6"

    # ── Public-deploy guardrails (only bite when a real client calls) ──
    # CORS allow-list, comma-separated. Dev defaults to the Next.js dev server;
    # the prod deploy sets this to the exact Vercel origin (see docs/deploy.md).
    allowed_origins: str = "http://localhost:3000,http://127.0.0.1:3000"
    # Per-IP sliding-window limit on the PAID endpoints (/answer,/agent[,/stream]).
    rate_limit_per_min: int = 8
    # Global hard ceiling on paid questions per UTC day β€” the cost circuit-breaker.
    # 300 Haiku agent-questions β‰ˆ a couple of dollars worst case; past it the API
    # returns 429 until midnight UTC instead of burning the key.
    daily_question_cap: int = 300

    @property
    def allowed_origins_list(self) -> list[str]:
        return [o.strip() for o in self.allowed_origins.split(",") if o.strip()]

    # ── Local / edge provider (Ollama, OpenAI-compatible API on :11434) ──
    # Used only when llm_provider == "local". base_url is the OpenAI-compat
    # endpoint, so the same backend would target vLLM/llama.cpp/LM Studio by
    # swapping this one value. local_use_tools is the kill-switch: True runs the
    # real agentic tool-loop (tests whether a 3B model can drive it); False runs
    # degraded synthesis-only over retrieved context (the documented fallback for
    # small models whose tool-calling is unreliable β€” see docs/handoff.md Day 5).
    local_model: str = "llama3.2:3b"
    local_base_url: str = "http://localhost:11434/v1"
    local_use_tools: bool = True

    model_config = SettingsConfigDict(env_file=REPO_ROOT / ".env", extra="ignore")

settings = Settings()  # validates at import time