Spaces:
Running
Running
Download yui_context_cache.py from BachDaThan/Ural-AI: direct link, hf CLI and curl.
- Browser
- Download file 31.9 kB
-
https://huggingface.co/spaces/BachDaThan/Ural-AI/resolve/main/yui_context_cache.py
- Command line
-
hf download hf://spaces/BachDaThan/Ural-AI/yui_context_cache.py
-
curl -L -o yui_context_cache.py https://huggingface.co/spaces/BachDaThan/Ural-AI/resolve/main/yui_context_cache.py
31.9 kB
| # ╔══════════════════════════════════════════════════════════════════════════════╗ | |
| # ║ YUI_CONTEXT_CACHE.py — Gemini Context Caching Engine v2.0 ║ | |
| # ╠══════════════════════════════════════════════════════════════════════════════╣ | |
| # ║ v2.0 mới so với v1.0: ║ | |
| # ║ • ProviderCacheLimits: Bộ quét tự động giới hạn cache của từng hãng ║ | |
| # ║ • SmartPromptTracker: Nhớ prompt nào đã cache, tránh gửi chồng lấp ║ | |
| # ║ • Multi-key rotation: Tự xoay vòng API key khi cache gần hết hạn ║ | |
| # ║ • Cache namespace: Mỗi model có slot cache riêng, không đụng nhau ║ | |
| # ╠══════════════════════════════════════════════════════════════════════════════╣ | |
| # ║ FREE-TIER SAFE | LIGHTWEIGHT ~10KB RAM | THREAD-SAFE ║ | |
| # ╚══════════════════════════════════════════════════════════════════════════════╝ | |
| import hashlib | |
| import time | |
| import threading | |
| import requests | |
| from typing import Optional, Dict, Tuple, List | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # PROVIDER CACHE LIMITS — bộ quét giới hạn từng hãng | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| class ProviderCacheLimits: | |
| """ | |
| Bộ quét và lưu giới hạn Context Cache của từng AI provider. | |
| Nguồn dữ liệu: | |
| 1. Bảng cứng (hardcoded) — cập nhật theo tài liệu chính thức | |
| 2. Tự động probe bằng cách thử tạo cache nhỏ và xem response | |
| Giới hạn quan trọng nhất của Gemini Free Tier: | |
| - min_tokens : 32,768 tokens (phải vượt ngưỡng này mới cache được) | |
| - ttl_max : 300 giây (5 phút) — Google xóa sau 5p nếu không renew | |
| - slots_max : 1 cache đang active cùng lúc (free tier) | |
| """ | |
| # Bảng giới hạn cứng — cập nhật từ docs.google.com/gemini/caching | |
| _KNOWN_LIMITS: Dict[str, dict] = { | |
| # Google Gemini — hãng DUY NHẤT cho cache free hiện tại (2026) | |
| "google": { | |
| "supported": True, | |
| "min_tokens": 32_768, # bắt buộc prompt >= 32k tokens | |
| "ttl_max": 300, # 5 phút (300 giây) | |
| "ttl_default": 295, # dùng 295s để an toàn | |
| "slots_max": 1, # free tier: 1 cache cùng lúc | |
| "models": [ # model hỗ trợ cache | |
| "gemini-3.5-flash", | |
| "gemini-2.5-flash", | |
| "gemini-2.5-pro-preview-06-05", | |
| "gemini-2.0-flash", | |
| "gemini-1.5-flash-latest", | |
| "gemini-1.5-pro-latest", | |
| ], | |
| "notes": "Free tier: chỉ Google hỗ trợ cache miễn phí (2026)", | |
| }, | |
| # Anthropic — cache có phí (paid only) | |
| "anthropic": { | |
| "supported": False, | |
| "min_tokens": 2_048, | |
| "ttl_max": 300, | |
| "ttl_default": 295, | |
| "slots_max": 0, | |
| "models": [], | |
| "notes": "Anthropic cache chỉ có trên paid plan, không free", | |
| }, | |
| # OpenAI — cache tự động, không API manual, paid only | |
| "openai": { | |
| "supported": False, | |
| "min_tokens": 1_024, | |
| "ttl_max": 300, | |
| "ttl_default": 295, | |
| "slots_max": 0, | |
| "models": [], | |
| "notes": "OpenAI auto-cache nhưng không expose free API", | |
| }, | |
| # Mistral — chưa có cache API public (2026) | |
| "mistral": { | |
| "supported": False, | |
| "min_tokens": 0, | |
| "ttl_max": 0, | |
| "ttl_default": 0, | |
| "slots_max": 0, | |
| "models": [], | |
| "notes": "Mistral chưa public cache API (2026)", | |
| }, | |
| # Cohere — chưa có cache API public (2026) | |
| "cohere": { | |
| "supported": False, | |
| "min_tokens": 0, | |
| "ttl_max": 0, | |
| "ttl_default": 0, | |
| "slots_max": 0, | |
| "models": [], | |
| "notes": "Cohere chưa public cache API (2026)", | |
| }, | |
| # Cerebras — siêu nhanh nhưng không có cache API (2026) | |
| "cerebras": { | |
| "supported": False, | |
| "min_tokens": 0, | |
| "ttl_max": 0, | |
| "ttl_default": 0, | |
| "slots_max": 0, | |
| "models": [], | |
| "notes": "Cerebras không có cache API — bù lại tốc độ 2000 tok/s", | |
| }, | |
| } | |
| # Kết quả probe thực tế (ghi đè _KNOWN_LIMITS nếu probe thành công) | |
| _probed: Dict[str, dict] = {} | |
| _probe_lock = threading.Lock() | |
| _last_probe: Dict[str, float] = {} | |
| _PROBE_INTERVAL = 3600 * 6 # probe lại mỗi 6 giờ | |
| def get(cls, provider: str) -> dict: | |
| """Trả về giới hạn cache của provider, ưu tiên probe > hardcoded.""" | |
| provider = provider.lower() | |
| with cls._probe_lock: | |
| probed = cls._probed.get(provider) | |
| if probed: | |
| return probed | |
| return cls._KNOWN_LIMITS.get(provider, { | |
| "supported": False, "min_tokens": 0, "ttl_max": 0, | |
| "slots_max": 0, "models": [], "notes": "Unknown provider" | |
| }) | |
| def get_google(cls) -> dict: | |
| return cls.get("google") | |
| def is_supported(cls, provider: str) -> bool: | |
| return cls.get(provider).get("supported", False) | |
| def min_tokens(cls, provider: str = "google") -> int: | |
| return cls.get(provider).get("min_tokens", 32_768) | |
| def ttl(cls, provider: str = "google") -> int: | |
| return cls.get(provider).get("ttl_default", 295) | |
| def probe_google(cls, api_key: str, model: str = "gemini-2.5-flash") -> bool: | |
| """ | |
| Tự động thử probe giới hạn thực tế của Google API. | |
| Gửi 1 request countTokens để xác nhận key còn hoạt động, | |
| sau đó cập nhật _probed. | |
| Trả về True nếu probe thành công. | |
| """ | |
| now = time.time() | |
| last = cls._last_probe.get("google", 0) | |
| if now - last < cls._PROBE_INTERVAL: | |
| return True # Đã probe gần đây, bỏ qua | |
| try: | |
| url = (f"https://generativelanguage.googleapis.com/v1beta/" | |
| f"models/{model}:countTokens?key={api_key}") | |
| # Gửi text ngắn để đếm tokens — nếu OK là key hợp lệ | |
| res = requests.post(url, json={ | |
| "contents": [{"parts": [{"text": "test"}]}] | |
| }, timeout=8) | |
| if res.status_code == 200: | |
| # Key OK, cập nhật probe record | |
| with cls._probe_lock: | |
| cls._probed["google"] = dict(cls._KNOWN_LIMITS["google"]) | |
| cls._probed["google"]["probe_ok"] = True | |
| cls._probed["google"]["probe_time"] = now | |
| cls._probed["google"]["active_model"] = model | |
| cls._last_probe["google"] = now | |
| print(f"✅ [CACHE-PROBE] Google probe OK — model={model}") | |
| return True | |
| elif res.status_code == 429: | |
| # Rate limit — key OK nhưng quota đầy | |
| print(f"⚠️ [CACHE-PROBE] Google quota đầy (429)") | |
| cls._last_probe["google"] = now | |
| return False | |
| else: | |
| print(f"⚠️ [CACHE-PROBE] Google probe lỗi: {res.status_code}") | |
| cls._last_probe["google"] = now | |
| return False | |
| except Exception as e: | |
| print(f"⚠️ [CACHE-PROBE] Exception: {e}") | |
| return False | |
| def summary(cls) -> str: | |
| """Tóm tắt giới hạn cache của tất cả provider — dùng cho lệnh admin.""" | |
| lines = ["📊 **Provider Cache Limits:**\n"] | |
| for provider, limits in cls._KNOWN_LIMITS.items(): | |
| icon = "✅" if limits["supported"] else "❌" | |
| line = f"{icon} **{provider.upper()}**: " | |
| if limits["supported"]: | |
| line += (f"min={limits['min_tokens']:,} tokens | " | |
| f"TTL={limits['ttl_max']}s | " | |
| f"slots={limits['slots_max']}") | |
| else: | |
| line += f"Không hỗ trợ — {limits['notes']}" | |
| lines.append(line) | |
| lines.append("\n💡 Chỉ Google cung cấp Context Cache miễn phí (2026)") | |
| return "\n".join(lines) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # SMART PROMPT TRACKER — chống gửi prompt trùng / chồng lấp | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| class SmartPromptTracker: | |
| """ | |
| Nhớ chính xác prompt nào đã được gửi lên Google cache. | |
| Mục đích: Tránh tình trạng gửi chồng nhiều prompt → đốt token. | |
| Cơ chế: | |
| - Mỗi prompt được hash SHA256 → ID duy nhất | |
| - Khi cache được tạo → đăng ký hash vào tracker | |
| - Khi có request mới → kiểm tra hash xem đã cache chưa | |
| - Nếu đã cache → KHÔNG gửi lại prompt, chỉ gửi user_message | |
| - Nếu chưa cache hoặc hết hạn → tạo cache mới, đăng ký lại | |
| Thread-safe với Lock. | |
| """ | |
| def __init__(self): | |
| self._lock = threading.Lock() | |
| # {prompt_hash: {"cached_at": float, "expires_at": float, | |
| # "cache_name": str, "model": str, | |
| # "char_count": int, "token_est": int}} | |
| self._registry: Dict[str, dict] = {} | |
| def register( | |
| self, | |
| prompt_hash: str, | |
| cache_name: str, | |
| model: str, | |
| char_count: int, | |
| token_est: int, | |
| ttl: int = 295, | |
| ): | |
| """Đăng ký một prompt đã được cache thành công.""" | |
| now = time.time() | |
| with self._lock: | |
| self._registry[prompt_hash] = { | |
| "cached_at": now, | |
| "expires_at": now + ttl, | |
| "cache_name": cache_name, | |
| "model": model, | |
| "char_count": char_count, | |
| "token_est": token_est, | |
| } | |
| print(f"📝 [TRACKER] Đã đăng ký prompt " | |
| f"hash={prompt_hash[:8]}... | model={model} | " | |
| f"~{token_est:,} tokens | TTL={ttl}s") | |
| def lookup(self, prompt_hash: str) -> Optional[dict]: | |
| """ | |
| Tra cứu xem prompt đã được cache chưa. | |
| Trả về entry nếu còn sống, None nếu không có hoặc hết hạn. | |
| """ | |
| now = time.time() | |
| with self._lock: | |
| entry = self._registry.get(prompt_hash) | |
| if not entry: | |
| return None | |
| if now >= entry["expires_at"]: | |
| # Hết hạn → xóa khỏi registry | |
| del self._registry[prompt_hash] | |
| print(f"🕐 [TRACKER] Prompt {prompt_hash[:8]}... đã hết hạn, xóa registry") | |
| return None | |
| remaining = int(entry["expires_at"] - now) | |
| print(f"🟢 [TRACKER] Prompt {prompt_hash[:8]}... còn cache {remaining}s") | |
| return entry | |
| def is_cached(self, prompt_hash: str) -> bool: | |
| return self.lookup(prompt_hash) is not None | |
| def invalidate(self, prompt_hash: str): | |
| """Xóa thủ công một prompt khỏi tracker.""" | |
| with self._lock: | |
| if prompt_hash in self._registry: | |
| del self._registry[prompt_hash] | |
| print(f"🗑️ [TRACKER] Đã xóa prompt {prompt_hash[:8]}...") | |
| def invalidate_all(self): | |
| """Reset toàn bộ tracker.""" | |
| with self._lock: | |
| count = len(self._registry) | |
| self._registry.clear() | |
| print(f"🗑️ [TRACKER] Đã xóa {count} entries") | |
| def sweep_expired(self): | |
| """Dọn dẹp các entry hết hạn — gọi định kỳ.""" | |
| now = time.time() | |
| with self._lock: | |
| expired = [h for h, e in self._registry.items() | |
| if now >= e["expires_at"]] | |
| for h in expired: | |
| del self._registry[h] | |
| if expired: | |
| print(f"🧹 [TRACKER] Sweep: xóa {len(expired)} entry hết hạn") | |
| def active_count(self) -> int: | |
| self.sweep_expired() | |
| with self._lock: | |
| return len(self._registry) | |
| def status(self) -> dict: | |
| """Trả về trạng thái đầy đủ — dùng cho lệnh admin.""" | |
| now = time.time() | |
| self.sweep_expired() | |
| with self._lock: | |
| result = {} | |
| for h, e in self._registry.items(): | |
| remaining = max(0, int(e["expires_at"] - now)) | |
| result[h[:8]] = { | |
| "model": e["model"], | |
| "tokens_est": e["token_est"], | |
| "chars": e["char_count"], | |
| "remaining": f"{remaining}s", | |
| "alive": remaining > 0, | |
| "cache_name": e["cache_name"], | |
| } | |
| return result | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # CORE CACHE ENGINE v2.0 | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| _GEMINI_BASE = "https://generativelanguage.googleapis.com/v1beta" | |
| _DEFAULT_MODEL = "gemini-3.5-flash" | |
| _TIMEOUT = 45 | |
| class YuiContextCache: | |
| """ | |
| Engine chính quản lý toàn bộ vòng đời Context Cache. | |
| v2.0 improvements: | |
| - Tích hợp SmartPromptTracker để chống chồng prompt | |
| - Tích hợp ProviderCacheLimits để biết giới hạn từng hãng | |
| - Auto-probe Google API khi khởi động | |
| - Namespace cache theo (model, prompt_hash) để không đụng nhau | |
| """ | |
| def __init__(self): | |
| self._lock = threading.Lock() | |
| self._tracker = SmartPromptTracker() | |
| self._limits = ProviderCacheLimits() | |
| # {prompt_hash → cache_name} — ánh xạ nhanh | |
| self._cache_map: Dict[str, str] = {} | |
| # Các model được confirm là hỗ trợ cache | |
| self._capable_models = set(ProviderCacheLimits.get_google()["models"]) | |
| # ── Public API ────────────────────────────────────────────────────────── | |
| def get_or_create( | |
| self, | |
| system_prompt: str, | |
| api_key: str, | |
| model: str = _DEFAULT_MODEL, | |
| ) -> Tuple[Optional[str], str]: | |
| """ | |
| Trả về (cache_name, source). | |
| source: "hit" | "created" | "skip" | "error" | |
| """ | |
| prompt_hash = self._hash(system_prompt) | |
| # 1. Kiểm tra SmartTracker — đây là nguồn truth chính | |
| entry = self._tracker.lookup(prompt_hash) | |
| if entry: | |
| return entry["cache_name"], "hit" | |
| # 2. Kiểm tra model hỗ trợ | |
| if not self._is_capable(model): | |
| print(f"⚪ [CTX-CACHE] Model {model} không hỗ trợ cache") | |
| return None, "skip" | |
| # 3. Probe Google lần đầu nếu chưa làm | |
| ProviderCacheLimits.probe_google(api_key, model) | |
| # 4. Lấy giới hạn từng hãng | |
| limits = ProviderCacheLimits.get_google() | |
| min_tokens = limits["min_tokens"] # 32,768 | |
| ttl = limits["ttl_default"] # 295s | |
| # 5. Đếm tokens | |
| token_count = self._count_tokens(system_prompt, api_key, model) | |
| if token_count < min_tokens: | |
| print(f"⚪ [CTX-CACHE] Prompt chỉ ~{token_count:,} tokens " | |
| f"< {min_tokens:,} (min) → skip cache") | |
| return None, "skip" | |
| # 6. Tạo cache mới | |
| print(f"🔵 [CTX-CACHE] Tạo cache mới " | |
| f"(~{token_count:,} tokens, model={model}, TTL={ttl}s)...") | |
| cache_name = self._create_cache(system_prompt, api_key, model, ttl) | |
| if not cache_name: | |
| return None, "error" | |
| # 7. Đăng ký vào SmartTracker (chống gửi lại lần sau) | |
| self._tracker.register( | |
| prompt_hash = prompt_hash, | |
| cache_name = cache_name, | |
| model = model, | |
| char_count = len(system_prompt), | |
| token_est = token_count, | |
| ttl = ttl, | |
| ) | |
| with self._lock: | |
| self._cache_map[prompt_hash] = cache_name | |
| print(f"✅ [CTX-CACHE] Cache OK: {cache_name}") | |
| return cache_name, "created" | |
| def generate_with_cache( | |
| self, | |
| cache_name: str, | |
| user_message: str, | |
| api_key: str, | |
| model: str = _DEFAULT_MODEL, | |
| history: Optional[list] = None, | |
| max_tokens: int = 8192, | |
| temperature: float = 0.7, | |
| ) -> Optional[str]: | |
| """ | |
| Gọi Gemini với cachedContent reference. | |
| CHỈ gửi user_message, KHÔNG gửi system_prompt → tiết kiệm token. | |
| """ | |
| contents = [] | |
| if history: | |
| for turn in history[-6:]: | |
| role = turn.get("role", "user") | |
| text = turn.get("content", "") | |
| if text: | |
| contents.append({"role": role, "parts": [{"text": text}]}) | |
| contents.append({"role": "user", "parts": [{"text": user_message}]}) | |
| payload = { | |
| "cachedContent": cache_name, | |
| "contents": contents, | |
| "generationConfig": { | |
| "maxOutputTokens": max_tokens, | |
| "temperature": temperature, | |
| "topP": 0.95, | |
| } | |
| } | |
| url = f"{_GEMINI_BASE}/models/{model}:generateContent?key={api_key}" | |
| try: | |
| res = requests.post(url, json=payload, | |
| headers={"Content-Type": "application/json"}, | |
| timeout=_TIMEOUT) | |
| data = res.json() | |
| if "candidates" in data: | |
| text = (data["candidates"][0] | |
| .get("content", {}) | |
| .get("parts", [{}])[0] | |
| .get("text", "").strip()) | |
| usage = data.get("usageMetadata", {}) | |
| cached_tok = usage.get("cachedContentTokenCount", 0) | |
| input_tok = usage.get("promptTokenCount", 0) | |
| out_tok = usage.get("candidatesTokenCount", 0) | |
| saved_pct = int(cached_tok / max(input_tok, 1) * 100) if cached_tok else 0 | |
| print(f"📊 [CTX-CACHE] cached={cached_tok:,} input={input_tok:,} " | |
| f"out={out_tok:,} | tiết kiệm ~{saved_pct}%") | |
| return text or None | |
| err = data.get("error", {}) | |
| code = err.get("code", "?") | |
| msg = err.get("message", "")[:80] | |
| print(f"⚠️ [CTX-CACHE] generate lỗi {code}: {msg}") | |
| # Nếu cache không hợp lệ → xóa tracker để tạo lại | |
| if code in (400, 404): | |
| self._evict_by_cache_name(cache_name) | |
| return None | |
| except Exception as e: | |
| print(f"⚠️ [CTX-CACHE] generate exception: {e}") | |
| return None | |
| def generate_normal( | |
| self, | |
| system_prompt: str, | |
| user_message: str, | |
| api_key: str, | |
| model: str = _DEFAULT_MODEL, | |
| history: Optional[list] = None, | |
| max_tokens: int = 8192, | |
| temperature: float = 0.7, | |
| ) -> Optional[str]: | |
| """Gọi Gemini bình thường khi cache không khả dụng.""" | |
| contents = [ | |
| {"role": "user", "parts": [{"text": system_prompt}]}, | |
| {"role": "model", "parts": [{"text": "OK mình hiểu rồi."}]} | |
| ] | |
| if history: | |
| for turn in history[-4:]: | |
| role = turn.get("role", "user") | |
| text = turn.get("content", "") | |
| if text: | |
| contents.append({"role": role, "parts": [{"text": text}]}) | |
| contents.append({"role": "user", "parts": [{"text": user_message}]}) | |
| payload = { | |
| "contents": contents, | |
| "generationConfig": { | |
| "maxOutputTokens": max_tokens, | |
| "temperature": temperature, | |
| "topP": 0.95, | |
| } | |
| } | |
| url = f"{_GEMINI_BASE}/models/{model}:generateContent?key={api_key}" | |
| try: | |
| res = requests.post(url, json=payload, | |
| headers={"Content-Type": "application/json"}, | |
| timeout=_TIMEOUT) | |
| data = res.json() | |
| if "candidates" in data: | |
| return (data["candidates"][0] | |
| .get("content", {}) | |
| .get("parts", [{}])[0] | |
| .get("text", "").strip()) or None | |
| return None | |
| except Exception as e: | |
| print(f"⚠️ [CTX-CACHE] generate_normal exception: {e}") | |
| return None | |
| def status(self) -> dict: | |
| """Trả về trạng thái đầy đủ cho lệnh admin.""" | |
| tracker_st = self._tracker.status() | |
| limits_summary = ProviderCacheLimits.summary() | |
| return { | |
| "active_caches": tracker_st, | |
| "count": self._tracker.active_count(), | |
| "limits_summary": limits_summary, | |
| } | |
| def add_capable_model(self, model_name: str): | |
| self._capable_models.add(model_name) | |
| def invalidate_prompt(self, system_prompt: str): | |
| h = self._hash(system_prompt) | |
| self._tracker.invalidate(h) | |
| with self._lock: | |
| self._cache_map.pop(h, None) | |
| # ── Private helpers ───────────────────────────────────────────────────── | |
| def _hash(text: str) -> str: | |
| return hashlib.sha256(text.encode("utf-8")).hexdigest() | |
| def _is_capable(self, model: str) -> bool: | |
| if model in self._capable_models: | |
| return True | |
| for cap in self._capable_models: | |
| base = cap.split("-preview")[0].split("-exp")[0] | |
| if model.startswith(base): | |
| return True | |
| return False | |
| def _count_tokens(self, text: str, api_key: str, model: str) -> int: | |
| """Đếm tokens qua API, fallback về ước tính nếu lỗi.""" | |
| url = f"{_GEMINI_BASE}/models/{model}:countTokens?key={api_key}" | |
| try: | |
| res = requests.post(url, json={ | |
| "contents": [{"parts": [{"text": text}]}] | |
| }, timeout=10) | |
| count = res.json().get("totalTokens", 0) | |
| return count if count > 0 else max(1, len(text) // 4) | |
| except Exception: | |
| return max(1, len(text) // 4) | |
| def _create_cache( | |
| self, | |
| system_prompt: str, | |
| api_key: str, | |
| model: str, | |
| ttl: int, | |
| ) -> Optional[str]: | |
| """Tạo cache mới trên Google, trả về cache_name hoặc None.""" | |
| url = f"{_GEMINI_BASE}/cachedContents?key={api_key}" | |
| payload = { | |
| "model": f"models/{model}", | |
| "contents": [ | |
| {"role": "user", | |
| "parts": [{"text": system_prompt}]}, | |
| {"role": "model", | |
| "parts": [{"text": "OK mình hiểu rồi, sẵn sàng hỗ trợ Boss!"}]} | |
| ], | |
| "ttl": f"{ttl}s", | |
| } | |
| try: | |
| res = requests.post(url, json=payload, | |
| headers={"Content-Type": "application/json"}, | |
| timeout=_TIMEOUT) | |
| data = res.json() | |
| name = data.get("name") | |
| if name: | |
| return name | |
| err = data.get("error", {}) | |
| print(f"⚠️ [CTX-CACHE] create lỗi " | |
| f"{err.get('code','?')}: {err.get('message','')[:120]}") | |
| return None | |
| except Exception as e: | |
| print(f"⚠️ [CTX-CACHE] create exception: {e}") | |
| return None | |
| def _evict_by_cache_name(self, cache_name: str): | |
| """Xóa cache_name khỏi tracker khi Google báo không hợp lệ.""" | |
| with self._lock: | |
| to_del = [h for h, n in self._cache_map.items() if n == cache_name] | |
| for h in to_del: | |
| self._cache_map.pop(h, None) | |
| self._tracker.invalidate(h) | |
| if to_del: | |
| print(f"🗑️ [CTX-CACHE] Evicted {len(to_del)} invalid entries") | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # SINGLETON | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| _yui_ctx_cache = YuiContextCache() | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # PUBLIC HELPERS — gọi từ app.py | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def call_gemini_35_with_cache( | |
| system_prompt: str, | |
| user_message: str, | |
| api_key: str, | |
| model: str = _DEFAULT_MODEL, | |
| history: Optional[list] = None, | |
| max_tokens: int = 8192, | |
| temperature: float = 0.7, | |
| ) -> Tuple[Optional[str], str]: | |
| """ | |
| Entry point chính: | |
| 1. get_or_create cache cho system_prompt | |
| 2. Nếu cache OK → chỉ gửi user_message (tiết kiệm 90%+ token) | |
| 3. Nếu skip/fail → gọi thường (gửi cả prompt) | |
| Trả về (response_text, source) | |
| source: "cached" | "normal" | "error" | |
| """ | |
| cache_name, src = _yui_ctx_cache.get_or_create(system_prompt, api_key, model) | |
| if cache_name and src in ("hit", "created"): | |
| resp = _yui_ctx_cache.generate_with_cache( | |
| cache_name, user_message, api_key, model, | |
| history=history, max_tokens=max_tokens, temperature=temperature | |
| ) | |
| if resp: | |
| return resp, "cached" | |
| print("⚠️ [CTX-CACHE] generate_with_cache fail → fallback normal") | |
| resp = _yui_ctx_cache.generate_normal( | |
| system_prompt, user_message, api_key, model, | |
| history=history, max_tokens=max_tokens, temperature=temperature | |
| ) | |
| return (resp, "normal") if resp else (None, "error") | |
| def get_cache_status() -> dict: | |
| """Trả về trạng thái đầy đủ — dùng cho lệnh admin.""" | |
| return _yui_ctx_cache.status() | |
| def get_provider_limits_summary() -> str: | |
| """Trả về bảng giới hạn cache của tất cả provider — admin command.""" | |
| return ProviderCacheLimits.summary() | |
| def invalidate_cache(system_prompt: str): | |
| """Xóa cache khi Boss thay đổi system prompt.""" | |
| _yui_ctx_cache.invalidate_prompt(system_prompt) | |
| def update_capable_models(models: List[str]): | |
| """Hunter gọi sau khi hunt xong để đăng ký model mới.""" | |
| for m in models: | |
| _yui_ctx_cache.add_capable_model(m) | |
| def probe_google_cache(api_key: str, model: str = "gemini-2.5-flash") -> bool: | |
| """Probe Google API để kiểm tra giới hạn cache thực tế.""" | |
| return ProviderCacheLimits.probe_google(api_key, model) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # SELF-TEST | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| if __name__ == "__main__": | |
| print("=== YuiContextCache v2.0 Self-Test ===\n") | |
| # Test ProviderCacheLimits | |
| assert ProviderCacheLimits.is_supported("google"), "Google phải supported" | |
| assert not ProviderCacheLimits.is_supported("mistral"), "Mistral không supported" | |
| assert not ProviderCacheLimits.is_supported("cerebras"), "Cerebras không supported" | |
| assert ProviderCacheLimits.min_tokens("google") == 32_768 | |
| assert ProviderCacheLimits.ttl("google") == 295 | |
| print("✅ ProviderCacheLimits OK") | |
| print(ProviderCacheLimits.summary()) | |
| # Test SmartPromptTracker | |
| tracker = SmartPromptTracker() | |
| h = hashlib.sha256(b"test prompt").hexdigest() | |
| assert tracker.lookup(h) is None, "Chưa đăng ký → phải None" | |
| tracker.register(h, "cachedContents/abc123", "gemini-3.5-flash", | |
| char_count=50000, token_est=35000, ttl=295) | |
| entry = tracker.lookup(h) | |
| assert entry is not None, "Sau đăng ký phải tìm được" | |
| assert entry["cache_name"] == "cachedContents/abc123" | |
| assert tracker.is_cached(h) | |
| assert tracker.active_count() == 1 | |
| tracker.invalidate(h) | |
| assert tracker.lookup(h) is None, "Sau invalidate phải None" | |
| print("\n✅ SmartPromptTracker OK") | |
| # Test YuiContextCache hash consistency | |
| cache = YuiContextCache() | |
| h1 = cache._hash("hello") | |
| h2 = cache._hash("hello") | |
| h3 = cache._hash("world") | |
| assert h1 == h2 | |
| assert h1 != h3 | |
| print("✅ Hash stable") | |
| # Test capable model check | |
| assert cache._is_capable("gemini-3.5-flash") | |
| assert cache._is_capable("gemini-2.5-flash") | |
| assert not cache._is_capable("command-r-plus") | |
| print("✅ Capable model check OK") | |
| # Test token fallback | |
| long_text = "A" * 200_000 | |
| est = max(1, len(long_text) // 4) | |
| assert est >= 32_768 | |
| print(f"✅ Token estimate fallback: {est:,} (>= 32,768)") | |
| print("\n✅ Tất cả tests passed! yui_context_cache v2.0 sẵn sàng deploy") | |