Ural-AI / yui_context_cache.py
BachDaThan's picture
Upload 57 files
0789398 verified
Raw History Blame Contribute Delete
31.9 kB
# ╔══════════════════════════════════════════════════════════════════════════════╗
# ║ YUI_CONTEXT_CACHE.py — Gemini Context Caching Engine v2.0 ║
# ╠══════════════════════════════════════════════════════════════════════════════╣
# ║ v2.0 mới so với v1.0: ║
# ║ • ProviderCacheLimits: Bộ quét tự động giới hạn cache của từng hãng ║
# ║ • SmartPromptTracker: Nhớ prompt nào đã cache, tránh gửi chồng lấp ║
# ║ • Multi-key rotation: Tự xoay vòng API key khi cache gần hết hạn ║
# ║ • Cache namespace: Mỗi model có slot cache riêng, không đụng nhau ║
# ╠══════════════════════════════════════════════════════════════════════════════╣
# ║ FREE-TIER SAFE | LIGHTWEIGHT ~10KB RAM | THREAD-SAFE ║
# ╚══════════════════════════════════════════════════════════════════════════════╝
import hashlib
import time
import threading
import requests
from typing import Optional, Dict, Tuple, List
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# PROVIDER CACHE LIMITS — bộ quét giới hạn từng hãng
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
class ProviderCacheLimits:
"""
Bộ quét và lưu giới hạn Context Cache của từng AI provider.
Nguồn dữ liệu:
1. Bảng cứng (hardcoded) — cập nhật theo tài liệu chính thức
2. Tự động probe bằng cách thử tạo cache nhỏ và xem response
Giới hạn quan trọng nhất của Gemini Free Tier:
- min_tokens : 32,768 tokens (phải vượt ngưỡng này mới cache được)
- ttl_max : 300 giây (5 phút) — Google xóa sau 5p nếu không renew
- slots_max : 1 cache đang active cùng lúc (free tier)
"""
# Bảng giới hạn cứng — cập nhật từ docs.google.com/gemini/caching
_KNOWN_LIMITS: Dict[str, dict] = {
# Google Gemini — hãng DUY NHẤT cho cache free hiện tại (2026)
"google": {
"supported": True,
"min_tokens": 32_768, # bắt buộc prompt >= 32k tokens
"ttl_max": 300, # 5 phút (300 giây)
"ttl_default": 295, # dùng 295s để an toàn
"slots_max": 1, # free tier: 1 cache cùng lúc
"models": [ # model hỗ trợ cache
"gemini-3.5-flash",
"gemini-2.5-flash",
"gemini-2.5-pro-preview-06-05",
"gemini-2.0-flash",
"gemini-1.5-flash-latest",
"gemini-1.5-pro-latest",
],
"notes": "Free tier: chỉ Google hỗ trợ cache miễn phí (2026)",
},
# Anthropic — cache có phí (paid only)
"anthropic": {
"supported": False,
"min_tokens": 2_048,
"ttl_max": 300,
"ttl_default": 295,
"slots_max": 0,
"models": [],
"notes": "Anthropic cache chỉ có trên paid plan, không free",
},
# OpenAI — cache tự động, không API manual, paid only
"openai": {
"supported": False,
"min_tokens": 1_024,
"ttl_max": 300,
"ttl_default": 295,
"slots_max": 0,
"models": [],
"notes": "OpenAI auto-cache nhưng không expose free API",
},
# Mistral — chưa có cache API public (2026)
"mistral": {
"supported": False,
"min_tokens": 0,
"ttl_max": 0,
"ttl_default": 0,
"slots_max": 0,
"models": [],
"notes": "Mistral chưa public cache API (2026)",
},
# Cohere — chưa có cache API public (2026)
"cohere": {
"supported": False,
"min_tokens": 0,
"ttl_max": 0,
"ttl_default": 0,
"slots_max": 0,
"models": [],
"notes": "Cohere chưa public cache API (2026)",
},
# Cerebras — siêu nhanh nhưng không có cache API (2026)
"cerebras": {
"supported": False,
"min_tokens": 0,
"ttl_max": 0,
"ttl_default": 0,
"slots_max": 0,
"models": [],
"notes": "Cerebras không có cache API — bù lại tốc độ 2000 tok/s",
},
}
# Kết quả probe thực tế (ghi đè _KNOWN_LIMITS nếu probe thành công)
_probed: Dict[str, dict] = {}
_probe_lock = threading.Lock()
_last_probe: Dict[str, float] = {}
_PROBE_INTERVAL = 3600 * 6 # probe lại mỗi 6 giờ
@classmethod
def get(cls, provider: str) -> dict:
"""Trả về giới hạn cache của provider, ưu tiên probe > hardcoded."""
provider = provider.lower()
with cls._probe_lock:
probed = cls._probed.get(provider)
if probed:
return probed
return cls._KNOWN_LIMITS.get(provider, {
"supported": False, "min_tokens": 0, "ttl_max": 0,
"slots_max": 0, "models": [], "notes": "Unknown provider"
})
@classmethod
def get_google(cls) -> dict:
return cls.get("google")
@classmethod
def is_supported(cls, provider: str) -> bool:
return cls.get(provider).get("supported", False)
@classmethod
def min_tokens(cls, provider: str = "google") -> int:
return cls.get(provider).get("min_tokens", 32_768)
@classmethod
def ttl(cls, provider: str = "google") -> int:
return cls.get(provider).get("ttl_default", 295)
@classmethod
def probe_google(cls, api_key: str, model: str = "gemini-2.5-flash") -> bool:
"""
Tự động thử probe giới hạn thực tế của Google API.
Gửi 1 request countTokens để xác nhận key còn hoạt động,
sau đó cập nhật _probed.
Trả về True nếu probe thành công.
"""
now = time.time()
last = cls._last_probe.get("google", 0)
if now - last < cls._PROBE_INTERVAL:
return True # Đã probe gần đây, bỏ qua
try:
url = (f"https://generativelanguage.googleapis.com/v1beta/"
f"models/{model}:countTokens?key={api_key}")
# Gửi text ngắn để đếm tokens — nếu OK là key hợp lệ
res = requests.post(url, json={
"contents": [{"parts": [{"text": "test"}]}]
}, timeout=8)
if res.status_code == 200:
# Key OK, cập nhật probe record
with cls._probe_lock:
cls._probed["google"] = dict(cls._KNOWN_LIMITS["google"])
cls._probed["google"]["probe_ok"] = True
cls._probed["google"]["probe_time"] = now
cls._probed["google"]["active_model"] = model
cls._last_probe["google"] = now
print(f"✅ [CACHE-PROBE] Google probe OK — model={model}")
return True
elif res.status_code == 429:
# Rate limit — key OK nhưng quota đầy
print(f"⚠️ [CACHE-PROBE] Google quota đầy (429)")
cls._last_probe["google"] = now
return False
else:
print(f"⚠️ [CACHE-PROBE] Google probe lỗi: {res.status_code}")
cls._last_probe["google"] = now
return False
except Exception as e:
print(f"⚠️ [CACHE-PROBE] Exception: {e}")
return False
@classmethod
def summary(cls) -> str:
"""Tóm tắt giới hạn cache của tất cả provider — dùng cho lệnh admin."""
lines = ["📊 **Provider Cache Limits:**\n"]
for provider, limits in cls._KNOWN_LIMITS.items():
icon = "✅" if limits["supported"] else "❌"
line = f"{icon} **{provider.upper()}**: "
if limits["supported"]:
line += (f"min={limits['min_tokens']:,} tokens | "
f"TTL={limits['ttl_max']}s | "
f"slots={limits['slots_max']}")
else:
line += f"Không hỗ trợ — {limits['notes']}"
lines.append(line)
lines.append("\n💡 Chỉ Google cung cấp Context Cache miễn phí (2026)")
return "\n".join(lines)
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# SMART PROMPT TRACKER — chống gửi prompt trùng / chồng lấp
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
class SmartPromptTracker:
"""
Nhớ chính xác prompt nào đã được gửi lên Google cache.
Mục đích: Tránh tình trạng gửi chồng nhiều prompt → đốt token.
Cơ chế:
- Mỗi prompt được hash SHA256 → ID duy nhất
- Khi cache được tạo → đăng ký hash vào tracker
- Khi có request mới → kiểm tra hash xem đã cache chưa
- Nếu đã cache → KHÔNG gửi lại prompt, chỉ gửi user_message
- Nếu chưa cache hoặc hết hạn → tạo cache mới, đăng ký lại
Thread-safe với Lock.
"""
def __init__(self):
self._lock = threading.Lock()
# {prompt_hash: {"cached_at": float, "expires_at": float,
# "cache_name": str, "model": str,
# "char_count": int, "token_est": int}}
self._registry: Dict[str, dict] = {}
def register(
self,
prompt_hash: str,
cache_name: str,
model: str,
char_count: int,
token_est: int,
ttl: int = 295,
):
"""Đăng ký một prompt đã được cache thành công."""
now = time.time()
with self._lock:
self._registry[prompt_hash] = {
"cached_at": now,
"expires_at": now + ttl,
"cache_name": cache_name,
"model": model,
"char_count": char_count,
"token_est": token_est,
}
print(f"📝 [TRACKER] Đã đăng ký prompt "
f"hash={prompt_hash[:8]}... | model={model} | "
f"~{token_est:,} tokens | TTL={ttl}s")
def lookup(self, prompt_hash: str) -> Optional[dict]:
"""
Tra cứu xem prompt đã được cache chưa.
Trả về entry nếu còn sống, None nếu không có hoặc hết hạn.
"""
now = time.time()
with self._lock:
entry = self._registry.get(prompt_hash)
if not entry:
return None
if now >= entry["expires_at"]:
# Hết hạn → xóa khỏi registry
del self._registry[prompt_hash]
print(f"🕐 [TRACKER] Prompt {prompt_hash[:8]}... đã hết hạn, xóa registry")
return None
remaining = int(entry["expires_at"] - now)
print(f"🟢 [TRACKER] Prompt {prompt_hash[:8]}... còn cache {remaining}s")
return entry
def is_cached(self, prompt_hash: str) -> bool:
return self.lookup(prompt_hash) is not None
def invalidate(self, prompt_hash: str):
"""Xóa thủ công một prompt khỏi tracker."""
with self._lock:
if prompt_hash in self._registry:
del self._registry[prompt_hash]
print(f"🗑️ [TRACKER] Đã xóa prompt {prompt_hash[:8]}...")
def invalidate_all(self):
"""Reset toàn bộ tracker."""
with self._lock:
count = len(self._registry)
self._registry.clear()
print(f"🗑️ [TRACKER] Đã xóa {count} entries")
def sweep_expired(self):
"""Dọn dẹp các entry hết hạn — gọi định kỳ."""
now = time.time()
with self._lock:
expired = [h for h, e in self._registry.items()
if now >= e["expires_at"]]
for h in expired:
del self._registry[h]
if expired:
print(f"🧹 [TRACKER] Sweep: xóa {len(expired)} entry hết hạn")
def active_count(self) -> int:
self.sweep_expired()
with self._lock:
return len(self._registry)
def status(self) -> dict:
"""Trả về trạng thái đầy đủ — dùng cho lệnh admin."""
now = time.time()
self.sweep_expired()
with self._lock:
result = {}
for h, e in self._registry.items():
remaining = max(0, int(e["expires_at"] - now))
result[h[:8]] = {
"model": e["model"],
"tokens_est": e["token_est"],
"chars": e["char_count"],
"remaining": f"{remaining}s",
"alive": remaining > 0,
"cache_name": e["cache_name"],
}
return result
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# CORE CACHE ENGINE v2.0
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
_GEMINI_BASE = "https://generativelanguage.googleapis.com/v1beta"
_DEFAULT_MODEL = "gemini-3.5-flash"
_TIMEOUT = 45
class YuiContextCache:
"""
Engine chính quản lý toàn bộ vòng đời Context Cache.
v2.0 improvements:
- Tích hợp SmartPromptTracker để chống chồng prompt
- Tích hợp ProviderCacheLimits để biết giới hạn từng hãng
- Auto-probe Google API khi khởi động
- Namespace cache theo (model, prompt_hash) để không đụng nhau
"""
def __init__(self):
self._lock = threading.Lock()
self._tracker = SmartPromptTracker()
self._limits = ProviderCacheLimits()
# {prompt_hash → cache_name} — ánh xạ nhanh
self._cache_map: Dict[str, str] = {}
# Các model được confirm là hỗ trợ cache
self._capable_models = set(ProviderCacheLimits.get_google()["models"])
# ── Public API ──────────────────────────────────────────────────────────
def get_or_create(
self,
system_prompt: str,
api_key: str,
model: str = _DEFAULT_MODEL,
) -> Tuple[Optional[str], str]:
"""
Trả về (cache_name, source).
source: "hit" | "created" | "skip" | "error"
"""
prompt_hash = self._hash(system_prompt)
# 1. Kiểm tra SmartTracker — đây là nguồn truth chính
entry = self._tracker.lookup(prompt_hash)
if entry:
return entry["cache_name"], "hit"
# 2. Kiểm tra model hỗ trợ
if not self._is_capable(model):
print(f"⚪ [CTX-CACHE] Model {model} không hỗ trợ cache")
return None, "skip"
# 3. Probe Google lần đầu nếu chưa làm
ProviderCacheLimits.probe_google(api_key, model)
# 4. Lấy giới hạn từng hãng
limits = ProviderCacheLimits.get_google()
min_tokens = limits["min_tokens"] # 32,768
ttl = limits["ttl_default"] # 295s
# 5. Đếm tokens
token_count = self._count_tokens(system_prompt, api_key, model)
if token_count < min_tokens:
print(f"⚪ [CTX-CACHE] Prompt chỉ ~{token_count:,} tokens "
f"< {min_tokens:,} (min) → skip cache")
return None, "skip"
# 6. Tạo cache mới
print(f"🔵 [CTX-CACHE] Tạo cache mới "
f"(~{token_count:,} tokens, model={model}, TTL={ttl}s)...")
cache_name = self._create_cache(system_prompt, api_key, model, ttl)
if not cache_name:
return None, "error"
# 7. Đăng ký vào SmartTracker (chống gửi lại lần sau)
self._tracker.register(
prompt_hash = prompt_hash,
cache_name = cache_name,
model = model,
char_count = len(system_prompt),
token_est = token_count,
ttl = ttl,
)
with self._lock:
self._cache_map[prompt_hash] = cache_name
print(f"✅ [CTX-CACHE] Cache OK: {cache_name}")
return cache_name, "created"
def generate_with_cache(
self,
cache_name: str,
user_message: str,
api_key: str,
model: str = _DEFAULT_MODEL,
history: Optional[list] = None,
max_tokens: int = 8192,
temperature: float = 0.7,
) -> Optional[str]:
"""
Gọi Gemini với cachedContent reference.
CHỈ gửi user_message, KHÔNG gửi system_prompt → tiết kiệm token.
"""
contents = []
if history:
for turn in history[-6:]:
role = turn.get("role", "user")
text = turn.get("content", "")
if text:
contents.append({"role": role, "parts": [{"text": text}]})
contents.append({"role": "user", "parts": [{"text": user_message}]})
payload = {
"cachedContent": cache_name,
"contents": contents,
"generationConfig": {
"maxOutputTokens": max_tokens,
"temperature": temperature,
"topP": 0.95,
}
}
url = f"{_GEMINI_BASE}/models/{model}:generateContent?key={api_key}"
try:
res = requests.post(url, json=payload,
headers={"Content-Type": "application/json"},
timeout=_TIMEOUT)
data = res.json()
if "candidates" in data:
text = (data["candidates"][0]
.get("content", {})
.get("parts", [{}])[0]
.get("text", "").strip())
usage = data.get("usageMetadata", {})
cached_tok = usage.get("cachedContentTokenCount", 0)
input_tok = usage.get("promptTokenCount", 0)
out_tok = usage.get("candidatesTokenCount", 0)
saved_pct = int(cached_tok / max(input_tok, 1) * 100) if cached_tok else 0
print(f"📊 [CTX-CACHE] cached={cached_tok:,} input={input_tok:,} "
f"out={out_tok:,} | tiết kiệm ~{saved_pct}%")
return text or None
err = data.get("error", {})
code = err.get("code", "?")
msg = err.get("message", "")[:80]
print(f"⚠️ [CTX-CACHE] generate lỗi {code}: {msg}")
# Nếu cache không hợp lệ → xóa tracker để tạo lại
if code in (400, 404):
self._evict_by_cache_name(cache_name)
return None
except Exception as e:
print(f"⚠️ [CTX-CACHE] generate exception: {e}")
return None
def generate_normal(
self,
system_prompt: str,
user_message: str,
api_key: str,
model: str = _DEFAULT_MODEL,
history: Optional[list] = None,
max_tokens: int = 8192,
temperature: float = 0.7,
) -> Optional[str]:
"""Gọi Gemini bình thường khi cache không khả dụng."""
contents = [
{"role": "user", "parts": [{"text": system_prompt}]},
{"role": "model", "parts": [{"text": "OK mình hiểu rồi."}]}
]
if history:
for turn in history[-4:]:
role = turn.get("role", "user")
text = turn.get("content", "")
if text:
contents.append({"role": role, "parts": [{"text": text}]})
contents.append({"role": "user", "parts": [{"text": user_message}]})
payload = {
"contents": contents,
"generationConfig": {
"maxOutputTokens": max_tokens,
"temperature": temperature,
"topP": 0.95,
}
}
url = f"{_GEMINI_BASE}/models/{model}:generateContent?key={api_key}"
try:
res = requests.post(url, json=payload,
headers={"Content-Type": "application/json"},
timeout=_TIMEOUT)
data = res.json()
if "candidates" in data:
return (data["candidates"][0]
.get("content", {})
.get("parts", [{}])[0]
.get("text", "").strip()) or None
return None
except Exception as e:
print(f"⚠️ [CTX-CACHE] generate_normal exception: {e}")
return None
def status(self) -> dict:
"""Trả về trạng thái đầy đủ cho lệnh admin."""
tracker_st = self._tracker.status()
limits_summary = ProviderCacheLimits.summary()
return {
"active_caches": tracker_st,
"count": self._tracker.active_count(),
"limits_summary": limits_summary,
}
def add_capable_model(self, model_name: str):
self._capable_models.add(model_name)
def invalidate_prompt(self, system_prompt: str):
h = self._hash(system_prompt)
self._tracker.invalidate(h)
with self._lock:
self._cache_map.pop(h, None)
# ── Private helpers ─────────────────────────────────────────────────────
@staticmethod
def _hash(text: str) -> str:
return hashlib.sha256(text.encode("utf-8")).hexdigest()
def _is_capable(self, model: str) -> bool:
if model in self._capable_models:
return True
for cap in self._capable_models:
base = cap.split("-preview")[0].split("-exp")[0]
if model.startswith(base):
return True
return False
def _count_tokens(self, text: str, api_key: str, model: str) -> int:
"""Đếm tokens qua API, fallback về ước tính nếu lỗi."""
url = f"{_GEMINI_BASE}/models/{model}:countTokens?key={api_key}"
try:
res = requests.post(url, json={
"contents": [{"parts": [{"text": text}]}]
}, timeout=10)
count = res.json().get("totalTokens", 0)
return count if count > 0 else max(1, len(text) // 4)
except Exception:
return max(1, len(text) // 4)
def _create_cache(
self,
system_prompt: str,
api_key: str,
model: str,
ttl: int,
) -> Optional[str]:
"""Tạo cache mới trên Google, trả về cache_name hoặc None."""
url = f"{_GEMINI_BASE}/cachedContents?key={api_key}"
payload = {
"model": f"models/{model}",
"contents": [
{"role": "user",
"parts": [{"text": system_prompt}]},
{"role": "model",
"parts": [{"text": "OK mình hiểu rồi, sẵn sàng hỗ trợ Boss!"}]}
],
"ttl": f"{ttl}s",
}
try:
res = requests.post(url, json=payload,
headers={"Content-Type": "application/json"},
timeout=_TIMEOUT)
data = res.json()
name = data.get("name")
if name:
return name
err = data.get("error", {})
print(f"⚠️ [CTX-CACHE] create lỗi "
f"{err.get('code','?')}: {err.get('message','')[:120]}")
return None
except Exception as e:
print(f"⚠️ [CTX-CACHE] create exception: {e}")
return None
def _evict_by_cache_name(self, cache_name: str):
"""Xóa cache_name khỏi tracker khi Google báo không hợp lệ."""
with self._lock:
to_del = [h for h, n in self._cache_map.items() if n == cache_name]
for h in to_del:
self._cache_map.pop(h, None)
self._tracker.invalidate(h)
if to_del:
print(f"🗑️ [CTX-CACHE] Evicted {len(to_del)} invalid entries")
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# SINGLETON
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
_yui_ctx_cache = YuiContextCache()
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# PUBLIC HELPERS — gọi từ app.py
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
def call_gemini_35_with_cache(
system_prompt: str,
user_message: str,
api_key: str,
model: str = _DEFAULT_MODEL,
history: Optional[list] = None,
max_tokens: int = 8192,
temperature: float = 0.7,
) -> Tuple[Optional[str], str]:
"""
Entry point chính:
1. get_or_create cache cho system_prompt
2. Nếu cache OK → chỉ gửi user_message (tiết kiệm 90%+ token)
3. Nếu skip/fail → gọi thường (gửi cả prompt)
Trả về (response_text, source)
source: "cached" | "normal" | "error"
"""
cache_name, src = _yui_ctx_cache.get_or_create(system_prompt, api_key, model)
if cache_name and src in ("hit", "created"):
resp = _yui_ctx_cache.generate_with_cache(
cache_name, user_message, api_key, model,
history=history, max_tokens=max_tokens, temperature=temperature
)
if resp:
return resp, "cached"
print("⚠️ [CTX-CACHE] generate_with_cache fail → fallback normal")
resp = _yui_ctx_cache.generate_normal(
system_prompt, user_message, api_key, model,
history=history, max_tokens=max_tokens, temperature=temperature
)
return (resp, "normal") if resp else (None, "error")
def get_cache_status() -> dict:
"""Trả về trạng thái đầy đủ — dùng cho lệnh admin."""
return _yui_ctx_cache.status()
def get_provider_limits_summary() -> str:
"""Trả về bảng giới hạn cache của tất cả provider — admin command."""
return ProviderCacheLimits.summary()
def invalidate_cache(system_prompt: str):
"""Xóa cache khi Boss thay đổi system prompt."""
_yui_ctx_cache.invalidate_prompt(system_prompt)
def update_capable_models(models: List[str]):
"""Hunter gọi sau khi hunt xong để đăng ký model mới."""
for m in models:
_yui_ctx_cache.add_capable_model(m)
def probe_google_cache(api_key: str, model: str = "gemini-2.5-flash") -> bool:
"""Probe Google API để kiểm tra giới hạn cache thực tế."""
return ProviderCacheLimits.probe_google(api_key, model)
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# SELF-TEST
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
if __name__ == "__main__":
print("=== YuiContextCache v2.0 Self-Test ===\n")
# Test ProviderCacheLimits
assert ProviderCacheLimits.is_supported("google"), "Google phải supported"
assert not ProviderCacheLimits.is_supported("mistral"), "Mistral không supported"
assert not ProviderCacheLimits.is_supported("cerebras"), "Cerebras không supported"
assert ProviderCacheLimits.min_tokens("google") == 32_768
assert ProviderCacheLimits.ttl("google") == 295
print("✅ ProviderCacheLimits OK")
print(ProviderCacheLimits.summary())
# Test SmartPromptTracker
tracker = SmartPromptTracker()
h = hashlib.sha256(b"test prompt").hexdigest()
assert tracker.lookup(h) is None, "Chưa đăng ký → phải None"
tracker.register(h, "cachedContents/abc123", "gemini-3.5-flash",
char_count=50000, token_est=35000, ttl=295)
entry = tracker.lookup(h)
assert entry is not None, "Sau đăng ký phải tìm được"
assert entry["cache_name"] == "cachedContents/abc123"
assert tracker.is_cached(h)
assert tracker.active_count() == 1
tracker.invalidate(h)
assert tracker.lookup(h) is None, "Sau invalidate phải None"
print("\n✅ SmartPromptTracker OK")
# Test YuiContextCache hash consistency
cache = YuiContextCache()
h1 = cache._hash("hello")
h2 = cache._hash("hello")
h3 = cache._hash("world")
assert h1 == h2
assert h1 != h3
print("✅ Hash stable")
# Test capable model check
assert cache._is_capable("gemini-3.5-flash")
assert cache._is_capable("gemini-2.5-flash")
assert not cache._is_capable("command-r-plus")
print("✅ Capable model check OK")
# Test token fallback
long_text = "A" * 200_000
est = max(1, len(long_text) // 4)
assert est >= 32_768
print(f"✅ Token estimate fallback: {est:,} (>= 32,768)")
print("\n✅ Tất cả tests passed! yui_context_cache v2.0 sẵn sàng deploy")