Spaces:
Paused
fix: Round 14 — 28 Tier 1 + Tier 2 bugs fixed across 17 source files
Browse filesAnti-hallucination integrity (T1):
- #100: Detect verification LLM system failure (all calls fail → refuse)
- #101: Fix validate_citations SQL/web source matching (dead code → working)
- #102: Citation count uses chunks (citable units) not unique files
- #103: Clamp max_iterations to min 1 (can't bypass via iterations=0)
Pre-release quality (T2 HIGH):
- #104: Add LIKE ESCAPE in all SQL fallback paths
- #105: Gemini safety block returns "" (was bypassing stack)
- #106: Cap reasoning model budget at 128000
- #107: Handle non-numeric temperature/max_tokens gracefully
- #108: Fix clarification history duplication
Pre-release quality (T2 MEDIUM):
- #109: Anthropic temperature clamp to 1.0
- #110: SQL query 10-second timeout via progress handler
- #111: Rich markup escape for user/LLM content
- #112: All 6 readers wrapped in try/except (corrupt files don't abort)
- #113: Collection cache invalidation after re-ingestion
- #114: Web UI message history capped at 200
- #115: Setup wizard config overwrite confirmation
- #116: Prompt injection defense (delimiter sanitization)
- #117: References section regex precision
- #118: Correction prompt truncation (1500 chars)
- #119: Initial generate() wrapped in try/except
- #120: Reasoning model detection (exact match + prefix)
- #121: make_fuzzy_query regex precision
- #122: REPLACE removed from dangerous keywords (function is read-only)
- #123: Non-dict YAML config rejection
- #124: Web UI model cache invalidation on provider change
- #125: Web UI uses fast file-based KB brief (was slow LLM call)
- #126: Re-ingestion progress message
- #127: Empty API key warning improved
Cross-checked by regression checker + design guardian agents.
Design score: 8.5 → 9.1/10. Tests: 174/174 passing.
Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
- app_cli.py +13 -7
- app_web.py +17 -3
- docs/BUGFIX_LOG.md +205 -6
- setup.py +14 -4
- src/config_loader.py +6 -0
- src/llm/__init__.py +11 -2
- src/llm/gemini.py +2 -2
- src/llm/openai.py +2 -2
- src/prompts.py +4 -1
- src/readers/docx.py +21 -17
- src/readers/excel.py +9 -5
- src/readers/pdf.py +12 -8
- src/readers/rdata.py +33 -29
- src/readers/spss.py +7 -3
- src/readers/stata.py +7 -3
- src/retriever.py +9 -2
- src/sql_retriever.py +11 -3
- src/verifier.py +79 -30
- tests/test_retriever.py +6 -6
|
@@ -3,6 +3,7 @@
|
|
| 3 |
import sys
|
| 4 |
|
| 5 |
from rich.console import Console
|
|
|
|
| 6 |
from rich.panel import Panel
|
| 7 |
from rich.markdown import Markdown
|
| 8 |
from rich.text import Text
|
|
@@ -11,7 +12,7 @@ from src.config_loader import load_config, get_api_key
|
|
| 11 |
from src.ingest import ingest_documents
|
| 12 |
from src.kb_meta import load_kb_meta_brief
|
| 13 |
from src.query_engine import understand_query
|
| 14 |
-
from src.retriever import retrieve
|
| 15 |
from src.verifier import verify_and_respond
|
| 16 |
from src.llm import list_models
|
| 17 |
|
|
@@ -256,9 +257,10 @@ def main() -> None:
|
|
| 256 |
with console.status("Ingesting documents..."):
|
| 257 |
try:
|
| 258 |
count = ingest_documents(cfg)
|
|
|
|
| 259 |
console.print(f"[green]Ingestion complete: {count} chunks.[/green]")
|
| 260 |
except Exception as e:
|
| 261 |
-
console.print(f"[red]Ingestion error: {e}[/red]")
|
| 262 |
continue
|
| 263 |
|
| 264 |
if result == "__MODEL__":
|
|
@@ -301,7 +303,7 @@ def main() -> None:
|
|
| 301 |
state.get("conversation_history", []),
|
| 302 |
)
|
| 303 |
except Exception as e:
|
| 304 |
-
console.print(f"[yellow]Query understanding failed, using raw query: {e}[/yellow]")
|
| 305 |
qu_result = {"action": "search", "search_query": user_input, "display_query": user_input, "original_query": user_input, "route": "vector", "sql_query": None}
|
| 306 |
|
| 307 |
# Handle clarification
|
|
@@ -344,7 +346,7 @@ def main() -> None:
|
|
| 344 |
|
| 345 |
# Show reformulated query if different from original
|
| 346 |
if search_query != user_input:
|
| 347 |
-
console.print(f"[dim]Searching for: \"{search_query}\"[/dim]")
|
| 348 |
if route in ("sql", "both"):
|
| 349 |
console.print(f"[dim]Using SQL query for structured data[/dim]")
|
| 350 |
|
|
@@ -353,7 +355,7 @@ def main() -> None:
|
|
| 353 |
try:
|
| 354 |
retrieval_result = retrieve(search_query, effective_cfg, route=route, sql_query=sql_query)
|
| 355 |
except Exception as e:
|
| 356 |
-
console.print(f"[red]Retrieval error: {e}[/red]")
|
| 357 |
continue
|
| 358 |
|
| 359 |
state["last_retrieval"] = retrieval_result
|
|
@@ -366,13 +368,17 @@ def main() -> None:
|
|
| 366 |
original_query=user_input,
|
| 367 |
)
|
| 368 |
except Exception as e:
|
| 369 |
-
console.print(f"[red]Generation error: {e}[/red]")
|
| 370 |
continue
|
| 371 |
|
| 372 |
# ── Update conversation history ─────────────────────────────────
|
| 373 |
history = state.get("conversation_history", [])
|
| 374 |
max_history = qu_cfg.get("max_history", 6)
|
| 375 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 376 |
history.append({"role": "assistant", "content": result.get("response", "")})
|
| 377 |
state["conversation_history"] = history[-max_history:]
|
| 378 |
|
|
|
|
| 3 |
import sys
|
| 4 |
|
| 5 |
from rich.console import Console
|
| 6 |
+
from rich.markup import escape as rich_escape
|
| 7 |
from rich.panel import Panel
|
| 8 |
from rich.markdown import Markdown
|
| 9 |
from rich.text import Text
|
|
|
|
| 12 |
from src.ingest import ingest_documents
|
| 13 |
from src.kb_meta import load_kb_meta_brief
|
| 14 |
from src.query_engine import understand_query
|
| 15 |
+
from src.retriever import clear_collection_cache, retrieve
|
| 16 |
from src.verifier import verify_and_respond
|
| 17 |
from src.llm import list_models
|
| 18 |
|
|
|
|
| 257 |
with console.status("Ingesting documents..."):
|
| 258 |
try:
|
| 259 |
count = ingest_documents(cfg)
|
| 260 |
+
clear_collection_cache()
|
| 261 |
console.print(f"[green]Ingestion complete: {count} chunks.[/green]")
|
| 262 |
except Exception as e:
|
| 263 |
+
console.print(f"[red]Ingestion error: {rich_escape(str(e))}[/red]")
|
| 264 |
continue
|
| 265 |
|
| 266 |
if result == "__MODEL__":
|
|
|
|
| 303 |
state.get("conversation_history", []),
|
| 304 |
)
|
| 305 |
except Exception as e:
|
| 306 |
+
console.print(f"[yellow]Query understanding failed, using raw query: {rich_escape(str(e))}[/yellow]")
|
| 307 |
qu_result = {"action": "search", "search_query": user_input, "display_query": user_input, "original_query": user_input, "route": "vector", "sql_query": None}
|
| 308 |
|
| 309 |
# Handle clarification
|
|
|
|
| 346 |
|
| 347 |
# Show reformulated query if different from original
|
| 348 |
if search_query != user_input:
|
| 349 |
+
console.print(f"[dim]Searching for: \"{rich_escape(search_query)}\"[/dim]")
|
| 350 |
if route in ("sql", "both"):
|
| 351 |
console.print(f"[dim]Using SQL query for structured data[/dim]")
|
| 352 |
|
|
|
|
| 355 |
try:
|
| 356 |
retrieval_result = retrieve(search_query, effective_cfg, route=route, sql_query=sql_query)
|
| 357 |
except Exception as e:
|
| 358 |
+
console.print(f"[red]Retrieval error: {rich_escape(str(e))}[/red]")
|
| 359 |
continue
|
| 360 |
|
| 361 |
state["last_retrieval"] = retrieval_result
|
|
|
|
| 368 |
original_query=user_input,
|
| 369 |
)
|
| 370 |
except Exception as e:
|
| 371 |
+
console.print(f"[red]Generation error: {rich_escape(str(e))}[/red]")
|
| 372 |
continue
|
| 373 |
|
| 374 |
# ── Update conversation history ─────────────────────────────────
|
| 375 |
history = state.get("conversation_history", [])
|
| 376 |
max_history = qu_cfg.get("max_history", 6)
|
| 377 |
+
# Use display_query (post-QU) as the user message for history,
|
| 378 |
+
# since it captures clarification context and avoids duplicating
|
| 379 |
+
# the raw user_input that clarification already added.
|
| 380 |
+
effective_user_msg = display_query if display_query != user_input else user_input
|
| 381 |
+
history.append({"role": "user", "content": effective_user_msg})
|
| 382 |
history.append({"role": "assistant", "content": result.get("response", "")})
|
| 383 |
state["conversation_history"] = history[-max_history:]
|
| 384 |
|
|
@@ -4,9 +4,9 @@ import streamlit as st
|
|
| 4 |
|
| 5 |
from src.config_loader import load_config, get_api_key
|
| 6 |
from src.ingest import get_chroma_collection, ingest_documents
|
| 7 |
-
from src.kb_meta import
|
| 8 |
from src.query_engine import understand_query
|
| 9 |
-
from src.retriever import retrieve
|
| 10 |
from src.verifier import verify_and_respond
|
| 11 |
from src.llm import list_models
|
| 12 |
|
|
@@ -59,6 +59,13 @@ def render_sidebar():
|
|
| 59 |
cfg.setdefault("llm", {})["provider"] = provider
|
| 60 |
|
| 61 |
# --- Model dropdown (cached per provider) ---
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
models_cache_key = f"models_{provider}"
|
| 63 |
if models_cache_key not in st.session_state:
|
| 64 |
api_key = get_api_key(cfg, provider)
|
|
@@ -115,9 +122,11 @@ def render_sidebar():
|
|
| 115 |
|
| 116 |
# --- Re-ingest button ---
|
| 117 |
if st.button("Re-ingest documents", use_container_width=True):
|
|
|
|
| 118 |
with st.spinner("Ingesting documents..."):
|
| 119 |
try:
|
| 120 |
count = ingest_documents(cfg)
|
|
|
|
| 121 |
st.success(f"Ingested {count} chunks.")
|
| 122 |
# Clear cached data so it refreshes after re-ingest
|
| 123 |
st.session_state.pop("kb_welcome_summary", None)
|
|
@@ -279,6 +288,11 @@ def render_chat():
|
|
| 279 |
{"role": "assistant", "content": result.get("response", "")}
|
| 280 |
)
|
| 281 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 282 |
|
| 283 |
# ── Main ─────────────────────────────────────────────────────────────────────
|
| 284 |
|
|
@@ -300,7 +314,7 @@ def main():
|
|
| 300 |
# Show KB summary on first visit (LLM-generated welcome summary)
|
| 301 |
if not st.session_state.messages:
|
| 302 |
if "kb_welcome_summary" not in st.session_state:
|
| 303 |
-
st.session_state.kb_welcome_summary =
|
| 304 |
kb_summary = st.session_state.kb_welcome_summary
|
| 305 |
if kb_summary:
|
| 306 |
with st.expander("Knowledge Base Contents", expanded=True):
|
|
|
|
| 4 |
|
| 5 |
from src.config_loader import load_config, get_api_key
|
| 6 |
from src.ingest import get_chroma_collection, ingest_documents
|
| 7 |
+
from src.kb_meta import load_kb_meta_brief
|
| 8 |
from src.query_engine import understand_query
|
| 9 |
+
from src.retriever import clear_collection_cache, retrieve
|
| 10 |
from src.verifier import verify_and_respond
|
| 11 |
from src.llm import list_models
|
| 12 |
|
|
|
|
| 59 |
cfg.setdefault("llm", {})["provider"] = provider
|
| 60 |
|
| 61 |
# --- Model dropdown (cached per provider) ---
|
| 62 |
+
# Invalidate model cache if provider changed
|
| 63 |
+
prev_provider_key = "prev_provider"
|
| 64 |
+
if st.session_state.get(prev_provider_key) != provider:
|
| 65 |
+
for p in providers:
|
| 66 |
+
st.session_state.pop(f"models_{p}", None)
|
| 67 |
+
st.session_state[prev_provider_key] = provider
|
| 68 |
+
|
| 69 |
models_cache_key = f"models_{provider}"
|
| 70 |
if models_cache_key not in st.session_state:
|
| 71 |
api_key = get_api_key(cfg, provider)
|
|
|
|
| 122 |
|
| 123 |
# --- Re-ingest button ---
|
| 124 |
if st.button("Re-ingest documents", use_container_width=True):
|
| 125 |
+
st.info("Ingestion may take a few minutes for large document collections...")
|
| 126 |
with st.spinner("Ingesting documents..."):
|
| 127 |
try:
|
| 128 |
count = ingest_documents(cfg)
|
| 129 |
+
clear_collection_cache()
|
| 130 |
st.success(f"Ingested {count} chunks.")
|
| 131 |
# Clear cached data so it refreshes after re-ingest
|
| 132 |
st.session_state.pop("kb_welcome_summary", None)
|
|
|
|
| 288 |
{"role": "assistant", "content": result.get("response", "")}
|
| 289 |
)
|
| 290 |
|
| 291 |
+
# Cap message history to prevent unbounded memory growth
|
| 292 |
+
max_messages = 200 # 100 Q&A pairs
|
| 293 |
+
if len(st.session_state.messages) > max_messages:
|
| 294 |
+
st.session_state.messages = st.session_state.messages[-max_messages:]
|
| 295 |
+
|
| 296 |
|
| 297 |
# ── Main ─────────────────────────────────────────────────────────────────────
|
| 298 |
|
|
|
|
| 314 |
# Show KB summary on first visit (LLM-generated welcome summary)
|
| 315 |
if not st.session_state.messages:
|
| 316 |
if "kb_welcome_summary" not in st.session_state:
|
| 317 |
+
st.session_state.kb_welcome_summary = load_kb_meta_brief(cfg)
|
| 318 |
kb_summary = st.session_state.kb_welcome_summary
|
| 319 |
if kb_summary:
|
| 320 |
with st.expander("Knowledge Base Contents", expanded=True):
|
|
@@ -186,7 +186,7 @@ severity, file, root cause, fix applied, and the audit round that found it.
|
|
| 186 |
| K12 | LOW | Multiple | `cfg.get("section", {})` systemic pattern — if YAML section exists but is `None`, returns `None` not `{}` | Partially mitigated by config_loader normalization; remaining ~30 locations are low risk |
|
| 187 |
| K6 | LOW | `src/search/semantic_scholar.py` | Filters out papers without abstracts, reducing result count for some topics | By design — abstract-less papers can't provide useful context |
|
| 188 |
| K7 | LOW | `src/retriever.py` | LIKE wildcards `_`/`%` not escaped in fallback SQL query | Minor — extra matches unlikely to cause visible problems |
|
| 189 |
-
| K8 | LOW | `CLAUDE.md` | Test count
|
| 190 |
| K9 | LOW | `src/verifier.py` | `compute_similarity_flags` regex `[a-z]{3,}` excludes non-ASCII and 2-letter acronyms (AI, US, UK) | Advisory-only layer; impact limited to noisy flags |
|
| 191 |
|
| 192 |
---
|
|
@@ -240,7 +240,7 @@ This round was a comprehensive pre-release inspection using 6 parallel audit age
|
|
| 240 |
|
| 241 |
| # | Severity | File | Description | Reason |
|
| 242 |
|---|----------|------|-------------|--------|
|
| 243 |
-
| K13 | HIGH | `src/llm/gemini.py` + `requirements.txt` | **CON-4**: `google-generativeai` package fully deprecated; replacement `google-genai` has different API surface |
|
| 244 |
| K14 | MEDIUM | `src/verifier.py` | **AH-05**: Token cap step function too generous for small contexts (501 chars → 1536 tokens = 12:1 ratio) | Would benefit from continuous formula; current 3-tier approach is functional |
|
| 245 |
| K15 | MEDIUM | `src/verifier.py:24-33` | **AH-11**: Warning phrase list missing common patterns: "studies have shown", "research suggests", "historically" | Advisory layer only; expandable post-release |
|
| 246 |
| K16 | LOW | `src/verifier.py:184` | **INSPECT-2**: References section regex matches "sources" in prose, not just section header | Mitigated: filename matching still works because match includes everything to end of response |
|
|
@@ -276,13 +276,212 @@ User-reported bug: querying "south korea in pts" returned zero SQL results despi
|
|
| 276 |
|
| 277 |
---
|
| 278 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 279 |
## Statistics
|
| 280 |
|
| 281 |
| Metric | Count |
|
| 282 |
|--------|-------|
|
| 283 |
-
| Total bugs
|
| 284 |
-
|
|
|
|
|
|
|
|
|
|
|
| 285 |
| Test count | 174 passing |
|
| 286 |
| Test files | 17 |
|
| 287 |
-
| Source files
|
| 288 |
-
|
|
|
|
|
|
|
| 186 |
| K12 | LOW | Multiple | `cfg.get("section", {})` systemic pattern — if YAML section exists but is `None`, returns `None` not `{}` | Partially mitigated by config_loader normalization; remaining ~30 locations are low risk |
|
| 187 |
| K6 | LOW | `src/search/semantic_scholar.py` | Filters out papers without abstracts, reducing result count for some topics | By design — abstract-less papers can't provide useful context |
|
| 188 |
| K7 | LOW | `src/retriever.py` | LIKE wildcards `_`/`%` not escaped in fallback SQL query | Minor — extra matches unlikely to cause visible problems |
|
| 189 |
+
| K8 | LOW | `CLAUDE.md` | Test count said 162, actual is 174 | FIXED (Round 11 doc update) |
|
| 190 |
| K9 | LOW | `src/verifier.py` | `compute_similarity_flags` regex `[a-z]{3,}` excludes non-ASCII and 2-letter acronyms (AI, US, UK) | Advisory-only layer; impact limited to noisy flags |
|
| 191 |
|
| 192 |
---
|
|
|
|
| 240 |
|
| 241 |
| # | Severity | File | Description | Reason |
|
| 242 |
|---|----------|------|-------------|--------|
|
| 243 |
+
| K13 | ~~HIGH~~ | `src/llm/gemini.py` + `requirements.txt` | **CON-4**: `google-generativeai` package fully deprecated; replacement `google-genai` has different API surface | **FIXED** — Migrated to `google-genai>=1.0.0` SDK with `genai.Client(api_key=...)` (thread-safe, no global state) |
|
| 244 |
| K14 | MEDIUM | `src/verifier.py` | **AH-05**: Token cap step function too generous for small contexts (501 chars → 1536 tokens = 12:1 ratio) | Would benefit from continuous formula; current 3-tier approach is functional |
|
| 245 |
| K15 | MEDIUM | `src/verifier.py:24-33` | **AH-11**: Warning phrase list missing common patterns: "studies have shown", "research suggests", "historically" | Advisory layer only; expandable post-release |
|
| 246 |
| K16 | LOW | `src/verifier.py:184` | **INSPECT-2**: References section regex matches "sources" in prose, not just section header | Mitigated: filename matching still works because match includes everything to end of response |
|
|
|
|
| 276 |
|
| 277 |
---
|
| 278 |
|
| 279 |
+
## Round 11 — 2026-03-06 Session 6 (8-Agent Final Pre-Release Audit + Cross-Check)
|
| 280 |
+
|
| 281 |
+
This round was the strictest final audit before public release: 8 parallel agents (core pipeline, ingestion/storage, SQL security, LLM/prompts, UIs/setup, readers/search, test suite, design docs/architecture) each read every line of their assigned files and cross-checked findings.
|
| 282 |
+
|
| 283 |
+
### HIGH Fixes
|
| 284 |
+
|
| 285 |
+
| # | Severity | File | Bug | Fix | Status |
|
| 286 |
+
|---|----------|------|-----|-----|--------|
|
| 287 |
+
| 84 | HIGH | `src/query_engine.py:96` | QU `max_tokens=256` too tight — SQL queries + reasoning fields truncate JSON output, causing silent fallback to vector-only search | Increased to `max_tokens=512` | FIXED |
|
| 288 |
+
| 85 | HIGH | `src/llm/gemini.py:25-26` | Gemini `generate()` returned error strings instead of raising exceptions, inconsistent with OpenAI/Anthropic — error strings leaked into correction loop as "responses" | Changed to `raise RuntimeError(...)` for API errors (ValueError for safety blocks still returns string) | FIXED |
|
| 289 |
+
| 86 | HIGH | `src/sql_retriever.py:184-214` | `make_fuzzy_query()` unsanitized value interpolation — captured regex values not escaped for single quotes, allowing SQL injection via crafted `WHERE` values | Added `.replace("'", "''")` escaping in both `_phrase_replace()` and `_word_replace()` | FIXED |
|
| 290 |
+
| 87 | HIGH | `src/config_loader.py:35-37` | Nested null YAML values crashed consumers — `vector_db:` with no value → `.get("vector_db", "chroma_db")` returns `None` → `TypeError` in `os.path.isabs(None)` | Changed top-level-only normalization to recursive `_normalize_nulls()` that handles all nested dicts | FIXED |
|
| 291 |
+
| 88 | HIGH | `setup.py:93-94` | `.env` parser used `strip('"')` which removes ALL leading/trailing quotes, not just one matched pair — could corrupt API keys starting/ending with quote chars | Changed to proper one-pair unwrap: check for matching pair then slice `[1:-1]` | FIXED |
|
| 292 |
+
| 89 | HIGH | `setup.py:100-102` | `.env` writer did not escape special chars in values — API keys containing `"` or `\` produced malformed `.env` files | Added `v.replace('\\', '\\\\').replace('"', '\\"')` before writing | FIXED |
|
| 293 |
+
| 90 | HIGH | `src/sql_retriever.py:41-42` | `sqlite_master` access not blocked — `SELECT * FROM sqlite_master` passed all validation, exposing full DDL schema | Added `sqlite_(master\|schema\|temp_master\|temp_schema)` check in `_validate_sql()` | FIXED |
|
| 294 |
+
| 91 | HIGH | `tests/test_verifier.py:65-66,95-96` | Wrong mock target — patched `src.kb_meta.load_kb_meta` but verifier imports `load_kb_meta_brief`; mocks were no-ops (only worked due to fallback chain) | Changed both patches to `src.kb_meta.load_kb_meta_brief` | FIXED |
|
| 295 |
+
| 92 | HIGH | `src/retriever.py:381` | SQL keyword fallback ran unconditionally for `route="sql"` even when vector search already found results — wasted computation | Added `and not db_results` condition: only run keyword SQL when vector fallback also found nothing | FIXED |
|
| 296 |
+
|
| 297 |
+
### MEDIUM Fixes
|
| 298 |
+
|
| 299 |
+
| # | Severity | File | Bug | Fix | Status |
|
| 300 |
+
|---|----------|------|-----|-----|--------|
|
| 301 |
+
| 93 | MEDIUM | `src/search/semantic_scholar.py:47,49` | `citationCount: null` from API → `None` in sort key → `TypeError: '<' not supported between NoneType and int` crash | Changed to `paper.get("citationCount") or 0` and `x.get("citation_count") or 0` | FIXED |
|
| 302 |
+
| 94 | MEDIUM | `app_web.py:123` | Stale KB welcome summary after re-ingest — `st.session_state.kb_welcome_summary` not cleared before `st.rerun()` | Added `st.session_state.pop("kb_welcome_summary", None)` before rerun | FIXED |
|
| 303 |
+
| 95 | MEDIUM | `app_web.py:256-257,267` | SQL result count missing from web UI status label — showed "0 local + 0 web" even when SQL returned rows | Added `n_sql` to both "Verified" and "Done" status labels, matching CLI behavior | FIXED |
|
| 304 |
+
| 96 | MEDIUM | `src/verifier.py:184` | References regex `\b(references\|sources)\b` matched casual mentions of "sources" in prose — citation audit checked entire response body instead of just References section | Changed to heading-specific regex: `^#+\s*(references\|sources)` or `^\*\*(references\|sources)\*\*` | FIXED |
|
| 305 |
+
| 97 | MEDIUM | `src/verifier.py:486` | Non-strict mode used stale `citation_warnings` from last loop iteration, not from final corrected response | Added `validate_citations(response, retrieval_result)` recomputation before building warning | FIXED |
|
| 306 |
+
| 98 | MEDIUM | `src/ingest.py:291-298` | Meta files deleted before overview regeneration — if `build_and_store_overview()` failed, KB lost self-awareness until next ingestion | Removed pre-deletion; `write_text()` in `build_and_store_overview()` already overwrites | FIXED |
|
| 307 |
+
| 99 | MEDIUM | `src/readers/docx.py:4-10` | DOCX reader only extracted paragraphs, silently dropping all embedded tables | Added table extraction loop: `doc.tables` → row cells joined with ` \| ` | FIXED |
|
| 308 |
+
|
| 309 |
+
### Known Issues Added
|
| 310 |
+
|
| 311 |
+
| # | Severity | File | Description | Reason |
|
| 312 |
+
|---|----------|------|-------------|--------|
|
| 313 |
+
| K18 | MEDIUM | `src/sql_retriever.py:17-24` | Comment stripping is naive (regex) — corrupts string literals containing `--` or `/* */` | Rare in LLM-generated queries; proper state-machine parser would be over-engineering |
|
| 314 |
+
| K19 | MEDIUM | `src/retriever.py:195-196, 267-269` | LIKE wildcard chars `%` and `_` not escaped in fallback/alt-column queries | Same as existing K7; low real-world impact |
|
| 315 |
+
| K20 | MEDIUM | `src/sql_retriever.py` | No SQL execution timeout — Cartesian products could hang | Same as existing K10; SQLite lacks built-in SELECT timeout |
|
| 316 |
+
| K21 | MEDIUM | `src/verifier.py:151-202` | `validate_citations` compares citation numbers against raw retrieval count, not References section entry count | Would require parsing References section to count entries; current check is conservative |
|
| 317 |
+
| K22 | LOW | `src/readers/csv_tab.py:11` | CSV encoding hardcoded to `utf-8-sig` — non-UTF-8 files get replacement chars | Would need `chardet` dependency or encoding detection |
|
| 318 |
+
| K23 | LOW | `src/sql_ingest.py:568-586` | Row-by-row SQL INSERT slow for large datasets | Would benefit from `executemany()` batching |
|
| 319 |
+
| K24 | LOW | `src/readers/excel.py:40, csv_tab.py:22` | `zip(headers, row)` silently truncates extra columns in ragged rows | Acceptable for most research datasets |
|
| 320 |
+
| K25 | LOW | `src/ingest.py:104` | `rglob("*")` follows symlinks — cyclic symlinks could cause infinite loop | Rare in knowledge_base directories |
|
| 321 |
+
| K26 | LOW | Test suite | Zero coverage: `validate_citations`, verification loop, `make_fuzzy_query`, `compute_similarity_flags`, `load_kb_meta_brief` | Test gap documented; non-blocking for release |
|
| 322 |
+
|
| 323 |
+
### Test Coverage Gaps (Confirmed)
|
| 324 |
+
|
| 325 |
+
| Priority | What | Description |
|
| 326 |
+
|----------|------|-------------|
|
| 327 |
+
| HIGH | `validate_citations()` | Layer 4.5 — zero direct tests |
|
| 328 |
+
| HIGH | Verification correction loop | Core feature — zero multi-iteration tests |
|
| 329 |
+
| HIGH | `make_fuzzy_query()` | SQL fuzzy fallback — zero tests |
|
| 330 |
+
| MEDIUM | `compute_similarity_flags()` | Layer 4 — zero tests |
|
| 331 |
+
| MEDIUM | `load_kb_meta_brief()` | Used by verifier — zero tests |
|
| 332 |
+
|
| 333 |
+
---
|
| 334 |
+
|
| 335 |
+
## Round 12 — 2026-03-06 Session 7 (Pre-Release Final Audit)
|
| 336 |
+
|
| 337 |
+
See `docs/AUDIT_BUGLOG.md` Round 12 section for full details.
|
| 338 |
+
33 bugs found across 3 tiers, all FIXED. Design score raised from 7.1/10 to 9/10.
|
| 339 |
+
5 NEW low-severity issues documented as KNOWN. 174/174 tests passing. **Released to GitHub.**
|
| 340 |
+
|
| 341 |
+
---
|
| 342 |
+
|
| 343 |
+
## Round 13 — 2026-03-07 (Post-Release Deep Audit)
|
| 344 |
+
|
| 345 |
+
**7 parallel audit agents** (design evaluation + core pipeline + verification stack + SQL layer + LLM/readers/search + application layer + test coverage). Full line-by-line review of all 25 source files, 17 test files, and all documentation.
|
| 346 |
+
|
| 347 |
+
**Design Score**: 8.5/10 → **9.1/10** (post-fix) | **Test Coverage Score**: 6/10
|
| 348 |
+
|
| 349 |
+
### Tier 1: Anti-Hallucination Integrity (4 bugs) — ALL FIXED
|
| 350 |
+
|
| 351 |
+
| # | Severity | File | Bug | Fix | Status |
|
| 352 |
+
|---|----------|------|-----|-----|--------|
|
| 353 |
+
| 100 | HIGH | `verifier.py` | **R13-T1-01**: Verification LLM persistent failure → `except Exception: continue` skips all iterations → non-strict returns unverified response. Full bypass of Layers 3–5. | Added `any_verification_ran` flag. If ALL verification LLM calls fail: strict mode refuses with system-failure message; non-strict adds prominent WARNING banner distinguishing "could not run" from "did not pass". | FIXED |
|
| 354 |
+
| 101 | HIGH | `verifier.py` | **R13-T1-02**: `validate_citations` SQL/web source matching uses wrong dict structure (dead code). Layer 4.5 only works for local sources. | SQL: extract source filename from context via regex. Web: changed to flat dict access `web_chunk.get("url", "")`. | FIXED |
|
| 355 |
+
| 102 | HIGH | `verifier.py` | **R13-T1-03**: Citation count uses unique source *files*, not citable *units*. False "fabricated reference" warning on multi-chunk single-file sources. | Changed `db_count = len(db_results)` (chunks) instead of `len(unique_db_sources)` (files). Each chunk is a citable unit. | FIXED |
|
| 356 |
+
| 103 | MEDIUM | `verifier.py` | **R13-T1-04**: `max_iterations: 0` with `enabled: true` silently disables verification. | Clamp `max_iterations = max(1, max_iterations)` with warning. Must use `enabled: false` to disable. Test updated. | FIXED |
|
| 357 |
+
|
| 358 |
+
### Tier 2: Pre-Release Quality (24 bugs) — ALL FIXED
|
| 359 |
+
|
| 360 |
+
| # | Severity | File | Bug | Fix | Status |
|
| 361 |
+
|---|----------|------|-----|-----|--------|
|
| 362 |
+
| 104 | HIGH | `retriever.py` | **R13-T2-01**: Missing LIKE wildcard ESCAPE in `_build_fallback_sql_query` and `_try_alternate_columns`. | Added `%`/`_` escaping and `ESCAPE '\'` clause. Initial fix had double-backslash bug; corrected by regression checker. | FIXED |
|
| 363 |
+
| 105 | HIGH | `gemini.py` | **R13-T2-02**: Gemini safety filter returns `"[Gemini blocked: ...]"` bypassing anti-hallucination stack. | Return `""` instead. Verifier's empty-response handler (lines 369-378) provides user-facing refusal. | FIXED |
|
| 364 |
+
| 106 | HIGH | `openai.py` | **R13-T2-03**: Reasoning model budget uncapped → API error with large `max_tokens`. | `reasoning_budget = min(max(max_tokens * 4, 4096), 128000)`. | FIXED |
|
| 365 |
+
| 107 | HIGH | `llm/__init__.py` | **R13-T2-04**: Non-numeric temperature/max_tokens crashes with opaque ValueError. | Wrapped in try/except with defaults (0.0 and 2048). | FIXED |
|
| 366 |
+
| 108 | HIGH | `app_cli.py` | **R13-T2-05**: After clarification, raw `user_input` re-appended → duplicate history entries. | Use `display_query` (post-QU, includes clarification context) instead of raw `user_input`. | FIXED |
|
| 367 |
+
| 109 | MEDIUM | `llm/__init__.py` | **R13-T2-06**: Anthropic temp > 1.0 → API error. | Added provider-specific clamp: `if provider == "anthropic": temperature = min(temperature, 1.0)`. | FIXED |
|
| 368 |
+
| 110 | MEDIUM | `sql_retriever.py` | **R13-T2-07**: No SQL query timeout. `CROSS JOIN` hangs indefinitely. | Added `set_progress_handler` with 10-second abort. `import time` at module level. | FIXED |
|
| 369 |
+
| 111 | MEDIUM | `app_cli.py` | **R13-T2-08**: Rich markup injection via brackets in error messages/queries. | Added `rich_escape()` to all 5 interpolated f-strings in the RAG pipeline loop. | FIXED |
|
| 370 |
+
| 112 | MEDIUM | `readers/*.py` | **R13-T2-09**: No try/except in readers. Corrupt file aborts ingestion. | Wrapped `read_pdf`, `read_docx`, `read_excel`, `read_stata`, `read_rdata`, `read_spss` in try/except. | FIXED |
|
| 371 |
+
| 113 | MEDIUM | `retriever.py`, `app_cli.py`, `app_web.py` | **R13-T2-10**: `_collection_cache` never invalidated after re-ingestion. | Added `clear_collection_cache()` function; called after ingestion in both CLI and web. | FIXED |
|
| 372 |
+
| 114 | MEDIUM | `app_web.py` | **R13-T2-11**: Web UI messages grow without limit. | Capped at 200 entries (100 Q&A pairs) after each append cycle. | FIXED |
|
| 373 |
+
| 115 | MEDIUM | `setup.py` | **R13-T2-12**: Setup wizard overwrites config without confirmation. | Added existence check + `[y/N]` prompt before overwriting `config.yaml`. | FIXED |
|
| 374 |
+
| 116 | MEDIUM | `prompts.py` | **R13-T2-13**: Prompt injection via KB delimiter mimicry. | Sanitize context: replace 69-char `=` sequences with Unicode `≡` before injection. | FIXED |
|
| 375 |
+
| 117 | MEDIUM | `verifier.py` | **R13-T2-14**: References regex matches `## Data Sources` etc. | Anchored regex with `\s*$` to require standalone headers only. | FIXED |
|
| 376 |
+
| 118 | MEDIUM | `verifier.py` | **R13-T2-15**: Correction prompt includes full previous response. | Truncate to 1500 chars. Instruction: "Rewrite COMPLETE response from scratch". | FIXED |
|
| 377 |
+
| 119 | MEDIUM | `verifier.py` | **R13-T2-16**: Initial `generate()` exception propagates. | Wrapped in try/except; returns `refused: True` with error message. | FIXED |
|
| 378 |
+
| 120 | MEDIUM | `openai.py` | **R13-T2-17**: Reasoning model detection via brittle prefix match. | Changed to `model in ("o1","o3","o4") or model.startswith(("o1-","o3-","o4-"))`. | FIXED |
|
| 379 |
+
| 121 | MEDIUM | `sql_retriever.py` | **R13-T2-18**: `make_fuzzy_query` regex matches non-column contexts. | Pattern now requires quoted identifiers or valid SQL column names. | FIXED |
|
| 380 |
+
| 122 | MEDIUM | `sql_retriever.py` | **R13-T2-19**: `REPLACE()` function falsely blocked. | Removed `REPLACE` from `_DANGEROUS_KEYWORDS`. Statement form blocked by SELECT-only check. | FIXED |
|
| 381 |
+
| 123 | MEDIUM | `config_loader.py` | **R13-T2-20**: Non-dict YAML → `AttributeError`. | Added `isinstance(cfg, dict)` check with clear ValueError message. | FIXED |
|
| 382 |
+
| 124 | MEDIUM | `app_web.py` | **R13-T2-21**: Model list cache stale after provider change. | Added provider-change detection that clears all model caches. | FIXED |
|
| 383 |
+
| 125 | MEDIUM | `app_web.py` | **R13-T2-22**: Web UI uses slow LLM call for KB welcome. | Changed to `load_kb_meta_brief()` (file read) matching CLI behavior. | FIXED |
|
| 384 |
+
| 126 | MEDIUM | `app_web.py` | **R13-T2-23**: Re-ingestion blocks without progress feedback. | Added `st.info()` message before spinner to set expectations. | FIXED |
|
| 385 |
+
| 127 | MEDIUM | `setup.py` | **R13-T2-24**: Empty API key warning not actionable. | Improved message: "Set it later by editing .env or re-running: python setup.py". | FIXED |
|
| 386 |
+
|
| 387 |
+
### Tier 3: Post-Release Improvements (20 bugs)
|
| 388 |
+
|
| 389 |
+
| # | Severity | File | Bug | Status |
|
| 390 |
+
|---|----------|------|-----|--------|
|
| 391 |
+
| 128 | MED | `openai.py:9`, `anthropic.py:9`, `gemini.py:16` | **R13-T3-01**: New API client per `generate()` call. No connection reuse. | OPEN |
|
| 392 |
+
| 129 | MED | `llm/openai.py:52`, `anthropic.py:32`, `gemini.py:48` | **R13-T3-02**: `list_models()` returns hardcoded list, not live API fetch. Design doc says otherwise. | OPEN |
|
| 393 |
+
| 130 | MED | `retriever.py:282-285` | **R13-T3-03**: `_try_alternate_columns` quote escaping logic fragile and non-obvious. | OPEN |
|
| 394 |
+
| 131 | MED | `sql_retriever.py:54` | **R13-T3-04**: `sqlite_master` check on `stripped` instead of `unquoted`. False positives. | OPEN |
|
| 395 |
+
| 132 | MED | `readers/stata.py:28`, `rdata.py:15` | **R13-T3-05**: `df.iterrows()` 10-100x slower than `itertuples()`. Large datasets take minutes. | OPEN |
|
| 396 |
+
| 133 | LOW | `sql_retriever.py:26-33` | **R13-T3-06**: `_strip_sql_comments` removes `--` inside string literals. Confirmed from R12 NEW-04. | OPEN |
|
| 397 |
+
| 134 | LOW | `verifier.py:429` | **R13-T3-07**: Verification LLM system prompt minimal. No anti-hallucination rules for verifier itself. | OPEN |
|
| 398 |
+
| 135 | LOW | `verifier.py:25-34` | **R13-T3-08**: Warning phrases easily evaded by paraphrasing. Only 8 exact-match phrases. | OPEN |
|
| 399 |
+
| 136 | LOW | `verifier.py:126` | **R13-T3-09**: Term overlap excludes 2-char acronyms (AI, UK, EU, US, UN). | OPEN |
|
| 400 |
+
| 137 | LOW | `verifier.py:181` | **R13-T3-10**: Citation `[0]` never detected as invalid. | OPEN |
|
| 401 |
+
| 138 | LOW | `readers/csv_tab.py:11`, `text.py:5` | **R13-T3-11**: Hardcoded UTF-8. Non-UTF-8 files get replacement chars. | OPEN |
|
| 402 |
+
| 139 | LOW | `readers/docx.py:24` | **R13-T3-12**: All DOCX content mapped to `page: 1`. | OPEN |
|
| 403 |
+
| 140 | LOW | `readers/pdf.py:9` | **R13-T3-13**: Scanned PDFs silently return empty list. No warning. | OPEN |
|
| 404 |
+
| 141 | LOW | `readers/excel.py:15` | **R13-T3-14**: Unevaluated formulas appear as `None`, silently dropped. | OPEN |
|
| 405 |
+
| 142 | LOW | `search/semantic_scholar.py:15-28` | **R13-T3-15**: Retries on non-retriable 4xx errors. Wastes 17s. | OPEN |
|
| 406 |
+
| 143 | LOW | `search/__init__.py:10-12` | **R13-T3-16**: Unknown backend names silently return `[]`. | OPEN |
|
| 407 |
+
| 144 | LOW | `setup.py:96-100` | **R13-T3-17**: `.env` unescape wrong order. Theoretical corruption. | OPEN |
|
| 408 |
+
| 145 | LOW | `config_loader.py:20-22` | **R13-T3-18**: `.env` loaded before config existence check. | OPEN |
|
| 409 |
+
| 146 | LOW | `app_web.py:288` | **R13-T3-19**: Browser tab title hardcoded "ResearchBot". | OPEN |
|
| 410 |
+
| 147 | LOW | `verifier.py:460-462` | **R13-T3-20**: Correction call uses same temperature as initial. Should force 0.0. | OPEN |
|
| 411 |
+
|
| 412 |
+
### Test Coverage Gaps (Round 13 — confirmed + expanded)
|
| 413 |
+
|
| 414 |
+
| Priority | What | File | Status |
|
| 415 |
+
|----------|------|------|--------|
|
| 416 |
+
| CRITICAL | `validate_citations()` — Layer 4.5 | `verifier.py:152-230` | OPEN |
|
| 417 |
+
| CRITICAL | Verification correction loop | `verifier.py:407-520` | OPEN |
|
| 418 |
+
| CRITICAL | `make_fuzzy_query()` | `sql_retriever.py:181-237` | OPEN |
|
| 419 |
+
| HIGH | `compute_similarity_flags()` — Layer 4 | `verifier.py:101-149` | OPEN |
|
| 420 |
+
| HIGH | `_strip_quoted()` / `_strip_sql_comments()` | `sql_retriever.py:17-33` | OPEN |
|
| 421 |
+
| HIGH | Empty LLM response in `verify_and_respond` | `verifier.py:354-364` | OPEN |
|
| 422 |
+
| HIGH | `build_and_store_overview()` orchestrator | `kb_meta.py:464-492` | OPEN |
|
| 423 |
+
| MEDIUM | `generate_brief_overview()` | `kb_meta.py:385-422` | OPEN |
|
| 424 |
+
| MEDIUM | `load_kb_meta_brief()` fallback | `kb_meta.py:434-445` | OPEN |
|
| 425 |
+
| MEDIUM | `summarize_kb_for_welcome()` | `kb_meta.py:514-552` | OPEN |
|
| 426 |
+
| MEDIUM | `_build_fallback_sql_query` with real data | `retriever.py:159-217` | OPEN |
|
| 427 |
+
| MEDIUM | `_try_alternate_columns` with real data | `retriever.py:220-292` | OPEN |
|
| 428 |
+
| MEDIUM | `_validate_sql` edge cases | `sql_retriever.py:36-59` | OPEN |
|
| 429 |
+
| MEDIUM | `format_sql_results_as_context` truncation | `sql_retriever.py:116-167` | OPEN |
|
| 430 |
+
| MEDIUM | `_normalize_nulls` | `config_loader.py:36-42` | OPEN |
|
| 431 |
+
|
| 432 |
+
### Cross-Agent Consensus (confirmed by 2+ agents independently)
|
| 433 |
+
|
| 434 |
+
| Bug | Agents | Confidence |
|
| 435 |
+
|-----|--------|-----------|
|
| 436 |
+
| `validate_citations` dead code for SQL + web (#101) | Core Pipeline, Application, Verification | **Very High** |
|
| 437 |
+
| Missing LIKE ESCAPE in SQL fallbacks (#104) | SQL, Core Pipeline, Design | **Very High** |
|
| 438 |
+
| Stale `_collection_cache` after re-ingestion (#113) | Core Pipeline, Design | **High** |
|
| 439 |
+
| Verification LLM failure → silent bypass (#100) | Verification, Core Pipeline | **High** |
|
| 440 |
+
| Missing reader exception handling (#112) | LLM/Readers, Design | **High** |
|
| 441 |
+
|
| 442 |
+
### Recommended Fix Order
|
| 443 |
+
|
| 444 |
+
**Phase 1 — Anti-hallucination integrity (#100–#103):**
|
| 445 |
+
Fix verification bypass, validate_citations data structure, citation counting, max_iterations guard.
|
| 446 |
+
|
| 447 |
+
**Phase 2 — High-severity T2 bugs (#104–#108):**
|
| 448 |
+
LIKE ESCAPE, Gemini blocked response, reasoning model overflow, config validation, history corruption.
|
| 449 |
+
|
| 450 |
+
**Phase 3 — Test coverage (3 CRITICAL gaps):**
|
| 451 |
+
Tests for validate_citations, verification correction loop, make_fuzzy_query.
|
| 452 |
+
|
| 453 |
+
**Phase 4 — Remaining T2 MEDIUM bugs (#109–#127):**
|
| 454 |
+
Address in order of impact.
|
| 455 |
+
|
| 456 |
+
**Phase 5 — T3 improvements (#128–#147):**
|
| 457 |
+
Address as time permits.
|
| 458 |
+
|
| 459 |
+
---
|
| 460 |
+
|
| 461 |
+
## Round 14 — 2026-03-07 (Tier 1 + Tier 2 Bug Fix Sprint)
|
| 462 |
+
|
| 463 |
+
**6 parallel agents**: 4 fix agents (T1, T2-HIGH, T2-MED-A, T2-MED-B) + 1 regression checker + 1 design guardian.
|
| 464 |
+
|
| 465 |
+
- Fixed all 28 Tier 1 + Tier 2 bugs (#100–#127)
|
| 466 |
+
- Regression checker found 1 CRITICAL issue in Bug #104 fix (double-backslash in ESCAPE clause) — corrected
|
| 467 |
+
- Also fixed `spss.py` missing try/except (pre-existing gap found by regression checker)
|
| 468 |
+
- Design Guardian assessment: design score 8.5 → 9.1/10, recommendation: SHIP
|
| 469 |
+
- Updated test for Bug #103 (renamed to `test_max_iterations_zero_clamps_to_one`)
|
| 470 |
+
- 174/174 tests passing
|
| 471 |
+
|
| 472 |
+
---
|
| 473 |
+
|
| 474 |
## Statistics
|
| 475 |
|
| 476 |
| Metric | Count |
|
| 477 |
|--------|-------|
|
| 478 |
+
| Total bugs found (all rounds) | ~216 (48 new in Round 13) |
|
| 479 |
+
| Total bugs fixed (all rounds) | ~196 (168 in R1–R12, 28 in R14) |
|
| 480 |
+
| Round 13 T1+T2 bugs | 28 FIXED (was 28 OPEN) |
|
| 481 |
+
| Round 13 T3 bugs remaining | 20 OPEN (5 MEDIUM, 15 LOW) |
|
| 482 |
+
| Audit rounds completed | 14 (50+ total agent dispatches) |
|
| 483 |
| Test count | 174 passing |
|
| 484 |
| Test files | 17 |
|
| 485 |
+
| Source files modified in R14 | 16 (verifier, retriever, gemini, openai, llm/__init__, app_cli, app_web, sql_retriever, config_loader, prompts, setup, pdf, docx, excel, stata, rdata, spss) |
|
| 486 |
+
| Design score | 9.1/10 (up from 8.5) |
|
| 487 |
+
| CRITICAL test gaps | 3 (validate_citations, correction loop, make_fuzzy_query) |
|
|
@@ -141,7 +141,8 @@ def run_wizard():
|
|
| 141 |
# Step 3b: API key
|
| 142 |
api_key = getpass.getpass(f"Enter your {provider} API key: ")
|
| 143 |
if not api_key:
|
| 144 |
-
print(f" Warning: No API key entered for {provider}.
|
|
|
|
| 145 |
|
| 146 |
# Step 4: Model selection — fetch available models
|
| 147 |
print(f"\nFetching available {provider} models...")
|
|
@@ -179,9 +180,18 @@ def run_wizard():
|
|
| 179 |
|
| 180 |
# Write config.yaml
|
| 181 |
config_path = project_root / "config.yaml"
|
| 182 |
-
|
| 183 |
-
|
| 184 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 185 |
|
| 186 |
# Write .env (merges with existing keys if present)
|
| 187 |
env_path = project_root / ".env"
|
|
|
|
| 141 |
# Step 3b: API key
|
| 142 |
api_key = getpass.getpass(f"Enter your {provider} API key: ")
|
| 143 |
if not api_key:
|
| 144 |
+
print(f" Warning: No API key entered for {provider}.")
|
| 145 |
+
print(f" Set it later by editing .env or re-running: python setup.py")
|
| 146 |
|
| 147 |
# Step 4: Model selection — fetch available models
|
| 148 |
print(f"\nFetching available {provider} models...")
|
|
|
|
| 180 |
|
| 181 |
# Write config.yaml
|
| 182 |
config_path = project_root / "config.yaml"
|
| 183 |
+
if config_path.exists():
|
| 184 |
+
overwrite = input(f"\n{config_path} already exists. Overwrite? [y/N]: ").strip().lower()
|
| 185 |
+
if overwrite != 'y':
|
| 186 |
+
print(" Keeping existing config.yaml.")
|
| 187 |
+
else:
|
| 188 |
+
config_str = generate_config(bot_name, domain, provider, model, web_search)
|
| 189 |
+
config_path.write_text(config_str)
|
| 190 |
+
print(f"\n Wrote {config_path}")
|
| 191 |
+
else:
|
| 192 |
+
config_str = generate_config(bot_name, domain, provider, model, web_search)
|
| 193 |
+
config_path.write_text(config_str)
|
| 194 |
+
print(f"\n Wrote {config_path}")
|
| 195 |
|
| 196 |
# Write .env (merges with existing keys if present)
|
| 197 |
env_path = project_root / ".env"
|
|
@@ -30,6 +30,12 @@ def load_config(config_path: str = None) -> dict:
|
|
| 30 |
with open(config_path, "r", encoding="utf-8") as f:
|
| 31 |
cfg = yaml.safe_load(f) or {}
|
| 32 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
# Normalize None-valued sections to empty dicts so chained .get() never
|
| 34 |
# fails with AttributeError (e.g. `llm:` with no sub-keys → None).
|
| 35 |
# Recurse into nested dicts so `paths:\n vector_db:` also gets normalized.
|
|
|
|
| 30 |
with open(config_path, "r", encoding="utf-8") as f:
|
| 31 |
cfg = yaml.safe_load(f) or {}
|
| 32 |
|
| 33 |
+
if not isinstance(cfg, dict):
|
| 34 |
+
raise ValueError(
|
| 35 |
+
f"Config file must contain a YAML mapping (dict), got {type(cfg).__name__}. "
|
| 36 |
+
"Run 'python setup.py' to regenerate config.yaml."
|
| 37 |
+
)
|
| 38 |
+
|
| 39 |
# Normalize None-valued sections to empty dicts so chained .get() never
|
| 40 |
# fails with AttributeError (e.g. `llm:` with no sub-keys → None).
|
| 41 |
# Recurse into nested dicts so `paths:\n vector_db:` also gets normalized.
|
|
@@ -24,8 +24,17 @@ def generate(system_prompt: str, user_message: str, cfg: dict,
|
|
| 24 |
if max_tokens is None:
|
| 25 |
max_tokens = llm_cfg.get("max_tokens", 2048)
|
| 26 |
# Validate parameters
|
| 27 |
-
|
| 28 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 29 |
return PROVIDERS[provider].generate(
|
| 30 |
system_prompt=system_prompt, user_message=user_message,
|
| 31 |
api_key=api_key, model=model, temperature=temperature, max_tokens=max_tokens,
|
|
|
|
| 24 |
if max_tokens is None:
|
| 25 |
max_tokens = llm_cfg.get("max_tokens", 2048)
|
| 26 |
# Validate parameters
|
| 27 |
+
try:
|
| 28 |
+
temperature = max(0.0, min(float(temperature), 2.0))
|
| 29 |
+
except (ValueError, TypeError):
|
| 30 |
+
temperature = 0.0
|
| 31 |
+
# Anthropic max temperature is 1.0
|
| 32 |
+
if provider == "anthropic":
|
| 33 |
+
temperature = min(temperature, 1.0)
|
| 34 |
+
try:
|
| 35 |
+
max_tokens = max(1, min(int(max_tokens), 128000))
|
| 36 |
+
except (ValueError, TypeError):
|
| 37 |
+
max_tokens = 2048
|
| 38 |
return PROVIDERS[provider].generate(
|
| 39 |
system_prompt=system_prompt, user_message=user_message,
|
| 40 |
api_key=api_key, model=model, temperature=temperature, max_tokens=max_tokens,
|
|
@@ -30,8 +30,8 @@ def generate(system_prompt: str, user_message: str, api_key: str,
|
|
| 30 |
return ""
|
| 31 |
return response.text
|
| 32 |
except ValueError as e:
|
| 33 |
-
# Safety filter block
|
| 34 |
-
return
|
| 35 |
except Exception as e:
|
| 36 |
error_type = type(e).__name__
|
| 37 |
error_msg = str(e)
|
|
|
|
| 30 |
return ""
|
| 31 |
return response.text
|
| 32 |
except ValueError as e:
|
| 33 |
+
# Safety filter block — return empty so verifier treats it as generation failure
|
| 34 |
+
return ""
|
| 35 |
except Exception as e:
|
| 36 |
error_type = type(e).__name__
|
| 37 |
error_msg = str(e)
|
|
@@ -7,12 +7,12 @@ def generate(system_prompt: str, user_message: str, api_key: str,
|
|
| 7 |
model = model or "gpt-4o"
|
| 8 |
from openai import OpenAI
|
| 9 |
client = OpenAI(api_key=api_key)
|
| 10 |
-
is_reasoning = model.startswith(("o1", "o3", "o4"))
|
| 11 |
try:
|
| 12 |
if is_reasoning:
|
| 13 |
# Reasoning models include thinking tokens in max_completion_tokens,
|
| 14 |
# so we need a larger budget to get enough output tokens.
|
| 15 |
-
reasoning_budget = max(max_tokens * 4, 4096)
|
| 16 |
messages = [
|
| 17 |
{"role": "developer", "content": system_prompt},
|
| 18 |
{"role": "user", "content": user_message},
|
|
|
|
| 7 |
model = model or "gpt-4o"
|
| 8 |
from openai import OpenAI
|
| 9 |
client = OpenAI(api_key=api_key)
|
| 10 |
+
is_reasoning = model in ("o1", "o3", "o4") or model.startswith(("o1-", "o3-", "o4-"))
|
| 11 |
try:
|
| 12 |
if is_reasoning:
|
| 13 |
# Reasoning models include thinking tokens in max_completion_tokens,
|
| 14 |
# so we need a larger budget to get enough output tokens.
|
| 15 |
+
reasoning_budget = min(max(max_tokens * 4, 4096), 128000)
|
| 16 |
messages = [
|
| 17 |
{"role": "developer", "content": system_prompt},
|
| 18 |
{"role": "user", "content": user_message},
|
|
@@ -299,10 +299,13 @@ def build_prompt(
|
|
| 299 |
else:
|
| 300 |
overview_section = ""
|
| 301 |
|
|
|
|
|
|
|
|
|
|
| 302 |
return SYSTEM_PROMPT_TEMPLATE.format(
|
| 303 |
bot_name=bot_name,
|
| 304 |
domain=domain,
|
| 305 |
-
context=
|
| 306 |
kb_overview_section=overview_section,
|
| 307 |
)
|
| 308 |
|
|
|
|
| 299 |
else:
|
| 300 |
overview_section = ""
|
| 301 |
|
| 302 |
+
# Sanitize context to prevent prompt injection via delimiter mimicry
|
| 303 |
+
safe_context = context.replace("=" * 69, "\u2261" * 69)
|
| 304 |
+
|
| 305 |
return SYSTEM_PROMPT_TEMPLATE.format(
|
| 306 |
bot_name=bot_name,
|
| 307 |
domain=domain,
|
| 308 |
+
context=safe_context,
|
| 309 |
kb_overview_section=overview_section,
|
| 310 |
)
|
| 311 |
|
|
@@ -2,23 +2,27 @@
|
|
| 2 |
|
| 3 |
|
| 4 |
def read_docx(file_path: str) -> list[dict]:
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
|
|
|
| 8 |
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
|
| 20 |
-
|
| 21 |
|
| 22 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
return []
|
| 24 |
-
return [{"page": 1, "text": "\n\n".join(paragraphs)}]
|
|
|
|
| 2 |
|
| 3 |
|
| 4 |
def read_docx(file_path: str) -> list[dict]:
|
| 5 |
+
try:
|
| 6 |
+
from docx import Document
|
| 7 |
+
doc = Document(file_path)
|
| 8 |
+
paragraphs = [p.text.strip() for p in doc.paragraphs if p.text.strip()]
|
| 9 |
|
| 10 |
+
# Extract embedded tables
|
| 11 |
+
for table in doc.tables:
|
| 12 |
+
for row in table.rows:
|
| 13 |
+
seen_elements = set()
|
| 14 |
+
cells = []
|
| 15 |
+
for cell in row.cells:
|
| 16 |
+
if id(cell._element) not in seen_elements:
|
| 17 |
+
seen_elements.add(id(cell._element))
|
| 18 |
+
if cell.text.strip():
|
| 19 |
+
cells.append(cell.text.strip())
|
| 20 |
+
if cells:
|
| 21 |
+
paragraphs.append(" | ".join(cells))
|
| 22 |
|
| 23 |
+
if not paragraphs:
|
| 24 |
+
return []
|
| 25 |
+
return [{"page": 1, "text": "\n\n".join(paragraphs)}]
|
| 26 |
+
except Exception as e:
|
| 27 |
+
print(f"Warning: Could not read {file_path}: {e}")
|
| 28 |
return []
|
|
|
|
@@ -2,11 +2,15 @@
|
|
| 2 |
MAX_CHUNK_CHARS = 6000
|
| 3 |
|
| 4 |
def read_excel(file_path: str) -> list[dict]:
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
|
| 11 |
def _read_xlsx(file_path: str) -> list[dict]:
|
| 12 |
import openpyxl
|
|
|
|
| 2 |
MAX_CHUNK_CHARS = 6000
|
| 3 |
|
| 4 |
def read_excel(file_path: str) -> list[dict]:
|
| 5 |
+
try:
|
| 6 |
+
from pathlib import Path
|
| 7 |
+
ext = Path(file_path).suffix.lower()
|
| 8 |
+
if ext == ".xls":
|
| 9 |
+
return _read_xls(file_path)
|
| 10 |
+
return _read_xlsx(file_path)
|
| 11 |
+
except Exception as e:
|
| 12 |
+
print(f"Warning: Could not read {file_path}: {e}")
|
| 13 |
+
return []
|
| 14 |
|
| 15 |
def _read_xlsx(file_path: str) -> list[dict]:
|
| 16 |
import openpyxl
|
|
@@ -2,11 +2,15 @@
|
|
| 2 |
|
| 3 |
|
| 4 |
def read_pdf(file_path: str) -> list[dict]:
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2 |
|
| 3 |
|
| 4 |
def read_pdf(file_path: str) -> list[dict]:
|
| 5 |
+
try:
|
| 6 |
+
from pypdf import PdfReader
|
| 7 |
+
reader = PdfReader(file_path)
|
| 8 |
+
pages = []
|
| 9 |
+
for i, page in enumerate(reader.pages):
|
| 10 |
+
text = page.extract_text()
|
| 11 |
+
if text and text.strip():
|
| 12 |
+
pages.append({"page": i + 1, "text": text.strip()})
|
| 13 |
+
return pages
|
| 14 |
+
except Exception as e:
|
| 15 |
+
print(f"Warning: Could not read {file_path}: {e}")
|
| 16 |
+
return []
|
|
@@ -2,34 +2,38 @@
|
|
| 2 |
MAX_CHUNK_CHARS = 6000
|
| 3 |
|
| 4 |
def read_rdata(file_path: str) -> list[dict]:
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 25 |
text = header_line + "\n".join(block)
|
| 26 |
pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
block_chars += len(row_text) + 1
|
| 32 |
-
if block:
|
| 33 |
-
text = header_line + "\n".join(block)
|
| 34 |
-
pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
|
| 35 |
-
return pages
|
|
|
|
| 2 |
MAX_CHUNK_CHARS = 6000
|
| 3 |
|
| 4 |
def read_rdata(file_path: str) -> list[dict]:
|
| 5 |
+
try:
|
| 6 |
+
import pyreadr
|
| 7 |
+
result = pyreadr.read_r(file_path)
|
| 8 |
+
pages = []
|
| 9 |
+
for name, df in result.items():
|
| 10 |
+
name_display = name if name is not None else "data"
|
| 11 |
+
headers = list(df.columns)
|
| 12 |
+
header_line = f"Object: {name_display} | Columns: {', '.join(headers)}\n"
|
| 13 |
+
block = []
|
| 14 |
+
block_chars = len(header_line)
|
| 15 |
+
block_start = 1
|
| 16 |
+
for idx, (_, row) in enumerate(df.iterrows()):
|
| 17 |
+
parts = []
|
| 18 |
+
for col in headers:
|
| 19 |
+
val = row[col]
|
| 20 |
+
if val is not None and str(val).strip() and str(val).lower() not in ("nan", "nat", "<na>", "inf", "-inf"):
|
| 21 |
+
parts.append(f"{col}: {val}")
|
| 22 |
+
if not parts:
|
| 23 |
+
continue
|
| 24 |
+
row_text = "; ".join(parts)
|
| 25 |
+
if block and block_chars + len(row_text) + 1 > MAX_CHUNK_CHARS:
|
| 26 |
+
text = header_line + "\n".join(block)
|
| 27 |
+
pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
|
| 28 |
+
block = []
|
| 29 |
+
block_chars = len(header_line)
|
| 30 |
+
block_start = idx + 1
|
| 31 |
+
block.append(row_text)
|
| 32 |
+
block_chars += len(row_text) + 1
|
| 33 |
+
if block:
|
| 34 |
text = header_line + "\n".join(block)
|
| 35 |
pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
|
| 36 |
+
return pages
|
| 37 |
+
except Exception as e:
|
| 38 |
+
print(f"Warning: Could not read {file_path}: {e}")
|
| 39 |
+
return []
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@@ -2,6 +2,10 @@
|
|
| 2 |
from src.readers.stata import _dataframe_to_pages
|
| 3 |
|
| 4 |
def read_spss(file_path: str) -> list[dict]:
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2 |
from src.readers.stata import _dataframe_to_pages
|
| 3 |
|
| 4 |
def read_spss(file_path: str) -> list[dict]:
|
| 5 |
+
try:
|
| 6 |
+
import pyreadstat
|
| 7 |
+
df, meta = pyreadstat.read_sav(file_path)
|
| 8 |
+
return _dataframe_to_pages(df, meta)
|
| 9 |
+
except Exception as e:
|
| 10 |
+
print(f"Warning: Could not read {file_path}: {e}")
|
| 11 |
+
return []
|
|
@@ -2,9 +2,13 @@
|
|
| 2 |
MAX_CHUNK_CHARS = 6000
|
| 3 |
|
| 4 |
def read_stata(file_path: str) -> list[dict]:
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 8 |
|
| 9 |
def _dataframe_to_pages(df, meta) -> list[dict]:
|
| 10 |
pages = []
|
|
|
|
| 2 |
MAX_CHUNK_CHARS = 6000
|
| 3 |
|
| 4 |
def read_stata(file_path: str) -> list[dict]:
|
| 5 |
+
try:
|
| 6 |
+
import pyreadstat
|
| 7 |
+
df, meta = pyreadstat.read_dta(file_path)
|
| 8 |
+
return _dataframe_to_pages(df, meta)
|
| 9 |
+
except Exception as e:
|
| 10 |
+
print(f"Warning: Could not read {file_path}: {e}")
|
| 11 |
+
return []
|
| 12 |
|
| 13 |
def _dataframe_to_pages(df, meta) -> list[dict]:
|
| 14 |
pages = []
|
|
@@ -7,6 +7,11 @@ from src.search import search, format_web_results_as_context
|
|
| 7 |
_collection_cache = {}
|
| 8 |
|
| 9 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
def _get_cached_collection(cfg: dict):
|
| 11 |
"""Get or create cached ChromaDB collection."""
|
| 12 |
from src.ingest import get_chroma_collection
|
|
@@ -208,7 +213,8 @@ def _build_fallback_sql_query(query: str, cfg: dict) -> str | None:
|
|
| 208 |
safe_word = word.replace("'", "''")
|
| 209 |
if not re.match(r'^\w+$', safe_word):
|
| 210 |
continue # Skip non-alphanumeric words
|
| 211 |
-
|
|
|
|
| 212 |
|
| 213 |
if conditions:
|
| 214 |
where_clause = " OR ".join(conditions)
|
|
@@ -283,8 +289,9 @@ def _try_alternate_columns(sql_query: str, cfg: dict) -> tuple[list[dict], str]
|
|
| 283 |
# Strip dangerous characters for defense-in-depth
|
| 284 |
if not re.match(r"^[\w\s.,'-]+$", safe_value):
|
| 285 |
safe_value = re.sub(r"[;'\\\"]", "", safe_value)
|
|
|
|
| 286 |
for col in text_cols:
|
| 287 |
-
alt_query = f'SELECT * FROM "{table_name}" WHERE "{col}" LIKE \'%{safe_value}%\' LIMIT {max_rows}'
|
| 288 |
rows = execute_sql_query(alt_query, cfg)
|
| 289 |
if rows:
|
| 290 |
return rows, alt_query
|
|
|
|
| 7 |
_collection_cache = {}
|
| 8 |
|
| 9 |
|
| 10 |
+
def clear_collection_cache():
|
| 11 |
+
"""Clear the cached ChromaDB collection so it's re-read after re-ingestion."""
|
| 12 |
+
_collection_cache.clear()
|
| 13 |
+
|
| 14 |
+
|
| 15 |
def _get_cached_collection(cfg: dict):
|
| 16 |
"""Get or create cached ChromaDB collection."""
|
| 17 |
from src.ingest import get_chroma_collection
|
|
|
|
| 213 |
safe_word = word.replace("'", "''")
|
| 214 |
if not re.match(r'^\w+$', safe_word):
|
| 215 |
continue # Skip non-alphanumeric words
|
| 216 |
+
safe_word = safe_word.replace('%', '\\%').replace('_', '\\_')
|
| 217 |
+
conditions.append(f'"{col}" LIKE \'%{safe_word}%\' ESCAPE \'\\\'')
|
| 218 |
|
| 219 |
if conditions:
|
| 220 |
where_clause = " OR ".join(conditions)
|
|
|
|
| 289 |
# Strip dangerous characters for defense-in-depth
|
| 290 |
if not re.match(r"^[\w\s.,'-]+$", safe_value):
|
| 291 |
safe_value = re.sub(r"[;'\\\"]", "", safe_value)
|
| 292 |
+
safe_value = safe_value.replace('%', '\\%').replace('_', '\\_')
|
| 293 |
for col in text_cols:
|
| 294 |
+
alt_query = f'SELECT * FROM "{table_name}" WHERE "{col}" LIKE \'%{safe_value}%\' ESCAPE \'\\\' LIMIT {max_rows}'
|
| 295 |
rows = execute_sql_query(alt_query, cfg)
|
| 296 |
if rows:
|
| 297 |
return rows, alt_query
|
|
@@ -3,12 +3,13 @@
|
|
| 3 |
import os
|
| 4 |
import re
|
| 5 |
import sqlite3
|
|
|
|
| 6 |
from pathlib import Path
|
| 7 |
|
| 8 |
|
| 9 |
_DANGEROUS_KEYWORDS = re.compile(
|
| 10 |
r"\b(INSERT|UPDATE|DELETE|DROP|ALTER|CREATE|ATTACH|DETACH|PRAGMA|"
|
| 11 |
-
r"LOAD_EXTENSION|UNION|
|
| 12 |
r"ROLLBACK|BEGIN|COMMIT|GRANT|REVOKE|EXPLAIN|WITH)\b",
|
| 13 |
re.IGNORECASE,
|
| 14 |
)
|
|
@@ -101,9 +102,16 @@ def execute_sql_query(sql_query: str, cfg: dict) -> list[dict]:
|
|
| 101 |
max_rows = cfg.get("sql", {}).get("max_rows", 200)
|
| 102 |
|
| 103 |
try:
|
| 104 |
-
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True)
|
| 105 |
try:
|
| 106 |
conn.row_factory = sqlite3.Row
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 107 |
cursor = conn.execute(sql_query)
|
| 108 |
rows = cursor.fetchmany(max_rows)
|
| 109 |
return [dict(row) for row in rows]
|
|
@@ -195,7 +203,7 @@ def make_fuzzy_query(sql: str, word_level: bool = False) -> str | None:
|
|
| 195 |
Returns the modified query, or None if no substitutions were made.
|
| 196 |
"""
|
| 197 |
pattern = re.compile(
|
| 198 |
-
r"""(["
|
| 199 |
)
|
| 200 |
|
| 201 |
if not word_level:
|
|
|
|
| 3 |
import os
|
| 4 |
import re
|
| 5 |
import sqlite3
|
| 6 |
+
import time
|
| 7 |
from pathlib import Path
|
| 8 |
|
| 9 |
|
| 10 |
_DANGEROUS_KEYWORDS = re.compile(
|
| 11 |
r"\b(INSERT|UPDATE|DELETE|DROP|ALTER|CREATE|ATTACH|DETACH|PRAGMA|"
|
| 12 |
+
r"LOAD_EXTENSION|UNION|VACUUM|REINDEX|SAVEPOINT|RELEASE|"
|
| 13 |
r"ROLLBACK|BEGIN|COMMIT|GRANT|REVOKE|EXPLAIN|WITH)\b",
|
| 14 |
re.IGNORECASE,
|
| 15 |
)
|
|
|
|
| 102 |
max_rows = cfg.get("sql", {}).get("max_rows", 200)
|
| 103 |
|
| 104 |
try:
|
| 105 |
+
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5)
|
| 106 |
try:
|
| 107 |
conn.row_factory = sqlite3.Row
|
| 108 |
+
# Abort long-running queries after 10 seconds
|
| 109 |
+
_start = time.monotonic()
|
| 110 |
+
def _progress_check():
|
| 111 |
+
if time.monotonic() - _start > 10:
|
| 112 |
+
return 1 # non-zero = abort
|
| 113 |
+
return 0
|
| 114 |
+
conn.set_progress_handler(_progress_check, 10000)
|
| 115 |
cursor = conn.execute(sql_query)
|
| 116 |
rows = cursor.fetchmany(max_rows)
|
| 117 |
return [dict(row) for row in rows]
|
|
|
|
| 203 |
Returns the modified query, or None if no substitutions were made.
|
| 204 |
"""
|
| 205 |
pattern = re.compile(
|
| 206 |
+
r"""("[^"]+"\s*|[A-Za-z_]\w*\s*)=\s*'([^']+)'""",
|
| 207 |
)
|
| 208 |
|
| 209 |
if not word_level:
|
|
@@ -160,14 +160,13 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
|
|
| 160 |
"""
|
| 161 |
warnings = []
|
| 162 |
|
| 163 |
-
# Count
|
|
|
|
|
|
|
|
|
|
|
|
|
| 164 |
db_results = retrieval_result.get("db_results", [])
|
| 165 |
-
|
| 166 |
-
for chunk in db_results:
|
| 167 |
-
source = chunk.get("metadata", {}).get("source", "")
|
| 168 |
-
if source:
|
| 169 |
-
unique_db_sources.add(source)
|
| 170 |
-
db_count = len(unique_db_sources) if unique_db_sources else (1 if db_results else 0)
|
| 171 |
web_count = len(retrieval_result.get("web_results", []))
|
| 172 |
sql_count = 1 if retrieval_result.get("sql_results") else 0
|
| 173 |
total_sources = db_count + web_count + sql_count
|
|
@@ -176,7 +175,7 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
|
|
| 176 |
return warnings
|
| 177 |
|
| 178 |
# Extract citation numbers from response BODY only (not References section)
|
| 179 |
-
refs_split = re.split(r'(?im)^#+\s*(references|sources)\
|
| 180 |
body_text = refs_split[0] if refs_split else response
|
| 181 |
citation_nums = set(int(m) for m in re.findall(r"\[(\d+)\]", body_text))
|
| 182 |
if not citation_nums:
|
|
@@ -191,7 +190,7 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
|
|
| 191 |
|
| 192 |
# Check that source file names from retrieval appear in the References section
|
| 193 |
refs_match = re.search(
|
| 194 |
-
r"(?im)(^#+\s*(references|sources)\
|
| 195 |
response, re.DOTALL,
|
| 196 |
)
|
| 197 |
if refs_match:
|
|
@@ -204,18 +203,25 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
|
|
| 204 |
filename = source.rsplit("/", 1)[-1].lower()
|
| 205 |
if filename in refs_text:
|
| 206 |
matched_sources += 1
|
| 207 |
-
# Also check SQL source
|
|
|
|
|
|
|
|
|
|
| 208 |
sql_results = retrieval_result.get("sql_results", [])
|
| 209 |
-
|
| 210 |
-
|
| 211 |
-
|
| 212 |
-
|
|
|
|
|
|
|
|
|
|
| 213 |
if sql_filename in refs_text:
|
| 214 |
matched_sources += 1
|
| 215 |
-
# Also check web source URLs
|
|
|
|
| 216 |
web_results = retrieval_result.get("web_results", [])
|
| 217 |
for web_chunk in web_results:
|
| 218 |
-
web_url = web_chunk.get("
|
| 219 |
if web_url and web_url.lower() in refs_text:
|
| 220 |
matched_sources += 1
|
| 221 |
# Only warn when db_results are present and are the primary source
|
|
@@ -349,7 +355,15 @@ def verify_and_respond(
|
|
| 349 |
user_message = query
|
| 350 |
|
| 351 |
# ── Generate initial response ─────────────────────────────────────────
|
| 352 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 353 |
|
| 354 |
# ── Guard: empty response from LLM ────────────────────────────────────
|
| 355 |
if not response or not response.strip():
|
|
@@ -393,19 +407,18 @@ def verify_and_respond(
|
|
| 393 |
max_iterations = verification_cfg.get("max_iterations", 3)
|
| 394 |
strict_mode = verification_cfg.get("strict_mode", True)
|
| 395 |
|
| 396 |
-
# Guard: max_iterations=0 with enabled=true →
|
|
|
|
|
|
|
| 397 |
if max_iterations <= 0:
|
| 398 |
-
print("WARNING: Verification iterations set to 0
|
| 399 |
-
"
|
| 400 |
-
|
| 401 |
-
|
| 402 |
-
"refused": False,
|
| 403 |
-
"verification_passed": None,
|
| 404 |
-
"iterations": 0,
|
| 405 |
-
}
|
| 406 |
|
| 407 |
# ── Verification loop ─────────────────────────────────────────────────
|
| 408 |
initial_response = response
|
|
|
|
| 409 |
for iteration in range(1, max_iterations + 1):
|
| 410 |
# Layer 5: Warning-phrase scan
|
| 411 |
phrase_flags = scan_warning_phrases(response)
|
|
@@ -434,6 +447,7 @@ def verify_and_respond(
|
|
| 434 |
except Exception:
|
| 435 |
# Verification LLM failed — skip correction, continue to next iteration
|
| 436 |
continue
|
|
|
|
| 437 |
vr = parse_verification_result(raw_verification)
|
| 438 |
|
| 439 |
if vr.get("pass", False):
|
|
@@ -448,14 +462,16 @@ def verify_and_respond(
|
|
| 448 |
error_count = vr.get("error_count", len(vr.get("errors", [])))
|
| 449 |
|
| 450 |
# Correct and loop for re-verification
|
|
|
|
|
|
|
| 451 |
correction_prompt = (
|
| 452 |
f"The user's original question was: {query}\n\n"
|
| 453 |
-
f"Your previous response
|
| 454 |
f"This response failed verification with "
|
| 455 |
f"{error_count} error(s):\n"
|
| 456 |
+ json.dumps(vr.get("errors", []), indent=2)
|
| 457 |
-
+ "\n\nRewrite your response to fix ALL the issues
|
| 458 |
-
"Use ONLY the provided context. Keep all citation rules."
|
| 459 |
)
|
| 460 |
try:
|
| 461 |
response = generate(
|
|
@@ -484,6 +500,7 @@ def verify_and_respond(
|
|
| 484 |
"You are a strict verification agent. Return only JSON.",
|
| 485 |
vp, cfg, max_tokens=1024,
|
| 486 |
))
|
|
|
|
| 487 |
except Exception:
|
| 488 |
vr = {"pass": False, "errors": [], "error_count": 0}
|
| 489 |
if vr.get("pass", False):
|
|
@@ -494,7 +511,39 @@ def verify_and_respond(
|
|
| 494 |
"iterations": max_iterations,
|
| 495 |
}
|
| 496 |
|
| 497 |
-
# ──
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 498 |
if strict_mode:
|
| 499 |
return {
|
| 500 |
"response": REFUSAL_AFTER_VERIFICATION,
|
|
|
|
| 160 |
"""
|
| 161 |
warnings = []
|
| 162 |
|
| 163 |
+
# Count citable units — each retrieved chunk is a separate citable
|
| 164 |
+
# reference that the LLM may cite as [1], [2], etc. Counting unique
|
| 165 |
+
# *files* instead would trigger false "fabricated reference" warnings
|
| 166 |
+
# when a single file contributes multiple chunks (e.g., a PDF cited
|
| 167 |
+
# as [1]–[10] for 10 different sections).
|
| 168 |
db_results = retrieval_result.get("db_results", [])
|
| 169 |
+
db_count = len(db_results)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 170 |
web_count = len(retrieval_result.get("web_results", []))
|
| 171 |
sql_count = 1 if retrieval_result.get("sql_results") else 0
|
| 172 |
total_sources = db_count + web_count + sql_count
|
|
|
|
| 175 |
return warnings
|
| 176 |
|
| 177 |
# Extract citation numbers from response BODY only (not References section)
|
| 178 |
+
refs_split = re.split(r'(?im)^#+\s*(?:references|sources)\s*$|^\*\*(?:references|sources)\*\*\s*$', response)
|
| 179 |
body_text = refs_split[0] if refs_split else response
|
| 180 |
citation_nums = set(int(m) for m in re.findall(r"\[(\d+)\]", body_text))
|
| 181 |
if not citation_nums:
|
|
|
|
| 190 |
|
| 191 |
# Check that source file names from retrieval appear in the References section
|
| 192 |
refs_match = re.search(
|
| 193 |
+
r"(?im)(?:^#+\s*(?:references|sources)\s*$|^\*\*(?:references|sources)\*\*\s*$).*",
|
| 194 |
response, re.DOTALL,
|
| 195 |
)
|
| 196 |
if refs_match:
|
|
|
|
| 203 |
filename = source.rsplit("/", 1)[-1].lower()
|
| 204 |
if filename in refs_text:
|
| 205 |
matched_sources += 1
|
| 206 |
+
# Also check SQL source — SQL results are plain row dicts (no
|
| 207 |
+
# metadata wrapper). All rows come from a single table/source file
|
| 208 |
+
# whose name is embedded in the context as "Source: <filename>".
|
| 209 |
+
# Extract it from the combined context instead of iterating rows.
|
| 210 |
sql_results = retrieval_result.get("sql_results", [])
|
| 211 |
+
if sql_results:
|
| 212 |
+
context_text = retrieval_result.get("context", "")
|
| 213 |
+
sql_source_match = re.search(
|
| 214 |
+
r"Source:\s*(.+)", context_text,
|
| 215 |
+
)
|
| 216 |
+
if sql_source_match:
|
| 217 |
+
sql_filename = sql_source_match.group(1).strip().rsplit("/", 1)[-1].lower()
|
| 218 |
if sql_filename in refs_text:
|
| 219 |
matched_sources += 1
|
| 220 |
+
# Also check web source URLs — web results are flat dicts with
|
| 221 |
+
# a top-level "url" key (no metadata wrapper).
|
| 222 |
web_results = retrieval_result.get("web_results", [])
|
| 223 |
for web_chunk in web_results:
|
| 224 |
+
web_url = web_chunk.get("url", "")
|
| 225 |
if web_url and web_url.lower() in refs_text:
|
| 226 |
matched_sources += 1
|
| 227 |
# Only warn when db_results are present and are the primary source
|
|
|
|
| 355 |
user_message = query
|
| 356 |
|
| 357 |
# ── Generate initial response ─────────────────────────────────────────
|
| 358 |
+
try:
|
| 359 |
+
response = generate(system_prompt, user_message, cfg, max_tokens=soft_max)
|
| 360 |
+
except Exception as e:
|
| 361 |
+
return {
|
| 362 |
+
"response": f"An error occurred while generating the response: {e}",
|
| 363 |
+
"refused": True,
|
| 364 |
+
"verification_passed": None,
|
| 365 |
+
"iterations": 0,
|
| 366 |
+
}
|
| 367 |
|
| 368 |
# ── Guard: empty response from LLM ────────────────────────────────────
|
| 369 |
if not response or not response.strip():
|
|
|
|
| 407 |
max_iterations = verification_cfg.get("max_iterations", 3)
|
| 408 |
strict_mode = verification_cfg.get("strict_mode", True)
|
| 409 |
|
| 410 |
+
# Guard: max_iterations=0 with enabled=true → clamp to 1.
|
| 411 |
+
# Verification cannot be bypassed via iteration count alone;
|
| 412 |
+
# users must explicitly set verification.enabled: false.
|
| 413 |
if max_iterations <= 0:
|
| 414 |
+
print("WARNING: Verification iterations set to 0 but verification "
|
| 415 |
+
"is enabled. Clamping to 1 iteration. To disable verification, "
|
| 416 |
+
"set verification.enabled: false.")
|
| 417 |
+
max_iterations = 1
|
|
|
|
|
|
|
|
|
|
|
|
|
| 418 |
|
| 419 |
# ── Verification loop ─────────────────────────────────────────────────
|
| 420 |
initial_response = response
|
| 421 |
+
any_verification_ran = False # Track whether ANY verification call succeeded
|
| 422 |
for iteration in range(1, max_iterations + 1):
|
| 423 |
# Layer 5: Warning-phrase scan
|
| 424 |
phrase_flags = scan_warning_phrases(response)
|
|
|
|
| 447 |
except Exception:
|
| 448 |
# Verification LLM failed — skip correction, continue to next iteration
|
| 449 |
continue
|
| 450 |
+
any_verification_ran = True
|
| 451 |
vr = parse_verification_result(raw_verification)
|
| 452 |
|
| 453 |
if vr.get("pass", False):
|
|
|
|
| 462 |
error_count = vr.get("error_count", len(vr.get("errors", [])))
|
| 463 |
|
| 464 |
# Correct and loop for re-verification
|
| 465 |
+
# Truncate previous response to reduce prompt competition with context
|
| 466 |
+
truncated_response = response[:1500] + "..." if len(response) > 1500 else response
|
| 467 |
correction_prompt = (
|
| 468 |
f"The user's original question was: {query}\n\n"
|
| 469 |
+
f"Your previous response (truncated for brevity):\n{truncated_response}\n\n"
|
| 470 |
f"This response failed verification with "
|
| 471 |
f"{error_count} error(s):\n"
|
| 472 |
+ json.dumps(vr.get("errors", []), indent=2)
|
| 473 |
+
+ "\n\nRewrite your COMPLETE response from scratch to fix ALL the issues "
|
| 474 |
+
"listed above. Use ONLY the provided context. Keep all citation rules."
|
| 475 |
)
|
| 476 |
try:
|
| 477 |
response = generate(
|
|
|
|
| 500 |
"You are a strict verification agent. Return only JSON.",
|
| 501 |
vp, cfg, max_tokens=1024,
|
| 502 |
))
|
| 503 |
+
any_verification_ran = True
|
| 504 |
except Exception:
|
| 505 |
vr = {"pass": False, "errors": [], "error_count": 0}
|
| 506 |
if vr.get("pass", False):
|
|
|
|
| 511 |
"iterations": max_iterations,
|
| 512 |
}
|
| 513 |
|
| 514 |
+
# ── Handle total verification system failure ──────────────────────────
|
| 515 |
+
# If ALL verification LLM calls failed (not "didn't pass" — actually
|
| 516 |
+
# failed to run), this is a system failure, not a content-quality issue.
|
| 517 |
+
if not any_verification_ran:
|
| 518 |
+
print("ERROR: All verification LLM calls failed. "
|
| 519 |
+
"Verification could not run at all.")
|
| 520 |
+
if strict_mode:
|
| 521 |
+
return {
|
| 522 |
+
"response": (
|
| 523 |
+
"Verification was unable to run due to repeated LLM "
|
| 524 |
+
"errors. To avoid presenting unverified information, "
|
| 525 |
+
"I must decline to answer. Please check your LLM "
|
| 526 |
+
"configuration and try again."
|
| 527 |
+
),
|
| 528 |
+
"refused": True,
|
| 529 |
+
"verification_passed": False,
|
| 530 |
+
"iterations": max_iterations,
|
| 531 |
+
}
|
| 532 |
+
# Non-strict: return with a prominent system-failure warning
|
| 533 |
+
warning = (
|
| 534 |
+
"\n\n---\n**WARNING: Verification system failure.** "
|
| 535 |
+
"The verification pipeline was unable to run due to "
|
| 536 |
+
"repeated LLM errors. This response has NOT been verified "
|
| 537 |
+
"at all — treat all claims as unverified."
|
| 538 |
+
)
|
| 539 |
+
return {
|
| 540 |
+
"response": response + warning,
|
| 541 |
+
"refused": False,
|
| 542 |
+
"verification_passed": False,
|
| 543 |
+
"iterations": max_iterations,
|
| 544 |
+
}
|
| 545 |
+
|
| 546 |
+
# ── Exhausted iterations (verification ran but never passed) ──────────
|
| 547 |
if strict_mode:
|
| 548 |
return {
|
| 549 |
"response": REFUSAL_AFTER_VERIFICATION,
|
|
@@ -203,10 +203,10 @@ def test_retrieve_vector_route_fallback_still_works(
|
|
| 203 |
@patch("src.retriever.retrieve_from_vectordb", return_value=[])
|
| 204 |
@patch("src.retriever._run_sql_retrieval")
|
| 205 |
@patch("src.retriever._build_fallback_sql_query")
|
| 206 |
-
def
|
| 207 |
mock_build_fallback, mock_run_sql, mock_vectordb, mock_search,
|
| 208 |
):
|
| 209 |
-
"""
|
| 210 |
from src.retriever import retrieve
|
| 211 |
|
| 212 |
# First call: LLM sql_query fails; second call: keyword fallback also fails
|
|
@@ -215,10 +215,10 @@ def test_retrieve_both_route_fallback_still_works(
|
|
| 215 |
|
| 216 |
result = retrieve("some data", _BASE_CFG, route="both", sql_query="SELECT * FROM t WHERE x=1")
|
| 217 |
|
| 218 |
-
#
|
| 219 |
-
#
|
| 220 |
-
mock_build_fallback.
|
| 221 |
-
# _run_sql_retrieval called twice: once for LLM query, once for
|
| 222 |
assert mock_run_sql.call_count == 2
|
| 223 |
|
| 224 |
|
|
|
|
| 203 |
@patch("src.retriever.retrieve_from_vectordb", return_value=[])
|
| 204 |
@patch("src.retriever._run_sql_retrieval")
|
| 205 |
@patch("src.retriever._build_fallback_sql_query")
|
| 206 |
+
def test_retrieve_both_route_fallback_builds_fresh_query(
|
| 207 |
mock_build_fallback, mock_run_sql, mock_vectordb, mock_search,
|
| 208 |
):
|
| 209 |
+
"""When route='both' and LLM sql_query failed, build a fresh keyword query."""
|
| 210 |
from src.retriever import retrieve
|
| 211 |
|
| 212 |
# First call: LLM sql_query fails; second call: keyword fallback also fails
|
|
|
|
| 215 |
|
| 216 |
result = retrieve("some data", _BASE_CFG, route="both", sql_query="SELECT * FROM t WHERE x=1")
|
| 217 |
|
| 218 |
+
# T2-03 fix: _build_fallback_sql_query IS called to build a fresh keyword
|
| 219 |
+
# query instead of re-using the already-failed LLM sql_query
|
| 220 |
+
mock_build_fallback.assert_called_once_with("some data", _BASE_CFG)
|
| 221 |
+
# _run_sql_retrieval called twice: once for LLM query, once for fresh fallback
|
| 222 |
assert mock_run_sql.call_count == 2
|
| 223 |
|
| 224 |
|