LIANJie-Jason Claude Opus 4.6 commited on
Commit
f95cdb7
·
1 Parent(s): 3710279

fix: Round 14 — 28 Tier 1 + Tier 2 bugs fixed across 17 source files

Browse files

Anti-hallucination integrity (T1):
- #100: Detect verification LLM system failure (all calls fail → refuse)
- #101: Fix validate_citations SQL/web source matching (dead code → working)
- #102: Citation count uses chunks (citable units) not unique files
- #103: Clamp max_iterations to min 1 (can't bypass via iterations=0)

Pre-release quality (T2 HIGH):
- #104: Add LIKE ESCAPE in all SQL fallback paths
- #105: Gemini safety block returns "" (was bypassing stack)
- #106: Cap reasoning model budget at 128000
- #107: Handle non-numeric temperature/max_tokens gracefully
- #108: Fix clarification history duplication

Pre-release quality (T2 MEDIUM):
- #109: Anthropic temperature clamp to 1.0
- #110: SQL query 10-second timeout via progress handler
- #111: Rich markup escape for user/LLM content
- #112: All 6 readers wrapped in try/except (corrupt files don't abort)
- #113: Collection cache invalidation after re-ingestion
- #114: Web UI message history capped at 200
- #115: Setup wizard config overwrite confirmation
- #116: Prompt injection defense (delimiter sanitization)
- #117: References section regex precision
- #118: Correction prompt truncation (1500 chars)
- #119: Initial generate() wrapped in try/except
- #120: Reasoning model detection (exact match + prefix)
- #121: make_fuzzy_query regex precision
- #122: REPLACE removed from dangerous keywords (function is read-only)
- #123: Non-dict YAML config rejection
- #124: Web UI model cache invalidation on provider change
- #125: Web UI uses fast file-based KB brief (was slow LLM call)
- #126: Re-ingestion progress message
- #127: Empty API key warning improved

Cross-checked by regression checker + design guardian agents.
Design score: 8.5 → 9.1/10. Tests: 174/174 passing.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>

app_cli.py CHANGED
@@ -3,6 +3,7 @@
3
  import sys
4
 
5
  from rich.console import Console
 
6
  from rich.panel import Panel
7
  from rich.markdown import Markdown
8
  from rich.text import Text
@@ -11,7 +12,7 @@ from src.config_loader import load_config, get_api_key
11
  from src.ingest import ingest_documents
12
  from src.kb_meta import load_kb_meta_brief
13
  from src.query_engine import understand_query
14
- from src.retriever import retrieve
15
  from src.verifier import verify_and_respond
16
  from src.llm import list_models
17
 
@@ -256,9 +257,10 @@ def main() -> None:
256
  with console.status("Ingesting documents..."):
257
  try:
258
  count = ingest_documents(cfg)
 
259
  console.print(f"[green]Ingestion complete: {count} chunks.[/green]")
260
  except Exception as e:
261
- console.print(f"[red]Ingestion error: {e}[/red]")
262
  continue
263
 
264
  if result == "__MODEL__":
@@ -301,7 +303,7 @@ def main() -> None:
301
  state.get("conversation_history", []),
302
  )
303
  except Exception as e:
304
- console.print(f"[yellow]Query understanding failed, using raw query: {e}[/yellow]")
305
  qu_result = {"action": "search", "search_query": user_input, "display_query": user_input, "original_query": user_input, "route": "vector", "sql_query": None}
306
 
307
  # Handle clarification
@@ -344,7 +346,7 @@ def main() -> None:
344
 
345
  # Show reformulated query if different from original
346
  if search_query != user_input:
347
- console.print(f"[dim]Searching for: \"{search_query}\"[/dim]")
348
  if route in ("sql", "both"):
349
  console.print(f"[dim]Using SQL query for structured data[/dim]")
350
 
@@ -353,7 +355,7 @@ def main() -> None:
353
  try:
354
  retrieval_result = retrieve(search_query, effective_cfg, route=route, sql_query=sql_query)
355
  except Exception as e:
356
- console.print(f"[red]Retrieval error: {e}[/red]")
357
  continue
358
 
359
  state["last_retrieval"] = retrieval_result
@@ -366,13 +368,17 @@ def main() -> None:
366
  original_query=user_input,
367
  )
368
  except Exception as e:
369
- console.print(f"[red]Generation error: {e}[/red]")
370
  continue
371
 
372
  # ── Update conversation history ─────────────────────────────────
373
  history = state.get("conversation_history", [])
374
  max_history = qu_cfg.get("max_history", 6)
375
- history.append({"role": "user", "content": user_input})
 
 
 
 
376
  history.append({"role": "assistant", "content": result.get("response", "")})
377
  state["conversation_history"] = history[-max_history:]
378
 
 
3
  import sys
4
 
5
  from rich.console import Console
6
+ from rich.markup import escape as rich_escape
7
  from rich.panel import Panel
8
  from rich.markdown import Markdown
9
  from rich.text import Text
 
12
  from src.ingest import ingest_documents
13
  from src.kb_meta import load_kb_meta_brief
14
  from src.query_engine import understand_query
15
+ from src.retriever import clear_collection_cache, retrieve
16
  from src.verifier import verify_and_respond
17
  from src.llm import list_models
18
 
 
257
  with console.status("Ingesting documents..."):
258
  try:
259
  count = ingest_documents(cfg)
260
+ clear_collection_cache()
261
  console.print(f"[green]Ingestion complete: {count} chunks.[/green]")
262
  except Exception as e:
263
+ console.print(f"[red]Ingestion error: {rich_escape(str(e))}[/red]")
264
  continue
265
 
266
  if result == "__MODEL__":
 
303
  state.get("conversation_history", []),
304
  )
305
  except Exception as e:
306
+ console.print(f"[yellow]Query understanding failed, using raw query: {rich_escape(str(e))}[/yellow]")
307
  qu_result = {"action": "search", "search_query": user_input, "display_query": user_input, "original_query": user_input, "route": "vector", "sql_query": None}
308
 
309
  # Handle clarification
 
346
 
347
  # Show reformulated query if different from original
348
  if search_query != user_input:
349
+ console.print(f"[dim]Searching for: \"{rich_escape(search_query)}\"[/dim]")
350
  if route in ("sql", "both"):
351
  console.print(f"[dim]Using SQL query for structured data[/dim]")
352
 
 
355
  try:
356
  retrieval_result = retrieve(search_query, effective_cfg, route=route, sql_query=sql_query)
357
  except Exception as e:
358
+ console.print(f"[red]Retrieval error: {rich_escape(str(e))}[/red]")
359
  continue
360
 
361
  state["last_retrieval"] = retrieval_result
 
368
  original_query=user_input,
369
  )
370
  except Exception as e:
371
+ console.print(f"[red]Generation error: {rich_escape(str(e))}[/red]")
372
  continue
373
 
374
  # ── Update conversation history ─────────────────────────────────
375
  history = state.get("conversation_history", [])
376
  max_history = qu_cfg.get("max_history", 6)
377
+ # Use display_query (post-QU) as the user message for history,
378
+ # since it captures clarification context and avoids duplicating
379
+ # the raw user_input that clarification already added.
380
+ effective_user_msg = display_query if display_query != user_input else user_input
381
+ history.append({"role": "user", "content": effective_user_msg})
382
  history.append({"role": "assistant", "content": result.get("response", "")})
383
  state["conversation_history"] = history[-max_history:]
384
 
app_web.py CHANGED
@@ -4,9 +4,9 @@ import streamlit as st
4
 
5
  from src.config_loader import load_config, get_api_key
6
  from src.ingest import get_chroma_collection, ingest_documents
7
- from src.kb_meta import summarize_kb_for_welcome
8
  from src.query_engine import understand_query
9
- from src.retriever import retrieve
10
  from src.verifier import verify_and_respond
11
  from src.llm import list_models
12
 
@@ -59,6 +59,13 @@ def render_sidebar():
59
  cfg.setdefault("llm", {})["provider"] = provider
60
 
61
  # --- Model dropdown (cached per provider) ---
 
 
 
 
 
 
 
62
  models_cache_key = f"models_{provider}"
63
  if models_cache_key not in st.session_state:
64
  api_key = get_api_key(cfg, provider)
@@ -115,9 +122,11 @@ def render_sidebar():
115
 
116
  # --- Re-ingest button ---
117
  if st.button("Re-ingest documents", use_container_width=True):
 
118
  with st.spinner("Ingesting documents..."):
119
  try:
120
  count = ingest_documents(cfg)
 
121
  st.success(f"Ingested {count} chunks.")
122
  # Clear cached data so it refreshes after re-ingest
123
  st.session_state.pop("kb_welcome_summary", None)
@@ -279,6 +288,11 @@ def render_chat():
279
  {"role": "assistant", "content": result.get("response", "")}
280
  )
281
 
 
 
 
 
 
282
 
283
  # ── Main ─────────────────────────────────────────────────────────────────────
284
 
@@ -300,7 +314,7 @@ def main():
300
  # Show KB summary on first visit (LLM-generated welcome summary)
301
  if not st.session_state.messages:
302
  if "kb_welcome_summary" not in st.session_state:
303
- st.session_state.kb_welcome_summary = summarize_kb_for_welcome(cfg)
304
  kb_summary = st.session_state.kb_welcome_summary
305
  if kb_summary:
306
  with st.expander("Knowledge Base Contents", expanded=True):
 
4
 
5
  from src.config_loader import load_config, get_api_key
6
  from src.ingest import get_chroma_collection, ingest_documents
7
+ from src.kb_meta import load_kb_meta_brief
8
  from src.query_engine import understand_query
9
+ from src.retriever import clear_collection_cache, retrieve
10
  from src.verifier import verify_and_respond
11
  from src.llm import list_models
12
 
 
59
  cfg.setdefault("llm", {})["provider"] = provider
60
 
61
  # --- Model dropdown (cached per provider) ---
62
+ # Invalidate model cache if provider changed
63
+ prev_provider_key = "prev_provider"
64
+ if st.session_state.get(prev_provider_key) != provider:
65
+ for p in providers:
66
+ st.session_state.pop(f"models_{p}", None)
67
+ st.session_state[prev_provider_key] = provider
68
+
69
  models_cache_key = f"models_{provider}"
70
  if models_cache_key not in st.session_state:
71
  api_key = get_api_key(cfg, provider)
 
122
 
123
  # --- Re-ingest button ---
124
  if st.button("Re-ingest documents", use_container_width=True):
125
+ st.info("Ingestion may take a few minutes for large document collections...")
126
  with st.spinner("Ingesting documents..."):
127
  try:
128
  count = ingest_documents(cfg)
129
+ clear_collection_cache()
130
  st.success(f"Ingested {count} chunks.")
131
  # Clear cached data so it refreshes after re-ingest
132
  st.session_state.pop("kb_welcome_summary", None)
 
288
  {"role": "assistant", "content": result.get("response", "")}
289
  )
290
 
291
+ # Cap message history to prevent unbounded memory growth
292
+ max_messages = 200 # 100 Q&A pairs
293
+ if len(st.session_state.messages) > max_messages:
294
+ st.session_state.messages = st.session_state.messages[-max_messages:]
295
+
296
 
297
  # ── Main ─────────────────────────────────────────────────────────────────────
298
 
 
314
  # Show KB summary on first visit (LLM-generated welcome summary)
315
  if not st.session_state.messages:
316
  if "kb_welcome_summary" not in st.session_state:
317
+ st.session_state.kb_welcome_summary = load_kb_meta_brief(cfg)
318
  kb_summary = st.session_state.kb_welcome_summary
319
  if kb_summary:
320
  with st.expander("Knowledge Base Contents", expanded=True):
docs/BUGFIX_LOG.md CHANGED
@@ -186,7 +186,7 @@ severity, file, root cause, fix applied, and the audit round that found it.
186
  | K12 | LOW | Multiple | `cfg.get("section", {})` systemic pattern — if YAML section exists but is `None`, returns `None` not `{}` | Partially mitigated by config_loader normalization; remaining ~30 locations are low risk |
187
  | K6 | LOW | `src/search/semantic_scholar.py` | Filters out papers without abstracts, reducing result count for some topics | By design — abstract-less papers can't provide useful context |
188
  | K7 | LOW | `src/retriever.py` | LIKE wildcards `_`/`%` not escaped in fallback SQL query | Minor — extra matches unlikely to cause visible problems |
189
- | K8 | LOW | `CLAUDE.md` | Test count says 162, should be 165 | Doc update needed |
190
  | K9 | LOW | `src/verifier.py` | `compute_similarity_flags` regex `[a-z]{3,}` excludes non-ASCII and 2-letter acronyms (AI, US, UK) | Advisory-only layer; impact limited to noisy flags |
191
 
192
  ---
@@ -240,7 +240,7 @@ This round was a comprehensive pre-release inspection using 6 parallel audit age
240
 
241
  | # | Severity | File | Description | Reason |
242
  |---|----------|------|-------------|--------|
243
- | K13 | HIGH | `src/llm/gemini.py` + `requirements.txt` | **CON-4**: `google-generativeai` package fully deprecated; replacement `google-genai` has different API surface | Requires full API migration; functional for now but will break when PyPI removes package |
244
  | K14 | MEDIUM | `src/verifier.py` | **AH-05**: Token cap step function too generous for small contexts (501 chars → 1536 tokens = 12:1 ratio) | Would benefit from continuous formula; current 3-tier approach is functional |
245
  | K15 | MEDIUM | `src/verifier.py:24-33` | **AH-11**: Warning phrase list missing common patterns: "studies have shown", "research suggests", "historically" | Advisory layer only; expandable post-release |
246
  | K16 | LOW | `src/verifier.py:184` | **INSPECT-2**: References section regex matches "sources" in prose, not just section header | Mitigated: filename matching still works because match includes everything to end of response |
@@ -276,13 +276,212 @@ User-reported bug: querying "south korea in pts" returned zero SQL results despi
276
 
277
  ---
278
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
279
  ## Statistics
280
 
281
  | Metric | Count |
282
  |--------|-------|
283
- | Total bugs fixed | ~119 (3 this round) |
284
- | Audit rounds completed | 10 (10 sessions, 30+ total agent dispatches) |
 
 
 
285
  | Test count | 174 passing |
286
  | Test files | 17 |
287
- | Source files audited | All 20+ |
288
- | Known issues remaining | 17 (0 CRITICAL, 1 HIGH, 6 MEDIUM, 10 LOW) |
 
 
186
  | K12 | LOW | Multiple | `cfg.get("section", {})` systemic pattern — if YAML section exists but is `None`, returns `None` not `{}` | Partially mitigated by config_loader normalization; remaining ~30 locations are low risk |
187
  | K6 | LOW | `src/search/semantic_scholar.py` | Filters out papers without abstracts, reducing result count for some topics | By design — abstract-less papers can't provide useful context |
188
  | K7 | LOW | `src/retriever.py` | LIKE wildcards `_`/`%` not escaped in fallback SQL query | Minor — extra matches unlikely to cause visible problems |
189
+ | K8 | LOW | `CLAUDE.md` | Test count said 162, actual is 174 | FIXED (Round 11 doc update) |
190
  | K9 | LOW | `src/verifier.py` | `compute_similarity_flags` regex `[a-z]{3,}` excludes non-ASCII and 2-letter acronyms (AI, US, UK) | Advisory-only layer; impact limited to noisy flags |
191
 
192
  ---
 
240
 
241
  | # | Severity | File | Description | Reason |
242
  |---|----------|------|-------------|--------|
243
+ | K13 | ~~HIGH~~ | `src/llm/gemini.py` + `requirements.txt` | **CON-4**: `google-generativeai` package fully deprecated; replacement `google-genai` has different API surface | **FIXED** — Migrated to `google-genai>=1.0.0` SDK with `genai.Client(api_key=...)` (thread-safe, no global state) |
244
  | K14 | MEDIUM | `src/verifier.py` | **AH-05**: Token cap step function too generous for small contexts (501 chars → 1536 tokens = 12:1 ratio) | Would benefit from continuous formula; current 3-tier approach is functional |
245
  | K15 | MEDIUM | `src/verifier.py:24-33` | **AH-11**: Warning phrase list missing common patterns: "studies have shown", "research suggests", "historically" | Advisory layer only; expandable post-release |
246
  | K16 | LOW | `src/verifier.py:184` | **INSPECT-2**: References section regex matches "sources" in prose, not just section header | Mitigated: filename matching still works because match includes everything to end of response |
 
276
 
277
  ---
278
 
279
+ ## Round 11 — 2026-03-06 Session 6 (8-Agent Final Pre-Release Audit + Cross-Check)
280
+
281
+ This round was the strictest final audit before public release: 8 parallel agents (core pipeline, ingestion/storage, SQL security, LLM/prompts, UIs/setup, readers/search, test suite, design docs/architecture) each read every line of their assigned files and cross-checked findings.
282
+
283
+ ### HIGH Fixes
284
+
285
+ | # | Severity | File | Bug | Fix | Status |
286
+ |---|----------|------|-----|-----|--------|
287
+ | 84 | HIGH | `src/query_engine.py:96` | QU `max_tokens=256` too tight — SQL queries + reasoning fields truncate JSON output, causing silent fallback to vector-only search | Increased to `max_tokens=512` | FIXED |
288
+ | 85 | HIGH | `src/llm/gemini.py:25-26` | Gemini `generate()` returned error strings instead of raising exceptions, inconsistent with OpenAI/Anthropic — error strings leaked into correction loop as "responses" | Changed to `raise RuntimeError(...)` for API errors (ValueError for safety blocks still returns string) | FIXED |
289
+ | 86 | HIGH | `src/sql_retriever.py:184-214` | `make_fuzzy_query()` unsanitized value interpolation — captured regex values not escaped for single quotes, allowing SQL injection via crafted `WHERE` values | Added `.replace("'", "''")` escaping in both `_phrase_replace()` and `_word_replace()` | FIXED |
290
+ | 87 | HIGH | `src/config_loader.py:35-37` | Nested null YAML values crashed consumers — `vector_db:` with no value → `.get("vector_db", "chroma_db")` returns `None` → `TypeError` in `os.path.isabs(None)` | Changed top-level-only normalization to recursive `_normalize_nulls()` that handles all nested dicts | FIXED |
291
+ | 88 | HIGH | `setup.py:93-94` | `.env` parser used `strip('"')` which removes ALL leading/trailing quotes, not just one matched pair — could corrupt API keys starting/ending with quote chars | Changed to proper one-pair unwrap: check for matching pair then slice `[1:-1]` | FIXED |
292
+ | 89 | HIGH | `setup.py:100-102` | `.env` writer did not escape special chars in values — API keys containing `"` or `\` produced malformed `.env` files | Added `v.replace('\\', '\\\\').replace('"', '\\"')` before writing | FIXED |
293
+ | 90 | HIGH | `src/sql_retriever.py:41-42` | `sqlite_master` access not blocked — `SELECT * FROM sqlite_master` passed all validation, exposing full DDL schema | Added `sqlite_(master\|schema\|temp_master\|temp_schema)` check in `_validate_sql()` | FIXED |
294
+ | 91 | HIGH | `tests/test_verifier.py:65-66,95-96` | Wrong mock target — patched `src.kb_meta.load_kb_meta` but verifier imports `load_kb_meta_brief`; mocks were no-ops (only worked due to fallback chain) | Changed both patches to `src.kb_meta.load_kb_meta_brief` | FIXED |
295
+ | 92 | HIGH | `src/retriever.py:381` | SQL keyword fallback ran unconditionally for `route="sql"` even when vector search already found results — wasted computation | Added `and not db_results` condition: only run keyword SQL when vector fallback also found nothing | FIXED |
296
+
297
+ ### MEDIUM Fixes
298
+
299
+ | # | Severity | File | Bug | Fix | Status |
300
+ |---|----------|------|-----|-----|--------|
301
+ | 93 | MEDIUM | `src/search/semantic_scholar.py:47,49` | `citationCount: null` from API → `None` in sort key → `TypeError: '<' not supported between NoneType and int` crash | Changed to `paper.get("citationCount") or 0` and `x.get("citation_count") or 0` | FIXED |
302
+ | 94 | MEDIUM | `app_web.py:123` | Stale KB welcome summary after re-ingest — `st.session_state.kb_welcome_summary` not cleared before `st.rerun()` | Added `st.session_state.pop("kb_welcome_summary", None)` before rerun | FIXED |
303
+ | 95 | MEDIUM | `app_web.py:256-257,267` | SQL result count missing from web UI status label — showed "0 local + 0 web" even when SQL returned rows | Added `n_sql` to both "Verified" and "Done" status labels, matching CLI behavior | FIXED |
304
+ | 96 | MEDIUM | `src/verifier.py:184` | References regex `\b(references\|sources)\b` matched casual mentions of "sources" in prose — citation audit checked entire response body instead of just References section | Changed to heading-specific regex: `^#+\s*(references\|sources)` or `^\*\*(references\|sources)\*\*` | FIXED |
305
+ | 97 | MEDIUM | `src/verifier.py:486` | Non-strict mode used stale `citation_warnings` from last loop iteration, not from final corrected response | Added `validate_citations(response, retrieval_result)` recomputation before building warning | FIXED |
306
+ | 98 | MEDIUM | `src/ingest.py:291-298` | Meta files deleted before overview regeneration — if `build_and_store_overview()` failed, KB lost self-awareness until next ingestion | Removed pre-deletion; `write_text()` in `build_and_store_overview()` already overwrites | FIXED |
307
+ | 99 | MEDIUM | `src/readers/docx.py:4-10` | DOCX reader only extracted paragraphs, silently dropping all embedded tables | Added table extraction loop: `doc.tables` → row cells joined with ` \| ` | FIXED |
308
+
309
+ ### Known Issues Added
310
+
311
+ | # | Severity | File | Description | Reason |
312
+ |---|----------|------|-------------|--------|
313
+ | K18 | MEDIUM | `src/sql_retriever.py:17-24` | Comment stripping is naive (regex) — corrupts string literals containing `--` or `/* */` | Rare in LLM-generated queries; proper state-machine parser would be over-engineering |
314
+ | K19 | MEDIUM | `src/retriever.py:195-196, 267-269` | LIKE wildcard chars `%` and `_` not escaped in fallback/alt-column queries | Same as existing K7; low real-world impact |
315
+ | K20 | MEDIUM | `src/sql_retriever.py` | No SQL execution timeout — Cartesian products could hang | Same as existing K10; SQLite lacks built-in SELECT timeout |
316
+ | K21 | MEDIUM | `src/verifier.py:151-202` | `validate_citations` compares citation numbers against raw retrieval count, not References section entry count | Would require parsing References section to count entries; current check is conservative |
317
+ | K22 | LOW | `src/readers/csv_tab.py:11` | CSV encoding hardcoded to `utf-8-sig` — non-UTF-8 files get replacement chars | Would need `chardet` dependency or encoding detection |
318
+ | K23 | LOW | `src/sql_ingest.py:568-586` | Row-by-row SQL INSERT slow for large datasets | Would benefit from `executemany()` batching |
319
+ | K24 | LOW | `src/readers/excel.py:40, csv_tab.py:22` | `zip(headers, row)` silently truncates extra columns in ragged rows | Acceptable for most research datasets |
320
+ | K25 | LOW | `src/ingest.py:104` | `rglob("*")` follows symlinks — cyclic symlinks could cause infinite loop | Rare in knowledge_base directories |
321
+ | K26 | LOW | Test suite | Zero coverage: `validate_citations`, verification loop, `make_fuzzy_query`, `compute_similarity_flags`, `load_kb_meta_brief` | Test gap documented; non-blocking for release |
322
+
323
+ ### Test Coverage Gaps (Confirmed)
324
+
325
+ | Priority | What | Description |
326
+ |----------|------|-------------|
327
+ | HIGH | `validate_citations()` | Layer 4.5 — zero direct tests |
328
+ | HIGH | Verification correction loop | Core feature — zero multi-iteration tests |
329
+ | HIGH | `make_fuzzy_query()` | SQL fuzzy fallback — zero tests |
330
+ | MEDIUM | `compute_similarity_flags()` | Layer 4 — zero tests |
331
+ | MEDIUM | `load_kb_meta_brief()` | Used by verifier — zero tests |
332
+
333
+ ---
334
+
335
+ ## Round 12 — 2026-03-06 Session 7 (Pre-Release Final Audit)
336
+
337
+ See `docs/AUDIT_BUGLOG.md` Round 12 section for full details.
338
+ 33 bugs found across 3 tiers, all FIXED. Design score raised from 7.1/10 to 9/10.
339
+ 5 NEW low-severity issues documented as KNOWN. 174/174 tests passing. **Released to GitHub.**
340
+
341
+ ---
342
+
343
+ ## Round 13 — 2026-03-07 (Post-Release Deep Audit)
344
+
345
+ **7 parallel audit agents** (design evaluation + core pipeline + verification stack + SQL layer + LLM/readers/search + application layer + test coverage). Full line-by-line review of all 25 source files, 17 test files, and all documentation.
346
+
347
+ **Design Score**: 8.5/10 → **9.1/10** (post-fix) | **Test Coverage Score**: 6/10
348
+
349
+ ### Tier 1: Anti-Hallucination Integrity (4 bugs) — ALL FIXED
350
+
351
+ | # | Severity | File | Bug | Fix | Status |
352
+ |---|----------|------|-----|-----|--------|
353
+ | 100 | HIGH | `verifier.py` | **R13-T1-01**: Verification LLM persistent failure → `except Exception: continue` skips all iterations → non-strict returns unverified response. Full bypass of Layers 3–5. | Added `any_verification_ran` flag. If ALL verification LLM calls fail: strict mode refuses with system-failure message; non-strict adds prominent WARNING banner distinguishing "could not run" from "did not pass". | FIXED |
354
+ | 101 | HIGH | `verifier.py` | **R13-T1-02**: `validate_citations` SQL/web source matching uses wrong dict structure (dead code). Layer 4.5 only works for local sources. | SQL: extract source filename from context via regex. Web: changed to flat dict access `web_chunk.get("url", "")`. | FIXED |
355
+ | 102 | HIGH | `verifier.py` | **R13-T1-03**: Citation count uses unique source *files*, not citable *units*. False "fabricated reference" warning on multi-chunk single-file sources. | Changed `db_count = len(db_results)` (chunks) instead of `len(unique_db_sources)` (files). Each chunk is a citable unit. | FIXED |
356
+ | 103 | MEDIUM | `verifier.py` | **R13-T1-04**: `max_iterations: 0` with `enabled: true` silently disables verification. | Clamp `max_iterations = max(1, max_iterations)` with warning. Must use `enabled: false` to disable. Test updated. | FIXED |
357
+
358
+ ### Tier 2: Pre-Release Quality (24 bugs) — ALL FIXED
359
+
360
+ | # | Severity | File | Bug | Fix | Status |
361
+ |---|----------|------|-----|-----|--------|
362
+ | 104 | HIGH | `retriever.py` | **R13-T2-01**: Missing LIKE wildcard ESCAPE in `_build_fallback_sql_query` and `_try_alternate_columns`. | Added `%`/`_` escaping and `ESCAPE '\'` clause. Initial fix had double-backslash bug; corrected by regression checker. | FIXED |
363
+ | 105 | HIGH | `gemini.py` | **R13-T2-02**: Gemini safety filter returns `"[Gemini blocked: ...]"` bypassing anti-hallucination stack. | Return `""` instead. Verifier's empty-response handler (lines 369-378) provides user-facing refusal. | FIXED |
364
+ | 106 | HIGH | `openai.py` | **R13-T2-03**: Reasoning model budget uncapped → API error with large `max_tokens`. | `reasoning_budget = min(max(max_tokens * 4, 4096), 128000)`. | FIXED |
365
+ | 107 | HIGH | `llm/__init__.py` | **R13-T2-04**: Non-numeric temperature/max_tokens crashes with opaque ValueError. | Wrapped in try/except with defaults (0.0 and 2048). | FIXED |
366
+ | 108 | HIGH | `app_cli.py` | **R13-T2-05**: After clarification, raw `user_input` re-appended → duplicate history entries. | Use `display_query` (post-QU, includes clarification context) instead of raw `user_input`. | FIXED |
367
+ | 109 | MEDIUM | `llm/__init__.py` | **R13-T2-06**: Anthropic temp > 1.0 → API error. | Added provider-specific clamp: `if provider == "anthropic": temperature = min(temperature, 1.0)`. | FIXED |
368
+ | 110 | MEDIUM | `sql_retriever.py` | **R13-T2-07**: No SQL query timeout. `CROSS JOIN` hangs indefinitely. | Added `set_progress_handler` with 10-second abort. `import time` at module level. | FIXED |
369
+ | 111 | MEDIUM | `app_cli.py` | **R13-T2-08**: Rich markup injection via brackets in error messages/queries. | Added `rich_escape()` to all 5 interpolated f-strings in the RAG pipeline loop. | FIXED |
370
+ | 112 | MEDIUM | `readers/*.py` | **R13-T2-09**: No try/except in readers. Corrupt file aborts ingestion. | Wrapped `read_pdf`, `read_docx`, `read_excel`, `read_stata`, `read_rdata`, `read_spss` in try/except. | FIXED |
371
+ | 113 | MEDIUM | `retriever.py`, `app_cli.py`, `app_web.py` | **R13-T2-10**: `_collection_cache` never invalidated after re-ingestion. | Added `clear_collection_cache()` function; called after ingestion in both CLI and web. | FIXED |
372
+ | 114 | MEDIUM | `app_web.py` | **R13-T2-11**: Web UI messages grow without limit. | Capped at 200 entries (100 Q&A pairs) after each append cycle. | FIXED |
373
+ | 115 | MEDIUM | `setup.py` | **R13-T2-12**: Setup wizard overwrites config without confirmation. | Added existence check + `[y/N]` prompt before overwriting `config.yaml`. | FIXED |
374
+ | 116 | MEDIUM | `prompts.py` | **R13-T2-13**: Prompt injection via KB delimiter mimicry. | Sanitize context: replace 69-char `=` sequences with Unicode `≡` before injection. | FIXED |
375
+ | 117 | MEDIUM | `verifier.py` | **R13-T2-14**: References regex matches `## Data Sources` etc. | Anchored regex with `\s*$` to require standalone headers only. | FIXED |
376
+ | 118 | MEDIUM | `verifier.py` | **R13-T2-15**: Correction prompt includes full previous response. | Truncate to 1500 chars. Instruction: "Rewrite COMPLETE response from scratch". | FIXED |
377
+ | 119 | MEDIUM | `verifier.py` | **R13-T2-16**: Initial `generate()` exception propagates. | Wrapped in try/except; returns `refused: True` with error message. | FIXED |
378
+ | 120 | MEDIUM | `openai.py` | **R13-T2-17**: Reasoning model detection via brittle prefix match. | Changed to `model in ("o1","o3","o4") or model.startswith(("o1-","o3-","o4-"))`. | FIXED |
379
+ | 121 | MEDIUM | `sql_retriever.py` | **R13-T2-18**: `make_fuzzy_query` regex matches non-column contexts. | Pattern now requires quoted identifiers or valid SQL column names. | FIXED |
380
+ | 122 | MEDIUM | `sql_retriever.py` | **R13-T2-19**: `REPLACE()` function falsely blocked. | Removed `REPLACE` from `_DANGEROUS_KEYWORDS`. Statement form blocked by SELECT-only check. | FIXED |
381
+ | 123 | MEDIUM | `config_loader.py` | **R13-T2-20**: Non-dict YAML → `AttributeError`. | Added `isinstance(cfg, dict)` check with clear ValueError message. | FIXED |
382
+ | 124 | MEDIUM | `app_web.py` | **R13-T2-21**: Model list cache stale after provider change. | Added provider-change detection that clears all model caches. | FIXED |
383
+ | 125 | MEDIUM | `app_web.py` | **R13-T2-22**: Web UI uses slow LLM call for KB welcome. | Changed to `load_kb_meta_brief()` (file read) matching CLI behavior. | FIXED |
384
+ | 126 | MEDIUM | `app_web.py` | **R13-T2-23**: Re-ingestion blocks without progress feedback. | Added `st.info()` message before spinner to set expectations. | FIXED |
385
+ | 127 | MEDIUM | `setup.py` | **R13-T2-24**: Empty API key warning not actionable. | Improved message: "Set it later by editing .env or re-running: python setup.py". | FIXED |
386
+
387
+ ### Tier 3: Post-Release Improvements (20 bugs)
388
+
389
+ | # | Severity | File | Bug | Status |
390
+ |---|----------|------|-----|--------|
391
+ | 128 | MED | `openai.py:9`, `anthropic.py:9`, `gemini.py:16` | **R13-T3-01**: New API client per `generate()` call. No connection reuse. | OPEN |
392
+ | 129 | MED | `llm/openai.py:52`, `anthropic.py:32`, `gemini.py:48` | **R13-T3-02**: `list_models()` returns hardcoded list, not live API fetch. Design doc says otherwise. | OPEN |
393
+ | 130 | MED | `retriever.py:282-285` | **R13-T3-03**: `_try_alternate_columns` quote escaping logic fragile and non-obvious. | OPEN |
394
+ | 131 | MED | `sql_retriever.py:54` | **R13-T3-04**: `sqlite_master` check on `stripped` instead of `unquoted`. False positives. | OPEN |
395
+ | 132 | MED | `readers/stata.py:28`, `rdata.py:15` | **R13-T3-05**: `df.iterrows()` 10-100x slower than `itertuples()`. Large datasets take minutes. | OPEN |
396
+ | 133 | LOW | `sql_retriever.py:26-33` | **R13-T3-06**: `_strip_sql_comments` removes `--` inside string literals. Confirmed from R12 NEW-04. | OPEN |
397
+ | 134 | LOW | `verifier.py:429` | **R13-T3-07**: Verification LLM system prompt minimal. No anti-hallucination rules for verifier itself. | OPEN |
398
+ | 135 | LOW | `verifier.py:25-34` | **R13-T3-08**: Warning phrases easily evaded by paraphrasing. Only 8 exact-match phrases. | OPEN |
399
+ | 136 | LOW | `verifier.py:126` | **R13-T3-09**: Term overlap excludes 2-char acronyms (AI, UK, EU, US, UN). | OPEN |
400
+ | 137 | LOW | `verifier.py:181` | **R13-T3-10**: Citation `[0]` never detected as invalid. | OPEN |
401
+ | 138 | LOW | `readers/csv_tab.py:11`, `text.py:5` | **R13-T3-11**: Hardcoded UTF-8. Non-UTF-8 files get replacement chars. | OPEN |
402
+ | 139 | LOW | `readers/docx.py:24` | **R13-T3-12**: All DOCX content mapped to `page: 1`. | OPEN |
403
+ | 140 | LOW | `readers/pdf.py:9` | **R13-T3-13**: Scanned PDFs silently return empty list. No warning. | OPEN |
404
+ | 141 | LOW | `readers/excel.py:15` | **R13-T3-14**: Unevaluated formulas appear as `None`, silently dropped. | OPEN |
405
+ | 142 | LOW | `search/semantic_scholar.py:15-28` | **R13-T3-15**: Retries on non-retriable 4xx errors. Wastes 17s. | OPEN |
406
+ | 143 | LOW | `search/__init__.py:10-12` | **R13-T3-16**: Unknown backend names silently return `[]`. | OPEN |
407
+ | 144 | LOW | `setup.py:96-100` | **R13-T3-17**: `.env` unescape wrong order. Theoretical corruption. | OPEN |
408
+ | 145 | LOW | `config_loader.py:20-22` | **R13-T3-18**: `.env` loaded before config existence check. | OPEN |
409
+ | 146 | LOW | `app_web.py:288` | **R13-T3-19**: Browser tab title hardcoded "ResearchBot". | OPEN |
410
+ | 147 | LOW | `verifier.py:460-462` | **R13-T3-20**: Correction call uses same temperature as initial. Should force 0.0. | OPEN |
411
+
412
+ ### Test Coverage Gaps (Round 13 — confirmed + expanded)
413
+
414
+ | Priority | What | File | Status |
415
+ |----------|------|------|--------|
416
+ | CRITICAL | `validate_citations()` — Layer 4.5 | `verifier.py:152-230` | OPEN |
417
+ | CRITICAL | Verification correction loop | `verifier.py:407-520` | OPEN |
418
+ | CRITICAL | `make_fuzzy_query()` | `sql_retriever.py:181-237` | OPEN |
419
+ | HIGH | `compute_similarity_flags()` — Layer 4 | `verifier.py:101-149` | OPEN |
420
+ | HIGH | `_strip_quoted()` / `_strip_sql_comments()` | `sql_retriever.py:17-33` | OPEN |
421
+ | HIGH | Empty LLM response in `verify_and_respond` | `verifier.py:354-364` | OPEN |
422
+ | HIGH | `build_and_store_overview()` orchestrator | `kb_meta.py:464-492` | OPEN |
423
+ | MEDIUM | `generate_brief_overview()` | `kb_meta.py:385-422` | OPEN |
424
+ | MEDIUM | `load_kb_meta_brief()` fallback | `kb_meta.py:434-445` | OPEN |
425
+ | MEDIUM | `summarize_kb_for_welcome()` | `kb_meta.py:514-552` | OPEN |
426
+ | MEDIUM | `_build_fallback_sql_query` with real data | `retriever.py:159-217` | OPEN |
427
+ | MEDIUM | `_try_alternate_columns` with real data | `retriever.py:220-292` | OPEN |
428
+ | MEDIUM | `_validate_sql` edge cases | `sql_retriever.py:36-59` | OPEN |
429
+ | MEDIUM | `format_sql_results_as_context` truncation | `sql_retriever.py:116-167` | OPEN |
430
+ | MEDIUM | `_normalize_nulls` | `config_loader.py:36-42` | OPEN |
431
+
432
+ ### Cross-Agent Consensus (confirmed by 2+ agents independently)
433
+
434
+ | Bug | Agents | Confidence |
435
+ |-----|--------|-----------|
436
+ | `validate_citations` dead code for SQL + web (#101) | Core Pipeline, Application, Verification | **Very High** |
437
+ | Missing LIKE ESCAPE in SQL fallbacks (#104) | SQL, Core Pipeline, Design | **Very High** |
438
+ | Stale `_collection_cache` after re-ingestion (#113) | Core Pipeline, Design | **High** |
439
+ | Verification LLM failure → silent bypass (#100) | Verification, Core Pipeline | **High** |
440
+ | Missing reader exception handling (#112) | LLM/Readers, Design | **High** |
441
+
442
+ ### Recommended Fix Order
443
+
444
+ **Phase 1 — Anti-hallucination integrity (#100–#103):**
445
+ Fix verification bypass, validate_citations data structure, citation counting, max_iterations guard.
446
+
447
+ **Phase 2 — High-severity T2 bugs (#104–#108):**
448
+ LIKE ESCAPE, Gemini blocked response, reasoning model overflow, config validation, history corruption.
449
+
450
+ **Phase 3 — Test coverage (3 CRITICAL gaps):**
451
+ Tests for validate_citations, verification correction loop, make_fuzzy_query.
452
+
453
+ **Phase 4 — Remaining T2 MEDIUM bugs (#109–#127):**
454
+ Address in order of impact.
455
+
456
+ **Phase 5 — T3 improvements (#128–#147):**
457
+ Address as time permits.
458
+
459
+ ---
460
+
461
+ ## Round 14 — 2026-03-07 (Tier 1 + Tier 2 Bug Fix Sprint)
462
+
463
+ **6 parallel agents**: 4 fix agents (T1, T2-HIGH, T2-MED-A, T2-MED-B) + 1 regression checker + 1 design guardian.
464
+
465
+ - Fixed all 28 Tier 1 + Tier 2 bugs (#100–#127)
466
+ - Regression checker found 1 CRITICAL issue in Bug #104 fix (double-backslash in ESCAPE clause) — corrected
467
+ - Also fixed `spss.py` missing try/except (pre-existing gap found by regression checker)
468
+ - Design Guardian assessment: design score 8.5 → 9.1/10, recommendation: SHIP
469
+ - Updated test for Bug #103 (renamed to `test_max_iterations_zero_clamps_to_one`)
470
+ - 174/174 tests passing
471
+
472
+ ---
473
+
474
  ## Statistics
475
 
476
  | Metric | Count |
477
  |--------|-------|
478
+ | Total bugs found (all rounds) | ~216 (48 new in Round 13) |
479
+ | Total bugs fixed (all rounds) | ~196 (168 in R1–R12, 28 in R14) |
480
+ | Round 13 T1+T2 bugs | 28 FIXED (was 28 OPEN) |
481
+ | Round 13 T3 bugs remaining | 20 OPEN (5 MEDIUM, 15 LOW) |
482
+ | Audit rounds completed | 14 (50+ total agent dispatches) |
483
  | Test count | 174 passing |
484
  | Test files | 17 |
485
+ | Source files modified in R14 | 16 (verifier, retriever, gemini, openai, llm/__init__, app_cli, app_web, sql_retriever, config_loader, prompts, setup, pdf, docx, excel, stata, rdata, spss) |
486
+ | Design score | 9.1/10 (up from 8.5) |
487
+ | CRITICAL test gaps | 3 (validate_citations, correction loop, make_fuzzy_query) |
setup.py CHANGED
@@ -141,7 +141,8 @@ def run_wizard():
141
  # Step 3b: API key
142
  api_key = getpass.getpass(f"Enter your {provider} API key: ")
143
  if not api_key:
144
- print(f" Warning: No API key entered for {provider}. You can set it later in .env")
 
145
 
146
  # Step 4: Model selection — fetch available models
147
  print(f"\nFetching available {provider} models...")
@@ -179,9 +180,18 @@ def run_wizard():
179
 
180
  # Write config.yaml
181
  config_path = project_root / "config.yaml"
182
- config_str = generate_config(bot_name, domain, provider, model, web_search)
183
- config_path.write_text(config_str)
184
- print(f"\n Wrote {config_path}")
 
 
 
 
 
 
 
 
 
185
 
186
  # Write .env (merges with existing keys if present)
187
  env_path = project_root / ".env"
 
141
  # Step 3b: API key
142
  api_key = getpass.getpass(f"Enter your {provider} API key: ")
143
  if not api_key:
144
+ print(f" Warning: No API key entered for {provider}.")
145
+ print(f" Set it later by editing .env or re-running: python setup.py")
146
 
147
  # Step 4: Model selection — fetch available models
148
  print(f"\nFetching available {provider} models...")
 
180
 
181
  # Write config.yaml
182
  config_path = project_root / "config.yaml"
183
+ if config_path.exists():
184
+ overwrite = input(f"\n{config_path} already exists. Overwrite? [y/N]: ").strip().lower()
185
+ if overwrite != 'y':
186
+ print(" Keeping existing config.yaml.")
187
+ else:
188
+ config_str = generate_config(bot_name, domain, provider, model, web_search)
189
+ config_path.write_text(config_str)
190
+ print(f"\n Wrote {config_path}")
191
+ else:
192
+ config_str = generate_config(bot_name, domain, provider, model, web_search)
193
+ config_path.write_text(config_str)
194
+ print(f"\n Wrote {config_path}")
195
 
196
  # Write .env (merges with existing keys if present)
197
  env_path = project_root / ".env"
src/config_loader.py CHANGED
@@ -30,6 +30,12 @@ def load_config(config_path: str = None) -> dict:
30
  with open(config_path, "r", encoding="utf-8") as f:
31
  cfg = yaml.safe_load(f) or {}
32
 
 
 
 
 
 
 
33
  # Normalize None-valued sections to empty dicts so chained .get() never
34
  # fails with AttributeError (e.g. `llm:` with no sub-keys → None).
35
  # Recurse into nested dicts so `paths:\n vector_db:` also gets normalized.
 
30
  with open(config_path, "r", encoding="utf-8") as f:
31
  cfg = yaml.safe_load(f) or {}
32
 
33
+ if not isinstance(cfg, dict):
34
+ raise ValueError(
35
+ f"Config file must contain a YAML mapping (dict), got {type(cfg).__name__}. "
36
+ "Run 'python setup.py' to regenerate config.yaml."
37
+ )
38
+
39
  # Normalize None-valued sections to empty dicts so chained .get() never
40
  # fails with AttributeError (e.g. `llm:` with no sub-keys → None).
41
  # Recurse into nested dicts so `paths:\n vector_db:` also gets normalized.
src/llm/__init__.py CHANGED
@@ -24,8 +24,17 @@ def generate(system_prompt: str, user_message: str, cfg: dict,
24
  if max_tokens is None:
25
  max_tokens = llm_cfg.get("max_tokens", 2048)
26
  # Validate parameters
27
- temperature = max(0.0, min(float(temperature), 2.0))
28
- max_tokens = max(1, min(int(max_tokens), 128000))
 
 
 
 
 
 
 
 
 
29
  return PROVIDERS[provider].generate(
30
  system_prompt=system_prompt, user_message=user_message,
31
  api_key=api_key, model=model, temperature=temperature, max_tokens=max_tokens,
 
24
  if max_tokens is None:
25
  max_tokens = llm_cfg.get("max_tokens", 2048)
26
  # Validate parameters
27
+ try:
28
+ temperature = max(0.0, min(float(temperature), 2.0))
29
+ except (ValueError, TypeError):
30
+ temperature = 0.0
31
+ # Anthropic max temperature is 1.0
32
+ if provider == "anthropic":
33
+ temperature = min(temperature, 1.0)
34
+ try:
35
+ max_tokens = max(1, min(int(max_tokens), 128000))
36
+ except (ValueError, TypeError):
37
+ max_tokens = 2048
38
  return PROVIDERS[provider].generate(
39
  system_prompt=system_prompt, user_message=user_message,
40
  api_key=api_key, model=model, temperature=temperature, max_tokens=max_tokens,
src/llm/gemini.py CHANGED
@@ -30,8 +30,8 @@ def generate(system_prompt: str, user_message: str, api_key: str,
30
  return ""
31
  return response.text
32
  except ValueError as e:
33
- # Safety filter block
34
- return f"[Gemini blocked: {e}]"
35
  except Exception as e:
36
  error_type = type(e).__name__
37
  error_msg = str(e)
 
30
  return ""
31
  return response.text
32
  except ValueError as e:
33
+ # Safety filter block — return empty so verifier treats it as generation failure
34
+ return ""
35
  except Exception as e:
36
  error_type = type(e).__name__
37
  error_msg = str(e)
src/llm/openai.py CHANGED
@@ -7,12 +7,12 @@ def generate(system_prompt: str, user_message: str, api_key: str,
7
  model = model or "gpt-4o"
8
  from openai import OpenAI
9
  client = OpenAI(api_key=api_key)
10
- is_reasoning = model.startswith(("o1", "o3", "o4"))
11
  try:
12
  if is_reasoning:
13
  # Reasoning models include thinking tokens in max_completion_tokens,
14
  # so we need a larger budget to get enough output tokens.
15
- reasoning_budget = max(max_tokens * 4, 4096)
16
  messages = [
17
  {"role": "developer", "content": system_prompt},
18
  {"role": "user", "content": user_message},
 
7
  model = model or "gpt-4o"
8
  from openai import OpenAI
9
  client = OpenAI(api_key=api_key)
10
+ is_reasoning = model in ("o1", "o3", "o4") or model.startswith(("o1-", "o3-", "o4-"))
11
  try:
12
  if is_reasoning:
13
  # Reasoning models include thinking tokens in max_completion_tokens,
14
  # so we need a larger budget to get enough output tokens.
15
+ reasoning_budget = min(max(max_tokens * 4, 4096), 128000)
16
  messages = [
17
  {"role": "developer", "content": system_prompt},
18
  {"role": "user", "content": user_message},
src/prompts.py CHANGED
@@ -299,10 +299,13 @@ def build_prompt(
299
  else:
300
  overview_section = ""
301
 
 
 
 
302
  return SYSTEM_PROMPT_TEMPLATE.format(
303
  bot_name=bot_name,
304
  domain=domain,
305
- context=context,
306
  kb_overview_section=overview_section,
307
  )
308
 
 
299
  else:
300
  overview_section = ""
301
 
302
+ # Sanitize context to prevent prompt injection via delimiter mimicry
303
+ safe_context = context.replace("=" * 69, "\u2261" * 69)
304
+
305
  return SYSTEM_PROMPT_TEMPLATE.format(
306
  bot_name=bot_name,
307
  domain=domain,
308
+ context=safe_context,
309
  kb_overview_section=overview_section,
310
  )
311
 
src/readers/docx.py CHANGED
@@ -2,23 +2,27 @@
2
 
3
 
4
  def read_docx(file_path: str) -> list[dict]:
5
- from docx import Document
6
- doc = Document(file_path)
7
- paragraphs = [p.text.strip() for p in doc.paragraphs if p.text.strip()]
 
8
 
9
- # Extract embedded tables
10
- for table in doc.tables:
11
- for row in table.rows:
12
- seen_elements = set()
13
- cells = []
14
- for cell in row.cells:
15
- if id(cell._element) not in seen_elements:
16
- seen_elements.add(id(cell._element))
17
- if cell.text.strip():
18
- cells.append(cell.text.strip())
19
- if cells:
20
- paragraphs.append(" | ".join(cells))
21
 
22
- if not paragraphs:
 
 
 
 
23
  return []
24
- return [{"page": 1, "text": "\n\n".join(paragraphs)}]
 
2
 
3
 
4
  def read_docx(file_path: str) -> list[dict]:
5
+ try:
6
+ from docx import Document
7
+ doc = Document(file_path)
8
+ paragraphs = [p.text.strip() for p in doc.paragraphs if p.text.strip()]
9
 
10
+ # Extract embedded tables
11
+ for table in doc.tables:
12
+ for row in table.rows:
13
+ seen_elements = set()
14
+ cells = []
15
+ for cell in row.cells:
16
+ if id(cell._element) not in seen_elements:
17
+ seen_elements.add(id(cell._element))
18
+ if cell.text.strip():
19
+ cells.append(cell.text.strip())
20
+ if cells:
21
+ paragraphs.append(" | ".join(cells))
22
 
23
+ if not paragraphs:
24
+ return []
25
+ return [{"page": 1, "text": "\n\n".join(paragraphs)}]
26
+ except Exception as e:
27
+ print(f"Warning: Could not read {file_path}: {e}")
28
  return []
 
src/readers/excel.py CHANGED
@@ -2,11 +2,15 @@
2
  MAX_CHUNK_CHARS = 6000
3
 
4
  def read_excel(file_path: str) -> list[dict]:
5
- from pathlib import Path
6
- ext = Path(file_path).suffix.lower()
7
- if ext == ".xls":
8
- return _read_xls(file_path)
9
- return _read_xlsx(file_path)
 
 
 
 
10
 
11
  def _read_xlsx(file_path: str) -> list[dict]:
12
  import openpyxl
 
2
  MAX_CHUNK_CHARS = 6000
3
 
4
  def read_excel(file_path: str) -> list[dict]:
5
+ try:
6
+ from pathlib import Path
7
+ ext = Path(file_path).suffix.lower()
8
+ if ext == ".xls":
9
+ return _read_xls(file_path)
10
+ return _read_xlsx(file_path)
11
+ except Exception as e:
12
+ print(f"Warning: Could not read {file_path}: {e}")
13
+ return []
14
 
15
  def _read_xlsx(file_path: str) -> list[dict]:
16
  import openpyxl
src/readers/pdf.py CHANGED
@@ -2,11 +2,15 @@
2
 
3
 
4
  def read_pdf(file_path: str) -> list[dict]:
5
- from pypdf import PdfReader
6
- reader = PdfReader(file_path)
7
- pages = []
8
- for i, page in enumerate(reader.pages):
9
- text = page.extract_text()
10
- if text and text.strip():
11
- pages.append({"page": i + 1, "text": text.strip()})
12
- return pages
 
 
 
 
 
2
 
3
 
4
  def read_pdf(file_path: str) -> list[dict]:
5
+ try:
6
+ from pypdf import PdfReader
7
+ reader = PdfReader(file_path)
8
+ pages = []
9
+ for i, page in enumerate(reader.pages):
10
+ text = page.extract_text()
11
+ if text and text.strip():
12
+ pages.append({"page": i + 1, "text": text.strip()})
13
+ return pages
14
+ except Exception as e:
15
+ print(f"Warning: Could not read {file_path}: {e}")
16
+ return []
src/readers/rdata.py CHANGED
@@ -2,34 +2,38 @@
2
  MAX_CHUNK_CHARS = 6000
3
 
4
  def read_rdata(file_path: str) -> list[dict]:
5
- import pyreadr
6
- result = pyreadr.read_r(file_path)
7
- pages = []
8
- for name, df in result.items():
9
- name_display = name if name is not None else "data"
10
- headers = list(df.columns)
11
- header_line = f"Object: {name_display} | Columns: {', '.join(headers)}\n"
12
- block = []
13
- block_chars = len(header_line)
14
- block_start = 1
15
- for idx, (_, row) in enumerate(df.iterrows()):
16
- parts = []
17
- for col in headers:
18
- val = row[col]
19
- if val is not None and str(val).strip() and str(val).lower() not in ("nan", "nat", "<na>", "inf", "-inf"):
20
- parts.append(f"{col}: {val}")
21
- if not parts:
22
- continue
23
- row_text = "; ".join(parts)
24
- if block and block_chars + len(row_text) + 1 > MAX_CHUNK_CHARS:
 
 
 
 
 
 
 
 
 
25
  text = header_line + "\n".join(block)
26
  pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
27
- block = []
28
- block_chars = len(header_line)
29
- block_start = idx + 1
30
- block.append(row_text)
31
- block_chars += len(row_text) + 1
32
- if block:
33
- text = header_line + "\n".join(block)
34
- pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
35
- return pages
 
2
  MAX_CHUNK_CHARS = 6000
3
 
4
  def read_rdata(file_path: str) -> list[dict]:
5
+ try:
6
+ import pyreadr
7
+ result = pyreadr.read_r(file_path)
8
+ pages = []
9
+ for name, df in result.items():
10
+ name_display = name if name is not None else "data"
11
+ headers = list(df.columns)
12
+ header_line = f"Object: {name_display} | Columns: {', '.join(headers)}\n"
13
+ block = []
14
+ block_chars = len(header_line)
15
+ block_start = 1
16
+ for idx, (_, row) in enumerate(df.iterrows()):
17
+ parts = []
18
+ for col in headers:
19
+ val = row[col]
20
+ if val is not None and str(val).strip() and str(val).lower() not in ("nan", "nat", "<na>", "inf", "-inf"):
21
+ parts.append(f"{col}: {val}")
22
+ if not parts:
23
+ continue
24
+ row_text = "; ".join(parts)
25
+ if block and block_chars + len(row_text) + 1 > MAX_CHUNK_CHARS:
26
+ text = header_line + "\n".join(block)
27
+ pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
28
+ block = []
29
+ block_chars = len(header_line)
30
+ block_start = idx + 1
31
+ block.append(row_text)
32
+ block_chars += len(row_text) + 1
33
+ if block:
34
  text = header_line + "\n".join(block)
35
  pages.append({"page": f"{name_display}_rows_{block_start}-{block_start + len(block) - 1}", "text": text})
36
+ return pages
37
+ except Exception as e:
38
+ print(f"Warning: Could not read {file_path}: {e}")
39
+ return []
 
 
 
 
 
src/readers/spss.py CHANGED
@@ -2,6 +2,10 @@
2
  from src.readers.stata import _dataframe_to_pages
3
 
4
  def read_spss(file_path: str) -> list[dict]:
5
- import pyreadstat
6
- df, meta = pyreadstat.read_sav(file_path)
7
- return _dataframe_to_pages(df, meta)
 
 
 
 
 
2
  from src.readers.stata import _dataframe_to_pages
3
 
4
  def read_spss(file_path: str) -> list[dict]:
5
+ try:
6
+ import pyreadstat
7
+ df, meta = pyreadstat.read_sav(file_path)
8
+ return _dataframe_to_pages(df, meta)
9
+ except Exception as e:
10
+ print(f"Warning: Could not read {file_path}: {e}")
11
+ return []
src/readers/stata.py CHANGED
@@ -2,9 +2,13 @@
2
  MAX_CHUNK_CHARS = 6000
3
 
4
  def read_stata(file_path: str) -> list[dict]:
5
- import pyreadstat
6
- df, meta = pyreadstat.read_dta(file_path)
7
- return _dataframe_to_pages(df, meta)
 
 
 
 
8
 
9
  def _dataframe_to_pages(df, meta) -> list[dict]:
10
  pages = []
 
2
  MAX_CHUNK_CHARS = 6000
3
 
4
  def read_stata(file_path: str) -> list[dict]:
5
+ try:
6
+ import pyreadstat
7
+ df, meta = pyreadstat.read_dta(file_path)
8
+ return _dataframe_to_pages(df, meta)
9
+ except Exception as e:
10
+ print(f"Warning: Could not read {file_path}: {e}")
11
+ return []
12
 
13
  def _dataframe_to_pages(df, meta) -> list[dict]:
14
  pages = []
src/retriever.py CHANGED
@@ -7,6 +7,11 @@ from src.search import search, format_web_results_as_context
7
  _collection_cache = {}
8
 
9
 
 
 
 
 
 
10
  def _get_cached_collection(cfg: dict):
11
  """Get or create cached ChromaDB collection."""
12
  from src.ingest import get_chroma_collection
@@ -208,7 +213,8 @@ def _build_fallback_sql_query(query: str, cfg: dict) -> str | None:
208
  safe_word = word.replace("'", "''")
209
  if not re.match(r'^\w+$', safe_word):
210
  continue # Skip non-alphanumeric words
211
- conditions.append(f'"{col}" LIKE \'%{safe_word}%\'')
 
212
 
213
  if conditions:
214
  where_clause = " OR ".join(conditions)
@@ -283,8 +289,9 @@ def _try_alternate_columns(sql_query: str, cfg: dict) -> tuple[list[dict], str]
283
  # Strip dangerous characters for defense-in-depth
284
  if not re.match(r"^[\w\s.,'-]+$", safe_value):
285
  safe_value = re.sub(r"[;'\\\"]", "", safe_value)
 
286
  for col in text_cols:
287
- alt_query = f'SELECT * FROM "{table_name}" WHERE "{col}" LIKE \'%{safe_value}%\' LIMIT {max_rows}'
288
  rows = execute_sql_query(alt_query, cfg)
289
  if rows:
290
  return rows, alt_query
 
7
  _collection_cache = {}
8
 
9
 
10
+ def clear_collection_cache():
11
+ """Clear the cached ChromaDB collection so it's re-read after re-ingestion."""
12
+ _collection_cache.clear()
13
+
14
+
15
  def _get_cached_collection(cfg: dict):
16
  """Get or create cached ChromaDB collection."""
17
  from src.ingest import get_chroma_collection
 
213
  safe_word = word.replace("'", "''")
214
  if not re.match(r'^\w+$', safe_word):
215
  continue # Skip non-alphanumeric words
216
+ safe_word = safe_word.replace('%', '\\%').replace('_', '\\_')
217
+ conditions.append(f'"{col}" LIKE \'%{safe_word}%\' ESCAPE \'\\\'')
218
 
219
  if conditions:
220
  where_clause = " OR ".join(conditions)
 
289
  # Strip dangerous characters for defense-in-depth
290
  if not re.match(r"^[\w\s.,'-]+$", safe_value):
291
  safe_value = re.sub(r"[;'\\\"]", "", safe_value)
292
+ safe_value = safe_value.replace('%', '\\%').replace('_', '\\_')
293
  for col in text_cols:
294
+ alt_query = f'SELECT * FROM "{table_name}" WHERE "{col}" LIKE \'%{safe_value}%\' ESCAPE \'\\\' LIMIT {max_rows}'
295
  rows = execute_sql_query(alt_query, cfg)
296
  if rows:
297
  return rows, alt_query
src/sql_retriever.py CHANGED
@@ -3,12 +3,13 @@
3
  import os
4
  import re
5
  import sqlite3
 
6
  from pathlib import Path
7
 
8
 
9
  _DANGEROUS_KEYWORDS = re.compile(
10
  r"\b(INSERT|UPDATE|DELETE|DROP|ALTER|CREATE|ATTACH|DETACH|PRAGMA|"
11
- r"LOAD_EXTENSION|UNION|REPLACE|VACUUM|REINDEX|SAVEPOINT|RELEASE|"
12
  r"ROLLBACK|BEGIN|COMMIT|GRANT|REVOKE|EXPLAIN|WITH)\b",
13
  re.IGNORECASE,
14
  )
@@ -101,9 +102,16 @@ def execute_sql_query(sql_query: str, cfg: dict) -> list[dict]:
101
  max_rows = cfg.get("sql", {}).get("max_rows", 200)
102
 
103
  try:
104
- conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True)
105
  try:
106
  conn.row_factory = sqlite3.Row
 
 
 
 
 
 
 
107
  cursor = conn.execute(sql_query)
108
  rows = cursor.fetchmany(max_rows)
109
  return [dict(row) for row in rows]
@@ -195,7 +203,7 @@ def make_fuzzy_query(sql: str, word_level: bool = False) -> str | None:
195
  Returns the modified query, or None if no substitutions were made.
196
  """
197
  pattern = re.compile(
198
- r"""(["']?\w+["']?\s*)=\s*'([^']+)'""",
199
  )
200
 
201
  if not word_level:
 
3
  import os
4
  import re
5
  import sqlite3
6
+ import time
7
  from pathlib import Path
8
 
9
 
10
  _DANGEROUS_KEYWORDS = re.compile(
11
  r"\b(INSERT|UPDATE|DELETE|DROP|ALTER|CREATE|ATTACH|DETACH|PRAGMA|"
12
+ r"LOAD_EXTENSION|UNION|VACUUM|REINDEX|SAVEPOINT|RELEASE|"
13
  r"ROLLBACK|BEGIN|COMMIT|GRANT|REVOKE|EXPLAIN|WITH)\b",
14
  re.IGNORECASE,
15
  )
 
102
  max_rows = cfg.get("sql", {}).get("max_rows", 200)
103
 
104
  try:
105
+ conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5)
106
  try:
107
  conn.row_factory = sqlite3.Row
108
+ # Abort long-running queries after 10 seconds
109
+ _start = time.monotonic()
110
+ def _progress_check():
111
+ if time.monotonic() - _start > 10:
112
+ return 1 # non-zero = abort
113
+ return 0
114
+ conn.set_progress_handler(_progress_check, 10000)
115
  cursor = conn.execute(sql_query)
116
  rows = cursor.fetchmany(max_rows)
117
  return [dict(row) for row in rows]
 
203
  Returns the modified query, or None if no substitutions were made.
204
  """
205
  pattern = re.compile(
206
+ r"""("[^"]+"\s*|[A-Za-z_]\w*\s*)=\s*'([^']+)'""",
207
  )
208
 
209
  if not word_level:
src/verifier.py CHANGED
@@ -160,14 +160,13 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
160
  """
161
  warnings = []
162
 
163
- # Count unique source documents (not raw chunks)
 
 
 
 
164
  db_results = retrieval_result.get("db_results", [])
165
- unique_db_sources = set()
166
- for chunk in db_results:
167
- source = chunk.get("metadata", {}).get("source", "")
168
- if source:
169
- unique_db_sources.add(source)
170
- db_count = len(unique_db_sources) if unique_db_sources else (1 if db_results else 0)
171
  web_count = len(retrieval_result.get("web_results", []))
172
  sql_count = 1 if retrieval_result.get("sql_results") else 0
173
  total_sources = db_count + web_count + sql_count
@@ -176,7 +175,7 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
176
  return warnings
177
 
178
  # Extract citation numbers from response BODY only (not References section)
179
- refs_split = re.split(r'(?im)^#+\s*(references|sources)\b|^\*\*(references|sources)\*\*', response)
180
  body_text = refs_split[0] if refs_split else response
181
  citation_nums = set(int(m) for m in re.findall(r"\[(\d+)\]", body_text))
182
  if not citation_nums:
@@ -191,7 +190,7 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
191
 
192
  # Check that source file names from retrieval appear in the References section
193
  refs_match = re.search(
194
- r"(?im)(^#+\s*(references|sources)\b|^\*\*(references|sources)\*\*).*",
195
  response, re.DOTALL,
196
  )
197
  if refs_match:
@@ -204,18 +203,25 @@ def validate_citations(response: str, retrieval_result: dict) -> list[str]:
204
  filename = source.rsplit("/", 1)[-1].lower()
205
  if filename in refs_text:
206
  matched_sources += 1
207
- # Also check SQL source files
 
 
 
208
  sql_results = retrieval_result.get("sql_results", [])
209
- for sql_chunk in sql_results:
210
- sql_source = sql_chunk.get("metadata", {}).get("source", "")
211
- if sql_source:
212
- sql_filename = sql_source.rsplit("/", 1)[-1].lower()
 
 
 
213
  if sql_filename in refs_text:
214
  matched_sources += 1
215
- # Also check web source URLs
 
216
  web_results = retrieval_result.get("web_results", [])
217
  for web_chunk in web_results:
218
- web_url = web_chunk.get("metadata", {}).get("url", "")
219
  if web_url and web_url.lower() in refs_text:
220
  matched_sources += 1
221
  # Only warn when db_results are present and are the primary source
@@ -349,7 +355,15 @@ def verify_and_respond(
349
  user_message = query
350
 
351
  # ── Generate initial response ─────────────────────────────────────────
352
- response = generate(system_prompt, user_message, cfg, max_tokens=soft_max)
 
 
 
 
 
 
 
 
353
 
354
  # ── Guard: empty response from LLM ────────────────────────────────────
355
  if not response or not response.strip():
@@ -393,19 +407,18 @@ def verify_and_respond(
393
  max_iterations = verification_cfg.get("max_iterations", 3)
394
  strict_mode = verification_cfg.get("strict_mode", True)
395
 
396
- # Guard: max_iterations=0 with enabled=true → skip verification
 
 
397
  if max_iterations <= 0:
398
- print("WARNING: Verification iterations set to 0. "
399
- "Anti-hallucination verification is effectively disabled.")
400
- return {
401
- "response": response,
402
- "refused": False,
403
- "verification_passed": None,
404
- "iterations": 0,
405
- }
406
 
407
  # ── Verification loop ─────────────────────────────────────────────────
408
  initial_response = response
 
409
  for iteration in range(1, max_iterations + 1):
410
  # Layer 5: Warning-phrase scan
411
  phrase_flags = scan_warning_phrases(response)
@@ -434,6 +447,7 @@ def verify_and_respond(
434
  except Exception:
435
  # Verification LLM failed — skip correction, continue to next iteration
436
  continue
 
437
  vr = parse_verification_result(raw_verification)
438
 
439
  if vr.get("pass", False):
@@ -448,14 +462,16 @@ def verify_and_respond(
448
  error_count = vr.get("error_count", len(vr.get("errors", [])))
449
 
450
  # Correct and loop for re-verification
 
 
451
  correction_prompt = (
452
  f"The user's original question was: {query}\n\n"
453
- f"Your previous response was:\n{response}\n\n"
454
  f"This response failed verification with "
455
  f"{error_count} error(s):\n"
456
  + json.dumps(vr.get("errors", []), indent=2)
457
- + "\n\nRewrite your response to fix ALL the issues listed above. "
458
- "Use ONLY the provided context. Keep all citation rules."
459
  )
460
  try:
461
  response = generate(
@@ -484,6 +500,7 @@ def verify_and_respond(
484
  "You are a strict verification agent. Return only JSON.",
485
  vp, cfg, max_tokens=1024,
486
  ))
 
487
  except Exception:
488
  vr = {"pass": False, "errors": [], "error_count": 0}
489
  if vr.get("pass", False):
@@ -494,7 +511,39 @@ def verify_and_respond(
494
  "iterations": max_iterations,
495
  }
496
 
497
- # ── Exhausted iterations ──────────────────────────────────────────────
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
498
  if strict_mode:
499
  return {
500
  "response": REFUSAL_AFTER_VERIFICATION,
 
160
  """
161
  warnings = []
162
 
163
+ # Count citable units — each retrieved chunk is a separate citable
164
+ # reference that the LLM may cite as [1], [2], etc. Counting unique
165
+ # *files* instead would trigger false "fabricated reference" warnings
166
+ # when a single file contributes multiple chunks (e.g., a PDF cited
167
+ # as [1]–[10] for 10 different sections).
168
  db_results = retrieval_result.get("db_results", [])
169
+ db_count = len(db_results)
 
 
 
 
 
170
  web_count = len(retrieval_result.get("web_results", []))
171
  sql_count = 1 if retrieval_result.get("sql_results") else 0
172
  total_sources = db_count + web_count + sql_count
 
175
  return warnings
176
 
177
  # Extract citation numbers from response BODY only (not References section)
178
+ refs_split = re.split(r'(?im)^#+\s*(?:references|sources)\s*$|^\*\*(?:references|sources)\*\*\s*$', response)
179
  body_text = refs_split[0] if refs_split else response
180
  citation_nums = set(int(m) for m in re.findall(r"\[(\d+)\]", body_text))
181
  if not citation_nums:
 
190
 
191
  # Check that source file names from retrieval appear in the References section
192
  refs_match = re.search(
193
+ r"(?im)(?:^#+\s*(?:references|sources)\s*$|^\*\*(?:references|sources)\*\*\s*$).*",
194
  response, re.DOTALL,
195
  )
196
  if refs_match:
 
203
  filename = source.rsplit("/", 1)[-1].lower()
204
  if filename in refs_text:
205
  matched_sources += 1
206
+ # Also check SQL source — SQL results are plain row dicts (no
207
+ # metadata wrapper). All rows come from a single table/source file
208
+ # whose name is embedded in the context as "Source: <filename>".
209
+ # Extract it from the combined context instead of iterating rows.
210
  sql_results = retrieval_result.get("sql_results", [])
211
+ if sql_results:
212
+ context_text = retrieval_result.get("context", "")
213
+ sql_source_match = re.search(
214
+ r"Source:\s*(.+)", context_text,
215
+ )
216
+ if sql_source_match:
217
+ sql_filename = sql_source_match.group(1).strip().rsplit("/", 1)[-1].lower()
218
  if sql_filename in refs_text:
219
  matched_sources += 1
220
+ # Also check web source URLs — web results are flat dicts with
221
+ # a top-level "url" key (no metadata wrapper).
222
  web_results = retrieval_result.get("web_results", [])
223
  for web_chunk in web_results:
224
+ web_url = web_chunk.get("url", "")
225
  if web_url and web_url.lower() in refs_text:
226
  matched_sources += 1
227
  # Only warn when db_results are present and are the primary source
 
355
  user_message = query
356
 
357
  # ── Generate initial response ─────────────────────────────────────────
358
+ try:
359
+ response = generate(system_prompt, user_message, cfg, max_tokens=soft_max)
360
+ except Exception as e:
361
+ return {
362
+ "response": f"An error occurred while generating the response: {e}",
363
+ "refused": True,
364
+ "verification_passed": None,
365
+ "iterations": 0,
366
+ }
367
 
368
  # ── Guard: empty response from LLM ────────────────────────────────────
369
  if not response or not response.strip():
 
407
  max_iterations = verification_cfg.get("max_iterations", 3)
408
  strict_mode = verification_cfg.get("strict_mode", True)
409
 
410
+ # Guard: max_iterations=0 with enabled=true → clamp to 1.
411
+ # Verification cannot be bypassed via iteration count alone;
412
+ # users must explicitly set verification.enabled: false.
413
  if max_iterations <= 0:
414
+ print("WARNING: Verification iterations set to 0 but verification "
415
+ "is enabled. Clamping to 1 iteration. To disable verification, "
416
+ "set verification.enabled: false.")
417
+ max_iterations = 1
 
 
 
 
418
 
419
  # ── Verification loop ─────────────────────────────────────────────────
420
  initial_response = response
421
+ any_verification_ran = False # Track whether ANY verification call succeeded
422
  for iteration in range(1, max_iterations + 1):
423
  # Layer 5: Warning-phrase scan
424
  phrase_flags = scan_warning_phrases(response)
 
447
  except Exception:
448
  # Verification LLM failed — skip correction, continue to next iteration
449
  continue
450
+ any_verification_ran = True
451
  vr = parse_verification_result(raw_verification)
452
 
453
  if vr.get("pass", False):
 
462
  error_count = vr.get("error_count", len(vr.get("errors", [])))
463
 
464
  # Correct and loop for re-verification
465
+ # Truncate previous response to reduce prompt competition with context
466
+ truncated_response = response[:1500] + "..." if len(response) > 1500 else response
467
  correction_prompt = (
468
  f"The user's original question was: {query}\n\n"
469
+ f"Your previous response (truncated for brevity):\n{truncated_response}\n\n"
470
  f"This response failed verification with "
471
  f"{error_count} error(s):\n"
472
  + json.dumps(vr.get("errors", []), indent=2)
473
+ + "\n\nRewrite your COMPLETE response from scratch to fix ALL the issues "
474
+ "listed above. Use ONLY the provided context. Keep all citation rules."
475
  )
476
  try:
477
  response = generate(
 
500
  "You are a strict verification agent. Return only JSON.",
501
  vp, cfg, max_tokens=1024,
502
  ))
503
+ any_verification_ran = True
504
  except Exception:
505
  vr = {"pass": False, "errors": [], "error_count": 0}
506
  if vr.get("pass", False):
 
511
  "iterations": max_iterations,
512
  }
513
 
514
+ # ── Handle total verification system failure ──────────────────────────
515
+ # If ALL verification LLM calls failed (not "didn't pass" — actually
516
+ # failed to run), this is a system failure, not a content-quality issue.
517
+ if not any_verification_ran:
518
+ print("ERROR: All verification LLM calls failed. "
519
+ "Verification could not run at all.")
520
+ if strict_mode:
521
+ return {
522
+ "response": (
523
+ "Verification was unable to run due to repeated LLM "
524
+ "errors. To avoid presenting unverified information, "
525
+ "I must decline to answer. Please check your LLM "
526
+ "configuration and try again."
527
+ ),
528
+ "refused": True,
529
+ "verification_passed": False,
530
+ "iterations": max_iterations,
531
+ }
532
+ # Non-strict: return with a prominent system-failure warning
533
+ warning = (
534
+ "\n\n---\n**WARNING: Verification system failure.** "
535
+ "The verification pipeline was unable to run due to "
536
+ "repeated LLM errors. This response has NOT been verified "
537
+ "at all — treat all claims as unverified."
538
+ )
539
+ return {
540
+ "response": response + warning,
541
+ "refused": False,
542
+ "verification_passed": False,
543
+ "iterations": max_iterations,
544
+ }
545
+
546
+ # ── Exhausted iterations (verification ran but never passed) ──────────
547
  if strict_mode:
548
  return {
549
  "response": REFUSAL_AFTER_VERIFICATION,
tests/test_retriever.py CHANGED
@@ -203,10 +203,10 @@ def test_retrieve_vector_route_fallback_still_works(
203
  @patch("src.retriever.retrieve_from_vectordb", return_value=[])
204
  @patch("src.retriever._run_sql_retrieval")
205
  @patch("src.retriever._build_fallback_sql_query")
206
- def test_retrieve_both_route_fallback_still_works(
207
  mock_build_fallback, mock_run_sql, mock_vectordb, mock_search,
208
  ):
209
- """Existing behavior: route='both', both empty, keyword SQL fallback runs."""
210
  from src.retriever import retrieve
211
 
212
  # First call: LLM sql_query fails; second call: keyword fallback also fails
@@ -215,10 +215,10 @@ def test_retrieve_both_route_fallback_still_works(
215
 
216
  result = retrieve("some data", _BASE_CFG, route="both", sql_query="SELECT * FROM t WHERE x=1")
217
 
218
- # For route="both", sql_query was already set so _build_fallback_sql_query
219
- # should NOT be called (existing behavior: reuses the sql_query from QU)
220
- mock_build_fallback.assert_not_called()
221
- # _run_sql_retrieval called twice: once for LLM query, once for fallback with same query
222
  assert mock_run_sql.call_count == 2
223
 
224
 
 
203
  @patch("src.retriever.retrieve_from_vectordb", return_value=[])
204
  @patch("src.retriever._run_sql_retrieval")
205
  @patch("src.retriever._build_fallback_sql_query")
206
+ def test_retrieve_both_route_fallback_builds_fresh_query(
207
  mock_build_fallback, mock_run_sql, mock_vectordb, mock_search,
208
  ):
209
+ """When route='both' and LLM sql_query failed, build a fresh keyword query."""
210
  from src.retriever import retrieve
211
 
212
  # First call: LLM sql_query fails; second call: keyword fallback also fails
 
215
 
216
  result = retrieve("some data", _BASE_CFG, route="both", sql_query="SELECT * FROM t WHERE x=1")
217
 
218
+ # T2-03 fix: _build_fallback_sql_query IS called to build a fresh keyword
219
+ # query instead of re-using the already-failed LLM sql_query
220
+ mock_build_fallback.assert_called_once_with("some data", _BASE_CFG)
221
+ # _run_sql_retrieval called twice: once for LLM query, once for fresh fallback
222
  assert mock_run_sql.call_count == 2
223
 
224