File size: 2,350 Bytes
7124acf | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 | """Which documents an analysis may use — the one definition (U-D3).
Go materialises the analysis-scope catalog from `analyses.data_bind`; every bound
pdf/docx/txt appears there as a source with `source_type="unstructured"` and
`source_id = documents.id` (`catalog/service.go::buildBoundSource`, verified
2026-09-23). This reads that row. It is the scope for document retrieval
(`unstructured_flow` and `retrieve_knowledge`) and the binding check for the
image-token endpoint.
**No scope row → no documents.** Decided with Rifqi 2026-09-23 to match the
code's existing behaviour for `structured_flow` and `check`
(`AnalysisScopedCatalogReader.read`, no user-scope fallback since 2026-07-13):
Go writes the row whenever anything is bound, so a miss means "nothing bound"
or a Go failure, and answering from unbound documents is the misleading-source
bug that change fixed. A failed read degrades the same way, logged.
"""
from __future__ import annotations
from typing import Any
from src.middlewares.logging import get_logger
logger = get_logger("document_scope")
async def bound_documents(
user_id: str, analysis_id: str | None, store: Any = None
) -> list[tuple[str, str]]:
"""(document_id, display name) for every document bound to the analysis.
Never raises. Empty when there is no analysis id, no scope row, or the read
failed (the last is logged with `degraded_seam`).
"""
if not analysis_id:
return []
if store is None:
from src.catalog.store import CatalogStore
store = CatalogStore()
try:
catalog = await store.get_by_analysis(analysis_id, user_id)
except Exception as e:
logger.warning(
"document scope read failed — no documents in scope",
analysis_id=analysis_id,
error=repr(e),
degraded_seam="document_scope",
)
return []
if catalog is None:
logger.info("no analysis scope row — no documents in scope", analysis_id=analysis_id)
return []
return [
(s.source_id, s.name)
for s in catalog.sources
if s.source_type == "unstructured"
]
async def bound_document_ids(
user_id: str, analysis_id: str | None, store: Any = None
) -> list[str]:
return [doc_id for doc_id, _ in await bound_documents(user_id, analysis_id, store)]
|