Spaces:
Running on Zero
Running on Zero
File size: 1,735 Bytes
b106f38 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 | """CLI Document Ingestion Utility.
Usage:
python scripts/ingest.py path/to/document1.pdf path/to/document2.pdf
"""
import argparse
import logging
from pathlib import Path
import sys
# Ensure project root is in sys.path
PROJECT_DIR = Path(__file__).resolve().parent.parent
if str(PROJECT_DIR) not in sys.path:
sys.path.insert(0, str(PROJECT_DIR))
from app.config import get_settings, setup_logging
from app.state import get_app_state
def main():
parser = argparse.ArgumentParser(description="Ingest and index PDF documents into the vector store.")
parser.add_argument("pdf_files", nargs="+", type=Path, help="Paths to PDF documents to index.")
args = parser.parse_args()
settings = get_settings()
setup_logging(settings.log_level)
logger = logging.getLogger("ingest_cli")
state = get_app_state()
logger.info("Starting ingestion for %d documents...", len(args.pdf_files))
success_count = 0
for file_path in args.pdf_files:
if not file_path.exists():
logger.error("File does not exist: %s", file_path)
continue
try:
res = state.index_pdf_file(file_path)
logger.info(
"Successfully indexed '%s': %d pages, %d chunks. Total store chunks: %d",
res["filename"],
res["pages_extracted"],
res["chunks_created"],
res["total_indexed_chunks"],
)
success_count += 1
except Exception as e:
logger.error("Failed to index '%s': %s", file_path.name, e)
logger.info("Ingestion finished. %d/%d documents successfully indexed.", success_count, len(args.pdf_files))
if __name__ == "__main__":
main()
|