import os import glob import re import uuid from dotenv import load_dotenv from backend.app.utils.chunking import semantic_chunking from backend.app.services.embedding_service import EmbeddingService from backend.app.services.qdrant_service import QdrantService load_dotenv(os.path.join(os.path.dirname(__file__), '..', 'app', '.env')) def parse_chapter_info(file_path: str): filename = os.path.basename(file_path) match = re.match(r'chapter-(\d+)-(.+)\.md', filename) if match: return int(match.group(1)), match.group(2).replace('-', ' ').title() return 0, "Unknown" CHAPTER_TITLES = [ "Introduction to Physical AI", "Basics of Humanoid Robotics", "ROS 2 Fundamentals", "Digital Twin Simulation", "Vision-Language-Action Systems", "Capstone: Simple AI-Robot Pipeline", ] def index_chapters(): print("Starting chapter indexing...") # Initialize services embedding_service = EmbeddingService() qdrant_service = QdrantService() chapter_files = sorted(glob.glob("frontend/docs/chapter-*.md")) if not chapter_files: print("No chapter files found in website/docs/. Please ensure chapters exist.") return for file_path in chapter_files: print(f"Processing {file_path}...") chapter_num, chapter_title = parse_chapter_info(file_path) with open(file_path, 'r', encoding='utf-8') as f: markdown_content = f.read() chunks = semantic_chunking(markdown_content) # Add chapter title card for better "chapter X" queries title_card = f"Chapter {chapter_num}: {chapter_title}. This chapter covers {chapter_title}." title_embedding = embedding_service.encode([title_card])[0] qdrant_service.upsert_chunks( ids=[str(uuid.uuid4())], vectors=[title_embedding], payloads=[{ "source": file_path, "chapter_number": chapter_num, "chapter_title": chapter_title, "heading": "Overview", "content": f"Chapter {chapter_num}: {chapter_title}", "chunk_number": -1, "type": "title_card", }] ) for i, chunk in enumerate(chunks): chunk_content = chunk["content"] metadata = chunk["metadata"] metadata["source"] = file_path metadata["chapter_number"] = chapter_num metadata["chapter_title"] = chapter_title metadata["chunk_number"] = i metadata["content"] = chunk_content point_id = str(uuid.uuid4()) embedding = embedding_service.encode([chunk_content])[0] qdrant_service.upsert_chunks(ids=[point_id], vectors=[embedding], payloads=[metadata]) print("Chapter indexing completed.") if __name__ == "__main__": index_chapters()