Spaces:
Runtime error
Runtime error
File size: 2,888 Bytes
36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 bbafb55 36425a4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 | import os
import glob
import re
import uuid
from dotenv import load_dotenv
from backend.app.utils.chunking import semantic_chunking
from backend.app.services.embedding_service import EmbeddingService
from backend.app.services.qdrant_service import QdrantService
load_dotenv(os.path.join(os.path.dirname(__file__), '..', 'app', '.env'))
def parse_chapter_info(file_path: str):
filename = os.path.basename(file_path)
match = re.match(r'chapter-(\d+)-(.+)\.md', filename)
if match:
return int(match.group(1)), match.group(2).replace('-', ' ').title()
return 0, "Unknown"
CHAPTER_TITLES = [
"Introduction to Physical AI",
"Basics of Humanoid Robotics",
"ROS 2 Fundamentals",
"Digital Twin Simulation",
"Vision-Language-Action Systems",
"Capstone: Simple AI-Robot Pipeline",
]
def index_chapters():
print("Starting chapter indexing...")
# Initialize services
embedding_service = EmbeddingService()
qdrant_service = QdrantService()
chapter_files = sorted(glob.glob("frontend/docs/chapter-*.md"))
if not chapter_files:
print("No chapter files found in website/docs/. Please ensure chapters exist.")
return
for file_path in chapter_files:
print(f"Processing {file_path}...")
chapter_num, chapter_title = parse_chapter_info(file_path)
with open(file_path, 'r', encoding='utf-8') as f:
markdown_content = f.read()
chunks = semantic_chunking(markdown_content)
# Add chapter title card for better "chapter X" queries
title_card = f"Chapter {chapter_num}: {chapter_title}. This chapter covers {chapter_title}."
title_embedding = embedding_service.encode([title_card])[0]
qdrant_service.upsert_chunks(
ids=[str(uuid.uuid4())],
vectors=[title_embedding],
payloads=[{
"source": file_path,
"chapter_number": chapter_num,
"chapter_title": chapter_title,
"heading": "Overview",
"content": f"Chapter {chapter_num}: {chapter_title}",
"chunk_number": -1,
"type": "title_card",
}]
)
for i, chunk in enumerate(chunks):
chunk_content = chunk["content"]
metadata = chunk["metadata"]
metadata["source"] = file_path
metadata["chapter_number"] = chapter_num
metadata["chapter_title"] = chapter_title
metadata["chunk_number"] = i
metadata["content"] = chunk_content
point_id = str(uuid.uuid4())
embedding = embedding_service.encode([chunk_content])[0]
qdrant_service.upsert_chunks(ids=[point_id], vectors=[embedding], payloads=[metadata])
print("Chapter indexing completed.")
if __name__ == "__main__":
index_chapters()
|