File size: 2,888 Bytes
36425a4
 
bbafb55
36425a4
 
 
 
 
 
 
bbafb55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36425a4
 
 
 
bbafb55
 
 
36425a4
bbafb55
36425a4
 
 
 
 
 
bbafb55
36425a4
 
 
 
 
bbafb55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
36425a4
 
 
bbafb55
 
 
36425a4
bbafb55
36425a4
 
bbafb55
36425a4
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
import os
import glob
import re
import uuid
from dotenv import load_dotenv

from backend.app.utils.chunking import semantic_chunking
from backend.app.services.embedding_service import EmbeddingService
from backend.app.services.qdrant_service import QdrantService

load_dotenv(os.path.join(os.path.dirname(__file__), '..', 'app', '.env'))

def parse_chapter_info(file_path: str):
    filename = os.path.basename(file_path)
    match = re.match(r'chapter-(\d+)-(.+)\.md', filename)
    if match:
        return int(match.group(1)), match.group(2).replace('-', ' ').title()
    return 0, "Unknown"

CHAPTER_TITLES = [
    "Introduction to Physical AI",
    "Basics of Humanoid Robotics",
    "ROS 2 Fundamentals",
    "Digital Twin Simulation",
    "Vision-Language-Action Systems",
    "Capstone: Simple AI-Robot Pipeline",
]

def index_chapters():
    print("Starting chapter indexing...")

    # Initialize services
    embedding_service = EmbeddingService()
    qdrant_service = QdrantService()

    chapter_files = sorted(glob.glob("frontend/docs/chapter-*.md"))
    if not chapter_files:
        print("No chapter files found in website/docs/. Please ensure chapters exist.")
        return

    for file_path in chapter_files:
        print(f"Processing {file_path}...")
        chapter_num, chapter_title = parse_chapter_info(file_path)
        with open(file_path, 'r', encoding='utf-8') as f:
            markdown_content = f.read()

        chunks = semantic_chunking(markdown_content)

        # Add chapter title card for better "chapter X" queries
        title_card = f"Chapter {chapter_num}: {chapter_title}. This chapter covers {chapter_title}."
        title_embedding = embedding_service.encode([title_card])[0]
        qdrant_service.upsert_chunks(
            ids=[str(uuid.uuid4())],
            vectors=[title_embedding],
            payloads=[{
                "source": file_path,
                "chapter_number": chapter_num,
                "chapter_title": chapter_title,
                "heading": "Overview",
                "content": f"Chapter {chapter_num}: {chapter_title}",
                "chunk_number": -1,
                "type": "title_card",
            }]
        )

        for i, chunk in enumerate(chunks):
            chunk_content = chunk["content"]
            metadata = chunk["metadata"]
            metadata["source"] = file_path
            metadata["chapter_number"] = chapter_num
            metadata["chapter_title"] = chapter_title
            metadata["chunk_number"] = i
            metadata["content"] = chunk_content
            
            point_id = str(uuid.uuid4())
            embedding = embedding_service.encode([chunk_content])[0]
            qdrant_service.upsert_chunks(ids=[point_id], vectors=[embedding], payloads=[metadata])

    print("Chapter indexing completed.")

if __name__ == "__main__":
    index_chapters()