Spaces:
Runtime error
Runtime error
Download scripts/index_chapters.py from Abdullahcoder54/Hackaton1_BOOK_chatbot: direct link, hf CLI and curl.
- Browser
- Download file 2.89 kB
-
https://huggingface.co/spaces/Abdullahcoder54/Hackaton1_BOOK_chatbot/resolve/main/scripts/index_chapters.py
- Command line
-
hf download hf://spaces/Abdullahcoder54/Hackaton1_BOOK_chatbot/scripts/index_chapters.py
-
curl -L -o index_chapters.py https://huggingface.co/spaces/Abdullahcoder54/Hackaton1_BOOK_chatbot/resolve/main/scripts/index_chapters.py
2.89 kB
| import os | |
| import glob | |
| import re | |
| import uuid | |
| from dotenv import load_dotenv | |
| from backend.app.utils.chunking import semantic_chunking | |
| from backend.app.services.embedding_service import EmbeddingService | |
| from backend.app.services.qdrant_service import QdrantService | |
| load_dotenv(os.path.join(os.path.dirname(__file__), '..', 'app', '.env')) | |
| def parse_chapter_info(file_path: str): | |
| filename = os.path.basename(file_path) | |
| match = re.match(r'chapter-(\d+)-(.+)\.md', filename) | |
| if match: | |
| return int(match.group(1)), match.group(2).replace('-', ' ').title() | |
| return 0, "Unknown" | |
| CHAPTER_TITLES = [ | |
| "Introduction to Physical AI", | |
| "Basics of Humanoid Robotics", | |
| "ROS 2 Fundamentals", | |
| "Digital Twin Simulation", | |
| "Vision-Language-Action Systems", | |
| "Capstone: Simple AI-Robot Pipeline", | |
| ] | |
| def index_chapters(): | |
| print("Starting chapter indexing...") | |
| # Initialize services | |
| embedding_service = EmbeddingService() | |
| qdrant_service = QdrantService() | |
| chapter_files = sorted(glob.glob("frontend/docs/chapter-*.md")) | |
| if not chapter_files: | |
| print("No chapter files found in website/docs/. Please ensure chapters exist.") | |
| return | |
| for file_path in chapter_files: | |
| print(f"Processing {file_path}...") | |
| chapter_num, chapter_title = parse_chapter_info(file_path) | |
| with open(file_path, 'r', encoding='utf-8') as f: | |
| markdown_content = f.read() | |
| chunks = semantic_chunking(markdown_content) | |
| # Add chapter title card for better "chapter X" queries | |
| title_card = f"Chapter {chapter_num}: {chapter_title}. This chapter covers {chapter_title}." | |
| title_embedding = embedding_service.encode([title_card])[0] | |
| qdrant_service.upsert_chunks( | |
| ids=[str(uuid.uuid4())], | |
| vectors=[title_embedding], | |
| payloads=[{ | |
| "source": file_path, | |
| "chapter_number": chapter_num, | |
| "chapter_title": chapter_title, | |
| "heading": "Overview", | |
| "content": f"Chapter {chapter_num}: {chapter_title}", | |
| "chunk_number": -1, | |
| "type": "title_card", | |
| }] | |
| ) | |
| for i, chunk in enumerate(chunks): | |
| chunk_content = chunk["content"] | |
| metadata = chunk["metadata"] | |
| metadata["source"] = file_path | |
| metadata["chapter_number"] = chapter_num | |
| metadata["chapter_title"] = chapter_title | |
| metadata["chunk_number"] = i | |
| metadata["content"] = chunk_content | |
| point_id = str(uuid.uuid4()) | |
| embedding = embedding_service.encode([chunk_content])[0] | |
| qdrant_service.upsert_chunks(ids=[point_id], vectors=[embedding], payloads=[metadata]) | |
| print("Chapter indexing completed.") | |
| if __name__ == "__main__": | |
| index_chapters() | |