Spaces:
Paused
Paused
Download src/dashboard/docs_source.py from ThomasHeisig/Brain-5D-Space: direct link, hf CLI and curl.
- Browser
- Download file 19.5 kB
-
https://huggingface.co/spaces/ThomasHeisig/Brain-5D-Space/resolve/main/src/dashboard/docs_source.py
- Command line
-
hf download hf://spaces/ThomasHeisig/Brain-5D-Space/src/dashboard/docs_source.py
-
curl -L -o docs_source.py https://huggingface.co/spaces/ThomasHeisig/Brain-5D-Space/resolve/main/src/dashboard/docs_source.py
19.5 kB
| """Safe read-only access to repository documentation and data files. | |
| Supports: .md, .txt, .docx, .xlsx, .csv, .json, .pdf (metadata only) | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import json | |
| from dataclasses import dataclass | |
| from datetime import datetime | |
| from enum import Enum | |
| from functools import lru_cache | |
| from importlib import import_module | |
| from pathlib import Path | |
| from typing import Any, Callable, cast | |
| from .models import JSONValue | |
| # Optional imports with proper fallbacks – using lower-case names to avoid Pylance constant redefinition warnings. | |
| DocxDocument: Any = None | |
| load_workbook: Callable[..., Any] | None = None | |
| pypdf: Any = None | |
| try: | |
| from docx import ( | |
| Document as _DocxDocument, # pyright: ignore[reportMissingTypeStubs] | |
| ) | |
| DocxDocument = _DocxDocument | |
| has_docx: bool = True | |
| except ImportError: | |
| has_docx = False | |
| try: | |
| from openpyxl import ( # type: ignore[import-untyped] | |
| load_workbook as _load_workbook, # pyright: ignore[reportMissingTypeStubs, reportUnknownVariableType] | |
| ) | |
| load_workbook = cast(Callable[..., Any], _load_workbook) | |
| has_openpyxl: bool = True | |
| except ImportError: | |
| has_openpyxl = False | |
| try: | |
| pypdf = import_module("pypdf") | |
| has_pypdf: bool = True | |
| except ImportError: | |
| has_pypdf = False | |
| __all__ = [ | |
| "DocumentationEntry", | |
| "DocumentationSource", | |
| "FileType", | |
| "create_docs_source", | |
| ] | |
| class FileType(Enum): | |
| """Supported document file types.""" | |
| MARKDOWN = "markdown" | |
| TEXT = "text" | |
| DOCX = "docx" | |
| XLSX = "xlsx" | |
| CSV = "csv" | |
| JSON = "json" | |
| PDF = "pdf" | |
| UNKNOWN = "unknown" | |
| def from_extension(cls, ext: str) -> FileType: | |
| """Map file extension to FileType.""" | |
| ext = ext.lower().lstrip(".") | |
| mapping: dict[str, FileType] = { | |
| "md": cls.MARKDOWN, | |
| "markdown": cls.MARKDOWN, | |
| "txt": cls.TEXT, | |
| "text": cls.TEXT, | |
| "docx": cls.DOCX, | |
| "xlsx": cls.XLSX, | |
| "xls": cls.XLSX, | |
| "csv": cls.CSV, | |
| "json": cls.JSON, | |
| "pdf": cls.PDF, | |
| } | |
| return mapping.get(ext, cls.UNKNOWN) | |
| def is_editable(self) -> bool: | |
| """Whether this file type can be edited in a text editor.""" | |
| return self in {FileType.MARKDOWN, FileType.TEXT, FileType.CSV, FileType.JSON} | |
| def is_binary(self) -> bool: | |
| """Whether this file type is binary.""" | |
| return self in {FileType.DOCX, FileType.XLSX, FileType.PDF} | |
| class DocumentationEntry: | |
| """Rich metadata for a documentation or data file.""" | |
| name: str | |
| path: str | |
| size_bytes: int | |
| file_type: FileType | |
| modified_time: str | |
| content_preview: str | None = None | |
| word_count: int | None = None | |
| line_count: int | None = None | |
| sheet_names: tuple[str, ...] | None = None | |
| supported: bool = True | |
| def to_json(self) -> dict[str, JSONValue]: | |
| """Return JSON-ready document metadata.""" | |
| result: dict[str, JSONValue] = { | |
| "name": self.name, | |
| "path": self.path, | |
| "size_bytes": self.size_bytes, | |
| "file_type": self.file_type.value, | |
| "modified_time": self.modified_time, | |
| "supported": self.supported, | |
| } | |
| if self.content_preview is not None: | |
| result["content_preview"] = self.content_preview | |
| if self.word_count is not None: | |
| result["word_count"] = self.word_count | |
| if self.line_count is not None: | |
| result["line_count"] = self.line_count | |
| if self.sheet_names is not None: | |
| result["sheet_names"] = list(self.sheet_names) | |
| return result | |
| class DocumentationSource: | |
| """Expose documentation and data files with multi-format support. | |
| Provides safe, read-only access to files below a fixed root directory. | |
| Supports content extraction for .md, .txt, .docx, .xlsx, .csv, .json. | |
| """ | |
| def __init__( | |
| self, | |
| docs_root: Path, | |
| max_preview_chars: int = 500, | |
| max_file_size_mb: int = 50, | |
| enable_caching: bool = True, | |
| ) -> None: | |
| """Initialize documentation source. | |
| Args: | |
| docs_root: Root directory for documentation files. | |
| max_preview_chars: Maximum characters for content preview. | |
| max_file_size_mb: Maximum file size to process (MB). | |
| enable_caching: Whether to cache file listings. | |
| """ | |
| self.docs_root = docs_root.resolve() | |
| self.max_preview_chars = max_preview_chars | |
| self.max_file_size_bytes = max_file_size_mb * 1024 * 1024 | |
| self.enable_caching = enable_caching | |
| def list_documents(self, recursive: bool = False) -> tuple[DocumentationEntry, ...]: | |
| """List all supported documents in stable order. | |
| Args: | |
| recursive: Whether to scan subdirectories recursively. | |
| Returns: | |
| Tuple of DocumentationEntry objects. | |
| """ | |
| if not self.docs_root.is_dir(): | |
| return () | |
| pattern = "**/*" if recursive else "*" | |
| entries: list[DocumentationEntry] = [] | |
| for path in sorted(self.docs_root.glob(pattern)): | |
| if not path.is_file(): | |
| continue | |
| if self._is_excluded(path): | |
| continue | |
| entry = self._build_entry(path) | |
| if entry is not None: | |
| entries.append(entry) | |
| return tuple(entries) | |
| def list_documents_by_type( | |
| self, file_type: FileType | |
| ) -> tuple[DocumentationEntry, ...]: | |
| """List documents filtered by file type.""" | |
| return tuple( | |
| entry | |
| for entry in self.list_documents(recursive=True) | |
| if entry.file_type == file_type | |
| ) | |
| def get_document(self, path: str) -> DocumentationEntry: | |
| """Get metadata for a specific document.""" | |
| resolved_path = self._resolve_path(path) | |
| if resolved_path is None or not resolved_path.is_file(): | |
| raise FileNotFoundError(f"Document not found: {path}") | |
| entry = self._build_entry(resolved_path) | |
| if entry is None: | |
| raise ValueError(f"Unsupported file type: {path}") | |
| return entry | |
| def read(self, path: str) -> str: | |
| """Read the full content of a document (convenience alias).""" | |
| return self.read_content(path) | |
| def read_content(self, path: str) -> str: | |
| """Read the full content of a document (text extraction).""" | |
| resolved_path = self._resolve_path(path) | |
| if resolved_path is None or not resolved_path.is_file(): | |
| raise FileNotFoundError(f"Document not found: {path}") | |
| file_type = FileType.from_extension(resolved_path.suffix) | |
| return self._extract_content(resolved_path, file_type, full=True) | |
| def read_preview(self, path: str) -> str: | |
| """Read a preview of the document content.""" | |
| resolved_path = self._resolve_path(path) | |
| if resolved_path is None or not resolved_path.is_file(): | |
| raise FileNotFoundError(f"Document not found: {path}") | |
| file_type = FileType.from_extension(resolved_path.suffix) | |
| return self._extract_content(resolved_path, file_type, full=False) | |
| def get_directory_structure(self) -> dict[str, Any]: | |
| """Get the full directory tree with metadata.""" | |
| if not self.docs_root.is_dir(): | |
| return {"path": str(self.docs_root), "children": []} | |
| return self._build_tree(self.docs_root) | |
| def _build_tree(self, path: Path, relative_path: str = "") -> dict[str, Any]: | |
| """Recursively build directory tree.""" | |
| result: dict[str, Any] = { | |
| "name": path.name if relative_path else str(path), | |
| "path": relative_path or ".", | |
| "type": "directory", | |
| "children": [], | |
| } | |
| for child in sorted(path.iterdir()): | |
| if self._is_excluded(child): | |
| continue | |
| if child.is_dir(): | |
| child_path = str(child.relative_to(self.docs_root)) | |
| result["children"].append(self._build_tree(child, child_path)) | |
| elif child.is_file(): | |
| entry = self._build_entry(child) | |
| if entry is not None: | |
| result["children"].append( | |
| { | |
| "name": child.name, | |
| "path": str(child.relative_to(self.docs_root)), | |
| "type": "file", | |
| "size_bytes": child.stat().st_size, | |
| "file_type": entry.file_type.value, | |
| "modified_time": entry.modified_time, | |
| } | |
| ) | |
| return result | |
| def _build_entry(self, path: Path) -> DocumentationEntry | None: | |
| """Build a DocumentationEntry from a file path.""" | |
| try: | |
| stat = path.stat() | |
| file_type = FileType.from_extension(path.suffix) | |
| supported = self._is_supported(file_type) | |
| # Base metadata | |
| name = path.name | |
| rel_path = str(path.relative_to(self.docs_root)) | |
| size_bytes = stat.st_size | |
| modified_time = datetime.fromtimestamp(stat.st_mtime).isoformat() | |
| entry_kwargs: dict[str, Any] = { | |
| "name": name, | |
| "path": rel_path, | |
| "size_bytes": size_bytes, | |
| "file_type": file_type, | |
| "modified_time": modified_time, | |
| "supported": supported, | |
| } | |
| # Skip content extraction for unsupported or large files | |
| if not supported or size_bytes > self.max_file_size_bytes: | |
| return DocumentationEntry(**entry_kwargs) # pyright: ignore[arg-type] | |
| # Extract content preview and metrics | |
| try: | |
| content = self._extract_content(path, file_type, full=False) | |
| if content: | |
| entry_kwargs["content_preview"] = content[: self.max_preview_chars] | |
| if file_type.is_editable: | |
| full_content = self._extract_content(path, file_type, full=True) | |
| if full_content: | |
| entry_kwargs["word_count"] = len(full_content.split()) | |
| entry_kwargs["line_count"] = full_content.count("\n") + 1 | |
| # Sheet names for Excel | |
| if ( | |
| file_type == FileType.XLSX | |
| and has_openpyxl | |
| and load_workbook is not None | |
| ): | |
| try: | |
| wb = load_workbook(path, read_only=True, data_only=True) | |
| entry_kwargs["sheet_names"] = tuple(wb.sheetnames) | |
| wb.close() | |
| except Exception: | |
| pass | |
| except Exception: | |
| pass | |
| return DocumentationEntry(**entry_kwargs) # pyright: ignore[arg-type] | |
| except Exception: | |
| return None | |
| def _extract_content( | |
| self, path: Path, file_type: FileType, full: bool = False | |
| ) -> str: | |
| """Extract text content from a file based on its type.""" | |
| if file_type == FileType.MARKDOWN or file_type == FileType.TEXT: | |
| return path.read_text(encoding="utf-8", errors="replace") | |
| if file_type == FileType.CSV: | |
| return self._extract_csv_content(path, full) | |
| if file_type == FileType.JSON: | |
| return self._extract_json_content(path, full) | |
| if file_type == FileType.DOCX and has_docx and DocxDocument is not None: | |
| return self._extract_docx_content(path, full) | |
| if file_type == FileType.XLSX and has_openpyxl and load_workbook is not None: | |
| return self._extract_xlsx_content(path, full) | |
| if file_type == FileType.PDF and has_pypdf and pypdf is not None: | |
| return self._extract_pdf_content(path, full) | |
| return f"Content extraction not available for {file_type.value} files." | |
| def _extract_csv_content(self, path: Path, full: bool) -> str: | |
| """Extract CSV content as readable text.""" | |
| try: | |
| lines: list[str] = [] | |
| with open(path, encoding="utf-8", errors="replace") as f: | |
| reader = csv.reader(f) | |
| if not full: | |
| for i, row in enumerate(reader): | |
| if i >= 10: | |
| lines.append("... (truncated)") | |
| break | |
| lines.append(" | ".join(row)) | |
| else: | |
| for row in reader: | |
| lines.append(" | ".join(row)) | |
| return "\n".join(lines) | |
| except Exception: | |
| return f"[Could not parse CSV: {path.name}]" | |
| def _extract_json_content(self, path: Path, full: bool) -> str: | |
| """Extract JSON content as formatted text.""" | |
| try: | |
| data = json.loads(path.read_text(encoding="utf-8", errors="replace")) | |
| if not full: | |
| if isinstance(data, list): | |
| # JSON data has arbitrary structure that Pylance cannot infer. | |
| preview = data[:2] # pyright: ignore | |
| if len(data) > 2: # pyright: ignore | |
| preview.append("...") # pyright: ignore | |
| return json.dumps(preview, indent=2, ensure_ascii=False) | |
| elif isinstance(data, dict): | |
| items = list(data.items()) # pyright: ignore | |
| preview_dict = dict(items[:5]) # pyright: ignore | |
| if len(items) > 5: # pyright: ignore | |
| preview_dict["..."] = ( | |
| f"({len(items) - 5} more keys)" # pyright: ignore | |
| ) | |
| return json.dumps(preview_dict, indent=2, ensure_ascii=False) | |
| return json.dumps(data, indent=2, ensure_ascii=False) | |
| except Exception: | |
| return f"[Could not parse JSON: {path.name}]" | |
| def _extract_docx_content(self, path: Path, full: bool) -> str: | |
| """Extract text from DOCX file.""" | |
| try: | |
| if DocxDocument is None: | |
| return "[python-docx not available]" | |
| doc = DocxDocument(str(path)) | |
| paragraphs: list[str] = [] | |
| for p in doc.paragraphs: | |
| if p.text.strip(): | |
| paragraphs.append(p.text) | |
| if not full: | |
| paragraphs = paragraphs[:20] | |
| if len(doc.paragraphs) > 20: | |
| paragraphs.append("... (truncated)") | |
| return "\n".join(paragraphs) | |
| except Exception: | |
| return f"[Could not read DOCX: {path.name}]" | |
| def _extract_xlsx_content(self, path: Path, full: bool) -> str: | |
| """Extract text from XLSX file.""" | |
| try: | |
| if load_workbook is None: | |
| return "[openpyxl not available]" | |
| wb = load_workbook(path, read_only=True, data_only=True) | |
| lines: list[str] = [] | |
| for sheet_name in wb.sheetnames: | |
| sheet = wb[sheet_name] | |
| lines.append(f"\n=== Sheet: {sheet_name} ===\n") | |
| max_rows = 50 if not full else 1000 | |
| row_count = 0 | |
| for row in sheet.iter_rows(values_only=True): | |
| if row_count >= max_rows: | |
| lines.append("... (truncated)") | |
| break | |
| row_str = " | ".join( | |
| str(cell) if cell is not None else "" for cell in row | |
| ) | |
| if row_str.strip(): | |
| lines.append(row_str) | |
| row_count += 1 | |
| wb.close() | |
| return "\n".join(lines) | |
| except Exception: | |
| return f"[Could not read XLSX: {path.name}]" | |
| def _extract_pdf_content(self, path: Path, full: bool) -> str: | |
| """Extract text from PDF file.""" | |
| try: | |
| if pypdf is None: | |
| return "[pypdf not available]" | |
| text_parts: list[str] = [] | |
| with open(path, "rb") as f: | |
| pdf = pypdf.PdfReader(f) | |
| max_pages = 5 if not full else len(pdf.pages) | |
| for i in range(min(max_pages, len(pdf.pages))): | |
| page = pdf.pages[i] | |
| page_text = page.extract_text() | |
| if page_text: | |
| text_parts.append(page_text) | |
| if not full and len(pdf.pages) > 5: | |
| text_parts.append("... (truncated)") | |
| return "\n".join(text_parts) | |
| except Exception: | |
| return f"[Could not read PDF: {path.name}]" | |
| def _resolve_path(self, path: str) -> Path | None: | |
| """Resolve a path relative to docs_root with security checks.""" | |
| # Basic security: prevent path traversal | |
| if ".." in path or path.startswith("/") or path.startswith("\\"): | |
| return None | |
| try: | |
| full_path = (self.docs_root / path).resolve() | |
| except ValueError: | |
| return None | |
| # Ensure path is within docs_root | |
| try: | |
| full_path.relative_to(self.docs_root) | |
| except ValueError: | |
| return None | |
| return full_path | |
| def _is_excluded(self, path: Path) -> bool: | |
| """Check if a path should be excluded.""" | |
| if path.name.startswith("."): | |
| return True | |
| exclude_dirs = {"__pycache__", ".git", ".venv", "node_modules", ".pytest_cache"} | |
| if path.name in exclude_dirs: | |
| return True | |
| return False | |
| def _is_supported(self, file_type: FileType) -> bool: | |
| """Check if a file type is supported for content extraction.""" | |
| if file_type in {FileType.MARKDOWN, FileType.TEXT, FileType.CSV, FileType.JSON}: | |
| return True | |
| if file_type == FileType.DOCX: | |
| return has_docx | |
| if file_type == FileType.XLSX: | |
| return has_openpyxl | |
| if file_type == FileType.PDF: | |
| return has_pypdf | |
| return False | |
| def _get_cached_list(self, recursive: bool) -> tuple[DocumentationEntry, ...]: | |
| """Cached version of list_documents for performance.""" | |
| return self.list_documents(recursive) | |
| def invalidate_cache(self) -> None: | |
| """Invalidate the internal cache.""" | |
| self._get_cached_list.cache_clear() | |
| def search_documents( | |
| self, query: str, file_types: list[FileType] | None = None | |
| ) -> tuple[DocumentationEntry, ...]: | |
| """Search documents by filename (basic search).""" | |
| query_lower = query.lower() | |
| results: list[DocumentationEntry] = [] | |
| for entry in self.list_documents(recursive=True): | |
| if file_types is not None and entry.file_type not in file_types: | |
| continue | |
| if query_lower in entry.name.lower(): | |
| results.append(entry) | |
| return tuple(results) | |
| def create_docs_source(docs_root: str | Path) -> DocumentationSource: | |
| """Factory function for creating a DocumentationSource.""" | |
| if isinstance(docs_root, str): | |
| docs_root = Path(docs_root) | |
| return DocumentationSource(docs_root) | |