rishik1111's picture
feat: Add VisionQuest Indic RAG core services, vector search, safety guardrails & web UI
83dc8bf
Raw
History Blame Contribute Delete
2.6 kB
"""
Chunk metadata schemas, tagging utilities, and language pre-filtering.
Strict Extensibility Requirement:
Language filtering operates dynamically against any language code in `config.LANGUAGES`.
"""
import re
from typing import Any, Dict, List, Optional
from pydantic import BaseModel, Field
class Chunk(BaseModel):
"""
Standardized schema for all indexed chunks across chunking strategies and languages.
"""
chunk_id: str
text: str
embed_text: str # The exact text passed to the embedding model
chunk_strategy: str # 'passage_native', 'sentence_window', 'semantic'
source_lang: str # 'hi', 'ta', 'en', etc.
token_count: int
source_query_ids: List[int] = Field(default_factory=list)
doc_id: Optional[str] = None
context_window: Optional[str] = None
metadata: Dict[str, Any] = Field(default_factory=dict)
def to_dict(self) -> Dict[str, Any]:
return self.model_dump()
def estimate_token_count(text: str) -> int:
"""
Estimate token count for multilingual text (word-based approximation with character factor).
"""
if not text:
return 0
words = text.split()
# For Indic scripts, characters/words roughly map to 1.3-1.5 subword tokens
return max(len(words), int(len(text) / 4.0))
def filter_chunks_by_language(chunks: List[Chunk], target_lang: str) -> List[Chunk]:
"""
Language-based pre-filter to restrict search or candidate selection to the query language.
"""
if not target_lang:
return chunks
target_clean = target_lang.lower().strip()
return [c for c in chunks if c.source_lang.lower() == target_clean]
def split_sentences_multilingual(text: str) -> List[str]:
"""
Robust multilingual sentence splitter handling Indic punctuation (।, ॥) and standard (. ! ?).
"""
if not text:
return []
# Split on Devanagari/Indic danda (।), double danda (॥), period, question mark, exclamation mark, newlines
raw_sentences = re.split(r'(?<=[।॥.!?\n])\s+', text)
cleaned = []
for s in raw_sentences:
s_str = s.strip()
if len(s_str) > 3:
cleaned.append(s_str)
return cleaned if cleaned else [text.strip()]
def calculate_overlap_tokens(text: str, overlap_percent: float = 0.15) -> str:
"""
Extract trailing overlap text (10-20% tokens) from the end of a chunk to prepend to the next.
"""
words = text.split()
if len(words) < 5:
return ""
num_overlap_words = max(1, int(len(words) * overlap_percent))
return " ".join(words[-num_overlap_words:])