Spaces:
Runtime error
Runtime error
Download theme.py from Nikpatil/chatbot_theme_identifier: direct link, hf CLI and curl.
- Browser
- Download file 11.2 kB
-
https://huggingface.co/spaces/Nikpatil/chatbot_theme_identifier/resolve/main/theme.py
- Command line
-
hf download hf://spaces/Nikpatil/chatbot_theme_identifier/theme.py
-
curl -L -o theme.py https://huggingface.co/spaces/Nikpatil/chatbot_theme_identifier/resolve/main/theme.py
11.2 kB
| import logging | |
| from typing import Dict, List, Optional | |
| from langchain_groq import ChatGroq | |
| from langchain.callbacks import StreamingStdOutCallbackHandler | |
| from langchain.callbacks.manager import CallbackManager | |
| # Configure Logging | |
| logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s') | |
| logger = logging.getLogger(__name__) | |
| class ThemeSynthesizer: | |
| """A service for extracting themes and insights from document collections using Groq Llama models.""" | |
| # Class constants for different models available on Groq | |
| DEFAULT_MODELS = { | |
| "small": "llama3-8b-8192", | |
| "medium": "llama3-70b-8192", | |
| "large": "llama3-70b-8192", | |
| "mixtral": "mixtral-8x7b-32768", | |
| "gemma": "gemma-7b-it" | |
| } | |
| # Template for individual document analysis | |
| DOCUMENT_ANALYSIS_TEMPLATE = """ | |
| You are an expert document analyst focused on precise extraction. | |
| Given the following document content, extract relevant information in response to this query: | |
| Query: {query} | |
| DOCUMENT CONTENT: | |
| {content} | |
| Your task: | |
| 1. Extract the most relevant information that answers the query | |
| 2. Be concise yet complete in your extraction | |
| 3. Include EXACT citations by page number and paragraph (e.g., "Page 3, Page 2") | |
| 4. Only provide information that is directly mentioned in the document | |
| 5. If no relevant information exists, state "No relevant information found" | |
| Your answer format must be: | |
| EXTRACTED ANSWER: [Your concise answer here] | |
| CITATION: [Exact page and paragraph reference] | |
| """ | |
| # Template for theme synthesis across documents | |
| THEME_SYNTHESIS_TEMPLATE = """ | |
| You are the expert research analyst synthesizing findings across multiple documents. | |
| Below are answers extracted from different documents in response to a specific query. | |
| Query: {query} | |
| DOCUMENT ANSWERS: | |
| {document_answers} | |
| Your task: | |
| 1. Identify 2-6 key themes or patterns across these documents | |
| 2. For each theme, provide a clear title and brief explanation | |
| 3. Be concise yet complete in your extraction | |
| 4. Mention specific document IDs (e.g, DOC001) that support each theme | |
| 5. Format your response as a chat message that synthesizes the information | |
| Format your response as follows: | |
| Theme 1 – [Theme Title]: | |
| Documents ([Document IDs]) [brief explanation with citations] | |
| Theme 2 – [Theme Title]: | |
| Documents ([Document IDs]) [brief explanation with citations] | |
| ...and so on for each theme. | |
| """ | |
| def __init__( | |
| self, | |
| model_size: str = "small", | |
| custom_model_name: Optional[str] = None, | |
| temperature: float = 0.2, | |
| max_tokens: int = 8192, | |
| streaming: bool = False, | |
| api_key: Optional[str] = None | |
| ): | |
| """Initialize the document analyzer with configurable parameters.""" | |
| # Check for API key | |
| if not api_key: | |
| raise ValueError("Groq API key is required. Get one from https://console.groq.com/") | |
| # Set up callbacks for streaming if enabled | |
| callbacks = None | |
| if streaming: | |
| callbacks = CallbackManager([StreamingStdOutCallbackHandler()]) | |
| # Select model based on size or custom name | |
| model_name = custom_model_name or self.DEFAULT_MODELS.get(model_size, self.DEFAULT_MODELS["small"]) | |
| # Initialize the Groq language model | |
| try: | |
| self.llm = ChatGroq( | |
| groq_api_key=api_key, | |
| model_name=model_name, | |
| temperature=temperature, | |
| max_tokens=max_tokens, | |
| streaming=streaming, | |
| callbacks=callbacks | |
| ) | |
| logger.info(f"Initialized Groq model: {model_name}") | |
| except Exception as e: | |
| logger.error(f"Failed to initialize Groq model: {e}") | |
| raise ValueError(f"Failed to initialize Groq model: {e}") | |
| # Store configuration | |
| self.config = { | |
| "model": model_name, | |
| "temperature": temperature, | |
| "max_tokens": max_tokens, | |
| "provider": "groq" | |
| } | |
| def analyze_single_document( | |
| self, | |
| document: Dict, | |
| query: str | |
| ) -> Dict: | |
| """ | |
| Process a single document to extract relevant information for the query. | |
| Args: | |
| document: Document dictionary with 'content' and 'metadata' | |
| query: User query to answer | |
| Returns: | |
| Dictionary with extracted answer and citation | |
| """ | |
| try: | |
| doc_id = document.get('metadata', {}).get('doc_id', 'Unknown') | |
| content = document.get('content', '') | |
| if not content.strip(): | |
| return { | |
| "doc_id": doc_id, | |
| "extracted_answer": "No content available in document", | |
| "citation": "N/A", | |
| "status": "error" | |
| } | |
| logger.info(f"Analyzing document {doc_id} for query: {query[:50]}...") | |
| # Format the prompt with query and content | |
| formatted_prompt = self.DOCUMENT_ANALYSIS_TEMPLATE.format( | |
| query=query, | |
| content=content | |
| ) | |
| # Run document analysis directly with the LLM | |
| response = self.llm.invoke(formatted_prompt) | |
| # Extract the result from the response | |
| if hasattr(response, 'content'): | |
| response = response.content | |
| elif isinstance(response, dict) and "text" in response: | |
| response = response["text"] | |
| # Parse response to extract answer and citation | |
| extracted_answer = "No relevant information found" | |
| citation = "N/A" | |
| # Simple parsing - can be made more robust | |
| if "EXTRACTED ANSWER:" in response: | |
| answer_parts = response.split("EXTRACTED ANSWER:") | |
| if len(answer_parts) > 1: | |
| citation_parts = answer_parts[1].split("CITATION:") | |
| extracted_answer = citation_parts[0].strip() | |
| if len(citation_parts) > 1: | |
| citation = citation_parts[1].strip() | |
| return { | |
| "doc_id": doc_id, | |
| "extracted_answer": extracted_answer, | |
| "citation": citation, | |
| "status": "success" | |
| } | |
| except Exception as e: | |
| logger.error(f"Error analyzing document: {e}", exc_info=True) | |
| return { | |
| "doc_id": document.get('metadata', {}).get('doc_id', 'Unknown'), | |
| "extracted_answer": f"Error during analysis: {str(e)}", | |
| "citation": "N/A", | |
| "status": "error" | |
| } | |
| def batch_analyze_documents( | |
| self, | |
| documents: List[Dict], | |
| query: str | |
| ) -> List[Dict]: | |
| """Process a batch of documents to extract relevant information for the query. | |
| Args: | |
| documents: List of document dictionaries with 'content' and 'metadata' | |
| query: User query to answer | |
| Returns: | |
| List of dictionaries with extracted answers and citations. | |
| """ | |
| try: | |
| results = [] | |
| for doc in documents: | |
| # Ensure Document has proper Id | |
| if 'metadata' not in doc: | |
| doc['metadata'] = {} | |
| if 'doc_id' not in doc['metadata']: | |
| doc['metadata']['doc_id'] = f"DOC{len(results)+1:03d}" | |
| # Analyze document | |
| result = self.analyze_single_document(doc, query) | |
| results.append(result) | |
| return results | |
| except Exception as e: | |
| logger.error(f"Error batch analyzing documents: {e}", exc_info=True) | |
| return [] | |
| def synthesize_themes( | |
| self, | |
| document_results: List[Dict], | |
| query: str | |
| ) -> Dict: | |
| """ | |
| Synthesize themes from individual document analysis results. | |
| Args: | |
| document_results: Results from batch_analyze_documents | |
| query: Original user query | |
| Returns: | |
| Dictionary with synthesized themes and status | |
| """ | |
| try: | |
| if not document_results: | |
| return { | |
| "themes": "No documents provided for analysis.", | |
| "status": "warning" | |
| } | |
| # Format document answers for the synthesis prompt | |
| formatted_answers = [] | |
| for res in document_results: | |
| doc_id = res.get("doc_id", "Unknown") | |
| answer = res.get("extracted_answer", "No answer available") | |
| citation = res.get("citation", "N/A") | |
| formatted_answer = f"Document ID: {doc_id}\nExtracted Answer: {answer}\nCitation: {citation}" | |
| formatted_answers.append(formatted_answer) | |
| combined_answers = "\n\n".join(formatted_answers) | |
| # Format the prompt with query and document answers | |
| formatted_prompt = self.THEME_SYNTHESIS_TEMPLATE.format( | |
| query=query, | |
| document_answers=combined_answers | |
| ) | |
| # Run theme synthesis directly with the LLM | |
| response = self.llm.invoke(formatted_prompt) | |
| # Extract the result from the response | |
| if hasattr(response, 'content'): | |
| response = response.content | |
| elif isinstance(response, dict) and "text" in response: | |
| response = response["text"] | |
| return { | |
| "themes": response.strip(), | |
| "status": "success", | |
| "stats": { | |
| "document_count": len(document_results), | |
| "query": query, | |
| "model": self.config["model"] | |
| } | |
| } | |
| except Exception as e: | |
| logger.error(f"Error synthesizing themes: {e}", exc_info=True) | |
| return { | |
| "themes": f"Error during theme synthesis: {str(e)}", | |
| "status": "error", | |
| "error": str(e) | |
| } | |
| def process_query( | |
| self, | |
| documents: List[Dict], | |
| query: str | |
| ) -> Dict: | |
| """ | |
| Complete document processing pipeline - analyze documents and synthesize themes. | |
| Args: | |
| documents: List of document dictionaries | |
| query: User query to answer | |
| Returns: | |
| Dictionary with individual document results and synthesized themes | |
| """ | |
| # Step 1: Analyze each document individually | |
| document_results = self.batch_analyze_documents(documents, query) | |
| # Step 2: Synthesize themes across documents | |
| theme_results = self.synthesize_themes(document_results, query) | |
| # Return complete results | |
| return { | |
| "document_results": document_results, | |
| "themes": theme_results.get("themes", ""), | |
| "status": "success" if theme_results.get("status") == "success" else "partial", | |
| "query": query, | |
| "model_info": self.config | |
| } |