import logging from typing import Dict, List, Optional from langchain_groq import ChatGroq from langchain.callbacks import StreamingStdOutCallbackHandler from langchain.callbacks.manager import CallbackManager # Configure Logging logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s') logger = logging.getLogger(__name__) class ThemeSynthesizer: """A service for extracting themes and insights from document collections using Groq Llama models.""" # Class constants for different models available on Groq DEFAULT_MODELS = { "small": "llama3-8b-8192", "medium": "llama3-70b-8192", "large": "llama3-70b-8192", "mixtral": "mixtral-8x7b-32768", "gemma": "gemma-7b-it" } # Template for individual document analysis DOCUMENT_ANALYSIS_TEMPLATE = """ You are an expert document analyst focused on precise extraction. Given the following document content, extract relevant information in response to this query: Query: {query} DOCUMENT CONTENT: {content} Your task: 1. Extract the most relevant information that answers the query 2. Be concise yet complete in your extraction 3. Include EXACT citations by page number and paragraph (e.g., "Page 3, Page 2") 4. Only provide information that is directly mentioned in the document 5. If no relevant information exists, state "No relevant information found" Your answer format must be: EXTRACTED ANSWER: [Your concise answer here] CITATION: [Exact page and paragraph reference] """ # Template for theme synthesis across documents THEME_SYNTHESIS_TEMPLATE = """ You are the expert research analyst synthesizing findings across multiple documents. Below are answers extracted from different documents in response to a specific query. Query: {query} DOCUMENT ANSWERS: {document_answers} Your task: 1. Identify 2-6 key themes or patterns across these documents 2. For each theme, provide a clear title and brief explanation 3. Be concise yet complete in your extraction 4. Mention specific document IDs (e.g, DOC001) that support each theme 5. Format your response as a chat message that synthesizes the information Format your response as follows: Theme 1 – [Theme Title]: Documents ([Document IDs]) [brief explanation with citations] Theme 2 – [Theme Title]: Documents ([Document IDs]) [brief explanation with citations] ...and so on for each theme. """ def __init__( self, model_size: str = "small", custom_model_name: Optional[str] = None, temperature: float = 0.2, max_tokens: int = 8192, streaming: bool = False, api_key: Optional[str] = None ): """Initialize the document analyzer with configurable parameters.""" # Check for API key if not api_key: raise ValueError("Groq API key is required. Get one from https://console.groq.com/") # Set up callbacks for streaming if enabled callbacks = None if streaming: callbacks = CallbackManager([StreamingStdOutCallbackHandler()]) # Select model based on size or custom name model_name = custom_model_name or self.DEFAULT_MODELS.get(model_size, self.DEFAULT_MODELS["small"]) # Initialize the Groq language model try: self.llm = ChatGroq( groq_api_key=api_key, model_name=model_name, temperature=temperature, max_tokens=max_tokens, streaming=streaming, callbacks=callbacks ) logger.info(f"Initialized Groq model: {model_name}") except Exception as e: logger.error(f"Failed to initialize Groq model: {e}") raise ValueError(f"Failed to initialize Groq model: {e}") # Store configuration self.config = { "model": model_name, "temperature": temperature, "max_tokens": max_tokens, "provider": "groq" } def analyze_single_document( self, document: Dict, query: str ) -> Dict: """ Process a single document to extract relevant information for the query. Args: document: Document dictionary with 'content' and 'metadata' query: User query to answer Returns: Dictionary with extracted answer and citation """ try: doc_id = document.get('metadata', {}).get('doc_id', 'Unknown') content = document.get('content', '') if not content.strip(): return { "doc_id": doc_id, "extracted_answer": "No content available in document", "citation": "N/A", "status": "error" } logger.info(f"Analyzing document {doc_id} for query: {query[:50]}...") # Format the prompt with query and content formatted_prompt = self.DOCUMENT_ANALYSIS_TEMPLATE.format( query=query, content=content ) # Run document analysis directly with the LLM response = self.llm.invoke(formatted_prompt) # Extract the result from the response if hasattr(response, 'content'): response = response.content elif isinstance(response, dict) and "text" in response: response = response["text"] # Parse response to extract answer and citation extracted_answer = "No relevant information found" citation = "N/A" # Simple parsing - can be made more robust if "EXTRACTED ANSWER:" in response: answer_parts = response.split("EXTRACTED ANSWER:") if len(answer_parts) > 1: citation_parts = answer_parts[1].split("CITATION:") extracted_answer = citation_parts[0].strip() if len(citation_parts) > 1: citation = citation_parts[1].strip() return { "doc_id": doc_id, "extracted_answer": extracted_answer, "citation": citation, "status": "success" } except Exception as e: logger.error(f"Error analyzing document: {e}", exc_info=True) return { "doc_id": document.get('metadata', {}).get('doc_id', 'Unknown'), "extracted_answer": f"Error during analysis: {str(e)}", "citation": "N/A", "status": "error" } def batch_analyze_documents( self, documents: List[Dict], query: str ) -> List[Dict]: """Process a batch of documents to extract relevant information for the query. Args: documents: List of document dictionaries with 'content' and 'metadata' query: User query to answer Returns: List of dictionaries with extracted answers and citations. """ try: results = [] for doc in documents: # Ensure Document has proper Id if 'metadata' not in doc: doc['metadata'] = {} if 'doc_id' not in doc['metadata']: doc['metadata']['doc_id'] = f"DOC{len(results)+1:03d}" # Analyze document result = self.analyze_single_document(doc, query) results.append(result) return results except Exception as e: logger.error(f"Error batch analyzing documents: {e}", exc_info=True) return [] def synthesize_themes( self, document_results: List[Dict], query: str ) -> Dict: """ Synthesize themes from individual document analysis results. Args: document_results: Results from batch_analyze_documents query: Original user query Returns: Dictionary with synthesized themes and status """ try: if not document_results: return { "themes": "No documents provided for analysis.", "status": "warning" } # Format document answers for the synthesis prompt formatted_answers = [] for res in document_results: doc_id = res.get("doc_id", "Unknown") answer = res.get("extracted_answer", "No answer available") citation = res.get("citation", "N/A") formatted_answer = f"Document ID: {doc_id}\nExtracted Answer: {answer}\nCitation: {citation}" formatted_answers.append(formatted_answer) combined_answers = "\n\n".join(formatted_answers) # Format the prompt with query and document answers formatted_prompt = self.THEME_SYNTHESIS_TEMPLATE.format( query=query, document_answers=combined_answers ) # Run theme synthesis directly with the LLM response = self.llm.invoke(formatted_prompt) # Extract the result from the response if hasattr(response, 'content'): response = response.content elif isinstance(response, dict) and "text" in response: response = response["text"] return { "themes": response.strip(), "status": "success", "stats": { "document_count": len(document_results), "query": query, "model": self.config["model"] } } except Exception as e: logger.error(f"Error synthesizing themes: {e}", exc_info=True) return { "themes": f"Error during theme synthesis: {str(e)}", "status": "error", "error": str(e) } def process_query( self, documents: List[Dict], query: str ) -> Dict: """ Complete document processing pipeline - analyze documents and synthesize themes. Args: documents: List of document dictionaries query: User query to answer Returns: Dictionary with individual document results and synthesized themes """ # Step 1: Analyze each document individually document_results = self.batch_analyze_documents(documents, query) # Step 2: Synthesize themes across documents theme_results = self.synthesize_themes(document_results, query) # Return complete results return { "document_results": document_results, "themes": theme_results.get("themes", ""), "status": "success" if theme_results.get("status") == "success" else "partial", "query": query, "model_info": self.config }