Nikpatil's picture
Upload 8 files
407b660 verified
Raw History Blame Contribute Delete
11.2 kB
import logging
from typing import Dict, List, Optional
from langchain_groq import ChatGroq
from langchain.callbacks import StreamingStdOutCallbackHandler
from langchain.callbacks.manager import CallbackManager
# Configure Logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
class ThemeSynthesizer:
"""A service for extracting themes and insights from document collections using Groq Llama models."""
# Class constants for different models available on Groq
DEFAULT_MODELS = {
"small": "llama3-8b-8192",
"medium": "llama3-70b-8192",
"large": "llama3-70b-8192",
"mixtral": "mixtral-8x7b-32768",
"gemma": "gemma-7b-it"
}
# Template for individual document analysis
DOCUMENT_ANALYSIS_TEMPLATE = """
You are an expert document analyst focused on precise extraction.
Given the following document content, extract relevant information in response to this query:
Query: {query}
DOCUMENT CONTENT:
{content}
Your task:
1. Extract the most relevant information that answers the query
2. Be concise yet complete in your extraction
3. Include EXACT citations by page number and paragraph (e.g., "Page 3, Page 2")
4. Only provide information that is directly mentioned in the document
5. If no relevant information exists, state "No relevant information found"
Your answer format must be:
EXTRACTED ANSWER: [Your concise answer here]
CITATION: [Exact page and paragraph reference]
"""
# Template for theme synthesis across documents
THEME_SYNTHESIS_TEMPLATE = """
You are the expert research analyst synthesizing findings across multiple documents.
Below are answers extracted from different documents in response to a specific query.
Query: {query}
DOCUMENT ANSWERS:
{document_answers}
Your task:
1. Identify 2-6 key themes or patterns across these documents
2. For each theme, provide a clear title and brief explanation
3. Be concise yet complete in your extraction
4. Mention specific document IDs (e.g, DOC001) that support each theme
5. Format your response as a chat message that synthesizes the information
Format your response as follows:
Theme 1 – [Theme Title]:
Documents ([Document IDs]) [brief explanation with citations]
Theme 2 – [Theme Title]:
Documents ([Document IDs]) [brief explanation with citations]
...and so on for each theme.
"""
def __init__(
self,
model_size: str = "small",
custom_model_name: Optional[str] = None,
temperature: float = 0.2,
max_tokens: int = 8192,
streaming: bool = False,
api_key: Optional[str] = None
):
"""Initialize the document analyzer with configurable parameters."""
# Check for API key
if not api_key:
raise ValueError("Groq API key is required. Get one from https://console.groq.com/")
# Set up callbacks for streaming if enabled
callbacks = None
if streaming:
callbacks = CallbackManager([StreamingStdOutCallbackHandler()])
# Select model based on size or custom name
model_name = custom_model_name or self.DEFAULT_MODELS.get(model_size, self.DEFAULT_MODELS["small"])
# Initialize the Groq language model
try:
self.llm = ChatGroq(
groq_api_key=api_key,
model_name=model_name,
temperature=temperature,
max_tokens=max_tokens,
streaming=streaming,
callbacks=callbacks
)
logger.info(f"Initialized Groq model: {model_name}")
except Exception as e:
logger.error(f"Failed to initialize Groq model: {e}")
raise ValueError(f"Failed to initialize Groq model: {e}")
# Store configuration
self.config = {
"model": model_name,
"temperature": temperature,
"max_tokens": max_tokens,
"provider": "groq"
}
def analyze_single_document(
self,
document: Dict,
query: str
) -> Dict:
"""
Process a single document to extract relevant information for the query.
Args:
document: Document dictionary with 'content' and 'metadata'
query: User query to answer
Returns:
Dictionary with extracted answer and citation
"""
try:
doc_id = document.get('metadata', {}).get('doc_id', 'Unknown')
content = document.get('content', '')
if not content.strip():
return {
"doc_id": doc_id,
"extracted_answer": "No content available in document",
"citation": "N/A",
"status": "error"
}
logger.info(f"Analyzing document {doc_id} for query: {query[:50]}...")
# Format the prompt with query and content
formatted_prompt = self.DOCUMENT_ANALYSIS_TEMPLATE.format(
query=query,
content=content
)
# Run document analysis directly with the LLM
response = self.llm.invoke(formatted_prompt)
# Extract the result from the response
if hasattr(response, 'content'):
response = response.content
elif isinstance(response, dict) and "text" in response:
response = response["text"]
# Parse response to extract answer and citation
extracted_answer = "No relevant information found"
citation = "N/A"
# Simple parsing - can be made more robust
if "EXTRACTED ANSWER:" in response:
answer_parts = response.split("EXTRACTED ANSWER:")
if len(answer_parts) > 1:
citation_parts = answer_parts[1].split("CITATION:")
extracted_answer = citation_parts[0].strip()
if len(citation_parts) > 1:
citation = citation_parts[1].strip()
return {
"doc_id": doc_id,
"extracted_answer": extracted_answer,
"citation": citation,
"status": "success"
}
except Exception as e:
logger.error(f"Error analyzing document: {e}", exc_info=True)
return {
"doc_id": document.get('metadata', {}).get('doc_id', 'Unknown'),
"extracted_answer": f"Error during analysis: {str(e)}",
"citation": "N/A",
"status": "error"
}
def batch_analyze_documents(
self,
documents: List[Dict],
query: str
) -> List[Dict]:
"""Process a batch of documents to extract relevant information for the query.
Args:
documents: List of document dictionaries with 'content' and 'metadata'
query: User query to answer
Returns:
List of dictionaries with extracted answers and citations.
"""
try:
results = []
for doc in documents:
# Ensure Document has proper Id
if 'metadata' not in doc:
doc['metadata'] = {}
if 'doc_id' not in doc['metadata']:
doc['metadata']['doc_id'] = f"DOC{len(results)+1:03d}"
# Analyze document
result = self.analyze_single_document(doc, query)
results.append(result)
return results
except Exception as e:
logger.error(f"Error batch analyzing documents: {e}", exc_info=True)
return []
def synthesize_themes(
self,
document_results: List[Dict],
query: str
) -> Dict:
"""
Synthesize themes from individual document analysis results.
Args:
document_results: Results from batch_analyze_documents
query: Original user query
Returns:
Dictionary with synthesized themes and status
"""
try:
if not document_results:
return {
"themes": "No documents provided for analysis.",
"status": "warning"
}
# Format document answers for the synthesis prompt
formatted_answers = []
for res in document_results:
doc_id = res.get("doc_id", "Unknown")
answer = res.get("extracted_answer", "No answer available")
citation = res.get("citation", "N/A")
formatted_answer = f"Document ID: {doc_id}\nExtracted Answer: {answer}\nCitation: {citation}"
formatted_answers.append(formatted_answer)
combined_answers = "\n\n".join(formatted_answers)
# Format the prompt with query and document answers
formatted_prompt = self.THEME_SYNTHESIS_TEMPLATE.format(
query=query,
document_answers=combined_answers
)
# Run theme synthesis directly with the LLM
response = self.llm.invoke(formatted_prompt)
# Extract the result from the response
if hasattr(response, 'content'):
response = response.content
elif isinstance(response, dict) and "text" in response:
response = response["text"]
return {
"themes": response.strip(),
"status": "success",
"stats": {
"document_count": len(document_results),
"query": query,
"model": self.config["model"]
}
}
except Exception as e:
logger.error(f"Error synthesizing themes: {e}", exc_info=True)
return {
"themes": f"Error during theme synthesis: {str(e)}",
"status": "error",
"error": str(e)
}
def process_query(
self,
documents: List[Dict],
query: str
) -> Dict:
"""
Complete document processing pipeline - analyze documents and synthesize themes.
Args:
documents: List of document dictionaries
query: User query to answer
Returns:
Dictionary with individual document results and synthesized themes
"""
# Step 1: Analyze each document individually
document_results = self.batch_analyze_documents(documents, query)
# Step 2: Synthesize themes across documents
theme_results = self.synthesize_themes(document_results, query)
# Return complete results
return {
"document_results": document_results,
"themes": theme_results.get("themes", ""),
"status": "success" if theme_results.get("status") == "success" else "partial",
"query": query,
"model_info": self.config
}