import sys import io import requests from bs4 import BeautifulSoup import pandas as pd from pypdf import PdfReader from langchain_core.tools import tool from langchain_community.tools import DuckDuckGoSearchRun import pptx import docx # If missing, run: pip install python-docx # 1. Native LangChain Search Tool web_search = DuckDuckGoSearchRun( description=( "A wrapper around DuckDuckGo Search. Useful for searching the web for current facts, " "academic papers, and conference proceedings. When looking for specific journal articles " "like Nature Scientific Reports 2012, do not search full complex sentences. Instead, " "pass clean, broad keyword queries (e.g., 'Nature Scientific Reports 2012 conference proceedings nano compound') " "to get the best results. If this tool returns no relevant text, immediately fall back to " "using `execute_python_code` to fetch resources programmatically." ) ) # 2. Resilient Web Scraping Tool @tool def fetch_webpage_content(url: str) -> str: """Useful when you need to read the full text content of a specific URL webpage. It strips away HTML tags and returns clean text snippets.""" try: headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} # Enforce strict 10s connect and read timeouts response = requests.get(url, headers=headers, timeout=(5, 10), stream=False) response.raise_for_status() soup = BeautifulSoup(response.text, "html.parser") # Remove script and style elements for script in soup(["script", "style"]): script.decompose() text = soup.get_text(separator=" ") lines = (line.strip() for line in text.splitlines()) chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) clean_text = "\n".join(chunk for chunk in chunks if chunk) return clean_text[:6000] except Exception as e: return f"Error reading page content: {e}" # 3. Robust Code Sandbox (Standard Output Redirected) @tool def execute_python_code(code: str) -> str: """Useful to run complex calculations, logical processing, or data manipulation. Write full code blocks using print() statements to view target evaluation values.""" old_stdout = sys.stdout redirected_output = sys.stdout = io.StringIO() local_vars = {} try: # Execute code in a clean workspace context exec(code, {"pd": pd, "openpyxl": openpyxl if 'openpyxl' in sys.modules else None}, local_vars) sys.stdout = old_stdout captured_out = redirected_output.getvalue() if not captured_out and local_vars: return f"Execution succeeded. Captured workspace variables: {str(local_vars)}" return captured_out if captured_out else "Execution complete with no output print statements." except Exception as e: sys.stdout = old_stdout return f"Execution Error: {e}" # 4. Specialized Document Analysis Tools @tool def read_local_pdf(file_path: str) -> str: """Extracts raw string text layout from a target local PDF file document.""" try: reader = PdfReader(file_path) extracted_text = "" for page in reader.pages[:10]: # Safe chunk limits extracted_text += page.extract_text() + "\n" return extracted_text[:6000] except Exception as e: return f"Failed to parse PDF document: {e}" @tool def inspect_excel_sheets(file_path: str) -> str: """Reads names of all tabs and displays preview slices of spreadsheets for data inspection.""" try: # Read file bytes into memory first to avoid thread deadlocks on Windows with open(file_path, "rb") as f: file_bytes = io.BytesIO(f.read()) xl = pd.ExcelFile(file_bytes) summary = f"Available sheets/tabs: {xl.sheet_names}\n\n" for sheet in xl.sheet_names[:3]: df = pd.read_excel(file_bytes, sheet_name=sheet) summary += f"--- Sheet: {sheet} (Shape: {df.shape}) ---\n" summary += df.head(5).to_string() + "\n\n" return summary[:6000] except Exception as e: return f"Failed to analyze spreadsheet structure: {e}" @tool def read_local_docx(file_path: str) -> str: """Useful to extract raw text paragraph contents layout from a target local Word document (.docx).""" try: # Read file bytes completely first to prevent thread deadlock issues on Windows systems with open(file_path, "rb") as f: file_bytes = io.BytesIO(f.read()) doc = docx.Document(file_bytes) full_text = [para.text for para in doc.paragraphs if para.text] return "\n".join(full_text)[:6000] except Exception as e: return f"Failed to parse Word document: {e}" @tool def read_local_pptx(file_path: str) -> str: """Useful to extract raw text slide-by-slide from a target PowerPoint presentation (.pptx).""" try: with open(file_path, "rb") as f: file_bytes = io.BytesIO(f.read()) prs = pptx.Presentation(file_bytes) summary = "" for i, slide in enumerate(prs.slides, start=1): slide_text = [] for shape in slide.shapes: if shape.has_text_frame: slide_text.append(shape.text_frame.text.strip()) text_content = " | ".join(t for t in slide_text if t) summary += f"Slide {i}: {text_content}\n" return summary[:6000] except Exception as e: return f"Failed to parse PowerPoint file: {e}" # Expose complete toolkit array to LangGraph agent executor all_tools = [ web_search, fetch_webpage_content, execute_python_code, read_local_pdf, inspect_excel_sheets, read_local_docx, read_local_pptx ]