Spaces:
Running on Zero
Running on Zero
| import sys | |
| import io | |
| import requests | |
| from bs4 import BeautifulSoup | |
| import pandas as pd | |
| from pypdf import PdfReader | |
| from langchain_core.tools import tool | |
| from langchain_community.tools import DuckDuckGoSearchRun | |
| import pptx | |
| import docx # If missing, run: pip install python-docx | |
| # 1. Native LangChain Search Tool | |
| web_search = DuckDuckGoSearchRun( | |
| description=( | |
| "A wrapper around DuckDuckGo Search. Useful for searching the web for current facts, " | |
| "academic papers, and conference proceedings. When looking for specific journal articles " | |
| "like Nature Scientific Reports 2012, do not search full complex sentences. Instead, " | |
| "pass clean, broad keyword queries (e.g., 'Nature Scientific Reports 2012 conference proceedings nano compound') " | |
| "to get the best results. If this tool returns no relevant text, immediately fall back to " | |
| "using `execute_python_code` to fetch resources programmatically." | |
| ) | |
| ) | |
| # 2. Resilient Web Scraping Tool | |
| def fetch_webpage_content(url: str) -> str: | |
| """Useful when you need to read the full text content of a specific URL webpage. | |
| It strips away HTML tags and returns clean text snippets.""" | |
| try: | |
| headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} | |
| # Enforce strict 10s connect and read timeouts | |
| response = requests.get(url, headers=headers, timeout=(5, 10), stream=False) | |
| response.raise_for_status() | |
| soup = BeautifulSoup(response.text, "html.parser") | |
| # Remove script and style elements | |
| for script in soup(["script", "style"]): | |
| script.decompose() | |
| text = soup.get_text(separator=" ") | |
| lines = (line.strip() for line in text.splitlines()) | |
| chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) | |
| clean_text = "\n".join(chunk for chunk in chunks if chunk) | |
| return clean_text[:6000] | |
| except Exception as e: | |
| return f"Error reading page content: {e}" | |
| # 3. Robust Code Sandbox (Standard Output Redirected) | |
| def execute_python_code(code: str) -> str: | |
| """Useful to run complex calculations, logical processing, or data manipulation. | |
| Write full code blocks using print() statements to view target evaluation values.""" | |
| old_stdout = sys.stdout | |
| redirected_output = sys.stdout = io.StringIO() | |
| local_vars = {} | |
| try: | |
| # Execute code in a clean workspace context | |
| exec(code, {"pd": pd, "openpyxl": openpyxl if 'openpyxl' in sys.modules else None}, local_vars) | |
| sys.stdout = old_stdout | |
| captured_out = redirected_output.getvalue() | |
| if not captured_out and local_vars: | |
| return f"Execution succeeded. Captured workspace variables: {str(local_vars)}" | |
| return captured_out if captured_out else "Execution complete with no output print statements." | |
| except Exception as e: | |
| sys.stdout = old_stdout | |
| return f"Execution Error: {e}" | |
| # 4. Specialized Document Analysis Tools | |
| def read_local_pdf(file_path: str) -> str: | |
| """Extracts raw string text layout from a target local PDF file document.""" | |
| try: | |
| reader = PdfReader(file_path) | |
| extracted_text = "" | |
| for page in reader.pages[:10]: # Safe chunk limits | |
| extracted_text += page.extract_text() + "\n" | |
| return extracted_text[:6000] | |
| except Exception as e: | |
| return f"Failed to parse PDF document: {e}" | |
| def inspect_excel_sheets(file_path: str) -> str: | |
| """Reads names of all tabs and displays preview slices of spreadsheets for data inspection.""" | |
| try: | |
| # Read file bytes into memory first to avoid thread deadlocks on Windows | |
| with open(file_path, "rb") as f: | |
| file_bytes = io.BytesIO(f.read()) | |
| xl = pd.ExcelFile(file_bytes) | |
| summary = f"Available sheets/tabs: {xl.sheet_names}\n\n" | |
| for sheet in xl.sheet_names[:3]: | |
| df = pd.read_excel(file_bytes, sheet_name=sheet) | |
| summary += f"--- Sheet: {sheet} (Shape: {df.shape}) ---\n" | |
| summary += df.head(5).to_string() + "\n\n" | |
| return summary[:6000] | |
| except Exception as e: | |
| return f"Failed to analyze spreadsheet structure: {e}" | |
| def read_local_docx(file_path: str) -> str: | |
| """Useful to extract raw text paragraph contents layout from a target local Word document (.docx).""" | |
| try: | |
| # Read file bytes completely first to prevent thread deadlock issues on Windows systems | |
| with open(file_path, "rb") as f: | |
| file_bytes = io.BytesIO(f.read()) | |
| doc = docx.Document(file_bytes) | |
| full_text = [para.text for para in doc.paragraphs if para.text] | |
| return "\n".join(full_text)[:6000] | |
| except Exception as e: | |
| return f"Failed to parse Word document: {e}" | |
| def read_local_pptx(file_path: str) -> str: | |
| """Useful to extract raw text slide-by-slide from a target PowerPoint presentation (.pptx).""" | |
| try: | |
| with open(file_path, "rb") as f: | |
| file_bytes = io.BytesIO(f.read()) | |
| prs = pptx.Presentation(file_bytes) | |
| summary = "" | |
| for i, slide in enumerate(prs.slides, start=1): | |
| slide_text = [] | |
| for shape in slide.shapes: | |
| if shape.has_text_frame: | |
| slide_text.append(shape.text_frame.text.strip()) | |
| text_content = " | ".join(t for t in slide_text if t) | |
| summary += f"Slide {i}: {text_content}\n" | |
| return summary[:6000] | |
| except Exception as e: | |
| return f"Failed to parse PowerPoint file: {e}" | |
| # Expose complete toolkit array to LangGraph agent executor | |
| all_tools = [ | |
| web_search, | |
| fetch_webpage_content, | |
| execute_python_code, | |
| read_local_pdf, | |
| inspect_excel_sheets, | |
| read_local_docx, | |
| read_local_pptx | |
| ] |