Files changed (1) hide show
  1. code2 +32 -0
code2 ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import requests
2
+ from bs4 import BeautifulSoup
3
+
4
+ @tool
5
+ def fetch_webpage_content(url: str) -> str:
6
+ """
7
+ Fetches the raw text content of a specific web URL and cleans it for reading.
8
+
9
+ Args:
10
+ url: The exact website URL string to scrape.
11
+ """
12
+ headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
13
+ try:
14
+ response = requests.get(url, headers=headers, timeout=15)
15
+ if response.status_code != 200:
16
+ return f"Failed to retrieve page. Status code: {response.status_code}"
17
+
18
+ soup = BeautifulSoup(response.text, 'html.parser')
19
+
20
+ # Remove non-text elements to save token space
21
+ for script in soup(["script", "style", "nav", "footer"]):
22
+ script.extract()
23
+
24
+ text = soup.get_text(separator=' ')
25
+ # Clean up whitespace chunking
26
+ lines = (line.strip() for line in text.splitlines())
27
+ chunks = (phrase.strip() for line in lines for phrase in line.split(" "))
28
+ clean_text = '\n'.join(chunk for chunk in chunks if chunk)
29
+
30
+ return clean_text[:8000] # Cap context to prevent token bloat
31
+ except Exception as e:
32
+ return f"Scraping error encountered: {str(e)}"