Download code2 from agents-course/Final_Assignment_Template: direct link, hf CLI and curl.
- Browser
- Download file 1.24 kB
-
https://huggingface.co/spaces/agents-course/Final_Assignment_Template/resolve/refs%2Fpr%2F589/code2
- Command line
-
hf download hf://spaces/agents-course/Final_Assignment_Template@refs/pr/589/code2
-
curl -L -o code2 https://huggingface.co/spaces/agents-course/Final_Assignment_Template/resolve/refs%2Fpr%2F589/code2
1.24 kB
| import requests | |
| from bs4 import BeautifulSoup | |
| def fetch_webpage_content(url: str) -> str: | |
| """ | |
| Fetches the raw text content of a specific web URL and cleans it for reading. | |
| Args: | |
| url: The exact website URL string to scrape. | |
| """ | |
| headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} | |
| try: | |
| response = requests.get(url, headers=headers, timeout=15) | |
| if response.status_code != 200: | |
| return f"Failed to retrieve page. Status code: {response.status_code}" | |
| soup = BeautifulSoup(response.text, 'html.parser') | |
| # Remove non-text elements to save token space | |
| for script in soup(["script", "style", "nav", "footer"]): | |
| script.extract() | |
| text = soup.get_text(separator=' ') | |
| # Clean up whitespace chunking | |
| lines = (line.strip() for line in text.splitlines()) | |
| chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) | |
| clean_text = '\n'.join(chunk for chunk in chunks if chunk) | |
| return clean_text[:8000] # Cap context to prevent token bloat | |
| except Exception as e: | |
| return f"Scraping error encountered: {str(e)}" | |