dev.altai / src /crawler.py
prince1604
Improve URL normalization logic and add API docs
3176f9b
Raw
History Blame Contribute Delete
22.8 kB
from curl_cffi import requests
import time
import random
from bs4 import BeautifulSoup
from urllib.parse import urljoin, urlparse
import sys
import logging
import os
import concurrent.futures
import cloudscraper
from fake_useragent import UserAgent
import threading
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
class Crawler:
def __init__(self, use_playwright=True):
# 1. Improved HTTP Client Setup
# impersonate="chrome120" is key. verify=False to speed up SSL handshakes.
self.session = requests.Session(impersonate="chrome120", verify=False)
self.ua = UserAgent()
# ssl_context=None might not be exposed, but we can try to optimize
self.scraper = cloudscraper.create_scraper(
browser={'browser': 'chrome', 'platform': 'windows', 'desktop': True}
)
# Concurrency control for Playwright (heavy resource)
self.playwright_semaphore = threading.Semaphore(3)
self.session.headers.update({
"Referer": "https://www.google.com/",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
"Accept-Language": "en-US,en;q=0.9",
"Upgrade-Insecure-Requests": "1",
"Sec-Fetch-Site": "none",
"Sec-Fetch-Mode": "navigate",
"Sec-Fetch-User": "?1",
"Sec-Fetch-Dest": "document",
})
self.visited_urls = set()
self.use_playwright_fallback = use_playwright
self.blocked_reason = None
def fetch(self, url):
"""
Tiered Fetching Strategy:
1. curl_cffi (Fastest, good evasion)
2. cloudscraper (Specialized for Cloudflare/WAF)
3. Playwright (Heaviest, comprehensive)
"""
# Tier 1: curl_cffi
content = self._fetch_http(url)
if content:
# Validate content isn't a block page
if not self._is_blocked(content):
return content
else:
logger.warning(f"Tier 1 fetched blocked content for {url}. Escalating...")
# Tier 2: Cloudscraper (Intermediate, handles JS challenges)
logger.info(f"Tier 1 failed. Trying Cloudscraper info for {url}...")
try:
resp = self.scraper.get(url, timeout=10)
if 200 <= resp.status_code < 300:
if not self._is_blocked(resp.text):
return resp.text
else:
logger.warning(f"Cloudscraper also blocked for {url}.")
except Exception as e:
logger.warning(f"Cloudscraper failed: {e}")
# Tier 3: Playwright Fallback
if self.use_playwright_fallback:
logger.info(f"Attempts failed. utilizing Playwright (Limit 3 concurrent) for {url}")
return self._fetch_playwright(url)
return None
def _is_blocked(self, content):
"""
Detects if the content is likely a WAF block page, CAPTCHA, or 'Just a moment'.
"""
if not content or len(content) < 500:
return True # Too small, suspicious
lower_content = content.lower()
block_keywords = [
"just a moment...",
"enable javascript",
"verify you are human",
"access denied",
"cloudflare",
"captcha",
"security check",
"turn on javascript",
"challenge.js",
"api-services-support@amazon.com",
"we just need to make sure you're not a robot",
"type the characters you see in this image"
]
if any(k in lower_content for k in block_keywords):
# Double check: sometimes legitimate pages mention these words.
# But usually, if it's < 5KB and has these words, it's a block.
if len(content) < 5000:
return True
return False
def _fetch_http(self, url):
"""
Fast HTTP fetch with aggressive timeouts and minimal retries.
"""
max_retries = 2
for attempt in range(max_retries):
try:
# Rotate User Agent
self.session.headers["User-Agent"] = self.ua.random
# Reduced timeout to 5 seconds
# verify=False is inherent in the session from init, but good to be explicit if needed (curl_cffi uses session setting)
response = self.session.get(url, timeout=5, allow_redirects=True)
# Check for 200 OK (or close to it)
if 200 <= response.status_code < 300:
return response.text
if response.status_code == 403:
# Fail fast on 403 to trigger fallback immediately if needed
return response.text
if response.status_code in [429, 500, 502, 503]:
if attempt < max_retries - 1:
# Minimal sleep for speed
time.sleep(0.5)
continue
return None
except Exception as e:
# If curl_cffi fails, we skip the slow standard requests fallback
# and return None to let the main loop decide (or trigger Playwright if configured)
logger.warning(f"Fast fetch failed for {url}: {e}")
return None
def _fetch_playwright(self, url):
"""
Fallback Strategy A: Use Playwright with 'playwright-stealth' library.
Includes ULTRA-VERBOSE Logging for debugging.
"""
logger.info(f"[-] initiating_playwright_fetch for: {url}")
try:
from playwright.sync_api import sync_playwright
# Try importing stealth, if fails, continue without it but warn
try:
from playwright_stealth import stealth_sync
has_stealth = True
logger.info("[-] playwright-stealth imported successfully")
except ImportError:
logger.warning("playwright-stealth not found. Running without stealth mode.")
has_stealth = False
logger.info("[-] playwright_libraries_imported_successfully")
except ImportError as e:
logger.error(f"[!] playwright_import_failed: {e}")
return None
try:
with self.playwright_semaphore:
logger.info("[-] semaphore_acquired: Starting Browser Session")
with sync_playwright() as p:
args = [
"--disable-blink-features=AutomationControlled",
"--no-sandbox",
"--disable-setuid-sandbox",
"--disable-dev-shm-usage",
"--disable-accelerated-2d-canvas",
"--no-first-run",
"--no-zygote",
"--disable-gpu",
"--mute-audio",
]
logger.info(f"[-] launching_browser with args: {len(args)} flags set")
browser = p.chromium.launch(headless=True, args=args)
target_ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.0.0 Safari/537.36"
logger.info(f"[-] creating_context with User-Agent: {target_ua}")
context = browser.new_context(
viewport={'width': 1920, 'height': 1080},
user_agent=target_ua,
locale="en-US",
timezone_id="America/New_York",
)
page = context.new_page()
logger.info("[-] page_created")
# Apply Stealth
if has_stealth:
stealth_sync(page)
logger.info("[-] stealth_sync_applied: 'navigator.webdriver' masked")
else:
logger.info("[-] stealth_disabled: library not found")
# Block resources
page.route("**/*.{png,jpg,jpeg,gif,svg,woff,woff2,ttf,otf}", lambda route: route.abort())
logger.info("[-] resource_blocking_active: Images/Fonts blocked")
try:
logger.info(f"[-] navigating_to_url: {url}")
response = page.goto(url, timeout=90000, wait_until="domcontentloaded")
status = response.status if response else "Unknown"
logger.info(f"[-] navigation_complete. Status: {status}")
# Initial wait
page.wait_for_timeout(3000)
# --- CLOUDFLARE CHECK ---
for attempt in range(3):
title = page.title()
content_sample = page.content().lower()[:500] # Log only start
logger.info(f"[-] check_attempt_{attempt+1}: Title='{title}'")
is_blocked = False
block_reason = ""
if "just a moment" in title.lower():
is_blocked = True
block_reason = "Title: Just a moment"
elif "challenge" in title.lower():
is_blocked = True
block_reason = "Title: Challenge"
elif "security" in title.lower():
is_blocked = True
block_reason = "Title: Security"
elif "verify you are human" in content_sample:
is_blocked = True
block_reason = "Content: Verify Human"
if is_blocked:
self.blocked_reason = block_reason
logger.warning(f"[!] WAF_DETECTED: {block_reason}. Initiating countermeasures...")
# 1. Mouse Action
logger.info("[-] countermeasures: performing_mouse_movements")
page.mouse.move(100, 100)
page.wait_for_timeout(500)
page.mouse.move(200, 200)
# 2. Click Checkboxes
logger.info("[-] countermeasures: scanning_for_iframes")
frame_clicked = False
for i, frame in enumerate(page.frames):
try:
# Checkbox
checkbox = frame.locator("input[type='checkbox']").first
if checkbox.is_visible():
logger.info(f"[-] frame_{i}: checkbox_found. CLICKING...")
checkbox.click()
frame_clicked = True
page.wait_for_timeout(2000)
# Button
verify_btn = frame.get_by_role("button", name="Verify you are human")
if verify_btn.is_visible():
logger.info(f"[-] frame_{i}: verify_button_found. CLICKING...")
verify_btn.click()
frame_clicked = True
except Exception as e:
logger.debug(f"[-] frame_{i}_scan_error: {e}")
if not frame_clicked:
logger.info("[-] countermeasures: no_interactive_elements_found_in_frames")
logger.info("[-] countermeasures: waiting_10s_for_reload")
page.wait_for_timeout(10000)
else:
logger.info("[-] verification_passed: Page seems clean")
break
# ------------------------
final_title = page.title()
logger.info(f"[-] final_page_title: {final_title}")
# Content validation
final_content = page.content()
size_kb = len(final_content) / 1024
logger.info(f"[-] content_captured: {size_kb:.2f} KB")
if len(final_content) < 1000:
logger.warning("[!] content_warning: Page content unusually small (<1KB)")
page.close()
context.close()
browser.close()
logger.info("[-] browser_session_closed_gracefully")
return final_content
except Exception as nav:
logger.error(f"[!] navigation_error: {nav}")
return None
finally:
try:
browser.close()
except:
pass
except Exception as e:
logger.error(f"Playwright critical error: {e}")
return None
def crawl_domain(self, start_url, max_pages=100, progress_callback=None):
"""
Crawls domain with robust link discovery and normalization.
"""
# Reset blocked state
self.blocked_reason = None
# Normalize start_url (remove trailing slash) to ensure consistency
# This fixes issues where https://amazon.in/ vs https://amazon.in behave differently
start_url = start_url.rstrip('/')
parsed_start = urlparse(start_url)
start_domain = parsed_start.netloc
base_domain = start_domain.replace('www.', '') # simplistic base domain
queue = [start_url]
self.visited_urls.add(start_url)
site_data = {}
pages_crawled = 0
logger.info(f"Starting crawl for domain: {base_domain}")
# Use ThreadPoolExecutor for parallel crawling with high concurrency
# Adjusted to 20 for HuggingFace Spaces (2 vCPU usually) to avoid context switching overhead
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
# Map of future -> url
future_to_url = {}
# Submit initial task
future = executor.submit(self._worker_crawl_page, start_url)
future_to_url[future] = start_url
# Loop processing completed futures
while future_to_url and pages_crawled < max_pages:
done, not_done = concurrent.futures.wait(
future_to_url.keys(),
return_when=concurrent.futures.FIRST_COMPLETED
)
for future in done:
url = future_to_url.pop(future)
try:
result = future.result()
except Exception as exc:
logger.error(f"{url} generated an exception: {exc}")
result = None
if not result:
if self.blocked_reason and pages_crawled == 0:
if url == start_url:
logger.error("Crawl blocked on first page. Aborting.")
return site_data, len(self.visited_urls), self.blocked_reason
continue
# Unpack result
_, images, raw_links = result
if url == start_url and len(raw_links) < 5:
logger.warning(f"Start URL {url} returned only {len(raw_links)} links. Likely JS-heavy or blocked. Forcing Playwright retry...")
pw_content = self._fetch_playwright(url)
if pw_content:
images = self.extract_images(pw_content, url)
soup = BeautifulSoup(pw_content, 'html.parser')
raw_links = [link.get('href') for link in soup.find_all('a', href=True)]
logger.info(f"Playwright retry found {len(raw_links)} links.")
logger.info(f"Crawled [{pages_crawled + 1}]: {url}")
site_data[url] = images
pages_crawled += 1
# --- Progress Update ---
if progress_callback:
try:
# Calculate current total images
current_total_images = sum(len(imgs) for imgs in site_data.values())
progress_callback(pages_crawled, current_total_images, url)
except Exception as cb_err:
logger.error(f"Callback error: {cb_err}")
# -----------------------
# Process Links
links_stats = {"total": len(raw_links), "kept": 0, "skipped": 0}
for href in raw_links:
full_url = urljoin(url, href)
parsed_url = urlparse(full_url)
# Normalize: Remove fragment, strip trailing slash to avoid duplicates (/about vs /about/)
full_url = parsed_url._replace(fragment="").geturl().rstrip('/')
link_domain = parsed_url.netloc
# Internal Check: Match base domain (handles www/non-www and subdomains)
is_internal = (
link_domain == start_domain or
link_domain.endswith('.' + base_domain) or
link_domain == base_domain
)
if is_internal:
path = parsed_url.path.lower()
excluded_exts = ['.jpg', '.jpeg', '.png', '.gif', '.css', '.js', '.ico', '.svg', '.pdf', '.zip', '.xml']
if any(path.endswith(ext) for ext in excluded_exts):
links_stats["skipped"] += 1
continue
# Filter out non-content paths
exclude_keywords = ['/account', '/login', '/signin', '/signup', '/cart', '/checkout', '/wishlist', '/auth', 'javascript:', 'mailto:', 'tel:']
if any(k in full_url.lower() for k in exclude_keywords):
links_stats["skipped"] += 1
continue
if full_url not in self.visited_urls:
self.visited_urls.add(full_url)
links_stats["kept"] += 1
if pages_crawled + len(future_to_url) < max_pages:
next_future = executor.submit(self._worker_crawl_page, full_url)
future_to_url[next_future] = full_url
else:
links_stats["skipped"] += 1
else:
links_stats["skipped"] += 1
logger.info(f"Link Discovery for {url}: Found {links_stats['total']}, Added {links_stats['kept']} new unique internal links.")
if pages_crawled >= max_pages:
break
for f in future_to_url:
f.cancel()
# Final block check: if we scraped 0 pages or only 1 page with 0 images and blocked_reason is set
# Final block check: if we scraped 0 pages or only 1 page with 0 images and blocked_reason is set
if pages_crawled == 0 and not self.blocked_reason:
# If we have 0 pages, it means the start_url failed to fetch completely.
self.blocked_reason = "Failed to access domain (All tiers failed)"
site_data = None # Ensure it's treated as failure
return site_data, len(self.visited_urls), self.blocked_reason
def extract_images(self, html_content, base_url):
if not html_content:
return []
soup = BeautifulSoup(html_content, 'html.parser')
images = []
for img in soup.find_all('img'):
raw_src = img.get('src')
if not raw_src:
continue
full_url = urljoin(base_url, raw_src)
images.append({'src': full_url, 'alt': img.get('alt', '')})
return images
def _worker_crawl_page(self, url):
"""
Worker method to fetch and parse a single page.
Returns (url, images, raw_links) or None.
"""
html_content = self.fetch(url)
if not html_content:
return None
images = self.extract_images(html_content, url)
soup = BeautifulSoup(html_content, 'html.parser')
raw_links = [link.get('href') for link in soup.find_all('a', href=True)]
return url, images, raw_links