| from curl_cffi import requests |
| import time |
| import random |
| from bs4 import BeautifulSoup |
| from urllib.parse import urljoin, urlparse |
| import sys |
| import logging |
| import os |
| import concurrent.futures |
| import cloudscraper |
| from fake_useragent import UserAgent |
| import threading |
|
|
| |
| logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') |
| logger = logging.getLogger(__name__) |
|
|
| class Crawler: |
| def __init__(self, use_playwright=True): |
| |
| |
| self.session = requests.Session(impersonate="chrome120", verify=False) |
| self.ua = UserAgent() |
| |
| self.scraper = cloudscraper.create_scraper( |
| browser={'browser': 'chrome', 'platform': 'windows', 'desktop': True} |
| ) |
| |
| |
| self.playwright_semaphore = threading.Semaphore(3) |
|
|
| self.session.headers.update({ |
| "Referer": "https://www.google.com/", |
| "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7", |
| "Accept-Language": "en-US,en;q=0.9", |
| "Upgrade-Insecure-Requests": "1", |
| "Sec-Fetch-Site": "none", |
| "Sec-Fetch-Mode": "navigate", |
| "Sec-Fetch-User": "?1", |
| "Sec-Fetch-Dest": "document", |
| }) |
| self.visited_urls = set() |
| self.use_playwright_fallback = use_playwright |
| self.blocked_reason = None |
|
|
| def fetch(self, url): |
| """ |
| Tiered Fetching Strategy: |
| 1. curl_cffi (Fastest, good evasion) |
| 2. cloudscraper (Specialized for Cloudflare/WAF) |
| 3. Playwright (Heaviest, comprehensive) |
| """ |
| |
| content = self._fetch_http(url) |
| if content: |
| |
| if not self._is_blocked(content): |
| return content |
| else: |
| logger.warning(f"Tier 1 fetched blocked content for {url}. Escalating...") |
| |
| |
| logger.info(f"Tier 1 failed. Trying Cloudscraper info for {url}...") |
| try: |
| resp = self.scraper.get(url, timeout=10) |
| if 200 <= resp.status_code < 300: |
| if not self._is_blocked(resp.text): |
| return resp.text |
| else: |
| logger.warning(f"Cloudscraper also blocked for {url}.") |
| except Exception as e: |
| logger.warning(f"Cloudscraper failed: {e}") |
|
|
| |
| if self.use_playwright_fallback: |
| logger.info(f"Attempts failed. utilizing Playwright (Limit 3 concurrent) for {url}") |
| return self._fetch_playwright(url) |
| |
| return None |
|
|
| def _is_blocked(self, content): |
| """ |
| Detects if the content is likely a WAF block page, CAPTCHA, or 'Just a moment'. |
| """ |
| if not content or len(content) < 500: |
| return True |
| |
| lower_content = content.lower() |
| block_keywords = [ |
| "just a moment...", |
| "enable javascript", |
| "verify you are human", |
| "access denied", |
| "cloudflare", |
| "captcha", |
| "security check", |
| "turn on javascript", |
| "challenge.js", |
| "api-services-support@amazon.com", |
| "we just need to make sure you're not a robot", |
| "type the characters you see in this image" |
| ] |
| |
| if any(k in lower_content for k in block_keywords): |
| |
| |
| if len(content) < 5000: |
| return True |
| |
| return False |
|
|
| def _fetch_http(self, url): |
| """ |
| Fast HTTP fetch with aggressive timeouts and minimal retries. |
| """ |
| max_retries = 2 |
| for attempt in range(max_retries): |
| try: |
| |
| self.session.headers["User-Agent"] = self.ua.random |
| |
| |
| |
| response = self.session.get(url, timeout=5, allow_redirects=True) |
| |
| |
| if 200 <= response.status_code < 300: |
| return response.text |
| |
| if response.status_code == 403: |
| |
| return response.text |
| |
| if response.status_code in [429, 500, 502, 503]: |
| if attempt < max_retries - 1: |
| |
| time.sleep(0.5) |
| continue |
|
|
| return None |
|
|
| except Exception as e: |
| |
| |
| logger.warning(f"Fast fetch failed for {url}: {e}") |
| |
| return None |
|
|
| def _fetch_playwright(self, url): |
| """ |
| Fallback Strategy A: Use Playwright with 'playwright-stealth' library. |
| Includes ULTRA-VERBOSE Logging for debugging. |
| """ |
| logger.info(f"[-] initiating_playwright_fetch for: {url}") |
| |
| try: |
| from playwright.sync_api import sync_playwright |
| |
| try: |
| from playwright_stealth import stealth_sync |
| has_stealth = True |
| logger.info("[-] playwright-stealth imported successfully") |
| except ImportError: |
| logger.warning("playwright-stealth not found. Running without stealth mode.") |
| has_stealth = False |
| |
| logger.info("[-] playwright_libraries_imported_successfully") |
| except ImportError as e: |
| logger.error(f"[!] playwright_import_failed: {e}") |
| return None |
|
|
| try: |
| with self.playwright_semaphore: |
| logger.info("[-] semaphore_acquired: Starting Browser Session") |
| with sync_playwright() as p: |
| args = [ |
| "--disable-blink-features=AutomationControlled", |
| "--no-sandbox", |
| "--disable-setuid-sandbox", |
| "--disable-dev-shm-usage", |
| "--disable-accelerated-2d-canvas", |
| "--no-first-run", |
| "--no-zygote", |
| "--disable-gpu", |
| "--mute-audio", |
| ] |
| |
| logger.info(f"[-] launching_browser with args: {len(args)} flags set") |
| browser = p.chromium.launch(headless=True, args=args) |
| |
| target_ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.0.0 Safari/537.36" |
| logger.info(f"[-] creating_context with User-Agent: {target_ua}") |
| |
| context = browser.new_context( |
| viewport={'width': 1920, 'height': 1080}, |
| user_agent=target_ua, |
| locale="en-US", |
| timezone_id="America/New_York", |
| ) |
| |
| page = context.new_page() |
| logger.info("[-] page_created") |
| |
| |
| if has_stealth: |
| stealth_sync(page) |
| logger.info("[-] stealth_sync_applied: 'navigator.webdriver' masked") |
| else: |
| logger.info("[-] stealth_disabled: library not found") |
| |
| |
| page.route("**/*.{png,jpg,jpeg,gif,svg,woff,woff2,ttf,otf}", lambda route: route.abort()) |
| logger.info("[-] resource_blocking_active: Images/Fonts blocked") |
|
|
| try: |
| logger.info(f"[-] navigating_to_url: {url}") |
| response = page.goto(url, timeout=90000, wait_until="domcontentloaded") |
| status = response.status if response else "Unknown" |
| logger.info(f"[-] navigation_complete. Status: {status}") |
| |
| |
| page.wait_for_timeout(3000) |
| |
| |
| for attempt in range(3): |
| title = page.title() |
| content_sample = page.content().lower()[:500] |
| logger.info(f"[-] check_attempt_{attempt+1}: Title='{title}'") |
| |
| is_blocked = False |
| block_reason = "" |
| |
| if "just a moment" in title.lower(): |
| is_blocked = True |
| block_reason = "Title: Just a moment" |
| elif "challenge" in title.lower(): |
| is_blocked = True |
| block_reason = "Title: Challenge" |
| elif "security" in title.lower(): |
| is_blocked = True |
| block_reason = "Title: Security" |
| elif "verify you are human" in content_sample: |
| is_blocked = True |
| block_reason = "Content: Verify Human" |
| |
| if is_blocked: |
| self.blocked_reason = block_reason |
| logger.warning(f"[!] WAF_DETECTED: {block_reason}. Initiating countermeasures...") |
| |
| |
| logger.info("[-] countermeasures: performing_mouse_movements") |
| page.mouse.move(100, 100) |
| page.wait_for_timeout(500) |
| page.mouse.move(200, 200) |
| |
| |
| logger.info("[-] countermeasures: scanning_for_iframes") |
| frame_clicked = False |
| for i, frame in enumerate(page.frames): |
| try: |
| |
| checkbox = frame.locator("input[type='checkbox']").first |
| if checkbox.is_visible(): |
| logger.info(f"[-] frame_{i}: checkbox_found. CLICKING...") |
| checkbox.click() |
| frame_clicked = True |
| page.wait_for_timeout(2000) |
| |
| |
| verify_btn = frame.get_by_role("button", name="Verify you are human") |
| if verify_btn.is_visible(): |
| logger.info(f"[-] frame_{i}: verify_button_found. CLICKING...") |
| verify_btn.click() |
| frame_clicked = True |
| except Exception as e: |
| logger.debug(f"[-] frame_{i}_scan_error: {e}") |
| |
| if not frame_clicked: |
| logger.info("[-] countermeasures: no_interactive_elements_found_in_frames") |
|
|
| logger.info("[-] countermeasures: waiting_10s_for_reload") |
| page.wait_for_timeout(10000) |
| else: |
| logger.info("[-] verification_passed: Page seems clean") |
| break |
| |
|
|
| final_title = page.title() |
| logger.info(f"[-] final_page_title: {final_title}") |
| |
| |
| final_content = page.content() |
| size_kb = len(final_content) / 1024 |
| logger.info(f"[-] content_captured: {size_kb:.2f} KB") |
| |
| if len(final_content) < 1000: |
| logger.warning("[!] content_warning: Page content unusually small (<1KB)") |
| |
| page.close() |
| context.close() |
| browser.close() |
| logger.info("[-] browser_session_closed_gracefully") |
| return final_content |
| |
| except Exception as nav: |
| logger.error(f"[!] navigation_error: {nav}") |
| return None |
| finally: |
| try: |
| browser.close() |
| except: |
| pass |
|
|
|
|
|
|
|
|
| except Exception as e: |
| logger.error(f"Playwright critical error: {e}") |
| return None |
|
|
| def crawl_domain(self, start_url, max_pages=100, progress_callback=None): |
| """ |
| Crawls domain with robust link discovery and normalization. |
| """ |
| |
| self.blocked_reason = None |
| |
| |
| |
| start_url = start_url.rstrip('/') |
| |
| parsed_start = urlparse(start_url) |
| start_domain = parsed_start.netloc |
| base_domain = start_domain.replace('www.', '') |
| |
| queue = [start_url] |
| self.visited_urls.add(start_url) |
| |
| site_data = {} |
| pages_crawled = 0 |
|
|
| logger.info(f"Starting crawl for domain: {base_domain}") |
|
|
| |
| |
| with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor: |
| |
| future_to_url = {} |
| |
| |
| future = executor.submit(self._worker_crawl_page, start_url) |
| future_to_url[future] = start_url |
| |
| |
| while future_to_url and pages_crawled < max_pages: |
| done, not_done = concurrent.futures.wait( |
| future_to_url.keys(), |
| return_when=concurrent.futures.FIRST_COMPLETED |
| ) |
| |
| for future in done: |
| url = future_to_url.pop(future) |
| |
| try: |
| result = future.result() |
| except Exception as exc: |
| logger.error(f"{url} generated an exception: {exc}") |
| result = None |
| |
| if not result: |
| if self.blocked_reason and pages_crawled == 0: |
| if url == start_url: |
| logger.error("Crawl blocked on first page. Aborting.") |
| return site_data, len(self.visited_urls), self.blocked_reason |
| continue |
| |
| |
| _, images, raw_links = result |
| |
| if url == start_url and len(raw_links) < 5: |
| logger.warning(f"Start URL {url} returned only {len(raw_links)} links. Likely JS-heavy or blocked. Forcing Playwright retry...") |
| pw_content = self._fetch_playwright(url) |
| if pw_content: |
| images = self.extract_images(pw_content, url) |
| soup = BeautifulSoup(pw_content, 'html.parser') |
| raw_links = [link.get('href') for link in soup.find_all('a', href=True)] |
| logger.info(f"Playwright retry found {len(raw_links)} links.") |
|
|
| logger.info(f"Crawled [{pages_crawled + 1}]: {url}") |
| site_data[url] = images |
| pages_crawled += 1 |
| |
| |
| if progress_callback: |
| try: |
| |
| current_total_images = sum(len(imgs) for imgs in site_data.values()) |
| progress_callback(pages_crawled, current_total_images, url) |
| except Exception as cb_err: |
| logger.error(f"Callback error: {cb_err}") |
| |
| |
| |
| links_stats = {"total": len(raw_links), "kept": 0, "skipped": 0} |
| |
| for href in raw_links: |
| full_url = urljoin(url, href) |
| parsed_url = urlparse(full_url) |
| |
| full_url = parsed_url._replace(fragment="").geturl().rstrip('/') |
| |
| link_domain = parsed_url.netloc |
| |
| |
| is_internal = ( |
| link_domain == start_domain or |
| link_domain.endswith('.' + base_domain) or |
| link_domain == base_domain |
| ) |
| |
| if is_internal: |
| path = parsed_url.path.lower() |
| excluded_exts = ['.jpg', '.jpeg', '.png', '.gif', '.css', '.js', '.ico', '.svg', '.pdf', '.zip', '.xml'] |
| |
| if any(path.endswith(ext) for ext in excluded_exts): |
| links_stats["skipped"] += 1 |
| continue |
|
|
| |
| exclude_keywords = ['/account', '/login', '/signin', '/signup', '/cart', '/checkout', '/wishlist', '/auth', 'javascript:', 'mailto:', 'tel:'] |
| if any(k in full_url.lower() for k in exclude_keywords): |
| links_stats["skipped"] += 1 |
| continue |
| |
| if full_url not in self.visited_urls: |
| self.visited_urls.add(full_url) |
| links_stats["kept"] += 1 |
| |
| if pages_crawled + len(future_to_url) < max_pages: |
| next_future = executor.submit(self._worker_crawl_page, full_url) |
| future_to_url[next_future] = full_url |
| else: |
| links_stats["skipped"] += 1 |
| else: |
| links_stats["skipped"] += 1 |
| |
| logger.info(f"Link Discovery for {url}: Found {links_stats['total']}, Added {links_stats['kept']} new unique internal links.") |
|
|
| if pages_crawled >= max_pages: |
| break |
|
|
| for f in future_to_url: |
| f.cancel() |
| |
| |
| |
| if pages_crawled == 0 and not self.blocked_reason: |
| |
| self.blocked_reason = "Failed to access domain (All tiers failed)" |
| site_data = None |
|
|
| return site_data, len(self.visited_urls), self.blocked_reason |
|
|
| def extract_images(self, html_content, base_url): |
| if not html_content: |
| return [] |
| soup = BeautifulSoup(html_content, 'html.parser') |
| images = [] |
| for img in soup.find_all('img'): |
| raw_src = img.get('src') |
| if not raw_src: |
| continue |
| full_url = urljoin(base_url, raw_src) |
| images.append({'src': full_url, 'alt': img.get('alt', '')}) |
| return images |
|
|
| def _worker_crawl_page(self, url): |
| """ |
| Worker method to fetch and parse a single page. |
| Returns (url, images, raw_links) or None. |
| """ |
| html_content = self.fetch(url) |
| if not html_content: |
| return None |
| |
| images = self.extract_images(html_content, url) |
| |
| soup = BeautifulSoup(html_content, 'html.parser') |
| raw_links = [link.get('href') for link in soup.find_all('a', href=True)] |
| |
| return url, images, raw_links |
|
|