Spaces:
Sleeping
Sleeping
Update app/listing_scraper.py
Browse files- app/listing_scraper.py +33 -282
app/listing_scraper.py
CHANGED
|
@@ -2,7 +2,7 @@ import logging
|
|
| 2 |
import re
|
| 3 |
import time
|
| 4 |
from typing import Callable, Any, Dict, List, Optional
|
| 5 |
-
from urllib.parse import urljoin
|
| 6 |
|
| 7 |
from browser_session import BrowserSession
|
| 8 |
|
|
@@ -14,15 +14,12 @@ def _parse_price(text: str) -> Optional[float]:
|
|
| 14 |
if not text:
|
| 15 |
return None
|
| 16 |
try:
|
| 17 |
-
# Match "Lowest Price: $7.49" pattern
|
| 18 |
match = re.search(r"Lowest\s+Price\s*:\s*\$?\s*([\d,.]+)", text, re.IGNORECASE)
|
| 19 |
if match:
|
| 20 |
return float(match.group(1).replace(",", ""))
|
| 21 |
-
# Fallback: any dollar amount
|
| 22 |
match = re.search(r"\$\s*([\d,.]+)", text)
|
| 23 |
if match:
|
| 24 |
return float(match.group(1).replace(",", ""))
|
| 25 |
-
# Last resort: any number
|
| 26 |
match = re.search(r"([\d,.]+)", text)
|
| 27 |
if match:
|
| 28 |
return float(match.group(1).replace(",", ""))
|
|
@@ -64,14 +61,6 @@ class ListingScraper:
|
|
| 64 |
return self.session.page
|
| 65 |
|
| 66 |
def _get_products_on_page(self) -> List[Dict[str, Any]]:
|
| 67 |
-
"""
|
| 68 |
-
Extract products from the listing page with timeout protection.
|
| 69 |
-
- Find each div.row containing a product
|
| 70 |
-
- Extract title from div.caption h4 a
|
| 71 |
-
- Extract price from div.price strong (Lowest Price: $X.XX)
|
| 72 |
-
- Extract link from product href
|
| 73 |
-
- ONLY return products where price >= min_price
|
| 74 |
-
"""
|
| 75 |
page = self.page
|
| 76 |
products: List[Dict[str, Any]] = []
|
| 77 |
|
|
@@ -100,21 +89,17 @@ class ListingScraper:
|
|
| 100 |
price_elem = row.locator("div.price strong").first
|
| 101 |
price_text = (price_elem.inner_text(timeout=500) or "").strip()
|
| 102 |
price = _parse_price(price_text)
|
| 103 |
-
logger.debug(f"Row {idx}: Extracted price text: {price_text} -> ${price}")
|
| 104 |
except Exception as e:
|
| 105 |
logger.debug(f"Row {idx}: Could not extract price: {e}")
|
| 106 |
|
| 107 |
if price is None:
|
| 108 |
self.skipped_unknown_price += 1
|
| 109 |
-
logger.debug(f"Row {idx}: Skipping '{title}' - no price found")
|
| 110 |
continue
|
| 111 |
|
| 112 |
if price < self.min_price:
|
| 113 |
self.skipped_below_threshold += 1
|
| 114 |
-
logger.debug(f"Row {idx}: Skip '{title}' (${price:.2f} < ${self.min_price:.2f})")
|
| 115 |
continue
|
| 116 |
|
| 117 |
-
logger.debug(f"Row {idx}: ✓ QUALIFY '{title}' @ ${price:.2f}")
|
| 118 |
products.append({"url": link, "title": title, "price": price})
|
| 119 |
|
| 120 |
except Exception as e:
|
|
@@ -126,104 +111,46 @@ class ListingScraper:
|
|
| 126 |
|
| 127 |
except Exception as e:
|
| 128 |
logger.error(f"Failed to extract products: {e}")
|
| 129 |
-
try:
|
| 130 |
-
self.session.screenshot("product_extraction_error")
|
| 131 |
-
except Exception:
|
| 132 |
-
pass
|
| 133 |
return []
|
| 134 |
|
| 135 |
def _next_page_url(self, current_url: str) -> Optional[str]:
|
| 136 |
"""Find next page link in pagination."""
|
| 137 |
page = self.page
|
| 138 |
-
# Prefer the 'Next' link inside pagination containers to avoid jumping to numbered anchors
|
| 139 |
-
container_selectors = [
|
| 140 |
-
"nav ul.pager",
|
| 141 |
-
"ul.pager",
|
| 142 |
-
"ul.pagination",
|
| 143 |
-
"div.pagination",
|
| 144 |
-
"nav.pagination",
|
| 145 |
-
"div.pagenav",
|
| 146 |
-
"div.pages",
|
| 147 |
-
"div.pager",
|
| 148 |
-
"aside.pagination",
|
| 149 |
-
]
|
| 150 |
|
| 151 |
def _valid_href(href: str) -> bool:
|
| 152 |
-
if not href:
|
| 153 |
return False
|
| 154 |
href_l = href.lower()
|
| 155 |
-
# Avoid direct product pages
|
| 156 |
if "dvd_view_" in href_l or "/dvd_view_" in href_l:
|
| 157 |
return False
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
|
| 161 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 162 |
|
| 163 |
-
for
|
| 164 |
try:
|
| 165 |
-
|
| 166 |
-
if
|
| 167 |
-
|
| 168 |
-
|
| 169 |
-
|
| 170 |
-
href = a.get_attribute("href") or ""
|
| 171 |
-
rel = (a.get_attribute("rel") or "").lower()
|
| 172 |
-
title = (a.get_attribute("title") or "").lower()
|
| 173 |
-
aria = (a.get_attribute("aria-label") or "").lower()
|
| 174 |
-
text = (a.inner_text() or "").strip().lower()
|
| 175 |
-
|
| 176 |
-
is_next = False
|
| 177 |
-
if rel == "next":
|
| 178 |
-
is_next = True
|
| 179 |
-
if "next" in title or "next" in aria:
|
| 180 |
-
is_next = True
|
| 181 |
-
# text may contain 'next' or start with an arrow symbol
|
| 182 |
-
if text.startswith("next") or text in (">", "»", ">>") or "next" in text:
|
| 183 |
-
is_next = True
|
| 184 |
-
|
| 185 |
-
if is_next and _valid_href(href):
|
| 186 |
-
return urljoin(current_url, href)
|
| 187 |
-
except Exception:
|
| 188 |
-
continue
|
| 189 |
except Exception:
|
| 190 |
continue
|
| 191 |
|
| 192 |
-
# Specific fallback: look for Next link inside nav ul.pager
|
| 193 |
-
try:
|
| 194 |
-
el = page.locator("nav ul.pager a:has-text('Next')").first
|
| 195 |
-
if el and el.is_visible(timeout=800):
|
| 196 |
-
href = el.get_attribute("href") or ""
|
| 197 |
-
if _valid_href(href):
|
| 198 |
-
return urljoin(current_url, href)
|
| 199 |
-
except Exception:
|
| 200 |
-
pass
|
| 201 |
-
|
| 202 |
-
# Broad fallback: look for any anchor with rel=next or text 'Next' anywhere on page
|
| 203 |
-
try:
|
| 204 |
-
el = page.locator("a[rel='next']").first
|
| 205 |
-
if el and el.is_visible(timeout=800):
|
| 206 |
-
href = el.get_attribute("href") or ""
|
| 207 |
-
if _valid_href(href):
|
| 208 |
-
return urljoin(current_url, href)
|
| 209 |
-
except Exception:
|
| 210 |
-
pass
|
| 211 |
-
|
| 212 |
-
try:
|
| 213 |
-
el = page.locator("a:has-text('Next')").first
|
| 214 |
-
if el and el.is_visible(timeout=800):
|
| 215 |
-
href = el.get_attribute("href") or ""
|
| 216 |
-
if _valid_href(href):
|
| 217 |
-
return urljoin(current_url, href)
|
| 218 |
-
except Exception:
|
| 219 |
-
pass
|
| 220 |
-
|
| 221 |
return None
|
| 222 |
|
| 223 |
def _resolve_next_page_url(self, current_url: str) -> Optional[str]:
|
| 224 |
-
"""Resolve the next page URL using the page counter first, then link-based fallback."""
|
| 225 |
page = self.page
|
| 226 |
|
|
|
|
| 227 |
try:
|
| 228 |
results_el = page.locator("div.col-sm-3.results").first
|
| 229 |
txt = (results_el.inner_text(timeout=800) or "").strip()
|
|
@@ -231,237 +158,61 @@ class ListingScraper:
|
|
| 231 |
if m:
|
| 232 |
cur = int(m.group(1))
|
| 233 |
total = int(m.group(2))
|
| 234 |
-
logger.debug(f"Pagination indicator: page {cur} of {total}")
|
| 235 |
if cur < total:
|
| 236 |
-
from urllib.parse import urlparse, parse_qs, urlencode, urlunparse
|
| 237 |
-
|
| 238 |
parsed = urlparse(current_url)
|
| 239 |
qs = parse_qs(parsed.query)
|
| 240 |
qs["page"] = [str(cur + 1)]
|
| 241 |
new_query = urlencode(qs, doseq=True)
|
| 242 |
-
|
| 243 |
(parsed.scheme, parsed.netloc, parsed.path, parsed.params, new_query, parsed.fragment)
|
| 244 |
)
|
| 245 |
-
logger.debug(f"Sequential next page URL -> {next_url}")
|
| 246 |
-
return next_url
|
| 247 |
-
logger.debug("Reached last page according to indicator")
|
| 248 |
return None
|
| 249 |
except Exception:
|
| 250 |
pass
|
| 251 |
|
| 252 |
-
|
| 253 |
-
|
| 254 |
-
logger.debug(f"Link-based next page URL -> {next_url}")
|
| 255 |
-
return next_url
|
| 256 |
-
|
| 257 |
-
def _resolve_next_page_url_with_timeout(self, current_url: str, timeout_ms: int = 5000) -> Optional[str]:
|
| 258 |
-
"""Resolve next page URL without crossing thread boundaries."""
|
| 259 |
-
try:
|
| 260 |
-
return self._resolve_next_page_url(current_url)
|
| 261 |
-
except Exception as e:
|
| 262 |
-
logger.warning(f"Error resolving next page URL: {e}")
|
| 263 |
-
return None
|
| 264 |
-
|
| 265 |
-
def _load_page_with_retry(self, url: str, page_num: int, max_retries: int = 3) -> bool:
|
| 266 |
-
"""
|
| 267 |
-
Load a page with exponential backoff retry logic and timeout protection.
|
| 268 |
-
Returns True if successful, False if all retries exhausted or timeout.
|
| 269 |
-
"""
|
| 270 |
-
retry_delays = [2, 5, 10] # Exponential backoff: 2s, 5s, 10s
|
| 271 |
-
|
| 272 |
-
for attempt in range(1, max_retries + 1):
|
| 273 |
-
try:
|
| 274 |
-
logger.info(f"Page load attempt {attempt}/{max_retries} for page {page_num}: {url}")
|
| 275 |
-
if self.session.goto(url):
|
| 276 |
-
logger.info(f"✓ Successfully loaded page {page_num} on attempt {attempt}")
|
| 277 |
-
try:
|
| 278 |
-
self.session.cleanup_page()
|
| 279 |
-
import gc
|
| 280 |
-
gc.collect()
|
| 281 |
-
except Exception as e:
|
| 282 |
-
logger.debug(f"Cleanup warning (non-critical): {e}")
|
| 283 |
-
return True
|
| 284 |
-
logger.warning(f"✗ Page load failed for page {page_num}, attempt {attempt}")
|
| 285 |
-
except Exception as e:
|
| 286 |
-
logger.warning(f"✗ Exception during page load for page {page_num}, attempt {attempt}: {e}")
|
| 287 |
-
|
| 288 |
-
if attempt < max_retries:
|
| 289 |
-
delay = retry_delays[attempt - 1]
|
| 290 |
-
logger.info(f"Waiting {delay}s before retry...")
|
| 291 |
-
time.sleep(delay)
|
| 292 |
-
|
| 293 |
-
logger.error(f"Failed to load page {page_num} after {max_retries} retries: {url}")
|
| 294 |
-
return False
|
| 295 |
|
| 296 |
def iter_qualifying_products(self, studio_url: str):
|
| 297 |
-
"""Stream qualifying products page by page
|
| 298 |
current_url = studio_url
|
| 299 |
page_num = 1
|
| 300 |
visited_urls = set()
|
| 301 |
-
consecutive_failures = 0
|
| 302 |
-
max_consecutive_failures = 3 # Stop if 3 pages in a row fail
|
| 303 |
pages_without_products = 0
|
| 304 |
-
max_empty_pages = 5 # Stop if 5 consecutive pages have no products
|
| 305 |
-
|
| 306 |
-
self.scan_start_time = time.time()
|
| 307 |
-
self.last_product_found_time = self.scan_start_time
|
| 308 |
|
| 309 |
-
|
| 310 |
-
logger.info(f"Limits: max_pages={self.max_pages}, page_timeout={self.page_timeout}s, total_timeout={self.total_timeout}s")
|
| 311 |
|
| 312 |
while current_url and page_num <= self.max_pages:
|
| 313 |
-
# Check for user stop signal
|
| 314 |
if self.stop_event and self.stop_event.is_set():
|
| 315 |
-
logger.info("Listing scan stopped by user")
|
| 316 |
break
|
| 317 |
|
| 318 |
-
# Check if we've exceeded total scan time
|
| 319 |
-
elapsed_time = time.time() - self.scan_start_time
|
| 320 |
-
if elapsed_time > self.total_timeout:
|
| 321 |
-
logger.warning(f"Total scan timeout exceeded ({elapsed_time:.0f}s > {self.total_timeout}s). Stopping scan.")
|
| 322 |
-
break
|
| 323 |
-
|
| 324 |
-
# Check for pagination loop
|
| 325 |
if current_url in visited_urls:
|
| 326 |
logger.warning("Pagination loop detected. Stopping scan.")
|
| 327 |
break
|
| 328 |
|
| 329 |
-
# Check if too many pages with no results
|
| 330 |
-
if pages_without_products >= max_empty_pages:
|
| 331 |
-
logger.warning(f"Too many empty pages ({pages_without_products}/{max_empty_pages}). Stopping scan.")
|
| 332 |
-
break
|
| 333 |
-
|
| 334 |
-
# Check for no products found in extended time
|
| 335 |
-
if self.collected == 0 and elapsed_time > 300: # 5 minutes
|
| 336 |
-
logger.warning("No products found after 5 minutes. Stopping scan.")
|
| 337 |
-
break
|
| 338 |
-
|
| 339 |
visited_urls.add(current_url)
|
| 340 |
logger.info(f"Scanning listing page {page_num}: {current_url}")
|
| 341 |
|
| 342 |
-
|
| 343 |
-
|
| 344 |
-
|
| 345 |
-
logger.warning(f"Page load failed. Consecutive failures: {consecutive_failures}/{max_consecutive_failures}")
|
| 346 |
-
|
| 347 |
-
if consecutive_failures >= max_consecutive_failures:
|
| 348 |
-
logger.error(f"Too many consecutive page failures ({consecutive_failures}). Stopping scan.")
|
| 349 |
-
break
|
| 350 |
-
|
| 351 |
-
# Try to proceed to next page instead of breaking
|
| 352 |
-
try:
|
| 353 |
-
next_url = self._resolve_next_page_url_with_timeout(current_url)
|
| 354 |
-
if next_url:
|
| 355 |
-
logger.info(f"Skipping failed page {page_num}, attempting next page")
|
| 356 |
-
current_url = next_url
|
| 357 |
-
page_num += 1
|
| 358 |
-
time.sleep(2) # Extra delay after failure
|
| 359 |
-
continue
|
| 360 |
-
else:
|
| 361 |
-
break
|
| 362 |
-
except Exception as e:
|
| 363 |
-
logger.error(f"Could not resolve next page after failure: {e}")
|
| 364 |
-
break
|
| 365 |
-
|
| 366 |
-
# Reset failure counter on successful page load
|
| 367 |
-
consecutive_failures = 0
|
| 368 |
-
time.sleep(1)
|
| 369 |
-
self.pages_scanned += 1
|
| 370 |
-
|
| 371 |
-
if self.status_callback:
|
| 372 |
-
try:
|
| 373 |
-
self.status_callback({
|
| 374 |
-
"state": "Scanning listing",
|
| 375 |
-
"current_page": page_num,
|
| 376 |
-
"found_items": self.collected,
|
| 377 |
-
"checked_items": self.checked,
|
| 378 |
-
})
|
| 379 |
-
except Exception:
|
| 380 |
-
pass
|
| 381 |
|
| 382 |
products = self._get_products_on_page()
|
| 383 |
-
|
| 384 |
-
|
| 385 |
-
|
| 386 |
-
pages_without_products += 1
|
| 387 |
-
if pages_without_products <= 2: # Only screenshot the first couple empty pages
|
| 388 |
-
try:
|
| 389 |
-
self.session.screenshot(f"empty_page_{page_num}")
|
| 390 |
-
except Exception:
|
| 391 |
-
pass
|
| 392 |
-
logger.warning(f"No qualifying products found on page {page_num} (empty pages: {pages_without_products}/{max_empty_pages})")
|
| 393 |
-
else:
|
| 394 |
-
pages_without_products = 0
|
| 395 |
-
self.last_product_found_time = time.time()
|
| 396 |
-
|
| 397 |
-
next_url = self._resolve_next_page_url_with_timeout(current_url)
|
| 398 |
-
if next_url:
|
| 399 |
-
logger.debug(f"Next listing page resolved before yielding products -> {next_url}")
|
| 400 |
-
else:
|
| 401 |
-
logger.debug("No next page found—scan may end after this page")
|
| 402 |
|
| 403 |
for p in products:
|
| 404 |
-
if self.stop_event and self.stop_event.is_set():
|
| 405 |
-
logger.info("Listing scan stopped by user")
|
| 406 |
-
return
|
| 407 |
-
|
| 408 |
self.checked += 1
|
| 409 |
self.collected += 1
|
| 410 |
p["url"] = urljoin(current_url, p.get("url") or "")
|
| 411 |
p["page_num"] = page_num
|
| 412 |
p["listing_url"] = current_url
|
| 413 |
-
|
| 414 |
-
if self.status_callback:
|
| 415 |
-
try:
|
| 416 |
-
self.status_callback({
|
| 417 |
-
"state": "Scanning listing",
|
| 418 |
-
"current_page": page_num,
|
| 419 |
-
"found_items": self.collected,
|
| 420 |
-
"checked_items": self.checked,
|
| 421 |
-
})
|
| 422 |
-
except Exception:
|
| 423 |
-
pass
|
| 424 |
-
|
| 425 |
yield p
|
| 426 |
|
| 427 |
-
if self.max_items and self.collected >= self.max_items:
|
| 428 |
-
logger.info(f"Reached max_items={self.max_items}; stopping listing scan")
|
| 429 |
-
return
|
| 430 |
-
|
| 431 |
if not next_url:
|
| 432 |
-
logger.info("No next page
|
| 433 |
break
|
| 434 |
|
| 435 |
current_url = next_url
|
| 436 |
-
page_num += 1
|
| 437 |
-
|
| 438 |
-
# Extra safety: check elapsed time again before next iteration
|
| 439 |
-
elapsed_time = time.time() - self.scan_start_time
|
| 440 |
-
if elapsed_time > self.total_timeout:
|
| 441 |
-
logger.warning(f"Total scan timeout reached ({elapsed_time:.0f}s). Stopping after {page_num-1} pages.")
|
| 442 |
-
break
|
| 443 |
-
|
| 444 |
-
def scan_listing(self, studio_url: str) -> List[Dict[str, Any]]:
|
| 445 |
-
"""
|
| 446 |
-
Scan all pages of the studio listing and collect qualifying products.
|
| 447 |
-
Only returns products where price >= min_price.
|
| 448 |
-
"""
|
| 449 |
-
qualifying = list(self.iter_qualifying_products(studio_url))
|
| 450 |
-
|
| 451 |
-
if self.status_callback:
|
| 452 |
-
try:
|
| 453 |
-
self.status_callback({
|
| 454 |
-
"state": "Listing scan complete",
|
| 455 |
-
"current_page": self.pages_scanned,
|
| 456 |
-
"found_items": len(qualifying),
|
| 457 |
-
"checked_items": self.checked,
|
| 458 |
-
})
|
| 459 |
-
except Exception:
|
| 460 |
-
pass
|
| 461 |
-
|
| 462 |
-
logger.info(
|
| 463 |
-
f"Listing scan complete: pages={self.pages_scanned}, "
|
| 464 |
-
f"checked={self.checked}, below_min={self.skipped_below_threshold}, "
|
| 465 |
-
f"no_price={self.skipped_unknown_price}, qualifying={len(qualifying)}"
|
| 466 |
-
)
|
| 467 |
-
return qualifying
|
|
|
|
| 2 |
import re
|
| 3 |
import time
|
| 4 |
from typing import Callable, Any, Dict, List, Optional
|
| 5 |
+
from urllib.parse import urljoin, urlparse, parse_qs, urlencode, urlunparse
|
| 6 |
|
| 7 |
from browser_session import BrowserSession
|
| 8 |
|
|
|
|
| 14 |
if not text:
|
| 15 |
return None
|
| 16 |
try:
|
|
|
|
| 17 |
match = re.search(r"Lowest\s+Price\s*:\s*\$?\s*([\d,.]+)", text, re.IGNORECASE)
|
| 18 |
if match:
|
| 19 |
return float(match.group(1).replace(",", ""))
|
|
|
|
| 20 |
match = re.search(r"\$\s*([\d,.]+)", text)
|
| 21 |
if match:
|
| 22 |
return float(match.group(1).replace(",", ""))
|
|
|
|
| 23 |
match = re.search(r"([\d,.]+)", text)
|
| 24 |
if match:
|
| 25 |
return float(match.group(1).replace(",", ""))
|
|
|
|
| 61 |
return self.session.page
|
| 62 |
|
| 63 |
def _get_products_on_page(self) -> List[Dict[str, Any]]:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 64 |
page = self.page
|
| 65 |
products: List[Dict[str, Any]] = []
|
| 66 |
|
|
|
|
| 89 |
price_elem = row.locator("div.price strong").first
|
| 90 |
price_text = (price_elem.inner_text(timeout=500) or "").strip()
|
| 91 |
price = _parse_price(price_text)
|
|
|
|
| 92 |
except Exception as e:
|
| 93 |
logger.debug(f"Row {idx}: Could not extract price: {e}")
|
| 94 |
|
| 95 |
if price is None:
|
| 96 |
self.skipped_unknown_price += 1
|
|
|
|
| 97 |
continue
|
| 98 |
|
| 99 |
if price < self.min_price:
|
| 100 |
self.skipped_below_threshold += 1
|
|
|
|
| 101 |
continue
|
| 102 |
|
|
|
|
| 103 |
products.append({"url": link, "title": title, "price": price})
|
| 104 |
|
| 105 |
except Exception as e:
|
|
|
|
| 111 |
|
| 112 |
except Exception as e:
|
| 113 |
logger.error(f"Failed to extract products: {e}")
|
|
|
|
|
|
|
|
|
|
|
|
|
| 114 |
return []
|
| 115 |
|
| 116 |
def _next_page_url(self, current_url: str) -> Optional[str]:
|
| 117 |
"""Find next page link in pagination."""
|
| 118 |
page = self.page
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 119 |
|
| 120 |
def _valid_href(href: str) -> bool:
|
| 121 |
+
if not href or href.strip() in ("#", "javascript:void(0);", "javascript:;"):
|
| 122 |
return False
|
| 123 |
href_l = href.lower()
|
|
|
|
| 124 |
if "dvd_view_" in href_l or "/dvd_view_" in href_l:
|
| 125 |
return False
|
| 126 |
+
return True
|
| 127 |
+
|
| 128 |
+
# Check explicit 'Next' selectors
|
| 129 |
+
next_selectors = [
|
| 130 |
+
"nav ul.pager a:has-text('Next')",
|
| 131 |
+
"ul.pagination a:has-text('Next')",
|
| 132 |
+
"a[rel='next']",
|
| 133 |
+
"a:has-text('Next')",
|
| 134 |
+
"a:has-text('»')",
|
| 135 |
+
"a:has-text('>')",
|
| 136 |
+
]
|
| 137 |
|
| 138 |
+
for sel in next_selectors:
|
| 139 |
try:
|
| 140 |
+
el = page.locator(sel).first
|
| 141 |
+
if el and el.is_visible(timeout=500):
|
| 142 |
+
href = el.get_attribute("href") or ""
|
| 143 |
+
if _valid_href(href):
|
| 144 |
+
return urljoin(current_url, href)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 145 |
except Exception:
|
| 146 |
continue
|
| 147 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 148 |
return None
|
| 149 |
|
| 150 |
def _resolve_next_page_url(self, current_url: str) -> Optional[str]:
|
|
|
|
| 151 |
page = self.page
|
| 152 |
|
| 153 |
+
# Method 1: Check UI text indicator e.g., "Page 5 of 12"
|
| 154 |
try:
|
| 155 |
results_el = page.locator("div.col-sm-3.results").first
|
| 156 |
txt = (results_el.inner_text(timeout=800) or "").strip()
|
|
|
|
| 158 |
if m:
|
| 159 |
cur = int(m.group(1))
|
| 160 |
total = int(m.group(2))
|
|
|
|
| 161 |
if cur < total:
|
|
|
|
|
|
|
| 162 |
parsed = urlparse(current_url)
|
| 163 |
qs = parse_qs(parsed.query)
|
| 164 |
qs["page"] = [str(cur + 1)]
|
| 165 |
new_query = urlencode(qs, doseq=True)
|
| 166 |
+
return urlunparse(
|
| 167 |
(parsed.scheme, parsed.netloc, parsed.path, parsed.params, new_query, parsed.fragment)
|
| 168 |
)
|
|
|
|
|
|
|
|
|
|
| 169 |
return None
|
| 170 |
except Exception:
|
| 171 |
pass
|
| 172 |
|
| 173 |
+
# Method 2: Link selector fallback
|
| 174 |
+
return self._next_page_url(current_url)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 175 |
|
| 176 |
def iter_qualifying_products(self, studio_url: str):
|
| 177 |
+
"""Stream qualifying products page by page."""
|
| 178 |
current_url = studio_url
|
| 179 |
page_num = 1
|
| 180 |
visited_urls = set()
|
|
|
|
|
|
|
| 181 |
pages_without_products = 0
|
|
|
|
|
|
|
|
|
|
|
|
|
| 182 |
|
| 183 |
+
self.scan_start_time = time.time()
|
|
|
|
| 184 |
|
| 185 |
while current_url and page_num <= self.max_pages:
|
|
|
|
| 186 |
if self.stop_event and self.stop_event.is_set():
|
|
|
|
| 187 |
break
|
| 188 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 189 |
if current_url in visited_urls:
|
| 190 |
logger.warning("Pagination loop detected. Stopping scan.")
|
| 191 |
break
|
| 192 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 193 |
visited_urls.add(current_url)
|
| 194 |
logger.info(f"Scanning listing page {page_num}: {current_url}")
|
| 195 |
|
| 196 |
+
if not self.session.goto(current_url):
|
| 197 |
+
logger.error(f"Failed to load page {page_num}")
|
| 198 |
+
break
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 199 |
|
| 200 |
products = self._get_products_on_page()
|
| 201 |
+
|
| 202 |
+
# Resolve next URL BEFORE yielding items so we maintain pagination state
|
| 203 |
+
next_url = self._resolve_next_page_url(current_url)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 204 |
|
| 205 |
for p in products:
|
|
|
|
|
|
|
|
|
|
|
|
|
| 206 |
self.checked += 1
|
| 207 |
self.collected += 1
|
| 208 |
p["url"] = urljoin(current_url, p.get("url") or "")
|
| 209 |
p["page_num"] = page_num
|
| 210 |
p["listing_url"] = current_url
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 211 |
yield p
|
| 212 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 213 |
if not next_url:
|
| 214 |
+
logger.info("No next page found. Pagination complete.")
|
| 215 |
break
|
| 216 |
|
| 217 |
current_url = next_url
|
| 218 |
+
page_num += 1
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|