Hammedalmodel commited on
Commit
f4dfd2f
·
verified ·
1 Parent(s): 2cd7bd9

Update app/listing_scraper.py

Browse files
Files changed (1) hide show
  1. app/listing_scraper.py +33 -282
app/listing_scraper.py CHANGED
@@ -2,7 +2,7 @@ import logging
2
  import re
3
  import time
4
  from typing import Callable, Any, Dict, List, Optional
5
- from urllib.parse import urljoin
6
 
7
  from browser_session import BrowserSession
8
 
@@ -14,15 +14,12 @@ def _parse_price(text: str) -> Optional[float]:
14
  if not text:
15
  return None
16
  try:
17
- # Match "Lowest Price: $7.49" pattern
18
  match = re.search(r"Lowest\s+Price\s*:\s*\$?\s*([\d,.]+)", text, re.IGNORECASE)
19
  if match:
20
  return float(match.group(1).replace(",", ""))
21
- # Fallback: any dollar amount
22
  match = re.search(r"\$\s*([\d,.]+)", text)
23
  if match:
24
  return float(match.group(1).replace(",", ""))
25
- # Last resort: any number
26
  match = re.search(r"([\d,.]+)", text)
27
  if match:
28
  return float(match.group(1).replace(",", ""))
@@ -64,14 +61,6 @@ class ListingScraper:
64
  return self.session.page
65
 
66
  def _get_products_on_page(self) -> List[Dict[str, Any]]:
67
- """
68
- Extract products from the listing page with timeout protection.
69
- - Find each div.row containing a product
70
- - Extract title from div.caption h4 a
71
- - Extract price from div.price strong (Lowest Price: $X.XX)
72
- - Extract link from product href
73
- - ONLY return products where price >= min_price
74
- """
75
  page = self.page
76
  products: List[Dict[str, Any]] = []
77
 
@@ -100,21 +89,17 @@ class ListingScraper:
100
  price_elem = row.locator("div.price strong").first
101
  price_text = (price_elem.inner_text(timeout=500) or "").strip()
102
  price = _parse_price(price_text)
103
- logger.debug(f"Row {idx}: Extracted price text: {price_text} -> ${price}")
104
  except Exception as e:
105
  logger.debug(f"Row {idx}: Could not extract price: {e}")
106
 
107
  if price is None:
108
  self.skipped_unknown_price += 1
109
- logger.debug(f"Row {idx}: Skipping '{title}' - no price found")
110
  continue
111
 
112
  if price < self.min_price:
113
  self.skipped_below_threshold += 1
114
- logger.debug(f"Row {idx}: Skip '{title}' (${price:.2f} < ${self.min_price:.2f})")
115
  continue
116
 
117
- logger.debug(f"Row {idx}: ✓ QUALIFY '{title}' @ ${price:.2f}")
118
  products.append({"url": link, "title": title, "price": price})
119
 
120
  except Exception as e:
@@ -126,104 +111,46 @@ class ListingScraper:
126
 
127
  except Exception as e:
128
  logger.error(f"Failed to extract products: {e}")
129
- try:
130
- self.session.screenshot("product_extraction_error")
131
- except Exception:
132
- pass
133
  return []
134
 
135
  def _next_page_url(self, current_url: str) -> Optional[str]:
136
  """Find next page link in pagination."""
137
  page = self.page
138
- # Prefer the 'Next' link inside pagination containers to avoid jumping to numbered anchors
139
- container_selectors = [
140
- "nav ul.pager",
141
- "ul.pager",
142
- "ul.pagination",
143
- "div.pagination",
144
- "nav.pagination",
145
- "div.pagenav",
146
- "div.pages",
147
- "div.pager",
148
- "aside.pagination",
149
- ]
150
 
151
  def _valid_href(href: str) -> bool:
152
- if not href:
153
  return False
154
  href_l = href.lower()
155
- # Avoid direct product pages
156
  if "dvd_view_" in href_l or "/dvd_view_" in href_l:
157
  return False
158
- # Prefer listing/search pages or links that include pagination params
159
- if any(token in href_l for token in ("dvd_search.php", "search_studioid", "page=", "order_by=", "search=")):
160
- return True
161
- return False
 
 
 
 
 
 
 
162
 
163
- for container in container_selectors:
164
  try:
165
- cont = page.locator(container).first
166
- if cont and cont.is_visible(timeout=800):
167
- anchors = cont.locator("a").all()
168
- for a in anchors:
169
- try:
170
- href = a.get_attribute("href") or ""
171
- rel = (a.get_attribute("rel") or "").lower()
172
- title = (a.get_attribute("title") or "").lower()
173
- aria = (a.get_attribute("aria-label") or "").lower()
174
- text = (a.inner_text() or "").strip().lower()
175
-
176
- is_next = False
177
- if rel == "next":
178
- is_next = True
179
- if "next" in title or "next" in aria:
180
- is_next = True
181
- # text may contain 'next' or start with an arrow symbol
182
- if text.startswith("next") or text in (">", "»", ">>") or "next" in text:
183
- is_next = True
184
-
185
- if is_next and _valid_href(href):
186
- return urljoin(current_url, href)
187
- except Exception:
188
- continue
189
  except Exception:
190
  continue
191
 
192
- # Specific fallback: look for Next link inside nav ul.pager
193
- try:
194
- el = page.locator("nav ul.pager a:has-text('Next')").first
195
- if el and el.is_visible(timeout=800):
196
- href = el.get_attribute("href") or ""
197
- if _valid_href(href):
198
- return urljoin(current_url, href)
199
- except Exception:
200
- pass
201
-
202
- # Broad fallback: look for any anchor with rel=next or text 'Next' anywhere on page
203
- try:
204
- el = page.locator("a[rel='next']").first
205
- if el and el.is_visible(timeout=800):
206
- href = el.get_attribute("href") or ""
207
- if _valid_href(href):
208
- return urljoin(current_url, href)
209
- except Exception:
210
- pass
211
-
212
- try:
213
- el = page.locator("a:has-text('Next')").first
214
- if el and el.is_visible(timeout=800):
215
- href = el.get_attribute("href") or ""
216
- if _valid_href(href):
217
- return urljoin(current_url, href)
218
- except Exception:
219
- pass
220
-
221
  return None
222
 
223
  def _resolve_next_page_url(self, current_url: str) -> Optional[str]:
224
- """Resolve the next page URL using the page counter first, then link-based fallback."""
225
  page = self.page
226
 
 
227
  try:
228
  results_el = page.locator("div.col-sm-3.results").first
229
  txt = (results_el.inner_text(timeout=800) or "").strip()
@@ -231,237 +158,61 @@ class ListingScraper:
231
  if m:
232
  cur = int(m.group(1))
233
  total = int(m.group(2))
234
- logger.debug(f"Pagination indicator: page {cur} of {total}")
235
  if cur < total:
236
- from urllib.parse import urlparse, parse_qs, urlencode, urlunparse
237
-
238
  parsed = urlparse(current_url)
239
  qs = parse_qs(parsed.query)
240
  qs["page"] = [str(cur + 1)]
241
  new_query = urlencode(qs, doseq=True)
242
- next_url = urlunparse(
243
  (parsed.scheme, parsed.netloc, parsed.path, parsed.params, new_query, parsed.fragment)
244
  )
245
- logger.debug(f"Sequential next page URL -> {next_url}")
246
- return next_url
247
- logger.debug("Reached last page according to indicator")
248
  return None
249
  except Exception:
250
  pass
251
 
252
- next_url = self._next_page_url(current_url)
253
- if next_url:
254
- logger.debug(f"Link-based next page URL -> {next_url}")
255
- return next_url
256
-
257
- def _resolve_next_page_url_with_timeout(self, current_url: str, timeout_ms: int = 5000) -> Optional[str]:
258
- """Resolve next page URL without crossing thread boundaries."""
259
- try:
260
- return self._resolve_next_page_url(current_url)
261
- except Exception as e:
262
- logger.warning(f"Error resolving next page URL: {e}")
263
- return None
264
-
265
- def _load_page_with_retry(self, url: str, page_num: int, max_retries: int = 3) -> bool:
266
- """
267
- Load a page with exponential backoff retry logic and timeout protection.
268
- Returns True if successful, False if all retries exhausted or timeout.
269
- """
270
- retry_delays = [2, 5, 10] # Exponential backoff: 2s, 5s, 10s
271
-
272
- for attempt in range(1, max_retries + 1):
273
- try:
274
- logger.info(f"Page load attempt {attempt}/{max_retries} for page {page_num}: {url}")
275
- if self.session.goto(url):
276
- logger.info(f"✓ Successfully loaded page {page_num} on attempt {attempt}")
277
- try:
278
- self.session.cleanup_page()
279
- import gc
280
- gc.collect()
281
- except Exception as e:
282
- logger.debug(f"Cleanup warning (non-critical): {e}")
283
- return True
284
- logger.warning(f"✗ Page load failed for page {page_num}, attempt {attempt}")
285
- except Exception as e:
286
- logger.warning(f"✗ Exception during page load for page {page_num}, attempt {attempt}: {e}")
287
-
288
- if attempt < max_retries:
289
- delay = retry_delays[attempt - 1]
290
- logger.info(f"Waiting {delay}s before retry...")
291
- time.sleep(delay)
292
-
293
- logger.error(f"Failed to load page {page_num} after {max_retries} retries: {url}")
294
- return False
295
 
296
  def iter_qualifying_products(self, studio_url: str):
297
- """Stream qualifying products page by page with comprehensive timeout protection."""
298
  current_url = studio_url
299
  page_num = 1
300
  visited_urls = set()
301
- consecutive_failures = 0
302
- max_consecutive_failures = 3 # Stop if 3 pages in a row fail
303
  pages_without_products = 0
304
- max_empty_pages = 5 # Stop if 5 consecutive pages have no products
305
-
306
- self.scan_start_time = time.time()
307
- self.last_product_found_time = self.scan_start_time
308
 
309
- logger.info(f"Starting listing scan with min_price=${self.min_price:.2f}")
310
- logger.info(f"Limits: max_pages={self.max_pages}, page_timeout={self.page_timeout}s, total_timeout={self.total_timeout}s")
311
 
312
  while current_url and page_num <= self.max_pages:
313
- # Check for user stop signal
314
  if self.stop_event and self.stop_event.is_set():
315
- logger.info("Listing scan stopped by user")
316
  break
317
 
318
- # Check if we've exceeded total scan time
319
- elapsed_time = time.time() - self.scan_start_time
320
- if elapsed_time > self.total_timeout:
321
- logger.warning(f"Total scan timeout exceeded ({elapsed_time:.0f}s > {self.total_timeout}s). Stopping scan.")
322
- break
323
-
324
- # Check for pagination loop
325
  if current_url in visited_urls:
326
  logger.warning("Pagination loop detected. Stopping scan.")
327
  break
328
 
329
- # Check if too many pages with no results
330
- if pages_without_products >= max_empty_pages:
331
- logger.warning(f"Too many empty pages ({pages_without_products}/{max_empty_pages}). Stopping scan.")
332
- break
333
-
334
- # Check for no products found in extended time
335
- if self.collected == 0 and elapsed_time > 300: # 5 minutes
336
- logger.warning("No products found after 5 minutes. Stopping scan.")
337
- break
338
-
339
  visited_urls.add(current_url)
340
  logger.info(f"Scanning listing page {page_num}: {current_url}")
341
 
342
- # Attempt to load page with retry logic
343
- if not self._load_page_with_retry(current_url, page_num, max_retries=3):
344
- consecutive_failures += 1
345
- logger.warning(f"Page load failed. Consecutive failures: {consecutive_failures}/{max_consecutive_failures}")
346
-
347
- if consecutive_failures >= max_consecutive_failures:
348
- logger.error(f"Too many consecutive page failures ({consecutive_failures}). Stopping scan.")
349
- break
350
-
351
- # Try to proceed to next page instead of breaking
352
- try:
353
- next_url = self._resolve_next_page_url_with_timeout(current_url)
354
- if next_url:
355
- logger.info(f"Skipping failed page {page_num}, attempting next page")
356
- current_url = next_url
357
- page_num += 1
358
- time.sleep(2) # Extra delay after failure
359
- continue
360
- else:
361
- break
362
- except Exception as e:
363
- logger.error(f"Could not resolve next page after failure: {e}")
364
- break
365
-
366
- # Reset failure counter on successful page load
367
- consecutive_failures = 0
368
- time.sleep(1)
369
- self.pages_scanned += 1
370
-
371
- if self.status_callback:
372
- try:
373
- self.status_callback({
374
- "state": "Scanning listing",
375
- "current_page": page_num,
376
- "found_items": self.collected,
377
- "checked_items": self.checked,
378
- })
379
- except Exception:
380
- pass
381
 
382
  products = self._get_products_on_page()
383
- logger.info(f" Found {len(products)} qualifying product(s) on page {page_num}")
384
-
385
- if not products:
386
- pages_without_products += 1
387
- if pages_without_products <= 2: # Only screenshot the first couple empty pages
388
- try:
389
- self.session.screenshot(f"empty_page_{page_num}")
390
- except Exception:
391
- pass
392
- logger.warning(f"No qualifying products found on page {page_num} (empty pages: {pages_without_products}/{max_empty_pages})")
393
- else:
394
- pages_without_products = 0
395
- self.last_product_found_time = time.time()
396
-
397
- next_url = self._resolve_next_page_url_with_timeout(current_url)
398
- if next_url:
399
- logger.debug(f"Next listing page resolved before yielding products -> {next_url}")
400
- else:
401
- logger.debug("No next page found—scan may end after this page")
402
 
403
  for p in products:
404
- if self.stop_event and self.stop_event.is_set():
405
- logger.info("Listing scan stopped by user")
406
- return
407
-
408
  self.checked += 1
409
  self.collected += 1
410
  p["url"] = urljoin(current_url, p.get("url") or "")
411
  p["page_num"] = page_num
412
  p["listing_url"] = current_url
413
-
414
- if self.status_callback:
415
- try:
416
- self.status_callback({
417
- "state": "Scanning listing",
418
- "current_page": page_num,
419
- "found_items": self.collected,
420
- "checked_items": self.checked,
421
- })
422
- except Exception:
423
- pass
424
-
425
  yield p
426
 
427
- if self.max_items and self.collected >= self.max_items:
428
- logger.info(f"Reached max_items={self.max_items}; stopping listing scan")
429
- return
430
-
431
  if not next_url:
432
- logger.info("No next page link found. Pagination complete.")
433
  break
434
 
435
  current_url = next_url
436
- page_num += 1
437
-
438
- # Extra safety: check elapsed time again before next iteration
439
- elapsed_time = time.time() - self.scan_start_time
440
- if elapsed_time > self.total_timeout:
441
- logger.warning(f"Total scan timeout reached ({elapsed_time:.0f}s). Stopping after {page_num-1} pages.")
442
- break
443
-
444
- def scan_listing(self, studio_url: str) -> List[Dict[str, Any]]:
445
- """
446
- Scan all pages of the studio listing and collect qualifying products.
447
- Only returns products where price >= min_price.
448
- """
449
- qualifying = list(self.iter_qualifying_products(studio_url))
450
-
451
- if self.status_callback:
452
- try:
453
- self.status_callback({
454
- "state": "Listing scan complete",
455
- "current_page": self.pages_scanned,
456
- "found_items": len(qualifying),
457
- "checked_items": self.checked,
458
- })
459
- except Exception:
460
- pass
461
-
462
- logger.info(
463
- f"Listing scan complete: pages={self.pages_scanned}, "
464
- f"checked={self.checked}, below_min={self.skipped_below_threshold}, "
465
- f"no_price={self.skipped_unknown_price}, qualifying={len(qualifying)}"
466
- )
467
- return qualifying
 
2
  import re
3
  import time
4
  from typing import Callable, Any, Dict, List, Optional
5
+ from urllib.parse import urljoin, urlparse, parse_qs, urlencode, urlunparse
6
 
7
  from browser_session import BrowserSession
8
 
 
14
  if not text:
15
  return None
16
  try:
 
17
  match = re.search(r"Lowest\s+Price\s*:\s*\$?\s*([\d,.]+)", text, re.IGNORECASE)
18
  if match:
19
  return float(match.group(1).replace(",", ""))
 
20
  match = re.search(r"\$\s*([\d,.]+)", text)
21
  if match:
22
  return float(match.group(1).replace(",", ""))
 
23
  match = re.search(r"([\d,.]+)", text)
24
  if match:
25
  return float(match.group(1).replace(",", ""))
 
61
  return self.session.page
62
 
63
  def _get_products_on_page(self) -> List[Dict[str, Any]]:
 
 
 
 
 
 
 
 
64
  page = self.page
65
  products: List[Dict[str, Any]] = []
66
 
 
89
  price_elem = row.locator("div.price strong").first
90
  price_text = (price_elem.inner_text(timeout=500) or "").strip()
91
  price = _parse_price(price_text)
 
92
  except Exception as e:
93
  logger.debug(f"Row {idx}: Could not extract price: {e}")
94
 
95
  if price is None:
96
  self.skipped_unknown_price += 1
 
97
  continue
98
 
99
  if price < self.min_price:
100
  self.skipped_below_threshold += 1
 
101
  continue
102
 
 
103
  products.append({"url": link, "title": title, "price": price})
104
 
105
  except Exception as e:
 
111
 
112
  except Exception as e:
113
  logger.error(f"Failed to extract products: {e}")
 
 
 
 
114
  return []
115
 
116
  def _next_page_url(self, current_url: str) -> Optional[str]:
117
  """Find next page link in pagination."""
118
  page = self.page
 
 
 
 
 
 
 
 
 
 
 
 
119
 
120
  def _valid_href(href: str) -> bool:
121
+ if not href or href.strip() in ("#", "javascript:void(0);", "javascript:;"):
122
  return False
123
  href_l = href.lower()
 
124
  if "dvd_view_" in href_l or "/dvd_view_" in href_l:
125
  return False
126
+ return True
127
+
128
+ # Check explicit 'Next' selectors
129
+ next_selectors = [
130
+ "nav ul.pager a:has-text('Next')",
131
+ "ul.pagination a:has-text('Next')",
132
+ "a[rel='next']",
133
+ "a:has-text('Next')",
134
+ "a:has-text('»')",
135
+ "a:has-text('>')",
136
+ ]
137
 
138
+ for sel in next_selectors:
139
  try:
140
+ el = page.locator(sel).first
141
+ if el and el.is_visible(timeout=500):
142
+ href = el.get_attribute("href") or ""
143
+ if _valid_href(href):
144
+ return urljoin(current_url, href)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
145
  except Exception:
146
  continue
147
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
148
  return None
149
 
150
  def _resolve_next_page_url(self, current_url: str) -> Optional[str]:
 
151
  page = self.page
152
 
153
+ # Method 1: Check UI text indicator e.g., "Page 5 of 12"
154
  try:
155
  results_el = page.locator("div.col-sm-3.results").first
156
  txt = (results_el.inner_text(timeout=800) or "").strip()
 
158
  if m:
159
  cur = int(m.group(1))
160
  total = int(m.group(2))
 
161
  if cur < total:
 
 
162
  parsed = urlparse(current_url)
163
  qs = parse_qs(parsed.query)
164
  qs["page"] = [str(cur + 1)]
165
  new_query = urlencode(qs, doseq=True)
166
+ return urlunparse(
167
  (parsed.scheme, parsed.netloc, parsed.path, parsed.params, new_query, parsed.fragment)
168
  )
 
 
 
169
  return None
170
  except Exception:
171
  pass
172
 
173
+ # Method 2: Link selector fallback
174
+ return self._next_page_url(current_url)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
175
 
176
  def iter_qualifying_products(self, studio_url: str):
177
+ """Stream qualifying products page by page."""
178
  current_url = studio_url
179
  page_num = 1
180
  visited_urls = set()
 
 
181
  pages_without_products = 0
 
 
 
 
182
 
183
+ self.scan_start_time = time.time()
 
184
 
185
  while current_url and page_num <= self.max_pages:
 
186
  if self.stop_event and self.stop_event.is_set():
 
187
  break
188
 
 
 
 
 
 
 
 
189
  if current_url in visited_urls:
190
  logger.warning("Pagination loop detected. Stopping scan.")
191
  break
192
 
 
 
 
 
 
 
 
 
 
 
193
  visited_urls.add(current_url)
194
  logger.info(f"Scanning listing page {page_num}: {current_url}")
195
 
196
+ if not self.session.goto(current_url):
197
+ logger.error(f"Failed to load page {page_num}")
198
+ break
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
199
 
200
  products = self._get_products_on_page()
201
+
202
+ # Resolve next URL BEFORE yielding items so we maintain pagination state
203
+ next_url = self._resolve_next_page_url(current_url)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
204
 
205
  for p in products:
 
 
 
 
206
  self.checked += 1
207
  self.collected += 1
208
  p["url"] = urljoin(current_url, p.get("url") or "")
209
  p["page_num"] = page_num
210
  p["listing_url"] = current_url
 
 
 
 
 
 
 
 
 
 
 
 
211
  yield p
212
 
 
 
 
 
213
  if not next_url:
214
+ logger.info("No next page found. Pagination complete.")
215
  break
216
 
217
  current_url = next_url
218
+ page_num += 1