Scraper: don't flag not_listed from partial scrapes
A page-load timeout was logged and skipped, so a scrape could 'succeed' with only page 1 of results — and the not_listed flagging then marked every page-2 hotel absent, suppressing their last known rates. - ScraperResult now tracks pages_requested/pages_ok - scrape_date skips not_listed flagging when pages failed or when the scrape saw <60% of the date's 7-day coverage baseline; scraped rates are still saved, unseen hotels keep last known rate + scrape time Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
de8c4ec257
commit
218b2f45f6
3 changed files with 62 additions and 6 deletions
|
|
@ -59,6 +59,8 @@ class ScraperResult:
|
|||
rates: List[RateData] = field(default_factory=list)
|
||||
error_message: Optional[str] = None
|
||||
page_content_sample: Optional[str] = None # For debugging
|
||||
pages_requested: int = 0 # Result pages we set out to fetch
|
||||
pages_ok: int = 0 # Pages that loaded and parsed cleanly
|
||||
|
||||
|
||||
class ScraperBackend(ABC):
|
||||
|
|
|
|||
|
|
@ -292,6 +292,7 @@ class PlaywrightLocalBackend(ScraperBackend):
|
|||
all_hotels = []
|
||||
all_rates = []
|
||||
seen_hotel_ids = set()
|
||||
pages_ok = 0
|
||||
|
||||
context = None
|
||||
page = None
|
||||
|
|
@ -315,10 +316,12 @@ class PlaywrightLocalBackend(ScraperBackend):
|
|||
|
||||
logger.info(f"Scraping page {page_num + 1}: {url}")
|
||||
|
||||
page_loaded = True
|
||||
try:
|
||||
await page.goto(url, wait_until='networkidle', timeout=30000)
|
||||
except Exception as e:
|
||||
logger.warning(f"Page load timeout, continuing: {e}")
|
||||
page_loaded = False
|
||||
|
||||
# Check for blocking
|
||||
content = await page.content()
|
||||
|
|
@ -331,7 +334,9 @@ class PlaywrightLocalBackend(ScraperBackend):
|
|||
block_reason=reason,
|
||||
hotels=all_hotels,
|
||||
rates=all_rates,
|
||||
page_content_sample=content[:1000]
|
||||
page_content_sample=content[:1000],
|
||||
pages_requested=pages,
|
||||
pages_ok=pages_ok,
|
||||
)
|
||||
|
||||
# Human-like scrolling
|
||||
|
|
@ -340,6 +345,12 @@ class PlaywrightLocalBackend(ScraperBackend):
|
|||
# Extract data
|
||||
hotels, rates = await self._extract_search_results(page, check_in)
|
||||
|
||||
# A page counts as clean if it loaded fully and parsed.
|
||||
# An empty page 1 means the results never rendered; an empty
|
||||
# later page can legitimately be the end of the results.
|
||||
if page_loaded and (hotels or page_num > 0):
|
||||
pages_ok += 1
|
||||
|
||||
# Deduplicate by booking_com_id
|
||||
for hotel, rate in zip(hotels, rates):
|
||||
if hotel.booking_com_id and hotel.booking_com_id not in seen_hotel_ids:
|
||||
|
|
@ -353,7 +364,9 @@ class PlaywrightLocalBackend(ScraperBackend):
|
|||
success=True,
|
||||
blocked=False,
|
||||
hotels=all_hotels,
|
||||
rates=all_rates
|
||||
rates=all_rates,
|
||||
pages_requested=pages,
|
||||
pages_ok=pages_ok,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
|
|
@ -363,7 +376,9 @@ class PlaywrightLocalBackend(ScraperBackend):
|
|||
blocked=False,
|
||||
error_message=str(e),
|
||||
hotels=all_hotels,
|
||||
rates=all_rates
|
||||
rates=all_rates,
|
||||
pages_requested=pages,
|
||||
pages_ok=pages_ok,
|
||||
)
|
||||
finally:
|
||||
if page:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue