rates/backend/services/scraper_backends/playwright_local.py
jtricerolph 349c795708 Scraper stealth: headful+xvfb, playwright-stealth, coherent UA, warm-up, click-paginate
Tier 1+2 anti-detection to stop Booking.com throttling page 2:
- Run Chromium headful under xvfb (Dockerfile) — headless leaks SwiftShader
  WebGL, empty plugins, missing chrome.runtime
- playwright-stealth patches navigator.webdriver/plugins/WebGL vendor
- Single coherent Chrome-121 identity: UA + matching sec-ch-ua client hints +
  platform (dropped the Firefox/Safari UA strings — a mismatched UA on a
  Chromium engine is a stronger tell than no rotation)
- Homepage warm-up so search requests carry real session cookies + cookie
  consent dismiss
- Paginate by clicking next (offset= deep-link was the page-2 tell), offset
  URL kept as fallback
- Viewport jitter, mouse movement, longer scroll, selector retry on lazy load

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-05 16:26:01 +00:00

523 lines
20 KiB
Python

"""
Local Playwright backend for booking.com scraping.
Uses Playwright with Chromium to scrape search results.
No proxy - direct connection. Suitable for low-volume scraping.
"""
import asyncio
import logging
import random
import re
from datetime import date
from decimal import Decimal, InvalidOperation
from typing import List, Optional
from urllib.parse import urlencode
from playwright.async_api import async_playwright, Browser, BrowserContext, Page
try:
from playwright_stealth import stealth_async
except ImportError: # pragma: no cover - stealth is optional at runtime
stealth_async = None
from .base import (
ScraperBackend,
ScraperResult,
HotelData,
RateData,
AvailabilityStatus
)
logger = logging.getLogger(__name__)
class PlaywrightLocalBackend(ScraperBackend):
"""
Local Playwright backend using Chromium.
Stealth measures:
- Runs headful (via xvfb in the container) — headless leaks SwiftShader
WebGL, empty plugins, missing chrome.runtime
- playwright-stealth patches navigator.webdriver, plugins, WebGL vendor
- ONE consistent modern-Chrome identity: UA + matching sec-ch-ua client
hints + platform (a mismatched UA is worse than none)
- Warms up via the homepage so search requests carry real session cookies
- Paginates by clicking "next" rather than deep-linking ?offset=25
- Random delays, viewport jitter, mouse movement, human-like scroll
"""
# A single coherent identity. The UA, the sec-ch-ua hints and the
# platform must all agree or the mismatch itself is a bot signal.
CHROME_VERSION = "121"
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/121.0.0.0 Safari/537.36"
)
CLIENT_HINT_HEADERS = {
"sec-ch-ua": '"Not A(Brand";v="99", "Google Chrome";v="121", "Chromium";v="121"',
"sec-ch-ua-mobile": "?0",
"sec-ch-ua-platform": '"Windows"',
"accept-language": "en-GB,en;q=0.9",
}
HOMEPAGE = "https://www.booking.com/index.en-gb.html"
def __init__(self, proxy_config: dict = None):
"""
Initialize the backend.
Args:
proxy_config: Optional proxy configuration (for future use)
"""
self.proxy_config = proxy_config
self._playwright = None
self._browser: Optional[Browser] = None
async def _ensure_browser(self) -> Browser:
"""Ensure browser is running, start if needed."""
if self._browser is None or not self._browser.is_connected():
self._playwright = await async_playwright().start()
self._browser = await self._playwright.chromium.launch(
# Headful under xvfb — far cleaner fingerprint than headless.
headless=False,
args=[
'--disable-blink-features=AutomationControlled',
'--no-sandbox',
'--disable-dev-shm-usage',
'--disable-features=IsolateOrigins,site-per-process',
'--start-maximized',
]
)
return self._browser
async def _create_context(self) -> BrowserContext:
"""Create a browser context with a single coherent Chrome identity."""
browser = await self._ensure_browser()
# Small viewport jitter so every session isn't pixel-identical
width = 1920 + random.randint(-40, 40)
height = 1080 + random.randint(-30, 30)
context = await browser.new_context(
user_agent=self.USER_AGENT,
viewport={'width': width, 'height': height},
locale='en-GB',
timezone_id='Europe/London',
extra_http_headers=self.CLIENT_HINT_HEADERS,
)
return context
async def _prepare_page(self, page: Page):
"""Apply stealth patches to a freshly created page."""
if stealth_async is not None:
try:
await stealth_async(page)
except Exception as e:
logger.warning(f"stealth_async failed, continuing without: {e}")
async def _warm_up(self, page: Page):
"""Visit the homepage first so search requests carry session cookies."""
try:
await page.goto(self.HOMEPAGE, wait_until='domcontentloaded', timeout=30000)
await asyncio.sleep(random.uniform(1.5, 3.0))
# Dismiss the cookie-consent dialog if present
for sel in ['#onetrust-accept-btn-handler',
'[aria-label="Accept"]',
'[data-testid="cookie-banner-accept"]']:
try:
btn = await page.query_selector(sel)
if btn:
await btn.click()
await asyncio.sleep(random.uniform(0.4, 0.9))
break
except Exception:
continue
except Exception as e:
logger.warning(f"Homepage warm-up failed, continuing: {e}")
def _build_search_url(
self,
location: str,
check_in: date,
check_out: date,
adults: int,
offset: int = 0
) -> str:
"""Build booking.com search URL with parameters."""
params = {
'ss': location,
'checkin': check_in.isoformat(),
'checkout': check_out.isoformat(),
'group_adults': adults,
'no_rooms': 1,
'group_children': 0,
}
if offset > 0:
params['offset'] = offset
return f"https://www.booking.com/searchresults.en-gb.html?{urlencode(params)}"
def _parse_price(self, price_text: str) -> Optional[Decimal]:
"""Parse price from text like '£150' or 'GBP 150'."""
if not price_text:
return None
# Remove currency symbols and extract number
cleaned = re.sub(r'[£$€,\s]', '', price_text)
# Find first number (including decimals)
match = re.search(r'[\d,]+(?:\.\d{2})?', cleaned)
if match:
try:
return Decimal(match.group().replace(',', ''))
except InvalidOperation:
return None
return None
def _extract_hotel_id(self, url: str) -> Optional[str]:
"""Extract hotel ID from booking.com URL."""
if not url:
return None
# URL format: /hotel/gb/hotel-name.en-gb.html or ?dest_id=123
# Try to extract from URL path
match = re.search(r'/hotel/[a-z]{2}/([^/]+)\.', url)
if match:
return match.group(1)
# Try dest_id parameter
match = re.search(r'dest_id=(-?\d+)', url)
if match:
return match.group(1)
return None
async def _human_like_scroll(self, page: Page):
"""Simulate human-like scrolling behavior."""
# Scroll down in increments
for _ in range(random.randint(3, 6)):
await page.mouse.wheel(0, random.randint(300, 600))
await asyncio.sleep(random.uniform(0.3, 0.8))
async def _human_like_mouse(self, page: Page):
"""A few random mouse moves — real users generate pointer events."""
try:
for _ in range(random.randint(2, 4)):
await page.mouse.move(random.randint(100, 1400), random.randint(150, 800))
await asyncio.sleep(random.uniform(0.1, 0.4))
except Exception:
pass
async def _go_to_next_page(self, page: Page) -> bool:
"""Click the pagination 'next' control. Returns True if navigation happened."""
selectors = [
'[data-testid="pagination-next-btn"]',
'button[aria-label="Next page"]',
'a[aria-label="Next page"]',
]
for sel in selectors:
try:
btn = await page.query_selector(sel)
if btn and await btn.is_enabled():
await btn.scroll_into_view_if_needed()
await asyncio.sleep(random.uniform(0.3, 0.7))
await btn.click()
# Results re-render in place; wait for network to settle
try:
await page.wait_for_load_state('networkidle', timeout=15000)
except Exception:
await asyncio.sleep(2)
return True
except Exception as e:
logger.debug(f"Next-page selector {sel} failed: {e}")
continue
return False
async def _extract_search_results(self, page: Page, rate_date: date) -> tuple[List[HotelData], List[RateData]]:
"""Extract hotel and rate data from search results page."""
hotels = []
rates = []
# Wait for property cards - booking.com uses data-testid. Retry once
# with a scroll nudge before giving up: cards can lazy-load late.
try:
await page.wait_for_selector('[data-testid="property-card"]', timeout=15000)
except Exception as e:
logger.warning(f"No property cards on first wait, retrying: {e}")
try:
await self._human_like_scroll(page)
await page.wait_for_selector('[data-testid="property-card"]', timeout=15000)
except Exception as e2:
logger.warning(f"No property cards found after retry: {e2}")
return hotels, rates
# Get all property cards
cards = await page.query_selector_all('[data-testid="property-card"]')
logger.info(f"Found {len(cards)} property cards")
for card in cards:
try:
hotel = HotelData(booking_com_id='', name='')
rate = RateData(rate_date=rate_date)
# Hotel name
name_el = await card.query_selector('[data-testid="title"]')
if name_el:
hotel.name = (await name_el.inner_text()).strip()
if not hotel.name:
continue # Skip if no name found
# Hotel URL and ID
link_el = await card.query_selector('[data-testid="title-link"]')
if link_el:
hotel.booking_com_url = await link_el.get_attribute('href')
hotel.booking_com_id = self._extract_hotel_id(hotel.booking_com_url) or ''
rate.booking_com_id = hotel.booking_com_id
# Star rating - look for star icons or rating text
stars_el = await card.query_selector('[data-testid="rating-stars"]')
if stars_el:
stars_text = await stars_el.get_attribute('aria-label') or ''
match = re.search(r'(\d+)', stars_text)
if match:
hotel.star_rating = Decimal(match.group(1))
# Review score
score_el = await card.query_selector('[data-testid="review-score"]')
if score_el:
score_text = await score_el.inner_text()
match = re.search(r'([\d.]+)', score_text)
if match:
try:
hotel.review_score = Decimal(match.group(1))
except InvalidOperation:
pass
# Check for no availability message FIRST
no_avail_el = await card.query_selector('[data-testid="availability-message"]')
if no_avail_el:
avail_text = (await no_avail_el.inner_text()).lower()
if 'no availability' in avail_text or 'sold out' in avail_text:
rate.availability_status = AvailabilityStatus.SOLD_OUT
hotels.append(hotel)
rates.append(rate)
continue
# Price
price_el = await card.query_selector('[data-testid="price-and-discounted-price"]')
if not price_el:
# Try alternative selector
price_el = await card.query_selector('[data-testid="price"]')
if price_el:
price_text = await price_el.inner_text()
rate.rate_gross = self._parse_price(price_text)
if rate.rate_gross:
rate.availability_status = AvailabilityStatus.AVAILABLE
# Room type
room_el = await card.query_selector('[data-testid="recommended-units"]')
if room_el:
rate.room_type = (await room_el.inner_text()).strip()
# Rate option badges - try multiple selectors
# Breakfast included
breakfast_el = await card.query_selector('[data-testid="breakfast-included"]')
if not breakfast_el:
# Check text content for breakfast mentions
card_text = (await card.inner_text()).lower()
rate.breakfast_included = 'breakfast included' in card_text
else:
rate.breakfast_included = True
# Free cancellation
cancel_el = await card.query_selector('[data-testid="cancellation-policy"]')
if cancel_el:
cancel_text = (await cancel_el.inner_text()).lower()
rate.free_cancellation = 'free cancellation' in cancel_text
else:
card_text = (await card.inner_text()).lower()
rate.free_cancellation = 'free cancellation' in card_text
# No prepayment
prepay_el = await card.query_selector('[data-testid="no-prepayment"]')
if prepay_el:
rate.no_prepayment = True
else:
card_text = (await card.inner_text()).lower()
rate.no_prepayment = 'no prepayment' in card_text
# Rooms left / scarcity indicator
scarcity_el = await card.query_selector('[data-testid="availability-rate"]')
if scarcity_el:
scarcity_text = await scarcity_el.inner_text()
match = re.search(r'(\d+)\s*room', scarcity_text.lower())
if match:
rate.rooms_left = int(match.group(1))
hotels.append(hotel)
rates.append(rate)
except Exception as e:
logger.warning(f"Error extracting hotel data: {e}")
continue
return hotels, rates
async def scrape_location_search(
self,
location: str,
check_in: date,
check_out: date,
adults: int = 2,
pages: int = 2
) -> ScraperResult:
"""
Scrape booking.com location search results.
Args:
location: Location name
check_in: Check-in date
check_out: Check-out date (check_in + 1 for single night rate)
adults: Number of adults
pages: Number of result pages to scrape
Returns:
ScraperResult with hotels and rates found
"""
all_hotels = []
all_rates = []
seen_hotel_ids = set()
pages_ok = 0
context = None
page = None
try:
context = await self._create_context()
page = await context.new_page()
await self._prepare_page(page)
# Land on the homepage first so the search carries session cookies
await self._warm_up(page)
for page_num in range(pages):
# Random delay between pages (3-7 seconds)
if page_num > 0:
delay = random.uniform(3, 7)
logger.info(f"Waiting {delay:.1f}s before page {page_num + 1}")
await asyncio.sleep(delay)
page_loaded = True
if page_num == 0:
# Navigate to page 1 by URL
url = self._build_search_url(location, check_in, check_out, adults)
logger.info(f"Scraping page {page_num + 1}: {url}")
try:
await page.goto(url, wait_until='domcontentloaded', timeout=30000)
except Exception as e:
logger.warning(f"Page load timeout, continuing: {e}")
page_loaded = False
else:
# Paginate by clicking "next" like a human — deep-linking
# ?offset=25 is a stronger bot signal and gets throttled
clicked = await self._go_to_next_page(page)
if not clicked:
# Fall back to offset URL if the control isn't found
url = self._build_search_url(
location, check_in, check_out, adults, offset=page_num * 25
)
logger.info(f"Next-button not found, offset fallback: {url}")
try:
await page.goto(url, wait_until='domcontentloaded', timeout=30000)
except Exception as e:
logger.warning(f"Page load timeout, continuing: {e}")
page_loaded = False
# Check for blocking
content = await page.content()
is_blocked, reason = self.detect_blocking(content)
if is_blocked:
logger.warning(f"Blocking detected: {reason}")
return ScraperResult(
success=False,
blocked=True,
block_reason=reason,
hotels=all_hotels,
rates=all_rates,
page_content_sample=content[:1000],
pages_requested=pages,
pages_ok=pages_ok,
)
# Human-like mouse movement + scrolling
await self._human_like_mouse(page)
await self._human_like_scroll(page)
# Extract data (retries the selector once on timeout)
hotels, rates = await self._extract_search_results(page, check_in)
# A page counts as clean if it loaded fully and parsed.
# An empty page 1 means the results never rendered; an empty
# later page can legitimately be the end of the results.
if page_loaded and (hotels or page_num > 0):
pages_ok += 1
# Deduplicate by booking_com_id
for hotel, rate in zip(hotels, rates):
if hotel.booking_com_id and hotel.booking_com_id not in seen_hotel_ids:
seen_hotel_ids.add(hotel.booking_com_id)
all_hotels.append(hotel)
all_rates.append(rate)
logger.info(f"Page {page_num + 1}: found {len(hotels)} hotels, {len(all_hotels)} total unique")
return ScraperResult(
success=True,
blocked=False,
hotels=all_hotels,
rates=all_rates,
pages_requested=pages,
pages_ok=pages_ok,
)
except Exception as e:
logger.error(f"Scrape error: {e}")
return ScraperResult(
success=False,
blocked=False,
error_message=str(e),
hotels=all_hotels,
rates=all_rates,
pages_requested=pages,
pages_ok=pages_ok,
)
finally:
if page:
await page.close()
if context:
await context.close()
async def scrape_hotel_page(
self,
hotel_url: str,
check_in: date,
check_out: date,
adults: int = 2
) -> ScraperResult:
"""
Scrape individual hotel page for detailed rates.
Future expansion - placeholder for now.
Will extract available_qty from room dropdowns.
"""
# Not implemented in Phase 2a
logger.warning("scrape_hotel_page not yet implemented")
return ScraperResult(
success=False,
error_message="Hotel page scraping not yet implemented"
)
async def close(self):
"""Clean up browser resources."""
if self._browser:
await self._browser.close()
self._browser = None
if self._playwright:
await self._playwright.stop()
self._playwright = None