Allow parallel workers for hotel-page scraper without proxy

The proxy-off → serial guard was correct for search-results (all workers
hit the same URL), but hotel-page workers each scrape a different hotel's
property page — parallel requests look like multi-tab browsing, not a
hammered aggregation endpoint. Added require_proxy param to
_effective_concurrency; hotel-page path passes require_proxy=False.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
jtricerolph 2026-07-10 00:18:56 +00:00
parent 0721769e03
commit 1b985e5a16

View file

@ -524,15 +524,20 @@ def _is_hotel_page_mode(db: Session) -> bool:
return (row and row.config_value) == 'playwright_hotel_page' return (row and row.config_value) == 'playwright_hotel_page'
def _effective_concurrency(db: Session, n_jobs: int) -> int: def _effective_concurrency(db: Session, n_jobs: int, require_proxy: bool = True) -> int:
"""Clamp configured concurrency to the workload, and force serial when the """Clamp configured concurrency to the workload.
proxy is off N workers would share one IP and hammer it, worse than 1."""
require_proxy=True (search-results mode): force serial without a proxy because
all workers would hit the same search URL from the same IP.
require_proxy=False (hotel-page mode): allow parallelism even without proxy
each worker hits a different hotel's URL so concurrent requests look like normal
multi-tab browsing rather than a hammered aggregation endpoint.
"""
configured = get_scraper_concurrency(db) configured = get_scraper_concurrency(db)
if configured <= 1 or n_jobs <= 1: if configured <= 1 or n_jobs <= 1:
return 1 return 1
# get_scraper_backend does no I/O beyond the config read; safe to probe. if require_proxy and not get_scraper_backend(db)._proxy_enabled():
if not get_scraper_backend(db)._proxy_enabled(): logger.info("Proxy disabled — running search-results scrape serially (parallelism needs per-worker IPs)")
logger.info("Proxy disabled — running scrape serially (parallelism needs per-worker IPs)")
return 1 return 1
return max(1, min(configured, n_jobs)) return max(1, min(configured, n_jobs))
@ -767,7 +772,7 @@ async def _run_hotel_page_scrape(
) )
db.commit() db.commit()
concurrency = _effective_concurrency(db, len(hotels)) concurrency = _effective_concurrency(db, len(hotels), require_proxy=False)
logger.info( logger.info(
f"Hotel-page scrape {from_date}..{to_date}: " f"Hotel-page scrape {from_date}..{to_date}: "
f"{len(hotels)} hotels × {len(dates)} dates = {len(hotel_date_jobs)} jobs, " f"{len(hotels)} hotels × {len(dates)} dates = {len(hotel_date_jobs)} jobs, "