import re import hashlib import asyncio import logging from urllib.parse import urlparse, urlencode, parse_qs, urlunparse from urllib.robotparser import RobotFileParser from playwright.async_api import async_playwright from playwright_stealth import stealth_async from bs4 import BeautifulSoup from core.proxy_service import proxy_service logger = logging.getLogger(__name__) # Default staleness threshold for the on-demand fast/refresh path # decision (CLAUDE.md §7). This is the canonical, adjustable location # per §7 — services/analysis_service.py imports it from here rather # than redefining it. CACHE_STALENESS_HOURS = 24 # Block only known third-party analytics/ad domains — never block the # competitor's own first-party assets. BLOCKED_DOMAINS = [ 'google-analytics.com', 'googletagmanager.com', 'doubleclick.net', 'facebook.net', 'facebook.com/tr', 'hotjar.com', 'intercom.io', 'zendesk.com', 'hubspot.com', 'marketo.com', 'segment.com', 'amplitude.com', 'mixpanel.com', 'optimizely.com', 'crazyegg.com', ] def extract_domain(url): """Strips scheme/path down to a bare domain — used for proxy_overrides and robots.txt lookups.""" return urlparse(url if '://' in url else f'//{url}').netloc.lower().lstrip('www.') def check_robots_txt(url): """ Checks if the given URL is allowed to be scraped per robots.txt. Returns True if scraping is permitted, False if disallowed. Defaults to True if robots.txt cannot be fetched (permissive default) — CLAUDE.md §30 Web Scraping Legal Policy. Data flow: url → extract root domain → fetch /robots.txt → parse with urllib.robotparser → check if Bettersight's user-agent is allowed for this path """ root = f'{urlparse(url).scheme}://{urlparse(url).netloc}' parser = RobotFileParser() parser.set_url(f'{root}/robots.txt') try: parser.read() return parser.can_fetch('*', url) except Exception: return True def normalise_html(content): """ Normalises scraped page content before hashing so cosmetic-only differences (whitespace, digit/word-case variation) never trigger a false change_detected. Operates on scrape()'s already-cleaned `text` field — script/style/nav/footer tags are stripped upstream in _render_with_proxy() — so this only needs to collapse whitespace and lowercase for a stable comparison. """ return re.sub(r'\s+', ' ', content).strip().lower() def compute_content_hash(content): """ Returns the SHA-256 hash of normalised page content. Used by the daily scrape job to detect competitor page changes without an external monitoring service (CLAUDE.md §4/§10 — replaces the Firecrawl Monitor feature, which self-hosted requires credit accounting and Supabase auth). """ return hashlib.sha256(normalise_html(content).encode('utf-8')).hexdigest() def clean_url(url): """Strips URL fragments and common tracking parameters.""" parsed = urlparse(url) strip_params = { 'utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'referrerView', 'ref', 'referrer', 'source', 'fbclid', 'gclid', 'mc_cid', 'mc_eid', '_ga', 'affiliate', } if parsed.query: params = parse_qs(parsed.query, keep_blank_values=True) cleaned = {k: v for k, v in params.items() if k not in strip_params} query = urlencode(cleaned, doseq=True) else: query = '' clean = urlunparse((parsed.scheme, parsed.netloc, parsed.path, parsed.params, query, '')) if clean != url: logger.info(f'URL cleaned: {url} → {clean}') return clean def _is_blocked(result): """ Detects common block signals from a scrape result. Returns True if the response looks like a bot wall rather than real content. Signals: low char count, HTTP 403/429, Cloudflare challenge, CAPTCHA text. """ if result.get('char_count', 0) < 500: return True if result.get('status_code') in (403, 429): return True block_signals = ['cf-browser-verification', 'captcha', 'access denied', 'blocked', 'robot'] return any(signal in result.get('text', '').lower() for signal in block_signals) async def _render_with_proxy(url, proxy): """ Launches a stealth Playwright browser through the given proxy config and extracts cleaned page text + tables. This is the actual page-rendering logic — split out from render_page() so it can be retried with a different proxy (Bright Data) on the first attempt being blocked, without duplicating the browser setup. Data flow: url + proxy config → Playwright chromium launch (proxied) → playwright-stealth applied → navigate → dismiss cookie banner → wait for content → scroll to trigger lazy-loading → HTML → BeautifulSoup cleaning → { text, tables, url, char_count, success } """ async with async_playwright() as p: browser = await p.chromium.launch( headless=True, proxy=proxy, args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-dev-shm-usage', '--disable-gpu'] ) context = await browser.new_context( user_agent=( 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) ' 'AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36' ), viewport={'width': 1366, 'height': 768}, locale='en-GB', timezone_id='Europe/London', extra_http_headers={ 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8', 'Accept-Language': 'en-GB,en;q=0.9', 'Accept-Encoding': 'gzip, deflate, br', 'DNT': '1', 'Upgrade-Insecure-Requests': '1', 'Sec-Fetch-Dest': 'document', 'Sec-Fetch-Mode': 'navigate', 'Sec-Fetch-Site': 'none', 'Sec-Fetch-User': '?1', } ) page = await context.new_page() # Comprehensive stealth (navigator.webdriver, plugins, canvas/WebGL # fingerprint, audio context, chrome runtime, ~20 evasions total) # replaces the old manual add_init_script block per CLAUDE.md §10. await stealth_async(page) async def block_trackers(route): if any(domain in route.request.url for domain in BLOCKED_DOMAINS): await route.abort() else: await route.continue_() await page.route('**/*', block_trackers) try: try: await page.goto(url, wait_until='domcontentloaded', timeout=30000) logger.info(f'Loaded: {url}') except Exception as e: logger.warning(f'domcontentloaded failed, trying networkidle: {str(e)[:100]}') await page.goto(url, wait_until='networkidle', timeout=60000) # Dismiss cookie consent banners try: await page.click( 'button:has-text("Accept All"), button:has-text("Accept all"), ' 'button:has-text("Accept"), button:has-text("OK"), ' 'button:has-text("I agree"), button:has-text("Agree"), ' '[id*="cookie"] button, [class*="cookie"] button, ' '[id*="consent"] button, [class*="consent"] button', timeout=3000 ) logger.info('Cookie banner dismissed') await page.wait_for_timeout(1000) except Exception: pass # Wait for meaningful content try: await page.wait_for_selector( 'h1, .trip-overview, main, article, #overview, ' '[class*="itinerary"], [class*="price"], [class*="trip"]', timeout=8000 ) except Exception: pass await page.wait_for_timeout(3000) # Two scroll passes to trigger lazy-loaded content for _ in range(2): await page.evaluate('window.scrollTo(0, document.body.scrollHeight)') await page.wait_for_timeout(1500) await page.evaluate('window.scrollTo(0, 0)') await page.wait_for_timeout(500) await page.evaluate('window.scrollTo(0, document.body.scrollHeight)') await page.wait_for_timeout(2000) html = await page.content() soup = BeautifulSoup(html, 'lxml') for tag in soup(['script', 'style', 'nav', 'footer', 'header', 'iframe', 'noscript', 'svg', 'button', 'form']): tag.decompose() main_content = ( soup.find('main') or soup.find('article') or soup.find(id='main') or soup.find(id='content') or soup.find(class_='content') or soup ) text = main_content.get_text(separator=' ', strip=True) text = re.sub(r'\s+', ' ', text).strip() tables = [t.get_text(separator=' | ', strip=True) for t in soup.find_all('table')] table_text = '\n'.join(tables) char_count = len(text) if char_count < 500: logger.warning(f'Low char count ({char_count}) for {url} — will attempt AI web fetch fallback') return { 'text': text[:25000], 'tables': table_text[:5000], 'url': url, 'char_count': char_count, 'success': True, } except Exception as e: logger.error(f'Render failed for {url}: {str(e)}') return {'text': f'Render failed: {str(e)}', 'tables': '', 'url': url, 'char_count': 0, 'success': False} finally: await browser.close() async def render_page(url, country=None): """ Renders a page using the smart proxy service with country targeting. `country` comes from competitors.primary_market in PocketBase. Detects blocks and auto-escalates to Bright Data if Webshare fails. Data flow: url + country → domain extracted → proxy_service.get_playwright_proxy_for_domain(domain, country) → _render_with_proxy() → if blocked → escalate_to_brightdata(domain) → retry once with Bright Data (no country targeting at fallback level) """ domain = extract_domain(url) proxy = proxy_service.get_playwright_proxy_for_domain(domain, country) result = await _render_with_proxy(url, proxy) if _is_blocked(result): logger.warning(f'Webshare blocked on {domain} — escalating to Bright Data') proxy_service.escalate_to_brightdata(domain) result = await _render_with_proxy(url, proxy_service._brightdata_playwright()) return result def scrape(url, country=None): """ Synchronous entry point for scraping one competitor page. Checks robots.txt first (legal/compliance gate per §30), then renders via Playwright, falling back to the AI web fetch path if Playwright's char count is too low (bot-protected sites). Data flow: url + country → clean_url() → check_robots_txt() → disallowed: return failure immediately (no scrape attempted) → allowed: render_page() (sync wrapper via nest_asyncio) → char_count < 1000: fetch_with_ai() fallback → final { text, tables, url, char_count, success } returned """ import nest_asyncio nest_asyncio.apply() url = clean_url(url) if not check_robots_txt(url): logger.warning(f'robots.txt disallows scraping {url} — skipping') return { 'text': 'Scraping disallowed by robots.txt for this URL', 'tables': '', 'url': url, 'char_count': 0, 'success': False, } loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) try: result = loop.run_until_complete(render_page(url, country=country)) finally: loop.close() if result.get('char_count', 0) < 1000 and result.get('success'): logger.info(f'Playwright char count low — trying AI web fetch for {url}') from core.ai_fetch import fetch_with_ai ai_result = fetch_with_ai(url) if ai_result: return ai_result return result