app.bettersight.io/backend/core/scraper.py

294 lines
11 KiB
Python

import re
import asyncio
import logging
from urllib.parse import urlparse, urlencode, parse_qs, urlunparse
from urllib.robotparser import RobotFileParser
from playwright.async_api import async_playwright
from playwright_stealth import stealth_async
from bs4 import BeautifulSoup
from core.proxy_service import proxy_service
logger = logging.getLogger(__name__)
# Default staleness threshold for the on-demand fast/refresh path
# decision (CLAUDE.md §7). This is the canonical, adjustable location
# per §7 — services/analysis_service.py imports it from here rather
# than redefining it.
CACHE_STALENESS_HOURS = 24
# Block only known third-party analytics/ad domains — never block the
# competitor's own first-party assets.
BLOCKED_DOMAINS = [
'google-analytics.com', 'googletagmanager.com', 'doubleclick.net',
'facebook.net', 'facebook.com/tr', 'hotjar.com', 'intercom.io',
'zendesk.com', 'hubspot.com', 'marketo.com', 'segment.com',
'amplitude.com', 'mixpanel.com', 'optimizely.com', 'crazyegg.com',
]
def extract_domain(url):
"""Strips scheme/path down to a bare domain — used for proxy_overrides and robots.txt lookups."""
return urlparse(url if '://' in url else f'//{url}').netloc.lower().lstrip('www.')
def check_robots_txt(url):
"""
Checks if the given URL is allowed to be scraped per robots.txt.
Returns True if scraping is permitted, False if disallowed.
Defaults to True if robots.txt cannot be fetched (permissive
default) — CLAUDE.md §30 Web Scraping Legal Policy.
Data flow:
url → extract root domain → fetch /robots.txt →
parse with urllib.robotparser →
check if Bettersight's user-agent is allowed for this path
"""
root = f'{urlparse(url).scheme}://{urlparse(url).netloc}'
parser = RobotFileParser()
parser.set_url(f'{root}/robots.txt')
try:
parser.read()
return parser.can_fetch('*', url)
except Exception:
return True
def clean_url(url):
"""Strips URL fragments and common tracking parameters."""
parsed = urlparse(url)
strip_params = {
'utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',
'referrerView', 'ref', 'referrer', 'source', 'fbclid', 'gclid',
'mc_cid', 'mc_eid', '_ga', 'affiliate',
}
if parsed.query:
params = parse_qs(parsed.query, keep_blank_values=True)
cleaned = {k: v for k, v in params.items() if k not in strip_params}
query = urlencode(cleaned, doseq=True)
else:
query = ''
clean = urlunparse((parsed.scheme, parsed.netloc, parsed.path, parsed.params, query, ''))
if clean != url:
logger.info(f'URL cleaned: {url}{clean}')
return clean
def _is_blocked(result):
"""
Detects common block signals from a scrape result. Returns True if
the response looks like a bot wall rather than real content.
Signals: low char count, HTTP 403/429, Cloudflare challenge, CAPTCHA text.
"""
if result.get('char_count', 0) < 500:
return True
if result.get('status_code') in (403, 429):
return True
block_signals = ['cf-browser-verification', 'captcha', 'access denied', 'blocked', 'robot']
return any(signal in result.get('text', '').lower() for signal in block_signals)
async def _render_with_proxy(url, proxy):
"""
Launches a stealth Playwright browser through the given proxy
config and extracts cleaned page text + tables. This is the actual
page-rendering logic — split out from render_page() so it can be
retried with a different proxy (Bright Data) on the first attempt
being blocked, without duplicating the browser setup.
Data flow:
url + proxy config → Playwright chromium launch (proxied) →
playwright-stealth applied → navigate → dismiss cookie banner →
wait for content → scroll to trigger lazy-loading →
HTML → BeautifulSoup cleaning → { text, tables, url, char_count, success }
"""
async with async_playwright() as p:
browser = await p.chromium.launch(
headless=True,
proxy=proxy,
args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-dev-shm-usage', '--disable-gpu']
)
context = await browser.new_context(
user_agent=(
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) '
'AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36'
),
viewport={'width': 1366, 'height': 768},
locale='en-GB',
timezone_id='Europe/London',
extra_http_headers={
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
'Accept-Language': 'en-GB,en;q=0.9',
'Accept-Encoding': 'gzip, deflate, br',
'DNT': '1',
'Upgrade-Insecure-Requests': '1',
'Sec-Fetch-Dest': 'document',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-Site': 'none',
'Sec-Fetch-User': '?1',
}
)
page = await context.new_page()
# Comprehensive stealth (navigator.webdriver, plugins, canvas/WebGL
# fingerprint, audio context, chrome runtime, ~20 evasions total)
# replaces the old manual add_init_script block per CLAUDE.md §10.
await stealth_async(page)
async def block_trackers(route):
if any(domain in route.request.url for domain in BLOCKED_DOMAINS):
await route.abort()
else:
await route.continue_()
await page.route('**/*', block_trackers)
try:
try:
await page.goto(url, wait_until='domcontentloaded', timeout=30000)
logger.info(f'Loaded: {url}')
except Exception as e:
logger.warning(f'domcontentloaded failed, trying networkidle: {str(e)[:100]}')
await page.goto(url, wait_until='networkidle', timeout=60000)
# Dismiss cookie consent banners
try:
await page.click(
'button:has-text("Accept All"), button:has-text("Accept all"), '
'button:has-text("Accept"), button:has-text("OK"), '
'button:has-text("I agree"), button:has-text("Agree"), '
'[id*="cookie"] button, [class*="cookie"] button, '
'[id*="consent"] button, [class*="consent"] button',
timeout=3000
)
logger.info('Cookie banner dismissed')
await page.wait_for_timeout(1000)
except Exception:
pass
# Wait for meaningful content
try:
await page.wait_for_selector(
'h1, .trip-overview, main, article, #overview, '
'[class*="itinerary"], [class*="price"], [class*="trip"]',
timeout=8000
)
except Exception:
pass
await page.wait_for_timeout(3000)
# Two scroll passes to trigger lazy-loaded content
for _ in range(2):
await page.evaluate('window.scrollTo(0, document.body.scrollHeight)')
await page.wait_for_timeout(1500)
await page.evaluate('window.scrollTo(0, 0)')
await page.wait_for_timeout(500)
await page.evaluate('window.scrollTo(0, document.body.scrollHeight)')
await page.wait_for_timeout(2000)
html = await page.content()
soup = BeautifulSoup(html, 'lxml')
for tag in soup(['script', 'style', 'nav', 'footer', 'header', 'iframe', 'noscript', 'svg', 'button', 'form']):
tag.decompose()
main_content = (
soup.find('main') or soup.find('article') or soup.find(id='main')
or soup.find(id='content') or soup.find(class_='content') or soup
)
text = main_content.get_text(separator=' ', strip=True)
text = re.sub(r'\s+', ' ', text).strip()
tables = [t.get_text(separator=' | ', strip=True) for t in soup.find_all('table')]
table_text = '\n'.join(tables)
char_count = len(text)
if char_count < 500:
logger.warning(f'Low char count ({char_count}) for {url} — will attempt AI web fetch fallback')
return {
'text': text[:25000],
'tables': table_text[:5000],
'url': url,
'char_count': char_count,
'success': True,
}
except Exception as e:
logger.error(f'Render failed for {url}: {str(e)}')
return {'text': f'Render failed: {str(e)}', 'tables': '', 'url': url, 'char_count': 0, 'success': False}
finally:
await browser.close()
async def render_page(url, country=None):
"""
Renders a page using the smart proxy service with country
targeting. `country` comes from competitors.primary_market in
PocketBase. Detects blocks and auto-escalates to Bright Data if
Webshare fails.
Data flow:
url + country → domain extracted →
proxy_service.get_playwright_proxy_for_domain(domain, country) →
_render_with_proxy() → if blocked → escalate_to_brightdata(domain) →
retry once with Bright Data (no country targeting at fallback level)
"""
domain = extract_domain(url)
proxy = proxy_service.get_playwright_proxy_for_domain(domain, country)
result = await _render_with_proxy(url, proxy)
if _is_blocked(result):
logger.warning(f'Webshare blocked on {domain} — escalating to Bright Data')
proxy_service.escalate_to_brightdata(domain)
result = await _render_with_proxy(url, proxy_service._brightdata_playwright())
return result
def scrape(url, country=None):
"""
Synchronous entry point for scraping one competitor page. Checks
robots.txt first (legal/compliance gate per §30), then renders via
Playwright, falling back to the AI web fetch path if Playwright's
char count is too low (bot-protected sites).
Data flow:
url + country → clean_url() → check_robots_txt() →
disallowed: return failure immediately (no scrape attempted) →
allowed: render_page() (sync wrapper via nest_asyncio) →
char_count < 1000: fetch_with_ai() fallback →
final { text, tables, url, char_count, success } returned
"""
import nest_asyncio
nest_asyncio.apply()
url = clean_url(url)
if not check_robots_txt(url):
logger.warning(f'robots.txt disallows scraping {url} — skipping')
return {
'text': 'Scraping disallowed by robots.txt for this URL',
'tables': '', 'url': url, 'char_count': 0, 'success': False,
}
loop = asyncio.new_event_loop()
asyncio.set_event_loop(loop)
try:
result = loop.run_until_complete(render_page(url, country=country))
finally:
loop.close()
if result.get('char_count', 0) < 1000 and result.get('success'):
logger.info(f'Playwright char count low — trying AI web fetch for {url}')
from core.ai_fetch import fetch_with_ai
ai_result = fetch_with_ai(url)
if ai_result:
return ai_result
return result