import logging from datetime import datetime, timedelta, timezone from repositories.competitor_repository import competitor_repository from repositories.product_repository import product_repository from repositories.price_history_repository import price_history_repository from repositories.scrape_repository import scrape_repository from services import alert_service logger = logging.getLogger(__name__) # A price move smaller than this is normal noise, not alert-worthy — # matches the >5% threshold in CLAUDE.md's n8n price-change-alert spec. PRICE_CHANGE_ALERT_THRESHOLD_PERCENT = 5 def needs_refresh(competitor): """ Decides whether a competitor needs a live re-scrape (REFRESH path) or can be served from PocketBase (FAST path). Data flow: competitor record → change_detected flag set? → True (refresh) → last_scraped missing or older than CACHE_STALENESS_HOURS? → True (refresh) → otherwise → False (fast path is safe) """ if competitor.get('change_detected'): return True last_scraped = competitor.get('last_scraped') if not last_scraped: return True try: scraped_at = datetime.fromisoformat(last_scraped.replace('Z', '+00:00')) except (ValueError, AttributeError): return True # CACHE_STALENESS_HOURS' canonical, adjustable home is core/scraper.py # per CLAUDE.md §7 ("in core/scraper.py — adjust here only") — imported # lazily here for the same reason every other core.* reference in this # file is lazy: this module must stay importable independent of Phase 2. from core.scraper import CACHE_STALENESS_HOURS age = datetime.now(timezone.utc) - scraped_at return age > timedelta(hours=CACHE_STALENESS_HOURS) def analyse_from_cache(tenant_id, competitor_id, context): """ FAST path analysis — uses cached PocketBase data instead of a live scrape. The LLM receives already-structured JSON and generates the analysis narrative, comments, and relevancy score only — it does NOT re-extract fields from raw page content. Data flow: competitor_id → PocketBase products (latest) → structured JSON + client trip context (destination/duration/style) → LLM prompt (analysis only, no extraction, via core.extractor) → narrative + comments + relevancy score returned """ products = product_repository.get_latest_for_competitor(competitor_id, limit=1) if not products: raise ValueError('No cached product data available for fast path') latest = products[0] # core.extractor is a Phase 2 module — imported lazily so this # service is importable before the scraping engine refactor lands. from core.extractor import extract_with_openrouter prompt = ( 'You are a travel industry analyst. Given this already-extracted ' 'competitor trip data, write a short comparative analysis against ' f"the client's product context: {context}.\n\nCompetitor data:\n{latest}" ) return extract_with_openrouter(prompt, content='') def detect_change(competitor_id, new_hash): """ Compares new_hash against the last content_hash recorded for this competitor. On a diff: sets competitor.change_detected = true and change_detected_at = now. Replaces Firecrawl Monitor's webhook role (CLAUDE.md §4/§10) — the daily/on-demand scrape itself is now the only place a change is ever detected. Data flow: competitor_id + new_hash → scrape_repository.get_last_hash_for_url() → no prior hash: True (first scrape — always process) → hash differs: competitor_repository.update() sets change flags → True → hash matches: no update → False """ last_hash = scrape_repository.get_last_hash_for_url(competitor_id) if last_hash is None: return True if last_hash != new_hash: competitor_repository.update(competitor_id, { 'change_detected': True, 'change_detected_at': datetime.now(timezone.utc).isoformat(), }) return True return False def run_competitor_analysis(tenant_id, competitor, context): """ Decides the fast vs refresh path for one competitor and executes it. Called per-competitor by jobs.run_research_job — failures here are caught by the caller so one competitor's failure never aborts the whole batch (CLAUDE.md §18 resilience rule). Data flow: competitor + context → needs_refresh() decision → FAST: analyse_from_cache() (seconds) → REFRESH: core.scraper.scrape() → compute_content_hash() → detect_change() against the last recorded hash → unchanged: change_detected cleared (confirmed stable since last look), skip LLM extraction, reuse cached narrative, bump last_scraped only (no wasted extraction cost) → changed: core.extractor.extract_with_openrouter() → PocketBase products/price_history updated → last_scraped bumped, change_detected left True (stays visible until the next scrape confirms no further change) → result dict returned """ if not needs_refresh(competitor): logger.info(f'{competitor["name"]} — fast path (cache fresh)') return {'path': 'fast', 'result': analyse_from_cache(tenant_id, competitor['id'], context)} logger.info(f'{competitor["name"]} — refresh path (stale or changed)') # Phase 2 modules — imported lazily, same reasoning as above. from core.scraper import scrape, compute_content_hash from core.extractor import extract_with_openrouter from industries.adventure_travel.prompt import build_prompt page_data = scrape(competitor['website'], country=competitor.get('primary_market')) if not page_data.get('success'): raise ValueError(page_data.get('text', 'Scrape failed')) content_hash = compute_content_hash(page_data['text']) changed = detect_change(competitor['id'], content_hash) # Every refresh scrape gets its own scrape_runs record purely to # persist content_hash for the next comparison — the outer batch # job's scrape_runs row (created before run_research_job) tracks # overall job status, not per-competitor hash history. scrape_repository.create({ 'tenant_id': tenant_id, 'competitor_id': competitor['id'], 'triggered_by': 'cron', 'status': 'complete', 'content_hash': content_hash, 'content_length': len(page_data['text']), 'hash_algorithm': 'sha256', 'competitors_total': 1, 'competitors_done': 1, 'results': {}, 'completed_at': datetime.now(timezone.utc).isoformat(), }) if not changed: # Confirmed no further change since the last scrape — this is # the moment change_detected gets cleared. If it was never true # to begin with, this is a harmless no-op. logger.info(f'{competitor["name"]} — content unchanged, skipping LLM extraction') competitor_repository.update(competitor['id'], { 'change_detected': False, 'last_scraped': datetime.now(timezone.utc).isoformat(), }) return {'path': 'refresh', 'result': analyse_from_cache(tenant_id, competitor['id'], context), 'unchanged': True} prompt = build_prompt( competitor['name'], page_data, context.get('product_name'), context.get('destination'), context.get('duration'), context.get('travel_style'), is_ngs=context.get('tab_type') == 'NGS' ) extracted = extract_with_openrouter(prompt, content=page_data['text']) product = _save_scrape_result(tenant_id, competitor, extracted, page_data) # change_detected is deliberately left as detect_change() set it # (True) rather than reset here — detection and extraction now # happen in the same scrape, so there's no separate "handled" event # to reset it on. It stays visible (CompetitorsPage's "⚠ Changed" # badge, Apps Script sidebar status, tomorrow's n8n cron filter) # until the *next* scrape confirms no further change and clears it # in the branch above — at worst one extra hash-only confirmation # scrape per genuine change, no LLM cost since it lands on the # unchanged branch. competitor_repository.update(competitor['id'], { 'last_scraped': datetime.now(timezone.utc).isoformat(), }) return {'path': 'refresh', 'result': extracted, 'product': product} def _save_scrape_result(tenant_id, competitor, extracted, page_data): """ Persists one competitor's freshly-extracted trip data, deciding between "this is a new product" and "this is an update to a product we already track" by matching on (competitor_id, trip_name) — the same dedup key CLAUDE.md's is_new detection and price-history diffing rely on (§15/§19 Phase 5). Data flow: extracted fields → product_repository.get_by_competitor_and_name() → no match: create product with is_new=True → alert_service.create_alert('new_product') → match found: compare majority_price_usd → changed: price_history record written, product updated, alert_service.create_alert('price_change') if the move exceeds PRICE_CHANGE_ALERT_THRESHOLD_PERCENT → unchanged: product's last_seen/fields refreshed, no alert → product record returned either way """ now = datetime.now(timezone.utc).isoformat() trip_name = extracted.get('tripName') new_price = extracted.get('majorityPrice') common_fields = { 'trip_name': trip_name, 'url': page_data['url'], 'destination': extracted.get('destination'), 'duration_days': extracted.get('duration'), 'majority_price_usd': new_price, 'price_low_usd': extracted.get('priceLow'), 'price_high_usd': extracted.get('priceHigh'), 'aud_price': extracted.get('audPrice'), 'cad_price': extracted.get('cadPrice'), 'eur_price': extracted.get('eurPrice'), 'gbp_price': extracted.get('gbpPrice'), 'date_used': extracted.get('dateUsed'), 'comments': extracted.get('comments'), 'relevancy': extracted.get('relevancy'), 'last_seen': now, 'is_active': True, } existing = product_repository.get_by_competitor_and_name(competitor['id'], trip_name) if not existing: product = product_repository.create({ 'tenant_id': tenant_id, 'competitor_id': competitor['id'], 'is_new': True, 'first_seen': now, **common_fields, }) alert_service.create_alert( tenant_id, competitor['id'], 'new_product', f"{competitor['name']} added a new product: {trip_name or 'Unnamed trip'}", product_id=product['id'], ) return product product = product_repository.update(existing['id'], {**common_fields, 'is_new': False}) old_price = existing.get('majority_price_usd') if old_price is not None and new_price is not None and old_price != new_price: change_amount = new_price - old_price change_percent = round((change_amount / old_price) * 100, 1) if old_price else None price_history_repository.create({ 'tenant_id': tenant_id, 'product_id': product['id'], 'majority_price_usd': new_price, 'price_low_usd': extracted.get('priceLow'), 'price_high_usd': extracted.get('priceHigh'), 'aud_price': extracted.get('audPrice'), 'cad_price': extracted.get('cadPrice'), 'eur_price': extracted.get('eurPrice'), 'gbp_price': extracted.get('gbpPrice'), 'date_used': extracted.get('dateUsed'), 'change_field': 'majorityPrice', 'change_amount': change_amount, 'change_percent': change_percent, }) if change_percent is not None and abs(change_percent) > PRICE_CHANGE_ALERT_THRESHOLD_PERCENT: direction = 'dropped' if change_percent < 0 else 'raised' alert_service.create_alert( tenant_id, competitor['id'], 'price_change', f"{competitor['name']} {direction} {trip_name or 'a product'} {change_percent}%", product_id=product['id'], change_amount=change_amount, change_percent=change_percent, ) return product