193 lines
7.0 KiB
Python
193 lines
7.0 KiB
Python
import asyncio
|
|
from core import scraper
|
|
|
|
|
|
def _run_async(coro):
|
|
"""
|
|
Runs a coroutine on a fresh, explicit event loop — mirrors exactly
|
|
how core.scraper.scrape() drives render_page() in production.
|
|
Deliberately NOT asyncio.run(): scrape()'s own nest_asyncio.apply()
|
|
call (exercised by other tests in this file) patches asyncio's
|
|
"current event loop" machinery process-wide, which makes a bare
|
|
asyncio.run() here liable to pick up and try to reuse a closed loop
|
|
left over from an earlier scrape() test, depending on test order.
|
|
"""
|
|
loop = asyncio.new_event_loop()
|
|
try:
|
|
return loop.run_until_complete(coro)
|
|
finally:
|
|
loop.close()
|
|
|
|
|
|
def test_extract_domain_strips_scheme_and_www():
|
|
assert scraper.extract_domain('https://www.intrepidtravel.com/peru') == 'intrepidtravel.com'
|
|
assert scraper.extract_domain('intrepidtravel.com') == 'intrepidtravel.com'
|
|
|
|
|
|
def test_clean_url_strips_tracking_params():
|
|
dirty = 'https://intrepidtravel.com/peru?utm_source=fb&utm_campaign=x&keep=1'
|
|
clean = scraper.clean_url(dirty)
|
|
assert 'utm_source' not in clean
|
|
assert 'utm_campaign' not in clean
|
|
assert 'keep=1' in clean
|
|
|
|
|
|
def test_clean_url_strips_fragment():
|
|
assert scraper.clean_url('https://intrepidtravel.com/peru#reviews') == 'https://intrepidtravel.com/peru'
|
|
|
|
|
|
def test_clean_url_unchanged_when_already_clean():
|
|
url = 'https://intrepidtravel.com/peru'
|
|
assert scraper.clean_url(url) == url
|
|
|
|
|
|
def test_is_blocked_true_on_low_char_count():
|
|
assert scraper._is_blocked({'char_count': 100, 'text': ''}) is True
|
|
|
|
|
|
def test_is_blocked_true_on_403_status():
|
|
assert scraper._is_blocked({'char_count': 2000, 'status_code': 403, 'text': ''}) is True
|
|
|
|
|
|
def test_is_blocked_true_on_captcha_text():
|
|
assert scraper._is_blocked({'char_count': 2000, 'text': 'Please complete this CAPTCHA to continue'}) is True
|
|
|
|
|
|
def test_is_blocked_false_for_normal_content():
|
|
assert scraper._is_blocked({'char_count': 5000, 'status_code': 200, 'text': 'Peru Classic 15 days...'}) is False
|
|
|
|
|
|
def test_check_robots_txt_permissive_default_on_fetch_failure(monkeypatch):
|
|
class _RaisingParser:
|
|
def set_url(self, url):
|
|
pass
|
|
|
|
def read(self):
|
|
raise Exception('unreachable')
|
|
|
|
monkeypatch.setattr(scraper, 'RobotFileParser', lambda: _RaisingParser())
|
|
assert scraper.check_robots_txt('https://intrepidtravel.com/peru') is True
|
|
|
|
|
|
def test_check_robots_txt_respects_disallow(monkeypatch):
|
|
class _DisallowParser:
|
|
def set_url(self, url):
|
|
pass
|
|
|
|
def read(self):
|
|
pass
|
|
|
|
def can_fetch(self, agent, url):
|
|
return False
|
|
|
|
monkeypatch.setattr(scraper, 'RobotFileParser', lambda: _DisallowParser())
|
|
assert scraper.check_robots_txt('https://intrepidtravel.com/peru') is False
|
|
|
|
|
|
def test_scrape_skips_when_robots_txt_disallows(monkeypatch):
|
|
monkeypatch.setattr(scraper, 'check_robots_txt', lambda url: False)
|
|
result = scraper.scrape('https://intrepidtravel.com/peru')
|
|
assert result['success'] is False
|
|
assert 'robots.txt' in result['text']
|
|
|
|
|
|
def test_scrape_returns_playwright_result_when_content_is_sufficient(monkeypatch):
|
|
monkeypatch.setattr(scraper, 'check_robots_txt', lambda url: True)
|
|
|
|
async def _fake_render_page(url, country=None):
|
|
return {'text': 'x' * 2000, 'tables': '', 'url': url, 'char_count': 2000, 'success': True}
|
|
|
|
monkeypatch.setattr(scraper, 'render_page', _fake_render_page)
|
|
|
|
result = scraper.scrape('https://intrepidtravel.com/peru')
|
|
assert result['char_count'] == 2000
|
|
assert result['success'] is True
|
|
|
|
|
|
def test_scrape_falls_back_to_ai_fetch_when_char_count_low(monkeypatch):
|
|
import sys
|
|
import types
|
|
|
|
monkeypatch.setattr(scraper, 'check_robots_txt', lambda url: True)
|
|
|
|
async def _fake_render_page(url, country=None):
|
|
return {'text': 'short', 'tables': '', 'url': url, 'char_count': 5, 'success': True}
|
|
|
|
monkeypatch.setattr(scraper, 'render_page', _fake_render_page)
|
|
|
|
fake_ai_fetch = types.ModuleType('core.ai_fetch')
|
|
fake_ai_fetch.fetch_with_ai = lambda url: {
|
|
'text': 'ai-fetched content', 'tables': '', 'url': url, 'char_count': 2000, 'success': True
|
|
}
|
|
monkeypatch.setitem(sys.modules, 'core.ai_fetch', fake_ai_fetch)
|
|
|
|
result = scraper.scrape('https://intrepidtravel.com/peru')
|
|
assert result['text'] == 'ai-fetched content'
|
|
|
|
|
|
def test_render_page_uses_webshare_proxy_when_not_blocked(monkeypatch):
|
|
monkeypatch.setattr(
|
|
scraper.proxy_service, 'get_playwright_proxy_for_domain',
|
|
lambda domain, country=None: {'server': 'webshare-proxy'}
|
|
)
|
|
calls = []
|
|
|
|
async def _fake_render_with_proxy(url, proxy):
|
|
calls.append(proxy)
|
|
return {'char_count': 5000, 'status_code': 200, 'text': 'good content', 'success': True, 'url': url}
|
|
|
|
monkeypatch.setattr(scraper, '_render_with_proxy', _fake_render_with_proxy)
|
|
escalated = []
|
|
monkeypatch.setattr(scraper.proxy_service, 'escalate_to_brightdata', lambda domain: escalated.append(domain))
|
|
|
|
result = _run_async(scraper.render_page('https://intrepidtravel.com/peru', country='GB'))
|
|
|
|
assert result['success'] is True
|
|
assert calls == [{'server': 'webshare-proxy'}]
|
|
assert escalated == []
|
|
|
|
|
|
def test_render_page_escalates_to_brightdata_when_blocked(monkeypatch):
|
|
monkeypatch.setattr(
|
|
scraper.proxy_service, 'get_playwright_proxy_for_domain',
|
|
lambda domain, country=None: {'server': 'webshare-proxy'}
|
|
)
|
|
monkeypatch.setattr(scraper.proxy_service, '_brightdata_playwright', lambda: {'server': 'brightdata-proxy'})
|
|
|
|
call_log = []
|
|
|
|
async def _fake_render_with_proxy(url, proxy):
|
|
call_log.append(proxy)
|
|
if proxy['server'] == 'webshare-proxy':
|
|
return {'char_count': 50, 'status_code': 403, 'text': 'blocked', 'success': True, 'url': url}
|
|
return {'char_count': 5000, 'status_code': 200, 'text': 'good content', 'success': True, 'url': url}
|
|
|
|
monkeypatch.setattr(scraper, '_render_with_proxy', _fake_render_with_proxy)
|
|
escalated = []
|
|
monkeypatch.setattr(scraper.proxy_service, 'escalate_to_brightdata', lambda domain: escalated.append(domain))
|
|
|
|
result = _run_async(scraper.render_page('https://intrepidtravel.com/peru', country='GB'))
|
|
|
|
assert result['text'] == 'good content'
|
|
assert escalated == ['intrepidtravel.com']
|
|
assert call_log == [{'server': 'webshare-proxy'}, {'server': 'brightdata-proxy'}]
|
|
|
|
|
|
def test_scrape_keeps_playwright_result_when_ai_fetch_returns_none(monkeypatch):
|
|
import sys
|
|
import types
|
|
|
|
monkeypatch.setattr(scraper, 'check_robots_txt', lambda url: True)
|
|
|
|
async def _fake_render_page(url, country=None):
|
|
return {'text': 'short', 'tables': '', 'url': url, 'char_count': 5, 'success': True}
|
|
|
|
monkeypatch.setattr(scraper, 'render_page', _fake_render_page)
|
|
|
|
fake_ai_fetch = types.ModuleType('core.ai_fetch')
|
|
fake_ai_fetch.fetch_with_ai = lambda url: None
|
|
monkeypatch.setitem(sys.modules, 'core.ai_fetch', fake_ai_fetch)
|
|
|
|
result = scraper.scrape('https://intrepidtravel.com/peru')
|
|
assert result['char_count'] == 5
|