Files
changedetection.io/changedetectionio/content_fetchers/base.py
T
dgtlmoonandClaude Opus 5 ba2280ba81 External CDP Browser engine - replaces the extra_browser_* hackery
An "extra browser" was a name + a ws(s):// endpoint in settings.requests.extra_browsers,
selected by a watch as the magic string 'extra_browser_<name>'. That string resolved to
html_webdriver plus a custom connection URL, which meant the protocol the endpoint was
spoken to came from env vars rather than from the entry: CDP over a WebSocket with
PLAYWRIGHT_DRIVER_URL set, CDP via pyppeteer with FAST_PUPPETEER_CHROME_FETCHER, and the
W3C WebDriver protocol over HTTP on a Selenium-only install - where a wss:// URL cannot
work at all. The form only ever accepted ws:// / wss://, so the feature was silently
broken on exactly the installs that could not honour it.

So it becomes an engine, html_external_cdp, which pins the protocol: a subclass of the
Playwright fetcher that takes its endpoint from the watch's browser config
(FetcherConfig.connection_url) instead of the environment. It is base-only
(ready_to_use=False) because an endpoint is required, so each endpoint is one browser
config ("variation") on the Browsers page - which is what the old settings list was.

update_36 migrates each extra_browsers row to such a variation, keyed by the SAME
'extra_browser_<name>' string watches already hold, so no watch, group override, API value
or global default needs rewriting; the legacy selector simply becomes a real browser-config
id. A row whose endpoint the model rejects is logged and skipped rather than taking the
update chain, and with it startup, down.

Knock-on cleanups, all of which delete a special case rather than add one:

 - The proxy opt-out for custom endpoints is now Fetcher.ignores_proxy_setting, asked of
   the engine, instead of a string-prefix test in call_browser().
 - A live browser-steps / visual-selector session asks the engine where to connect
   (Fetcher.browser_steps_connection_url, overridden by html_external_cdp) and refuses an
   engine whose supports_browser_steps is False, instead of reading the env var itself and
   silently stepping a browser the watch does not check with. That refusal is real: on a
   Selenium install html_webdriver cannot drive a live session.
 - is_valid_browser_selector() answers "may a watch store this in fetch_backend?" in one
   place; the API (create/update/import), the quick-add form validator and the bulk "set
   browser" operation each had their own copy, which is how they came to disagree about
   whether a browser-config id was acceptable.
 - api-spec.yaml's fetch_backend pattern enumerated extra_browser_* while rejecting
   browser-config ids and every engine newer than html_webdriver. Valid values are
   per-install, so the schema now bounds the string and the handlers do the real check.
 - html_external_cdp registers unconditionally (unlike html_playwright_builtin): migrated
   configs name it, so it must resolve even without the playwright library, or those
   watches would quietly fetch with the plain HTTP client. The library is imported lazily
   inside run(), and an unavailable engine now warns instead of falling back silently.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 19:39:16 +02:00

309 lines
14 KiB
Python

import os
from abc import abstractmethod
from loguru import logger
from pydantic import BaseModel
from changedetectionio.content_fetchers import BrowserStepsStepException
from changedetectionio.strtobool import strtobool
class FetcherCapabilities(BaseModel):
"""Typed view of what a content fetcher can do.
Single source of truth for the fetcher capability flags. The flags live as
class attributes on each fetcher (supports_browser_steps etc); build a
validated instance from a fetcher class with FetcherCapabilities.from_fetcher(),
and call .model_dump() where a plain dict is expected (templates, plugin API).
"""
supports_browser_steps: bool = False # Can execute browser automation steps
supports_screenshots: bool = False # Can capture page screenshots
supports_xpath_element_data: bool = False # Can extract xpath element positions for visual selector
supports_request_blocking: bool = False # Can block requests by resource-type / URL pattern
supports_browser_type: bool = False # Can choose the browser engine (chromium/firefox/webkit)
supports_delete_created_files: bool = False # Launches locally & can clean up its temp files
supports_request_timeout: bool = False # Plain HTTP client: honours a per-profile request timeout
supports_custom_user_agent: bool = False # Honours a per-profile User-Agent (all fetchers)
supports_connection_url: bool = False # Connects to an endpoint named by the browser config
@classmethod
def from_fetcher(cls, fetcher_class):
"""Build capabilities from a fetcher class (or None -> all False)."""
return cls(**{
name: getattr(fetcher_class, name, False)
for name in cls.model_fields
})
@property
def is_browser(self):
"""A real browser: can screenshot / extract element positions (visual selector)."""
return bool(self.supports_screenshots or self.supports_xpath_element_data)
@property
def can_host_variation(self):
"""True if this engine has per-profile behaviour worth saving as a browser config -
a real browser (viewport/locale/…), the plain client's request timeout, or a custom
user-agent (all fetchers). Single source of truth for the /browsers 'Add variation' +
'Edit' actions."""
return bool(self.is_browser or self.supports_request_timeout or self.supports_custom_user_agent)
def get_playwright_bypass_csp():
"""Return whether Playwright-compatible browser contexts should bypass CSP.
Bypassing CSP remains enabled by default for backward compatibility. Some
remote CDP implementations do not support ``Page.setBypassCSP``; operators
can disable the option by setting ``PLAYWRIGHT_BYPASS_CSP=false``.
"""
return strtobool(os.getenv('PLAYWRIGHT_BYPASS_CSP', 'true'))
def manage_user_agent(headers, current_ua=''):
"""
Basic setting of user-agent
NOTE!!!!!! The service that does the actual Chrome fetching should handle any anti-robot techniques
THERE ARE MANY WAYS THAT IT CAN BE DETECTED AS A ROBOT!!
This does not take care of
- Scraping of 'navigator' (platform, productSub, vendor, oscpu etc etc) browser object (navigator.appVersion) etc
- TCP/IP fingerprint JA3 etc
- Graphic rendering fingerprinting
- Your IP being obviously in a pool of bad actors
- Too many requests
- Scraping of SCH-UA browser replies (thanks google!!)
- Scraping of ServiceWorker, new window calls etc
See https://filipvitas.medium.com/how-to-set-user-agent-header-with-puppeteer-js-and-not-fail-28c7a02165da
Puppeteer requests https://github.com/dgtlmoon/pyppeteerstealth
:param page:
:param headers:
:return:
"""
# Ask it what the user agent is, if its obviously ChromeHeadless, switch it to the default
ua_in_custom_headers = headers.get('User-Agent')
if ua_in_custom_headers:
return ua_in_custom_headers
if not ua_in_custom_headers and current_ua:
current_ua = current_ua.replace('HeadlessChrome', 'Chrome')
return current_ua
return None
class Fetcher():
# The fully-resolved concrete backend name this fetcher was chosen as
# (e.g. 'html_requests', 'html_webdriver'). Set by resolve_content_fetcher()
# so downstream consumers don't have to re-derive it from the class name.
backend_name = None
# Resolved per-watch browser behaviour (FetcherConfig), injected after construction.
# None by default so a fetcher that never reads it is completely unaffected.
browser_config = None
# True for engines that reach the web through someone else's service (an external CDP
# endpoint), where layering our own per-watch proxy on top is wrong rather than merely
# redundant. Read by call_browser() when it decides whether to pass a proxy URL.
ignores_proxy_setting = False
@classmethod
def browser_steps_connection_url(cls, browser_config=None):
"""CDP endpoint a live browser-steps / visual-selector session connects to.
A live session is always driven with Playwright whatever the engine fetches with, so the
default is the system driver. An engine whose browser lives somewhere else overrides this
and answers from the watch's browser config (see external_cdp).
"""
return os.getenv('PLAYWRIGHT_DRIVER_URL', '').strip('"')
# Whether this fetcher is usable directly, out-of-the-box, as a built-in "browser" (has sane
# env defaults). False means it's a *base only* - it must be configured via a browser config
# first (e.g. html_playwright_builtin needs a browser_type chosen), so it's offered as a base
# in the Add Browser form but NOT shown as a directly-selectable built-in browser.
ready_to_use = True
# Every fetcher sends an HTTP User-Agent (browsers via the request_headers channel -> the
# browser context, the plain client directly), so all support a per-profile UA override.
# Set here on the base so browser configs can carry a user_agent for any engine.
supports_custom_user_agent = True
browser_connection_is_custom = None
browser_connection_url = None
browser_steps = None
browser_steps_screenshot_path = None
content = None
error = None
fetcher_description = "No description"
headers = {}
favicon_blob = None
instock_data = None
instock_data_js = ""
screenshot_format = None
status_code = None
webdriver_js_execute_code = None
worker_id = None
xpath_data = None
xpath_element_js = ""
# Will be needed in the future by the VisualSelector, always get this where possible.
screenshot = False
system_http_proxy = os.getenv('HTTP_PROXY')
system_https_proxy = os.getenv('HTTPS_PROXY')
# Time ONTOP of the system defined env minimum time
render_extract_delay = 0
# Fetcher capability flags - subclasses should override these
# These indicate what features the fetcher supports
supports_browser_steps = False # Can execute browser automation steps
supports_screenshots = False # Can capture page screenshots
supports_xpath_element_data = False # Can extract xpath element positions/data for visual selector
# Screenshot element locking - prevents layout shifts during screenshot capture
# Only needed for visual comparison (image_ssim_diff processor)
# Locks element dimensions in the first viewport to prevent headers/ads from resizing
lock_viewport_elements = False # Default: disabled for performance
def __init__(self, **kwargs):
if kwargs and 'screenshot_format' in kwargs:
self.screenshot_format = kwargs.get('screenshot_format')
# Allow lock_viewport_elements to be set via kwargs
if kwargs and 'lock_viewport_elements' in kwargs:
self.lock_viewport_elements = kwargs.get('lock_viewport_elements')
# Which async worker is driving this fetch, subclasses use it to keep per-worker browser
# state (profile dirs etc) apart, stays None when we're not called from a worker
if kwargs and 'worker_id' in kwargs:
self.worker_id = kwargs.get('worker_id')
@classmethod
def get_status_icon_data(cls):
"""Return data for status icon to display in the watch overview.
This method can be overridden by subclasses to provide custom status icons.
Returns:
dict or None: Dictionary with icon data:
{
'filename': 'icon-name.svg', # Icon filename
'alt': 'Alt text', # Alt attribute
'title': 'Tooltip text', # Title attribute
'style': 'height: 1em;' # Optional inline CSS
}
Or None if no icon
"""
return None
def clear_content(self):
"""
Explicitly clear all content from memory to free up heap space.
Call this after content has been saved to disk.
"""
self.content = None
if hasattr(self, 'raw_content'):
self.raw_content = None
self.screenshot = None
self.xpath_data = None
# Keep headers and status_code as they're small
@abstractmethod
def get_error(self):
return self.error
@abstractmethod
async def run(self,
fetch_favicon=True,
current_include_filters=None,
empty_pages_are_a_change=False,
ignore_status_codes=False,
is_binary=False,
request_body=None,
request_headers=None,
request_method=None,
timeout=None,
url=None,
watch_uuid=None,
):
# Should set self.error, self.status_code and self.content
pass
@abstractmethod
async def quit(self, watch=None):
return
@abstractmethod
def get_last_status_code(self):
return self.status_code
@abstractmethod
def screenshot_step(self, step_n):
if self.browser_steps_screenshot_path and not os.path.isdir(self.browser_steps_screenshot_path):
logger.debug(f"> Creating data dir {self.browser_steps_screenshot_path}")
os.mkdir(self.browser_steps_screenshot_path)
return None
@abstractmethod
# Return true/false if this checker is ready to run, in the case it needs todo some special config check etc
def is_ready(self):
return True
def get_all_headers(self):
"""
Get all headers but ensure all keys are lowercase
:return:
"""
return {k.lower(): v for k, v in self.headers.items()}
async def iterate_browser_steps(self, start_url=None):
from changedetectionio.browser_steps.browser_steps import steppable_browser_interface, browser_steps_get_valid_steps
from playwright._impl._errors import TimeoutError, Error
from changedetectionio.jinja2_custom import render as jinja_render
step_n = 0
if self.browser_steps:
interface = steppable_browser_interface(start_url=start_url)
interface.page = self.page
valid_steps = browser_steps_get_valid_steps(self.browser_steps)
for step in valid_steps:
step_n += 1
logger.debug(f">> Iterating check - browser Step n {step_n} - {step['operation']}...")
await self.screenshot_step("before-" + str(step_n))
await self.save_step_html("before-" + str(step_n))
try:
optional_value = step['optional_value']
selector = step['selector']
# Support for jinja2 template in step values, with date module added
if '{%' in step['optional_value'] or '{{' in step['optional_value']:
optional_value = jinja_render(template_str=step['optional_value'])
if '{%' in step['selector'] or '{{' in step['selector']:
selector = jinja_render(template_str=step['selector'])
await getattr(interface, "call_action")(action_name=step['operation'],
selector=selector,
optional_value=optional_value)
await self.screenshot_step(step_n)
await self.save_step_html(step_n)
except (Error, TimeoutError, ValueError) as e:
# ValueError is what validate_fetch_url_async() raises when a step's URL is
# refused (file://, private IP, bad scheme) - report it against the offending
# step number like any other step failure, rather than failing the whole watch
# with an opaque error.
logger.debug(str(e))
# Stop processing here
raise BrowserStepsStepException(step_n=step_n, original_e=e)
# It's always good to reset these
def delete_browser_steps_screenshots(self):
import glob
if self.browser_steps_screenshot_path is not None:
dest = os.path.join(self.browser_steps_screenshot_path, 'step_*.jpeg')
files = glob.glob(dest)
for f in files:
if os.path.isfile(f):
os.unlink(f)
def save_step_html(self, step_n):
if self.browser_steps_screenshot_path and not os.path.isdir(self.browser_steps_screenshot_path):
logger.debug(f"> Creating data dir {self.browser_steps_screenshot_path}")
os.mkdir(self.browser_steps_screenshot_path)
pass