mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-10-04 02:58:10 +00:00
An "extra browser" was a name + a ws(s):// endpoint in settings.requests.extra_browsers,
selected by a watch as the magic string 'extra_browser_<name>'. That string resolved to
html_webdriver plus a custom connection URL, which meant the protocol the endpoint was
spoken to came from env vars rather than from the entry: CDP over a WebSocket with
PLAYWRIGHT_DRIVER_URL set, CDP via pyppeteer with FAST_PUPPETEER_CHROME_FETCHER, and the
W3C WebDriver protocol over HTTP on a Selenium-only install - where a wss:// URL cannot
work at all. The form only ever accepted ws:// / wss://, so the feature was silently
broken on exactly the installs that could not honour it.
So it becomes an engine, html_external_cdp, which pins the protocol: a subclass of the
Playwright fetcher that takes its endpoint from the watch's browser config
(FetcherConfig.connection_url) instead of the environment. It is base-only
(ready_to_use=False) because an endpoint is required, so each endpoint is one browser
config ("variation") on the Browsers page - which is what the old settings list was.
update_36 migrates each extra_browsers row to such a variation, keyed by the SAME
'extra_browser_<name>' string watches already hold, so no watch, group override, API value
or global default needs rewriting; the legacy selector simply becomes a real browser-config
id. A row whose endpoint the model rejects is logged and skipped rather than taking the
update chain, and with it startup, down.
Knock-on cleanups, all of which delete a special case rather than add one:
- The proxy opt-out for custom endpoints is now Fetcher.ignores_proxy_setting, asked of
the engine, instead of a string-prefix test in call_browser().
- A live browser-steps / visual-selector session asks the engine where to connect
(Fetcher.browser_steps_connection_url, overridden by html_external_cdp) and refuses an
engine whose supports_browser_steps is False, instead of reading the env var itself and
silently stepping a browser the watch does not check with. That refusal is real: on a
Selenium install html_webdriver cannot drive a live session.
- is_valid_browser_selector() answers "may a watch store this in fetch_backend?" in one
place; the API (create/update/import), the quick-add form validator and the bulk "set
browser" operation each had their own copy, which is how they came to disagree about
whether a browser-config id was acceptable.
- api-spec.yaml's fetch_backend pattern enumerated extra_browser_* while rejecting
browser-config ids and every engine newer than html_webdriver. Valid values are
per-install, so the schema now bounds the string and the handlers do the real check.
- html_external_cdp registers unconditionally (unlike html_playwright_builtin): migrated
configs name it, so it must resolve even without the playwright library, or those
watches would quietly fetch with the plain HTTP client. The library is imported lazily
inside run(), and an unavailable engine now warns instead of falling back silently.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
309 lines
14 KiB
Python
309 lines
14 KiB
Python
import os
|
|
from abc import abstractmethod
|
|
from loguru import logger
|
|
from pydantic import BaseModel
|
|
|
|
from changedetectionio.content_fetchers import BrowserStepsStepException
|
|
from changedetectionio.strtobool import strtobool
|
|
|
|
|
|
class FetcherCapabilities(BaseModel):
|
|
"""Typed view of what a content fetcher can do.
|
|
|
|
Single source of truth for the fetcher capability flags. The flags live as
|
|
class attributes on each fetcher (supports_browser_steps etc); build a
|
|
validated instance from a fetcher class with FetcherCapabilities.from_fetcher(),
|
|
and call .model_dump() where a plain dict is expected (templates, plugin API).
|
|
"""
|
|
supports_browser_steps: bool = False # Can execute browser automation steps
|
|
supports_screenshots: bool = False # Can capture page screenshots
|
|
supports_xpath_element_data: bool = False # Can extract xpath element positions for visual selector
|
|
supports_request_blocking: bool = False # Can block requests by resource-type / URL pattern
|
|
supports_browser_type: bool = False # Can choose the browser engine (chromium/firefox/webkit)
|
|
supports_delete_created_files: bool = False # Launches locally & can clean up its temp files
|
|
supports_request_timeout: bool = False # Plain HTTP client: honours a per-profile request timeout
|
|
supports_custom_user_agent: bool = False # Honours a per-profile User-Agent (all fetchers)
|
|
supports_connection_url: bool = False # Connects to an endpoint named by the browser config
|
|
|
|
@classmethod
|
|
def from_fetcher(cls, fetcher_class):
|
|
"""Build capabilities from a fetcher class (or None -> all False)."""
|
|
return cls(**{
|
|
name: getattr(fetcher_class, name, False)
|
|
for name in cls.model_fields
|
|
})
|
|
|
|
@property
|
|
def is_browser(self):
|
|
"""A real browser: can screenshot / extract element positions (visual selector)."""
|
|
return bool(self.supports_screenshots or self.supports_xpath_element_data)
|
|
|
|
@property
|
|
def can_host_variation(self):
|
|
"""True if this engine has per-profile behaviour worth saving as a browser config -
|
|
a real browser (viewport/locale/…), the plain client's request timeout, or a custom
|
|
user-agent (all fetchers). Single source of truth for the /browsers 'Add variation' +
|
|
'Edit' actions."""
|
|
return bool(self.is_browser or self.supports_request_timeout or self.supports_custom_user_agent)
|
|
|
|
|
|
def get_playwright_bypass_csp():
|
|
"""Return whether Playwright-compatible browser contexts should bypass CSP.
|
|
|
|
Bypassing CSP remains enabled by default for backward compatibility. Some
|
|
remote CDP implementations do not support ``Page.setBypassCSP``; operators
|
|
can disable the option by setting ``PLAYWRIGHT_BYPASS_CSP=false``.
|
|
"""
|
|
return strtobool(os.getenv('PLAYWRIGHT_BYPASS_CSP', 'true'))
|
|
|
|
|
|
def manage_user_agent(headers, current_ua=''):
|
|
"""
|
|
Basic setting of user-agent
|
|
|
|
NOTE!!!!!! The service that does the actual Chrome fetching should handle any anti-robot techniques
|
|
THERE ARE MANY WAYS THAT IT CAN BE DETECTED AS A ROBOT!!
|
|
This does not take care of
|
|
- Scraping of 'navigator' (platform, productSub, vendor, oscpu etc etc) browser object (navigator.appVersion) etc
|
|
- TCP/IP fingerprint JA3 etc
|
|
- Graphic rendering fingerprinting
|
|
- Your IP being obviously in a pool of bad actors
|
|
- Too many requests
|
|
- Scraping of SCH-UA browser replies (thanks google!!)
|
|
- Scraping of ServiceWorker, new window calls etc
|
|
|
|
See https://filipvitas.medium.com/how-to-set-user-agent-header-with-puppeteer-js-and-not-fail-28c7a02165da
|
|
Puppeteer requests https://github.com/dgtlmoon/pyppeteerstealth
|
|
|
|
:param page:
|
|
:param headers:
|
|
:return:
|
|
"""
|
|
# Ask it what the user agent is, if its obviously ChromeHeadless, switch it to the default
|
|
ua_in_custom_headers = headers.get('User-Agent')
|
|
if ua_in_custom_headers:
|
|
return ua_in_custom_headers
|
|
|
|
if not ua_in_custom_headers and current_ua:
|
|
current_ua = current_ua.replace('HeadlessChrome', 'Chrome')
|
|
return current_ua
|
|
|
|
return None
|
|
|
|
class Fetcher():
|
|
# The fully-resolved concrete backend name this fetcher was chosen as
|
|
# (e.g. 'html_requests', 'html_webdriver'). Set by resolve_content_fetcher()
|
|
# so downstream consumers don't have to re-derive it from the class name.
|
|
backend_name = None
|
|
# Resolved per-watch browser behaviour (FetcherConfig), injected after construction.
|
|
# None by default so a fetcher that never reads it is completely unaffected.
|
|
browser_config = None
|
|
# True for engines that reach the web through someone else's service (an external CDP
|
|
# endpoint), where layering our own per-watch proxy on top is wrong rather than merely
|
|
# redundant. Read by call_browser() when it decides whether to pass a proxy URL.
|
|
ignores_proxy_setting = False
|
|
|
|
@classmethod
|
|
def browser_steps_connection_url(cls, browser_config=None):
|
|
"""CDP endpoint a live browser-steps / visual-selector session connects to.
|
|
|
|
A live session is always driven with Playwright whatever the engine fetches with, so the
|
|
default is the system driver. An engine whose browser lives somewhere else overrides this
|
|
and answers from the watch's browser config (see external_cdp).
|
|
"""
|
|
return os.getenv('PLAYWRIGHT_DRIVER_URL', '').strip('"')
|
|
|
|
# Whether this fetcher is usable directly, out-of-the-box, as a built-in "browser" (has sane
|
|
# env defaults). False means it's a *base only* - it must be configured via a browser config
|
|
# first (e.g. html_playwright_builtin needs a browser_type chosen), so it's offered as a base
|
|
# in the Add Browser form but NOT shown as a directly-selectable built-in browser.
|
|
ready_to_use = True
|
|
# Every fetcher sends an HTTP User-Agent (browsers via the request_headers channel -> the
|
|
# browser context, the plain client directly), so all support a per-profile UA override.
|
|
# Set here on the base so browser configs can carry a user_agent for any engine.
|
|
supports_custom_user_agent = True
|
|
browser_connection_is_custom = None
|
|
browser_connection_url = None
|
|
browser_steps = None
|
|
browser_steps_screenshot_path = None
|
|
content = None
|
|
error = None
|
|
fetcher_description = "No description"
|
|
headers = {}
|
|
favicon_blob = None
|
|
instock_data = None
|
|
instock_data_js = ""
|
|
screenshot_format = None
|
|
status_code = None
|
|
webdriver_js_execute_code = None
|
|
worker_id = None
|
|
xpath_data = None
|
|
xpath_element_js = ""
|
|
|
|
# Will be needed in the future by the VisualSelector, always get this where possible.
|
|
screenshot = False
|
|
system_http_proxy = os.getenv('HTTP_PROXY')
|
|
system_https_proxy = os.getenv('HTTPS_PROXY')
|
|
|
|
# Time ONTOP of the system defined env minimum time
|
|
render_extract_delay = 0
|
|
|
|
# Fetcher capability flags - subclasses should override these
|
|
# These indicate what features the fetcher supports
|
|
supports_browser_steps = False # Can execute browser automation steps
|
|
supports_screenshots = False # Can capture page screenshots
|
|
supports_xpath_element_data = False # Can extract xpath element positions/data for visual selector
|
|
|
|
# Screenshot element locking - prevents layout shifts during screenshot capture
|
|
# Only needed for visual comparison (image_ssim_diff processor)
|
|
# Locks element dimensions in the first viewport to prevent headers/ads from resizing
|
|
lock_viewport_elements = False # Default: disabled for performance
|
|
|
|
def __init__(self, **kwargs):
|
|
if kwargs and 'screenshot_format' in kwargs:
|
|
self.screenshot_format = kwargs.get('screenshot_format')
|
|
|
|
# Allow lock_viewport_elements to be set via kwargs
|
|
if kwargs and 'lock_viewport_elements' in kwargs:
|
|
self.lock_viewport_elements = kwargs.get('lock_viewport_elements')
|
|
|
|
# Which async worker is driving this fetch, subclasses use it to keep per-worker browser
|
|
# state (profile dirs etc) apart, stays None when we're not called from a worker
|
|
if kwargs and 'worker_id' in kwargs:
|
|
self.worker_id = kwargs.get('worker_id')
|
|
|
|
|
|
@classmethod
|
|
def get_status_icon_data(cls):
|
|
"""Return data for status icon to display in the watch overview.
|
|
|
|
This method can be overridden by subclasses to provide custom status icons.
|
|
|
|
Returns:
|
|
dict or None: Dictionary with icon data:
|
|
{
|
|
'filename': 'icon-name.svg', # Icon filename
|
|
'alt': 'Alt text', # Alt attribute
|
|
'title': 'Tooltip text', # Title attribute
|
|
'style': 'height: 1em;' # Optional inline CSS
|
|
}
|
|
Or None if no icon
|
|
"""
|
|
return None
|
|
|
|
def clear_content(self):
|
|
"""
|
|
Explicitly clear all content from memory to free up heap space.
|
|
Call this after content has been saved to disk.
|
|
"""
|
|
self.content = None
|
|
if hasattr(self, 'raw_content'):
|
|
self.raw_content = None
|
|
self.screenshot = None
|
|
self.xpath_data = None
|
|
# Keep headers and status_code as they're small
|
|
|
|
@abstractmethod
|
|
def get_error(self):
|
|
return self.error
|
|
|
|
@abstractmethod
|
|
async def run(self,
|
|
fetch_favicon=True,
|
|
current_include_filters=None,
|
|
empty_pages_are_a_change=False,
|
|
ignore_status_codes=False,
|
|
is_binary=False,
|
|
request_body=None,
|
|
request_headers=None,
|
|
request_method=None,
|
|
timeout=None,
|
|
url=None,
|
|
watch_uuid=None,
|
|
):
|
|
# Should set self.error, self.status_code and self.content
|
|
pass
|
|
|
|
@abstractmethod
|
|
async def quit(self, watch=None):
|
|
return
|
|
|
|
@abstractmethod
|
|
def get_last_status_code(self):
|
|
return self.status_code
|
|
|
|
@abstractmethod
|
|
def screenshot_step(self, step_n):
|
|
if self.browser_steps_screenshot_path and not os.path.isdir(self.browser_steps_screenshot_path):
|
|
logger.debug(f"> Creating data dir {self.browser_steps_screenshot_path}")
|
|
os.mkdir(self.browser_steps_screenshot_path)
|
|
return None
|
|
|
|
@abstractmethod
|
|
# Return true/false if this checker is ready to run, in the case it needs todo some special config check etc
|
|
def is_ready(self):
|
|
return True
|
|
|
|
def get_all_headers(self):
|
|
"""
|
|
Get all headers but ensure all keys are lowercase
|
|
:return:
|
|
"""
|
|
return {k.lower(): v for k, v in self.headers.items()}
|
|
|
|
async def iterate_browser_steps(self, start_url=None):
|
|
from changedetectionio.browser_steps.browser_steps import steppable_browser_interface, browser_steps_get_valid_steps
|
|
from playwright._impl._errors import TimeoutError, Error
|
|
from changedetectionio.jinja2_custom import render as jinja_render
|
|
step_n = 0
|
|
|
|
if self.browser_steps:
|
|
interface = steppable_browser_interface(start_url=start_url)
|
|
interface.page = self.page
|
|
valid_steps = browser_steps_get_valid_steps(self.browser_steps)
|
|
|
|
for step in valid_steps:
|
|
step_n += 1
|
|
logger.debug(f">> Iterating check - browser Step n {step_n} - {step['operation']}...")
|
|
await self.screenshot_step("before-" + str(step_n))
|
|
await self.save_step_html("before-" + str(step_n))
|
|
|
|
try:
|
|
optional_value = step['optional_value']
|
|
selector = step['selector']
|
|
# Support for jinja2 template in step values, with date module added
|
|
if '{%' in step['optional_value'] or '{{' in step['optional_value']:
|
|
optional_value = jinja_render(template_str=step['optional_value'])
|
|
if '{%' in step['selector'] or '{{' in step['selector']:
|
|
selector = jinja_render(template_str=step['selector'])
|
|
|
|
await getattr(interface, "call_action")(action_name=step['operation'],
|
|
selector=selector,
|
|
optional_value=optional_value)
|
|
await self.screenshot_step(step_n)
|
|
await self.save_step_html(step_n)
|
|
except (Error, TimeoutError, ValueError) as e:
|
|
# ValueError is what validate_fetch_url_async() raises when a step's URL is
|
|
# refused (file://, private IP, bad scheme) - report it against the offending
|
|
# step number like any other step failure, rather than failing the whole watch
|
|
# with an opaque error.
|
|
logger.debug(str(e))
|
|
# Stop processing here
|
|
raise BrowserStepsStepException(step_n=step_n, original_e=e)
|
|
|
|
# It's always good to reset these
|
|
def delete_browser_steps_screenshots(self):
|
|
import glob
|
|
if self.browser_steps_screenshot_path is not None:
|
|
dest = os.path.join(self.browser_steps_screenshot_path, 'step_*.jpeg')
|
|
files = glob.glob(dest)
|
|
for f in files:
|
|
if os.path.isfile(f):
|
|
os.unlink(f)
|
|
|
|
def save_step_html(self, step_n):
|
|
if self.browser_steps_screenshot_path and not os.path.isdir(self.browser_steps_screenshot_path):
|
|
logger.debug(f"> Creating data dir {self.browser_steps_screenshot_path}")
|
|
os.mkdir(self.browser_steps_screenshot_path)
|
|
pass
|