Files
changedetection.io/changedetectionio/content_fetchers/__init__.py
T
dgtlmoonandClaude Opus 5 ba2280ba81 External CDP Browser engine - replaces the extra_browser_* hackery
An "extra browser" was a name + a ws(s):// endpoint in settings.requests.extra_browsers,
selected by a watch as the magic string 'extra_browser_<name>'. That string resolved to
html_webdriver plus a custom connection URL, which meant the protocol the endpoint was
spoken to came from env vars rather than from the entry: CDP over a WebSocket with
PLAYWRIGHT_DRIVER_URL set, CDP via pyppeteer with FAST_PUPPETEER_CHROME_FETCHER, and the
W3C WebDriver protocol over HTTP on a Selenium-only install - where a wss:// URL cannot
work at all. The form only ever accepted ws:// / wss://, so the feature was silently
broken on exactly the installs that could not honour it.

So it becomes an engine, html_external_cdp, which pins the protocol: a subclass of the
Playwright fetcher that takes its endpoint from the watch's browser config
(FetcherConfig.connection_url) instead of the environment. It is base-only
(ready_to_use=False) because an endpoint is required, so each endpoint is one browser
config ("variation") on the Browsers page - which is what the old settings list was.

update_36 migrates each extra_browsers row to such a variation, keyed by the SAME
'extra_browser_<name>' string watches already hold, so no watch, group override, API value
or global default needs rewriting; the legacy selector simply becomes a real browser-config
id. A row whose endpoint the model rejects is logged and skipped rather than taking the
update chain, and with it startup, down.

Knock-on cleanups, all of which delete a special case rather than add one:

 - The proxy opt-out for custom endpoints is now Fetcher.ignores_proxy_setting, asked of
   the engine, instead of a string-prefix test in call_browser().
 - A live browser-steps / visual-selector session asks the engine where to connect
   (Fetcher.browser_steps_connection_url, overridden by html_external_cdp) and refuses an
   engine whose supports_browser_steps is False, instead of reading the env var itself and
   silently stepping a browser the watch does not check with. That refusal is real: on a
   Selenium install html_webdriver cannot drive a live session.
 - is_valid_browser_selector() answers "may a watch store this in fetch_backend?" in one
   place; the API (create/update/import), the quick-add form validator and the bulk "set
   browser" operation each had their own copy, which is how they came to disagree about
   whether a browser-config id was acceptable.
 - api-spec.yaml's fetch_backend pattern enumerated extra_browser_* while rejecting
   browser-config ids and every engine newer than html_webdriver. Valid values are
   per-install, so the schema now bounds the string and the handlers do the real check.
 - html_external_cdp registers unconditionally (unlike html_playwright_builtin): migrated
   configs name it, so it must resolve even without the playwright library, or those
   watches would quietly fetch with the plain HTTP client. The library is imported lazily
   inside run(), and an unavailable engine now warns instead of falling back silently.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 19:39:16 +02:00

203 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import sys
from changedetectionio.strtobool import strtobool
from loguru import logger
from changedetectionio.content_fetchers.exceptions import BrowserStepsStepException
import os
# Visual Selector scraper - 'Button' is there because some sites have <button>OUT OF STOCK</button>.
visualselector_xpath_selectors = 'div,span,form,table,tbody,tr,td,a,p,ul,li,h1,h2,h3,h4,header,footer,section,article,aside,details,main,nav,section,summary,button'
# Import hookimpl from centralized pluggy interface
from changedetectionio.pluggy_interface import hookimpl
SCREENSHOT_MAX_HEIGHT_DEFAULT = 20000
SCREENSHOT_DEFAULT_QUALITY = 40
# Maximum total height for the final image (When in stitch mode).
# We limit this to 16000px due to the huge amount of RAM that was being used
# Example: 16000 × 1400 × 3 = 67,200,000 bytes ≈ 64.1 MB (not including buffers in PIL etc)
SCREENSHOT_MAX_TOTAL_HEIGHT = int(os.getenv("SCREENSHOT_MAX_HEIGHT", SCREENSHOT_MAX_HEIGHT_DEFAULT))
# The size at which we will switch to stitching method, when below this (and
# MAX_TOTAL_HEIGHT which can be set by a user) we will use the default
# screenshot method.
# Increased from 8000 to 10000 for better performance (fewer chunks = faster)
# Most modern GPUs support 16384x16384 textures, so 1280x10000 is safe
SCREENSHOT_SIZE_STITCH_THRESHOLD = int(os.getenv("SCREENSHOT_CHUNK_HEIGHT", 10000))
# available_fetchers() will scan this implementation looking for anything starting with html_
# this information is used in the form selections
from changedetectionio.content_fetchers.requests import fetcher as html_requests
import importlib.resources
XPATH_ELEMENT_JS = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('xpath_element_scraper.js').read_text(encoding='utf-8')
INSTOCK_DATA_JS = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('stock-not-in-stock.js').read_text(encoding='utf-8')
FAVICON_FETCHER_JS = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('favicon-fetcher.js').read_text(encoding='utf-8')
def available_fetchers():
# See the if statement at the bottom of this file for how we switch between playwright and webdriver
import inspect
p = []
# Get built-in fetchers (but skip plugin fetchers that were added via setattr)
for name, obj in inspect.getmembers(sys.modules[__name__], inspect.isclass):
if inspect.isclass(obj):
# @todo html_ is maybe better as fetcher_ or something
# In this case, make sure to edit the default one in store.py and fetch_site_status.py
if name.startswith('html_'):
# Skip plugin fetchers that were already registered
if name not in _plugin_fetchers:
t = tuple([name, obj.fetcher_description])
p.append(t)
# Get plugin fetchers from cache (already loaded at module init)
for name, fetcher_class in _plugin_fetchers.items():
if hasattr(fetcher_class, 'fetcher_description'):
t = tuple([name, fetcher_class.fetcher_description])
p.append(t)
else:
logger.warning(f"Plugin fetcher '{name}' does not have fetcher_description attribute")
return p
def get_plugin_fetchers():
"""Load and return all plugin fetchers from the centralized plugin manager."""
from changedetectionio.pluggy_interface import plugin_manager
fetchers = {}
try:
# Call the register_content_fetcher hook from all registered plugins
results = plugin_manager.hook.register_content_fetcher()
for result in results:
if result:
name, fetcher_class = result
fetchers[name] = fetcher_class
# Register in current module so hasattr() checks work
setattr(sys.modules[__name__], name, fetcher_class)
logger.info(f"Registered plugin fetcher: {name} - {getattr(fetcher_class, 'fetcher_description', 'No description')}")
except Exception as e:
logger.error(f"Error loading plugin fetchers: {e}")
return fetchers
# Initialize plugins at module load time
_plugin_fetchers = get_plugin_fetchers()
def _log_fetcher_capabilities(fetcher_class, backend_name, uuid=None):
"""logger.info the capabilities of a resolved content fetcher class.
Returns the FetcherCapabilities instance so callers can reuse it.
"""
from changedetectionio.content_fetchers.base import FetcherCapabilities
caps = FetcherCapabilities.from_fetcher(fetcher_class)
logger.info(
f"Content fetcher '{backend_name}' ({getattr(fetcher_class, 'fetcher_description', 'No description')})"
f"{f' for watch {uuid}' if uuid else ''} - capabilities: "
f"browser_steps={caps.supports_browser_steps}, "
f"screenshots={caps.supports_screenshots}, "
f"xpath_element_data={caps.supports_xpath_element_data}"
)
return caps
def resolve_content_fetcher(watch, datastore):
"""Build the concrete content-fetcher class + config for a watch.
*Which* browser/engine is selected is owned by the Watch model (watch.get_fetch_backend:
PDF / group override / watch / 'system' -> global Default browser). This function takes that
resolved selector and turns it into a fetcher instance, collapsing the special forms:
- a user browser-config id -> its base_fetcher engine + its FetcherConfig
- html_webdriver + browser_steps -> playwright override (puppeteer steps incomplete)
- a deleted browser-config id -> raises BrowserConfigDoesntExist
Returns:
tuple: (fetcher_class, backend_name, custom_browser_connection_url, browser_config)
where `backend_name` is the fully-resolved concrete backend name the caller should
stamp onto the fetcher instance as `.backend_name`, and `browser_config` is the
resolved FetcherConfig to inject as `.browser_config`.
"""
this_module = sys.modules[__name__]
from changedetectionio.model.browser_config import FetcherConfig, BrowserConfigDoesntExist
# Default behaviour = empty config (built-in engines / system default).
browser_config = FetcherConfig()
# THE single resolved selector for this watch (PDF / group override / watch / 'system' ->
# global default) - the Watch owns this chain so every codepath agrees. The value is a
# built-in engine name or the stable id of a user browser config.
selected = watch.get_fetch_backend
store = getattr(datastore, 'browser_config_store', None)
if store is not None:
entry, prefer_fetch_backend, browser_config = store.engine_and_config(selected)
else:
entry, prefer_fetch_backend = None, selected
if entry is None:
# Not a stored browser config, so the only valid value left is a built-in engine name.
# Anything else is a reference to a browser config that has been deleted - fail loudly
# instead of silently defaulting.
if selected and selected != 'system' and not hasattr(this_module, selected):
raise BrowserConfigDoesntExist(config_id=selected, uuid=watch.get('uuid'))
prefer_fetch_backend = selected
# An external browser's endpoint lives on its browser config (FetcherConfig.connection_url,
# read at connect time by html_external_cdp), not in a per-fetcher constructor argument -
# which is what the old 'extra_browser_<name>' selector needed this hook for.
custom_browser_connection_url = None
# PDF watches are already forced to 'html_requests' by Watch.get_fetch_backend (playwright
# can't render a PDF in-page yet - @todo https://github.com/dgtlmoon/changedetection.io/issues/2019),
# so no extra handling is needed here.
# Grab the right kind of 'fetcher' class (playwright, requests, plugin-provided, etc)
if prefer_fetch_backend and hasattr(this_module, prefer_fetch_backend):
# @todo TEMPORARY HACK - SWITCH BACK TO PLAYWRIGHT FOR BROWSERSTEPS
if prefer_fetch_backend == 'html_webdriver' and getattr(watch, 'has_browser_steps', False):
# This is never supported in selenium anyway
logger.warning(
"Using playwright fetcher override for possible puppeteer request in browsersteps, "
"because puppetteer:browser steps is incomplete.")
from changedetectionio.content_fetchers.playwright import fetcher as playwright_fetcher
fetcher_obj = playwright_fetcher
else:
fetcher_obj = getattr(this_module, prefer_fetch_backend)
else:
# What it referenced doesn't exist. Falling back to the plain client keeps the check
# running, but it fetches with something other than what the watch asked for - so say so
# rather than letting a browser watch quietly become a plaintext one.
logger.warning(f"Fetcher '{prefer_fetch_backend}' is not available in this install - "
f"falling back to html_requests for watch {watch.get('uuid')}")
fetcher_obj = getattr(this_module, "html_requests")
_log_fetcher_capabilities(fetcher_obj, prefer_fetch_backend, uuid=watch.get('uuid'))
return fetcher_obj, prefer_fetch_backend, custom_browser_connection_url, browser_config
# Decide which is the 'real' HTML webdriver, this is more a system wide config
# rather than site-specific.
use_playwright_as_chrome_fetcher = os.getenv('PLAYWRIGHT_DRIVER_URL', False)
if use_playwright_as_chrome_fetcher:
# @note - For now, browser steps always uses playwright
if not strtobool(os.getenv('FAST_PUPPETEER_CHROME_FETCHER', 'False')):
logger.debug('Using Playwright library as fetcher')
from .playwright import fetcher as html_webdriver
else:
logger.debug('Using direct Python Puppeteer library as fetcher')
from .puppeteer import fetcher as html_webdriver
else:
logger.debug("Falling back to selenium as fetcher")
from .webdriver_selenium import fetcher as html_webdriver
# Register built-in fetchers as plugins after all imports are complete
from changedetectionio.pluggy_interface import register_builtin_fetchers
register_builtin_fetchers()