mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-10-04 19:18:04 +00:00
An "extra browser" was a name + a ws(s):// endpoint in settings.requests.extra_browsers,
selected by a watch as the magic string 'extra_browser_<name>'. That string resolved to
html_webdriver plus a custom connection URL, which meant the protocol the endpoint was
spoken to came from env vars rather than from the entry: CDP over a WebSocket with
PLAYWRIGHT_DRIVER_URL set, CDP via pyppeteer with FAST_PUPPETEER_CHROME_FETCHER, and the
W3C WebDriver protocol over HTTP on a Selenium-only install - where a wss:// URL cannot
work at all. The form only ever accepted ws:// / wss://, so the feature was silently
broken on exactly the installs that could not honour it.
So it becomes an engine, html_external_cdp, which pins the protocol: a subclass of the
Playwright fetcher that takes its endpoint from the watch's browser config
(FetcherConfig.connection_url) instead of the environment. It is base-only
(ready_to_use=False) because an endpoint is required, so each endpoint is one browser
config ("variation") on the Browsers page - which is what the old settings list was.
update_36 migrates each extra_browsers row to such a variation, keyed by the SAME
'extra_browser_<name>' string watches already hold, so no watch, group override, API value
or global default needs rewriting; the legacy selector simply becomes a real browser-config
id. A row whose endpoint the model rejects is logged and skipped rather than taking the
update chain, and with it startup, down.
Knock-on cleanups, all of which delete a special case rather than add one:
- The proxy opt-out for custom endpoints is now Fetcher.ignores_proxy_setting, asked of
the engine, instead of a string-prefix test in call_browser().
- A live browser-steps / visual-selector session asks the engine where to connect
(Fetcher.browser_steps_connection_url, overridden by html_external_cdp) and refuses an
engine whose supports_browser_steps is False, instead of reading the env var itself and
silently stepping a browser the watch does not check with. That refusal is real: on a
Selenium install html_webdriver cannot drive a live session.
- is_valid_browser_selector() answers "may a watch store this in fetch_backend?" in one
place; the API (create/update/import), the quick-add form validator and the bulk "set
browser" operation each had their own copy, which is how they came to disagree about
whether a browser-config id was acceptable.
- api-spec.yaml's fetch_backend pattern enumerated extra_browser_* while rejecting
browser-config ids and every engine newer than html_webdriver. Valid values are
per-install, so the schema now bounds the string and the handlers do the real check.
- html_external_cdp registers unconditionally (unlike html_playwright_builtin): migrated
configs name it, so it must resolve even without the playwright library, or those
watches would quietly fetch with the plain HTTP client. The library is imported lazily
inside run(), and an unavailable engine now warns instead of falling back silently.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
203 lines
10 KiB
Python
203 lines
10 KiB
Python
import sys
|
||
from changedetectionio.strtobool import strtobool
|
||
from loguru import logger
|
||
from changedetectionio.content_fetchers.exceptions import BrowserStepsStepException
|
||
import os
|
||
|
||
# Visual Selector scraper - 'Button' is there because some sites have <button>OUT OF STOCK</button>.
|
||
visualselector_xpath_selectors = 'div,span,form,table,tbody,tr,td,a,p,ul,li,h1,h2,h3,h4,header,footer,section,article,aside,details,main,nav,section,summary,button'
|
||
|
||
# Import hookimpl from centralized pluggy interface
|
||
from changedetectionio.pluggy_interface import hookimpl
|
||
|
||
SCREENSHOT_MAX_HEIGHT_DEFAULT = 20000
|
||
SCREENSHOT_DEFAULT_QUALITY = 40
|
||
|
||
# Maximum total height for the final image (When in stitch mode).
|
||
# We limit this to 16000px due to the huge amount of RAM that was being used
|
||
# Example: 16000 × 1400 × 3 = 67,200,000 bytes ≈ 64.1 MB (not including buffers in PIL etc)
|
||
SCREENSHOT_MAX_TOTAL_HEIGHT = int(os.getenv("SCREENSHOT_MAX_HEIGHT", SCREENSHOT_MAX_HEIGHT_DEFAULT))
|
||
|
||
# The size at which we will switch to stitching method, when below this (and
|
||
# MAX_TOTAL_HEIGHT which can be set by a user) we will use the default
|
||
# screenshot method.
|
||
# Increased from 8000 to 10000 for better performance (fewer chunks = faster)
|
||
# Most modern GPUs support 16384x16384 textures, so 1280x10000 is safe
|
||
SCREENSHOT_SIZE_STITCH_THRESHOLD = int(os.getenv("SCREENSHOT_CHUNK_HEIGHT", 10000))
|
||
|
||
# available_fetchers() will scan this implementation looking for anything starting with html_
|
||
# this information is used in the form selections
|
||
from changedetectionio.content_fetchers.requests import fetcher as html_requests
|
||
|
||
|
||
import importlib.resources
|
||
XPATH_ELEMENT_JS = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('xpath_element_scraper.js').read_text(encoding='utf-8')
|
||
INSTOCK_DATA_JS = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('stock-not-in-stock.js').read_text(encoding='utf-8')
|
||
FAVICON_FETCHER_JS = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('favicon-fetcher.js').read_text(encoding='utf-8')
|
||
|
||
|
||
def available_fetchers():
|
||
# See the if statement at the bottom of this file for how we switch between playwright and webdriver
|
||
import inspect
|
||
p = []
|
||
|
||
# Get built-in fetchers (but skip plugin fetchers that were added via setattr)
|
||
for name, obj in inspect.getmembers(sys.modules[__name__], inspect.isclass):
|
||
if inspect.isclass(obj):
|
||
# @todo html_ is maybe better as fetcher_ or something
|
||
# In this case, make sure to edit the default one in store.py and fetch_site_status.py
|
||
if name.startswith('html_'):
|
||
# Skip plugin fetchers that were already registered
|
||
if name not in _plugin_fetchers:
|
||
t = tuple([name, obj.fetcher_description])
|
||
p.append(t)
|
||
|
||
# Get plugin fetchers from cache (already loaded at module init)
|
||
for name, fetcher_class in _plugin_fetchers.items():
|
||
if hasattr(fetcher_class, 'fetcher_description'):
|
||
t = tuple([name, fetcher_class.fetcher_description])
|
||
p.append(t)
|
||
else:
|
||
logger.warning(f"Plugin fetcher '{name}' does not have fetcher_description attribute")
|
||
|
||
return p
|
||
|
||
|
||
def get_plugin_fetchers():
|
||
"""Load and return all plugin fetchers from the centralized plugin manager."""
|
||
from changedetectionio.pluggy_interface import plugin_manager
|
||
|
||
fetchers = {}
|
||
try:
|
||
# Call the register_content_fetcher hook from all registered plugins
|
||
results = plugin_manager.hook.register_content_fetcher()
|
||
for result in results:
|
||
if result:
|
||
name, fetcher_class = result
|
||
fetchers[name] = fetcher_class
|
||
# Register in current module so hasattr() checks work
|
||
setattr(sys.modules[__name__], name, fetcher_class)
|
||
logger.info(f"Registered plugin fetcher: {name} - {getattr(fetcher_class, 'fetcher_description', 'No description')}")
|
||
except Exception as e:
|
||
logger.error(f"Error loading plugin fetchers: {e}")
|
||
|
||
return fetchers
|
||
|
||
|
||
# Initialize plugins at module load time
|
||
_plugin_fetchers = get_plugin_fetchers()
|
||
|
||
|
||
def _log_fetcher_capabilities(fetcher_class, backend_name, uuid=None):
|
||
"""logger.info the capabilities of a resolved content fetcher class.
|
||
|
||
Returns the FetcherCapabilities instance so callers can reuse it.
|
||
"""
|
||
from changedetectionio.content_fetchers.base import FetcherCapabilities
|
||
caps = FetcherCapabilities.from_fetcher(fetcher_class)
|
||
logger.info(
|
||
f"Content fetcher '{backend_name}' ({getattr(fetcher_class, 'fetcher_description', 'No description')})"
|
||
f"{f' for watch {uuid}' if uuid else ''} - capabilities: "
|
||
f"browser_steps={caps.supports_browser_steps}, "
|
||
f"screenshots={caps.supports_screenshots}, "
|
||
f"xpath_element_data={caps.supports_xpath_element_data}"
|
||
)
|
||
return caps
|
||
|
||
|
||
def resolve_content_fetcher(watch, datastore):
|
||
"""Build the concrete content-fetcher class + config for a watch.
|
||
|
||
*Which* browser/engine is selected is owned by the Watch model (watch.get_fetch_backend:
|
||
PDF / group override / watch / 'system' -> global Default browser). This function takes that
|
||
resolved selector and turns it into a fetcher instance, collapsing the special forms:
|
||
- a user browser-config id -> its base_fetcher engine + its FetcherConfig
|
||
- html_webdriver + browser_steps -> playwright override (puppeteer steps incomplete)
|
||
- a deleted browser-config id -> raises BrowserConfigDoesntExist
|
||
|
||
Returns:
|
||
tuple: (fetcher_class, backend_name, custom_browser_connection_url, browser_config)
|
||
where `backend_name` is the fully-resolved concrete backend name the caller should
|
||
stamp onto the fetcher instance as `.backend_name`, and `browser_config` is the
|
||
resolved FetcherConfig to inject as `.browser_config`.
|
||
"""
|
||
this_module = sys.modules[__name__]
|
||
from changedetectionio.model.browser_config import FetcherConfig, BrowserConfigDoesntExist
|
||
|
||
# Default behaviour = empty config (built-in engines / system default).
|
||
browser_config = FetcherConfig()
|
||
|
||
# THE single resolved selector for this watch (PDF / group override / watch / 'system' ->
|
||
# global default) - the Watch owns this chain so every codepath agrees. The value is a
|
||
# built-in engine name or the stable id of a user browser config.
|
||
selected = watch.get_fetch_backend
|
||
|
||
store = getattr(datastore, 'browser_config_store', None)
|
||
if store is not None:
|
||
entry, prefer_fetch_backend, browser_config = store.engine_and_config(selected)
|
||
else:
|
||
entry, prefer_fetch_backend = None, selected
|
||
if entry is None:
|
||
# Not a stored browser config, so the only valid value left is a built-in engine name.
|
||
# Anything else is a reference to a browser config that has been deleted - fail loudly
|
||
# instead of silently defaulting.
|
||
if selected and selected != 'system' and not hasattr(this_module, selected):
|
||
raise BrowserConfigDoesntExist(config_id=selected, uuid=watch.get('uuid'))
|
||
prefer_fetch_backend = selected
|
||
|
||
# An external browser's endpoint lives on its browser config (FetcherConfig.connection_url,
|
||
# read at connect time by html_external_cdp), not in a per-fetcher constructor argument -
|
||
# which is what the old 'extra_browser_<name>' selector needed this hook for.
|
||
custom_browser_connection_url = None
|
||
|
||
# PDF watches are already forced to 'html_requests' by Watch.get_fetch_backend (playwright
|
||
# can't render a PDF in-page yet - @todo https://github.com/dgtlmoon/changedetection.io/issues/2019),
|
||
# so no extra handling is needed here.
|
||
|
||
# Grab the right kind of 'fetcher' class (playwright, requests, plugin-provided, etc)
|
||
if prefer_fetch_backend and hasattr(this_module, prefer_fetch_backend):
|
||
# @todo TEMPORARY HACK - SWITCH BACK TO PLAYWRIGHT FOR BROWSERSTEPS
|
||
if prefer_fetch_backend == 'html_webdriver' and getattr(watch, 'has_browser_steps', False):
|
||
# This is never supported in selenium anyway
|
||
logger.warning(
|
||
"Using playwright fetcher override for possible puppeteer request in browsersteps, "
|
||
"because puppetteer:browser steps is incomplete.")
|
||
from changedetectionio.content_fetchers.playwright import fetcher as playwright_fetcher
|
||
fetcher_obj = playwright_fetcher
|
||
else:
|
||
fetcher_obj = getattr(this_module, prefer_fetch_backend)
|
||
else:
|
||
# What it referenced doesn't exist. Falling back to the plain client keeps the check
|
||
# running, but it fetches with something other than what the watch asked for - so say so
|
||
# rather than letting a browser watch quietly become a plaintext one.
|
||
logger.warning(f"Fetcher '{prefer_fetch_backend}' is not available in this install - "
|
||
f"falling back to html_requests for watch {watch.get('uuid')}")
|
||
fetcher_obj = getattr(this_module, "html_requests")
|
||
|
||
_log_fetcher_capabilities(fetcher_obj, prefer_fetch_backend, uuid=watch.get('uuid'))
|
||
|
||
return fetcher_obj, prefer_fetch_backend, custom_browser_connection_url, browser_config
|
||
|
||
|
||
# Decide which is the 'real' HTML webdriver, this is more a system wide config
|
||
# rather than site-specific.
|
||
use_playwright_as_chrome_fetcher = os.getenv('PLAYWRIGHT_DRIVER_URL', False)
|
||
if use_playwright_as_chrome_fetcher:
|
||
# @note - For now, browser steps always uses playwright
|
||
if not strtobool(os.getenv('FAST_PUPPETEER_CHROME_FETCHER', 'False')):
|
||
logger.debug('Using Playwright library as fetcher')
|
||
from .playwright import fetcher as html_webdriver
|
||
else:
|
||
logger.debug('Using direct Python Puppeteer library as fetcher')
|
||
from .puppeteer import fetcher as html_webdriver
|
||
|
||
else:
|
||
logger.debug("Falling back to selenium as fetcher")
|
||
from .webdriver_selenium import fetcher as html_webdriver
|
||
|
||
|
||
# Register built-in fetchers as plugins after all imports are complete
|
||
from changedetectionio.pluggy_interface import register_builtin_fetchers
|
||
register_builtin_fetchers()
|
||
|