mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-10-05 03:28:07 +00:00
An "extra browser" was a name + a ws(s):// endpoint in settings.requests.extra_browsers,
selected by a watch as the magic string 'extra_browser_<name>'. That string resolved to
html_webdriver plus a custom connection URL, which meant the protocol the endpoint was
spoken to came from env vars rather than from the entry: CDP over a WebSocket with
PLAYWRIGHT_DRIVER_URL set, CDP via pyppeteer with FAST_PUPPETEER_CHROME_FETCHER, and the
W3C WebDriver protocol over HTTP on a Selenium-only install - where a wss:// URL cannot
work at all. The form only ever accepted ws:// / wss://, so the feature was silently
broken on exactly the installs that could not honour it.
So it becomes an engine, html_external_cdp, which pins the protocol: a subclass of the
Playwright fetcher that takes its endpoint from the watch's browser config
(FetcherConfig.connection_url) instead of the environment. It is base-only
(ready_to_use=False) because an endpoint is required, so each endpoint is one browser
config ("variation") on the Browsers page - which is what the old settings list was.
update_36 migrates each extra_browsers row to such a variation, keyed by the SAME
'extra_browser_<name>' string watches already hold, so no watch, group override, API value
or global default needs rewriting; the legacy selector simply becomes a real browser-config
id. A row whose endpoint the model rejects is logged and skipped rather than taking the
update chain, and with it startup, down.
Knock-on cleanups, all of which delete a special case rather than add one:
- The proxy opt-out for custom endpoints is now Fetcher.ignores_proxy_setting, asked of
the engine, instead of a string-prefix test in call_browser().
- A live browser-steps / visual-selector session asks the engine where to connect
(Fetcher.browser_steps_connection_url, overridden by html_external_cdp) and refuses an
engine whose supports_browser_steps is False, instead of reading the env var itself and
silently stepping a browser the watch does not check with. That refusal is real: on a
Selenium install html_webdriver cannot drive a live session.
- is_valid_browser_selector() answers "may a watch store this in fetch_backend?" in one
place; the API (create/update/import), the quick-add form validator and the bulk "set
browser" operation each had their own copy, which is how they came to disagree about
whether a browser-config id was acceptable.
- api-spec.yaml's fetch_backend pattern enumerated extra_browser_* while rejecting
browser-config ids and every engine newer than html_webdriver. Valid values are
per-install, so the schema now bounds the string and the handlers do the real check.
- html_external_cdp registers unconditionally (unlike html_playwright_builtin): migrated
configs name it, so it must resolve even without the playwright library, or those
watches would quietly fetch with the plain HTTP client. The library is imported lazily
inside run(), and an unavailable engine now warns instead of falling back silently.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
444 lines
19 KiB
Python
444 lines
19 KiB
Python
|
|
# HORRIBLE HACK BUT WORKS :-) PR anyone?
|
|
#
|
|
# Why?
|
|
# `browsersteps_playwright_browser_interface.chromium.connect_over_cdp()` will only run once without async()
|
|
# - this flask app is not async()
|
|
# - A single timeout/keepalive which applies to the session made at .connect_over_cdp()
|
|
#
|
|
# So it means that we must unfortunately for now just keep a single timer since .connect_over_cdp() was run
|
|
# and know when that reaches timeout/keepalive :( when that time is up, restart the connection and tell the user
|
|
# that their time is up, insert another coin. (reload)
|
|
#
|
|
#
|
|
|
|
from changedetectionio.strtobool import strtobool
|
|
from flask import Blueprint, request, make_response
|
|
import os
|
|
|
|
from changedetectionio.store import ChangeDetectionStore
|
|
from changedetectionio.blueprint import plaintext_response
|
|
from changedetectionio.flask_app import login_optionally_required
|
|
from changedetectionio.content_fetchers.exceptions import BrowserConnectError
|
|
from changedetectionio.validate_url import validate_fetch_url_async
|
|
from loguru import logger
|
|
|
|
browsersteps_sessions = {}
|
|
browsersteps_watch_to_session = {} # Maps watch_uuid -> browsersteps_session_id
|
|
io_interface_context = None
|
|
import json
|
|
import hashlib
|
|
from flask import Response
|
|
import asyncio
|
|
import threading
|
|
import time
|
|
|
|
# Dedicated event loop for ALL browser steps sessions
|
|
_browser_steps_loop = None
|
|
_browser_steps_thread = None
|
|
_browser_steps_loop_lock = threading.Lock()
|
|
|
|
def _start_browser_steps_loop():
|
|
"""Start a dedicated event loop for browser steps in its own thread"""
|
|
global _browser_steps_loop
|
|
|
|
# Create and set the event loop for this thread
|
|
loop = asyncio.new_event_loop()
|
|
asyncio.set_event_loop(loop)
|
|
_browser_steps_loop = loop
|
|
|
|
logger.debug("Browser steps event loop started")
|
|
|
|
try:
|
|
# Run the loop forever - handles all browsersteps sessions
|
|
loop.run_forever()
|
|
except Exception as e:
|
|
logger.error(f"Browser steps event loop error: {e}")
|
|
finally:
|
|
try:
|
|
# Cancel all remaining tasks
|
|
pending = asyncio.all_tasks(loop)
|
|
for task in pending:
|
|
task.cancel()
|
|
|
|
# Wait for tasks to finish cancellation
|
|
if pending:
|
|
loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
|
|
except Exception as e:
|
|
logger.debug(f"Error during browser steps loop cleanup: {e}")
|
|
finally:
|
|
loop.close()
|
|
logger.debug("Browser steps event loop closed")
|
|
|
|
def _ensure_browser_steps_loop():
|
|
"""Ensure the browser steps event loop is running"""
|
|
global _browser_steps_loop, _browser_steps_thread
|
|
|
|
with _browser_steps_loop_lock:
|
|
if _browser_steps_thread is None or not _browser_steps_thread.is_alive():
|
|
logger.debug("Starting browser steps event loop thread")
|
|
_browser_steps_thread = threading.Thread(
|
|
target=_start_browser_steps_loop,
|
|
daemon=True,
|
|
name="BrowserStepsEventLoop"
|
|
)
|
|
_browser_steps_thread.start()
|
|
|
|
# Wait for the loop to be ready
|
|
timeout = 5.0
|
|
start_time = time.time()
|
|
while _browser_steps_loop is None:
|
|
if time.time() - start_time > timeout:
|
|
raise RuntimeError("Browser steps event loop failed to start")
|
|
time.sleep(0.01)
|
|
|
|
logger.debug("Browser steps event loop thread started and ready")
|
|
|
|
def run_async_in_browser_loop(coro):
|
|
"""Run async coroutine using the dedicated browser steps event loop"""
|
|
_ensure_browser_steps_loop()
|
|
|
|
if _browser_steps_loop and not _browser_steps_loop.is_closed():
|
|
logger.debug("Browser steps using dedicated event loop")
|
|
future = asyncio.run_coroutine_threadsafe(coro, _browser_steps_loop)
|
|
return future.result()
|
|
else:
|
|
raise RuntimeError("Browser steps event loop is not available")
|
|
|
|
async def _close_session_resources(session_data, label=''):
|
|
"""Close all browser resources for a session in the correct order.
|
|
|
|
browserstepper.cleanup() closes page+context but not the browser itself.
|
|
For CloakBrowser, browser.close() is what stops the local Chromium process via pw.stop().
|
|
For the default CDP path, playwright_context.stop() shuts down the playwright instance.
|
|
"""
|
|
browserstepper = session_data.get('browserstepper')
|
|
if browserstepper:
|
|
try:
|
|
await browserstepper.cleanup()
|
|
except Exception as e:
|
|
logger.error(f"Error cleaning up browserstepper{label}: {e}")
|
|
|
|
browser = session_data.get('browser')
|
|
if browser:
|
|
try:
|
|
await asyncio.wait_for(browser.close(), timeout=5.0)
|
|
except Exception as e:
|
|
logger.warning(f"Error closing browser{label}: {e}")
|
|
|
|
playwright_context = session_data.get('playwright_context')
|
|
if playwright_context:
|
|
try:
|
|
await playwright_context.stop()
|
|
except Exception as e:
|
|
logger.warning(f"Error stopping playwright context{label}: {e}")
|
|
|
|
|
|
async def acquire_browser_for_fetcher(fetcher_name, proxy=None, keepalive_ms=None, browser_config=None):
|
|
"""The live Playwright browser to step/preview with, from the engine the watch fetches with.
|
|
|
|
The engine decides how: it either launches its own (get_browsersteps_browser) or names the
|
|
CDP endpoint to connect to (browser_steps_connection_url). An engine that cannot drive a live
|
|
session says so via supports_browser_steps and is refused - previewing a watch with a browser
|
|
it does not check with would show the operator something untrue.
|
|
"""
|
|
from changedetectionio import content_fetchers
|
|
from changedetectionio.content_fetchers.base import FetcherCapabilities
|
|
from playwright.async_api import async_playwright
|
|
|
|
fetcher_class = getattr(content_fetchers, fetcher_name, None) if fetcher_name else None
|
|
if not FetcherCapabilities.from_fetcher(fetcher_class).supports_browser_steps:
|
|
raise BrowserConnectError(msg=f"'{fetcher_name}' cannot drive a live browser session "
|
|
f"(browser steps / visual selector) - choose a browser that can.")
|
|
|
|
# Engines that run their own browser locally hand one back themselves.
|
|
if hasattr(fetcher_class, 'get_browsersteps_browser'):
|
|
result = await fetcher_class.get_browsersteps_browser(proxy=proxy, keepalive_ms=keepalive_ms)
|
|
if result is not None:
|
|
logger.debug(f"acquire_browser_for_fetcher: '{fetcher_name}' supplied its own browser")
|
|
return result
|
|
|
|
base_url = fetcher_class.browser_steps_connection_url(browser_config)
|
|
logger.debug(f"acquire_browser_for_fetcher: connecting over CDP for '{fetcher_name}'")
|
|
playwright_context = await async_playwright().start()
|
|
a = "?" if '?' not in base_url else '&'
|
|
browser = await playwright_context.chromium.connect_over_cdp(base_url + a + f"timeout={keepalive_ms}",
|
|
timeout=keepalive_ms)
|
|
return browser, playwright_context
|
|
|
|
|
|
def cleanup_expired_sessions():
|
|
"""Remove expired browsersteps sessions and cleanup their resources"""
|
|
global browsersteps_sessions, browsersteps_watch_to_session
|
|
|
|
expired_session_ids = []
|
|
|
|
# Find expired sessions
|
|
for session_id, session_data in browsersteps_sessions.items():
|
|
browserstepper = session_data.get('browserstepper')
|
|
if browserstepper and browserstepper.has_expired:
|
|
expired_session_ids.append(session_id)
|
|
|
|
# Cleanup expired sessions
|
|
for session_id in expired_session_ids:
|
|
logger.debug(f"Cleaning up expired browsersteps session {session_id}")
|
|
session_data = browsersteps_sessions[session_id]
|
|
|
|
try:
|
|
run_async_in_browser_loop(_close_session_resources(session_data, label=f" for session {session_id}"))
|
|
except Exception as e:
|
|
logger.error(f"Error cleaning up session {session_id}: {e}")
|
|
|
|
# Remove from sessions dict
|
|
del browsersteps_sessions[session_id]
|
|
|
|
# Remove from watch mapping
|
|
for watch_uuid, mapped_session_id in list(browsersteps_watch_to_session.items()):
|
|
if mapped_session_id == session_id:
|
|
del browsersteps_watch_to_session[watch_uuid]
|
|
break
|
|
|
|
if expired_session_ids:
|
|
logger.info(f"Cleaned up {len(expired_session_ids)} expired browsersteps session(s)")
|
|
|
|
def cleanup_session_for_watch(watch_uuid):
|
|
"""Cleanup a specific browsersteps session for a watch UUID"""
|
|
global browsersteps_sessions, browsersteps_watch_to_session
|
|
|
|
session_id = browsersteps_watch_to_session.get(watch_uuid)
|
|
if not session_id:
|
|
logger.debug(f"No browsersteps session found for watch {watch_uuid}")
|
|
return
|
|
|
|
logger.debug(f"Cleaning up browsersteps session {session_id} for watch {watch_uuid}")
|
|
|
|
session_data = browsersteps_sessions.get(session_id)
|
|
if session_data:
|
|
try:
|
|
run_async_in_browser_loop(_close_session_resources(session_data, label=f" for watch {watch_uuid}"))
|
|
except Exception as e:
|
|
logger.error(f"Error cleaning up session {session_id} for watch {watch_uuid}: {e}")
|
|
|
|
# Remove from sessions dict
|
|
del browsersteps_sessions[session_id]
|
|
|
|
# Remove from watch mapping
|
|
del browsersteps_watch_to_session[watch_uuid]
|
|
|
|
logger.debug(f"Cleaned up session for watch {watch_uuid}")
|
|
|
|
# Opportunistically cleanup any other expired sessions
|
|
cleanup_expired_sessions()
|
|
|
|
def construct_blueprint(datastore: ChangeDetectionStore):
|
|
browser_steps_blueprint = Blueprint('browser_steps', __name__, template_folder="templates")
|
|
|
|
async def start_browsersteps_session(watch_uuid):
|
|
from changedetectionio.browser_steps import browser_steps
|
|
import time
|
|
from playwright.async_api import async_playwright
|
|
|
|
keepalive_seconds = int(os.getenv('BROWSERSTEPS_MINUTES_KEEPALIVE', 10)) * 60
|
|
keepalive_ms = ((keepalive_seconds + 3) * 1000)
|
|
|
|
browsersteps_start_session = {'start_time': time.time()}
|
|
|
|
# Build proxy dict first — needed by both the CDP path and fetcher-specific launchers
|
|
proxy_id = datastore.get_preferred_proxy_for_watch(uuid=watch_uuid)
|
|
proxy = None
|
|
if proxy_id:
|
|
proxy_url = datastore.proxy_list.get(proxy_id, {}).get('url')
|
|
if proxy_url:
|
|
from urllib.parse import urlparse
|
|
parsed = urlparse(proxy_url)
|
|
proxy = {'server': proxy_url}
|
|
if parsed.username:
|
|
proxy['username'] = parsed.username
|
|
if parsed.password:
|
|
proxy['password'] = parsed.password
|
|
logger.debug(f"Browser Steps: UUID {watch_uuid} selected proxy {proxy_url}")
|
|
|
|
# Resolve the fetcher backend for this watch so we can ask it to launch its own browser
|
|
# if it supports that (e.g. CloakBrowser, which runs locally rather than via CDP).
|
|
watch = datastore.data['watching'][watch_uuid]
|
|
|
|
# Live preview sessions also return rendered screenshots to the caller and never pass
|
|
# through difference_detection_processor.call_browser(), so validate before we even spend
|
|
# a browser on it - otherwise a watch pointed at a private address is refused at real check
|
|
# time but happily previewed (and exfiltrated) here.
|
|
await validate_fetch_url_async(watch.link)
|
|
|
|
# get_fetch_backend is the fully-resolved SELECTOR (group override / watch / 'system' ->
|
|
# global default) - but it may be a browser-config id, not an engine name. Map it to the
|
|
# concrete engine (for launching) AND its FetcherConfig (viewport/locale/timezone) so the
|
|
# live browser-steps session matches what the watch actually fetches with.
|
|
_entry, fetcher_name, browser_config = datastore.browser_config_store.engine_and_config(watch.get_fetch_backend)
|
|
|
|
browser, playwright_context = await acquire_browser_for_fetcher(
|
|
fetcher_name, proxy=proxy, keepalive_ms=keepalive_ms, browser_config=browser_config)
|
|
|
|
browsersteps_start_session['browser'] = browser
|
|
browsersteps_start_session['playwright_context'] = playwright_context
|
|
|
|
browserstepper = browser_steps.browsersteps_live_ui(
|
|
playwright_browser=browser,
|
|
proxy=proxy,
|
|
start_url=watch.link,
|
|
headers=watch.get('headers'),
|
|
browser_config=browser_config,
|
|
)
|
|
await browserstepper.connect(proxy=proxy)
|
|
browsersteps_start_session['browserstepper'] = browserstepper
|
|
|
|
return browsersteps_start_session
|
|
|
|
|
|
@browser_steps_blueprint.route("/browsersteps_start_session", methods=['POST'])
|
|
@login_optionally_required
|
|
def browsersteps_start_session():
|
|
# A new session was requested, return sessionID
|
|
import uuid
|
|
browsersteps_session_id = str(uuid.uuid4())
|
|
watch_uuid = request.args.get('uuid')
|
|
|
|
if not watch_uuid:
|
|
return make_response('No Watch UUID specified', 500)
|
|
|
|
# Cleanup any existing session for this watch
|
|
cleanup_session_for_watch(watch_uuid)
|
|
|
|
logger.debug("Starting connection with playwright")
|
|
logger.debug("browser_steps.py connecting")
|
|
|
|
try:
|
|
# Run the async function in the dedicated browser steps event loop
|
|
browsersteps_sessions[browsersteps_session_id] = run_async_in_browser_loop(
|
|
start_browsersteps_session(watch_uuid)
|
|
)
|
|
|
|
# Store the mapping of watch_uuid -> browsersteps_session_id
|
|
browsersteps_watch_to_session[watch_uuid] = browsersteps_session_id
|
|
|
|
except Exception as e:
|
|
if 'ECONNREFUSED' in str(e):
|
|
return make_response('Unable to start the Playwright Browser session, is sockpuppetbrowser running? Network configuration is OK?', 401)
|
|
else:
|
|
# Other errors, bad URL syntax, bad reply etc
|
|
return make_response(str(e), 401)
|
|
|
|
logger.debug("Starting connection with playwright - done")
|
|
return {'browsersteps_session_id': browsersteps_session_id}
|
|
|
|
@browser_steps_blueprint.route("/browsersteps_image", methods=['GET'])
|
|
@login_optionally_required
|
|
def browser_steps_fetch_screenshot_image():
|
|
from flask import (
|
|
make_response,
|
|
request,
|
|
send_from_directory,
|
|
)
|
|
uuid = request.args.get('uuid')
|
|
step_n = int(request.args.get('step_n'))
|
|
|
|
watch = datastore.data['watching'].get(uuid)
|
|
filename = f"step_before-{step_n}.jpeg" if request.args.get('type', '') == 'before' else f"step_{step_n}.jpeg"
|
|
|
|
if step_n and watch and os.path.isfile(os.path.join(watch.data_dir, filename)):
|
|
response = make_response(send_from_directory(directory=watch.data_dir, path=filename))
|
|
response.headers['Content-type'] = 'image/jpeg'
|
|
response.headers['Cache-Control'] = 'no-cache, no-store, must-revalidate'
|
|
response.headers['Pragma'] = 'no-cache'
|
|
response.headers['Expires'] = 0
|
|
return response
|
|
|
|
else:
|
|
return make_response('Unable to fetch image, is the URL correct? does the watch exist? does the step_type-n.jpeg exist?', 401)
|
|
|
|
# A request for an action was received
|
|
@browser_steps_blueprint.route("/browsersteps_update", methods=['POST'])
|
|
@login_optionally_required
|
|
def browsersteps_ui_update():
|
|
import base64
|
|
|
|
remaining = 0
|
|
uuid = request.args.get('uuid')
|
|
goto_website_url_first_step = request.args.get('goto_website_url_first_step')
|
|
|
|
browsersteps_session_id = request.args.get('browsersteps_session_id')
|
|
|
|
if not browsersteps_session_id:
|
|
return make_response('No browsersteps_session_id specified', 500)
|
|
|
|
if not browsersteps_sessions.get(browsersteps_session_id):
|
|
return make_response('No session exists under that ID', 500)
|
|
|
|
is_last_step = False
|
|
|
|
# @todo - should always be an existing session
|
|
if goto_website_url_first_step:
|
|
logger.debug("Going to site (requested automatically before stepping)..")
|
|
step_operation = "Goto site"
|
|
step_selector = None
|
|
step_optional_value = None
|
|
else:
|
|
step_operation = request.form.get('operation')
|
|
step_selector = request.form.get('selector')
|
|
step_optional_value = request.form.get('optional_value')
|
|
is_last_step = strtobool(request.form.get('is_last_step'))
|
|
|
|
try:
|
|
# Run the async call_action method in the dedicated browser steps event loop
|
|
run_async_in_browser_loop(
|
|
browsersteps_sessions[browsersteps_session_id]['browserstepper'].call_action(
|
|
action_name=step_operation,
|
|
selector=step_selector,
|
|
optional_value=step_optional_value
|
|
)
|
|
)
|
|
|
|
except Exception as e:
|
|
logger.error(f"Exception when calling step operation {step_operation} {str(e)}")
|
|
# Try to find something of value to give back to the user.
|
|
# text/plain: the message can contain the user's own selectors/values, so it
|
|
# must not be parsed as HTML by the browser (GHSA-23mp-8222-96fr pattern).
|
|
return plaintext_response(str(e).splitlines()[0], 401)
|
|
|
|
# Screenshots and other info only needed on requesting a step (POST)
|
|
try:
|
|
# Run the async get_current_state method in the dedicated browser steps event loop
|
|
(screenshot, xpath_data) = run_async_in_browser_loop(
|
|
browsersteps_sessions[browsersteps_session_id]['browserstepper'].get_current_state()
|
|
)
|
|
|
|
if is_last_step:
|
|
watch = datastore.data['watching'].get(uuid)
|
|
u = browsersteps_sessions[browsersteps_session_id]['browserstepper'].page.url
|
|
if watch and u:
|
|
watch.save_screenshot(screenshot=screenshot)
|
|
watch.save_xpath_data(data=xpath_data)
|
|
|
|
except Exception as e:
|
|
return plaintext_response(f"Error fetching screenshot and element data - {str(e)}", 401)
|
|
|
|
# SEND THIS BACK TO THE BROWSER
|
|
output = {
|
|
"screenshot": f"data:image/jpeg;base64,{base64.b64encode(screenshot).decode('ascii')}",
|
|
"xpath_data": xpath_data,
|
|
"session_age_start": browsersteps_sessions[browsersteps_session_id]['browserstepper'].age_start,
|
|
"browser_time_remaining": round(remaining)
|
|
}
|
|
json_data = json.dumps(output)
|
|
|
|
# Generate an ETag (hash of the response body)
|
|
etag_hash = hashlib.md5(json_data.encode('utf-8')).hexdigest()
|
|
|
|
# Create the response with ETag
|
|
response = Response(json_data, mimetype="application/json; charset=UTF-8")
|
|
response.set_etag(etag_hash)
|
|
|
|
return response
|
|
|
|
return browser_steps_blueprint
|
|
|
|
|