Merge branch 'GHSA-56fq-63vj-9992-add-watch-ui-snapshot' into 0-60-1-cleanups
Publish Python 🐍distribution 📦 to PyPI and TestPyPI / Build distribution 📦 (push) Canceled after 0s
ChangeDetection.io App Test / lint-code (push) Canceled after 0s
ChangeDetection.io App Test / lint-translations (push) Canceled after 0s
ChangeDetection.io App Test / lint-template-i18n (push) Canceled after 0s
Publish Python 🐍distribution 📦 to PyPI and TestPyPI / Test the built package works basically. (push) Canceled after 0s
Publish Python 🐍distribution 📦 to PyPI and TestPyPI / Publish Python 🐍 distribution 📦 to PyPI (push) Canceled after 0s
ChangeDetection.io App Test / test-application-3-11 (push) Canceled after 0s
ChangeDetection.io App Test / test-application-3-12 (push) Canceled after 0s
ChangeDetection.io App Test / test-application-3-13 (push) Canceled after 0s
ChangeDetection.io App Test / test-application-3-14 (push) Canceled after 0s

This commit is contained in:
dgtlmoon
2026-09-04 12:06:35 +02:00
5 changed files with 282 additions and 12 deletions
@@ -42,7 +42,7 @@ def construct_blueprint(datastore: ChangeDetectionStore):
system_default_browser=browser_config.system_default_description(datastore),
)
@add_watch_ui_blueprint.route("/snapshot", methods=['GET'])
@add_watch_ui_blueprint.route("/snapshot", methods=['POST'])
@login_optionally_required
def add_watch_ui_snapshot():
"""One-shot live fetch of an arbitrary URL for the Add Watch visual selector.
@@ -52,6 +52,10 @@ def construct_blueprint(datastore: ChangeDetectionStore):
connect, "Goto site", grab the screenshot + xpath element data, then tear
the browser down again. Element selection then happens client-side on the
returned data, exactly like the watch Edit page's visual selector.
POST-only and CSRF protected on purpose: this drives a real browser fetch and
writes a temporary watch dir, so as a GET it could be triggered cross-origin
(or by any tag/link that issues a GET) without the operator's consent.
"""
import base64
from changedetectionio.blueprint.browser_steps import (
@@ -71,7 +75,7 @@ def construct_blueprint(datastore: ChangeDetectionStore):
# backslash/parser-differential rejection of GHSA-rph4-96w6-q594 (GHSA-56fq-63vj-9992).
# Note this fetch never reaches difference_detection_processor.call_browser(), so it gets
# no gating from there - it has to validate for itself.
url = (request.args.get('url') or '').strip()
url = (request.form.get('url') or '').strip()
ok, reason = is_fetch_url_allowed(url)
if not ok:
logger.warning(f"Add-watch snapshot: refused '{url}' - {reason}")
@@ -82,7 +86,7 @@ def construct_blueprint(datastore: ChangeDetectionStore):
# Either way it has to be able to render a preview - the plain HTTP client
# produces no screenshot and no element data, so previewing with it is pointless
# (and it used to be the silent default here, see the system-default bug).
fetcher_name = (request.args.get('fetch_backend') or '').strip() or browser_config.default_visual_browser(datastore)
fetcher_name = (request.form.get('fetch_backend') or '').strip() or browser_config.default_visual_browser(datastore)
if not fetcher_name or not browser_config.is_visual_capable(fetcher_name, datastore):
logger.warning(f"Add-watch snapshot: refused browser '{fetcher_name}' for '{url}'")
return make_response('No interactive browser available that can render a live preview '
@@ -59,9 +59,18 @@ $(document).ready(() => {
$.ajax({
url: add_watch_snapshot_url,
// POST, never GET - this makes the server-side browser fetch a URL of our
// choosing, so it must not be triggerable cross-origin. csrf.js adds the
// X-CSRFToken header to every non-GET ajax call; the CSRF field on the form
// is sent too so it works even if that handler hasn't run yet.
method: 'POST',
// Preview with the browser picked in the list - that same browser is what
// gets saved on the watch, so what you see here is what it will check with.
data: {url: url, fetch_backend: $('input[name="fetch_backend"]:checked').val() || ''},
data: {
url: url,
fetch_backend: $('input[name="fetch_backend"]:checked').val() || '',
csrf_token: $('#new-watch-form input[name="csrf_token"]').val() || '',
},
dataType: 'json',
}).done((data) => {
showState('ready');
@@ -79,26 +79,41 @@ def test_snapshot_refuses_browser_that_cannot_preview(client, live_server, measu
from changedetectionio.blueprint.add_watch_ui import browser_config
monkeypatch.setattr(browser_config, 'is_visual_capable', lambda name, datastore: False)
snapshot_url = url_for('add_watch_ui.add_watch_ui_snapshot')
# Nothing capable, and no explicit browser asked for -> nothing to preview with
res = client.get(url_for('add_watch_ui.add_watch_ui_snapshot', url='https://example.com'))
res = client.post(snapshot_url, data={'url': 'https://example.com'})
assert res.status_code == 400
assert b'No interactive browser' in res.data
# Explicitly asking for a browser that can't preview is refused just the same
res = client.get(url_for('add_watch_ui.add_watch_ui_snapshot', url='https://example.com',
fetch_backend='html_requests'))
res = client.post(snapshot_url, data={'url': 'https://example.com',
'fetch_backend': 'html_requests'})
assert res.status_code == 400
# A made-up name never resolves to a capable fetcher either (real capability lookup here)
monkeypatch.undo()
res = client.get(url_for('add_watch_ui.add_watch_ui_snapshot', url='https://example.com',
fetch_backend='../../etc/passwd'))
res = client.post(snapshot_url, data={'url': 'https://example.com',
'fetch_backend': '../../etc/passwd'})
assert res.status_code == 400
res = client.get(url_for('add_watch_ui.add_watch_ui_snapshot', url='https://example.com',
fetch_backend='os'))
res = client.post(snapshot_url, data={'url': 'https://example.com',
'fetch_backend': 'os'})
assert res.status_code == 400
def test_snapshot_is_post_only(client, live_server, measure_memory_usage, datastore_path):
"""A GET must not reach the endpoint at all.
/snapshot drives a real server-side browser fetch and hands the rendered result back in
the response (GHSA-56fq-63vj-9992). As a GET that is reachable by anything that can make
the operator's browser issue a request - an <img>/<iframe>/link from another site - with
no CSRF token in play. POST-only + CSRFProtect means only our own page can trigger it.
"""
# Method mismatch surfaces as 404 here rather than 405
res = client.get(url_for('add_watch_ui.add_watch_ui_snapshot') + '?url=https://example.com')
assert res.status_code in (404, 405)
def test_submit_rejects_unknown_fetcher(client, live_server, measure_memory_usage, datastore_path):
"""A posted browser is checked server side, so a doctored form can't pin a junk fetcher."""
datastore = _datastore(client)
@@ -0,0 +1,242 @@
#!/usr/bin/env python3
"""Tests for the shared "may the server fetch this URL?" gate.
# run from dir above changedetectionio/ dir
# python3 -m unittest changedetectionio.tests.unit.test_fetch_url_gate
Every server-side fetch entry point routes through validate_url.is_fetch_url_allowed(). Before it
existed, the file:// and private-IP rules were enforced inline in call_browser() only, so any fetch
path that did not go through call_browser() was unprotected:
* a "Goto URL" browser step could read file:///etc/passwd (GHSA-hm22-wg2m-35v4)
* /add-watch-ui/snapshot url= could fetch internal hosts (GHSA-56fq-63vj-9992)
These tests pin the gate's rules AND the browser-step choke point, so a future fetch path that
forgets to call the gate is the only way to regress it.
"""
import asyncio
import unittest
from unittest.mock import patch
from changedetectionio.browser_steps.browser_steps import steppable_browser_interface
from changedetectionio.validate_url import (
is_fetch_url_allowed,
is_special_purpose_ip,
validate_fetch_url,
validate_fetch_url_async,
)
# tests/conftest.py sets ALLOW_IANA_RESTRICTED_ADDRESSES=true for the functional suite, so the
# locked-down default has to be re-asserted explicitly rather than assumed.
LOCKED_DOWN = {'ALLOW_IANA_RESTRICTED_ADDRESSES': 'false', 'ALLOW_FILE_URI': 'false'}
OPTED_IN = {'ALLOW_IANA_RESTRICTED_ADDRESSES': 'true', 'ALLOW_FILE_URI': 'true'}
class TestFetchUrlGate(unittest.TestCase):
def assertBlocked(self, url):
ok, reason = is_fetch_url_allowed(url)
self.assertFalse(ok, f"URL '{url}' should have been blocked")
self.assertTrue(reason, f"URL '{url}' was blocked without a reason to show the user")
def assertAllowed(self, url):
ok, reason = is_fetch_url_allowed(url)
self.assertTrue(ok, f"URL '{url}' should have been allowed, got: {reason}")
def test_file_uri_blocked_by_default(self):
with patch.dict('os.environ', LOCKED_DOWN):
# All the spellings that reach the same local file
for url in ('file:///etc/passwd', 'FILE:///etc/passwd', 'file:/etc/passwd', 'file://etc/passwd'):
with self.subTest(url=url):
self.assertBlocked(url)
def test_file_uri_allowed_when_operator_opts_in(self):
with patch.dict('os.environ', OPTED_IN):
self.assertAllowed('file:///etc/passwd')
def test_file_uri_blocked_even_if_safe_protocol_regex_was_loosened(self):
"""An operator who widens SAFE_PROTOCOL_REGEX for some other scheme must not get local
file reads thrown in for free - hence the explicit file: check ahead of is_safe_valid_url()."""
env = dict(LOCKED_DOWN, SAFE_PROTOCOL_REGEX='^(http|https|ftp|file):')
with patch.dict('os.environ', env):
self.assertBlocked('file:///etc/passwd')
def test_private_and_reserved_addresses_blocked_by_default(self):
with patch.dict('os.environ', LOCKED_DOWN):
for url in ('http://127.0.0.1:5000/',
'http://localhost/',
'http://169.254.169.254/latest/meta-data/', # cloud metadata
'http://192.168.1.1/',
'http://10.0.0.1/',
'http://[::1]/'):
with self.subTest(url=url):
self.assertBlocked(url)
def test_cgnat_and_other_non_global_addresses_blocked_by_default(self):
"""GHSA-gwph-fp79-379w - the 0.54.1 predicate only tested is_private/is_loopback/
is_link_local/is_reserved, none of which are True for RFC 6598 CGNAT space, so
100.64.0.0/10 (an ISP's other subscribers, CPE admin panels, CGNAT gateways) stayed
fetchable. These are IP literals, so no DNS is involved and CI cannot flake."""
with patch.dict('os.environ', LOCKED_DOWN):
for url in ('http://100.64.0.1/', # RFC 6598 CGNAT, first usable
'http://100.127.255.254/', # RFC 6598 CGNAT, last usable
'http://100.100.100.100/', # inside CGNAT (Alibaba Cloud metadata)
'http://192.88.99.1/', # RFC 7526 deprecated 6to4 relay anycast
'http://224.0.0.1/', # IPv4 multicast all-hosts
'http://[ff02::1]/'): # IPv6 multicast all-nodes
with self.subTest(url=url):
self.assertBlocked(url)
def test_cgnat_allowed_when_operator_opts_in(self):
"""CGNAT is legitimate for operators monitoring their own carrier network, so the
opt-in has to release it the same way it releases 127.0.0.1."""
with patch.dict('os.environ', OPTED_IN):
self.assertAllowed('http://100.64.0.1/')
def test_special_purpose_ip_classification(self):
"""The predicate itself, without DNS - one place to pin what is and is not fetchable."""
for ip in ('100.64.0.1', '100.127.255.254', '192.88.99.1', '224.0.0.1', 'ff02::1',
'127.0.0.1', '10.0.0.1', '169.254.169.254', '192.168.1.1', '::1',
'0.0.0.0', '255.255.255.255', '198.18.0.1', 'fc00::1', 'fe80::1',
'::ffff:100.64.0.1', # CGNAT wrapped as an IPv4-mapped IPv6 address
'2002:6440:1::'): # CGNAT wrapped as a 6to4 address
with self.subTest(ip=ip):
blocked, why = is_special_purpose_ip(ip)
self.assertTrue(blocked, f"{ip} should be refused")
self.assertTrue(why, f"{ip} was refused without a stated reason")
for ip in ('1.1.1.1', '8.8.8.8', '93.184.216.34', '2606:4700:4700::1111'):
with self.subTest(ip=ip):
blocked, why = is_special_purpose_ip(ip)
self.assertFalse(blocked, f"public address {ip} was refused as '{why}'")
def test_cgnat_boundaries_are_exact(self):
"""100.64.0.0/10 ends at 100.127.255.255 - 100.63.x and 100.128.x are ordinary public
space and must not be collateral damage from a /8-sized over-block."""
for ip in ('100.63.255.255', '100.128.0.0'):
with self.subTest(ip=ip):
blocked, why = is_special_purpose_ip(ip)
self.assertFalse(blocked, f"public address {ip} was refused as '{why}'")
def test_private_addresses_allowed_when_operator_opts_in(self):
with patch.dict('os.environ', OPTED_IN):
self.assertAllowed('http://127.0.0.1:5000/')
def test_source_prefix_is_stripped_before_the_hostname_check(self):
"""Load-bearing, not cosmetic: urlparse('source:http://127.0.0.1/') reports NO hostname,
so leaving the prefix on would hand the private-IP check nothing to look at and let it pass."""
with patch.dict('os.environ', LOCKED_DOWN):
self.assertBlocked('source:http://127.0.0.1/')
self.assertBlocked('SOURCE:http://169.254.169.254/')
self.assertBlocked('source:file:///etc/passwd')
def test_jinja2_is_rendered_before_the_hostname_check(self):
"""The fetch uses the rendered URL, so the rendered URL is what must be judged - otherwise
a template expression hides the real target from the check."""
with patch.dict('os.environ', LOCKED_DOWN):
self.assertBlocked("http://{{ '127.0.0.1' }}/")
self.assertBlocked("http://{% if 1 %}127.0.0.1{% endif %}/")
def test_parser_differential_payload_always_rejected(self):
"""GHSA-rph4-96w6-q594: urlparse sees PUBLIC, urllib3 connects to INTERNAL. A backslash has
no legitimate use in a URL, so this is refused even with both opt-ins enabled."""
for env in (LOCKED_DOWN, OPTED_IN):
with self.subTest(env=env), patch.dict('os.environ', env):
self.assertBlocked('http://127.0.0.1:8888\\@example.com/')
def test_unsupported_schemes_rejected(self):
with patch.dict('os.environ', LOCKED_DOWN):
for url in ('javascript:alert(1)', 'data:text/html,<h1>x', 'chrome://version'):
with self.subTest(url=url):
self.assertBlocked(url)
def test_empty_input_rejected(self):
with patch.dict('os.environ', LOCKED_DOWN):
for url in ('', ' ', None):
with self.subTest(url=url):
self.assertBlocked(url)
def test_ordinary_public_urls_still_allowed(self):
# Unresolvable hostnames are allowed by design (DNS may be down, domain not yet live), so
# these pass with or without working DNS in CI.
with patch.dict('os.environ', LOCKED_DOWN):
for url in ('https://example.com/',
'source:https://example.com/',
'https://example.com/path?a=b&c=d#frag'):
with self.subTest(url=url):
self.assertAllowed(url)
def test_validate_fetch_url_raises_with_the_reason(self):
with patch.dict('os.environ', LOCKED_DOWN):
with self.assertRaises(ValueError):
validate_fetch_url('file:///etc/passwd')
validate_fetch_url('https://example.com/') # must not raise
def test_validate_fetch_url_async_raises_with_the_reason(self):
with patch.dict('os.environ', LOCKED_DOWN):
with self.assertRaises(ValueError):
asyncio.run(validate_fetch_url_async('http://127.0.0.1/'))
asyncio.run(validate_fetch_url_async('https://example.com/')) # must not raise
class _RecordingPage:
"""Stands in for the Playwright page so we can assert navigation never happened."""
def __init__(self):
self.goto_calls = []
async def goto(self, url, **kwargs):
self.goto_calls.append(url)
return None
async def wait_for_timeout(self, ms):
return None
class TestBrowserStepGotoUrlGate(unittest.TestCase):
"""GHSA-hm22-wg2m-35v4 - browser step values are raw user input and were never validated.
action_goto_url() is the single choke point for every navigation we initiate (the "Goto URL"
step, "Goto site", the live Browser Steps UI and the Add Watch preview all land here), so the
assertion that matters is that page.goto() is never reached for a refused URL.
"""
def _interface(self, start_url='https://example.com/'):
interface = steppable_browser_interface(start_url=start_url)
interface.page = _RecordingPage()
return interface
def test_goto_url_step_cannot_read_local_files(self):
with patch.dict('os.environ', LOCKED_DOWN):
interface = self._interface()
with self.assertRaises(ValueError):
asyncio.run(interface.action_goto_url(value='file:///etc/passwd'))
self.assertEqual(interface.page.goto_calls, [], "Chromium was navigated to a refused URL")
def test_goto_url_step_cannot_reach_private_addresses(self):
with patch.dict('os.environ', LOCKED_DOWN):
for url in ('http://127.0.0.1:5000/', 'http://169.254.169.254/latest/meta-data/'):
with self.subTest(url=url):
interface = self._interface()
with self.assertRaises(ValueError):
asyncio.run(interface.action_goto_url(value=url))
self.assertEqual(interface.page.goto_calls, [])
def test_goto_site_step_validates_the_start_url_too(self):
with patch.dict('os.environ', LOCKED_DOWN):
interface = self._interface(start_url='source:http://127.0.0.1/')
with self.assertRaises(ValueError):
asyncio.run(interface.action_goto_site())
self.assertEqual(interface.page.goto_calls, [])
def test_permitted_url_still_navigates(self):
with patch.dict('os.environ', LOCKED_DOWN):
interface = self._interface()
asyncio.run(interface.action_goto_url(value='https://example.com/'))
self.assertEqual(interface.page.goto_calls, ['https://example.com/'])
if __name__ == '__main__':
unittest.main()
+1 -1
View File
@@ -8,7 +8,7 @@ msgid ""
msgstr ""
"Project-Id-Version: changedetection.io 0.60.2\n"
"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n"
"POT-Creation-Date: 2026-09-03 21:31+0200\n"
"POT-Creation-Date: 2026-09-04 10:59+0200\n"
"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n"
"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n"
"Language-Team: LANGUAGE <LL@li.org>\n"