diff --git a/.github/workflows/containers.yml b/.github/workflows/containers.yml
index 95b5ce77c..5e73c48ee 100644
--- a/.github/workflows/containers.yml
+++ b/.github/workflows/containers.yml
@@ -41,7 +41,7 @@ jobs:
steps:
- uses: actions/checkout@v7
- name: Set up Python 3.11
- uses: actions/setup-python@v6
+ uses: actions/setup-python@v7
with:
python-version: 3.11
diff --git a/.github/workflows/pypi-release.yml b/.github/workflows/pypi-release.yml
index 502cbbfa1..e23d1bbab 100644
--- a/.github/workflows/pypi-release.yml
+++ b/.github/workflows/pypi-release.yml
@@ -9,7 +9,7 @@ jobs:
steps:
- uses: actions/checkout@v7
- name: Set up Python
- uses: actions/setup-python@v6
+ uses: actions/setup-python@v7
with:
python-version: "3.11"
- name: Install pypa/build
@@ -39,7 +39,7 @@ jobs:
name: python-package-distributions
path: dist/
- name: Set up Python 3.11
- uses: actions/setup-python@v6
+ uses: actions/setup-python@v7
with:
python-version: '3.11'
diff --git a/.github/workflows/test-container-build.yml b/.github/workflows/test-container-build.yml
index 5f841915e..bbfc8ef45 100644
--- a/.github/workflows/test-container-build.yml
+++ b/.github/workflows/test-container-build.yml
@@ -46,7 +46,7 @@ jobs:
steps:
- uses: actions/checkout@v7
- name: Set up Python 3.11
- uses: actions/setup-python@v6
+ uses: actions/setup-python@v7
with:
python-version: 3.11
diff --git a/.github/workflows/test-only.yml b/.github/workflows/test-only.yml
index 7a0eb49b4..375c95e43 100644
--- a/.github/workflows/test-only.yml
+++ b/.github/workflows/test-only.yml
@@ -33,7 +33,7 @@ jobs:
done
- name: Lint .pot template with dennis
run: |
- pip install "$(grep -E '^dennis ?>=' requirements.txt)"
+ pip install "$(grep -E '^dennis ?>=' requirements-dev.txt)"
dennis-cmd lint --strict changedetectionio/translations/messages.pot
- name: Lint .po files with dennis
run: |
@@ -52,6 +52,19 @@ jobs:
git diff --stat changedetectionio/translations
exit 1
fi
+ - name: Check translation overlay
+ # Deliberately after extract_messages above, so overrides are validated against a freshly
+ # extracted messages.pot. An overlay entry is keyed on the exact upstream msgid: when a
+ # string is reworded upstream the override stops matching and silently reverts to upstream
+ # wording. This is the only thing that makes that visible.
+ # See changedetectionio/translations_overlay/README.md
+ if: hashFiles('changedetectionio/translations_overlay/**/*.po') != ''
+ run: |
+ find changedetectionio/translations_overlay -name "*.po" | while read f; do
+ echo "Checking $f"
+ msgfmt --check-format -o /dev/null "$f"
+ done
+ python changedetectionio/translations_overlay/manage.py check
lint-template-i18n:
runs-on: ubuntu-latest
@@ -103,15 +116,6 @@ jobs:
sys.exit(1)
PYEOF
- test-application-3-10:
- # Only run on push to master (including PR merges)
- if: github.event_name == 'push' && github.ref == 'refs/heads/master'
- needs: [lint-code, lint-translations, lint-template-i18n]
- uses: ./.github/workflows/test-stack-reusable-workflow.yml
- with:
- python-version: '3.10'
-
-
test-application-3-11:
# Always run
needs: [lint-code, lint-translations, lint-template-i18n]
diff --git a/.github/workflows/test-stack-reusable-workflow.yml b/.github/workflows/test-stack-reusable-workflow.yml
index 911cbb0fc..350a3f854 100644
--- a/.github/workflows/test-stack-reusable-workflow.yml
+++ b/.github/workflows/test-stack-reusable-workflow.yml
@@ -24,7 +24,7 @@ jobs:
- uses: actions/checkout@v7
- name: Set up Python ${{ env.PYTHON_VERSION }}
- uses: actions/setup-python@v6
+ uses: actions/setup-python@v7
with:
python-version: ${{ env.PYTHON_VERSION }}
@@ -62,6 +62,18 @@ jobs:
echo "---- Built for Python ${{ env.PYTHON_VERSION }} -----"
docker run test-changedetectionio bash -c 'pip list'
+ - name: License compliance - no strong copyleft in shipped image
+ run: |
+ # Runs against the built image, i.e. exactly what ships, with the entrypoint
+ # bypassed so EXTRA_PACKAGES cannot pull in the opt-in AGPL osint plugin.
+ # --partial-match is required: plain --fail-on is an exact string match and
+ # would silently miss declarations like "GPL-2.0-or-later".
+ docker run --rm --entrypoint /bin/bash test-changedetectionio -c '
+ pip install --quiet pip-licenses &&
+ pip-licenses --partial-match \
+ --fail-on="GPL-2.0;GPL-3.0;AGPL;GNU General Public License;GNU Affero General Public License"
+ '
+
- name: We should be Python ${{ env.PYTHON_VERSION }} ...
run: |
docker run test-changedetectionio bash -c 'python3 --version'
@@ -229,10 +241,20 @@ jobs:
- name: Playwright - Specific tests in built container
run: |
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest -vv --capture=tee-sys --showlocals --tb=long --live-server-host=0.0.0.0 --live-server-port=5004 tests/fetchers/test_content.py'
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest -vv --capture=tee-sys --showlocals --tb=long --live-server-host=0.0.0.0 --live-server-port=5004 tests/test_errorhandling.py'
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest -vv --capture=tee-sys --showlocals --tb=long --live-server-host=0.0.0.0 --live-server-port=5004 tests/visualselector/test_fetch_data.py'
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest -vv --capture=tee-sys --showlocals --tb=long --live-server-host=0.0.0.0 --live-server-port=5004 tests/fetchers/test_custom_js_before_content.py'
+ # Deliberately fail fast - the first failure is nearly always the real problem and it
+ # keeps the run short. Each file is wrapped in its own log group and the failing one is
+ # named explicitly, because everything after it is SKIPPED rather than run, and that is
+ # otherwise easy to misread as "the whole browser suite broke".
+ for t in tests/fetchers/test_content.py tests/test_errorhandling.py tests/visualselector/test_fetch_data.py tests/fetchers/test_custom_js_before_content.py tests/fetchers/test_renavigation.py; do
+ echo "::group::pytest $t"
+ if ! docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio \
+ bash -c "cd changedetectionio;pytest -vv --capture=tee-sys --showlocals --tb=long --live-server-host=0.0.0.0 --live-server-port=5004 $t"; then
+ echo "::endgroup::"
+ echo "::error::$t FAILED - stopping here. Any later test file and the remaining steps of this job were SKIPPED, not failed."
+ exit 1
+ fi
+ echo "::endgroup::"
+ done
- name: Playwright - Headers and requests
run: |
@@ -270,10 +292,17 @@ jobs:
- name: Pyppeteer - Specific tests in built container
run: |
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "FAST_PUPPETEER_CHROME_FETCHER=True" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest --live-server-host=0.0.0.0 --live-server-port=5004 tests/fetchers/test_content.py'
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "FAST_PUPPETEER_CHROME_FETCHER=True" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest --live-server-host=0.0.0.0 --live-server-port=5004 tests/test_errorhandling.py'
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "FAST_PUPPETEER_CHROME_FETCHER=True" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest --live-server-host=0.0.0.0 --live-server-port=5004 tests/visualselector/test_fetch_data.py'
- docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "FAST_PUPPETEER_CHROME_FETCHER=True" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio bash -c 'cd changedetectionio;pytest --live-server-host=0.0.0.0 --live-server-port=5004 tests/fetchers/test_custom_js_before_content.py'
+ # Fail fast, but name the file that failed - see the note in the playwright job above
+ for t in tests/fetchers/test_content.py tests/test_errorhandling.py tests/visualselector/test_fetch_data.py tests/fetchers/test_custom_js_before_content.py tests/fetchers/test_renavigation.py; do
+ echo "::group::pytest $t"
+ if ! docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "FAST_PUPPETEER_CHROME_FETCHER=True" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio \
+ bash -c "cd changedetectionio;pytest --live-server-host=0.0.0.0 --live-server-port=5004 $t"; then
+ echo "::endgroup::"
+ echo "::error::$t FAILED - stopping here. Any later test file and the remaining steps of this job were SKIPPED, not failed."
+ exit 1
+ fi
+ echo "::endgroup::"
+ done
- name: Pyppeteer - Headers and requests checks
run: |
@@ -727,7 +756,7 @@ jobs:
fetch-depth: 0 # Fetch all history and tags for upgrade testing
- name: Set up Python ${{ env.PYTHON_VERSION }}
- uses: actions/setup-python@v6
+ uses: actions/setup-python@v7
with:
python-version: ${{ env.PYTHON_VERSION }}
diff --git a/.ruff.toml b/.ruff.toml
index 259be2fb4..04b7adf33 100644
--- a/.ruff.toml
+++ b/.ruff.toml
@@ -1,5 +1,5 @@
# Minimum supported version
-target-version = "py310"
+target-version = "py311"
# Formatting options
line-length = 100
diff --git a/COMMERCIAL_LICENCE.md b/COMMERCIAL_LICENCE.md
deleted file mode 100644
index fa59b2eae..000000000
--- a/COMMERCIAL_LICENCE.md
+++ /dev/null
@@ -1,54 +0,0 @@
-# Generally
-
-In any commercial activity involving 'Hosting' (as defined herein), whether in part or in full, this license must be executed and adhered to.
-
-# Commercial License Agreement
-
-This Commercial License Agreement ("Agreement") is entered into by and between Web Technologies s.r.o. here-in ("Licensor") and (your company or personal name) _____________ ("Licensee"). This Agreement sets forth the terms and conditions under which Licensor provides its software ("Software") and services to Licensee for the purpose of reselling the software either in part or full, as part of any commercial activity where the activity involves a third party.
-
-### Definition of Hosting
-
-For the purposes of this Agreement, "hosting" means making the functionality of the Program or modified version available to third parties as a service. This includes, without limitation:
-- Enabling third parties to interact with the functionality of the Program or modified version remotely through a computer network.
-- Offering a service the value of which entirely or primarily derives from the value of the Program or modified version.
-- Offering a service that accomplishes for users the primary purpose of the Program or modified version.
-
-## 1. Grant of License
-Subject to the terms and conditions of this Agreement, Licensor grants Licensee a non-exclusive, non-transferable license to install, use, and resell the Software. Licensee may:
-- Resell the Software as part of a service offering or as a standalone product.
-- Host the Software on a server and provide it as a hosted service (e.g., Software as a Service - SaaS).
-- Integrate the Software into a larger product or service that is then sold or provided for commercial purposes, where the software is used either in part or full.
-
-## 2. License Fees
-Licensee agrees to pay Licensor the license fees specified in the ordering document. License fees are due and payable as specified in the ordering document. The fees may include initial licensing costs and recurring fees based on the number of end users, instances of the Software resold, or revenue generated from the resale activities.
-
-## 3. Resale Conditions
-Licensee must comply with the following conditions when reselling the Software, whether the software is resold in part or full:
-- Provide end users with access to the source code under the same open-source license conditions as provided by Licensor.
-- Clearly state in all marketing and sales materials that the Software is provided under a commercial license from Licensor, and provide a link back to https://changedetection.io.
-- Ensure end users are aware of and agree to the terms of the commercial license prior to resale.
-- Do not sublicense or transfer the Software to third parties except as part of an authorized resale activity.
-
-## 4. Hosting and Provision of Services
-Licensee may host the Software (either in part or full) on its servers and provide it as a hosted service to end users. The following conditions apply:
-- Licensee must ensure that all hosted versions of the Software comply with the terms of this Agreement.
-- Licensee must provide Licensor with regular reports detailing the number of end users and instances of the hosted service.
-- Any modifications to the Software made by Licensee for hosting purposes must be made available to end users under the same open-source license conditions, unless agreed otherwise.
-
-## 5. Services
-Licensor will provide support and maintenance services as described in the support policy referenced in the ordering document should such an agreement be signed by all parties. Additional fees may apply for support services provided to end users resold by Licensee.
-
-## 6. Reporting and Audits
-Licensee agrees to provide Licensor with regular reports detailing the number of instances, end users, and revenue generated from the resale of the Software. Licensor reserves the right to audit Licensee’s records to ensure compliance with this Agreement.
-
-## 7. Term and Termination
-This Agreement shall commence on the effective date and continue for the period set forth in the ordering document unless terminated earlier in accordance with this Agreement. Either party may terminate this Agreement if the other party breaches any material term and fails to cure such breach within thirty (30) days after receipt of written notice.
-
-## 8. Limitation of Liability and Disclaimer of Warranty
-Executing this commercial license does not waive the Limitation of Liability or Disclaimer of Warranty as stated in the open-source LICENSE provided with the Software. The Software is provided "as is," without warranty of any kind, express or implied, including but not limited to the warranties of merchantability, fitness for a particular purpose, and noninfringement. In no event shall the authors or copyright holders be liable for any claim, damages, or other liability, whether in an action of contract, tort, or otherwise, arising from, out of, or in connection with the Software or the use or other dealings in the Software.
-
-## 9. Governing Law
-This Agreement shall be governed by and construed in accordance with the laws of the Czech Republic.
-
-## Contact Information
-For commercial licensing inquiries, please contact contact@changedetection.io and dgtlmoon@gmail.com.
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index da3d99a96..4bb696cfc 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -4,6 +4,15 @@ I am no professional flask developer, if you know a better way that something ca
Otherwise, it's always best to PR into the `master` branch.
+Install the development and test dependencies with `pip install -r requirements-dev.txt`.
+
Please be sure that all new functionality has a matching test!
Use `pytest` to validate/test, you can run the existing tests as `pytest tests/test_notification.py` for example
+
+### New dependencies
+
+Please check the licence before adding anything to `requirements.txt`. MIT, BSD,
+Apache-2.0, LGPL and MPL are fine; GPL and AGPL are not, since they would
+relicense the whole project. CI enforces this. There is usually a permissive
+equivalent - ask in the PR if you're not sure.
diff --git a/Dockerfile b/Dockerfile
index 341da9361..3c7c777da 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -101,6 +101,31 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
libxrender-dev \
&& apt-get clean && rm -rf /var/lib/apt/lists/*
+# Actually generate the locales. Installing the `locales` package above only
+# ships /etc/locale.gen - it does not build any locale, so the image had just
+# C, C.utf8 and POSIX. That made the `ENV LC_ALL=en_US.UTF-8` below unsatisfiable:
+# locale.setlocale() in flask_app.py failed, fell back to C, and the
+# format_number_locale / format_int_locale Jinja filters silently lost their
+# thousands separators - 1234567.89 rendered as "1234567.89" rather than
+# "1,234,567.89" in the restock/price overview, which is the very thing the
+# `locales` package was added for.
+#
+# More than en_US is generated so that operators can override LC_ALL / LANG and
+# get formatting for their own region (de_DE gives 1.234.567,89, fr_FR gives
+# 1 234 567,89). Costs ~21MB and ~16s of build time.
+#
+# This list mirrors the UI translations in changedetectionio/translations - one
+# glibc locale per language we ship a translation for, so any language a user
+# can pick in the UI also has a working locale. Keep the two in sync when adding
+# a translation. The territory for each bare language code comes from CLDR's
+# likely-subtags (cs -> cs_CZ, ja -> ja_JP, ko -> ko_KR, uk -> uk_UA, zh ->
+# zh_CN, zh_Hant_TW -> zh_TW), NOT from uppercasing the language code.
+RUN for l in cs_CZ de_DE en_GB en_US es_ES fr_FR id_ID it_IT ja_JP ko_KR \
+ pl_PL pt_BR ru_RU tr_TR uk_UA zh_CN zh_TW; do \
+ sed -i "s/^# *${l}.UTF-8 UTF-8/${l}.UTF-8 UTF-8/" /etc/locale.gen; \
+ done \
+ && locale-gen
+
# https://stackoverflow.com/questions/58701233/docker-logs-erroneously-appears-empty-until-container-stops
ENV PYTHONUNBUFFERED=1
diff --git a/LICENSE b/LICENSE
index 8410206a2..a6e09a451 100644
--- a/LICENSE
+++ b/LICENSE
@@ -186,7 +186,7 @@
same "printed page" as the copyright notice for easier
identification within third-party archives.
- Copyright 2025 Web Technologies s.r.o.
+ Copyright (c) Leigh Morresi and the changedetection.io contributors
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
diff --git a/MANIFEST.in b/MANIFEST.in
index dfcf0cdd6..3b5fe8c33 100644
--- a/MANIFEST.in
+++ b/MANIFEST.in
@@ -13,6 +13,7 @@ recursive-include changedetectionio/store *
recursive-include changedetectionio/templates *
recursive-include changedetectionio/tests *
recursive-include changedetectionio/translations *
+recursive-include changedetectionio/translations_overlay *
recursive-include changedetectionio/widgets *
prune changedetectionio/static/package-lock.json
prune changedetectionio/static/styles/node_modules
diff --git a/changedetectionio/__init__.py b/changedetectionio/__init__.py
index 1263e7635..f2722c535 100644
--- a/changedetectionio/__init__.py
+++ b/changedetectionio/__init__.py
@@ -2,7 +2,7 @@
# Read more https://github.com/dgtlmoon/changedetection.io/wiki
# Semver means never use .01, or 00. Should be .1.
-__version__ = '0.55.8'
+__version__ = '0.60.7'
from changedetectionio.strtobool import strtobool
from json.decoder import JSONDecodeError
@@ -628,9 +628,13 @@ def main():
@app.context_processor
def inject_template_globals():
from changedetectionio.llm.evaluator import get_llm_config as _get_llm_config
- return dict(right_sticky="v"+__version__,
+ from flask_login import current_user
+ has_password = datastore.data['settings']['application']['password'] != False
+ # Don't reveal the running version to anonymous visitors when password protection is enabled (#2190)
+ show_version = current_user.is_authenticated or not has_password
+ return dict(right_sticky="v"+__version__ if show_version else None,
new_version_available=app.config['NEW_VERSION_AVAILABLE'],
- has_password=datastore.data['settings']['application']['password'] != False,
+ has_password=has_password,
socket_io_enabled=datastore.data['settings']['application'].get('ui', {}).get('socket_io_enabled', True),
all_paused=datastore.data['settings']['application'].get('all_paused', False),
all_muted=datastore.data['settings']['application'].get('all_muted', False),
diff --git a/changedetectionio/api/Import.py b/changedetectionio/api/Import.py
index d622a6d58..86ff7a3cc 100644
--- a/changedetectionio/api/Import.py
+++ b/changedetectionio/api/Import.py
@@ -195,6 +195,15 @@ class Import(Resource):
urls_to_import.append(url)
+ # PAGE_WATCH_LIMIT - refuse the whole batch rather than importing an arbitrary prefix of
+ # it, so a 429 always means "nothing was created" and the caller can retry as-is
+ watch_limit = self.datastore.watch_limit
+ if watch_limit is not None:
+ current_watch_count = len(self.datastore.data['watching'])
+ if current_watch_count + len(urls_to_import) > watch_limit:
+ return (f"Watch limit reached ({current_watch_count}/{watch_limit} watches), importing "
+ f"{len(urls_to_import)} URL(s) would exceed it. No watches were imported.", 429)
+
# For small imports, process synchronously for immediate feedback
if len(urls_to_import) < IMPORT_SWITCH_TO_BACKGROUND_THRESHOLD:
added = []
diff --git a/changedetectionio/api/Tags.py b/changedetectionio/api/Tags.py
index ae09dfe29..0b001442e 100644
--- a/changedetectionio/api/Tags.py
+++ b/changedetectionio/api/Tags.py
@@ -10,6 +10,21 @@ from . import auth
from . import validate_openapi_request, strip_internal_api_fields
+def validate_tag_colour(json_data):
+ """Return an error message when tag_colour is set to anything but a hex colour.
+
+ The value is rendered into a
+ {%- endif %}
{{ pagination.links }}
diff --git a/changedetectionio/browser_steps/browser_steps.py b/changedetectionio/browser_steps/browser_steps.py
index 40b33ae66..73d0e621a 100644
--- a/changedetectionio/browser_steps/browser_steps.py
+++ b/changedetectionio/browser_steps/browser_steps.py
@@ -5,8 +5,42 @@ from random import randint
from loguru import logger
from changedetectionio.content_fetchers import SCREENSHOT_MAX_HEIGHT_DEFAULT
-from changedetectionio.content_fetchers.base import manage_user_agent
+from changedetectionio.content_fetchers.base import get_playwright_bypass_csp, manage_user_agent
from changedetectionio.jinja2_custom import render as jinja_render
+from changedetectionio.validate_url import validate_fetch_url_async
+
+def track_latest_navigation_response(page):
+ """Record the latest main-frame document response seen on this page, and return the holder.
+
+ Idempotent on purpose - every navigation would otherwise add another 'response' listener, and
+ that event fires once per HTTP response (hundreds on a heavy page), so the callbacks are worth
+ not duplicating. One tracker per page is installed and then shared by the fetcher and by every
+ action_goto_url() call on it.
+
+ Returns a dict that holds {'response': }, or None if the
+ page does not support event listeners (the unit test stubs, mainly).
+ """
+ if not hasattr(page, 'on'):
+ return None
+
+ existing = getattr(page, '_cdio_latest_navigation_response', None)
+ if existing is not None:
+ return existing
+
+ latest = {}
+
+ def _keep(response):
+ try:
+ if response.frame == page.main_frame and response.request.is_navigation_request():
+ latest['response'] = response
+ except Exception as e:
+ # Never let a bookkeeping listener break a fetch
+ logger.debug(f"Could not record navigation response: {e}")
+
+ page.on("response", _keep)
+ page._cdio_latest_navigation_response = latest
+ return latest
+
def browser_steps_get_valid_steps(browser_steps: list):
if browser_steps is not None and len(browser_steps):
@@ -135,9 +169,32 @@ class steppable_browser_interface():
if not value:
logger.warning("No URL provided for goto_url action")
return None
-
+
+ # Every browser navigation we initiate funnels through here - the "Goto URL" step, the
+ # "Goto site" step, the live Browser Steps UI and the Add Watch snapshot preview - so this
+ # is the one place that has to enforce the fetch rules. Step values are plain user-supplied
+ # strings (forms.SingleBrowserStep.optional_value, or the browser_steps[] API field) and are
+ # NOT covered by the watch URL validation, which is what made file:///etc/passwd readable
+ # and private-IP SSRF possible via a browser step (GHSA-hm22-wg2m-35v4).
+ await validate_fetch_url_async(value)
+
+ # Chrome 153+ refuses to commit a navigation when an error status arrives with a
+ # zero-length body, so page.goto() raises net::ERR_HTTP_RESPONSE_CODE_FAILURE instead of
+ # handing back the response. The response was received fine, we just never get it as a
+ # return value, so fall back to the page's navigation-response tracker and hand that back -
+ # callers then report a real "Error - 404" instead of a raw net:: string.
+ navigation_response = track_latest_navigation_response(self.page)
+
now = time.time()
- response = await self.page.goto(value, timeout=0, wait_until='load')
+ try:
+ response = await self.page.goto(value, timeout=0, wait_until='load')
+ except Exception as e:
+ if 'ERR_HTTP_RESPONSE_CODE_FAILURE' not in str(e) or not navigation_response:
+ raise
+ response = navigation_response['response']
+ logger.debug(f"Navigation was aborted by the browser (empty body on an error status), "
+ f"recovered status {response.status} from the response event")
+
logger.debug(f"Time to goto URL {time.time()-now:.2f}s")
return response
@@ -361,7 +418,7 @@ class browsersteps_live_ui(steppable_browser_interface):
# @todo handle multiple contexts, bind a unique id from the browser on each req?
context_kwargs = dict(
accept_downloads=False, # Should never be needed
- bypass_csp=True, # This is needed to enable JavaScript execution on GitHub and others
+ bypass_csp=get_playwright_bypass_csp(),
extra_http_headers=self.headers,
ignore_https_errors=True,
proxy=proxy,
@@ -514,4 +571,3 @@ class browsersteps_live_ui(steppable_browser_interface):
pass
return (screenshot, xpath_data)
-
diff --git a/changedetectionio/conditions/__init__.py b/changedetectionio/conditions/__init__.py
index 691ba3185..ff054d98d 100644
--- a/changedetectionio/conditions/__init__.py
+++ b/changedetectionio/conditions/__init__.py
@@ -34,7 +34,13 @@ CUSTOM_OPERATIONS = {
def filter_complete_rules(ruleset):
rules = [
rule for rule in ruleset
- if all(value not in ("", False, "None", None) for value in [rule["operator"], rule["field"], rule["value"]])
+ if all(
+ rule.get(k) is not None
+ and rule.get(k) is not False
+ and rule.get(k) != ""
+ and rule.get(k) != "None"
+ for k in ("operator", "field", "value")
+ )
]
return rules
@@ -54,12 +60,20 @@ def convert_to_jsonlogic(logic_operator: str, rule_dict: list):
field = condition["field"]
value = condition["value"]
- if not operator or operator == 'None' or not value or not field:
+ if (
+ not operator
+ or operator == 'None'
+ or not field
+ or value is None
+ or value is False
+ or value == ""
+ or value == "None"
+ ):
raise EmptyConditionRuleRowNotUsable()
# Convert value to int/float if possible
try:
- if isinstance(value, str) and "." in value and str != "None":
+ if isinstance(value, str) and "." in value and value != "None":
value = float(value)
else:
value = int(value)
diff --git a/changedetectionio/conditions/default_plugin.py b/changedetectionio/conditions/default_plugin.py
index ffc1f7296..3dc456952 100644
--- a/changedetectionio/conditions/default_plugin.py
+++ b/changedetectionio/conditions/default_plugin.py
@@ -1,7 +1,6 @@
import re
import pluggy
-from price_parser import Price
from loguru import logger
from flask_babel import lazy_gettext as _l
@@ -70,6 +69,7 @@ def register_field_choices():
@hookimpl
def add_data(current_watch_uuid, application_datastruct, ephemeral_data):
+ from price_parser import Price
res = {}
if 'text' in ephemeral_data:
res['page_filtered_text'] = ephemeral_data['text']
diff --git a/changedetectionio/conditions/plugins/levenshtein_plugin.py b/changedetectionio/conditions/plugins/levenshtein_plugin.py
index 7f5f6e2f1..f5fb8623d 100644
--- a/changedetectionio/conditions/plugins/levenshtein_plugin.py
+++ b/changedetectionio/conditions/plugins/levenshtein_plugin.py
@@ -14,7 +14,11 @@ global_hookimpl = pluggy.HookimplMarker("changedetectionio")
def levenshtein_ratio_recent_history(watch, incoming_text=None):
try:
- from Levenshtein import ratio, distance
+ # rapidfuzz (MIT) instead of Levenshtein (GPL-2.0-or-later), to keep the
+ # shipped deps free of strong copyleft. Indel.normalized_similarity is the
+ # exact equivalent of Levenshtein.ratio - Levenshtein.normalized_similarity is not.
+ from rapidfuzz.distance.Levenshtein import distance
+ from rapidfuzz.distance.Indel import normalized_similarity as ratio
k = list(watch.history.keys())
a = None
b = None
diff --git a/changedetectionio/content_fetchers/base.py b/changedetectionio/content_fetchers/base.py
index 7213bf355..8c7dddf97 100644
--- a/changedetectionio/content_fetchers/base.py
+++ b/changedetectionio/content_fetchers/base.py
@@ -4,6 +4,7 @@ from loguru import logger
from pydantic import BaseModel
from changedetectionio.content_fetchers import BrowserStepsStepException
+from changedetectionio.strtobool import strtobool
class FetcherCapabilities(BaseModel):
@@ -45,6 +46,16 @@ class FetcherCapabilities(BaseModel):
return bool(self.is_browser or self.supports_request_timeout or self.supports_custom_user_agent)
+def get_playwright_bypass_csp():
+ """Return whether Playwright-compatible browser contexts should bypass CSP.
+
+ Bypassing CSP remains enabled by default for backward compatibility. Some
+ remote CDP implementations do not support ``Page.setBypassCSP``; operators
+ can disable the option by setting ``PLAYWRIGHT_BYPASS_CSP=false``.
+ """
+ return strtobool(os.getenv('PLAYWRIGHT_BYPASS_CSP', 'true'))
+
+
def manage_user_agent(headers, current_ua=''):
"""
Basic setting of user-agent
@@ -110,6 +121,7 @@ class Fetcher():
screenshot_format = None
status_code = None
webdriver_js_execute_code = None
+ worker_id = None
xpath_data = None
xpath_element_js = ""
@@ -140,6 +152,11 @@ class Fetcher():
if kwargs and 'lock_viewport_elements' in kwargs:
self.lock_viewport_elements = kwargs.get('lock_viewport_elements')
+ # Which async worker is driving this fetch, subclasses use it to keep per-worker browser
+ # state (profile dirs etc) apart, stays None when we're not called from a worker
+ if kwargs and 'worker_id' in kwargs:
+ self.worker_id = kwargs.get('worker_id')
+
@classmethod
def get_status_icon_data(cls):
@@ -250,7 +267,11 @@ class Fetcher():
optional_value=optional_value)
await self.screenshot_step(step_n)
await self.save_step_html(step_n)
- except (Error, TimeoutError) as e:
+ except (Error, TimeoutError, ValueError) as e:
+ # ValueError is what validate_fetch_url_async() raises when a step's URL is
+ # refused (file://, private IP, bad scheme) - report it against the offending
+ # step number like any other step failure, rather than failing the whole watch
+ # with an opaque error.
logger.debug(str(e))
# Stop processing here
raise BrowserStepsStepException(step_n=step_n, original_e=e)
diff --git a/changedetectionio/content_fetchers/playwright.py b/changedetectionio/content_fetchers/playwright.py
index 59e97137c..6acef304a 100644
--- a/changedetectionio/content_fetchers/playwright.py
+++ b/changedetectionio/content_fetchers/playwright.py
@@ -8,7 +8,7 @@ from loguru import logger
from changedetectionio.content_fetchers import SCREENSHOT_MAX_HEIGHT_DEFAULT, visualselector_xpath_selectors, \
SCREENSHOT_SIZE_STITCH_THRESHOLD, SCREENSHOT_MAX_TOTAL_HEIGHT, XPATH_ELEMENT_JS, INSTOCK_DATA_JS, FAVICON_FETCHER_JS
-from changedetectionio.content_fetchers.base import Fetcher, manage_user_agent
+from changedetectionio.content_fetchers.base import Fetcher, get_playwright_bypass_csp, manage_user_agent
from changedetectionio.content_fetchers.exceptions import PageUnloadable, Non200ErrorCodeReceived, EmptyReply, ScreenshotUnavailable, \
BrowserStepsStepException
@@ -304,7 +304,9 @@ class fetcher(Fetcher):
# (locale also drives the Accept-Language header - see issues #4210 #1412; timezone #3303)
context_kwargs = dict(
accept_downloads=False, # Should never be needed
- bypass_csp=True, # This is needed to enable JavaScript execution on GitHub and others
+ # Enabled by default because sites such as GitHub need it for injected JavaScript.
+ # Some CDP implementations do not support Page.setBypassCSP, so allow operators to disable it.
+ bypass_csp=get_playwright_bypass_csp(),
extra_http_headers=request_headers,
ignore_https_errors=True,
proxy=self.proxy,
@@ -324,6 +326,20 @@ class fetcher(Fetcher):
self.page = await context.new_page()
+ # Track the LATEST main-frame document response for the whole fetch, not just the one
+ # goto() returns. This app compares the text of the page the browser ends up on, and a
+ # site that gates with an interstitial (503/429 + meta-refresh) navigates to the real
+ # page *during* the extra_wait below - judging the fetch on the first response fails a
+ # watch whose content is present and fine. Same for plain client-side redirects.
+ # Shared with action_goto_url() so only one 'response' listener exists on the page.
+ from changedetectionio.browser_steps.browser_steps import track_latest_navigation_response
+ # Must be an identity check - the tracker hands back the same (initially empty, so
+ # falsy) dict the listener writes into, and `or {}` would quietly swap in a different
+ # one that never gets updated.
+ latest_navigation_response = track_latest_navigation_response(self.page)
+ if latest_navigation_response is None:
+ latest_navigation_response = {}
+
# Listen for all console events and handle errors
self.page.on("console", lambda msg: logger.debug(f"Playwright console: Watch URL: {url} {msg.type}: {msg.text} {msg.args}"))
@@ -364,6 +380,25 @@ class fetcher(Fetcher):
extra_wait = int(os.getenv("WEBDRIVER_DELAY_BEFORE_CONTENT_READY", 5)) + self.render_extract_delay
await self.page.wait_for_timeout(extra_wait * 1000)
+ # A meta-refresh or client-side redirect usually lands during that wait, so judge the
+ # fetch on the document we are actually about to extract rather than the first one.
+ latest = latest_navigation_response.get('response')
+ if latest is not None and latest is not response:
+ logger.debug(f"Page navigated again while waiting, judging the fetch on {latest.url} "
+ f"(status {latest.status}) instead of the first response for {url}")
+ response = latest
+ try:
+ self.headers = await response.all_headers()
+ except Exception as e:
+ logger.debug(f"Could not refresh headers from the final document: {e}")
+
+ # Don't extract while a navigation is mid-flight, that is what produces
+ # "Execution context was destroyed, most likely because of a navigation"
+ try:
+ await self.page.wait_for_load_state('load', timeout=extra_wait * 1000)
+ except Exception as e:
+ logger.debug(f"Page did not reach a settled load state, continuing anyway: {e}")
+
try:
self.status_code = response.status
except Exception as e:
@@ -501,4 +536,3 @@ class PlaywrightFetcherPlugin:
playwright_plugin = PlaywrightFetcherPlugin()
-
diff --git a/changedetectionio/content_fetchers/puppeteer.py b/changedetectionio/content_fetchers/puppeteer.py
index 41775ce16..4fa2cac53 100644
--- a/changedetectionio/content_fetchers/puppeteer.py
+++ b/changedetectionio/content_fetchers/puppeteer.py
@@ -1,5 +1,4 @@
import asyncio
-import gc
import json
import os
import websockets.exceptions
@@ -10,11 +9,18 @@ from loguru import logger
from changedetectionio.content_fetchers import SCREENSHOT_MAX_HEIGHT_DEFAULT, visualselector_xpath_selectors, \
SCREENSHOT_SIZE_STITCH_THRESHOLD, SCREENSHOT_DEFAULT_QUALITY, XPATH_ELEMENT_JS, INSTOCK_DATA_JS, \
SCREENSHOT_MAX_TOTAL_HEIGHT, FAVICON_FETCHER_JS
-from changedetectionio.content_fetchers.base import Fetcher, manage_user_agent
+from changedetectionio import gc_debounce
+from changedetectionio.content_fetchers.base import Fetcher, get_playwright_bypass_csp, manage_user_agent
from changedetectionio.content_fetchers.exceptions import PageUnloadable, Non200ErrorCodeReceived, EmptyReply, BrowserFetchTimedOut, \
BrowserConnectError
+async def _configure_puppeteer_csp(page):
+ """Enable CSP bypass without requiring unsupported CDP methods when disabled."""
+ if get_playwright_bypass_csp():
+ await page.setBypassCSP(True)
+
+
# Bug 3 in Playwright screenshot handling
# Some bug where it gives the wrong screenshot size, but making a request with the clip set first seems to solve it
@@ -198,6 +204,10 @@ class fetcher(Fetcher):
def __init__(self, proxy_override=None, custom_browser_connection_url=None, **kwargs):
super().__init__(**kwargs)
+ # Renderer crashes recorded during the fetch, see the 'error' handler in fetch_page().
+ # Set up here so run()'s finally can always report, even if we never got as far as a page.
+ self.page_errors = []
+
if custom_browser_connection_url:
self.browser_connection_is_custom = True
self.browser_connection_url = custom_browser_connection_url
@@ -227,6 +237,7 @@ class fetcher(Fetcher):
async def quit(self, watch=None):
watch_uuid = watch.get('uuid') if watch else 'unknown'
+ closed_something = bool(getattr(self, 'page', None) or getattr(self, 'browser', None))
# Close page
try:
@@ -254,8 +265,16 @@ class fetcher(Fetcher):
logger.info(f"[{watch_uuid}] Cleanup puppeteer complete")
- # Force garbage collection to release resources
- gc.collect()
+ # Only collect if this call actually closed something.
+ #
+ # quit() runs twice per check - from run()'s finally, then again from the worker's
+ # safety net - and it sets self.page/self.browser to None in its own finally
+ # blocks. The second call therefore closes nothing, creates no garbage and breaks
+ # no cycles, but still paid for a full stop-the-world collection: measured at 88
+ # calls across 51 checks, roughly half of them reclaiming nothing. The pyppeteer
+ # page/connection/session graph is genuinely cyclic, so the first call still runs.
+ if closed_something:
+ gc_debounce.collect('puppeteer.quit')
async def fetch_page(self,
current_include_filters,
@@ -313,6 +332,23 @@ class fetcher(Fetcher):
self.browser = None
raise
+ # A renderer crash makes pyppeteer emit Page 'error' (PageError('Page crashed!')).
+ # pyee re-raises an 'error' emission that has no listener, and that raise escapes into
+ # Connection._onMessage, whose catch-all disposes the entire connection - so one dead
+ # tab takes the whole browser with it and every later call reports the misleading
+ # "Session closed. Most likely the page has been closed." Attaching a listener keeps
+ # the failure local, named, and recoverable.
+ #
+ # A single page load can emit 'error' more than once (an iframe renderer going down, then
+ # the main one), so collect them all rather than keeping only the last.
+ self.page_errors = []
+
+ def _handle_page_error(e):
+ self.page_errors.append(e)
+ logger.error(f"[{watch_uuid}] Page error (the renderer likely crashed, often OOM): {e}")
+
+ self.page.on('error', _handle_page_error)
+
# Add console handler to capture console.log from favicon fetcher
#self.page.on('console', lambda msg: logger.debug(f"Browser console [{msg.type}]: {msg.text}"))
@@ -348,7 +384,7 @@ class fetcher(Fetcher):
# Attempt to strip 'HeadlessChrome' etc
await self.page.setUserAgent(manage_user_agent(headers=request_headers, current_ua=await self.page.evaluate('navigator.userAgent')))
- await self.page.setBypassCSP(True)
+ await _configure_puppeteer_csp(self.page)
if request_headers:
await self.page.setExtraHTTPHeaders(request_headers)
@@ -373,56 +409,192 @@ class fetcher(Fetcher):
# Enable Network domain to detect when first bytes arrive
await self.page._client.send('Network.enable')
- # Now set up the frame navigation handlers
- async def handle_frame_navigation(event=None):
- # Wait n seconds after the frameStartedLoading, not from any frameStartedLoading/frameStartedNavigating
- logger.debug(f"Frame navigated: {event}")
- w = extra_wait - 2 if extra_wait > 4 else 2
- logger.debug(f"Waiting {w} seconds before calling Page.stopLoading...")
- await asyncio.sleep(w)
+ # Navigate (bounded), then wait the configured "wait n seconds before extracting text"
+ # delay, then stop whatever is still loading, then extract. The delay is measured from
+ # when navigation finished, not from when it started, because the point of it is to let
+ # JS-rendered content appear *after* load - anchoring it to the start would quietly give a
+ # slow-loading page almost no settle time.
+ #
+ # Only that delay is a user-facing setting. The navigation bound above is a safety net with
+ # a sane default, not a tuning knob, so there is still one number for users to think about.
+ #
+ # There is no way to know a page is "finished" - plenty of sites navigate as part of their
+ # normal design, and some sit forever on a subresource that never answers. So the delay
+ # restarts if the MAIN frame replaces its document (a redirect or interstitial gets the
+ # same settle time the first document got), iframes do not restart it, and it is capped so
+ # a page that re-navigates in a loop cannot extend it indefinitely.
+ max_content_ready_resets = int(os.getenv("BROWSER_CONTENT_READY_MAX_RESETS", 2))
- # Check if page still exists (might have been closed due to error during sleep)
- if not self.page or not hasattr(self.page, '_client'):
- logger.debug("Page already closed, skipping stopLoading")
- return
+ async def wait_for_content_ready_then_stop_loading():
+ main_frame_id = self.page.mainFrame._id
+ renavigated = asyncio.Event()
- logger.debug("Issuing stopLoading command...")
- await self.page._client.send('Page.stopLoading')
- logger.debug("stopLoading command sent!")
+ def _on_main_frame_navigation(event):
+ if event.get('frameId') == main_frame_id:
+ renavigated.set()
- async def setup_frame_handlers_on_first_response(event):
- # Only trigger for the main document response
- if event.get('type') == 'Document':
- logger.debug("First response received, setting up frame handlers for forced page stop load.")
- self.page._client.on('Page.frameStartedNavigating', lambda e: asyncio.create_task(handle_frame_navigation(e)))
- self.page._client.on('Page.frameStartedLoading', lambda e: asyncio.create_task(handle_frame_navigation(e)))
- self.page._client.on('Page.frameStoppedLoading', lambda e: logger.debug(f"Frame stopped loading: {e}"))
- logger.debug("First response received, setting up frame handlers for forced page stop load DONE SETUP")
- # De-register this listener - we only need it once
- self.page._client.remove_listener('Network.responseReceived', setup_frame_handlers_on_first_response)
+ self.page._client.on('Page.frameStartedLoading', _on_main_frame_navigation)
+ self.page._client.on('Page.frameStoppedLoading', lambda e: logger.debug(f"Frame stopped loading: {e}"))
+ try:
+ resets = 0
+ while True:
+ renavigated.clear()
+ try:
+ await asyncio.wait_for(renavigated.wait(), timeout=extra_wait)
+ # Main frame started a new document
+ resets += 1
+ if resets > max_content_ready_resets:
+ # The cap is there to stop a page that re-navigates in a loop from
+ # extending the fetch forever - it is NOT permission to extract
+ # immediately. Breaking straight out here landed on whatever document
+ # happened to be mid-flight, with zero settle time: measured against a
+ # page that hops every 500ms, the fetch ended after 2.7s holding 130
+ # bytes of an intermediate hop, no final document and no JS-rendered
+ # content, while logging "content-ready wait of 12s elapsed".
+ #
+ # So spend the delay one last time, just without arming another reset.
+ # Total stays bounded at (max_resets + 2) * extra_wait, and whatever we
+ # extract has had the same settle time every other fetch gets.
+ logger.debug(f"Main frame re-navigated {resets} times (cap "
+ f"{max_content_ready_resets}), waiting {extra_wait}s once "
+ f"more without restarting, then extracting regardless")
+ await asyncio.sleep(extra_wait)
+ break
+ logger.debug(f"Main frame started a new document, restarting the {extra_wait}s "
+ f"content-ready wait ({resets}/{max_content_ready_resets})")
+ except asyncio.TimeoutError:
+ # Quiet for the whole delay - the page is as ready as it is going to get
+ break
+ finally:
+ self.page._client.remove_listener('Page.frameStartedLoading', _on_main_frame_navigation)
- # Listen for first response to trigger frame handler setup
- self.page._client.on('Network.responseReceived', setup_frame_handlers_on_first_response)
+ # Stop whatever is still in flight so the DOM and screenshot come from what rendered,
+ # rather than waiting on a subresource that may never answer
+ try:
+ logger.debug(f"Content-ready wait finished, issuing Page.stopLoading before extracting")
+ await self.page._client.send('Page.stopLoading')
+ logger.debug("stopLoading command sent!")
+
+ # stopLoading stops the network, not script execution. A page whose JS has pegged
+ # the renderer's main thread (a runaway loop, a rAF that never settles) holds that
+ # thread indefinitely, and every CDP call that needs to run script then queues
+ # behind it and never returns - page.content, the xPath scraper, the favicon
+ # fetcher. The fetch dies at PUPPETEER_MAX_PROCESSING_TIMEOUT_SECONDS having
+ # extracted nothing, with a core spinning the entire time.
+ #
+ # Nothing else recovers this. Runtime.evaluate's own `timeout` parameter bounds an
+ # evaluation once it starts, not time spent queued behind the running task, and
+ # wrapping the call in asyncio.wait_for is worse than useless: cancelling a
+ # pyppeteer request mid-flight leaves the connection unusable ("Target closed" on
+ # everything after it). Terminating execution is what releases the thread -
+ # measured against a deliberately spinning page, extraction went from timing out
+ # to returning the full DOM in 0.0s and the renderer dropped from 1.00 to 0.08
+ # cores. Safe here because stopLoading has already declared "give me what
+ # rendered", and the content-ready wait above has already had its chance to let
+ # late JS-rendered content appear.
+ await self.page._client.send('Runtime.terminateExecution')
+ logger.debug("Runtime.terminateExecution sent, any runaway page script is stopped")
+ except Exception as e:
+ logger.debug(f"Page.stopLoading/Runtime.terminateExecution skipped, page is most "
+ f"likely already gone: {e}")
+
+ # Track the LATEST main-frame document response for the whole fetch, not just the one that
+ # goto() happens to return. This app compares the text of the page the browser ends up on,
+ # and plenty of sites navigate again after the first response:
+ # - an interstitial answering 503/429 with a meta-refresh into the real 200 page, where
+ # judging the first response fails a watch whose content is sitting right there
+ # - a plain client-side redirect to another host (slated.com -> get.slated.com)
+ # It also covers Chrome 153+, which refuses to commit a navigation when an error status
+ # arrives with a zero-length body: goto() raises net::ERR_HTTP_RESPONSE_CODE_FAILURE rather
+ # than returning the response, but the response itself still arrives on this event.
+ navigation_response = {}
+
+ def _keep_navigation_response(response):
+ # Note pyppeteer exposes these as properties, unlike playwright where they are methods
+ if response.frame == self.page.mainFrame and response.request.isNavigationRequest:
+ navigation_response['response'] = response
+
+ self.page.on('response', _keep_navigation_response)
+
+ # pyppeteer's navigation watcher is bound to the loaderId of the navigation it started. If
+ # the page replaces that document (redirect/interstitial) the 'load' it waits for never
+ # arrives for that loaderId, so goto() never returns - and with timeout=0 it would block
+ # until the hard PUPPETEER_MAX_PROCESSING_TIMEOUT_SECONDS kill, burning a worker slot for
+ # minutes on a page that is fully loaded. Bound it, then fall back to the document we can
+ # see. Verified against slated.com and getastra.com, which hang indefinitely otherwise.
+ nav_timeout = int(os.getenv("BROWSER_NAVIGATION_TIMEOUT_SECONDS", 30))
response = None
attempt=0
- while not response:
- logger.debug(f"Attempting page fetch {url} attempt {attempt}")
- asyncio.create_task(handle_frame_navigation())
- response = await self.page.goto(url, timeout=0)
- await asyncio.sleep(1 + extra_wait)
- # Check if page still exists before sending command
- if self.page and hasattr(self.page, '_client'):
- await self.page._client.send('Page.stopLoading')
+ try:
+ while not response:
+ logger.debug(f"Attempting page fetch {url} attempt {attempt}")
+ # Race goto() against the main frame actually firing 'load'. In the re-navigation
+ # case goto() can never resolve, but the replacement document does fire 'load' -
+ # usually within a few seconds - so this returns then instead of sitting out the
+ # whole nav_timeout. Whichever arrives first means "the document is loaded".
+ main_frame_loaded = asyncio.Event()
+ main_frame_id = self.page.mainFrame._id
- if response:
- break
- if not response:
- logger.warning("Page did not fetch! trying again!")
- if response is None and attempt>=2:
- logger.warning(f"Content Fetcher > Response object was none (as in, the response from the browser was empty, not just the content) exiting attempt {attempt}")
- raise EmptyReply(url=url, status_code=None)
- attempt+=1
+ def _on_lifecycle(event):
+ if event.get('name') == 'load' and event.get('frameId') == main_frame_id:
+ main_frame_loaded.set()
+
+ self.page._client.on('Page.lifecycleEvent', _on_lifecycle)
+ goto_task = asyncio.ensure_future(self.page.goto(url, timeout=0))
+ load_task = asyncio.ensure_future(main_frame_loaded.wait())
+ try:
+ done, _pending = await asyncio.wait({goto_task, load_task},
+ timeout=nav_timeout,
+ return_when=asyncio.FIRST_COMPLETED)
+
+ if goto_task in done:
+ try:
+ response = goto_task.result()
+ except Exception as e:
+ if 'ERR_HTTP_RESPONSE_CODE_FAILURE' not in str(e) or not navigation_response:
+ raise
+ response = navigation_response['response']
+ logger.debug(f"Navigation was aborted by the browser (empty body on an error status), "
+ f"recovered status {response.status} from the response event")
+ else:
+ # Either the replacement document loaded, or we ran out of patience
+ response = navigation_response.get('response')
+ if not response:
+ raise BrowserFetchTimedOut(msg=f"Browser did not finish navigating to {url} within "
+ f"{nav_timeout}s and no main-frame response was seen.")
+ why = ("the page replaced the document it started on" if load_task in done
+ else f"navigation did not settle within {nav_timeout}s")
+ logger.warning(f"Continuing with the document actually loaded ({why}) - "
+ f"status {response.status} for {response.url}")
+ finally:
+ self.page._client.remove_listener('Page.lifecycleEvent', _on_lifecycle)
+ for t in (goto_task, load_task):
+ if not t.done():
+ t.cancel()
+ if response:
+ break
+ if not response:
+ logger.warning("Page did not fetch! trying again!")
+ if response is None and attempt>=2:
+ logger.warning(f"Content Fetcher > Response object was none (as in, the response from the browser was empty, not just the content) exiting attempt {attempt}")
+ raise EmptyReply(url=url, status_code=None)
+ attempt+=1
+
+ # Navigation is done; now honour "wait n seconds before extracting text" and then
+ # force-stop whatever is still loading, so extraction always gets what rendered.
+ # Awaited inline rather than fired off as a task, so nothing can outlive the fetch.
+ await wait_for_content_ready_then_stop_loading()
+
+ # That wait is where a meta-refresh interstitial typically swaps in the real page, so
+ # re-check which document we are actually on before judging the status code.
+ latest = navigation_response.get('response')
+ if latest is not None and latest is not response:
+ logger.debug(f"Page navigated again while waiting, judging the fetch on {latest.url} "
+ f"(status {latest.status}) instead of {response.url} (status {response.status})")
+ response = latest
+ finally:
+ self.page.remove_listener('response', _keep_navigation_response)
self.headers = response.headers
@@ -483,8 +655,7 @@ class fetcher(Fetcher):
self.screenshot = await capture_full_page(page=self.page, screenshot_format=self.screenshot_format, watch_uuid=watch_uuid, lock_viewport_elements=self.lock_viewport_elements)
# Force garbage collection - pyppeteer base64 decode creates temporary buffers
- import gc
- gc.collect()
+ gc_debounce.collect('puppeteer.after_screenshot')
self.xpath_data = await self.page.evaluate(XPATH_ELEMENT_JS, {
"visualselector_xpath_selectors": visualselector_xpath_selectors,
"max_height": MAX_TOTAL_HEIGHT
@@ -539,6 +710,15 @@ class fetcher(Fetcher):
except asyncio.TimeoutError:
raise (BrowserFetchTimedOut(msg=f"Browser connected but was unable to process the page in {max_time} seconds."))
finally:
+ # Nothing consumes page_errors yet, but a crashed renderer usually means the content we
+ # just extracted is partial or stale, so always say so - otherwise the only clue is a
+ # confusing downstream error (or worse, a silently wrong "change detected").
+ if self.page_errors:
+ logger.warning(
+ f"[{watch_uuid}] {len(self.page_errors)} page error(s) during this fetch of '{url}', "
+ f"content may be incomplete: {'; '.join(str(e) for e in self.page_errors)}"
+ )
+
# Internal cleanup on any exception/timeout - call quit() immediately
# This prevents connection leaks during exception bursts
# Worker.py's quit() call becomes a redundant safety net (idempotent)
diff --git a/changedetectionio/content_fetchers/requests.py b/changedetectionio/content_fetchers/requests.py
index 9d51fe961..cdc965a5d 100644
--- a/changedetectionio/content_fetchers/requests.py
+++ b/changedetectionio/content_fetchers/requests.py
@@ -9,7 +9,7 @@ import asyncio
from changedetectionio import strtobool
from changedetectionio.content_fetchers.exceptions import BrowserStepsInUnsupportedFetcher, EmptyReply, Non200ErrorCodeReceived
from changedetectionio.content_fetchers.base import Fetcher
-from changedetectionio.validate_url import is_private_hostname, is_url_private_or_parser_confused
+from changedetectionio.validate_url import is_fetch_url_allowed, is_private_hostname, is_url_private_or_parser_confused
# "html_requests" is listed as the default fetcher in store.py!
@@ -103,13 +103,12 @@ class fetcher(Fetcher):
try:
# Fresh DNS check at fetch time — catches DNS rebinding regardless of add-time cache.
- # Validates every hostname both urlparse and urllib3 see, so parser-differential
- # payloads (GHSA-rph4-96w6-q594) cannot smuggle an internal target past the gate.
- if not allow_iana_restricted:
- if is_url_private_or_parser_confused(url):
- raise Exception(f"Fetch blocked: '{url}' resolves to a private/reserved IP address "
- f"or contains a parser-differential payload. "
- f"Set ALLOW_IANA_RESTRICTED_ADDRESSES=true to allow.")
+ # Shared with every other fetch entry point, so the scheme allowlist and the
+ # parser-differential rejection (GHSA-rph4-96w6-q594) stay in step here too.
+ # Per-redirect-hop re-validation is done separately in the loop below.
+ ok, reason = is_fetch_url_allowed(url)
+ if not ok:
+ raise Exception(reason)
r = session.request(method=request_method,
data=request_body.encode('utf-8') if type(request_body) is str else request_body,
diff --git a/changedetectionio/content_fetchers/res/favicon-fetcher.js b/changedetectionio/content_fetchers/res/favicon-fetcher.js
index afd564dbc..2ba757563 100644
--- a/changedetectionio/content_fetchers/res/favicon-fetcher.js
+++ b/changedetectionio/content_fetchers/res/favicon-fetcher.js
@@ -54,57 +54,98 @@
return 0;
});
- const timeoutMs = 2000;
+ // All candidates are fetched concurrently under one shared deadline.
+ //
+ // Sequentially, each icon got its own fresh 2s AbortController, so a site declaring
+ // five variants spent 10s+ here - inside the page, holding a browser
+ // and a worker the whole time. Simply capping the total made it worse: giving up early
+ // returns no icon, nothing gets saved, favicon_is_expired() stays true and the cost is
+ // paid again on the very next check, forever. Fetching in parallel bounds the wall time
+ // *and* still finds a working icon, so it saves and the watch stops asking.
+ //
+ // Measured against a page with five hanging icons, one 404 and one good one:
+ // sequential, 2s each : 10.1s, found the icon
+ // total cap only : 3.0s, found nothing (then repeats every check)
+ // parallel + deadline : ~3s, found the icon
+ const TOTAL_BUDGET_MS = 3000;
// 1 MB — matches the server-side limit in bump_favicon()
const MAX_BYTES = 1 * 1024 * 1024;
- for (const icon of icons) {
+ const toBase64 = (blob) => new Promise(resolve => {
+ // Always resolves. The previous version resolved only from onloadend and read
+ // reader.result unguarded, so a FileReader failure threw inside the callback and left
+ // the promise permanently pending - the whole favicon fetch then hung with nothing
+ // bounding it, because clearTimeout had already fired.
+ try {
+ const reader = new FileReader();
+ reader.onerror = () => resolve(null);
+ reader.onloadend = () => {
+ try {
+ const result = reader.result;
+ resolve(result ? String(result).split(',')[1] : null);
+ } catch (e) {
+ resolve(null);
+ }
+ };
+ reader.readAsDataURL(blob);
+ } catch (e) {
+ resolve(null);
+ }
+ });
+
+ const controller = new AbortController();
+ const budget = setTimeout(() => controller.abort(), TOTAL_BUDGET_MS);
+
+ const fetchOne = async (icon) => {
try {
// Inline data URI — no network fetch needed, data is already here
if (icon.href.startsWith('data:')) {
const match = icon.href.match(/^data:([^;]+);base64,([A-Za-z0-9+/=]+)$/);
- if (!match) continue;
+ if (!match) return null;
const mime_type = match[1];
const base64 = match[2];
// Rough size check: base64 is ~4/3 the binary size
- if (base64.length * 0.75 > MAX_BYTES) continue;
+ if (base64.length * 0.75 > MAX_BYTES) return null;
return { url: icon.href, mime_type, base64 };
}
- const controller = new AbortController();
- const timeout = setTimeout(() => controller.abort(), timeoutMs);
-
const resp = await fetch(icon.href, {
signal: controller.signal,
redirect: 'follow'
});
- clearTimeout(timeout);
+ if (!resp.ok) return null;
- if (!resp.ok) {
- continue;
- }
+ // Skip an oversized icon before pulling its body down the wire, where the server
+ // tells us the size up front.
+ const declared = parseInt(resp.headers.get('content-length') || '0', 10);
+ if (declared > MAX_BYTES) return null;
+ // Still covered by the shared signal: aborting errors the body stream too. The
+ // previous version cleared its timer before this line, leaving a slow or
+ // never-ending body read completely unguarded.
const blob = await resp.blob();
+ if (blob.size > MAX_BYTES) return null;
- if (blob.size > MAX_BYTES) continue;
-
- // Convert blob to base64
- const reader = new FileReader();
- return await new Promise(resolve => {
- reader.onloadend = () => {
- resolve({
- url: icon.href,
- mime_type: blob.type,
- base64: reader.result.split(",")[1]
- });
- };
- reader.readAsDataURL(blob);
- });
+ const base64 = await toBase64(blob);
+ if (!base64) return null;
+ return { url: icon.href, mime_type: blob.type, base64 };
} catch (e) {
- continue;
+ return null;
}
+ };
+
+ try {
+ const settled = await Promise.all(icons.map(fetchOne));
+ // icons[] is already in preference order (largest, then apple-touch-icon), so the
+ // first success in that order is the one we want - not merely the fastest to answer.
+ const best = settled.find(r => r);
+ if (best) return best;
+ } catch (e) {
+ // fall through to "nothing found"
+ } finally {
+ clearTimeout(budget);
}
// nothing found
diff --git a/changedetectionio/content_fetchers/res/stock-not-in-stock.js b/changedetectionio/content_fetchers/res/stock-not-in-stock.js
index 82673c105..4fd20d10e 100644
--- a/changedetectionio/content_fetchers/res/stock-not-in-stock.js
+++ b/changedetectionio/content_fetchers/res/stock-not-in-stock.js
@@ -108,7 +108,14 @@ async () => {
'品切れ',
'已售',
'已售完',
- '품절'
+ '품절',
+ "товар закінчився",
+ "немає в наявності",
+ "нема в наявності",
+ "закінчився",
+ "знято з виробництва",
+ "недоступний",
+ "немає на складі"
];
diff --git a/changedetectionio/content_fetchers/res/xpath_element_scraper.js b/changedetectionio/content_fetchers/res/xpath_element_scraper.js
index 25bd5ad98..12753ca3f 100644
--- a/changedetectionio/content_fetchers/res/xpath_element_scraper.js
+++ b/changedetectionio/content_fetchers/res/xpath_element_scraper.js
@@ -13,7 +13,10 @@ async (options) => {
// Include the getXpath script directly, easier than fetching
function getxpath(e) {
var n = e;
- if (n && n.id) return '//*[@id="' + n.id + '"]';
+ // A double quote cannot be expressed inside a double-quoted xpath literal, and an angle
+ // bracket would carry page markup into whatever displays the selector later, so for those
+ // ids build the positional path below instead of an unusable '//*[@id="..."]'.
+ if (n && n.id && !/["<>]/.test(n.id)) return '//*[@id="' + n.id + '"]';
for (var o = []; n && Node.ELEMENT_NODE === n.nodeType;) {
for (var i = 0, r = !1, d = n.previousSibling; d;) d.nodeType !== Node.DOCUMENT_TYPE_NODE && d.nodeName === n.nodeName && i++, d = d.previousSibling;
for (d = n.nextSibling; d;) {
diff --git a/changedetectionio/flask_app.py b/changedetectionio/flask_app.py
index 461e53f1d..f9b0c73e9 100644
--- a/changedetectionio/flask_app.py
+++ b/changedetectionio/flask_app.py
@@ -1,6 +1,7 @@
#!/usr/bin/env python3
-import flask_login
+import gc
+import hashlib
import locale
import os
import queue
@@ -8,19 +9,17 @@ import re
import sys
import threading
import time
+from pathlib import Path
+from threading import Event
+
+import flask_login
import timeago
from blinker import signal
-from pathlib import Path
-
-from changedetectionio.strtobool import strtobool
-from threading import Event
-from changedetectionio.queue_handlers import RecheckPriorityQueue, NotificationQueue
-from changedetectionio import worker_pool
-
from flask import (
Flask,
abort,
flash,
+ g,
redirect,
render_template,
request,
@@ -28,23 +27,48 @@ from flask import (
session,
url_for,
)
-from flask_restful import abort, Api
+from flask.sessions import SecureCookieSessionInterface
from flask_cors import CORS
+from flask_restful import Api, abort
+
+from changedetectionio import worker_pool
+from changedetectionio.queue_handlers import NotificationQueue, RecheckPriorityQueue
+from changedetectionio.strtobool import strtobool
# Create specific signals for application events
# Make this a global singleton to avoid multiple signal objects
watch_check_update = signal('watch_check_update', doc='Signal sent when a watch check is completed')
+from flask_babel import Babel, get_locale, gettext
from flask_wtf import CSRFProtect
-from flask_babel import Babel, gettext, get_locale
from loguru import logger
-from changedetectionio import __version__
-from changedetectionio import queuedWatchMetaData
-from changedetectionio.api import Watch, WatchHistory, WatchSingleHistory, WatchHistoryDiff, CreateWatch, Import, SystemInfo, Tag, Tags, Notifications, WatchFavicon, Spec
+from changedetectionio import __version__, queuedWatchMetaData
+from changedetectionio.api import (
+ CreateWatch,
+ Import,
+ Notifications,
+ Spec,
+ SystemInfo,
+ Tag,
+ Tags,
+ Watch,
+ WatchFavicon,
+ WatchHistory,
+ WatchHistoryDiff,
+ WatchSingleHistory,
+)
from changedetectionio.api.Search import Search
-from .time_handler import is_within_schedule
-from changedetectionio.languages import get_available_languages, get_language_codes, get_flag_for_locale, get_timeago_locale
+from changedetectionio.blueprint.menu_modes import MENU_SIDEBAR_ACTIONMODES, MENU_SIDEBAR_ACTIONMODES_DEFAULT
from changedetectionio.favicon_utils import get_favicon_mime_type
+from changedetectionio.languages import (
+ get_available_languages,
+ get_flag_for_locale,
+ get_language_codes,
+ get_timeago_locale,
+)
+
+from .thread_supervisor import start_supervised_thread
+from .time_handler import default_timezone_name, is_within_schedule
IN_PYTEST = "pytest" in sys.modules or "PYTEST_CURRENT_TEST" in os.environ
@@ -54,24 +78,23 @@ datastore = None
ticker_thread = None
extra_stylesheets = []
-# Use bulletproof janus-based queues for sync/async reliability
+# Use bulletproof janus-based queues for sync/async reliability
update_q = RecheckPriorityQueue()
notification_q = NotificationQueue()
MAX_QUEUE_SIZE = 5000
-app = Flask(__name__,
- static_url_path="",
- static_folder="static",
- template_folder="templates")
+app = Flask(__name__, static_url_path="", static_folder="static", template_folder="templates")
# Will be initialized in changedetection_app
socketio_server = None
# Enable CORS, especially useful for the Chrome extension to operate from anywhere
CORS(app)
-from werkzeug.routing import BaseConverter, ValidationError
from uuid import UUID
+from werkzeug.routing import BaseConverter, ValidationError
+
+
class StrictUUIDConverter(BaseConverter):
# Special sentinel values allowed in addition to strict UUIDs
_ALLOWED_SENTINELS = frozenset({'first'})
@@ -91,6 +114,7 @@ class StrictUUIDConverter(BaseConverter):
def to_url(self, value) -> str:
return str(value)
+
# app setup (once)
app.url_map.converters["uuid_str"] = StrictUUIDConverter
@@ -99,8 +123,16 @@ app.url_map.converters["uuid_str"] = StrictUUIDConverter
# It's better to use compression on your reverse proxy (nginx etc) instead.
if strtobool(os.getenv("FLASK_ENABLE_COMPRESSION")):
from flask_compress import Compress as FlaskCompress
+
app.config['COMPRESS_MIN_SIZE'] = 2096
- app.config['COMPRESS_MIMETYPES'] = ['text/html', 'text/css', 'text/javascript', 'application/json', 'application/javascript', 'image/svg+xml']
+ app.config['COMPRESS_MIMETYPES'] = [
+ 'text/html',
+ 'text/css',
+ 'text/javascript',
+ 'application/json',
+ 'application/javascript',
+ 'image/svg+xml',
+ ]
# Use gzip only - smaller memory footprint than zstd/brotli (4-8KB vs 200-500KB contexts)
app.config['COMPRESS_ALGORITHM'] = ['gzip']
compress = FlaskCompress()
@@ -109,7 +141,8 @@ if strtobool(os.getenv("FLASK_ENABLE_COMPRESSION")):
app.config['TEMPLATES_AUTO_RELOAD'] = False
-# Stop browser caching of assets
+# Default to revalidate-always for anything served with send_file(); static_content() then
+# opts the fingerprinted asset URLs into real caching (see _fingerprint_static_urls).
app.config['SEND_FILE_MAX_AGE_DEFAULT'] = 0
app.config.exit = Event()
@@ -119,7 +152,28 @@ if os.getenv('FLASK_SERVER_NAME'):
app.config['SERVER_NAME'] = os.getenv('FLASK_SERVER_NAME')
# Babel/i18n configuration
-app.config['BABEL_TRANSLATION_DIRECTORIES'] = str(Path(__file__).parent / 'translations')
+#
+# BABEL_TRANSLATION_DIRECTORIES is a ';' separated list. For each locale, Flask-Babel loads one
+# catalog per directory and merges them in order (Domain.get_translations -> babel
+# Translations.merge -> dict.update), so a later directory overrides an earlier one *per message*
+# rather than replacing the catalog.
+#
+# That gives deployments an optional overlay layer: a catalog holding only the handful of msgids
+# whose wording needs to differ (branding, or wording that makes no sense for that deployment,
+# e.g. instructions to set an ENV var that the operator controls). Every other string in the same
+# language still comes from the base catalog, and a language with no overlay file is untouched.
+# Because the overlay keys off the upstream msgid, templates keep the upstream string verbatim and
+# carry no diff at all.
+#
+# Absent or empty overlay directory == no behaviour change.
+# See changedetectionio/translations_overlay/README.md
+_translation_directories = [str(Path(__file__).parent / 'translations')]
+_translation_overlay = os.getenv(
+ 'TRANSLATION_OVERLAY_DIR', str(Path(__file__).parent / 'translations_overlay')
+)
+if os.path.isdir(_translation_overlay):
+ _translation_directories.append(_translation_overlay)
+app.config['BABEL_TRANSLATION_DIRECTORIES'] = ';'.join(_translation_directories)
app.config['BABEL_DEFAULT_LOCALE'] = 'en_GB'
# Session configuration
@@ -128,17 +182,20 @@ app.config['BABEL_DEFAULT_LOCALE'] = 'en_GB'
# - Flask-Login's remember=True creates a separate authentication cookie
# - Setting PERMANENT_SESSION_LIFETIME controls how long the Flask session cookie lasts
from datetime import timedelta
+
app.config['PERMANENT_SESSION_LIFETIME'] = timedelta(days=3650) # ~10 years (effectively unlimited)
-#app.config["EXPLAIN_TEMPLATE_LOADING"] = True
+# app.config["EXPLAIN_TEMPLATE_LOADING"] = True
app.jinja_env.add_extension('jinja2.ext.loopcontrols')
+
# Configure Jinja2 to search for templates in plugin directories
def _configure_plugin_templates():
"""Configure Jinja2 loader to include plugin template directories."""
from jinja2 import ChoiceLoader, FileSystemLoader
+
from changedetectionio.pluggy_interface import get_plugin_template_paths
# Get plugin template paths
@@ -151,35 +208,65 @@ def _configure_plugin_templates():
loaders.append(FileSystemLoader(path))
app.jinja_loader = ChoiceLoader(loaders)
- logger.info(f"Configured Jinja2 to search {len(plugin_template_paths)} plugin template directories")
+ logger.info(
+ f"Configured Jinja2 to search {len(plugin_template_paths)} plugin template directories"
+ )
+
# Configure plugin templates (called after plugins are loaded)
_configure_plugin_templates()
csrf = CSRFProtect()
csrf.init_app(app)
-notification_debug_log=[]
-# Locale for correct presentation of prices etc
+notification_debug_log = []
+
+# Locale for correct presentation of prices etc.
+#
+# Deliberately NOT locale.LC_ALL - LC_COLLATE must stay in the "C" locale.
+#
+# elementpath implements the XPath string functions on top of locale.strxfrm:
+#
+# def contains(self, a, b): return self.strxfrm(b) in self.strxfrm(a)
+#
+# Under LC_COLLATE=C, strxfrm() is the identity function and that substring test means what it
+# says. Under any real locale it returns a binary collation key, and a substring of a collation
+# key is not the collation key of the substring - so contains(), starts-with(), ends-with() and
+# substring-before/after() silently return false for EVERY input. Every xPath filter using
+# contains() then matches nothing and the watch reports "no filters were found" on a page whose
+# HTML plainly contains the target (#4437).
+#
+# That stayed hidden until the image actually generated its locales: before then this call raised
+# locale.Error, we logged a warning and stayed in C. Once en_US.UTF-8 existed the call succeeded
+# and took LC_COLLATE with it. Setting the presentation categories individually keeps what this
+# block is for - 1234567 still renders as "1,234,567" - without touching collation.
+#
+# Per XPath 3.1 the default collation is codepoint and must not consult LC_COLLATE at all, so
+# this is arguably an elementpath bug; html_tools.xpath_filter() pins the collation explicitly as
+# well, so a filter is correct even if an operator sets LC_COLLATE themselves.
default_locale = locale.getdefaultlocale()
logger.info(f"System locale default is {default_locale}")
-try:
- locale.setlocale(locale.LC_ALL, default_locale)
-except locale.Error:
- logger.warning(f"Unable to set locale {default_locale}, locale is not installed maybe?")
+for _category in (locale.LC_CTYPE, locale.LC_NUMERIC, locale.LC_MONETARY, locale.LC_TIME):
+ try:
+ locale.setlocale(_category, default_locale)
+ except locale.Error:
+ logger.warning(f"Unable to set locale {default_locale} for category {_category}, "
+ f"locale is not installed maybe?")
watch_api = Api(app, decorators=[csrf.exempt])
+
def init_app_secret(datastore_path):
secret = ""
path = os.path.join(datastore_path, "secret.txt")
try:
- with open(path, "r", encoding='utf-8') as f:
+ with open(path, encoding='utf-8') as f:
secret = f.read()
except FileNotFoundError:
import secrets
+
with open(path, "w", encoding='utf-8') as f:
secret = secrets.token_hex(32)
f.write(secret)
@@ -192,9 +279,117 @@ def get_darkmode_state():
css_dark_mode = request.cookies.get('css_dark_mode', 'false')
return 'true' if css_dark_mode and strtobool(css_dark_mode) else 'false'
+
@app.template_global()
def get_css_version():
- return __version__
+ """Cache-busting token for static assets.
+
+ Changes on every upgrade (so browsers refetch CSS/JS) but is not the raw
+ version string - the raw version was leaking to anonymous visitors on the
+ login page via `?v=x.y.z`, which allows exposed instances to be fingerprinted
+ for known-vulnerable releases (#2190). Salted with the per-installation
+ app_guid so it can't be reversed to the version.
+ """
+ salt = datastore.data.get('app_guid', '') if datastore else ''
+ return hashlib.sha256(f"{salt}{__version__}".encode()).hexdigest()[:10]
+
+
+# Static groups that are plain files on disk under changedetectionio/static// - the
+# same bytes for every visitor, so they can be fingerprinted and cached hard. Deliberately
+# excludes the dynamic groups handled inside static_content() ('screenshot', 'favicon',
+# 'visual_selector_data', 'plugin'), which are per-watch and/or password protected.
+STATIC_CACHEABLE_GROUPS = frozenset(['favicons', 'images', 'js', 'styles'])
+
+# How long a fingerprint is trusted before the file is stat()ed again. The lookup sits in the
+# hot path (one watch-list render emits hundreds of asset url_for() calls) so it can't stat
+# per URL, but the URLs we hand out are served `immutable` - an edited .js/.css that never
+# re-fingerprinted would be pinned in the browser for a year. A few seconds of staleness is
+# the compromise: invisible in production (files only change on upgrade, and the container
+# restarts) and self-correcting while developing.
+STATIC_FINGERPRINT_TTL = 10.0
+_static_fingerprints = {}
+_static_fingerprints_expires = 0.0
+
+
+def get_static_fingerprint(group, filename):
+ """Short token identifying this exact revision of a static file, for `?v=` cache busting.
+
+ Built from the file's own mtime+size rather than get_css_version()'s app-version token:
+ a version-wide token silently serves a stale asset whenever content changes without a
+ release (local dev, a patched image, a rebuilt styles.css), which is not survivable once
+ the response says `immutable`. Returns '' when the file can't be stat()ed, so the request
+ stays unversioned (and revalidating) rather than being pinned under a made-up token.
+ """
+ global _static_fingerprints_expires
+
+ now = time.monotonic()
+ if now > _static_fingerprints_expires:
+ _static_fingerprints.clear()
+ _static_fingerprints_expires = now + STATIC_FINGERPRINT_TTL
+
+ key = (group, filename)
+ token = _static_fingerprints.get(key)
+ if token is None:
+ try:
+ st = os.stat(os.path.join(app.static_folder, group, filename))
+ token = f"{int(st.st_mtime)}-{st.st_size}"
+ except OSError:
+ token = ''
+ _static_fingerprints[key] = token
+
+ return token
+
+
+@app.url_defaults
+def _fingerprint_static_urls(endpoint, values):
+ """Pin every static asset URL to the revision of the file it resolves to.
+
+ Doing it here rather than in the templates means an asset can't be added without its
+ cache-buster - the `?v=` is what lets static_content() answer with a year-long
+ `immutable` instead of making the browser revalidate on every page load.
+ """
+ if endpoint != 'static_content' or 'v' in values:
+ return
+
+ if values.get('group') in STATIC_CACHEABLE_GROUPS:
+ token = get_static_fingerprint(values['group'], values.get('filename', ''))
+ if token:
+ values['v'] = token
+
+
+class PublicStaticAssetSessionInterface(SecureCookieSessionInterface):
+ """Keeps "Vary: Cookie" and the session cookie refresh off public static asset responses.
+
+ Flask tags any response whose session was touched with "Vary: Cookie", and flask_login's
+ auth check touches it on every single request. Since Flask 3.1.3 the request context sets
+ `session.accessed` itself, so a view or an after_request hook can't opt out - the header is
+ added in save_session(), which runs last. It has to go for the files marked by
+ static_content(): our session cookie is permanent and re-signed (fresh timestamp) on every
+ response, so the Cookie request header keeps changing, and a browser honouring
+ "Vary: Cookie" would then miss its cache on every asset of every page load - the immutable
+ caching would never be used at all. Nothing in those groups depends on the session.
+ """
+
+ def save_session(self, app, session, response):
+ public_asset = g.get('public_static_asset', False)
+
+ if public_asset and not session.modified:
+ # Same bytes for every visitor and nothing to persist: skip the cookie refresh
+ # and the Vary entirely.
+ return
+
+ super().save_session(app, session, response)
+
+ if public_asset and 'Set-Cookie' in response.headers:
+ # Shouldn't happen (these requests don't write to the session), but if something
+ # ever does, the response now carries one visitor's cookie - it must not be stored
+ # by a shared cache under the long-lived header static_content() just set.
+ response.cache_control.public = False
+ response.cache_control.private = True
+
+
+app.session_interface = PublicStaticAssetSessionInterface()
+
@app.template_global()
def browser_config_locked():
@@ -204,14 +399,6 @@ def browser_config_locked():
Loading and using browsers.json is unaffected."""
return bool(strtobool(os.getenv('LOCKED_BROWSER_CONFIG', 'False')))
-@app.template_global()
-def add_watch_ui_available():
- """True when the Add-Watch-with-a-browser flow is usable - i.e. at least one interactive
- browser (screenshots + visual selector) is available. Used to hide the sidebar link when the
- only fetcher is the plain HTTP client (which can't drive the live visual preview)."""
- from changedetectionio.model.browser_config import has_visual_browser
- return has_visual_browser(datastore)
-
@app.template_global('filtered_action_url')
def _filtered_action_url(endpoint, **overrides):
"""Build a URL to `endpoint` carrying the CURRENT watch-list filters (query args)
@@ -224,23 +411,42 @@ def _filtered_action_url(endpoint, **overrides):
args = {k: v for k, v in args.items() if v not in (None, '', 0, '0')}
return url_for(endpoint, **args)
+
@app.template_global('filter_url')
def _filter_url(**overrides):
"""Watch-list filter link (shorthand for filtered_action_url('watchlist.index'))."""
return _filtered_action_url('watchlist.index', **overrides)
+
@app.template_global()
def get_sidebar_mode_class():
- """Body class that drives the left-rail behaviour (see parts/_action_sidebar.scss).
+ """Body class(es) that drive the left-rail behaviour (see parts/_action_sidebar.scss).
- 'collapsed' -> slim icon rail that expands on hover/focus (actionsidebar-minimal)
- 'pinned' -> rail always expanded with labels visible (actionside-bar-on)
+ Only the modes offered by MENU_SIDEBAR_ACTIONMODES are honoured - anything else in the
+ datastore (a stale value from an older release, hand-edited JSON) falls back to
+ MENU_SIDEBAR_ACTIONMODES_DEFAULT rather than leaking through as a body class.
+
+ 'expandable' -> icon-only rail, rolls out over the content on hover/focus
+ 'pinned-expanded' -> rail always expanded, labels visible at rest
+ 'minimal' -> icon-only rail that never expands
"""
- mode = datastore.data['settings']['application'].get('ui', {}).get('sidebar_mode', 'collapsed')
- # Pinned mode is permanently expanded, so it carries 'action-side-bar-expanded'
- # from the start. In collapsed mode that class is toggled on hover/focus by
- # static/js/sidebar.js.
- return 'actionside-bar-on action-side-bar-expanded' if mode == 'pinned' else 'actionsidebar-minimal'
+
+ # 'actionsidebar-minimal' - collapsed icon rail (hover-to-expand lives in CSS + static/js/sidebar.js)
+ # 'actionsidebar-no-expand' - opts that rail out of hover-to-expand
+ # 'actionside-bar-on' - always-open rail
+ # 'actionsidebar-expanded'- expanded logo/stats block
+ body_classes = {
+ 'expandable': 'actionsidebar-minimal',
+ 'pinned-expanded': 'actionside-bar-on actionsidebar-expanded',
+ 'minimal': 'actionsidebar-minimal actionsidebar-no-expand',
+ }
+
+ mode = datastore.data['settings']['application'].get('ui', {}).get('sidebar_mode')
+ if mode not in {choice for choice, _label in MENU_SIDEBAR_ACTIONMODES} or mode not in body_classes:
+ mode = MENU_SIDEBAR_ACTIONMODES_DEFAULT
+
+ return body_classes[mode]
+
@app.template_global()
def get_blueprint_class():
@@ -250,6 +456,7 @@ def get_blueprint_class():
bp = request.blueprint or ''
return ('blueprint-' + bp.replace('.', '-')) if bp else ''
+
@app.template_global()
def get_socketio_path():
"""Generate the correct Socket.IO path prefix for the client"""
@@ -261,14 +468,18 @@ def get_socketio_path():
# Socket.IO will be available at {prefix}/socket.io/
return prefix
+
@app.template_global('is_safe_valid_url')
def _is_safe_valid_url(test_url):
from .validate_url import is_safe_valid_url
+
return is_safe_valid_url(test_url)
+
@app.template_global('get_html_head_extras')
def _get_html_head_extras():
from .pluggy_interface import collect_html_head_extras
+
return collect_html_head_extras()
@@ -279,43 +490,50 @@ def _jinja2_filter_format_number_locale(value: float) -> str:
formatted_value = locale.format_string("%.2f", value, grouping=True)
return formatted_value
+
@app.template_filter('format_int_locale')
def _jinja2_filter_format_int_locale(value) -> str:
"Locale-grouped integer, e.g. 1000 -> 1,000 (no decimals — for counts)"
return locale.format_string("%d", int(value), grouping=True)
+
@app.template_filter('regex_search')
def _jinja2_filter_regex_search(value, pattern):
import re
+
return re.search(pattern, str(value)) is not None
+
@app.template_global('is_checking_now')
def _watch_is_checking_now(watch_obj, format="%Y-%m-%d %H:%M:%S"):
return worker_pool.is_watch_running(watch_obj['uuid'])
+
@app.template_global('get_watch_queue_position')
def _get_watch_queue_position(watch_obj):
"""Get the position of a watch in the queue"""
uuid = watch_obj['uuid']
return update_q.get_uuid_position(uuid)
+
@app.template_global('get_current_worker_count')
def _get_current_worker_count():
"""Get the current number of operational workers"""
return worker_pool.get_worker_count()
+
@app.template_global('get_worker_status_info')
def _get_worker_status_info():
"""Get detailed worker status information for display"""
status = worker_pool.get_worker_status()
running_uuids = worker_pool.get_running_uuids()
-
+
return {
'count': status['worker_count'],
'type': status['worker_type'],
'active_workers': len(running_uuids),
'processing_watches': running_uuids,
- 'loop_running': status.get('async_loop_running', None)
+ 'loop_running': status.get('async_loop_running', None),
}
@@ -323,7 +541,6 @@ def _get_worker_status_info():
# running or something similar.
@app.template_filter('format_last_checked_time')
def _jinja2_filter_datetime(watch_obj, format="%Y-%m-%d %H:%M:%S"):
-
if watch_obj['last_checked'] == 0:
return gettext('Not yet')
@@ -333,7 +550,10 @@ def _jinja2_filter_datetime(watch_obj, format="%Y-%m-%d %H:%M:%S"):
return timeago.format(int(watch_obj['last_checked']), time.time(), locale)
except:
# Fallback to English if locale not supported by timeago
- return timeago.format(int(watch_obj['last_checked']), time.time(), 'en_short' if short else 'en')
+ return timeago.format(
+ int(watch_obj['last_checked']), time.time(), 'en_short' if short else 'en'
+ )
+
@app.template_filter('format_timestamp_timeago')
def _jinja2_filter_datetimestamp(timestamp, format="%Y-%m-%d %H:%M:%S"):
@@ -353,16 +573,18 @@ def _jinja2_filter_datetimestamp(timestamp, format="%Y-%m-%d %H:%M:%S"):
def _jinja2_filter_pagination_slice(arr, skip):
per_page = datastore.data['settings']['application'].get('pager_size', 50)
if per_page:
- return arr[skip:skip + per_page]
+ return arr[skip : skip + per_page]
return arr
+
@app.template_filter('format_seconds_ago')
def _jinja2_filter_seconds_precise(timestamp):
if timestamp == False:
return gettext('Not yet')
- return format(int(time.time()-timestamp), ',d')
+ return format(int(time.time() - timestamp), ',d')
+
@app.template_filter('format_duration')
def _jinja2_filter_format_duration(seconds):
@@ -461,10 +683,11 @@ def _jinja2_filter_fetcher_status_icons(fetcher_name):
Returns:
str: HTML string containing status icon elements
"""
+ from flask import url_for
+ from markupsafe import Markup
+
from changedetectionio import content_fetchers
from changedetectionio.pluggy_interface import collect_fetcher_status_icons
- from markupsafe import Markup
- from flask import url_for
icon_data = None
@@ -489,6 +712,7 @@ def _jinja2_filter_fetcher_status_icons(fetcher_name):
except:
# Fallback: build URL manually respecting APPLICATION_ROOT
from flask import request
+
app_root = request.script_root if hasattr(request, 'script_root') else ''
icon_url = f"{app_root}/static/{group}/{icon_data['filename']}"
@@ -498,8 +722,10 @@ def _jinja2_filter_fetcher_status_icons(fetcher_name):
return ''
+
_RE_SANITIZE_TAG = re.compile(r'[^a-zA-Z0-9]')
+
@app.template_filter('sanitize_tag_class')
def _jinja2_filter_sanitize_tag_class(tag_title):
"""Sanitize a tag title to create a valid CSS class name.
@@ -518,23 +744,33 @@ def _jinja2_filter_sanitize_tag_class(tag_title):
sanitized = 'tag' + sanitized
return sanitized if sanitized else 'tag'
+
# Import login_optionally_required from auth_decorator
-from changedetectionio.auth_decorator import SHARED_DIFF_READ_ONLY_ENDPOINTS, login_optionally_required
+from changedetectionio.auth_decorator import (
+ SHARED_DIFF_READ_ONLY_ENDPOINTS,
+ login_optionally_required,
+)
+
# When nobody is logged in Flask-Login's current_user is set to an AnonymousUser object.
class User(flask_login.UserMixin):
- id=None
+ id = None
def set_password(self, password):
return True
+
def get_user(self, email="defaultuser@changedetection.io"):
return self
+
def is_authenticated(self):
return True
+
def is_active(self):
return True
+
def is_anonymous(self):
return False
+
def get_id(self):
return str(self.id)
@@ -557,7 +793,7 @@ class User(flask_login.UserMixin):
'sha256',
password.encode('utf-8'), # Convert the password to bytes
salt_from_storage,
- 100000
+ 100000,
)
new_key = salt_from_storage + new_key
@@ -580,6 +816,7 @@ def clean_startup_state(datastore):
# (and break for browser-steps watches).
try:
from changedetectionio import content_fetchers
+
valid_fetchers = {name for name, _desc in content_fetchers.available_fetchers()}
# A user browser-config id is also a valid default (it maps to an engine).
valid_fetchers |= set(datastore.browser_config_store.all().keys())
@@ -656,18 +893,30 @@ def changedetection_app(config=None, datastore_o=None):
_=gettext,
get_locale=get_locale,
get_flag_for_locale=get_flag_for_locale,
- available_languages=available_languages
+ available_languages=available_languages,
)
@app.context_processor
def inject_llm_features_disabled():
from changedetectionio.llm.evaluator import is_llm_features_disabled
+
return dict(llm_features_disabled=is_llm_features_disabled())
+ @app.context_processor
+ def inject_has_visual_browser():
+ # Whether any installed content fetcher can render the Add-Watch live preview -
+ # sidebar-nav.html hides the Add-Watch link without one. Same capability lookup the
+ # page's browser picker and /snapshot use, so they can't disagree.
+ from changedetectionio.blueprint.add_watch_ui import browser_config
+
+ return dict(has_visual_browser=browser_config.has_visual_browser(datastore))
+
# Set up a request hook to check authentication for all routes
@app.before_request
def check_authentication():
- has_password_enabled = datastore.data['settings']['application'].get('password') or os.getenv("SALTED_PASS", False)
+ has_password_enabled = datastore.data['settings']['application'].get(
+ 'password'
+ ) or os.getenv("SALTED_PASS", False)
if has_password_enabled and not flask_login.current_user.is_authenticated:
# Permitted
@@ -677,13 +926,22 @@ def changedetection_app(config=None, datastore_o=None):
# Permitted - static flag icons need to load on login page
elif request.endpoint and request.endpoint == 'static_flags':
return None
- # Permitted - language selection should work on login page
- elif request.endpoint and request.endpoint == 'set_language':
+ # Permitted - language selection should work on login page.
+ # Both halves of the language modal must be exempt: it renders for anonymous
+ # users (base.html deliberately leaves it outside the is_authenticated guard),
+ # so exempting only set_language let you pick a language but bounced
+ # "Auto-detect from browser" to /login without clearing the session locale.
+ elif request.endpoint and request.endpoint in (
+ 'set_language',
+ 'ui.delete_locale_language_session_var_if_it_exists',
+ ):
return None
# Permitted
elif request.endpoint and 'login' in request.endpoint:
return None
- elif request.endpoint in SHARED_DIFF_READ_ONLY_ENDPOINTS and datastore.data['settings']['application'].get('shared_diff_access'):
+ elif request.endpoint in SHARED_DIFF_READ_ONLY_ENDPOINTS and datastore.data['settings'][
+ 'application'
+ ].get('shared_diff_access'):
return None
elif request.method in flask_login.config.EXEMPT_METHODS:
return None
@@ -701,44 +959,87 @@ def changedetection_app(config=None, datastore_o=None):
else:
return login_manager.unauthorized()
+ # #4299: werkzeug's send_file() (via make_conditional) injects a Date
+ # header into the WSGI response for conditional/static responses, and the
+ # Werkzeug built-in server (allow_unsafe_werkzeug=True) then writes its own
+ # Date via BaseHTTPRequestHandler.send_response() — emitting the Date
+ # header line twice, which RFC 9110 forbids and nginx rejects ("upstream
+ # sent duplicate header line"). Strip the application-side copy so the
+ # server's single header is what reaches the wire.
+ @app.after_request
+ def strip_duplicate_date_header(response):
+ if request.environ.get('SERVER_SOFTWARE', '').startswith('Werkzeug'):
+ response.headers.pop("Date", None)
+ return response
- watch_api.add_resource(WatchHistoryDiff,
- '/api/v1/watch//difference//',
- resource_class_kwargs={'datastore': datastore})
- watch_api.add_resource(WatchSingleHistory,
- '/api/v1/watch//history/',
- resource_class_kwargs={'datastore': datastore, 'update_q': update_q})
- watch_api.add_resource(WatchFavicon,
- '/api/v1/watch//favicon',
- resource_class_kwargs={'datastore': datastore})
- watch_api.add_resource(WatchHistory,
- '/api/v1/watch//history',
- resource_class_kwargs={'datastore': datastore})
+ # Dynamic/authenticated pages (forms carrying a CSRF token, watch data, settings) must not
+ # be stored by an intermediate CDN or reverse proxy. Flask already sends "Vary: Cookie" on
+ # these, but an edge cache configured to key purely on URL will ignore it and can serve a
+ # stale CSRF token (breaking form submits) or one session's page to another visitor.
+ # Only fills in the header when the route didn't set one, so the explicit Cache-Control on
+ # static assets, screenshots, favicons and plugin files is left untouched. Note that
+ # werkzeug's send_file() always sets Cache-Control, so file responses never reach here.
+ @app.after_request
+ def add_no_cache_headers(response):
+ if 'Cache-Control' not in response.headers:
+ response.headers['Cache-Control'] = 'no-cache, no-store, must-revalidate'
+ return response
- watch_api.add_resource(CreateWatch, '/api/v1/watch',
- resource_class_kwargs={'datastore': datastore, 'update_q': update_q})
+ watch_api.add_resource(
+ WatchHistoryDiff,
+ '/api/v1/watch//difference//',
+ resource_class_kwargs={'datastore': datastore},
+ )
+ watch_api.add_resource(
+ WatchSingleHistory,
+ '/api/v1/watch//history/',
+ resource_class_kwargs={'datastore': datastore, 'update_q': update_q},
+ )
+ watch_api.add_resource(
+ WatchFavicon,
+ '/api/v1/watch//favicon',
+ resource_class_kwargs={'datastore': datastore},
+ )
+ watch_api.add_resource(
+ WatchHistory,
+ '/api/v1/watch//history',
+ resource_class_kwargs={'datastore': datastore},
+ )
- watch_api.add_resource(Watch, '/api/v1/watch/',
- resource_class_kwargs={'datastore': datastore, 'update_q': update_q})
+ watch_api.add_resource(
+ CreateWatch,
+ '/api/v1/watch',
+ resource_class_kwargs={'datastore': datastore, 'update_q': update_q},
+ )
- watch_api.add_resource(SystemInfo, '/api/v1/systeminfo',
- resource_class_kwargs={'datastore': datastore, 'update_q': update_q})
+ watch_api.add_resource(
+ Watch,
+ '/api/v1/watch/',
+ resource_class_kwargs={'datastore': datastore, 'update_q': update_q},
+ )
- watch_api.add_resource(Import,
- '/api/v1/import',
- resource_class_kwargs={'datastore': datastore})
+ watch_api.add_resource(
+ SystemInfo,
+ '/api/v1/systeminfo',
+ resource_class_kwargs={'datastore': datastore, 'update_q': update_q},
+ )
- watch_api.add_resource(Tags, '/api/v1/tags',
- resource_class_kwargs={'datastore': datastore})
+ watch_api.add_resource(Import, '/api/v1/import', resource_class_kwargs={'datastore': datastore})
- watch_api.add_resource(Tag, '/api/v1/tag', '/api/v1/tag/',
- resource_class_kwargs={'datastore': datastore, 'update_q': update_q})
-
- watch_api.add_resource(Search, '/api/v1/search',
- resource_class_kwargs={'datastore': datastore})
+ watch_api.add_resource(Tags, '/api/v1/tags', resource_class_kwargs={'datastore': datastore})
- watch_api.add_resource(Notifications, '/api/v1/notifications',
- resource_class_kwargs={'datastore': datastore})
+ watch_api.add_resource(
+ Tag,
+ '/api/v1/tag',
+ '/api/v1/tag/',
+ resource_class_kwargs={'datastore': datastore, 'update_q': update_q},
+ )
+
+ watch_api.add_resource(Search, '/api/v1/search', resource_class_kwargs={'datastore': datastore})
+
+ watch_api.add_resource(
+ Notifications, '/api/v1/notifications', resource_class_kwargs={'datastore': datastore}
+ )
watch_api.add_resource(Spec, '/api/v1/full-spec')
@@ -753,7 +1054,7 @@ def changedetection_app(config=None, datastore_o=None):
# Pass the current request path so users are redirected back after login
return redirect(url_for('login', redirect=request.path))
- @app.route('/logout')
+ @app.route('/logout', methods=['POST'])
def logout():
flask_login.logout_user()
@@ -767,7 +1068,7 @@ def changedetection_app(config=None, datastore_o=None):
# Otherwise just go to watchlist
return redirect(url_for('watchlist.index'))
- @app.route('/set-language/')
+ @app.route('/set-language/', methods=['POST'])
def set_language(locale):
"""Set the user's preferred language in the session"""
if not request.cookies:
@@ -786,6 +1087,7 @@ def changedetection_app(config=None, datastore_o=None):
# We must refresh to clear this cache so the new locale takes effect immediately
# This is especially important for tests where multiple requests happen rapidly
from flask_babel import refresh
+
refresh()
else:
logger.error(f"Invalid locale {locale}, available: {language_codes}")
@@ -827,7 +1129,7 @@ def changedetection_app(config=None, datastore_o=None):
password = request.form.get('password')
- if (user.check_password(password)):
+ if user.check_password(password):
flask_login.login_user(user, remember=True)
# Redirect to the validated URL after successful login
return redirect(validated_redirect)
@@ -848,9 +1150,10 @@ def changedetection_app(config=None, datastore_o=None):
@app.route("/static/flags/", methods=['GET'])
def static_flags(flag_path):
"""Handle flag icon files with subdirectories"""
- from flask import make_response
import re
+ from flask import make_response
+
# flag_path comes in as "1x1/de.svg" or "4x3/de.svg"
if re.match(r'^(1x1|4x3)/[a-z0-9-]+\.svg$', flag_path.lower()):
# Reconstruct the path safely with additional validation
@@ -873,6 +1176,9 @@ def changedetection_app(config=None, datastore_o=None):
response = make_response(send_from_directory(f"static/flags/{subdir}", svg_file))
response.headers['Content-type'] = 'image/svg+xml'
response.headers['Cache-Control'] = 'max-age=86400, public' # Cache for 24 hours
+ # Same for everyone, and the language modal pulls a few hundred of them - see
+ # PublicStaticAssetSessionInterface for why the "Vary: Cookie" has to go.
+ g.public_static_asset = True
return response
except FileNotFoundError:
abort(404)
@@ -881,9 +1187,10 @@ def changedetection_app(config=None, datastore_o=None):
@app.route("/static//", methods=['GET'])
def static_content(group, filename):
- from flask import make_response
import re
+ from flask import make_response
+
# Strict sanitization: only allow a-z, 0-9, and underscore (blocks .. and other traversal)
group = re.sub(r'[^a-z0-9_-]+', '', group.lower())
filename = filename
@@ -894,16 +1201,27 @@ def changedetection_app(config=None, datastore_o=None):
if group == 'screenshot':
# Could be sensitive, follow password requirements
- if datastore.data['settings']['application']['password'] and not flask_login.current_user.is_authenticated:
+ if (
+ datastore.data['settings']['application']['password']
+ and not flask_login.current_user.is_authenticated
+ ):
if not datastore.data['settings']['application'].get('shared_diff_access'):
abort(403)
- screenshot_filename = "last-screenshot.png" if not request.args.get('error_screenshot') else "last-error-screenshot.png"
+ screenshot_filename = (
+ "last-screenshot.png"
+ if not request.args.get('error_screenshot')
+ else "last-error-screenshot.png"
+ )
# These files should be in our subdirectory
try:
# set nocache, set content-type
- response = make_response(send_from_directory(os.path.join(datastore_o.datastore_path, filename), screenshot_filename))
+ response = make_response(
+ send_from_directory(
+ os.path.join(datastore_o.datastore_path, filename), screenshot_filename
+ )
+ )
response.headers['Content-type'] = 'image/png'
response.headers['Cache-Control'] = 'no-cache, no-store, must-revalidate'
response.headers['Pragma'] = 'no-cache'
@@ -915,7 +1233,10 @@ def changedetection_app(config=None, datastore_o=None):
if group == 'favicon':
# Could be sensitive, follow password requirements
- if datastore.data['settings']['application']['password'] and not flask_login.current_user.is_authenticated:
+ if (
+ datastore.data['settings']['application']['password']
+ and not flask_login.current_user.is_authenticated
+ ):
abort(403)
# Get the watch object
watch = datastore.data['watching'].get(filename)
@@ -929,17 +1250,24 @@ def changedetection_app(config=None, datastore_o=None):
mime = get_favicon_mime_type(filepath)
if 'text' in mime:
- logger.debug(f"Aborting favicon request for {filename} because mimetype might be text (bad mimetype) '{mime}'")
+ logger.debug(
+ f"Aborting favicon request for {filename} because mimetype might be text (bad mimetype) '{mime}'"
+ )
abort(404)
response = make_response(send_from_directory(watch.data_dir, favicon_filename))
response.headers['Content-type'] = mime
- response.headers['Cache-Control'] = 'max-age=300, must-revalidate' # Cache for 5 minutes, then revalidate
+ response.headers['Cache-Control'] = (
+ 'max-age=300, must-revalidate' # Cache for 5 minutes, then revalidate
+ )
return response
if group == 'visual_selector_data':
# Could be sensitive, follow password requirements
- if datastore.data['settings']['application']['password'] and not flask_login.current_user.is_authenticated:
+ if (
+ datastore.data['settings']['application']['password']
+ and not flask_login.current_user.is_authenticated
+ ):
abort(403)
# These files should be in our subdirectory
@@ -949,11 +1277,15 @@ def changedetection_app(config=None, datastore_o=None):
watch_directory = str(os.path.join(datastore_o.datastore_path, filename))
response = None
if os.path.isfile(os.path.join(watch_directory, "elements.deflate")):
- response = make_response(send_from_directory(watch_directory, "elements.deflate"))
+ response = make_response(
+ send_from_directory(watch_directory, "elements.deflate")
+ )
response.headers['Content-Type'] = 'application/json'
response.headers['Content-Encoding'] = 'deflate'
else:
- logger.error(f'Request elements.deflate at "{watch_directory}" but was not found.')
+ logger.error(
+ f'Request elements.deflate at "{watch_directory}" but was not found.'
+ )
abort(404)
if response:
@@ -969,9 +1301,10 @@ def changedetection_app(config=None, datastore_o=None):
# Handle plugin group specially
if group == 'plugin':
# Serve files from plugin static directories
- from changedetectionio.pluggy_interface import plugin_manager
import os as os_check
+ from changedetectionio.pluggy_interface import plugin_manager
+
for plugin_name, plugin_obj in plugin_manager.list_name_plugin():
if hasattr(plugin_obj, 'plugin_static_path'):
try:
@@ -982,7 +1315,9 @@ def changedetection_app(config=None, datastore_o=None):
if os_check.path.isfile(plugin_file_path):
# Found the file in a plugin
response = make_response(send_from_directory(static_path, filename))
- response.headers['Cache-Control'] = 'max-age=3600, public' # Cache for 1 hour
+ response.headers['Cache-Control'] = (
+ 'max-age=3600, public' # Cache for 1 hour
+ )
return response
except Exception as e:
logger.debug(f"Error checking plugin {plugin_name} for static file: {e}")
@@ -993,54 +1328,115 @@ def changedetection_app(config=None, datastore_o=None):
# These files should be in our subdirectory
try:
- return send_from_directory(f"static/{group}", path=filename)
+ response = make_response(send_from_directory(f"static/{group}", path=filename))
except FileNotFoundError:
abort(404)
+ # SEND_FILE_MAX_AGE_DEFAULT=0 means werkzeug hands these out as "no-cache, max-age=0",
+ # so every asset on every page load costs a request - a 304, but still a round trip.
+ # A `?v=` matching the file's current fingerprint (added by _fingerprint_static_urls)
+ # means the caller asked for this exact revision and can keep it for good; the next
+ # upgrade changes the URL, not the cache entry. Anything unversioned - older cached
+ # HTML, a hand-typed or third-party URL - keeps revalidating, where the ETag werkzeug
+ # already set turns the round trip into a 304 rather than a re-download.
+ if group in STATIC_CACHEABLE_GROUPS and request.args.get('v') == get_static_fingerprint(
+ group, filename
+ ):
+ response.headers['Cache-Control'] = 'public, max-age=31536000, immutable'
+ # werkzeug derived an "Expires: " from SEND_FILE_MAX_AGE_DEFAULT=0. Cache-Control
+ # wins over it for anything HTTP/1.1, but leaving the two contradicting each other
+ # means an HTTP/1.0-era cache treats the file as already stale.
+ response.expires = int(time.time()) + 31536000
+ else:
+ response.headers['Cache-Control'] = 'public, max-age=0, must-revalidate'
+
+ # Identical for every visitor (the password-protected groups returned further up), so
+ # let PublicStaticAssetSessionInterface strip the "Vary: Cookie" that would otherwise
+ # key each of these on the caller's cookies and defeat the caching above.
+ g.public_static_asset = True
+
+ return response
import changedetectionio.blueprint.browser_steps as browser_steps
- app.register_blueprint(browser_steps.construct_blueprint(datastore), url_prefix='/browser-steps')
- from changedetectionio.blueprint.imports import construct_blueprint as construct_import_blueprint
- app.register_blueprint(construct_import_blueprint(datastore, update_q, queuedWatchMetaData), url_prefix='/imports')
+ app.register_blueprint(
+ browser_steps.construct_blueprint(datastore), url_prefix='/browser-steps'
+ )
+
+ from changedetectionio.blueprint.imports import (
+ construct_blueprint as construct_import_blueprint,
+ )
+
+ app.register_blueprint(
+ construct_import_blueprint(datastore, update_q, queuedWatchMetaData), url_prefix='/imports'
+ )
+
+ from changedetectionio.blueprint.add_watch_ui import (
+ construct_blueprint as construct_add_watch_ui_blueprint,
+ )
- from changedetectionio.blueprint.add_watch_ui import construct_blueprint as construct_add_watch_ui_blueprint
app.register_blueprint(construct_add_watch_ui_blueprint(datastore), url_prefix='/add-watch-ui')
import changedetectionio.blueprint.price_data_follower as price_data_follower
- app.register_blueprint(price_data_follower.construct_blueprint(datastore, update_q), url_prefix='/price_data_follower')
+
+ app.register_blueprint(
+ price_data_follower.construct_blueprint(datastore, update_q),
+ url_prefix='/price_data_follower',
+ )
import changedetectionio.blueprint.tags as tags
+
app.register_blueprint(tags.construct_blueprint(datastore), url_prefix='/tags')
import changedetectionio.blueprint.check_proxies as check_proxies
- app.register_blueprint(check_proxies.construct_blueprint(datastore=datastore), url_prefix='/check_proxy')
+
+ app.register_blueprint(
+ check_proxies.construct_blueprint(datastore=datastore), url_prefix='/check_proxy'
+ )
import changedetectionio.blueprint.backups as backups
+
app.register_blueprint(backups.construct_blueprint(datastore), url_prefix='/backups')
import changedetectionio.blueprint.settings as settings
+
app.register_blueprint(settings.construct_blueprint(datastore), url_prefix='/settings')
import changedetectionio.conditions.blueprint as conditions
+
app.register_blueprint(conditions.construct_blueprint(datastore), url_prefix='/conditions')
import changedetectionio.blueprint.rss.blueprint as rss
+
app.register_blueprint(rss.construct_blueprint(datastore), url_prefix='/rss')
# watchlist UI buttons etc
import changedetectionio.blueprint.ui as ui
- app.register_blueprint(ui.construct_blueprint(datastore, update_q, worker_pool, queuedWatchMetaData, watch_check_update))
+
+ app.register_blueprint(
+ ui.construct_blueprint(
+ datastore, update_q, worker_pool, queuedWatchMetaData, watch_check_update
+ )
+ )
import changedetectionio.blueprint.watchlist as watchlist
- app.register_blueprint(watchlist.construct_blueprint(datastore=datastore, update_q=update_q, queuedWatchMetaData=queuedWatchMetaData), url_prefix='')
+
+ app.register_blueprint(
+ watchlist.construct_blueprint(
+ datastore=datastore, update_q=update_q, queuedWatchMetaData=queuedWatchMetaData
+ ),
+ url_prefix='',
+ )
# Initialize Socket.IO server conditionally based on settings
- socket_io_enabled = datastore.data['settings']['application'].get('ui', {}).get('socket_io_enabled', True)
+ socket_io_enabled = (
+ datastore.data['settings']['application'].get('ui', {}).get('socket_io_enabled', True)
+ )
if socket_io_enabled and app.config.get('batch_mode'):
socket_io_enabled = False
if socket_io_enabled:
from changedetectionio.realtime.socket_server import init_socketio
+
global socketio_server
socketio_server = init_socketio(app, datastore)
logger.info("Socket.IO server initialized")
@@ -1052,68 +1448,70 @@ def changedetection_app(config=None, datastore_o=None):
@app.route('/gc-cleanup', methods=['GET'])
@login_optionally_required
def gc_cleanup():
- from changedetectionio.gc_cleanup import memory_cleanup
from flask import jsonify
+ from changedetectionio.gc_cleanup import memory_cleanup
+
result = memory_cleanup(app)
- return jsonify({"status": "success", "message": "Memory cleanup completed", "result": result})
+ return jsonify(
+ {"status": "success", "message": "Memory cleanup completed", "result": result}
+ )
# Worker health check endpoint
@app.route('/worker-health', methods=['GET'])
@login_optionally_required
def worker_health():
from flask import jsonify
-
- expected_workers = int(os.getenv("FETCH_WORKERS", datastore.data['settings']['requests']['workers']))
-
+
+ expected_workers = int(
+ os.getenv("FETCH_WORKERS", datastore.data['settings']['requests']['workers'])
+ )
+
# Get basic status
status = worker_pool.get_worker_status()
-
+
# Perform health check
health_result = worker_pool.check_worker_health(
expected_count=expected_workers,
update_q=update_q,
notification_q=notification_q,
app=app,
- datastore=datastore
+ datastore=datastore,
+ )
+
+ return jsonify(
+ {
+ "status": "success",
+ "worker_status": status,
+ "health_check": health_result,
+ "expected_workers": expected_workers,
+ }
)
-
- return jsonify({
- "status": "success",
- "worker_status": status,
- "health_check": health_result,
- "expected_workers": expected_workers
- })
# Queue status endpoint
@app.route('/queue-status', methods=['GET'])
@login_optionally_required
def queue_status():
from flask import jsonify, request
-
+
# Get specific UUID position if requested
target_uuid = request.args.get('uuid')
-
+
if target_uuid:
position_info = update_q.get_uuid_position(target_uuid)
- return jsonify({
- "status": "success",
- "uuid": target_uuid,
- "queue_position": position_info
- })
+ return jsonify(
+ {"status": "success", "uuid": target_uuid, "queue_position": position_info}
+ )
else:
# Get pagination parameters
limit = request.args.get('limit', type=int)
offset = request.args.get('offset', type=int, default=0)
summary_only = request.args.get('summary', type=bool, default=False)
-
+
if summary_only:
# Fast summary for large queues
summary = update_q.get_queue_summary()
- return jsonify({
- "status": "success",
- "queue_summary": summary
- })
+ return jsonify({"status": "success", "queue_summary": summary})
else:
# Get queued items with pagination support
if limit is None:
@@ -1121,19 +1519,35 @@ def changedetection_app(config=None, datastore_o=None):
queue_size = update_q.qsize()
if queue_size > 100:
limit = 50
- logger.warning(f"Large queue ({queue_size} items) detected, limiting to {limit} items. Use ?limit=N for more.")
-
+ logger.warning(
+ f"Large queue ({queue_size} items) detected, limiting to {limit} items. Use ?limit=N for more."
+ )
+
all_queued = update_q.get_all_queued_uuids(limit=limit, offset=offset)
- return jsonify({
- "status": "success",
- "queue_size": update_q.qsize(),
- "queued_data": all_queued
- })
+ return jsonify(
+ {"status": "success", "queue_size": update_q.qsize(), "queued_data": all_queued}
+ )
if strtobool(os.getenv('HISTORY_SNAPSHOT_FILE_ALLOW_OUTSIDE_WATCH_DATADIR', 'False')):
- logger.warning("SECURITY WARNING: HISTORY_SNAPSHOT_FILE_ALLOW_OUTSIDE_WATCH_DATADIR is enabled — "
- "snapshot reads are NOT confined to the watch data directory. "
- "This disables protection against path traversal via restored backups (GHSA-8757-69j2-hx56).")
+ logger.warning(
+ "SECURITY WARNING: HISTORY_SNAPSHOT_FILE_ALLOW_OUTSIDE_WATCH_DATADIR is enabled — "
+ "snapshot reads are NOT confined to the watch data directory. "
+ "This disables protection against path traversal via restored backups (GHSA-8757-69j2-hx56)."
+ )
+
+ # Memory/CPU management -
+ # Freeze the startup object graph into the "permanent generation" so that the cyclic
+ # garbage collector never traverses it again, this allows more of the app to swap into 'cold' RAM
+ if (
+ 'pytest' not in sys.modules
+ and 'PYTEST_CURRENT_TEST' not in os.environ
+ and not strtobool(os.getenv('DISABLE_GC_FREEZE', 'no'))
+ ):
+ gc.collect()
+ gc.freeze()
+ logger.debug(
+ f"GC: froze {gc.get_freeze_count()} startup objects into the permanent generation"
+ )
# Start the async workers during app initialization
# Can be overridden by ENV or use the default settings
@@ -1145,23 +1559,38 @@ def changedetection_app(config=None, datastore_o=None):
batch_mode = app.config.get('batch_mode', False)
if not batch_mode:
# @todo handle ctrl break
- ticker_thread = threading.Thread(target=ticker_thread_check_time_launch_checks, daemon=True, name="TickerThread-ScheduleChecker").start()
+ # Supervised: if the ticker ever returns or raises it is logged CRITICAL and
+ # restarted. A bare Thread cannot be restarted once its target returns, and a
+ # dead ticker means no watch is ever checked again while the process keeps
+ # looking healthy. Note this keeps a real Thread handle - Thread(...).start()
+ # returns None, so the old assignment left `ticker_thread` permanently None.
+ ticker_thread = start_supervised_thread(
+ target=ticker_thread_check_time_launch_checks,
+ name="TickerThread-ScheduleChecker",
+ exit_event=app.config.exit,
+ # sigshutdown_handler() sets both of these; check both so a restart
+ # can never race an in-progress shutdown.
+ is_shutting_down=lambda: bool(getattr(datastore, 'stop_thread', False)),
+ )
# Start configurable number of notification workers (default 1)
notification_workers = int(os.getenv("NOTIFICATION_WORKERS", "1"))
for i in range(notification_workers):
threading.Thread(
- target=notification_runner,
- args=(i,),
- daemon=True,
- name=f"NotificationRunner-{i}"
+ target=notification_runner, args=(i,), daemon=True, name=f"NotificationRunner-{i}"
).start()
logger.info(f"Started {notification_workers} notification worker(s)")
in_pytest = "pytest" in sys.modules or "PYTEST_CURRENT_TEST" in os.environ
# Check for new release version, but not when running in test/build or pytest
- if not os.getenv("GITHUB_REF", False) and not strtobool(os.getenv('DISABLE_VERSION_CHECK', 'no')) and not in_pytest:
- threading.Thread(target=check_for_new_version, daemon=True, name="VersionChecker").start()
+ if (
+ not os.getenv("GITHUB_REF", False)
+ and not strtobool(os.getenv('DISABLE_VERSION_CHECK', 'no'))
+ and not in_pytest
+ ):
+ threading.Thread(
+ target=check_for_new_version, daemon=True, name="VersionChecker"
+ ).start()
else:
logger.info("Batch mode: Skipping ticker thread, notification runner, and version checker")
@@ -1174,6 +1603,7 @@ def changedetection_app(config=None, datastore_o=None):
def check_for_new_version():
import requests
import urllib3
+
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
session = requests.Session()
@@ -1181,11 +1611,14 @@ def check_for_new_version():
while not app.config.exit.is_set():
try:
- r = session.post("https://changedetection.io/check-ver.php",
- data={'version': __version__,
- 'app_guid': datastore.data['app_guid'],
- 'watch_count': len(datastore.data['watching'])
- })
+ r = session.post(
+ "https://changedetection.io/check-ver.php",
+ data={
+ 'version': __version__,
+ 'app_guid': datastore.data['app_guid'],
+ 'watch_count': len(datastore.data['watching']),
+ },
+ )
except:
pass
@@ -1201,8 +1634,9 @@ def check_for_new_version():
def notification_runner(worker_id=0):
global notification_debug_log
- from datetime import datetime
import json
+ from datetime import datetime
+
with app.app_context():
while not app.config.exit.is_set():
try:
@@ -1212,7 +1646,6 @@ def notification_runner(worker_id=0):
app.config.exit.wait(1)
else:
-
now = datetime.now()
sent_obj = None
@@ -1220,41 +1653,63 @@ def notification_runner(worker_id=0):
from changedetectionio.notification.handler import process_notification
# Fallback to system config if not set
- if not n_object.get('notification_body') and datastore.data['settings']['application'].get('notification_body'):
- n_object['notification_body'] = datastore.data['settings']['application'].get('notification_body')
+ if not n_object.get('notification_body') and datastore.data['settings'][
+ 'application'
+ ].get('notification_body'):
+ n_object['notification_body'] = datastore.data['settings'][
+ 'application'
+ ].get('notification_body')
- if not n_object.get('notification_title') and datastore.data['settings']['application'].get('notification_title'):
- n_object['notification_title'] = datastore.data['settings']['application'].get('notification_title')
+ if not n_object.get('notification_title') and datastore.data['settings'][
+ 'application'
+ ].get('notification_title'):
+ n_object['notification_title'] = datastore.data['settings'][
+ 'application'
+ ].get('notification_title')
- if not n_object.get('notification_format') and datastore.data['settings']['application'].get('notification_format'):
- n_object['notification_format'] = datastore.data['settings']['application'].get('notification_format')
+ if not n_object.get('notification_format') and datastore.data['settings'][
+ 'application'
+ ].get('notification_format'):
+ n_object['notification_format'] = datastore.data['settings'][
+ 'application'
+ ].get('notification_format')
if n_object.get('notification_urls', {}):
sent_obj = process_notification(n_object, datastore)
except Exception as e:
- logger.error(f"Notification worker {worker_id} - Watch URL: {n_object['watch_url']} Error {str(e)}")
+ logger.error(
+ f"Notification worker {worker_id} - Watch URL: {n_object['watch_url']} Error {str(e)}"
+ )
# UUID wont be present when we submit a 'test' from the global settings
if 'uuid' in n_object:
- datastore.update_watch(uuid=n_object['uuid'],
- update_obj={'last_notification_error': "Notification error detected, goto notification log."})
+ datastore.update_watch(
+ uuid=n_object['uuid'],
+ update_obj={
+ 'last_notification_error': "Notification error detected, goto notification log."
+ },
+ )
log_lines = str(e).splitlines()
notification_debug_log += log_lines
with app.app_context():
- app.config['watch_check_update_SIGNAL'].send(app_context=app, watch_uuid=n_object.get('uuid'))
+ app.config['watch_check_update_SIGNAL'].send(
+ app_context=app, watch_uuid=n_object.get('uuid')
+ )
# Process notifications
- notification_debug_log+= ["{} - SENDING - {}".format(now.strftime("%c"), json.dumps(sent_obj))]
+ notification_debug_log += [
+ "{} - SENDING - {}".format(now.strftime("%c"), json.dumps(sent_obj))
+ ]
# Trim the log length
notification_debug_log = notification_debug_log[-100:]
-
# Threaded runner, look for new watches to feed into the Queue.
def ticker_thread_check_time_launch_checks():
import random
+
proxy_last_called_time = {}
last_health_check = 0
@@ -1265,32 +1720,30 @@ def ticker_thread_check_time_launch_checks():
WAIT_TIME_BETWEEN_LOOP = 1.0 if not IN_PYTEST else 0.01
if IN_PYTEST:
# The time between loops should be less than the first .sleep/wait in def wait_for_all_checks() of tests/util.py
- logger.warning(f"Looks like we're in PYTEST! Setting time between searching for items to add to the queue to {WAIT_TIME_BETWEEN_LOOP}s")
+ logger.warning(
+ f"Looks like we're in PYTEST! Setting time between searching for items to add to the queue to {WAIT_TIME_BETWEEN_LOOP}s"
+ )
while not app.config.exit.is_set():
-
# Periodic worker health check (every 60 seconds)
now = time.time()
if now - last_health_check > 60:
- expected_workers = int(os.getenv("FETCH_WORKERS", datastore.data['settings']['requests']['workers']))
+ expected_workers = int(
+ os.getenv("FETCH_WORKERS", datastore.data['settings']['requests']['workers'])
+ )
health_result = worker_pool.check_worker_health(
expected_count=expected_workers,
update_q=update_q,
notification_q=notification_q,
app=app,
- datastore=datastore
+ datastore=datastore,
)
-
+
if health_result['status'] != 'healthy':
logger.warning(f"Worker health check: {health_result['message']}")
last_health_check = now
- # Check if all checks are paused
- if datastore.data['settings']['application'].get('all_paused', False):
- app.config.exit.wait(1)
- continue
-
# Get a list of watches by UUID that are currently fetching data
running_uuids = worker_pool.get_running_uuids()
@@ -1303,10 +1756,13 @@ def ticker_thread_check_time_launch_checks():
try:
# Get a list of watches sorted by last_checked, [1] because it gets passed a tuple
# This is so we examine the most over-due first
- for k in sorted(datastore.data['watching'].items(), key=lambda item: item[1].get('last_checked',0)):
+ for k in sorted(
+ datastore.data['watching'].items(),
+ key=lambda item: item[1].get('last_checked', 0),
+ ):
watch_uuid_list.append(k[0])
- except RuntimeError as e:
+ except RuntimeError:
# RuntimeError: dictionary changed size during iteration
time.sleep(0.1)
watch_uuid_list = []
@@ -1321,9 +1777,16 @@ def ticker_thread_check_time_launch_checks():
if watch_index % 100 == 0:
current_queue_size = update_q.qsize()
if current_queue_size >= MAX_QUEUE_SIZE:
- logger.debug(f"Queue size limit reached ({current_queue_size}/{MAX_QUEUE_SIZE}), stopping scheduler this iteration.")
+ logger.debug(
+ f"Queue size limit reached ({current_queue_size}/{MAX_QUEUE_SIZE}), stopping scheduler this iteration."
+ )
break
+ # Check if all checks are paused - this loop could get stuck on very long lists of watches, best to check here.
+ if datastore.data['settings']['application'].get('all_paused', False):
+ app.config.exit.wait(1)
+ break
+
now = time.time()
watch = datastore.data['watching'].get(uuid)
if not watch:
@@ -1338,31 +1801,46 @@ def ticker_thread_check_time_launch_checks():
# Time schedule limit - Decide between watch or global settings
scheduler_source = None
if watch.get('time_between_check_use_default'):
- time_schedule_limit = datastore.data['settings']['requests'].get('time_schedule_limit', {})
+ time_schedule_limit = datastore.data['settings']['requests'].get(
+ 'time_schedule_limit', {}
+ )
scheduler_source = 'system/global settings'
else:
time_schedule_limit = watch.get('time_schedule_limit')
scheduler_source = 'watch'
- tz_name = datastore.data['settings']['application'].get('scheduler_timezone_default', os.getenv('TZ', 'UTC').strip())
+ tz_name = default_timezone_name(
+ datastore.data['settings']['application'].get('scheduler_timezone_default')
+ )
if time_schedule_limit and time_schedule_limit.get('enabled'):
- logger.trace(f"{uuid} Time scheduler - Using scheduler settings from {scheduler_source}")
+ logger.trace(
+ f"{uuid} Time scheduler - Using scheduler settings from {scheduler_source}"
+ )
try:
- result = is_within_schedule(time_schedule_limit=time_schedule_limit,
- default_tz=tz_name
- )
+ result = is_within_schedule(
+ time_schedule_limit=time_schedule_limit, default_tz=tz_name
+ )
if not result:
logger.trace(f"{uuid} Time scheduler - not within schedule skipping.")
continue
except Exception as e:
+ # `continue`, never `return` — this runs inside the ticker thread's
+ # main `while not exit.is_set()` loop, so returning here killed the
+ # scheduler outright and no watch was ever checked again until
+ # restart. One watch with a bad schedule must not stop the others.
logger.error(
- f"{uuid} - Recheck scheduler, error handling timezone, check skipped - TZ name '{tz_name}' - {str(e)}")
- return False
+ f"{uuid} - Recheck scheduler, error handling timezone, check skipped - TZ name '{tz_name}' - {str(e)}"
+ )
+ continue
# If they supplied an individual entry minutes to threshold.
- threshold = recheck_time_system_seconds if watch.get('time_between_check_use_default') else watch.threshold_seconds()
+ threshold = (
+ recheck_time_system_seconds
+ if watch.get('time_between_check_use_default')
+ else watch.threshold_seconds()
+ )
# #580 - Jitter plus/minus amount of time to make the check seem more random to the server
jitter = datastore.data['settings']['requests'].get('jitter_seconds', 0)
@@ -1372,23 +1850,29 @@ def ticker_thread_check_time_launch_checks():
seconds_since_last_recheck = now - watch['last_checked']
- if seconds_since_last_recheck >= (threshold + watch.jitter_seconds) and seconds_since_last_recheck >= recheck_time_minimum_seconds:
- if not uuid in running_uuids and uuid not in queued_uuids:
-
+ if (
+ seconds_since_last_recheck >= (threshold + watch.jitter_seconds)
+ and seconds_since_last_recheck >= recheck_time_minimum_seconds
+ ):
+ if uuid not in running_uuids and uuid not in queued_uuids:
# Proxies can be set to have a limit on seconds between which they can be called
watch_proxy = datastore.get_preferred_proxy_for_watch(uuid=uuid)
if watch_proxy and watch_proxy in list(datastore.proxy_list.keys()):
# Proxy may also have some threshold minimum
- proxy_list_reuse_time_minimum = int(datastore.proxy_list.get(watch_proxy, {}).get('reuse_time_minimum', 0))
+ proxy_list_reuse_time_minimum = int(
+ datastore.proxy_list.get(watch_proxy, {}).get('reuse_time_minimum', 0)
+ )
if proxy_list_reuse_time_minimum:
proxy_last_used_time = proxy_last_called_time.get(watch_proxy, 0)
time_since_proxy_used = int(time.time() - proxy_last_used_time)
if time_since_proxy_used < proxy_list_reuse_time_minimum:
# Not enough time difference reached, skip this watch
- logger.debug(f"> Skipped UUID {uuid} "
- f"using proxy '{watch_proxy}', not "
- f"enough time between proxy requests "
- f"{time_since_proxy_used}s/{proxy_list_reuse_time_minimum}s")
+ logger.debug(
+ f"> Skipped UUID {uuid} "
+ f"using proxy '{watch_proxy}', not "
+ f"enough time between proxy requests "
+ f"{time_since_proxy_used}s/{proxy_list_reuse_time_minimum}s"
+ )
continue
else:
# Record the last used time
@@ -1398,20 +1882,23 @@ def ticker_thread_check_time_launch_checks():
priority = int(time.time())
# Into the queue with you
- queued_successfully = worker_pool.queue_item_async_safe(update_q,
- queuedWatchMetaData.PrioritizedItem(priority=priority,
- item={'uuid': uuid})
- )
+ queued_successfully = worker_pool.queue_item_async_safe(
+ update_q,
+ queuedWatchMetaData.PrioritizedItem(priority=priority, item={'uuid': uuid}),
+ )
if queued_successfully:
logger.debug(
f"> Queued watch UUID {uuid} "
f"Checked at {watch['last_checked']} "
f"queued at {now:0.2f} priority {priority} "
f"jitter {watch.jitter_seconds:0.2f}s, "
- f"{now - watch['last_checked']:0.2f}s since Checked")
+ f"{now - watch['last_checked']:0.2f}s since Checked"
+ )
else:
- logger.critical(f"CRITICAL: Failed to queue watch UUID {uuid} in ticker thread!")
-
+ logger.critical(
+ f"CRITICAL: Failed to queue watch UUID {uuid} in ticker thread!"
+ )
+
# Reset for next time
watch.jitter_seconds = 0
diff --git a/changedetectionio/forms.py b/changedetectionio/forms.py
index e07b662eb..eff42701b 100644
--- a/changedetectionio/forms.py
+++ b/changedetectionio/forms.py
@@ -4,9 +4,16 @@ from loguru import logger
from wtforms.widgets.core import TimeInput
from flask_babel import lazy_gettext as _l, gettext
+from changedetectionio.blueprint.menu_modes import MENU_SIDEBAR_ACTIONMODES, MENU_SIDEBAR_ACTIONMODES_DEFAULT
from changedetectionio.blueprint.rss import RSS_FORMAT_TYPES, RSS_TEMPLATE_TYPE_OPTIONS, RSS_TEMPLATE_HTML_DEFAULT
from changedetectionio.llm.ui_strings import LLM_INTENT_WATCH_PLACEHOLDER
-from changedetectionio.llm.evaluator import DEFAULT_CHANGE_SUMMARY_PROMPT, LLM_DEFAULT_MAX_SUMMARY_TOKENS, LLM_DEFAULT_THINKING_BUDGET
+from changedetectionio.llm.evaluator import (
+ DEFAULT_CHANGE_SUMMARY_PROMPT,
+ LLM_DEFAULT_MAX_SUMMARY_TOKENS,
+ LLM_DEFAULT_THINKING_BUDGET,
+ LLM_PROMPT_MODE_APPEND,
+ LLM_PROMPT_MODE_REPLACE,
+)
from changedetectionio.conditions.form import ConditionFormRow
from changedetectionio.notification_service import NotificationContextData
from changedetectionio.strtobool import strtobool
@@ -481,6 +488,42 @@ class ValidateContentFetcherIsReady(object):
# raise ValidationError(message % (field.data, e))
+class ValidateKnownContentFetcher(object):
+ """The posted fetch_backend has to name a fetcher this install actually has.
+
+ Deliberately *not* a live-preview capability check. This validator sits on the
+ shared quick-add form, whose POST endpoint is also how the watch list (and tests,
+ and scripts) add a watch with any legal backend - 'html_requests' included. Which
+ browsers the Add-Watch page *offers* is a rendering decision (see the add_watch_ui
+ blueprint's browser_config), and whether one can render a live preview is enforced
+ where that matters, in /snapshot.
+
+ Optional: no value posted means "leave it on the system default", as before.
+ """
+
+ def __init__(self, message=None):
+ self.message = message
+
+ def __call__(self, form, field):
+ from flask import current_app
+ from changedetectionio import content_fetchers
+
+ if not field.data:
+ return
+
+ allowed = {'system'} | {name for name, _description in content_fetchers.available_fetchers()}
+ datastore = current_app.config.get('DATASTORE')
+ if datastore:
+ allowed |= {value for value, _label in datastore.extra_browsers}
+ # A saved browser config is picked by its id, not an engine name (it maps to one) -
+ # the Add-Watch browser list offers these, so the shared quick-add has to accept them.
+ allowed |= set(datastore.browser_config_store.all().keys())
+
+ if field.data not in allowed:
+ logger.warning(f"Rejected unknown fetch_backend {field.data!r} - known: {sorted(allowed)}")
+ raise ValidationError(self.message or gettext("Unknown fetch method."))
+
+
class ValidateNotificationBodyAndTitleWhenURLisSet(object):
"""
Validates that they entered something in both notification title+body when the URL is set
@@ -657,12 +700,18 @@ class ValidateCSSJSONXPATHInput(object):
raise ValidationError("XPath not permitted in this field!")
from lxml import etree, html
import elementpath
- from changedetectionio.html_tools import SafeXPath3Parser
- tree = html.fromstring("")
+ from changedetectionio.html_tools import get_safe_xpath3_parser, lxml_guard, lxml_html_parser, \
+ XPATH_CODEPOINT_COLLATION
line = line.replace('xpath:', '')
try:
- elementpath.select(tree, line.strip(), parser=SafeXPath3Parser)
+ # Runs on a Flask request thread - must share the worker's lxml lock.
+ with lxml_guard():
+ tree = html.fromstring("", parser=lxml_html_parser())
+ # Same collation the filter will actually run under, so validation
+ # cannot accept an expression that then behaves differently at check time.
+ elementpath.select(tree, line.strip(), parser=get_safe_xpath3_parser(),
+ default_collation=XPATH_CODEPOINT_COLLATION)
except elementpath.ElementPathError as e:
message = field.gettext('\'%(expression)s\' is not a valid XPath expression. (%(error)s)')
raise ValidationError(message % {'expression': line, 'error': str(e)})
@@ -673,11 +722,14 @@ class ValidateCSSJSONXPATHInput(object):
if not self.allow_xpath:
raise ValidationError("XPath not permitted in this field!")
from lxml import etree, html
- tree = html.fromstring("")
+ from changedetectionio.html_tools import lxml_guard, lxml_html_parser
line = re.sub(r'^xpath1:', '', line)
try:
- tree.xpath(line.strip())
+ # Runs on a Flask request thread - must share the worker's lxml lock.
+ with lxml_guard():
+ tree = html.fromstring("", parser=lxml_html_parser())
+ tree.xpath(line.strip())
except etree.XPathEvalError as e:
message = field.gettext('\'%(expression)s\' is not a valid XPath expression. (%(error)s)')
raise ValidationError(message % {'expression': line, 'error': str(e)})
@@ -773,11 +825,41 @@ class ValidateStartsWithRegex(object):
if not self.pattern.match(stripped):
raise ValidationError(self.message or _l("Invalid value."))
+def visual_browser_choices():
+ """Browsers that can render the Add-Watch live preview, as RadioField choices.
+
+ Lazy import (the add_watch_ui blueprint imports this module) and empty outside an
+ app context, because WTForms evaluates a choices callable on field construction.
+ """
+ from flask import current_app, has_app_context
+ from changedetectionio.blueprint.add_watch_ui import browser_config
+
+ if not has_app_context():
+ return []
+ datastore = current_app.config.get('DATASTORE')
+ return browser_config.radio_choices(datastore) if datastore else []
+
+
class quickWatchForm(Form):
url = StringField('URL', validators=[validateURL()])
tags = StringTagUUID(_l('Group tag'), validators=[validators.Optional()])
watch_submit_button = SubmitField(_l('Watch'), render_kw={"class": "pure-button pure-button-primary"})
processor = RadioField(_l('Processor'), choices=lambda: processors.available_processors(), default=processors.get_default_processor)
+ # Only the Add-Watch page renders this; the watch-list quick-add posts nothing, which
+ # leaves the new watch on 'system' exactly as before.
+ #
+ # A radio list rather than a dropdown: fetcher descriptions run long (they include the
+ # driver URL) and a wrapping label reads fine in a narrow pane, where a