mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-10-05 11:38:22 +00:00
An "extra browser" was a name + a ws(s):// endpoint in settings.requests.extra_browsers,
selected by a watch as the magic string 'extra_browser_<name>'. That string resolved to
html_webdriver plus a custom connection URL, which meant the protocol the endpoint was
spoken to came from env vars rather than from the entry: CDP over a WebSocket with
PLAYWRIGHT_DRIVER_URL set, CDP via pyppeteer with FAST_PUPPETEER_CHROME_FETCHER, and the
W3C WebDriver protocol over HTTP on a Selenium-only install - where a wss:// URL cannot
work at all. The form only ever accepted ws:// / wss://, so the feature was silently
broken on exactly the installs that could not honour it.
So it becomes an engine, html_external_cdp, which pins the protocol: a subclass of the
Playwright fetcher that takes its endpoint from the watch's browser config
(FetcherConfig.connection_url) instead of the environment. It is base-only
(ready_to_use=False) because an endpoint is required, so each endpoint is one browser
config ("variation") on the Browsers page - which is what the old settings list was.
update_36 migrates each extra_browsers row to such a variation, keyed by the SAME
'extra_browser_<name>' string watches already hold, so no watch, group override, API value
or global default needs rewriting; the legacy selector simply becomes a real browser-config
id. A row whose endpoint the model rejects is logged and skipped rather than taking the
update chain, and with it startup, down.
Knock-on cleanups, all of which delete a special case rather than add one:
- The proxy opt-out for custom endpoints is now Fetcher.ignores_proxy_setting, asked of
the engine, instead of a string-prefix test in call_browser().
- A live browser-steps / visual-selector session asks the engine where to connect
(Fetcher.browser_steps_connection_url, overridden by html_external_cdp) and refuses an
engine whose supports_browser_steps is False, instead of reading the env var itself and
silently stepping a browser the watch does not check with. That refusal is real: on a
Selenium install html_webdriver cannot drive a live session.
- is_valid_browser_selector() answers "may a watch store this in fetch_backend?" in one
place; the API (create/update/import), the quick-add form validator and the bulk "set
browser" operation each had their own copy, which is how they came to disagree about
whether a browser-config id was acceptable.
- api-spec.yaml's fetch_backend pattern enumerated extra_browser_* while rejecting
browser-config ids and every engine newer than html_webdriver. Valid values are
per-install, so the schema now bounds the string and the handlers do the real check.
- html_external_cdp registers unconditionally (unlike html_playwright_builtin): migrated
configs name it, so it must resolve even without the playwright library, or those
watches would quietly fetch with the plain HTTP client. The library is imported lazily
inside run(), and an unavailable engine now warns instead of falling back silently.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
1024 lines
48 KiB
Python
1024 lines
48 KiB
Python
"""
|
|
Schema update migrations for the datastore.
|
|
|
|
This module contains all schema version upgrade methods (update_1 through update_N).
|
|
These are mixed into ChangeDetectionStore to keep the main store file focused.
|
|
|
|
IMPORTANT: Each update could be run even when they have a new install and the schema is correct.
|
|
Therefore - each `update_n` should be very careful about checking if it needs to actually run.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import shutil
|
|
import tarfile
|
|
import time
|
|
from loguru import logger
|
|
from copy import deepcopy
|
|
from pydantic import ValidationError
|
|
|
|
|
|
# Try to import orjson for faster JSON serialization
|
|
try:
|
|
import orjson
|
|
HAS_ORJSON = True
|
|
except ImportError:
|
|
HAS_ORJSON = False
|
|
|
|
from ..html_tools import TRANSLATE_WHITESPACE_TABLE
|
|
from ..processors.restock_diff import Restock, get_price_from_history_str
|
|
from ..blueprint.rss import RSS_CONTENT_FORMAT_DEFAULT
|
|
from ..model import USE_SYSTEM_DEFAULT_NOTIFICATION_FORMAT_FOR_WATCH
|
|
|
|
def create_backup_tarball(datastore_path, update_number):
|
|
"""
|
|
Create a tarball backup of the entire datastore structure before running an update.
|
|
|
|
Includes:
|
|
- All {uuid}/watch.json files
|
|
- All {uuid}/tag.json files
|
|
- changedetection.json (settings, if it exists)
|
|
- url-watches.json (legacy format, if it exists)
|
|
- Directory structure preserved
|
|
|
|
Args:
|
|
datastore_path: Path to datastore directory
|
|
update_number: Update number being applied
|
|
|
|
Returns:
|
|
str: Path to created tarball, or None if backup failed
|
|
|
|
Restoration:
|
|
To restore from a backup:
|
|
cd /path/to/datastore
|
|
tar -xzf before-update-N-timestamp.tar.gz
|
|
This will restore all watch.json and tag.json files and settings to their pre-update state.
|
|
"""
|
|
timestamp = int(time.time())
|
|
backup_filename = f"before-update-{update_number}-{timestamp}.tar.gz"
|
|
backup_path = os.path.join(datastore_path, backup_filename)
|
|
|
|
try:
|
|
logger.info(f"Creating backup tarball: {backup_filename}")
|
|
|
|
with tarfile.open(backup_path, "w:gz") as tar:
|
|
# Backup changedetection.json if it exists (new format)
|
|
changedetection_json = os.path.join(datastore_path, "changedetection.json")
|
|
if os.path.isfile(changedetection_json):
|
|
tar.add(changedetection_json, arcname="changedetection.json")
|
|
logger.debug("Added changedetection.json to backup")
|
|
|
|
# Backup url-watches.json if it exists (legacy format)
|
|
url_watches_json = os.path.join(datastore_path, "url-watches.json")
|
|
if os.path.isfile(url_watches_json):
|
|
tar.add(url_watches_json, arcname="url-watches.json")
|
|
logger.debug("Added url-watches.json to backup")
|
|
|
|
# Backup all watch/tag directories with their JSON files
|
|
# This preserves the UUID directory structure
|
|
watch_count = 0
|
|
tag_count = 0
|
|
for entry in os.listdir(datastore_path):
|
|
entry_path = os.path.join(datastore_path, entry)
|
|
|
|
# Skip if not a directory
|
|
if not os.path.isdir(entry_path):
|
|
continue
|
|
|
|
# Skip hidden directories and backup directories
|
|
if entry.startswith('.') or entry.startswith('before-update-'):
|
|
continue
|
|
|
|
# Backup watch.json if exists
|
|
watch_json = os.path.join(entry_path, "watch.json")
|
|
if os.path.isfile(watch_json):
|
|
tar.add(watch_json, arcname=f"{entry}/watch.json")
|
|
watch_count += 1
|
|
|
|
if watch_count % 100 == 0:
|
|
logger.debug(f"Backed up {watch_count} watch.json files...")
|
|
|
|
# Backup tag.json if exists
|
|
tag_json = os.path.join(entry_path, "tag.json")
|
|
if os.path.isfile(tag_json):
|
|
tar.add(tag_json, arcname=f"{entry}/tag.json")
|
|
tag_count += 1
|
|
|
|
logger.success(f"Backup created: {backup_filename} ({watch_count} watches from disk, {tag_count} tags from disk)")
|
|
return backup_path
|
|
|
|
except Exception as e:
|
|
logger.error(f"Failed to create backup tarball: {e}")
|
|
# Try to clean up partial backup
|
|
if os.path.exists(backup_path):
|
|
try:
|
|
os.unlink(backup_path)
|
|
except:
|
|
pass
|
|
return None
|
|
|
|
|
|
class DatastoreUpdatesMixin:
|
|
"""
|
|
Mixin class containing all schema update methods.
|
|
|
|
This class is inherited by ChangeDetectionStore to provide schema migration functionality.
|
|
Each update_N method upgrades the schema from version N-1 to version N.
|
|
"""
|
|
|
|
def get_updates_available(self):
|
|
"""
|
|
Discover all available update methods.
|
|
|
|
Returns:
|
|
list: Sorted list of update version numbers (e.g., [1, 2, 3, ..., 26])
|
|
"""
|
|
import inspect
|
|
updates_available = []
|
|
for i, o in inspect.getmembers(self, predicate=inspect.ismethod):
|
|
m = re.search(r'update_(\d+)$', i)
|
|
if m:
|
|
updates_available.append(int(m.group(1)))
|
|
updates_available.sort()
|
|
|
|
return updates_available
|
|
|
|
def run_updates(self, current_schema_version=None):
|
|
import sys
|
|
"""
|
|
Run all pending schema updates sequentially.
|
|
|
|
Args:
|
|
current_schema_version: Optional current schema version. If provided, only run updates
|
|
greater than this version. If None, uses the schema version from
|
|
the datastore. If no schema version exists in datastore and it appears
|
|
to be a fresh install, sets to latest update number (no updates needed).
|
|
|
|
IMPORTANT: Each update could be run even when they have a new install and the schema is correct.
|
|
Therefore - each `update_n` should be very careful about checking if it needs to actually run.
|
|
|
|
Process:
|
|
1. Get list of available updates
|
|
2. For each update > current schema version:
|
|
- Create backup of datastore
|
|
- Run update method
|
|
- Update schema version and commit settings
|
|
- Commit all watches and tags
|
|
3. If any update fails, stop processing
|
|
4. All changes saved via individual .commit() calls
|
|
"""
|
|
updates_available = self.get_updates_available()
|
|
if self.data.get('watching'):
|
|
test_watch = self.data['watching'].get(next(iter(self.data.get('watching', {}))))
|
|
from ..model.Watch import model
|
|
|
|
if not isinstance(test_watch, model):
|
|
import sys
|
|
logger.critical("Cannot run updates! Watch structure must be re-hydrated back to a Watch model object!")
|
|
sys.exit(1)
|
|
|
|
if self.data['settings']['application'].get('tags',{}):
|
|
test_tag = self.data['settings']['application'].get('tags',{}).get(next(iter(self.data['settings']['application'].get('tags',{}))))
|
|
from ..model.Tag import model as tag_model
|
|
|
|
if not isinstance(test_tag, tag_model):
|
|
import sys
|
|
logger.critical("Cannot run updates! Watch tag/group structure must be re-hydrated back to a Tag model object!")
|
|
sys.exit(1)
|
|
|
|
# Determine current schema version
|
|
if current_schema_version is None:
|
|
# Check if schema_version exists in datastore
|
|
current_schema_version = self.data['settings']['application'].get('schema_version')
|
|
|
|
if current_schema_version is None:
|
|
# No schema version found - could be a fresh install or very old datastore
|
|
# If this is a fresh/new config with no watches, assume it's up-to-date
|
|
# and set to latest update number (no updates needed)
|
|
if len(self.data['watching']) == 0:
|
|
# Get the highest update number from available update methods
|
|
latest_update = updates_available[-1] if updates_available else 0
|
|
logger.info(f"No schema version found and no watches exist - assuming fresh install, setting schema_version to {latest_update}")
|
|
self.data['settings']['application']['schema_version'] = latest_update
|
|
self.commit()
|
|
return # No updates needed for fresh install
|
|
else:
|
|
# Has watches but no schema version - likely old datastore, run all updates
|
|
logger.warning("No schema version found but watches exist - running all updates from version 0")
|
|
current_schema_version = 0
|
|
|
|
logger.info(f"Current schema version: {current_schema_version}")
|
|
|
|
updates_ran = []
|
|
|
|
for update_n in updates_available:
|
|
if update_n > current_schema_version:
|
|
logger.critical(f"Applying update_{update_n}")
|
|
|
|
# Create tarball backup of entire datastore structure
|
|
# This includes all watch.json files, settings, and preserves directory structure
|
|
backup_path = create_backup_tarball(self.datastore_path, update_n)
|
|
if backup_path:
|
|
logger.info(f"Backup created at: {backup_path}")
|
|
else:
|
|
logger.warning("Backup creation failed, but continuing with update")
|
|
|
|
try:
|
|
update_method = getattr(self, f"update_{update_n}")()
|
|
except Exception as e:
|
|
logger.critical(f"Error while trying update_{update_n}")
|
|
logger.exception(e)
|
|
sys.exit(1)
|
|
else:
|
|
# Bump the version
|
|
self.data['settings']['application']['schema_version'] = update_n
|
|
self.commit()
|
|
|
|
logger.success(f"Update {update_n} completed")
|
|
|
|
# Track which updates ran
|
|
updates_ran.append(update_n)
|
|
|
|
# ============================================================================
|
|
# Individual Update Methods
|
|
# ============================================================================
|
|
|
|
def update_1(self):
|
|
"""Convert minutes to seconds on settings and each watch."""
|
|
if self.data['settings']['requests'].get('minutes_between_check'):
|
|
self.data['settings']['requests']['time_between_check']['minutes'] = self.data['settings']['requests']['minutes_between_check']
|
|
# Remove the default 'hours' that is set from the model
|
|
self.data['settings']['requests']['time_between_check']['hours'] = None
|
|
|
|
for uuid, watch in self.data['watching'].items():
|
|
if 'minutes_between_check' in watch:
|
|
# Only upgrade individual watch time if it was set
|
|
if watch.get('minutes_between_check', False):
|
|
self.data['watching'][uuid]['time_between_check']['minutes'] = watch['minutes_between_check']
|
|
|
|
def update_2(self):
|
|
"""
|
|
Move the history list to a flat text file index.
|
|
Better than SQLite because this list is only appended to, and works across NAS / NFS type setups.
|
|
"""
|
|
# @todo test running this on a newly updated one (when this already ran)
|
|
for uuid, watch in self.data['watching'].items():
|
|
history = []
|
|
|
|
if watch.get('history', False):
|
|
for d, p in watch['history'].items():
|
|
d = int(d) # Used to be keyed as str, we'll fix this now too
|
|
history.append("{},{}\n".format(d, p))
|
|
|
|
if len(history):
|
|
target_path = os.path.join(self.datastore_path, uuid)
|
|
if os.path.exists(target_path):
|
|
with open(os.path.join(target_path, "history.txt"), "w") as f:
|
|
f.writelines(history)
|
|
else:
|
|
logger.warning(f"Datastore history directory {target_path} does not exist, skipping history import.")
|
|
|
|
# No longer needed, dynamically pulled from the disk when needed.
|
|
# But we should set it back to a empty dict so we don't break if this schema runs on an earlier version.
|
|
# In the distant future we can remove this entirely
|
|
self.data['watching'][uuid]['history'] = {}
|
|
|
|
def update_3(self):
|
|
"""We incorrectly stored last_changed when there was not a change, and then confused the output list table."""
|
|
# see https://github.com/dgtlmoon/changedetection.io/pull/835
|
|
return
|
|
|
|
def update_4(self):
|
|
"""`last_changed` not needed, we pull that information from the history.txt index."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
try:
|
|
# Remove it from the struct
|
|
del(watch['last_changed'])
|
|
except:
|
|
continue
|
|
return
|
|
|
|
def update_5(self):
|
|
"""
|
|
If the watch notification body, title look the same as the global one, unset it, so the watch defaults back to using the main settings.
|
|
In other words - the watch notification_title and notification_body are not needed if they are the same as the default one.
|
|
"""
|
|
current_system_body = self.data['settings']['application']['notification_body'].translate(TRANSLATE_WHITESPACE_TABLE)
|
|
current_system_title = self.data['settings']['application']['notification_body'].translate(TRANSLATE_WHITESPACE_TABLE)
|
|
for uuid, watch in self.data['watching'].items():
|
|
try:
|
|
watch_body = watch.get('notification_body', '')
|
|
if watch_body and watch_body.translate(TRANSLATE_WHITESPACE_TABLE) == current_system_body:
|
|
# Looks the same as the default one, so unset it
|
|
watch['notification_body'] = None
|
|
|
|
watch_title = watch.get('notification_title', '')
|
|
if watch_title and watch_title.translate(TRANSLATE_WHITESPACE_TABLE) == current_system_title:
|
|
# Looks the same as the default one, so unset it
|
|
watch['notification_title'] = None
|
|
except Exception as e:
|
|
continue
|
|
return
|
|
|
|
def update_7(self):
|
|
"""
|
|
We incorrectly used common header overrides that should only apply to Requests.
|
|
These are now handled in content_fetcher::html_requests and shouldnt be passed to Playwright/Selenium.
|
|
"""
|
|
# These were hard-coded in early versions
|
|
for v in ['User-Agent', 'Accept', 'Accept-Encoding', 'Accept-Language']:
|
|
if self.data['settings']['headers'].get(v):
|
|
del self.data['settings']['headers'][v]
|
|
|
|
def update_8(self):
|
|
"""Convert filters to a list of filters css_filter -> include_filters."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
try:
|
|
existing_filter = watch.get('css_filter', '')
|
|
if existing_filter:
|
|
watch['include_filters'] = [existing_filter]
|
|
except:
|
|
continue
|
|
return
|
|
|
|
def update_9(self):
|
|
"""Convert old static notification tokens to jinja2 tokens."""
|
|
# Each watch
|
|
# only { } not {{ or }}
|
|
r = r'(?<!{){(?!{)(\w+)(?<!})}(?!})'
|
|
for uuid, watch in self.data['watching'].items():
|
|
try:
|
|
n_body = watch.get('notification_body', '')
|
|
if n_body:
|
|
watch['notification_body'] = re.sub(r, r'{{\1}}', n_body)
|
|
|
|
n_title = watch.get('notification_title')
|
|
if n_title:
|
|
watch['notification_title'] = re.sub(r, r'{{\1}}', n_title)
|
|
|
|
n_urls = watch.get('notification_urls')
|
|
if n_urls:
|
|
for i, url in enumerate(n_urls):
|
|
watch['notification_urls'][i] = re.sub(r, r'{{\1}}', url)
|
|
|
|
except:
|
|
continue
|
|
|
|
# System wide
|
|
n_body = self.data['settings']['application'].get('notification_body')
|
|
if n_body:
|
|
self.data['settings']['application']['notification_body'] = re.sub(r, r'{{\1}}', n_body)
|
|
|
|
n_title = self.data['settings']['application'].get('notification_title')
|
|
if n_body:
|
|
self.data['settings']['application']['notification_title'] = re.sub(r, r'{{\1}}', n_title)
|
|
|
|
n_urls = self.data['settings']['application'].get('notification_urls')
|
|
if n_urls:
|
|
for i, url in enumerate(n_urls):
|
|
self.data['settings']['application']['notification_urls'][i] = re.sub(r, r'{{\1}}', url)
|
|
|
|
return
|
|
|
|
def update_10(self):
|
|
"""Some setups may have missed the correct default, so it shows the wrong config in the UI, although it will default to system-wide."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
try:
|
|
if not watch.get('fetch_backend', ''):
|
|
watch['fetch_backend'] = 'system'
|
|
except:
|
|
continue
|
|
return
|
|
|
|
def update_12(self):
|
|
"""Create tag objects and their references from existing tag text."""
|
|
i = 0
|
|
for uuid, watch in self.data['watching'].items():
|
|
# Split out and convert old tag string
|
|
tag = watch.get('tag')
|
|
if tag:
|
|
tag_uuids = []
|
|
for t in tag.split(','):
|
|
tag_uuids.append(self.add_tag(title=t))
|
|
|
|
self.data['watching'][uuid]['tags'] = tag_uuids
|
|
|
|
def update_13(self):
|
|
"""#1775 - Update 11 did not update the records correctly when adding 'date_created' values for sorting."""
|
|
i = 0
|
|
for uuid, watch in self.data['watching'].items():
|
|
if not watch.get('date_created'):
|
|
self.data['watching'][uuid]['date_created'] = i
|
|
i += 1
|
|
return
|
|
|
|
def update_14(self):
|
|
"""#1774 - protect xpath1 against migration."""
|
|
for awatch in self.data["watching"]:
|
|
if self.data["watching"][awatch]['include_filters']:
|
|
for num, selector in enumerate(self.data["watching"][awatch]['include_filters']):
|
|
if selector.startswith('/'):
|
|
self.data["watching"][awatch]['include_filters'][num] = 'xpath1:' + selector
|
|
if selector.startswith('xpath:'):
|
|
self.data["watching"][awatch]['include_filters'][num] = selector.replace('xpath:', 'xpath1:', 1)
|
|
|
|
def update_15(self):
|
|
"""Use more obvious default time setting."""
|
|
for uuid in self.data["watching"]:
|
|
if self.data["watching"][uuid]['time_between_check'] == self.data['settings']['requests']['time_between_check']:
|
|
# What the old logic was, which was pretty confusing
|
|
self.data["watching"][uuid]['time_between_check_use_default'] = True
|
|
elif all(value is None or value == 0 for value in self.data["watching"][uuid]['time_between_check'].values()):
|
|
self.data["watching"][uuid]['time_between_check_use_default'] = True
|
|
else:
|
|
# Something custom here
|
|
self.data["watching"][uuid]['time_between_check_use_default'] = False
|
|
|
|
def update_16(self):
|
|
"""Correctly set datatype for older installs where 'tag' was string and update_12 did not catch it."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
if isinstance(watch.get('tags'), str):
|
|
self.data['watching'][uuid]['tags'] = []
|
|
|
|
def update_17(self):
|
|
"""Migrate old 'in_stock' values to the new Restock."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
if 'in_stock' in watch:
|
|
watch['restock'] = Restock({'in_stock': watch.get('in_stock')})
|
|
del watch['in_stock']
|
|
|
|
def update_18(self):
|
|
"""Migrate old restock settings."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
if not watch.get('restock_settings'):
|
|
# So we enable price following by default
|
|
self.data['watching'][uuid]['restock_settings'] = {'follow_price_changes': True}
|
|
|
|
# Migrate and cleanoff old value
|
|
self.data['watching'][uuid]['restock_settings']['in_stock_processing'] = 'in_stock_only' if watch.get(
|
|
'in_stock_only') else 'all_changes'
|
|
|
|
if self.data['watching'][uuid].get('in_stock_only'):
|
|
del (self.data['watching'][uuid]['in_stock_only'])
|
|
|
|
def update_19(self):
|
|
"""Compress old elements.json to elements.deflate, saving disk, this compression is pretty fast."""
|
|
import zlib
|
|
|
|
for uuid, watch in self.data['watching'].items():
|
|
json_path = os.path.join(self.datastore_path, uuid, "elements.json")
|
|
deflate_path = os.path.join(self.datastore_path, uuid, "elements.deflate")
|
|
|
|
if os.path.exists(json_path):
|
|
with open(json_path, "rb") as f_j:
|
|
with open(deflate_path, "wb") as f_d:
|
|
logger.debug(f"Compressing {str(json_path)} to {str(deflate_path)}..")
|
|
f_d.write(zlib.compress(f_j.read()))
|
|
os.unlink(json_path)
|
|
|
|
def update_20(self):
|
|
"""Migrate extract_title_as_title to use_page_title_in_list."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
if self.data['watching'][uuid].get('extract_title_as_title'):
|
|
self.data['watching'][uuid]['use_page_title_in_list'] = self.data['watching'][uuid].get('extract_title_as_title')
|
|
del self.data['watching'][uuid]['extract_title_as_title']
|
|
|
|
if self.data['settings']['application'].get('extract_title_as_title'):
|
|
# Ensure 'ui' key exists (defensive for edge cases where base_config merge didn't happen)
|
|
if 'ui' not in self.data['settings']['application']:
|
|
self.data['settings']['application']['ui'] = {
|
|
'use_page_title_in_list': True,
|
|
'open_diff_in_new_tab': True,
|
|
'socket_io_enabled': True,
|
|
'favicons_enabled': True
|
|
}
|
|
self.data['settings']['application']['ui']['use_page_title_in_list'] = self.data['settings']['application'].get('extract_title_as_title')
|
|
|
|
def update_21(self):
|
|
"""Migrate timezone to scheduler_timezone_default."""
|
|
if self.data['settings']['application'].get('timezone'):
|
|
self.data['settings']['application']['scheduler_timezone_default'] = self.data['settings']['application'].get('timezone')
|
|
del self.data['settings']['application']['timezone']
|
|
|
|
def update_23(self):
|
|
"""Some notification formats got the wrong name type."""
|
|
|
|
def re_run(formats):
|
|
sys_n_format = self.data['settings']['application'].get('notification_format')
|
|
key_exists_as_value = next((k for k, v in formats.items() if v == sys_n_format), None)
|
|
if key_exists_as_value: # key of "Plain text"
|
|
logger.success(f"['settings']['application']['notification_format'] '{sys_n_format}' -> '{key_exists_as_value}'")
|
|
self.data['settings']['application']['notification_format'] = key_exists_as_value
|
|
|
|
for uuid, watch in self.data['watching'].items():
|
|
n_format = self.data['watching'][uuid].get('notification_format')
|
|
key_exists_as_value = next((k for k, v in formats.items() if v == n_format), None)
|
|
if key_exists_as_value and key_exists_as_value != USE_SYSTEM_DEFAULT_NOTIFICATION_FORMAT_FOR_WATCH: # key of "Plain text"
|
|
logger.success(f"['watching'][{uuid}]['notification_format'] '{n_format}' -> '{key_exists_as_value}'")
|
|
self.data['watching'][uuid]['notification_format'] = key_exists_as_value # should be 'text' or whatever
|
|
|
|
for uuid, tag in self.data['settings']['application']['tags'].items():
|
|
n_format = self.data['settings']['application']['tags'][uuid].get('notification_format')
|
|
key_exists_as_value = next((k for k, v in formats.items() if v == n_format), None)
|
|
if key_exists_as_value and key_exists_as_value != USE_SYSTEM_DEFAULT_NOTIFICATION_FORMAT_FOR_WATCH: # key of "Plain text"
|
|
logger.success(
|
|
f"['settings']['application']['tags'][{uuid}]['notification_format'] '{n_format}' -> '{key_exists_as_value}'")
|
|
self.data['settings']['application']['tags'][uuid][
|
|
'notification_format'] = key_exists_as_value # should be 'text' or whatever
|
|
|
|
from ..notification import valid_notification_formats
|
|
formats = deepcopy(valid_notification_formats)
|
|
re_run(formats)
|
|
# And in previous versions, it was "text" instead of Plain text, Markdown instead of "Markdown to HTML"
|
|
formats['text'] = 'Text'
|
|
formats['markdown'] = 'Markdown'
|
|
re_run(formats)
|
|
|
|
def update_24(self):
|
|
"""RSS types should be inline with the same names as notification types."""
|
|
rss_format = self.data['settings']['application'].get('rss_content_format')
|
|
if not rss_format or 'text' in rss_format:
|
|
# might have been 'plaintext, 'plain text' or something
|
|
self.data['settings']['application']['rss_content_format'] = RSS_CONTENT_FORMAT_DEFAULT
|
|
elif 'html' in rss_format:
|
|
self.data['settings']['application']['rss_content_format'] = 'htmlcolor'
|
|
else:
|
|
# safe fallback to text
|
|
self.data['settings']['application']['rss_content_format'] = RSS_CONTENT_FORMAT_DEFAULT
|
|
|
|
def update_25(self):
|
|
"""Different processors now hold their own history.txt."""
|
|
for uuid, watch in self.data['watching'].items():
|
|
processor = self.data['watching'][uuid].get('processor')
|
|
if processor != 'text_json_diff':
|
|
old_history_txt = os.path.join(self.datastore_path, "history.txt")
|
|
target_history_name = f"history-{processor}.txt"
|
|
if os.path.isfile(old_history_txt) and not os.path.isfile(target_history_name):
|
|
new_history_txt = os.path.join(self.datastore_path, target_history_name)
|
|
logger.debug(f"Renaming history index {old_history_txt} to {new_history_txt}...")
|
|
shutil.move(old_history_txt, new_history_txt)
|
|
|
|
def migrate_legacy_db_format(self):
|
|
"""
|
|
Migration: Individual watch persistence (COPY-based, safe rollback).
|
|
|
|
Loads legacy url-watches.json format and migrates to:
|
|
- {uuid}/watch.json (per watch)
|
|
- changedetection.json (settings only)
|
|
|
|
IMPORTANT:
|
|
- A tarball backup (before-update-26-timestamp.tar.gz) is created before migration
|
|
- url-watches.json is LEFT INTACT for rollback safety
|
|
- Users can roll back by simply downgrading to the previous version
|
|
- Or restore from tarball: tar -xzf before-update-26-*.tar.gz
|
|
|
|
This is a dedicated migration release - users upgrade at their own pace.
|
|
"""
|
|
logger.critical("=" * 80)
|
|
logger.critical("Running migration: Individual watch persistence (update_26)")
|
|
logger.critical("COPY-based migration: url-watches.json will remain intact for rollback")
|
|
logger.critical("=" * 80)
|
|
|
|
# Populate settings from legacy data
|
|
logger.info("Populating settings from legacy data...")
|
|
watch_count = len(self.data['watching'])
|
|
logger.success(f"Loaded {watch_count} watches from legacy format")
|
|
|
|
# Phase 1: Save all watches to individual files
|
|
logger.critical(f"Phase 1/4: Saving {watch_count} watches to individual watch.json files...")
|
|
|
|
saved_count = 0
|
|
for uuid, watch in self.data['watching'].items():
|
|
try:
|
|
watch.commit()
|
|
saved_count += 1
|
|
|
|
if saved_count % 100 == 0:
|
|
logger.info(f" Progress: {saved_count}/{watch_count} watches migrated...")
|
|
|
|
except Exception as e:
|
|
logger.error(f"Failed to save watch {uuid}: {e}")
|
|
raise Exception(
|
|
f"Migration failed: Could not save watch {uuid}. "
|
|
f"url-watches.json remains intact, safe to retry. Error: {e}"
|
|
)
|
|
|
|
logger.critical(f"Phase 1 complete: Saved {saved_count} watches")
|
|
|
|
# Phase 2: Verify all files exist
|
|
logger.critical("Phase 2/4: Verifying all watch.json files were created...")
|
|
|
|
missing = []
|
|
for uuid in self.data['watching'].keys():
|
|
watch_json = os.path.join(self.datastore_path, uuid, "watch.json")
|
|
if not os.path.isfile(watch_json):
|
|
missing.append(uuid)
|
|
|
|
if missing:
|
|
raise Exception(
|
|
f"Migration failed: {len(missing)} watch files missing: {missing[:5]}... "
|
|
f"url-watches.json remains intact, safe to retry."
|
|
)
|
|
|
|
logger.critical(f"Phase 2 complete: Verified {watch_count} watch files")
|
|
|
|
# Phase 3: Create new settings file
|
|
logger.critical("Phase 3/4: Creating changedetection.json...")
|
|
|
|
try:
|
|
self._save_settings()
|
|
except Exception as e:
|
|
logger.error(f"Failed to create changedetection.json: {e}")
|
|
raise Exception(
|
|
f"Migration failed: Could not create changedetection.json. "
|
|
f"url-watches.json remains intact, safe to retry. Error: {e}"
|
|
)
|
|
|
|
# Phase 4: Verify settings file exists
|
|
logger.critical("Phase 4/4: Verifying changedetection.json exists...")
|
|
changedetection_json_new_schema=os.path.join(self.datastore_path, "changedetection.json")
|
|
if not os.path.isfile(changedetection_json_new_schema):
|
|
import sys
|
|
logger.critical("Migration failed, changedetection.json not found after update ran!")
|
|
sys.exit(1)
|
|
|
|
|
|
logger.critical("Phase 4 complete: Verified changedetection.json exists")
|
|
|
|
# Success! Now reload from new format
|
|
logger.critical("Reloading datastore from new format...")
|
|
# write it to disk, it will be saved without ['watching'] in the JSON db because we find it from disk glob
|
|
self._save_settings()
|
|
logger.success("Datastore reloaded from new format successfully")
|
|
logger.critical("=" * 80)
|
|
logger.critical("MIGRATION COMPLETED SUCCESSFULLY!")
|
|
logger.critical("=" * 80)
|
|
logger.info("")
|
|
logger.info("New format:")
|
|
logger.info(f" - {watch_count} individual watch.json files created")
|
|
logger.info(f" - changedetection.json created (settings only)")
|
|
logger.info("")
|
|
logger.info("Rollback safety:")
|
|
logger.info(" - url-watches.json preserved for rollback")
|
|
logger.info(" - To rollback: downgrade to previous version and restart")
|
|
logger.info(" - No manual file operations needed")
|
|
logger.info("")
|
|
logger.info("Optional cleanup (after testing new version):")
|
|
logger.info(f" - rm {os.path.join(self.datastore_path, 'url-watches.json')}")
|
|
logger.info("")
|
|
|
|
def update_26(self):
|
|
self.migrate_legacy_db_format()
|
|
|
|
# Re-run tag to JSON migration
|
|
def update_29(self):
|
|
|
|
"""
|
|
Migrate tags to individual tag.json files.
|
|
|
|
Tags are currently saved only in changedetection.json (settings).
|
|
This migration ALSO saves them to individual {uuid}/tag.json files,
|
|
similar to how watches are stored (dual storage).
|
|
|
|
Benefits:
|
|
- Allows atomic tag updates without rewriting entire settings
|
|
- Enables independent tag versioning/backup
|
|
- Maintains backwards compatibility (tags stay in settings too)
|
|
"""
|
|
logger.critical("=" * 80)
|
|
logger.critical("Running migration: Individual tag persistence (update_28)")
|
|
logger.critical("Creating individual tag.json files")
|
|
logger.critical("=" * 80)
|
|
|
|
tags = self.data['settings']['application'].get('tags', {})
|
|
tag_count = len(tags)
|
|
|
|
if tag_count == 0:
|
|
logger.info("No tags found, skipping migration")
|
|
return
|
|
|
|
logger.info(f"Migrating {tag_count} tags to individual tag.json files...")
|
|
|
|
saved_count = 0
|
|
failed_count = 0
|
|
|
|
for uuid, tag_data in tags.items():
|
|
if os.path.isfile(os.path.join(self.datastore_path, uuid, "tag.json")):
|
|
logger.debug(f"Tag {uuid} tag.json exists, skipping")
|
|
continue
|
|
try:
|
|
tag_data.commit()
|
|
saved_count += 1
|
|
if saved_count % 10 == 0:
|
|
logger.info(f" Progress: {saved_count}/{tag_count} tags migrated...")
|
|
|
|
except Exception as e:
|
|
logger.error(f"Failed to save tag {uuid} ({tag_data.get('title', 'unknown')}): {e}")
|
|
failed_count += 1
|
|
|
|
if failed_count > 0:
|
|
logger.warning(f"Migration complete: {saved_count} tags saved, {failed_count} tags FAILED")
|
|
else:
|
|
logger.success(f"Migration complete: {saved_count} tags saved to individual tag.json files")
|
|
|
|
# Tags remain in settings for backwards compatibility AND easy access
|
|
# On next load, _load_tags() will read from tag.json files and merge with settings
|
|
logger.info("Tags saved to both settings AND individual tag.json files")
|
|
logger.info("Future tag edits will update both locations (dual storage)")
|
|
logger.critical("=" * 80)
|
|
|
|
# write it to disk, it will be saved without ['tags'] in the JSON db because we find it from disk glob
|
|
# (left this out by accident in previous update, added tags={} in the changedetection.json save_to_disk)
|
|
self._save_settings()
|
|
|
|
def update_30(self):
|
|
"""Migrate restock_settings out of watch.json into restock_diff.json processor config file.
|
|
|
|
Previously, restock_diff processor settings (in_stock_processing, follow_price_changes, etc.)
|
|
were stored directly in the watch dict (watch.json). They now belong in a separate per-watch
|
|
processor config file (restock_diff.json) consistent with the processor_config_* API system.
|
|
|
|
For tags: restock_settings key is renamed to processor_config_restock_diff in the tag dict,
|
|
matching what the API writes when updating a tag.
|
|
|
|
Safe to re-run: skips watches that already have a restock_diff.json, skips tags that already
|
|
have processor_config_restock_diff set.
|
|
"""
|
|
import json
|
|
|
|
# --- Watches ---
|
|
for uuid, watch in self.data['watching'].items():
|
|
if watch.get('processor') != 'restock_diff':
|
|
continue
|
|
restock_settings = watch.get('restock_settings')
|
|
if not restock_settings:
|
|
continue
|
|
|
|
data_dir = watch.data_dir
|
|
if data_dir:
|
|
watch.ensure_data_dir_exists()
|
|
filepath = os.path.join(data_dir, 'restock_diff.json')
|
|
if not os.path.isfile(filepath):
|
|
with open(filepath, 'w', encoding='utf-8') as f:
|
|
json.dump({'restock_diff': restock_settings}, f, indent=2)
|
|
logger.info(f"update_30: migrated restock_settings → {filepath}")
|
|
|
|
del self.data['watching'][uuid]['restock_settings']
|
|
watch.commit()
|
|
|
|
# --- Tags ---
|
|
for tag_uuid, tag in self.data['settings']['application']['tags'].items():
|
|
restock_settings = tag.get('restock_settings')
|
|
if not restock_settings or tag.get('processor_config_restock_diff'):
|
|
continue
|
|
tag['processor_config_restock_diff'] = restock_settings
|
|
del tag['restock_settings']
|
|
tag.commit()
|
|
logger.info(f"update_30: migrated tag {tag_uuid} restock_settings → processor_config_restock_diff")
|
|
|
|
def update_31(self):
|
|
"""Fold any flat application.llm_* key into nested application.llm.<stripped>.
|
|
|
|
Before: a handful of LLM settings (llm_enabled, llm_thinking_budget, …) lived
|
|
directly on settings.application alongside everything else, while the provider
|
|
config (model, api_key, …) was already nested under settings.application.llm.
|
|
Unifies them under one parent so the LLMSettings pydantic model has a single
|
|
home to read/write.
|
|
|
|
Flat key wins on conflict (most-recent form-saved value). Idempotent.
|
|
"""
|
|
application = self.data['settings']['application']
|
|
present = [k for k in list(application) if k.startswith('llm_')]
|
|
if not present:
|
|
return
|
|
|
|
nested = application.get('llm') or {}
|
|
for flat in present:
|
|
nested[flat.removeprefix('llm_')] = application.pop(flat)
|
|
application['llm'] = nested
|
|
logger.info(f"update_31: folded {len(present)} flat llm_* keys into application.llm.* "
|
|
f"({', '.join(present)})")
|
|
|
|
def update_32(self):
|
|
"""Drop max_tokens_per_check and rename max_tokens_cumulative → max_tokens_per_count_period.
|
|
|
|
max_tokens_per_check was never reachable from the UI (form field declared but
|
|
never rendered or saved) and overlapped with the cumulative cap. Removing it.
|
|
|
|
max_tokens_cumulative was misleading — the field was used as a per-watch
|
|
per-period cap, not lifetime. Renamed so the semantic is clear and so a
|
|
future configurable period (day/week/month) doesn't force another rename.
|
|
|
|
Both keys are unreached from real installs (no UI path on prior releases);
|
|
this migration is mostly for branches and devs running pre-release commits.
|
|
"""
|
|
llm = self.data['settings']['application'].get('llm') or {}
|
|
if not llm:
|
|
return
|
|
changed = False
|
|
if 'max_tokens_per_check' in llm:
|
|
del llm['max_tokens_per_check']
|
|
changed = True
|
|
if 'max_tokens_cumulative' in llm:
|
|
llm.setdefault('max_tokens_per_count_period', llm.pop('max_tokens_cumulative'))
|
|
changed = True
|
|
if changed:
|
|
self.data['settings']['application']['llm'] = llm
|
|
logger.info("update_32: cleaned up obsolete max_tokens_per_check / renamed max_tokens_cumulative")
|
|
|
|
def update_33(self):
|
|
"""Restock: consolidate the old price-history fields into a single 'last_price'.
|
|
|
|
Earlier schemas carried 'original_price' (misnamed - it was re-stamped with the current
|
|
price every check, so it actually held the previous check's price) and, on the UI branch,
|
|
'prev_price' (price before the last change, for the watch-list arrow). Both are replaced by
|
|
'last_price' = the price at the previous check, which now drives BOTH the % threshold and
|
|
the up/down arrow (get_price_change_percent), with no history reads at render time.
|
|
|
|
Backfill last_price from the second-to-last history snapshot so the arrow is correct
|
|
immediately; fall back to the old original_price; then drop the obsolete keys. Idempotent.
|
|
|
|
"""
|
|
migrated = 0
|
|
for uuid, watch in self.data['watching'].items():
|
|
if watch.get('processor') != 'restock_diff':
|
|
continue
|
|
restock = watch.get('restock')
|
|
if not isinstance(restock, dict):
|
|
continue
|
|
|
|
# Best-effort backfill of last_price = previous price (second-to-last history snapshot)
|
|
try:
|
|
versions = list(watch.history.keys())
|
|
except Exception:
|
|
versions = []
|
|
|
|
if len(versions) >= 1 and not restock.get('price'):
|
|
snapshot = watch.get_history_snapshot(timestamp=versions[-1])
|
|
restock['price'] = get_price_from_history_str(history_str=snapshot)
|
|
logger.trace(f"UUID {uuid} restock current price set to '{restock['last_price']}'")
|
|
|
|
if len(versions) >= 2:
|
|
snapshot = watch.get_history_snapshot(timestamp=versions[-2])
|
|
if snapshot:
|
|
restock['last_price'] = get_price_from_history_str(history_str=snapshot)
|
|
logger.trace(f"UUID {uuid} restock last_price set to '{restock['last_price']}'")
|
|
|
|
# Fall back to the old preserved value if history gave us nothing
|
|
if not restock.get('last_price') and restock.get('original_price') is not None:
|
|
restock['last_price'] = restock.get('original_price')
|
|
|
|
restock.pop('original_price', None)
|
|
restock.pop('prev_price', None)
|
|
watch.commit()
|
|
|
|
def update_34(self):
|
|
"""Make the global 'Default browser' a concrete, valid selection for the /browsers tab.
|
|
|
|
All browser choice is managed on /browsers now; the default is
|
|
settings.application.fetch_backend (the single source of truth that the per-row radio
|
|
writes and every watch/group set to 'system' resolves to). Older installs may hold a
|
|
blank/missing value, the sentinel 'system', or a browser-config id that has since been
|
|
deleted - any of which would leave the /browsers "Default" radio with nothing selected
|
|
(and a watch on 'system' with no concrete engine). Normalise those to a concrete built-in
|
|
engine, honouring DEFAULT_FETCH_BACKEND (the same env var fresh installs use), else
|
|
'html_requests'.
|
|
|
|
Concrete values already stored - a built-in engine name (e.g. 'html_webdriver') or a
|
|
still-existing saved browser-config id - are left untouched. Idempotent.
|
|
"""
|
|
app = self.data['settings']['application']
|
|
current = app.get('fetch_backend')
|
|
|
|
def _is_valid_default(value):
|
|
if not value or value == 'system':
|
|
return False
|
|
# A saved browser config? (which is what a migrated extra browser is - update_36)
|
|
if self.browser_config_store.get(value):
|
|
return True
|
|
# A built-in engine that actually exists in this build?
|
|
from changedetectionio import content_fetchers
|
|
return hasattr(content_fetchers, value)
|
|
|
|
if _is_valid_default(current):
|
|
return # already a concrete, resolvable default - nothing to do
|
|
|
|
default = os.getenv('DEFAULT_FETCH_BACKEND', 'html_requests') or 'html_requests'
|
|
app['fetch_backend'] = default
|
|
logger.info(
|
|
f"update_34: normalised global Default browser (fetch_backend) from '{current}' to '{default}'"
|
|
)
|
|
|
|
def update_35(self):
|
|
"""Migrate the per-engine request timeout + default User-Agent out of global settings into
|
|
browser configs keyed by the engine name (html_requests / html_webdriver), so all fetch
|
|
behaviour lives on the /browsers tab.
|
|
|
|
Watches/global defaults set to those engine names pick the same-keyed config up
|
|
automatically (BrowserConfigStore.engine_and_config), so nothing needs repointing. Only
|
|
migrates values not already present on an existing (user-edited) config, then drops the old
|
|
settings keys. Idempotent: once they're gone there is nothing left to move.
|
|
"""
|
|
req = self.data['settings']['requests']
|
|
timeout = req.get('timeout')
|
|
default_ua = req.get('default_ua') or {}
|
|
if timeout is None and not default_ua:
|
|
return # already migrated / nothing to move
|
|
|
|
from changedetectionio import content_fetchers
|
|
descriptions = dict(content_fetchers.available_fetchers())
|
|
store = self.browser_config_store
|
|
|
|
def _merge(engine, updates):
|
|
updates = {k: v for k, v in updates.items() if v}
|
|
if not updates:
|
|
return
|
|
existing = store.get(engine)
|
|
bc = dict((existing or {}).get('browser_config') or {})
|
|
for k, v in updates.items():
|
|
bc.setdefault(k, v) # never clobber a value a user already set on the config
|
|
label = (existing or {}).get('label') or str(descriptions.get(engine, engine))
|
|
store.upsert(engine, label=label, base_fetcher=engine, browser_config=bc)
|
|
logger.info(f"update_35: migrated {sorted(updates)} into browser config '{engine}'")
|
|
|
|
_merge('html_requests', {'timeout': timeout, 'user_agent': default_ua.get('html_requests')})
|
|
_merge('html_webdriver', {'user_agent': default_ua.get('html_webdriver')})
|
|
|
|
req.pop('timeout', None)
|
|
req.pop('default_ua', None)
|
|
logger.info("update_35: removed migrated requests.timeout / requests.default_ua from settings")
|
|
|
|
def update_36(self):
|
|
"""Migrate settings.requests.extra_browsers into browser configs on the /browsers tab.
|
|
|
|
Each 'extra browser' was a name + a ws(s):// endpoint, selected by a watch as the magic
|
|
string 'extra_browser_<name>'. That string was resolved to html_webdriver + a custom
|
|
connection URL, which meant the protocol the endpoint was spoken to depended on env vars
|
|
(Playwright/Puppeteer = CDP, Selenium = W3C WebDriver, where a wss:// URL cannot work).
|
|
Each one now becomes an ordinary browser config based on html_external_cdp, which pins
|
|
the protocol to the engine.
|
|
|
|
Keyed by the SAME 'extra_browser_<name>' string the watches already hold, so no watch,
|
|
group override, API value or global default needs rewriting - the legacy selector simply
|
|
becomes a real browser-config id (update_35 set the precedent of non-uuid keys; anything
|
|
created from the UI afterwards is a uuid).
|
|
|
|
Idempotent: gated on the settings key still being there.
|
|
"""
|
|
req = self.data['settings']['requests']
|
|
if 'extra_browsers' not in req:
|
|
return # already migrated / never had any
|
|
|
|
store = self.browser_config_store
|
|
# Same predicate the old datastore.extra_browsers property used - the settings form keeps
|
|
# five blank FieldList slots, and a row without both halves was never selectable.
|
|
rows = [r for r in (req.get('extra_browsers') or [])
|
|
if r.get('browser_name') and r.get('browser_connection_url')]
|
|
|
|
existing_labels = {(e.get('label') or '').strip().lower()
|
|
for e in store.all().values()}
|
|
seen = set()
|
|
migrated = 0
|
|
for row in rows:
|
|
name = row['browser_name'].strip()
|
|
config_id = f"extra_browser_{name}"
|
|
if config_id in seen:
|
|
# Two rows could share a name; the old resolver just took the first match.
|
|
logger.warning(f"update_36: ignoring duplicate extra browser '{name}'")
|
|
continue
|
|
seen.add(config_id)
|
|
if store.get(config_id):
|
|
continue # already migrated (a re-run with the settings key still present)
|
|
|
|
# A label that collides with an existing browser would make this entry unsaveable
|
|
# later, because the /browsers form rejects duplicate names.
|
|
label = name
|
|
suffix = 2
|
|
while label.strip().lower() in existing_labels:
|
|
label = f"{name} ({suffix})"
|
|
suffix += 1
|
|
existing_labels.add(label.strip().lower())
|
|
|
|
try:
|
|
store.upsert(config_id,
|
|
label=label,
|
|
base_fetcher='html_external_cdp',
|
|
browser_config={'connection_url': row['browser_connection_url'].strip()})
|
|
except ValidationError as e:
|
|
# An endpoint that fails FetcherConfig's rules (the old settings could be
|
|
# hand-edited or restored from anywhere) must not take the whole update chain -
|
|
# and with it startup - down. Say so loudly and carry on; the operator can add
|
|
# the browser on the Browsers page.
|
|
logger.error(f"update_36: could not migrate extra browser '{name}' "
|
|
f"(endpoint rejected: {e}) - add it on the Browsers page instead")
|
|
continue
|
|
migrated += 1
|
|
logger.info(f"update_36: migrated extra browser '{name}' to browser config "
|
|
f"'{config_id}' (label '{label}')")
|
|
|
|
req.pop('extra_browsers', None)
|
|
logger.info(f"update_36: migrated {migrated} extra browser(s) and removed "
|
|
f"settings.requests.extra_browsers")
|
|
|
|
|