diff --git a/changedetectionio/model/Watch.py b/changedetectionio/model/Watch.py index 37b04c496..79e0f8737 100644 --- a/changedetectionio/model/Watch.py +++ b/changedetectionio/model/Watch.py @@ -10,9 +10,13 @@ from pathlib import Path from loguru import logger from .. import jinja2_custom as safe_jinja -from ..diff import ADDED_PLACEMARKER_OPEN from ..html_tools import TRANSLATE_WHITESPACE_TABLE +FAVICON_RESAVE_THRESHOLD_SECONDS=86400 +BROTLI_COMPRESS_SIZE_THRESHOLD = int(os.getenv('SNAPSHOT_BROTLI_COMPRESSION_THRESHOLD', 1024)) + +minimum_seconds_recheck_time = int(os.getenv('MINIMUM_SECONDS_RECHECK_TIME', 3)) +mtable = {'seconds': 1, 'minutes': 60, 'hours': 3600, 'days': 86400, 'weeks': 86400 * 7} def _brotli_compress_worker(conn, filepath, mode=None): """ @@ -29,6 +33,7 @@ def _brotli_compress_worker(conn, filepath, mode=None): try: # Receive data from parent process via pipe (avoids pickle overhead) contents = conn.recv() + logger.debug(f"Starting brotli compression of {len(contents)} bytes.") if mode is not None: compressed_data = brotli.compress(contents, mode=mode) @@ -40,9 +45,10 @@ def _brotli_compress_worker(conn, filepath, mode=None): # Send success status back conn.send(True) + logger.debug(f"Finished brotli compression - From {len(contents)} to {len(compressed_data)} bytes.") # No need for explicit cleanup - process exit frees all memory except Exception as e: - logger.error(f"Brotli compression worker failed: {e}") + logger.critical(f"Brotli compression worker failed: {e}") conn.send(False) finally: conn.close() @@ -66,7 +72,6 @@ def _brotli_subprocess_save(contents, filepath, mode=None, timeout=30, fallback_ Raises: Exception: if compression fails and fallback_uncompressed is False """ - import brotli import multiprocessing import sys @@ -144,11 +149,6 @@ def _brotli_subprocess_save(contents, filepath, mode=None, timeout=30, fallback_ else: raise Exception(f"Brotli compression subprocess failed for {filepath}") -FAVICON_RESAVE_THRESHOLD_SECONDS=86400 - - -minimum_seconds_recheck_time = int(os.getenv('MINIMUM_SECONDS_RECHECK_TIME', 3)) -mtable = {'seconds': 1, 'minutes': 60, 'hours': 3600, 'days': 86400, 'weeks': 86400 * 7} class model(watch_base): __newest_history_key = None @@ -492,7 +492,6 @@ class model(watch_base): self.ensure_data_dir_exists() - threshold = int(os.getenv('SNAPSHOT_BROTLI_COMPRESSION_THRESHOLD', 1024)) skip_brotli = strtobool(os.getenv('DISABLE_BROTLI_TEXT_SNAPSHOT', 'False')) # Binary data - detect file type and save without compression @@ -516,7 +515,7 @@ class model(watch_base): # Text data - use brotli compression if enabled and above threshold else: - if not skip_brotli and len(contents) > threshold: + if not skip_brotli and len(contents) > BROTLI_COMPRESS_SIZE_THRESHOLD: # Compressed text import brotli snapshot_fname = f"{snapshot_id}.txt.br" diff --git a/changedetectionio/run_basic_tests.sh b/changedetectionio/run_basic_tests.sh index 26e33c231..3a6f801f6 100755 --- a/changedetectionio/run_basic_tests.sh +++ b/changedetectionio/run_basic_tests.sh @@ -105,4 +105,5 @@ echo "Hello world" > /tmp/test-file.txt ALLOW_FILE_URI=yes pytest -vv -s tests/test_security.py - +# Run it again so that brotli kicks in +TEST_WITH_BROTLI=1 SNAPSHOT_BROTLI_COMPRESSION_THRESHOLD=100 pytest tests/test_history_consistency.py -vv -l -s diff --git a/changedetectionio/tests/test_history_consistency.py b/changedetectionio/tests/test_history_consistency.py index ff69797f1..75ec2b1c7 100644 --- a/changedetectionio/tests/test_history_consistency.py +++ b/changedetectionio/tests/test_history_consistency.py @@ -4,23 +4,33 @@ import time import os import json from flask import url_for + +from build.lib.changedetectionio.strtobool import strtobool from .util import wait_for_all_checks, delete_all_watches from urllib.parse import urlparse, parse_qs + def test_consistent_history(client, live_server, measure_memory_usage, datastore_path): - # live_server_setup(live_server) # Setup on conftest per function + # live_server_setup(live_server) # Setup on conftest per function workers = int(os.getenv("FETCH_WORKERS", 10)) - r = range(1, 10+workers) + r = range(1, 10 + workers) + uuids = set() + import brotli for one in r: - test_url = url_for('test_endpoint', content_type="text/html", content=str(one), _external=True) - res = client.post( - url_for("imports.import_page"), - data={"urls": test_url}, - follow_redirects=True - ) + if strtobool(os.getenv("TEST_WITH_BROTLI")): + # A very long string that WILL trigger Brotli compression of the snapshot + # BROTLI_COMPRESS_SIZE_THRESHOLD should be set to say 200 + from ..model.Watch import BROTLI_COMPRESS_SIZE_THRESHOLD + content = str(one) + " " + str(one) * (BROTLI_COMPRESS_SIZE_THRESHOLD + 10) + else: + # Just enough to test datastore + content = str(one) - assert b"1 Imported" in res.data + test_url = url_for('test_endpoint', content_type="text/html", content=content, _external=True) + uuids.add(client.application.config.get('DATASTORE').add_watch(url=test_url, extras={'title': str(one)})) + + client.get(url_for("ui.form_watch_checknow"), follow_redirects=True) wait_for_all_checks(client) @@ -45,13 +55,17 @@ def test_consistent_history(client, live_server, measure_memory_usage, datastore # assert the right amount of watches was found in the JSON assert len(json_obj['watching']) == len(r), "Correct number of watches was found in the JSON" - i=0 + + i = 0 # each one should have a history.txt containing just one line for w in json_obj['watching'].keys(): - i+=1 + i += 1 history_txt_index_file = os.path.join(live_server.app.config['DATASTORE'].datastore_path, w, 'history.txt') assert os.path.isfile(history_txt_index_file), f"History.txt should exist where I expect it at {history_txt_index_file}" + # Should be no errors (could be from brotli etc) + assert not live_server.app.config['DATASTORE'].data['watching'][w].get('last_error') + # Same like in model.Watch with open(history_txt_index_file, "r") as f: tmp_history = dict(i.strip().split(',', 2) for i in f.readlines()) @@ -63,15 +77,21 @@ def test_consistent_history(client, live_server, measure_memory_usage, datastore # Find the snapshot one for fname in files_in_watch_dir: if fname != 'history.txt' and 'html' not in fname: + if strtobool(os.getenv("TEST_WITH_BROTLI")): + assert fname.endswith('.br'), "Forced TEST_WITH_BROTLI then it should be a .br filename" + + full_snapshot_history_path = os.path.join(live_server.app.config['DATASTORE'].datastore_path, w, fname) # contents should match what we requested as content returned from the test url - with open(os.path.join(live_server.app.config['DATASTORE'].datastore_path, w, fname), 'r') as snapshot_f: - contents = snapshot_f.read() - watch_url = json_obj['watching'][w]['url'] - u = urlparse(watch_url) - q = parse_qs(u[4]) - assert q['content'][0] == contents.strip(), f"Snapshot file {fname} should contain {q['content'][0]}" - + if fname.endswith('.br'): + with open(full_snapshot_history_path, 'rb') as f: + contents = brotli.decompress(f.read()).decode('utf-8') + else: + with open(full_snapshot_history_path, 'r') as snapshot_f: + contents = snapshot_f.read() + watch_title = json_obj['watching'][w]['title'] + assert json_obj['watching'][w]['title'], "Watch should have a title set" + assert contents.startswith(watch_title + " "), f"Snapshot file {fname} should start with '{watch_title} '" assert len(files_in_watch_dir) == 3, "Should be just three files in the dir, html.br snapshot, history.txt and the extracted text snapshot" diff --git a/changedetectionio/tests/test_xpath_selector_unit.py b/changedetectionio/tests/test_xpath_selector_unit.py index a09a65393..9a7701463 100644 --- a/changedetectionio/tests/test_xpath_selector_unit.py +++ b/changedetectionio/tests/test_xpath_selector_unit.py @@ -1,8 +1,10 @@ import sys import os import pytest + +from changedetectionio import html_tools + sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -import html_tools # test generation guide. # 1. Do not include encoding in the xml declaration if the test object is a str type.