mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-09-28 08:16:03 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
62f425bdb5 | ||
|
|
16b19a4ab7 | ||
|
|
48723a09dd | ||
|
|
718295f30b | ||
|
|
7cc62f8321 | ||
|
|
cd66f9ed54 | ||
|
|
8d938b5966 | ||
|
|
0d6b254fdb |
@@ -205,7 +205,7 @@ jobs:
|
||||
# keeps the run short. Each file is wrapped in its own log group and the failing one is
|
||||
# named explicitly, because everything after it is SKIPPED rather than run, and that is
|
||||
# otherwise easy to misread as "the whole browser suite broke".
|
||||
for t in tests/fetchers/test_content.py tests/test_errorhandling.py tests/visualselector/test_fetch_data.py tests/fetchers/test_custom_js_before_content.py; do
|
||||
for t in tests/fetchers/test_content.py tests/test_errorhandling.py tests/visualselector/test_fetch_data.py tests/fetchers/test_custom_js_before_content.py tests/fetchers/test_renavigation.py; do
|
||||
echo "::group::pytest $t"
|
||||
if ! docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio \
|
||||
bash -c "cd changedetectionio;pytest -vv --capture=tee-sys --showlocals --tb=long --live-server-host=0.0.0.0 --live-server-port=5004 $t"; then
|
||||
@@ -253,7 +253,7 @@ jobs:
|
||||
- name: Pyppeteer - Specific tests in built container
|
||||
run: |
|
||||
# Fail fast, but name the file that failed - see the note in the playwright job above
|
||||
for t in tests/fetchers/test_content.py tests/test_errorhandling.py tests/visualselector/test_fetch_data.py tests/fetchers/test_custom_js_before_content.py; do
|
||||
for t in tests/fetchers/test_content.py tests/test_errorhandling.py tests/visualselector/test_fetch_data.py tests/fetchers/test_custom_js_before_content.py tests/fetchers/test_renavigation.py; do
|
||||
echo "::group::pytest $t"
|
||||
if ! docker run --rm -e "FLASK_SERVER_NAME=cdio" -e "FAST_PUPPETEER_CHROME_FETCHER=True" -e "PLAYWRIGHT_DRIVER_URL=ws://sockpuppetbrowser:3000" --network changedet-network --hostname=cdio test-changedetectionio \
|
||||
bash -c "cd changedetectionio;pytest --live-server-host=0.0.0.0 --live-server-port=5004 $t"; then
|
||||
|
||||
+25
@@ -101,6 +101,31 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libxrender-dev \
|
||||
&& apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Actually generate the locales. Installing the `locales` package above only
|
||||
# ships /etc/locale.gen - it does not build any locale, so the image had just
|
||||
# C, C.utf8 and POSIX. That made the `ENV LC_ALL=en_US.UTF-8` below unsatisfiable:
|
||||
# locale.setlocale() in flask_app.py failed, fell back to C, and the
|
||||
# format_number_locale / format_int_locale Jinja filters silently lost their
|
||||
# thousands separators - 1234567.89 rendered as "1234567.89" rather than
|
||||
# "1,234,567.89" in the restock/price overview, which is the very thing the
|
||||
# `locales` package was added for.
|
||||
#
|
||||
# More than en_US is generated so that operators can override LC_ALL / LANG and
|
||||
# get formatting for their own region (de_DE gives 1.234.567,89, fr_FR gives
|
||||
# 1 234 567,89). Costs ~21MB and ~16s of build time.
|
||||
#
|
||||
# This list mirrors the UI translations in changedetectionio/translations - one
|
||||
# glibc locale per language we ship a translation for, so any language a user
|
||||
# can pick in the UI also has a working locale. Keep the two in sync when adding
|
||||
# a translation. The territory for each bare language code comes from CLDR's
|
||||
# likely-subtags (cs -> cs_CZ, ja -> ja_JP, ko -> ko_KR, uk -> uk_UA, zh ->
|
||||
# zh_CN, zh_Hant_TW -> zh_TW), NOT from uppercasing the language code.
|
||||
RUN for l in cs_CZ de_DE en_GB en_US es_ES fr_FR id_ID it_IT ja_JP ko_KR \
|
||||
pl_PL pt_BR ru_RU tr_TR uk_UA zh_CN zh_TW; do \
|
||||
sed -i "s/^# *${l}.UTF-8 UTF-8/${l}.UTF-8 UTF-8/" /etc/locale.gen; \
|
||||
done \
|
||||
&& locale-gen
|
||||
|
||||
|
||||
# https://stackoverflow.com/questions/58701233/docker-logs-erroneously-appears-empty-until-container-stops
|
||||
ENV PYTHONUNBUFFERED=1
|
||||
|
||||
@@ -296,6 +296,20 @@ class fetcher(Fetcher):
|
||||
|
||||
self.page = await context.new_page()
|
||||
|
||||
# Track the LATEST main-frame document response for the whole fetch, not just the one
|
||||
# goto() returns. This app compares the text of the page the browser ends up on, and a
|
||||
# site that gates with an interstitial (503/429 + meta-refresh) navigates to the real
|
||||
# page *during* the extra_wait below - judging the fetch on the first response fails a
|
||||
# watch whose content is present and fine. Same for plain client-side redirects.
|
||||
# Shared with action_goto_url() so only one 'response' listener exists on the page.
|
||||
from changedetectionio.browser_steps.browser_steps import track_latest_navigation_response
|
||||
# Must be an identity check - the tracker hands back the same (initially empty, so
|
||||
# falsy) dict the listener writes into, and `or {}` would quietly swap in a different
|
||||
# one that never gets updated.
|
||||
latest_navigation_response = track_latest_navigation_response(self.page)
|
||||
if latest_navigation_response is None:
|
||||
latest_navigation_response = {}
|
||||
|
||||
# Listen for all console events and handle errors
|
||||
self.page.on("console", lambda msg: logger.debug(f"Playwright console: Watch URL: {url} {msg.type}: {msg.text} {msg.args}"))
|
||||
|
||||
@@ -336,6 +350,25 @@ class fetcher(Fetcher):
|
||||
extra_wait = int(os.getenv("WEBDRIVER_DELAY_BEFORE_CONTENT_READY", 5)) + self.render_extract_delay
|
||||
await self.page.wait_for_timeout(extra_wait * 1000)
|
||||
|
||||
# A meta-refresh or client-side redirect usually lands during that wait, so judge the
|
||||
# fetch on the document we are actually about to extract rather than the first one.
|
||||
latest = latest_navigation_response.get('response')
|
||||
if latest is not None and latest is not response:
|
||||
logger.debug(f"Page navigated again while waiting, judging the fetch on {latest.url} "
|
||||
f"(status {latest.status}) instead of the first response for {url}")
|
||||
response = latest
|
||||
try:
|
||||
self.headers = await response.all_headers()
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not refresh headers from the final document: {e}")
|
||||
|
||||
# Don't extract while a navigation is mid-flight, that is what produces
|
||||
# "Execution context was destroyed, most likely because of a navigation"
|
||||
try:
|
||||
await self.page.wait_for_load_state('load', timeout=extra_wait * 1000)
|
||||
except Exception as e:
|
||||
logger.debug(f"Page did not reach a settled load state, continuing anyway: {e}")
|
||||
|
||||
try:
|
||||
self.status_code = response.status
|
||||
except Exception as e:
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
import asyncio
|
||||
import gc
|
||||
import json
|
||||
import os
|
||||
import websockets.exceptions
|
||||
@@ -10,6 +9,7 @@ from loguru import logger
|
||||
from changedetectionio.content_fetchers import SCREENSHOT_MAX_HEIGHT_DEFAULT, visualselector_xpath_selectors, \
|
||||
SCREENSHOT_SIZE_STITCH_THRESHOLD, SCREENSHOT_DEFAULT_QUALITY, XPATH_ELEMENT_JS, INSTOCK_DATA_JS, \
|
||||
SCREENSHOT_MAX_TOTAL_HEIGHT, FAVICON_FETCHER_JS
|
||||
from changedetectionio import gc_debounce
|
||||
from changedetectionio.content_fetchers.base import Fetcher, get_playwright_bypass_csp, manage_user_agent
|
||||
from changedetectionio.content_fetchers.exceptions import PageUnloadable, Non200ErrorCodeReceived, EmptyReply, BrowserFetchTimedOut, \
|
||||
BrowserConnectError
|
||||
@@ -236,6 +236,7 @@ class fetcher(Fetcher):
|
||||
|
||||
async def quit(self, watch=None):
|
||||
watch_uuid = watch.get('uuid') if watch else 'unknown'
|
||||
closed_something = bool(getattr(self, 'page', None) or getattr(self, 'browser', None))
|
||||
|
||||
# Close page
|
||||
try:
|
||||
@@ -263,8 +264,16 @@ class fetcher(Fetcher):
|
||||
|
||||
logger.info(f"[{watch_uuid}] Cleanup puppeteer complete")
|
||||
|
||||
# Force garbage collection to release resources
|
||||
gc.collect()
|
||||
# Only collect if this call actually closed something.
|
||||
#
|
||||
# quit() runs twice per check - from run()'s finally, then again from the worker's
|
||||
# safety net - and it sets self.page/self.browser to None in its own finally
|
||||
# blocks. The second call therefore closes nothing, creates no garbage and breaks
|
||||
# no cycles, but still paid for a full stop-the-world collection: measured at 88
|
||||
# calls across 51 checks, roughly half of them reclaiming nothing. The pyppeteer
|
||||
# page/connection/session graph is genuinely cyclic, so the first call still runs.
|
||||
if closed_something:
|
||||
gc_debounce.collect('puppeteer.quit')
|
||||
|
||||
async def fetch_page(self,
|
||||
current_include_filters,
|
||||
@@ -399,43 +408,69 @@ class fetcher(Fetcher):
|
||||
# Enable Network domain to detect when first bytes arrive
|
||||
await self.page._client.send('Network.enable')
|
||||
|
||||
# Now set up the frame navigation handlers
|
||||
async def handle_frame_navigation(event=None):
|
||||
# Wait n seconds after the frameStartedLoading, not from any frameStartedLoading/frameStartedNavigating
|
||||
logger.debug(f"Frame navigated: {event}")
|
||||
w = extra_wait - 2 if extra_wait > 4 else 2
|
||||
logger.debug(f"Waiting {w} seconds before calling Page.stopLoading...")
|
||||
await asyncio.sleep(w)
|
||||
# Navigate (bounded), then wait the configured "wait n seconds before extracting text"
|
||||
# delay, then stop whatever is still loading, then extract. The delay is measured from
|
||||
# when navigation finished, not from when it started, because the point of it is to let
|
||||
# JS-rendered content appear *after* load - anchoring it to the start would quietly give a
|
||||
# slow-loading page almost no settle time.
|
||||
#
|
||||
# Only that delay is a user-facing setting. The navigation bound above is a safety net with
|
||||
# a sane default, not a tuning knob, so there is still one number for users to think about.
|
||||
#
|
||||
# There is no way to know a page is "finished" - plenty of sites navigate as part of their
|
||||
# normal design, and some sit forever on a subresource that never answers. So the delay
|
||||
# restarts if the MAIN frame replaces its document (a redirect or interstitial gets the
|
||||
# same settle time the first document got), iframes do not restart it, and it is capped so
|
||||
# a page that re-navigates in a loop cannot extend it indefinitely.
|
||||
max_content_ready_resets = int(os.getenv("BROWSER_CONTENT_READY_MAX_RESETS", 2))
|
||||
|
||||
# Check if page still exists (might have been closed due to error during sleep)
|
||||
if not self.page or not hasattr(self.page, '_client'):
|
||||
logger.debug("Page already closed, skipping stopLoading")
|
||||
return
|
||||
async def wait_for_content_ready_then_stop_loading():
|
||||
main_frame_id = self.page.mainFrame._id
|
||||
renavigated = asyncio.Event()
|
||||
|
||||
logger.debug("Issuing stopLoading command...")
|
||||
await self.page._client.send('Page.stopLoading')
|
||||
logger.debug("stopLoading command sent!")
|
||||
def _on_main_frame_navigation(event):
|
||||
if event.get('frameId') == main_frame_id:
|
||||
renavigated.set()
|
||||
|
||||
async def setup_frame_handlers_on_first_response(event):
|
||||
# Only trigger for the main document response
|
||||
if event.get('type') == 'Document':
|
||||
logger.debug("First response received, setting up frame handlers for forced page stop load.")
|
||||
self.page._client.on('Page.frameStartedNavigating', lambda e: asyncio.create_task(handle_frame_navigation(e)))
|
||||
self.page._client.on('Page.frameStartedLoading', lambda e: asyncio.create_task(handle_frame_navigation(e)))
|
||||
self.page._client.on('Page.frameStoppedLoading', lambda e: logger.debug(f"Frame stopped loading: {e}"))
|
||||
logger.debug("First response received, setting up frame handlers for forced page stop load DONE SETUP")
|
||||
# De-register this listener - we only need it once
|
||||
self.page._client.remove_listener('Network.responseReceived', setup_frame_handlers_on_first_response)
|
||||
self.page._client.on('Page.frameStartedLoading', _on_main_frame_navigation)
|
||||
self.page._client.on('Page.frameStoppedLoading', lambda e: logger.debug(f"Frame stopped loading: {e}"))
|
||||
try:
|
||||
resets = 0
|
||||
while True:
|
||||
renavigated.clear()
|
||||
try:
|
||||
await asyncio.wait_for(renavigated.wait(), timeout=extra_wait)
|
||||
# Main frame started a new document
|
||||
resets += 1
|
||||
if resets > max_content_ready_resets:
|
||||
logger.debug(f"Main frame keeps re-navigating, not restarting the content-ready wait again")
|
||||
break
|
||||
logger.debug(f"Main frame started a new document, restarting the {extra_wait}s "
|
||||
f"content-ready wait ({resets}/{max_content_ready_resets})")
|
||||
except asyncio.TimeoutError:
|
||||
# Quiet for the whole delay - the page is as ready as it is going to get
|
||||
break
|
||||
finally:
|
||||
self.page._client.remove_listener('Page.frameStartedLoading', _on_main_frame_navigation)
|
||||
|
||||
# Listen for first response to trigger frame handler setup
|
||||
self.page._client.on('Network.responseReceived', setup_frame_handlers_on_first_response)
|
||||
# Stop whatever is still in flight so the DOM and screenshot come from what rendered,
|
||||
# rather than waiting on a subresource that may never answer
|
||||
try:
|
||||
logger.debug(f"Content-ready wait of {extra_wait}s elapsed, issuing Page.stopLoading before extracting")
|
||||
await self.page._client.send('Page.stopLoading')
|
||||
logger.debug("stopLoading command sent!")
|
||||
except Exception as e:
|
||||
logger.debug(f"Page.stopLoading skipped, page is most likely already gone: {e}")
|
||||
|
||||
# Chrome 153+ refuses to commit a navigation when an error status arrives with a
|
||||
# zero-length body, so goto() raises net::ERR_HTTP_RESPONSE_CODE_FAILURE instead of handing
|
||||
# back the response. The response was received fine, we just never get it as a return value,
|
||||
# so keep the main-frame response from the 'response' event and use that instead - the
|
||||
# status check below then reports a real "Error - 404" instead of a raw net:: string.
|
||||
# Kept as the latest matching response so a redirect chain reports its final hop.
|
||||
# Track the LATEST main-frame document response for the whole fetch, not just the one that
|
||||
# goto() happens to return. This app compares the text of the page the browser ends up on,
|
||||
# and plenty of sites navigate again after the first response:
|
||||
# - an interstitial answering 503/429 with a meta-refresh into the real 200 page, where
|
||||
# judging the first response fails a watch whose content is sitting right there
|
||||
# - a plain client-side redirect to another host (slated.com -> get.slated.com)
|
||||
# It also covers Chrome 153+, which refuses to commit a navigation when an error status
|
||||
# arrives with a zero-length body: goto() raises net::ERR_HTTP_RESPONSE_CODE_FAILURE rather
|
||||
# than returning the response, but the response itself still arrives on this event.
|
||||
navigation_response = {}
|
||||
|
||||
def _keep_navigation_response(response):
|
||||
@@ -445,25 +480,62 @@ class fetcher(Fetcher):
|
||||
|
||||
self.page.on('response', _keep_navigation_response)
|
||||
|
||||
# pyppeteer's navigation watcher is bound to the loaderId of the navigation it started. If
|
||||
# the page replaces that document (redirect/interstitial) the 'load' it waits for never
|
||||
# arrives for that loaderId, so goto() never returns - and with timeout=0 it would block
|
||||
# until the hard PUPPETEER_MAX_PROCESSING_TIMEOUT_SECONDS kill, burning a worker slot for
|
||||
# minutes on a page that is fully loaded. Bound it, then fall back to the document we can
|
||||
# see. Verified against slated.com and getastra.com, which hang indefinitely otherwise.
|
||||
nav_timeout = int(os.getenv("BROWSER_NAVIGATION_TIMEOUT_SECONDS", 30))
|
||||
|
||||
response = None
|
||||
attempt=0
|
||||
try:
|
||||
while not response:
|
||||
logger.debug(f"Attempting page fetch {url} attempt {attempt}")
|
||||
asyncio.create_task(handle_frame_navigation())
|
||||
try:
|
||||
response = await self.page.goto(url, timeout=0)
|
||||
except Exception as e:
|
||||
if 'ERR_HTTP_RESPONSE_CODE_FAILURE' not in str(e) or not navigation_response:
|
||||
raise
|
||||
response = navigation_response['response']
|
||||
logger.debug(f"Navigation was aborted by the browser (empty body on an error status), "
|
||||
f"recovered status {response.status} from the response event")
|
||||
await asyncio.sleep(1 + extra_wait)
|
||||
# Check if page still exists before sending command
|
||||
if self.page and hasattr(self.page, '_client'):
|
||||
await self.page._client.send('Page.stopLoading')
|
||||
# Race goto() against the main frame actually firing 'load'. In the re-navigation
|
||||
# case goto() can never resolve, but the replacement document does fire 'load' -
|
||||
# usually within a few seconds - so this returns then instead of sitting out the
|
||||
# whole nav_timeout. Whichever arrives first means "the document is loaded".
|
||||
main_frame_loaded = asyncio.Event()
|
||||
main_frame_id = self.page.mainFrame._id
|
||||
|
||||
def _on_lifecycle(event):
|
||||
if event.get('name') == 'load' and event.get('frameId') == main_frame_id:
|
||||
main_frame_loaded.set()
|
||||
|
||||
self.page._client.on('Page.lifecycleEvent', _on_lifecycle)
|
||||
goto_task = asyncio.ensure_future(self.page.goto(url, timeout=0))
|
||||
load_task = asyncio.ensure_future(main_frame_loaded.wait())
|
||||
try:
|
||||
done, _pending = await asyncio.wait({goto_task, load_task},
|
||||
timeout=nav_timeout,
|
||||
return_when=asyncio.FIRST_COMPLETED)
|
||||
|
||||
if goto_task in done:
|
||||
try:
|
||||
response = goto_task.result()
|
||||
except Exception as e:
|
||||
if 'ERR_HTTP_RESPONSE_CODE_FAILURE' not in str(e) or not navigation_response:
|
||||
raise
|
||||
response = navigation_response['response']
|
||||
logger.debug(f"Navigation was aborted by the browser (empty body on an error status), "
|
||||
f"recovered status {response.status} from the response event")
|
||||
else:
|
||||
# Either the replacement document loaded, or we ran out of patience
|
||||
response = navigation_response.get('response')
|
||||
if not response:
|
||||
raise BrowserFetchTimedOut(msg=f"Browser did not finish navigating to {url} within "
|
||||
f"{nav_timeout}s and no main-frame response was seen.")
|
||||
why = ("the page replaced the document it started on" if load_task in done
|
||||
else f"navigation did not settle within {nav_timeout}s")
|
||||
logger.warning(f"Continuing with the document actually loaded ({why}) - "
|
||||
f"status {response.status} for {response.url}")
|
||||
finally:
|
||||
self.page._client.remove_listener('Page.lifecycleEvent', _on_lifecycle)
|
||||
for t in (goto_task, load_task):
|
||||
if not t.done():
|
||||
t.cancel()
|
||||
if response:
|
||||
break
|
||||
if not response:
|
||||
@@ -472,6 +544,19 @@ class fetcher(Fetcher):
|
||||
logger.warning(f"Content Fetcher > Response object was none (as in, the response from the browser was empty, not just the content) exiting attempt {attempt}")
|
||||
raise EmptyReply(url=url, status_code=None)
|
||||
attempt+=1
|
||||
|
||||
# Navigation is done; now honour "wait n seconds before extracting text" and then
|
||||
# force-stop whatever is still loading, so extraction always gets what rendered.
|
||||
# Awaited inline rather than fired off as a task, so nothing can outlive the fetch.
|
||||
await wait_for_content_ready_then_stop_loading()
|
||||
|
||||
# That wait is where a meta-refresh interstitial typically swaps in the real page, so
|
||||
# re-check which document we are actually on before judging the status code.
|
||||
latest = navigation_response.get('response')
|
||||
if latest is not None and latest is not response:
|
||||
logger.debug(f"Page navigated again while waiting, judging the fetch on {latest.url} "
|
||||
f"(status {latest.status}) instead of {response.url} (status {response.status})")
|
||||
response = latest
|
||||
finally:
|
||||
self.page.remove_listener('response', _keep_navigation_response)
|
||||
|
||||
@@ -534,8 +619,7 @@ class fetcher(Fetcher):
|
||||
self.screenshot = await capture_full_page(page=self.page, screenshot_format=self.screenshot_format, watch_uuid=watch_uuid, lock_viewport_elements=self.lock_viewport_elements)
|
||||
|
||||
# Force garbage collection - pyppeteer base64 decode creates temporary buffers
|
||||
import gc
|
||||
gc.collect()
|
||||
gc_debounce.collect('puppeteer.after_screenshot')
|
||||
self.xpath_data = await self.page.evaluate(XPATH_ELEMENT_JS, {
|
||||
"visualselector_xpath_selectors": visualselector_xpath_selectors,
|
||||
"max_height": MAX_TOTAL_HEIGHT
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
"""
|
||||
Debounced explicit garbage collection for the watch-check hot path.
|
||||
|
||||
Several places call gc.collect() after a check to keep C-level memory (pyppeteer buffers,
|
||||
libxml2 documents, PIL, brotli) from accumulating. Individually each is reasonable. Run
|
||||
concurrently by many fetch workers they become a storm: with FETCH_WORKERS=50 and roughly
|
||||
five call sites per check, the process spends most of its time stopped in the collector,
|
||||
because every gc.collect() is a full stop-the-world pass that walks the entire heap while
|
||||
holding the GIL.
|
||||
|
||||
Measured on 153 real puppeteer checks of a live site, FETCH_WORKERS=50, an `//div` filter
|
||||
so the lxml document tree is realistic:
|
||||
|
||||
collects objects freed gc time checks/sec CPU/check RSS plateau
|
||||
one per call site 790 2,594,098 91.3s 0.766 1.373s 279.6MB
|
||||
debounced to 1s 48 2,219,010 7.3s 1.433 0.731s 275.3MB
|
||||
none at all 0 0 0.0s 1.503 0.685s 287.7MB
|
||||
|
||||
Debouncing keeps 86% of the reclamation for 6% of the collections, and resident memory
|
||||
ends up LOWER than collecting every time. It works because the collector is process-wide:
|
||||
any worker's collection breaks every other worker's cycles too, so with many workers the
|
||||
calls are overwhelmingly redundant duplicates rather than independently necessary.
|
||||
|
||||
Removing them entirely was also measured. It is slightly faster still, but it was the only
|
||||
configuration whose RSS had not plateaued by the end of the run, so it is not offered.
|
||||
|
||||
Collecting a younger generation was measured and rejected: gen 0 freed 1,136 objects
|
||||
against the full pass's 2,594,098, because objects that survive a 10-30 second fetch have
|
||||
already been promoted out of gen 0. It is cheap because it does almost nothing.
|
||||
|
||||
Environment:
|
||||
EXPLICIT_GC_MIN_INTERVAL seconds between explicit collections, process-wide.
|
||||
Default 1.0. Set 0 to collect at every call site as before.
|
||||
EXPLICIT_GC_COLLECT set false to disable explicit collection entirely. A
|
||||
measurement switch for attributing a slowdown, not a
|
||||
recommended setting - expect resident memory to drift.
|
||||
"""
|
||||
|
||||
import gc
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
|
||||
from changedetectionio.strtobool import strtobool
|
||||
|
||||
ENABLED = strtobool(os.getenv('EXPLICIT_GC_COLLECT', 'true'))
|
||||
MIN_INTERVAL = float(os.getenv('EXPLICIT_GC_MIN_INTERVAL', '1.0') or 0)
|
||||
|
||||
_last_collect = 0.0
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def collect(where=None):
|
||||
"""Explicit collection for the per-check hot path, rate-limited process-wide.
|
||||
|
||||
`where` is a short label for the call site, kept so callers read clearly and so a
|
||||
future caller can log it. Returns the number of objects collected, or 0 when the call
|
||||
was debounced or disabled - matching gc.collect()'s return, so this is a drop-in
|
||||
replacement for it.
|
||||
"""
|
||||
global _last_collect
|
||||
|
||||
if not ENABLED:
|
||||
return 0
|
||||
|
||||
if MIN_INTERVAL > 0:
|
||||
now = time.monotonic()
|
||||
with _lock:
|
||||
if now - _last_collect < MIN_INTERVAL:
|
||||
return 0
|
||||
_last_collect = now
|
||||
|
||||
return gc.collect()
|
||||
@@ -30,6 +30,7 @@ from changedetectionio.validate_url import is_safe_valid_url
|
||||
|
||||
from changedetectionio.strtobool import strtobool
|
||||
from changedetectionio.jinja2_custom import render as jinja_render
|
||||
from changedetectionio import gc_debounce
|
||||
from . import watch_base
|
||||
from .persistence import EntityPersistenceMixin
|
||||
import os
|
||||
@@ -107,9 +108,21 @@ def _brotli_save(contents, filepath, mode=None, fallback_uncompressed=False):
|
||||
|
||||
logger.debug(f"Finished brotli compression - From {original_size} to {total_compressed_size} bytes.")
|
||||
|
||||
# Cleanup: Delete compressor, force Python GC, then force C-level memory release
|
||||
# Cleanup: drop the compressor, then force C-level memory back to the OS.
|
||||
#
|
||||
# There is deliberately no gc.collect() here. brotli.Compressor is not gc-tracked,
|
||||
# so the collector can never reclaim it - `del` frees it immediately by refcount.
|
||||
# Measured over 60 x 2.2MB compressions, RSS growth was:
|
||||
#
|
||||
# neither +0.9MB
|
||||
# gc.collect() only +0.2MB
|
||||
# malloc_trim() only +0.0MB <- does all of the work
|
||||
# both (previous) +0.0MB <- the collect contributed nothing
|
||||
#
|
||||
# malloc_trim below is the load-bearing line: brotli's retention is glibc holding
|
||||
# freed arenas, which only a trim returns. The collect cost ~31ms of stop-the-world
|
||||
# per snapshot save for no reclamation.
|
||||
del compressor
|
||||
gc.collect()
|
||||
|
||||
# Force release of C-level memory back to OS (since brotli is a C library)
|
||||
try:
|
||||
@@ -689,7 +702,7 @@ class model(EntityPersistenceMixin, watch_base):
|
||||
|
||||
# reimport
|
||||
bump = self.history
|
||||
gc.collect()
|
||||
gc_debounce.collect('watch.history_bump')
|
||||
|
||||
# Save some text file to the appropriate path and bump the history
|
||||
# result_obj from fetch_site_status.run()
|
||||
|
||||
@@ -7,7 +7,6 @@
|
||||
// The Search button is rendered in the left rail and the mobile drawer.
|
||||
const openSearchButtons = document.querySelectorAll('.js-open-search-modal');
|
||||
const closeSearchButton = document.getElementById('close-search-modal');
|
||||
const searchForm = document.getElementById('search-form');
|
||||
const searchInput = document.getElementById('search-modal-input');
|
||||
|
||||
if (!searchModal || openSearchButtons.length === 0) {
|
||||
@@ -63,6 +62,15 @@
|
||||
|
||||
// Close modal when clicking the backdrop
|
||||
searchModal.addEventListener('click', function(e) {
|
||||
// Only real pointer clicks can land on the backdrop. Keyboard-synthesised clicks
|
||||
// report detail 0 and coordinates of 0,0, which the geometry test below reads as
|
||||
// "outside the dialog" - and implicit form submission (Enter in the input) fires
|
||||
// exactly such a click at the Search button. That closed the modal and blanked
|
||||
// the input mid-dispatch, so the submit that followed hit an empty `required`
|
||||
// field and was rejected: Enter appeared to just dismiss the form.
|
||||
if (e.detail === 0) {
|
||||
return;
|
||||
}
|
||||
const rect = searchModal.getBoundingClientRect();
|
||||
const isInDialog = (
|
||||
rect.top <= e.clientY &&
|
||||
@@ -93,43 +101,8 @@
|
||||
}
|
||||
});
|
||||
|
||||
// Handle Enter key in search input
|
||||
if (searchInput) {
|
||||
searchInput.addEventListener('keydown', function(e) {
|
||||
if (e.key === 'Enter') {
|
||||
e.preventDefault();
|
||||
if (searchForm) {
|
||||
// Trigger form submission programmatically
|
||||
searchForm.dispatchEvent(new Event('submit'));
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Handle form submission
|
||||
if (searchForm) {
|
||||
searchForm.addEventListener('submit', function(e) {
|
||||
e.preventDefault();
|
||||
|
||||
// Get form data
|
||||
const formData = new FormData(searchForm);
|
||||
const searchQuery = formData.get('q');
|
||||
const tags = formData.get('tags');
|
||||
|
||||
// Build URL
|
||||
const params = new URLSearchParams();
|
||||
if (searchQuery) {
|
||||
params.append('q', searchQuery);
|
||||
}
|
||||
if (tags) {
|
||||
params.append('tags', tags);
|
||||
}
|
||||
|
||||
// Navigate to search results (always redirect to watchlist home)
|
||||
// Use base_path if available (for sub-path deployments like /enlighten-richerx)
|
||||
const basePath = typeof base_path !== 'undefined' ? base_path : '';
|
||||
window.location.href = basePath + '/?' + params.toString();
|
||||
});
|
||||
}
|
||||
// Submission is left to the browser: the form carries a server-rendered action
|
||||
// (correct under a reverse-proxy sub-path) and Enter in the input triggers implicit
|
||||
// submission via the footer's submit button, which also runs `required` validation.
|
||||
});
|
||||
})();
|
||||
|
||||
@@ -95,7 +95,9 @@
|
||||
overflow-y: auto;
|
||||
padding-top: 60px;
|
||||
|
||||
|
||||
.action-label {
|
||||
display: block !important;
|
||||
}
|
||||
#cdio-logo {
|
||||
color: var(--color-text);
|
||||
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -317,11 +317,18 @@
|
||||
<h2 class="modal-title" id="search-modal-title">{{ _('Search') }}</h2>
|
||||
</div>
|
||||
<div class="modal-body">
|
||||
<form id="search-form" method="GET">
|
||||
{# Plain GET submit to the watchlist - url_for() carries the reverse-proxy sub-path
|
||||
(SCRIPT_NAME), so no client-side URL building is needed. #}
|
||||
<form id="search-form" method="GET" action="{{ url_for('watchlist.index') }}">
|
||||
<div class="pure-control-group">
|
||||
<label for="search-modal-input">{% if active_tag_uuid %}{{ _("URL or Title in '%(title)s'", title=active_tag.title) }}{% else %}{{ _('URL or Title') }}{% endif %}</label>
|
||||
{# Matches watch_passes_search() in blueprint/watchlist/filters.py - title, URL and last error text #}
|
||||
<label for="search-modal-input">{{ _('URL, title or error text') }}</label>
|
||||
<input id="search-modal-input" class="m-d" name="q" placeholder="{{ _('Enter search term...') }}" required type="text" value="" autofocus>
|
||||
<input name="tags" type="hidden" value="{% if active_tag_uuid %}{{active_tag_uuid}}{% endif %}">
|
||||
{# 'tag' (not 'tags') - that's the arg the watchlist filters on #}
|
||||
{% if active_tag_uuid %}
|
||||
<input name="tag" type="hidden" value="{{ active_tag_uuid }}">
|
||||
<span class="pure-form-message-inline">{{ _("Searching in current group '%(title)s' only", title=active_tag.title) }}</span>
|
||||
{% endif %}
|
||||
</div>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
"""The browser fetchers must judge a fetch on the document they end up extracting.
|
||||
|
||||
Plenty of sites answer the first request with an interstitial carrying an error status and a
|
||||
client-side redirect, then serve the real page. Judging the fetch on the first navigation fails a
|
||||
watch whose content is present and fine (reported against fotokoch.de: 503 + meta refresh -> 200),
|
||||
and on the pyppeteer fetcher a replaced document used to hang goto() until the hard processing
|
||||
timeout because its navigation watcher is bound to the loaderId it started on.
|
||||
"""
|
||||
|
||||
import os
|
||||
from flask import url_for
|
||||
from ..util import wait_for_all_checks
|
||||
|
||||
|
||||
def _cdio(url):
|
||||
# The browser runs in another container in CI and reaches the test server as 'cdio'
|
||||
return url.replace('localhost.localdomain', 'cdio').replace('localhost', 'cdio')
|
||||
|
||||
|
||||
def test_interstitial_redirect_is_followed(client, live_server, measure_memory_usage, datastore_path):
|
||||
assert os.getenv('PLAYWRIGHT_DRIVER_URL'), "Needs PLAYWRIGHT_DRIVER_URL set for this test"
|
||||
|
||||
res = client.post(
|
||||
url_for("settings.settings_page"),
|
||||
data={
|
||||
"application-empty_pages_are_a_change": "",
|
||||
"requests-time_between_check-minutes": 180,
|
||||
'application-fetch_backend': "html_webdriver",
|
||||
},
|
||||
follow_redirects=True
|
||||
)
|
||||
assert b"Settings updated." in res.data
|
||||
|
||||
test_url = _cdio(url_for('test_interstitial', key='renav', _external=True))
|
||||
|
||||
res = client.post(
|
||||
url_for("imports.import_page"),
|
||||
data={"urls": test_url},
|
||||
follow_redirects=True
|
||||
)
|
||||
assert b"1 Imported" in res.data
|
||||
wait_for_all_checks(client)
|
||||
|
||||
# The interstitial answered 503, so judging the first navigation would have failed the watch
|
||||
uuid = next(iter(live_server.app.config['DATASTORE'].data['watching']))
|
||||
watch = live_server.app.config['DATASTORE'].data['watching'][uuid]
|
||||
assert not watch.get('last_error'), \
|
||||
f"Watch was judged on the interstitial instead of the page it landed on: {watch.get('last_error')}"
|
||||
|
||||
res = client.get(url_for("watchlist.index"))
|
||||
assert b'Error - 503' not in res.data
|
||||
|
||||
assert watch.history_n >= 1, "Fetch succeeded but no snapshot was stored"
|
||||
snapshot = watch.get_history_snapshot(list(watch.history.keys())[-1])
|
||||
assert 'The real page content is here' in snapshot
|
||||
assert 'Browser check in progress' not in snapshot
|
||||
|
||||
client.post(url_for("ui.form_delete", uuid="all"), follow_redirects=True)
|
||||
@@ -1,5 +1,6 @@
|
||||
from flask import url_for
|
||||
from .util import set_original_response, set_modified_response, live_server_setup
|
||||
import re
|
||||
import time
|
||||
|
||||
|
||||
@@ -71,3 +72,43 @@ def test_search_in_tag_limit(client, live_server, measure_memory_usage, datastor
|
||||
assert urls[0].split(' ')[0].encode('utf-8') in res.data, urls[0].encode('utf-8')
|
||||
assert urls[1].split(' ')[0].encode('utf-8') not in res.data, urls[0].encode('utf-8')
|
||||
|
||||
|
||||
|
||||
def test_search_modal_form_action(client, live_server, measure_memory_usage, datastore_path):
|
||||
# The search modal submits as a plain GET form, so its action has to carry the
|
||||
# reverse-proxy sub-path (SCRIPT_NAME), otherwise search jumps to the host root.
|
||||
res = client.get(url_for("watchlist.index"))
|
||||
assert b'<form id="search-form" method="GET" action="/">' in res.data
|
||||
|
||||
res = client.get("/", base_url="http://localhost/sub-path")
|
||||
assert b'<form id="search-form" method="GET" action="/sub-path/">' in res.data
|
||||
|
||||
|
||||
def test_search_modal_tag_field_is_filterable(client, live_server, measure_memory_usage, datastore_path):
|
||||
# The modal carries the active tag as a hidden field so a search stays scoped to the
|
||||
# tag you were viewing. The field name has to be the one the watchlist filters on.
|
||||
urls = ['https://localhost:12300?first-result=1 tag-one',
|
||||
'https://localhost:5000?second-result=1 tag-two'
|
||||
]
|
||||
res = client.post(
|
||||
url_for("imports.import_page"),
|
||||
data={"urls": "\r\n".join(urls)},
|
||||
follow_redirects=True
|
||||
)
|
||||
assert b"2 Imported" in res.data
|
||||
|
||||
res = client.get(url_for("watchlist.index") + "?tag=tag-one")
|
||||
form = re.search(rb'<form id="search-form".*?</form>', res.data, re.DOTALL)
|
||||
assert form, "search modal form not rendered"
|
||||
field = re.search(rb'<input name="([^"]+)" type="hidden" value="([^"]+)"', form.group(0))
|
||||
assert field, f"no populated hidden tag field in {form.group(0)}"
|
||||
name, value = field.group(1).decode(), field.group(2).decode()
|
||||
|
||||
# The scoping is spelled out in the modal, so narrowed results aren't a surprise
|
||||
assert b'Searching in current group' in form.group(0)
|
||||
assert b'tag-one' in form.group(0)
|
||||
|
||||
# 'localhost' matches both watches, so only the tag field can narrow it down
|
||||
res = client.get(url_for("watchlist.index") + f"?q=localhost&{name}={value}")
|
||||
assert urls[0].split(' ')[0].encode('utf-8') in res.data
|
||||
assert urls[1].split(' ')[0].encode('utf-8') not in res.data, f"'{name}' is not filtered on by the watchlist"
|
||||
|
||||
@@ -249,6 +249,31 @@ def new_live_server_setup(live_server):
|
||||
import secrets
|
||||
return "Random content - {}\n".format(secrets.token_hex(64))
|
||||
|
||||
# Re-navigation gate: the first hit answers with an error status AND a client-side meta
|
||||
# refresh, the second serves the real page. This mirrors sites that gate visitors they have
|
||||
# not seen recently (reported against fotokoch.de, which answers 503 + meta refresh and then
|
||||
# serves a 200). The browser follows the refresh, so the fetch has to be judged on the
|
||||
# document we actually end up extracting rather than on the interstitial.
|
||||
# Keyed on last-seen time rather than a hit count, so EVERY fresh check starts out gated -
|
||||
# a counter would serve a clean 200 to the second check and let the test pass without the fix.
|
||||
_interstitial_last_seen = {}
|
||||
|
||||
@live_server.app.route('/test-interstitial')
|
||||
def test_interstitial():
|
||||
key = request.args.get('key', 'default')
|
||||
now = time.time()
|
||||
seen_recently = (now - _interstitial_last_seen.get(key, 0)) < 10
|
||||
_interstitial_last_seen[key] = now
|
||||
if not seen_recently:
|
||||
resp = make_response(
|
||||
'<html><head><meta http-equiv="refresh" content="1"></head>'
|
||||
'<body>Browser check in progress, you will be redirected</body></html>', 503)
|
||||
else:
|
||||
resp = make_response(
|
||||
'<html><body><h1>The real page content is here</h1></body></html>', 200)
|
||||
resp.headers['Content-Type'] = 'text/html'
|
||||
return resp
|
||||
|
||||
@live_server.app.route('/test-endpoint2')
|
||||
def test_endpoint2():
|
||||
return "<html><body>some basic content</body></html>"
|
||||
|
||||
Binary file not shown.
@@ -4571,18 +4571,18 @@ msgid "Search"
|
||||
msgstr "Hledat"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr ""
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, název nebo text chyby"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Hledá se pouze ve skupině '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4625,18 +4625,18 @@ msgid "Search"
|
||||
msgstr "Suchen"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL oder Titel in '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL oder Titel"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, Titel oder Fehlertext"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Suchbegriff eingeben..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Suche nur in Gruppe '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4563,18 +4563,18 @@ msgid "Search"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr ""
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, title or error text"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Searching in current group '%(title)s' only"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4563,18 +4563,18 @@ msgid "Search"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr ""
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, title or error text"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Searching in current group '%(title)s' only"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4640,18 +4640,18 @@ msgid "Search"
|
||||
msgstr "Buscar"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL o título en '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL o título"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, título o texto de error"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Introduzca el término de búsqueda..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Buscando solo en el grupo '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4578,18 +4578,18 @@ msgid "Search"
|
||||
msgstr "Rechercher"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr ""
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, titre ou texte d'erreur"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Recherche uniquement dans le groupe '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4684,18 +4684,18 @@ msgid "Search"
|
||||
msgstr "Cari"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL atau Judul di '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL atau Judul"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, judul atau teks kesalahan"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Masukkan istilah pencarian..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Mencari hanya dalam grup '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4565,18 +4565,18 @@ msgid "Search"
|
||||
msgstr "Cerca"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr ""
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, titolo o testo dell'errore"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Ricerca solo nel gruppo '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4596,18 +4596,18 @@ msgid "Search"
|
||||
msgstr "検索"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "'%(title)s' 内のURLまたはタイトル"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URLまたはタイトル"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL、タイトル、エラーテキスト"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "検索語を入力..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "グループ「%(title)s」内のみを検索"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4573,18 +4573,18 @@ msgid "Search"
|
||||
msgstr "검색"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "'%(title)s'의 URL 또는 제목"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL 또는 제목"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, 제목 또는 오류 텍스트"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "검색어를 입력하세요..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "'%(title)s' 그룹 내에서만 검색"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
@@ -8,7 +8,7 @@ msgid ""
|
||||
msgstr ""
|
||||
"Project-Id-Version: changedetection.io 0.60.4\n"
|
||||
"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n"
|
||||
"POT-Creation-Date: 2026-09-12 15:41+0200\n"
|
||||
"POT-Creation-Date: 2026-09-12 17:18+0200\n"
|
||||
"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n"
|
||||
"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n"
|
||||
"Language-Team: LANGUAGE <LL@li.org>\n"
|
||||
@@ -4562,18 +4562,18 @@ msgid "Search"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgid "URL, title or error text"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr ""
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4745,18 +4745,18 @@ msgid "Search"
|
||||
msgstr "Wyszukiwanie"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "Adres URL lub tytuł w „%(title)s”"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "Adres URL lub tytuł"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, tytuł lub treść błędu"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Wpisz szukane hasło..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Szukanie tylko w grupie '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4615,18 +4615,18 @@ msgid "Search"
|
||||
msgstr "Pesquisar"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL ou Título em '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL ou Título"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, título ou texto de erro"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Digite o termo de busca..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Buscando apenas no grupo '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4693,18 +4693,18 @@ msgid "Search"
|
||||
msgstr "Поиск"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL-адрес или заголовок в формате «%(title)s»"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL или заголовок"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, заголовок или текст ошибки"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Введите поисковый запрос..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Поиск только в группе '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4620,18 +4620,18 @@ msgid "Search"
|
||||
msgstr "Ara"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "'%(title)s' içinde URL veya Başlık"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL veya Başlık"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, başlık veya hata metni"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Arama terimini girin..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Yalnızca '%(title)s' grubunda aranıyor"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4597,18 +4597,18 @@ msgid "Search"
|
||||
msgstr "Пошук"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL або Назва у '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL або Назва"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL, заголовок або текст помилки"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "Введіть пошуковий запит..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "Пошук лише в групі '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4571,18 +4571,18 @@ msgid "Search"
|
||||
msgstr "搜索"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "'%(title)s' 中的 URL 或标题"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL 或标题"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL、标题或错误文本"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "输入搜索关键词..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "仅在分组「%(title)s」中搜索"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
Binary file not shown.
@@ -4574,18 +4574,18 @@ msgid "Search"
|
||||
msgstr "搜尋"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "URL or Title in '%(title)s'"
|
||||
msgstr "URL 或標題 在 '%(title)s'"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "URL or Title"
|
||||
msgstr "URL 或標題"
|
||||
msgid "URL, title or error text"
|
||||
msgstr "URL、標題或錯誤文字"
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
msgid "Enter search term..."
|
||||
msgstr "輸入搜尋關鍵字 ..."
|
||||
|
||||
#: changedetectionio/templates/base.html
|
||||
#, python-format
|
||||
msgid "Searching in current group '%(title)s' only"
|
||||
msgstr "僅在群組「%(title)s」中搜尋"
|
||||
|
||||
#: changedetectionio/templates/edit/include_llm_intent.html
|
||||
msgid ""
|
||||
"<strong>On</strong> – every watch in this group uses the AI settings below, unless it fills in its own. "
|
||||
|
||||
@@ -6,6 +6,7 @@ from changedetectionio import html_tools
|
||||
from changedetectionio import worker_pool
|
||||
from changedetectionio.queuedWatchMetaData import PrioritizedItem
|
||||
from changedetectionio.pluggy_interface import apply_update_handler_alter, apply_update_finalize
|
||||
from changedetectionio import gc_debounce
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
@@ -655,9 +656,8 @@ async def async_update_worker(worker_id, q, notification_q, app, datastore, exec
|
||||
del update_handler
|
||||
update_handler = None
|
||||
|
||||
# Force garbage collection
|
||||
import gc
|
||||
gc.collect()
|
||||
# Force garbage collection (debounced process-wide, see gc_debounce)
|
||||
gc_debounce.collect('worker.after_processing')
|
||||
|
||||
except Exception as e:
|
||||
# Store the processing exception for plugin finalization hook
|
||||
@@ -702,8 +702,7 @@ async def async_update_worker(worker_id, q, notification_q, app, datastore, exec
|
||||
del contents
|
||||
|
||||
# Force garbage collection after all references are cleared
|
||||
import gc
|
||||
gc.collect()
|
||||
gc_debounce.collect('worker.cleanup_finally')
|
||||
|
||||
logger.debug(f"Worker {worker_id} completed watch {uuid} in {time.time()-fetch_start_time:.2f}s")
|
||||
except Exception as cleanup_error:
|
||||
|
||||
Reference in New Issue
Block a user