From f682a80c43a375f688b875b712ba6c7103326c3a Mon Sep 17 00:00:00 2001 From: dgtlmoon Date: Thu, 27 Feb 2025 15:34:02 +0100 Subject: [PATCH] BrowserSteps - Speed up scraping, refactor screenshot handling for very long pages --- .../blueprint/browser_steps/__init__.py | 29 ++++++------- .../blueprint/browser_steps/browser_steps.py | 43 ++++++++----------- .../content_fetchers/playwright.py | 14 ++++-- 3 files changed, 42 insertions(+), 44 deletions(-) diff --git a/changedetectionio/blueprint/browser_steps/__init__.py b/changedetectionio/blueprint/browser_steps/__init__.py index a472ba4b0..3e5c50c51 100644 --- a/changedetectionio/blueprint/browser_steps/__init__.py +++ b/changedetectionio/blueprint/browser_steps/__init__.py @@ -160,14 +160,13 @@ def construct_blueprint(datastore: ChangeDetectionStore): if not browsersteps_sessions.get(browsersteps_session_id): return make_response('No session exists under that ID', 500) - + is_last_step = False # Actions - step/apply/etc, do the thing and return state if request.method == 'POST': # @todo - should always be an existing session step_operation = request.form.get('operation') step_selector = request.form.get('selector') step_optional_value = request.form.get('optional_value') - step_n = int(request.form.get('step_n')) is_last_step = strtobool(request.form.get('is_last_step')) # @todo try.. accept.. nice errors not popups.. @@ -182,16 +181,6 @@ def construct_blueprint(datastore: ChangeDetectionStore): # Try to find something of value to give back to the user return make_response(str(e).splitlines()[0], 401) - # Get visual selector ready/update its data (also use the current filter info from the page?) - # When the last 'apply' button was pressed - # @todo this adds overhead because the xpath selection is happening twice - u = browsersteps_sessions[browsersteps_session_id]['browserstepper'].page.url - if is_last_step and u: - (screenshot, xpath_data) = browsersteps_sessions[browsersteps_session_id]['browserstepper'].request_visualselector_data() - watch = datastore.data['watching'].get(uuid) - if watch: - watch.save_screenshot(screenshot=screenshot) - watch.save_xpath_data(data=xpath_data) # if not this_session.page: # cleanup_playwright_session() @@ -199,10 +188,20 @@ def construct_blueprint(datastore: ChangeDetectionStore): # Screenshots and other info only needed on requesting a step (POST) try: - state = browsersteps_sessions[browsersteps_session_id]['browserstepper'].get_current_state() + (screenshot, xpath_data) = browsersteps_sessions[browsersteps_session_id]['browserstepper'].get_current_state() + if is_last_step: + watch = datastore.data['watching'].get(uuid) + u = browsersteps_sessions[browsersteps_session_id]['browserstepper'].page.url + if watch and u: + watch.save_screenshot(screenshot=screenshot) + watch.save_xpath_data(data=xpath_data) + except playwright._impl._api_types.Error as e: return make_response("Browser session ran out of time :( Please reload this page."+str(e), 401) + except Exception as e: + return make_response("Error fetching screenshot and element data - " + str(e), 401) + # SEND THIS BACK TO THE BROWSER # Use send_file() which is way faster than read/write loop on bytes import json from tempfile import mkstemp @@ -210,8 +209,8 @@ def construct_blueprint(datastore: ChangeDetectionStore): tmp_fd, tmp_file = mkstemp(text=True, suffix=".json", prefix="changedetectionio-") output = json.dumps({'screenshot': "data:image/jpeg;base64,{}".format( - base64.b64encode(state[0]).decode('ascii')), - 'xpath_data': state[1], + base64.b64encode(screenshot).decode('ascii')), + 'xpath_data': xpath_data, 'session_age_start': browsersteps_sessions[browsersteps_session_id]['browserstepper'].age_start, 'browser_time_remaining': round(remaining) }) diff --git a/changedetectionio/blueprint/browser_steps/browser_steps.py b/changedetectionio/blueprint/browser_steps/browser_steps.py index f533edf0b..00a30a36c 100644 --- a/changedetectionio/blueprint/browser_steps/browser_steps.py +++ b/changedetectionio/blueprint/browser_steps/browser_steps.py @@ -1,14 +1,15 @@ -#!/usr/bin/env python3 - import os import time import re from random import randint from loguru import logger +from changedetectionio.content_fetchers.helpers import capture_stitched_together_full_page, SCREENSHOT_SIZE_STITCH_THRESHOLD from changedetectionio.content_fetchers.base import manage_user_agent from changedetectionio.safe_jinja import render as jinja_render + + # Two flags, tell the JS which of the "Selector" or "Value" field should be enabled in the front end # 0- off, 1- on browser_step_ui_config = {'Choose one': '0 0', @@ -279,6 +280,7 @@ class browsersteps_live_ui(steppable_browser_interface): logger.debug(f"Time to browser setup {time.time()-now:.2f}s") self.page.wait_for_timeout(1 * 1000) + def mark_as_closed(self): logger.debug("Page closed, cleaning up..") @@ -296,39 +298,30 @@ class browsersteps_live_ui(steppable_browser_interface): now = time.time() self.page.wait_for_timeout(1 * 1000) - # The actual screenshot - screenshot = self.page.screenshot(type='jpeg', full_page=True, quality=40) + full_height = self.page.evaluate("document.documentElement.scrollHeight") + + if full_height >= SCREENSHOT_SIZE_STITCH_THRESHOLD: + logger.warning(f"Page full Height: {full_height}px longer than {SCREENSHOT_SIZE_STITCH_THRESHOLD}px, using 'stitched screenshot method'.") + screenshot = capture_stitched_together_full_page(self.page) + else: + screenshot = self.page.screenshot(type='jpeg', full_page=True, quality=40) + + logger.debug(f"Time to get screenshot from browser {time.time() - now:.2f}s") + + now = time.time() self.page.evaluate("var include_filters=''") # Go find the interactive elements # @todo in the future, something smarter that can scan for elements with .click/focus etc event handlers? elements = 'a,button,input,select,textarea,i,th,td,p,li,h1,h2,h3,h4,div,span' xpath_element_js = xpath_element_js.replace('%ELEMENTS%', elements) + xpath_data = self.page.evaluate("async () => {" + xpath_element_js + "}") # So the JS will find the smallest one first xpath_data['size_pos'] = sorted(xpath_data['size_pos'], key=lambda k: k['width'] * k['height'], reverse=True) - logger.debug(f"Time to complete get_current_state of browser {time.time()-now:.2f}s") - # except + logger.debug(f"Time to scrape xpath element data in browser {time.time()-now:.2f}s") + # playwright._impl._api_types.Error: Browser closed. # @todo show some countdown timer? return (screenshot, xpath_data) - def request_visualselector_data(self): - """ - Does the same that the playwright operation in content_fetcher does - This is used to just bump the VisualSelector data so it' ready to go if they click on the tab - @todo refactor and remove duplicate code, add include_filters - :param xpath_data: - :param screenshot: - :param current_include_filters: - :return: - """ - import importlib.resources - self.page.evaluate("var include_filters=''") - xpath_element_js = importlib.resources.files("changedetectionio.content_fetchers.res").joinpath('xpath_element_scraper.js').read_text() - from changedetectionio.content_fetchers import visualselector_xpath_selectors - xpath_element_js = xpath_element_js.replace('%ELEMENTS%', visualselector_xpath_selectors) - xpath_data = self.page.evaluate("async () => {" + xpath_element_js + "}") - screenshot = self.page.screenshot(type='jpeg', full_page=True, quality=int(os.getenv("SCREENSHOT_QUALITY", 72))) - - return (screenshot, xpath_data) diff --git a/changedetectionio/content_fetchers/playwright.py b/changedetectionio/content_fetchers/playwright.py index 53be33f1d..70a3c6972 100644 --- a/changedetectionio/content_fetchers/playwright.py +++ b/changedetectionio/content_fetchers/playwright.py @@ -4,6 +4,7 @@ from urllib.parse import urlparse from loguru import logger +from changedetectionio.content_fetchers.helpers import capture_stitched_together_full_page, SCREENSHOT_SIZE_STITCH_THRESHOLD from changedetectionio.content_fetchers.base import Fetcher, manage_user_agent from changedetectionio.content_fetchers.exceptions import PageUnloadable, Non200ErrorCodeReceived, EmptyReply, ScreenshotUnavailable @@ -199,10 +200,15 @@ class fetcher(Fetcher): # acceptable screenshot quality here try: # The actual screenshot - this always base64 and needs decoding! horrible! huge CPU usage - self.screenshot = self.page.screenshot(type='jpeg', - full_page=True, - quality=int(os.getenv("SCREENSHOT_QUALITY", 72)), - ) + full_height = self.page.evaluate("document.documentElement.scrollHeight") + + if full_height >= SCREENSHOT_SIZE_STITCH_THRESHOLD: + logger.warning( + f"Page full Height: {full_height}px longer than {SCREENSHOT_SIZE_STITCH_THRESHOLD}px, using 'stitched screenshot method'.") + self.screenshot = capture_stitched_together_full_page(self.page) + else: + self.screenshot = self.page.screenshot(type='jpeg', full_page=True, quality=int(os.getenv("SCREENSHOT_QUALITY", 30))) + except Exception as e: # It's likely the screenshot was too long/big and something crashed raise ScreenshotUnavailable(url=url, status_code=self.status_code)