Files
changedetection.io/changedetectionio/blueprint/ui/diff.py
T
dgtlmoon 11916a89e6 LLM - UI nonblocking (#4469)
* Make LLM summary from UI non blocking

* Increase logging
2026-09-20 16:00:12 +02:00

830 lines
38 KiB
Python

from flask import Blueprint, request, redirect, url_for, flash, render_template, make_response, send_from_directory
from flask_babel import gettext
import re
import importlib
from loguru import logger
from markupsafe import Markup
from changedetectionio.diff import (
REMOVED_STYLE, ADDED_STYLE, REMOVED_INNER_STYLE, ADDED_INNER_STYLE,
REMOVED_PLACEMARKER_OPEN, REMOVED_PLACEMARKER_CLOSED,
ADDED_PLACEMARKER_OPEN, ADDED_PLACEMARKER_CLOSED,
CHANGED_PLACEMARKER_OPEN, CHANGED_PLACEMARKER_CLOSED,
CHANGED_INTO_PLACEMARKER_OPEN, CHANGED_INTO_PLACEMARKER_CLOSED
)
from changedetectionio.store import ChangeDetectionStore
from changedetectionio.blueprint import plaintext_response
from changedetectionio.auth_decorator import login_optionally_required
def _clean_litellm_error(exc) -> str:
"""Return a short, human-readable error string from a litellm exception.
litellm embeds the raw provider JSON in str(exc), which can be hundreds of
characters of verbose quota detail. We try to pull just the provider's
'message' field; failing that we return the first non-empty line with the
'litellm.XxxError:' class prefix stripped.
"""
import json, re
raw = str(exc)
# Try to parse the embedded JSON block (starts at first '{')
brace = raw.find('{')
if brace >= 0:
try:
payload = json.loads(raw[brace:])
msg = (payload.get('error') or {}).get('message') or ''
if msg:
# Take only the first sentence / line — provider messages can be long
return msg.split('\n')[0].split('. ')[0].strip() + '.'
except Exception:
pass
# Fallback: strip the "litellm.XxxError: litellm.XxxError: providerException - " prefix
first_line = raw.split('\n')[0]
first_line = re.sub(r'^(litellm\.\w+:\s*)+', '', first_line)
first_line = re.sub(r'\w+Exception\s*-\s*', '', first_line).strip()
return first_line or raw.split('\n')[0]
def construct_blueprint(datastore: ChangeDetectionStore):
diff_blueprint = Blueprint('ui_diff', __name__, template_folder="../ui/templates")
@diff_blueprint.app_template_filter('diff_unescape_difference_spans')
def diff_unescape_difference_spans(content):
"""Emulate Jinja2's auto-escape, then selectively unescape our diff spans."""
from markupsafe import escape
if not content:
return Markup('')
# Step 1: Escape everything like Jinja2 would (this makes it XSS-safe)
escaped_content = escape(str(content))
# Step 2: Unescape only our exact diff spans generated by apply_html_color_to_body()
# Pattern matches the exact structure:
# <span style="{STYLE}" role="{ROLE}" aria-label="{LABEL}" title="{TITLE}">
# Unescape outer span opening tags with full attributes (role, aria-label, title)
# Matches removed/added/changed/changed_into spans
result = re.sub(
rf'&lt;span style=&#34;({re.escape(REMOVED_STYLE)}|{re.escape(ADDED_STYLE)})&#34; '
rf'role=&#34;(deletion|insertion|note)&#34; '
rf'aria-label=&#34;([^&]+?)&#34; '
rf'title=&#34;([^&]+?)&#34;&gt;',
r'<span style="\1" role="\2" aria-label="\3" title="\4">',
str(escaped_content),
flags=re.IGNORECASE
)
# Unescape inner span opening tags (without additional attributes)
# This matches the darker background styles for changed parts within lines
result = re.sub(
rf'&lt;span style=&#34;({re.escape(REMOVED_INNER_STYLE)}|{re.escape(ADDED_INNER_STYLE)})&#34;&gt;',
r'<span style="\1">',
result,
flags=re.IGNORECASE
)
# Unescape closing tags (but only as many as we opened)
open_count = result.count('<span style=')
close_count = str(escaped_content).count('&lt;/span&gt;')
# Replace up to the number of spans we opened
for _ in range(min(open_count, close_count)):
result = result.replace('&lt;/span&gt;', '</span>', 1)
return Markup(result)
@diff_blueprint.route("/diff/<uuid_str:uuid>", methods=['GET'])
@login_optionally_required
def diff_history_page(uuid):
"""
Render the history/diff page for a watch.
This route is processor-aware: it delegates rendering to the processor's
difference.py module, allowing different processor types to provide
custom visualizations:
- text_json_diff: Text/HTML diff with syntax highlighting
- restock_diff: Could show price charts and stock history
- image_diff: Could show image comparison slider/overlay
Each processor implements processors/{type}/difference.py::render()
If a processor doesn't have a difference module, falls back to text_json_diff.
"""
if uuid == 'first':
uuid = list(datastore.data['watching'].keys()).pop()
try:
watch = datastore.data['watching'][uuid]
except KeyError:
flash(gettext("No history found for the specified link, bad link?"), "error")
return redirect(url_for('watchlist.index'))
dates = list(watch.history.keys())
if not dates or len(dates) < 2:
flash(gettext("Not enough history (2 snapshots required) to show difference page for this watch."), "error")
return redirect(url_for('watchlist.index'))
# Get the processor type for this watch
processor_name = watch.get('processor', 'text_json_diff')
# Try to get the processor's difference module (works for both built-in and plugin processors)
from changedetectionio.processors import get_processor_submodule
processor_module = get_processor_submodule(processor_name, 'difference')
# Call the processor's render() function
if processor_module and hasattr(processor_module, 'render'):
return processor_module.render(
watch=watch,
datastore=datastore,
request=request,
url_for=url_for,
render_template=render_template,
flash=flash,
redirect=redirect
)
# Fallback: if processor doesn't have difference module, use text_json_diff as default
from changedetectionio.processors.text_json_diff.difference import render as default_render
return default_render(
watch=watch,
datastore=datastore,
request=request,
url_for=url_for,
render_template=render_template,
flash=flash,
redirect=redirect
)
@diff_blueprint.route("/diff/<uuid_str:uuid>/llm-summary/prompt", methods=['GET'])
@login_optionally_required
def diff_llm_summary_prompt(uuid):
"""Return the effective LLM summary prompt for a watch immediately (no LLM call)."""
from flask import jsonify
watch = datastore.data['watching'].get(uuid)
if not watch:
return jsonify({'prompt': ''}), 404
try:
from changedetectionio.llm.evaluator import get_effective_summary_prompt
prompt = get_effective_summary_prompt(watch, datastore)
except Exception:
prompt = ''
return jsonify({'prompt': prompt})
# ── AI change summary ────────────────────────────────────────────────────────────────────
# Generation is slow (an LLM round-trip), so it does not happen in the request thread:
#
# POST /diff/<uuid>/llm-summary start it (or return an already-cached summary)
# GET /diff/<uuid>/llm-summary poll for the result
#
# POST is deliberate: starting a generation spends tokens and marks the watch viewed, which
# must not be reachable from a bare cross-site GET (an <img src=...> would do it). Flask-WTF's
# CSRFProtect covers every non-GET route, and static/js/csrf.js puts the token on the header
# of every jQuery non-GET call, so the browser side needs nothing extra.
#
# The GET poll stays side-effect free: it reads the cache and the in-flight registry and never
# starts work, which is what makes it safe to leave as a GET.
def _summary_request_context(uuid):
"""Resolve everything both routes need, without reading snapshots or building a diff.
Returns (context_dict, None) or (None, (json_body, http_status)) when the request cannot
be served at all. Kept cheap so the poll route stays cheap: the cache key does not depend
on the diff text, only on the version pair, the effective prompt, the diff prefs and the
model - so a poll never has to re-run difflib over two snapshots.
"""
from changedetectionio.llm.evaluator import (
DiffPrefs, build_summary_cache_prompt, get_effective_summary_prompt, get_llm_settings,
resolve_llm_timeout,
)
try:
watch = datastore.data['watching'][uuid]
except KeyError:
return None, ({'summary': None, 'error': 'Watch not found', 'status': 'error'}, 404)
llm_cfg = datastore.data.get('settings', {}).get('application', {}).get('llm', {})
if not llm_cfg.get('model'):
return None, ({'summary': None, 'error': 'LLM not configured', 'status': 'error'}, 400)
dates = list(watch.history.keys())
if len(dates) < 2:
return None, ({'summary': None, 'error': 'Not enough history', 'status': 'error'}, 400)
# Default baseline for the watchlist "Summary" link (when no explicit from_version
# is requested) is configurable at Settings > AI. Default 'second_last_version'
# compares the previous snapshot; 'since_last_viewed' uses the operator's last view.
if llm_cfg.get('watchlist_overview_summary', 'second_last_version') == 'since_last_viewed':
best_from = watch.get_from_version_based_on_last_viewed
default_from = best_from if best_from else dates[-2]
else:
default_from = dates[-2]
prefs = DiffPrefs.from_request_args(request.args)
from_version = request.args.get('from_version', default_from)
to_version = request.args.get('to_version', dates[-1])
settings = get_llm_settings(datastore)
# Diff-pref flags + system prompt + active model are part of the cache key
# so prompt or model changes bust the cache.
cache_prompt = build_summary_cache_prompt(
effective_prompt=get_effective_summary_prompt(watch, datastore),
max_summary_tokens=settings.max_summary_tokens,
prefs=prefs,
model=settings.model,
)
return {
'watch': watch,
'dates': dates,
'prefs': prefs,
'from_version': from_version,
'to_version': to_version,
'cache_prompt': cache_prompt,
# Same granularity as the on-disk cache filename (version pair + prompt hash), so an
# in-flight job is deduplicated exactly when it would write to the same cache entry.
'job_key': f"{uuid}|{from_version}|{to_version}|"
f"{watch._llm_summary_prompt_hash(cache_prompt)}",
# How long this deployment is prepared to wait for the model on one call. The
# browser must not give up before this, or it reports a failure for a job that is
# still running and will still write its summary - a local Ollama/vLLM endpoint
# gets 1800s here against the client's old fixed 180s. See _pending_reply().
'llm_timeout': resolve_llm_timeout(llm_cfg),
}, None
def _build_summary_diff(ctx):
"""Build the diff text to summarise. The expensive half - POST only."""
import difflib
watch = ctx['watch']
prefs = ctx['prefs']
from_version, to_version = ctx['from_version'], ctx['to_version']
def _prep(text):
"""Optionally normalise whitespace on each line before diffing."""
if not prefs.ignore_whitespace:
return text.splitlines()
return [' '.join(line.split()) for line in text.splitlines()]
def _make_unified_diff(a_text, b_text):
lines = list(difflib.unified_diff(_prep(a_text), _prep(b_text), lineterm='', n=3))
return '\n'.join(lines[2:]) if len(lines) > 2 else '\n'.join(lines)
def _apply_filters(diff_text):
"""Strip +/- lines the user has hidden in the UI so the LLM matches what they see."""
if prefs.show_removed and prefs.show_added:
return diff_text
out = []
for line in diff_text.splitlines():
if line.startswith('-') and not prefs.show_removed:
continue
if line.startswith('+') and not prefs.show_added:
continue
out.append(line)
return '\n'.join(out)
try:
from_text = watch.get_history_snapshot(timestamp=from_version)
to_text = watch.get_history_snapshot(timestamp=to_version)
except Exception as e:
return None, None, ({'summary': None, 'error': f'Could not read snapshots: {e}',
'status': 'error'}, 500)
if prefs.all_changes:
# Build sequential diffs for every intermediate snapshot between from and to
# so the LLM sees the full timeline of changes, not just start→end
sorted_dates = sorted(ctx['dates'])
try:
start_idx = sorted_dates.index(from_version)
end_idx = sorted_dates.index(to_version)
except ValueError:
start_idx, end_idx = 0, len(sorted_dates) - 1
steps = sorted_dates[start_idx:end_idx + 1]
segments = []
for i in range(len(steps) - 1):
a_ts, b_ts = steps[i], steps[i + 1]
try:
a_text = watch.get_history_snapshot(timestamp=a_ts) or ''
b_text = watch.get_history_snapshot(timestamp=b_ts) or ''
except Exception:
continue
seg = _apply_filters(_make_unified_diff(a_text, b_text))
if seg.strip():
segments.append(f'=== {a_ts} → {b_ts} ===\n{seg}')
diff_text = '\n\n'.join(segments) if segments else ''
else:
diff_text = _apply_filters(_make_unified_diff(from_text, to_text))
return diff_text, to_text, None
def _mark_viewed(uuid):
import time
datastore.set_last_viewed(uuid, int(time.time()))
# Slack on top of the model's own deadline: the worker still has to come back through
# litellm and write the summary to the on-disk cache after the last token.
SUMMARY_DEADLINE_MARGIN = 15
def _pending_reply(ctx):
"""The 202 body, carrying the point in time after which the summary is not coming.
The client cannot work this out for itself. It knows neither the configured timeout
(300s for a cloud provider, 1800s for a local endpoint) nor when the job started - which
is often long before *this* browser asked, because another tab, or this same tab before
a reload, already kicked it off. So the deadline is anchored to the job's real start and
is the same answer no matter who polls or when.
Two forms of the same instant, because neither alone is enough:
timeout_at absolute epoch seconds - stable across polls, and what to display.
expires_in seconds from now - what to actually count down with, since a browser clock
that is minutes off would otherwise expire the wait immediately.
A job still queued behind another generation has not started its timeout yet, so it
measures from now and its deadline keeps moving until a worker picks it up.
"""
import time
from changedetectionio.llm.summary_jobs import summary_jobs
started = summary_jobs.run_started_at(ctx['job_key'])
now = time.time()
timeout_at = int((started or now) + ctx['llm_timeout'] + SUMMARY_DEADLINE_MARGIN)
return {
'summary': None,
'error': None,
'status': 'pending',
'timeout_at': timeout_at,
'expires_in': max(0, timeout_at - int(now)),
}
@diff_blueprint.route("/diff/<uuid_str:uuid>/llm-summary", methods=['POST'])
@login_optionally_required
def diff_llm_summary(uuid):
"""
Start (or serve from cache) an AI summary of the diff between two snapshots.
Returns JSON:
200 {"summary": "...", "cached": true, "status": "done"} already generated
202 {"summary": null, "status": "pending", generating, poll the GET
"timeout_at": epoch, "expires_in": secs} (when to stop waiting)
4xx/5xx {"summary": null, "error": "...", "status": "error"} cannot be generated
"""
from flask import jsonify
ctx, err = _summary_request_context(uuid)
if err:
body, code = err
return jsonify(body), code
watch = ctx['watch']
from_version, to_version = ctx['from_version'], ctx['to_version']
cache_prompt, job_key = ctx['cache_prompt'], ctx['job_key']
# Check cache — keyed by version pair + prompt hash (invalidates if prompt changes)
cached = watch.get_llm_diff_summary(from_version, to_version, prompt=cache_prompt)
if cached:
_mark_viewed(uuid)
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
from changedetectionio.llm.summary_jobs import SummaryJobFailed, summary_jobs
# A failure from a previous attempt that no poll collected (e.g. the user navigated away).
# Deliver it once rather than silently starting a fresh attempt on the same broken config.
stored = summary_jobs.take_error(job_key)
if stored:
code, message = stored
return jsonify({'summary': None, 'error': message, 'status': 'error'}), code
if summary_jobs.is_pending(job_key):
logger.info(f"AI summary already in progress for {uuid} ({from_version}->{to_version}), "
f"returning pending - not starting a second generation")
_mark_viewed(uuid)
return jsonify(_pending_reply(ctx)), 202
from changedetectionio.llm.evaluator import (
LLMInputTooLargeError, get_global_token_budget_month,
is_global_token_budget_exceeded, summarise_change,
)
# Check global monthly token budget before making an LLM call
if is_global_token_budget_exceeded(datastore):
budget = get_global_token_budget_month(datastore)
llm_cfg = datastore.data.get('settings', {}).get('application', {}).get('llm', {})
used = llm_cfg.get('tokens_this_month', 0)
return jsonify({
'summary': None,
'error': gettext(
'Monthly AI token budget of %(budget)s tokens reached (%(used)s used). Resets next month.',
budget=f'{budget:,}',
used=f'{used:,}',
),
'budget_exceeded': True,
'status': 'error',
}), 429
diff_text, to_text, err = _build_summary_diff(ctx)
if err:
body, code = err
return jsonify(body), code
if not diff_text.strip():
return jsonify({'summary': None, 'error': 'No differences found', 'status': 'error'})
def _job():
"""Runs on the summary_jobs pool - no request context available in here."""
import time as _time
_started = _time.time()
logger.info(f"AI summary generation started for {uuid} ({from_version}->{to_version}), "
f"{len(diff_text)} chars of diff")
try:
summary = summarise_change(watch, datastore, diff=diff_text, current_snapshot=to_text)
except LLMInputTooLargeError as e:
raise SummaryJobFailed(str(e), http_status=400)
except Exception as e:
logger.error(f"LLM summary generation failed for {uuid}: {e}")
raise SummaryJobFailed(_clean_litellm_error(e), http_status=500)
if not summary:
raise SummaryJobFailed('LLM returned empty summary', http_status=200)
# Persisted before the job is marked done, so a poll can never see "nothing running,
# nothing cached" and start paying for the same summary a second time.
watch.save_llm_diff_summary(summary, from_version, to_version, prompt=cache_prompt)
logger.info(f"AI summary generation finished for {uuid} in "
f"{_time.time() - _started:.1f}s ({len(summary)} chars)")
# Re-read the cache. Between the read above and here we have read two snapshots and run
# difflib, which is long enough for a job started by another tab to have finished and
# written the summary - and submitting now would pay the LLM for it a second time.
cached = watch.get_llm_diff_summary(from_version, to_version, prompt=cache_prompt)
if cached:
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
def _announce(_key):
"""Tell any connected browser the job is done so it does not have to poll.
A blinker signal rather than a direct emit: the socket server subscribes to it when
realtime is enabled, and when it is disabled nothing is listening and this is a no-op.
The payload is only an identifier - the client fetches the summary through the normal
authenticated poll route, so the text never goes out over a broadcast channel.
"""
from blinker import signal
signal('llm_summary_ready').send(
watch_uuid=uuid, from_version=from_version, to_version=to_version,
)
if not summary_jobs.submit(job_key, _job, on_settled=_announce):
# Another request submitted the same key while we were building the diff. The registry
# refused ours, so nothing was sent to the LLM twice.
logger.info(f"AI summary for {uuid} was already started by a concurrent request, "
f"returning pending")
summary_jobs.submit(job_key, _job, on_settled=_announce)
_mark_viewed(uuid)
return jsonify(_pending_reply(ctx)), 202
@diff_blueprint.route("/diff/<uuid_str:uuid>/llm-summary", methods=['GET'])
@login_optionally_required
def diff_llm_summary_poll(uuid):
"""
Poll for a summary started by the POST above. Read-only: never calls the LLM.
Returns JSON:
200 {"summary": "...", "cached": true, "status": "done"} ready
202 {"summary": null, "status": "pending", still generating
"timeout_at": epoch, "expires_in": secs}
200 {"summary": null, "status": "idle"} nothing running - re-POST
4xx/5xx {"summary": null, "error": "...", "status": "error"} the job failed
"""
from flask import jsonify
ctx, err = _summary_request_context(uuid)
if err:
body, code = err
return jsonify(body), code
cached = ctx['watch'].get_llm_diff_summary(
ctx['from_version'], ctx['to_version'], prompt=ctx['cache_prompt']
)
if cached:
_mark_viewed(uuid)
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
from changedetectionio.llm.summary_jobs import summary_jobs
stored = summary_jobs.take_error(ctx['job_key'])
if stored:
code, message = stored
return jsonify({'summary': None, 'error': message, 'status': 'error'}), code
if summary_jobs.is_pending(ctx['job_key']):
return jsonify(_pending_reply(ctx)), 202
# Re-read the cache before giving up. The job can finish in the window between the read
# above and the is_pending check - it writes the summary and *then* clears the flag - so
# 'idle' here would send the client back to POST for a summary already sitting on disk.
cached = ctx['watch'].get_llm_diff_summary(
ctx['from_version'], ctx['to_version'], prompt=ctx['cache_prompt']
)
if cached:
_mark_viewed(uuid)
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
# Genuinely nothing: the process restarted mid-generation. Tell the client to start again
# rather than poll forever.
return jsonify({'summary': None, 'error': None, 'status': 'idle'})
@diff_blueprint.route("/diff/<uuid_str:uuid>/processor-data", methods=['GET'])
@login_optionally_required
def diff_history_page_processor_data(uuid):
"""
Return processor-specific JSON data for the history/diff page (e.g. the restock
price/stock timeline that the graph JS fetches).
Processor-aware: delegates to processors/{type}/difference.py::get_data(), so the
heavy data stays out of the rendered HTML (same rationale as the preview asset route).
Works for built-in and plugin processors via get_processor_submodule(). Returns 404
if the watch's processor doesn't implement get_data().
"""
from flask import jsonify, abort
if uuid == 'first':
uuid = list(datastore.data['watching'].keys()).pop()
try:
watch = datastore.data['watching'][uuid]
except KeyError:
return jsonify({'error': 'Watch not found'}), 404
processor_name = watch.get('processor', 'text_json_diff')
from changedetectionio.processors import get_processor_submodule
processor_module = get_processor_submodule(processor_name, 'difference')
if processor_module and hasattr(processor_module, 'get_data'):
return jsonify(processor_module.get_data(watch=watch, datastore=datastore, request=request))
abort(404, description=f"Processor '{processor_name}' does not provide difference data")
@diff_blueprint.route("/diff/<uuid_str:uuid>/processor-export.xlsx", methods=['GET'])
@login_optionally_required
def diff_history_page_processor_export(uuid):
"""
Download the processor's history as an .xlsx (e.g. the restock price/stock timeline).
Processor-aware: delegates to processors/{type}/difference.py::export_xlsx(), which
returns (bytes, filename). 404 if the processor doesn't implement it.
"""
from flask import make_response, abort
if uuid == 'first':
uuid = list(datastore.data['watching'].keys()).pop()
try:
watch = datastore.data['watching'][uuid]
except KeyError:
flash(gettext("No history found for the specified link, bad link?"), "error")
return redirect(url_for('watchlist.index'))
processor_name = watch.get('processor', 'text_json_diff')
from changedetectionio.processors import get_processor_submodule
processor_module = get_processor_submodule(processor_name, 'difference')
if processor_module and hasattr(processor_module, 'export_xlsx'):
data, filename = processor_module.export_xlsx(watch=watch, datastore=datastore)
resp = make_response(data)
resp.headers['Content-Type'] = 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet'
resp.headers['Content-Disposition'] = f'attachment; filename="{filename}"'
return resp
abort(404, description=f"Processor '{processor_name}' does not support xlsx export")
@diff_blueprint.route("/diff/<uuid_str:uuid>/extract", methods=['GET'])
@login_optionally_required
def diff_history_page_extract_GET(uuid):
"""
Render the data extraction form for a watch.
This route is processor-aware: it delegates to the processor's
extract.py module, allowing different processor types to provide
custom extraction interfaces.
Each processor implements processors/{type}/extract.py::render_form()
If a processor doesn't have an extract module, falls back to text_json_diff.
"""
if uuid == 'first':
uuid = list(datastore.data['watching'].keys()).pop()
try:
watch = datastore.data['watching'][uuid]
except KeyError:
flash(gettext("No history found for the specified link, bad link?"), "error")
return redirect(url_for('watchlist.index'))
# Get the processor type for this watch
processor_name = watch.get('processor', 'text_json_diff')
# Try to get the processor's extract module (works for both built-in and plugin processors)
from changedetectionio.processors import get_processor_submodule
processor_module = get_processor_submodule(processor_name, 'extract')
# Call the processor's render_form() function
if processor_module and hasattr(processor_module, 'render_form'):
return processor_module.render_form(
watch=watch,
datastore=datastore,
request=request,
url_for=url_for,
render_template=render_template,
flash=flash,
redirect=redirect
)
# Fallback: if processor doesn't have extract module, use base processors.extract as default
from changedetectionio.processors.extract import render_form as default_render_form
return default_render_form(
watch=watch,
datastore=datastore,
request=request,
url_for=url_for,
render_template=render_template,
flash=flash,
redirect=redirect
)
@diff_blueprint.route("/diff/<uuid_str:uuid>/extract", methods=['POST'])
@login_optionally_required
def diff_history_page_extract_POST(uuid):
"""
Process the data extraction request.
This route is processor-aware: it delegates to the processor's
extract.py module, allowing different processor types to provide
custom extraction logic.
Each processor implements processors/{type}/extract.py::process_extraction()
If a processor doesn't have an extract module, falls back to text_json_diff.
"""
if uuid == 'first':
uuid = list(datastore.data['watching'].keys()).pop()
try:
watch = datastore.data['watching'][uuid]
except KeyError:
flash(gettext("No history found for the specified link, bad link?"), "error")
return redirect(url_for('watchlist.index'))
# Get the processor type for this watch
processor_name = watch.get('processor', 'text_json_diff')
# Try to get the processor's extract module (works for both built-in and plugin processors)
from changedetectionio.processors import get_processor_submodule
processor_module = get_processor_submodule(processor_name, 'extract')
# Call the processor's process_extraction() function
if processor_module and hasattr(processor_module, 'process_extraction'):
return processor_module.process_extraction(
watch=watch,
datastore=datastore,
request=request,
url_for=url_for,
make_response=make_response,
send_from_directory=send_from_directory,
flash=flash,
redirect=redirect
)
# Fallback: if processor doesn't have extract module, use base processors.extract as default
from changedetectionio.processors.extract import process_extraction as default_process_extraction
return default_process_extraction(
watch=watch,
datastore=datastore,
request=request,
url_for=url_for,
make_response=make_response,
send_from_directory=send_from_directory,
flash=flash,
redirect=redirect
)
@diff_blueprint.route("/diff/<uuid_str:uuid>/download-patch", methods=['GET'])
@login_optionally_required
def download_patch(uuid):
"""
Generate and return a unified diff patch file between two snapshots.
Query params: from_version, to_version (timestamp strings from watch history).
Returns the patch as a downloadable .patch file — the same content fed to the LLM.
"""
import difflib
try:
watch = datastore.data['watching'][uuid]
except KeyError:
return plaintext_response('Watch not found', 404)
dates = list(watch.history.keys())
if len(dates) < 2:
return plaintext_response('Not enough history', 400)
from_version = request.args.get('from_version', dates[-2])
to_version = request.args.get('to_version', dates[-1])
# Validate before use. get_history_snapshot() does a plain dict lookup, so an
# unknown key raises KeyError whose str() carries the raw query parameter - which
# the error path below used to reflect into a text/html response, giving reflected
# XSS (GHSA-23mp-8222-96fr). This mirrors the check the API resource already
# performs for the same operation in api/Watch.py.
for timestamp in (from_version, to_version):
if timestamp not in watch.history:
return plaintext_response('Snapshot not found', 404)
try:
from_text = watch.get_history_snapshot(timestamp=from_version)
to_text = watch.get_history_snapshot(timestamp=to_version)
except Exception as e:
# Never reflect the exception text. Besides the KeyError guarded above, this
# also catches brotli decode failures and the data-dir containment guard in
# get_history_snapshot(), whose messages carry filesystem paths. Send the
# detail to the log and keep the response text/plain regardless, matching the
# pattern already applied to rss/single_watch.py, so nothing that reaches the
# body can be parsed as HTML by the browser.
logger.error(f"download-patch: could not read snapshots for watch {uuid}: {e}")
return plaintext_response('Could not read snapshots', 500)
diff_lines = list(difflib.unified_diff(
from_text.splitlines(keepends=True),
to_text.splitlines(keepends=True),
fromfile=f'snapshot-{from_version}',
tofile=f'snapshot-{to_version}',
lineterm='',
))
patch_text = ''.join(diff_lines) if diff_lines else '(no differences)\n'
response = make_response(patch_text)
response.headers['Content-Type'] = 'text/plain; charset=utf-8'
return response
@diff_blueprint.route("/diff/<uuid_str:uuid>/processor-asset/<string:asset_name>", methods=['GET'])
@login_optionally_required
def processor_asset(uuid, asset_name):
"""
Serve processor-specific binary assets (images, files, etc.).
This route is processor-aware: it delegates to the processor's
difference.py module, allowing different processor types to serve
custom assets without embedding them as base64 in templates.
This solves memory issues with large binary data (e.g., screenshots)
by streaming them as separate HTTP responses instead of embedding
in the HTML template.
Each processor implements processors/{type}/difference.py::get_asset()
which returns (binary_data, content_type, cache_control_header).
Example URLs:
- /diff/{uuid}/processor-asset/before
- /diff/{uuid}/processor-asset/after
- /diff/{uuid}/processor-asset/rendered_diff
"""
if uuid == 'first':
uuid = list(datastore.data['watching'].keys()).pop()
try:
watch = datastore.data['watching'][uuid]
except KeyError:
flash(gettext("No history found for the specified link, bad link?"), "error")
return redirect(url_for('watchlist.index'))
# Get the processor type for this watch
processor_name = watch.get('processor', 'text_json_diff')
# Try to get the processor's difference module (works for both built-in and plugin processors)
from changedetectionio.processors import get_processor_submodule
processor_module = get_processor_submodule(processor_name, 'difference')
# Call the processor's get_asset() function
if processor_module and hasattr(processor_module, 'get_asset'):
result = processor_module.get_asset(
asset_name=asset_name,
watch=watch,
datastore=datastore,
request=request
)
if result is None:
from flask import abort
abort(404, description=f"Asset '{asset_name}' not found")
binary_data, content_type, cache_control = result
response = make_response(binary_data)
response.headers['Content-Type'] = content_type
if cache_control:
response.headers['Cache-Control'] = cache_control
return response
else:
logger.warning(f"Processor {processor_name} does not implement get_asset()")
from flask import abort
abort(404, description=f"Processor '{processor_name}' does not support assets")
return diff_blueprint