mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-09-28 16:26:42 +00:00
830 lines
38 KiB
Python
830 lines
38 KiB
Python
from flask import Blueprint, request, redirect, url_for, flash, render_template, make_response, send_from_directory
|
|
from flask_babel import gettext
|
|
|
|
import re
|
|
import importlib
|
|
from loguru import logger
|
|
from markupsafe import Markup
|
|
|
|
from changedetectionio.diff import (
|
|
REMOVED_STYLE, ADDED_STYLE, REMOVED_INNER_STYLE, ADDED_INNER_STYLE,
|
|
REMOVED_PLACEMARKER_OPEN, REMOVED_PLACEMARKER_CLOSED,
|
|
ADDED_PLACEMARKER_OPEN, ADDED_PLACEMARKER_CLOSED,
|
|
CHANGED_PLACEMARKER_OPEN, CHANGED_PLACEMARKER_CLOSED,
|
|
CHANGED_INTO_PLACEMARKER_OPEN, CHANGED_INTO_PLACEMARKER_CLOSED
|
|
)
|
|
from changedetectionio.store import ChangeDetectionStore
|
|
from changedetectionio.blueprint import plaintext_response
|
|
from changedetectionio.auth_decorator import login_optionally_required
|
|
|
|
|
|
def _clean_litellm_error(exc) -> str:
|
|
"""Return a short, human-readable error string from a litellm exception.
|
|
|
|
litellm embeds the raw provider JSON in str(exc), which can be hundreds of
|
|
characters of verbose quota detail. We try to pull just the provider's
|
|
'message' field; failing that we return the first non-empty line with the
|
|
'litellm.XxxError:' class prefix stripped.
|
|
"""
|
|
import json, re
|
|
raw = str(exc)
|
|
# Try to parse the embedded JSON block (starts at first '{')
|
|
brace = raw.find('{')
|
|
if brace >= 0:
|
|
try:
|
|
payload = json.loads(raw[brace:])
|
|
msg = (payload.get('error') or {}).get('message') or ''
|
|
if msg:
|
|
# Take only the first sentence / line — provider messages can be long
|
|
return msg.split('\n')[0].split('. ')[0].strip() + '.'
|
|
except Exception:
|
|
pass
|
|
# Fallback: strip the "litellm.XxxError: litellm.XxxError: providerException - " prefix
|
|
first_line = raw.split('\n')[0]
|
|
first_line = re.sub(r'^(litellm\.\w+:\s*)+', '', first_line)
|
|
first_line = re.sub(r'\w+Exception\s*-\s*', '', first_line).strip()
|
|
return first_line or raw.split('\n')[0]
|
|
|
|
|
|
def construct_blueprint(datastore: ChangeDetectionStore):
|
|
diff_blueprint = Blueprint('ui_diff', __name__, template_folder="../ui/templates")
|
|
|
|
@diff_blueprint.app_template_filter('diff_unescape_difference_spans')
|
|
def diff_unescape_difference_spans(content):
|
|
"""Emulate Jinja2's auto-escape, then selectively unescape our diff spans."""
|
|
from markupsafe import escape
|
|
|
|
if not content:
|
|
return Markup('')
|
|
|
|
# Step 1: Escape everything like Jinja2 would (this makes it XSS-safe)
|
|
escaped_content = escape(str(content))
|
|
|
|
# Step 2: Unescape only our exact diff spans generated by apply_html_color_to_body()
|
|
# Pattern matches the exact structure:
|
|
# <span style="{STYLE}" role="{ROLE}" aria-label="{LABEL}" title="{TITLE}">
|
|
|
|
# Unescape outer span opening tags with full attributes (role, aria-label, title)
|
|
# Matches removed/added/changed/changed_into spans
|
|
result = re.sub(
|
|
rf'<span style="({re.escape(REMOVED_STYLE)}|{re.escape(ADDED_STYLE)})" '
|
|
rf'role="(deletion|insertion|note)" '
|
|
rf'aria-label="([^&]+?)" '
|
|
rf'title="([^&]+?)">',
|
|
r'<span style="\1" role="\2" aria-label="\3" title="\4">',
|
|
str(escaped_content),
|
|
flags=re.IGNORECASE
|
|
)
|
|
|
|
# Unescape inner span opening tags (without additional attributes)
|
|
# This matches the darker background styles for changed parts within lines
|
|
result = re.sub(
|
|
rf'<span style="({re.escape(REMOVED_INNER_STYLE)}|{re.escape(ADDED_INNER_STYLE)})">',
|
|
r'<span style="\1">',
|
|
result,
|
|
flags=re.IGNORECASE
|
|
)
|
|
|
|
# Unescape closing tags (but only as many as we opened)
|
|
open_count = result.count('<span style=')
|
|
close_count = str(escaped_content).count('</span>')
|
|
|
|
# Replace up to the number of spans we opened
|
|
for _ in range(min(open_count, close_count)):
|
|
result = result.replace('</span>', '</span>', 1)
|
|
|
|
return Markup(result)
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>", methods=['GET'])
|
|
@login_optionally_required
|
|
def diff_history_page(uuid):
|
|
"""
|
|
Render the history/diff page for a watch.
|
|
|
|
This route is processor-aware: it delegates rendering to the processor's
|
|
difference.py module, allowing different processor types to provide
|
|
custom visualizations:
|
|
- text_json_diff: Text/HTML diff with syntax highlighting
|
|
- restock_diff: Could show price charts and stock history
|
|
- image_diff: Could show image comparison slider/overlay
|
|
|
|
Each processor implements processors/{type}/difference.py::render()
|
|
If a processor doesn't have a difference module, falls back to text_json_diff.
|
|
"""
|
|
|
|
if uuid == 'first':
|
|
uuid = list(datastore.data['watching'].keys()).pop()
|
|
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
flash(gettext("No history found for the specified link, bad link?"), "error")
|
|
return redirect(url_for('watchlist.index'))
|
|
|
|
dates = list(watch.history.keys())
|
|
if not dates or len(dates) < 2:
|
|
flash(gettext("Not enough history (2 snapshots required) to show difference page for this watch."), "error")
|
|
return redirect(url_for('watchlist.index'))
|
|
|
|
# Get the processor type for this watch
|
|
processor_name = watch.get('processor', 'text_json_diff')
|
|
|
|
# Try to get the processor's difference module (works for both built-in and plugin processors)
|
|
from changedetectionio.processors import get_processor_submodule
|
|
processor_module = get_processor_submodule(processor_name, 'difference')
|
|
|
|
# Call the processor's render() function
|
|
if processor_module and hasattr(processor_module, 'render'):
|
|
return processor_module.render(
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request,
|
|
url_for=url_for,
|
|
render_template=render_template,
|
|
flash=flash,
|
|
redirect=redirect
|
|
)
|
|
|
|
# Fallback: if processor doesn't have difference module, use text_json_diff as default
|
|
from changedetectionio.processors.text_json_diff.difference import render as default_render
|
|
return default_render(
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request,
|
|
url_for=url_for,
|
|
render_template=render_template,
|
|
flash=flash,
|
|
redirect=redirect
|
|
)
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/llm-summary/prompt", methods=['GET'])
|
|
@login_optionally_required
|
|
def diff_llm_summary_prompt(uuid):
|
|
"""Return the effective LLM summary prompt for a watch immediately (no LLM call)."""
|
|
from flask import jsonify
|
|
watch = datastore.data['watching'].get(uuid)
|
|
if not watch:
|
|
return jsonify({'prompt': ''}), 404
|
|
try:
|
|
from changedetectionio.llm.evaluator import get_effective_summary_prompt
|
|
prompt = get_effective_summary_prompt(watch, datastore)
|
|
except Exception:
|
|
prompt = ''
|
|
return jsonify({'prompt': prompt})
|
|
|
|
# ── AI change summary ────────────────────────────────────────────────────────────────────
|
|
# Generation is slow (an LLM round-trip), so it does not happen in the request thread:
|
|
#
|
|
# POST /diff/<uuid>/llm-summary start it (or return an already-cached summary)
|
|
# GET /diff/<uuid>/llm-summary poll for the result
|
|
#
|
|
# POST is deliberate: starting a generation spends tokens and marks the watch viewed, which
|
|
# must not be reachable from a bare cross-site GET (an <img src=...> would do it). Flask-WTF's
|
|
# CSRFProtect covers every non-GET route, and static/js/csrf.js puts the token on the header
|
|
# of every jQuery non-GET call, so the browser side needs nothing extra.
|
|
#
|
|
# The GET poll stays side-effect free: it reads the cache and the in-flight registry and never
|
|
# starts work, which is what makes it safe to leave as a GET.
|
|
|
|
def _summary_request_context(uuid):
|
|
"""Resolve everything both routes need, without reading snapshots or building a diff.
|
|
|
|
Returns (context_dict, None) or (None, (json_body, http_status)) when the request cannot
|
|
be served at all. Kept cheap so the poll route stays cheap: the cache key does not depend
|
|
on the diff text, only on the version pair, the effective prompt, the diff prefs and the
|
|
model - so a poll never has to re-run difflib over two snapshots.
|
|
"""
|
|
from changedetectionio.llm.evaluator import (
|
|
DiffPrefs, build_summary_cache_prompt, get_effective_summary_prompt, get_llm_settings,
|
|
resolve_llm_timeout,
|
|
)
|
|
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
return None, ({'summary': None, 'error': 'Watch not found', 'status': 'error'}, 404)
|
|
|
|
llm_cfg = datastore.data.get('settings', {}).get('application', {}).get('llm', {})
|
|
if not llm_cfg.get('model'):
|
|
return None, ({'summary': None, 'error': 'LLM not configured', 'status': 'error'}, 400)
|
|
|
|
dates = list(watch.history.keys())
|
|
if len(dates) < 2:
|
|
return None, ({'summary': None, 'error': 'Not enough history', 'status': 'error'}, 400)
|
|
|
|
# Default baseline for the watchlist "Summary" link (when no explicit from_version
|
|
# is requested) is configurable at Settings > AI. Default 'second_last_version'
|
|
# compares the previous snapshot; 'since_last_viewed' uses the operator's last view.
|
|
if llm_cfg.get('watchlist_overview_summary', 'second_last_version') == 'since_last_viewed':
|
|
best_from = watch.get_from_version_based_on_last_viewed
|
|
default_from = best_from if best_from else dates[-2]
|
|
else:
|
|
default_from = dates[-2]
|
|
|
|
prefs = DiffPrefs.from_request_args(request.args)
|
|
from_version = request.args.get('from_version', default_from)
|
|
to_version = request.args.get('to_version', dates[-1])
|
|
|
|
settings = get_llm_settings(datastore)
|
|
# Diff-pref flags + system prompt + active model are part of the cache key
|
|
# so prompt or model changes bust the cache.
|
|
cache_prompt = build_summary_cache_prompt(
|
|
effective_prompt=get_effective_summary_prompt(watch, datastore),
|
|
max_summary_tokens=settings.max_summary_tokens,
|
|
prefs=prefs,
|
|
model=settings.model,
|
|
)
|
|
|
|
return {
|
|
'watch': watch,
|
|
'dates': dates,
|
|
'prefs': prefs,
|
|
'from_version': from_version,
|
|
'to_version': to_version,
|
|
'cache_prompt': cache_prompt,
|
|
# Same granularity as the on-disk cache filename (version pair + prompt hash), so an
|
|
# in-flight job is deduplicated exactly when it would write to the same cache entry.
|
|
'job_key': f"{uuid}|{from_version}|{to_version}|"
|
|
f"{watch._llm_summary_prompt_hash(cache_prompt)}",
|
|
# How long this deployment is prepared to wait for the model on one call. The
|
|
# browser must not give up before this, or it reports a failure for a job that is
|
|
# still running and will still write its summary - a local Ollama/vLLM endpoint
|
|
# gets 1800s here against the client's old fixed 180s. See _pending_reply().
|
|
'llm_timeout': resolve_llm_timeout(llm_cfg),
|
|
}, None
|
|
|
|
def _build_summary_diff(ctx):
|
|
"""Build the diff text to summarise. The expensive half - POST only."""
|
|
import difflib
|
|
|
|
watch = ctx['watch']
|
|
prefs = ctx['prefs']
|
|
from_version, to_version = ctx['from_version'], ctx['to_version']
|
|
|
|
def _prep(text):
|
|
"""Optionally normalise whitespace on each line before diffing."""
|
|
if not prefs.ignore_whitespace:
|
|
return text.splitlines()
|
|
return [' '.join(line.split()) for line in text.splitlines()]
|
|
|
|
def _make_unified_diff(a_text, b_text):
|
|
lines = list(difflib.unified_diff(_prep(a_text), _prep(b_text), lineterm='', n=3))
|
|
return '\n'.join(lines[2:]) if len(lines) > 2 else '\n'.join(lines)
|
|
|
|
def _apply_filters(diff_text):
|
|
"""Strip +/- lines the user has hidden in the UI so the LLM matches what they see."""
|
|
if prefs.show_removed and prefs.show_added:
|
|
return diff_text
|
|
out = []
|
|
for line in diff_text.splitlines():
|
|
if line.startswith('-') and not prefs.show_removed:
|
|
continue
|
|
if line.startswith('+') and not prefs.show_added:
|
|
continue
|
|
out.append(line)
|
|
return '\n'.join(out)
|
|
|
|
try:
|
|
from_text = watch.get_history_snapshot(timestamp=from_version)
|
|
to_text = watch.get_history_snapshot(timestamp=to_version)
|
|
except Exception as e:
|
|
return None, None, ({'summary': None, 'error': f'Could not read snapshots: {e}',
|
|
'status': 'error'}, 500)
|
|
|
|
if prefs.all_changes:
|
|
# Build sequential diffs for every intermediate snapshot between from and to
|
|
# so the LLM sees the full timeline of changes, not just start→end
|
|
sorted_dates = sorted(ctx['dates'])
|
|
try:
|
|
start_idx = sorted_dates.index(from_version)
|
|
end_idx = sorted_dates.index(to_version)
|
|
except ValueError:
|
|
start_idx, end_idx = 0, len(sorted_dates) - 1
|
|
|
|
steps = sorted_dates[start_idx:end_idx + 1]
|
|
segments = []
|
|
for i in range(len(steps) - 1):
|
|
a_ts, b_ts = steps[i], steps[i + 1]
|
|
try:
|
|
a_text = watch.get_history_snapshot(timestamp=a_ts) or ''
|
|
b_text = watch.get_history_snapshot(timestamp=b_ts) or ''
|
|
except Exception:
|
|
continue
|
|
seg = _apply_filters(_make_unified_diff(a_text, b_text))
|
|
if seg.strip():
|
|
segments.append(f'=== {a_ts} → {b_ts} ===\n{seg}')
|
|
|
|
diff_text = '\n\n'.join(segments) if segments else ''
|
|
else:
|
|
diff_text = _apply_filters(_make_unified_diff(from_text, to_text))
|
|
|
|
return diff_text, to_text, None
|
|
|
|
def _mark_viewed(uuid):
|
|
import time
|
|
datastore.set_last_viewed(uuid, int(time.time()))
|
|
|
|
# Slack on top of the model's own deadline: the worker still has to come back through
|
|
# litellm and write the summary to the on-disk cache after the last token.
|
|
SUMMARY_DEADLINE_MARGIN = 15
|
|
|
|
def _pending_reply(ctx):
|
|
"""The 202 body, carrying the point in time after which the summary is not coming.
|
|
|
|
The client cannot work this out for itself. It knows neither the configured timeout
|
|
(300s for a cloud provider, 1800s for a local endpoint) nor when the job started - which
|
|
is often long before *this* browser asked, because another tab, or this same tab before
|
|
a reload, already kicked it off. So the deadline is anchored to the job's real start and
|
|
is the same answer no matter who polls or when.
|
|
|
|
Two forms of the same instant, because neither alone is enough:
|
|
timeout_at absolute epoch seconds - stable across polls, and what to display.
|
|
expires_in seconds from now - what to actually count down with, since a browser clock
|
|
that is minutes off would otherwise expire the wait immediately.
|
|
A job still queued behind another generation has not started its timeout yet, so it
|
|
measures from now and its deadline keeps moving until a worker picks it up.
|
|
"""
|
|
import time
|
|
from changedetectionio.llm.summary_jobs import summary_jobs
|
|
|
|
started = summary_jobs.run_started_at(ctx['job_key'])
|
|
now = time.time()
|
|
timeout_at = int((started or now) + ctx['llm_timeout'] + SUMMARY_DEADLINE_MARGIN)
|
|
return {
|
|
'summary': None,
|
|
'error': None,
|
|
'status': 'pending',
|
|
'timeout_at': timeout_at,
|
|
'expires_in': max(0, timeout_at - int(now)),
|
|
}
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/llm-summary", methods=['POST'])
|
|
@login_optionally_required
|
|
def diff_llm_summary(uuid):
|
|
"""
|
|
Start (or serve from cache) an AI summary of the diff between two snapshots.
|
|
|
|
Returns JSON:
|
|
200 {"summary": "...", "cached": true, "status": "done"} already generated
|
|
202 {"summary": null, "status": "pending", generating, poll the GET
|
|
"timeout_at": epoch, "expires_in": secs} (when to stop waiting)
|
|
4xx/5xx {"summary": null, "error": "...", "status": "error"} cannot be generated
|
|
"""
|
|
from flask import jsonify
|
|
|
|
ctx, err = _summary_request_context(uuid)
|
|
if err:
|
|
body, code = err
|
|
return jsonify(body), code
|
|
|
|
watch = ctx['watch']
|
|
from_version, to_version = ctx['from_version'], ctx['to_version']
|
|
cache_prompt, job_key = ctx['cache_prompt'], ctx['job_key']
|
|
|
|
# Check cache — keyed by version pair + prompt hash (invalidates if prompt changes)
|
|
cached = watch.get_llm_diff_summary(from_version, to_version, prompt=cache_prompt)
|
|
if cached:
|
|
_mark_viewed(uuid)
|
|
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
|
|
|
|
from changedetectionio.llm.summary_jobs import SummaryJobFailed, summary_jobs
|
|
|
|
# A failure from a previous attempt that no poll collected (e.g. the user navigated away).
|
|
# Deliver it once rather than silently starting a fresh attempt on the same broken config.
|
|
stored = summary_jobs.take_error(job_key)
|
|
if stored:
|
|
code, message = stored
|
|
return jsonify({'summary': None, 'error': message, 'status': 'error'}), code
|
|
|
|
if summary_jobs.is_pending(job_key):
|
|
logger.info(f"AI summary already in progress for {uuid} ({from_version}->{to_version}), "
|
|
f"returning pending - not starting a second generation")
|
|
_mark_viewed(uuid)
|
|
return jsonify(_pending_reply(ctx)), 202
|
|
|
|
from changedetectionio.llm.evaluator import (
|
|
LLMInputTooLargeError, get_global_token_budget_month,
|
|
is_global_token_budget_exceeded, summarise_change,
|
|
)
|
|
|
|
# Check global monthly token budget before making an LLM call
|
|
if is_global_token_budget_exceeded(datastore):
|
|
budget = get_global_token_budget_month(datastore)
|
|
llm_cfg = datastore.data.get('settings', {}).get('application', {}).get('llm', {})
|
|
used = llm_cfg.get('tokens_this_month', 0)
|
|
return jsonify({
|
|
'summary': None,
|
|
'error': gettext(
|
|
'Monthly AI token budget of %(budget)s tokens reached (%(used)s used). Resets next month.',
|
|
budget=f'{budget:,}',
|
|
used=f'{used:,}',
|
|
),
|
|
'budget_exceeded': True,
|
|
'status': 'error',
|
|
}), 429
|
|
|
|
diff_text, to_text, err = _build_summary_diff(ctx)
|
|
if err:
|
|
body, code = err
|
|
return jsonify(body), code
|
|
|
|
if not diff_text.strip():
|
|
return jsonify({'summary': None, 'error': 'No differences found', 'status': 'error'})
|
|
|
|
def _job():
|
|
"""Runs on the summary_jobs pool - no request context available in here."""
|
|
import time as _time
|
|
_started = _time.time()
|
|
logger.info(f"AI summary generation started for {uuid} ({from_version}->{to_version}), "
|
|
f"{len(diff_text)} chars of diff")
|
|
try:
|
|
summary = summarise_change(watch, datastore, diff=diff_text, current_snapshot=to_text)
|
|
except LLMInputTooLargeError as e:
|
|
raise SummaryJobFailed(str(e), http_status=400)
|
|
except Exception as e:
|
|
logger.error(f"LLM summary generation failed for {uuid}: {e}")
|
|
raise SummaryJobFailed(_clean_litellm_error(e), http_status=500)
|
|
|
|
if not summary:
|
|
raise SummaryJobFailed('LLM returned empty summary', http_status=200)
|
|
|
|
# Persisted before the job is marked done, so a poll can never see "nothing running,
|
|
# nothing cached" and start paying for the same summary a second time.
|
|
watch.save_llm_diff_summary(summary, from_version, to_version, prompt=cache_prompt)
|
|
logger.info(f"AI summary generation finished for {uuid} in "
|
|
f"{_time.time() - _started:.1f}s ({len(summary)} chars)")
|
|
|
|
# Re-read the cache. Between the read above and here we have read two snapshots and run
|
|
# difflib, which is long enough for a job started by another tab to have finished and
|
|
# written the summary - and submitting now would pay the LLM for it a second time.
|
|
cached = watch.get_llm_diff_summary(from_version, to_version, prompt=cache_prompt)
|
|
if cached:
|
|
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
|
|
|
|
def _announce(_key):
|
|
"""Tell any connected browser the job is done so it does not have to poll.
|
|
|
|
A blinker signal rather than a direct emit: the socket server subscribes to it when
|
|
realtime is enabled, and when it is disabled nothing is listening and this is a no-op.
|
|
The payload is only an identifier - the client fetches the summary through the normal
|
|
authenticated poll route, so the text never goes out over a broadcast channel.
|
|
"""
|
|
from blinker import signal
|
|
signal('llm_summary_ready').send(
|
|
watch_uuid=uuid, from_version=from_version, to_version=to_version,
|
|
)
|
|
|
|
if not summary_jobs.submit(job_key, _job, on_settled=_announce):
|
|
# Another request submitted the same key while we were building the diff. The registry
|
|
# refused ours, so nothing was sent to the LLM twice.
|
|
logger.info(f"AI summary for {uuid} was already started by a concurrent request, "
|
|
f"returning pending")
|
|
summary_jobs.submit(job_key, _job, on_settled=_announce)
|
|
_mark_viewed(uuid)
|
|
return jsonify(_pending_reply(ctx)), 202
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/llm-summary", methods=['GET'])
|
|
@login_optionally_required
|
|
def diff_llm_summary_poll(uuid):
|
|
"""
|
|
Poll for a summary started by the POST above. Read-only: never calls the LLM.
|
|
|
|
Returns JSON:
|
|
200 {"summary": "...", "cached": true, "status": "done"} ready
|
|
202 {"summary": null, "status": "pending", still generating
|
|
"timeout_at": epoch, "expires_in": secs}
|
|
200 {"summary": null, "status": "idle"} nothing running - re-POST
|
|
4xx/5xx {"summary": null, "error": "...", "status": "error"} the job failed
|
|
"""
|
|
from flask import jsonify
|
|
|
|
ctx, err = _summary_request_context(uuid)
|
|
if err:
|
|
body, code = err
|
|
return jsonify(body), code
|
|
|
|
cached = ctx['watch'].get_llm_diff_summary(
|
|
ctx['from_version'], ctx['to_version'], prompt=ctx['cache_prompt']
|
|
)
|
|
if cached:
|
|
_mark_viewed(uuid)
|
|
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
|
|
|
|
from changedetectionio.llm.summary_jobs import summary_jobs
|
|
|
|
stored = summary_jobs.take_error(ctx['job_key'])
|
|
if stored:
|
|
code, message = stored
|
|
return jsonify({'summary': None, 'error': message, 'status': 'error'}), code
|
|
|
|
if summary_jobs.is_pending(ctx['job_key']):
|
|
return jsonify(_pending_reply(ctx)), 202
|
|
|
|
# Re-read the cache before giving up. The job can finish in the window between the read
|
|
# above and the is_pending check - it writes the summary and *then* clears the flag - so
|
|
# 'idle' here would send the client back to POST for a summary already sitting on disk.
|
|
cached = ctx['watch'].get_llm_diff_summary(
|
|
ctx['from_version'], ctx['to_version'], prompt=ctx['cache_prompt']
|
|
)
|
|
if cached:
|
|
_mark_viewed(uuid)
|
|
return jsonify({'summary': cached, 'error': None, 'cached': True, 'status': 'done'})
|
|
|
|
# Genuinely nothing: the process restarted mid-generation. Tell the client to start again
|
|
# rather than poll forever.
|
|
return jsonify({'summary': None, 'error': None, 'status': 'idle'})
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/processor-data", methods=['GET'])
|
|
@login_optionally_required
|
|
def diff_history_page_processor_data(uuid):
|
|
"""
|
|
Return processor-specific JSON data for the history/diff page (e.g. the restock
|
|
price/stock timeline that the graph JS fetches).
|
|
|
|
Processor-aware: delegates to processors/{type}/difference.py::get_data(), so the
|
|
heavy data stays out of the rendered HTML (same rationale as the preview asset route).
|
|
Works for built-in and plugin processors via get_processor_submodule(). Returns 404
|
|
if the watch's processor doesn't implement get_data().
|
|
"""
|
|
from flask import jsonify, abort
|
|
|
|
if uuid == 'first':
|
|
uuid = list(datastore.data['watching'].keys()).pop()
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
return jsonify({'error': 'Watch not found'}), 404
|
|
|
|
processor_name = watch.get('processor', 'text_json_diff')
|
|
from changedetectionio.processors import get_processor_submodule
|
|
processor_module = get_processor_submodule(processor_name, 'difference')
|
|
|
|
if processor_module and hasattr(processor_module, 'get_data'):
|
|
return jsonify(processor_module.get_data(watch=watch, datastore=datastore, request=request))
|
|
|
|
abort(404, description=f"Processor '{processor_name}' does not provide difference data")
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/processor-export.xlsx", methods=['GET'])
|
|
@login_optionally_required
|
|
def diff_history_page_processor_export(uuid):
|
|
"""
|
|
Download the processor's history as an .xlsx (e.g. the restock price/stock timeline).
|
|
Processor-aware: delegates to processors/{type}/difference.py::export_xlsx(), which
|
|
returns (bytes, filename). 404 if the processor doesn't implement it.
|
|
"""
|
|
from flask import make_response, abort
|
|
|
|
if uuid == 'first':
|
|
uuid = list(datastore.data['watching'].keys()).pop()
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
flash(gettext("No history found for the specified link, bad link?"), "error")
|
|
return redirect(url_for('watchlist.index'))
|
|
|
|
processor_name = watch.get('processor', 'text_json_diff')
|
|
from changedetectionio.processors import get_processor_submodule
|
|
processor_module = get_processor_submodule(processor_name, 'difference')
|
|
|
|
if processor_module and hasattr(processor_module, 'export_xlsx'):
|
|
data, filename = processor_module.export_xlsx(watch=watch, datastore=datastore)
|
|
resp = make_response(data)
|
|
resp.headers['Content-Type'] = 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet'
|
|
resp.headers['Content-Disposition'] = f'attachment; filename="{filename}"'
|
|
return resp
|
|
|
|
abort(404, description=f"Processor '{processor_name}' does not support xlsx export")
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/extract", methods=['GET'])
|
|
@login_optionally_required
|
|
def diff_history_page_extract_GET(uuid):
|
|
"""
|
|
Render the data extraction form for a watch.
|
|
|
|
This route is processor-aware: it delegates to the processor's
|
|
extract.py module, allowing different processor types to provide
|
|
custom extraction interfaces.
|
|
|
|
Each processor implements processors/{type}/extract.py::render_form()
|
|
If a processor doesn't have an extract module, falls back to text_json_diff.
|
|
"""
|
|
|
|
|
|
if uuid == 'first':
|
|
uuid = list(datastore.data['watching'].keys()).pop()
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
flash(gettext("No history found for the specified link, bad link?"), "error")
|
|
return redirect(url_for('watchlist.index'))
|
|
|
|
# Get the processor type for this watch
|
|
processor_name = watch.get('processor', 'text_json_diff')
|
|
|
|
# Try to get the processor's extract module (works for both built-in and plugin processors)
|
|
from changedetectionio.processors import get_processor_submodule
|
|
processor_module = get_processor_submodule(processor_name, 'extract')
|
|
|
|
# Call the processor's render_form() function
|
|
if processor_module and hasattr(processor_module, 'render_form'):
|
|
return processor_module.render_form(
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request,
|
|
url_for=url_for,
|
|
render_template=render_template,
|
|
flash=flash,
|
|
redirect=redirect
|
|
)
|
|
|
|
# Fallback: if processor doesn't have extract module, use base processors.extract as default
|
|
from changedetectionio.processors.extract import render_form as default_render_form
|
|
return default_render_form(
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request,
|
|
url_for=url_for,
|
|
render_template=render_template,
|
|
flash=flash,
|
|
redirect=redirect
|
|
)
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/extract", methods=['POST'])
|
|
@login_optionally_required
|
|
def diff_history_page_extract_POST(uuid):
|
|
"""
|
|
Process the data extraction request.
|
|
|
|
This route is processor-aware: it delegates to the processor's
|
|
extract.py module, allowing different processor types to provide
|
|
custom extraction logic.
|
|
|
|
Each processor implements processors/{type}/extract.py::process_extraction()
|
|
If a processor doesn't have an extract module, falls back to text_json_diff.
|
|
"""
|
|
|
|
if uuid == 'first':
|
|
uuid = list(datastore.data['watching'].keys()).pop()
|
|
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
flash(gettext("No history found for the specified link, bad link?"), "error")
|
|
return redirect(url_for('watchlist.index'))
|
|
|
|
# Get the processor type for this watch
|
|
processor_name = watch.get('processor', 'text_json_diff')
|
|
|
|
# Try to get the processor's extract module (works for both built-in and plugin processors)
|
|
from changedetectionio.processors import get_processor_submodule
|
|
processor_module = get_processor_submodule(processor_name, 'extract')
|
|
|
|
# Call the processor's process_extraction() function
|
|
if processor_module and hasattr(processor_module, 'process_extraction'):
|
|
return processor_module.process_extraction(
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request,
|
|
url_for=url_for,
|
|
make_response=make_response,
|
|
send_from_directory=send_from_directory,
|
|
flash=flash,
|
|
redirect=redirect
|
|
)
|
|
|
|
# Fallback: if processor doesn't have extract module, use base processors.extract as default
|
|
from changedetectionio.processors.extract import process_extraction as default_process_extraction
|
|
return default_process_extraction(
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request,
|
|
url_for=url_for,
|
|
make_response=make_response,
|
|
send_from_directory=send_from_directory,
|
|
flash=flash,
|
|
redirect=redirect
|
|
)
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/download-patch", methods=['GET'])
|
|
@login_optionally_required
|
|
def download_patch(uuid):
|
|
"""
|
|
Generate and return a unified diff patch file between two snapshots.
|
|
Query params: from_version, to_version (timestamp strings from watch history).
|
|
Returns the patch as a downloadable .patch file — the same content fed to the LLM.
|
|
"""
|
|
import difflib
|
|
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
return plaintext_response('Watch not found', 404)
|
|
|
|
dates = list(watch.history.keys())
|
|
if len(dates) < 2:
|
|
return plaintext_response('Not enough history', 400)
|
|
|
|
from_version = request.args.get('from_version', dates[-2])
|
|
to_version = request.args.get('to_version', dates[-1])
|
|
|
|
# Validate before use. get_history_snapshot() does a plain dict lookup, so an
|
|
# unknown key raises KeyError whose str() carries the raw query parameter - which
|
|
# the error path below used to reflect into a text/html response, giving reflected
|
|
# XSS (GHSA-23mp-8222-96fr). This mirrors the check the API resource already
|
|
# performs for the same operation in api/Watch.py.
|
|
for timestamp in (from_version, to_version):
|
|
if timestamp not in watch.history:
|
|
return plaintext_response('Snapshot not found', 404)
|
|
|
|
try:
|
|
from_text = watch.get_history_snapshot(timestamp=from_version)
|
|
to_text = watch.get_history_snapshot(timestamp=to_version)
|
|
except Exception as e:
|
|
# Never reflect the exception text. Besides the KeyError guarded above, this
|
|
# also catches brotli decode failures and the data-dir containment guard in
|
|
# get_history_snapshot(), whose messages carry filesystem paths. Send the
|
|
# detail to the log and keep the response text/plain regardless, matching the
|
|
# pattern already applied to rss/single_watch.py, so nothing that reaches the
|
|
# body can be parsed as HTML by the browser.
|
|
logger.error(f"download-patch: could not read snapshots for watch {uuid}: {e}")
|
|
return plaintext_response('Could not read snapshots', 500)
|
|
|
|
diff_lines = list(difflib.unified_diff(
|
|
from_text.splitlines(keepends=True),
|
|
to_text.splitlines(keepends=True),
|
|
fromfile=f'snapshot-{from_version}',
|
|
tofile=f'snapshot-{to_version}',
|
|
lineterm='',
|
|
))
|
|
patch_text = ''.join(diff_lines) if diff_lines else '(no differences)\n'
|
|
|
|
response = make_response(patch_text)
|
|
response.headers['Content-Type'] = 'text/plain; charset=utf-8'
|
|
return response
|
|
|
|
@diff_blueprint.route("/diff/<uuid_str:uuid>/processor-asset/<string:asset_name>", methods=['GET'])
|
|
@login_optionally_required
|
|
def processor_asset(uuid, asset_name):
|
|
"""
|
|
Serve processor-specific binary assets (images, files, etc.).
|
|
|
|
This route is processor-aware: it delegates to the processor's
|
|
difference.py module, allowing different processor types to serve
|
|
custom assets without embedding them as base64 in templates.
|
|
|
|
This solves memory issues with large binary data (e.g., screenshots)
|
|
by streaming them as separate HTTP responses instead of embedding
|
|
in the HTML template.
|
|
|
|
Each processor implements processors/{type}/difference.py::get_asset()
|
|
which returns (binary_data, content_type, cache_control_header).
|
|
|
|
Example URLs:
|
|
- /diff/{uuid}/processor-asset/before
|
|
- /diff/{uuid}/processor-asset/after
|
|
- /diff/{uuid}/processor-asset/rendered_diff
|
|
"""
|
|
|
|
if uuid == 'first':
|
|
uuid = list(datastore.data['watching'].keys()).pop()
|
|
|
|
try:
|
|
watch = datastore.data['watching'][uuid]
|
|
except KeyError:
|
|
flash(gettext("No history found for the specified link, bad link?"), "error")
|
|
return redirect(url_for('watchlist.index'))
|
|
|
|
# Get the processor type for this watch
|
|
processor_name = watch.get('processor', 'text_json_diff')
|
|
|
|
# Try to get the processor's difference module (works for both built-in and plugin processors)
|
|
from changedetectionio.processors import get_processor_submodule
|
|
processor_module = get_processor_submodule(processor_name, 'difference')
|
|
|
|
# Call the processor's get_asset() function
|
|
if processor_module and hasattr(processor_module, 'get_asset'):
|
|
result = processor_module.get_asset(
|
|
asset_name=asset_name,
|
|
watch=watch,
|
|
datastore=datastore,
|
|
request=request
|
|
)
|
|
|
|
if result is None:
|
|
from flask import abort
|
|
abort(404, description=f"Asset '{asset_name}' not found")
|
|
|
|
binary_data, content_type, cache_control = result
|
|
|
|
response = make_response(binary_data)
|
|
response.headers['Content-Type'] = content_type
|
|
if cache_control:
|
|
response.headers['Cache-Control'] = cache_control
|
|
return response
|
|
else:
|
|
logger.warning(f"Processor {processor_name} does not implement get_asset()")
|
|
from flask import abort
|
|
abort(404, description=f"Processor '{processor_name}' does not support assets")
|
|
|
|
return diff_blueprint
|