From 90d68f7ca74bb6820b4feba0a9cba5ddd67c6046 Mon Sep 17 00:00:00 2001 From: dgtlmoon Date: Wed, 19 Nov 2025 15:19:50 +0100 Subject: [PATCH] RSS encoding fixes --- .../content_fetchers/requests.py | 20 ++++++++++++++++--- changedetectionio/rss_tools.py | 2 -- 2 files changed, 17 insertions(+), 5 deletions(-) diff --git a/changedetectionio/content_fetchers/requests.py b/changedetectionio/content_fetchers/requests.py index 514883b7..f5b9d51e 100644 --- a/changedetectionio/content_fetchers/requests.py +++ b/changedetectionio/content_fetchers/requests.py @@ -1,6 +1,7 @@ from loguru import logger import hashlib import os +import re import asyncio from changedetectionio import strtobool from changedetectionio.content_fetchers.exceptions import BrowserStepsInUnsupportedFetcher, EmptyReply, Non200ErrorCodeReceived @@ -76,9 +77,22 @@ class fetcher(Fetcher): if not is_binary: # Don't run this for PDF (and requests identified as binary) takes a _long_ time if not r.headers.get('content-type') or not 'charset=' in r.headers.get('content-type'): - encoding = chardet.detect(r.content)['encoding'] - if encoding: - r.encoding = encoding + # For XML/RSS feeds, check the XML declaration for encoding attribute + # This is more reliable than chardet which can misdetect UTF-8 as MacRoman + content_type = r.headers.get('content-type', '').lower() + if 'xml' in content_type or 'rss' in content_type: + # Look for + xml_encoding_match = re.search(rb'<\?xml[^>]+encoding=["\']([^"\']+)["\']', r.content[:200]) + if xml_encoding_match: + r.encoding = xml_encoding_match.group(1).decode('ascii') + else: + # Default to UTF-8 for XML if no encoding found + r.encoding = 'utf-8' + else: + # For other content types, use chardet + encoding = chardet.detect(r.content)['encoding'] + if encoding: + r.encoding = encoding self.headers = r.headers diff --git a/changedetectionio/rss_tools.py b/changedetectionio/rss_tools.py index 80e34269..64e40d12 100644 --- a/changedetectionio/rss_tools.py +++ b/changedetectionio/rss_tools.py @@ -146,8 +146,6 @@ RSS_ENTRY_TEMPLATE = """ Summary:
{%- endif -%} {{ entry.summary | safe }} -{%- else -%} -<none> {%- endif -%} """