More non standard encoding fixes

This commit is contained in:
dgtlmoon
2026-03-05 11:31:03 +01:00
parent 8e83643c70
commit 1453119516
3 changed files with 57 additions and 7 deletions
+13 -4
View File
@@ -148,10 +148,19 @@ class fetcher(Fetcher):
# Default to UTF-8 for XML if no encoding found
r.encoding = 'utf-8'
else:
# For other content types, use chardet
encoding = chardet.detect(r.content)['encoding']
if encoding:
r.encoding = encoding
# Try UTF-8 first - the vast majority of modern pages are UTF-8.
# chardet can misdetect UTF-8 content as UTF-7 or other encodings,
# which causes surrogates/mojibake and is also slow (scans entire body).
# See: https://github.com/dgtlmoon/changedetection.io/issues/3952
try:
r.content.decode('utf-8') # try to decode, validation only
r.encoding = 'utf-8' # If it got this far, set it to utf-8
except UnicodeDecodeError:
# Not valid UTF-8, fall back to chardet
encoding = chardet.detect(r.content)['encoding']
logger.warning(f"URL: {url} Did not decode as UTF-8, got UnicodeDecodeError, guessed new encoding as '{encoding}' via chardet")
if encoding:
r.encoding = encoding
self.headers = r.headers
+2 -3
View File
@@ -266,9 +266,8 @@ class difference_detection_processor():
# Covers all fetchers (requests, playwright, puppeteer, selenium) in one place.
# See: https://github.com/dgtlmoon/changedetection.io/issues/3952
# DISABLED FOR NOW, TRY TO SEE IF THE TEST BREAKS ON GITHUB
# if self.fetcher.content and isinstance(self.fetcher.content, str):
# self.fetcher.content = self.fetcher.content.encode('utf-8', errors='replace').decode('utf-8')
if self.fetcher.content and isinstance(self.fetcher.content, str):
self.fetcher.content = self.fetcher.content.encode('utf-8', errors='replace').decode('utf-8')
# After init, call run_changedetection() which will do the actual change-detection
+42
View File
@@ -33,6 +33,48 @@ def test_surrogate_characters_in_content_are_sanitized():
hashlib.md5(sanitized.encode('utf-8')).hexdigest()
def test_utf8_content_without_charset_header(client, live_server, datastore_path):
"""Server returns UTF-8 content but no charset in Content-Type header.
chardet can misdetect such pages as UTF-7 (Python 3.14 then produces surrogates).
Our fix tries UTF-8 first before falling back to chardet.
See: https://github.com/dgtlmoon/changedetection.io/issues/3952
"""
from .util import write_test_file_and_sync
# UTF-8 encoded content with non-ASCII chars - no charset will be in the header
html = '<html><body><p>Español</p><p>Français</p><p>日本語</p></body></html>'
write_test_file_and_sync(os.path.join(datastore_path, "endpoint-content.txt"), html.encode('utf-8'), mode='wb')
test_url = url_for('test_endpoint', content_type="text/html", _external=True)
client.application.config.get('DATASTORE').add_watch(url=test_url)
client.get(url_for("ui.form_watch_checknow"), follow_redirects=True)
wait_for_all_checks(client)
res = client.get(url_for("ui.ui_preview.preview_page", uuid="first"), follow_redirects=True)
# Should decode correctly as UTF-8, not produce mojibake (Español) or replacement chars
assert 'Español'.encode('utf-8') in res.data
assert 'Français'.encode('utf-8') in res.data
assert '日本語'.encode('utf-8') in res.data
def test_shiftjis_content_without_charset_header(client, live_server, datastore_path):
"""Server returns Shift-JIS encoded content with no charset header.
UTF-8 decode will fail, so we fall back to chardet which should detect Shift-JIS.
"""
from .util import write_test_file_and_sync
japanese_text = '日本語のページ'
html = f'<html><body><p>{japanese_text}</p></body></html>'
write_test_file_and_sync(os.path.join(datastore_path, "endpoint-content.txt"), html.encode('shift_jis'), mode='wb')
test_url = url_for('test_endpoint', content_type="text/html", _external=True)
client.application.config.get('DATASTORE').add_watch(url=test_url)
client.get(url_for("ui.form_watch_checknow"), follow_redirects=True)
wait_for_all_checks(client)
res = client.get(url_for("ui.ui_preview.preview_page", uuid="first"), follow_redirects=True)
# chardet should detect Shift-JIS and decode correctly to Unicode
assert japanese_text.encode('utf-8') in res.data
def set_html_response(datastore_path):
test_return_data = """
<html><body><span class="nav_second_img_text">