mirror of
https://github.com/dgtlmoon/changedetection.io.git
synced 2026-09-24 22:36:33 +00:00
Improving UTF-8 handling for xpath selectors
This commit is contained in:
@@ -182,8 +182,10 @@ def xpath_filter(xpath_filter, html_content, append_pretty_line_formatting=False
|
||||
if is_rss:
|
||||
# So that we can keep CDATA for cdata_in_document_to_text() to process
|
||||
parser = etree.XMLParser(strip_cdata=False)
|
||||
|
||||
tree = html.fromstring(bytes(html_content, encoding='utf-8'), parser=parser)
|
||||
# For XML/RSS content, use etree.fromstring to properly handle XML declarations
|
||||
tree = etree.fromstring(html_content.encode('utf-8') if isinstance(html_content, str) else html_content, parser=parser)
|
||||
else:
|
||||
tree = html.fromstring(html_content, parser=parser)
|
||||
html_block = ""
|
||||
|
||||
# Build namespace map for XPath queries
|
||||
@@ -216,7 +218,10 @@ def xpath_filter(xpath_filter, html_content, append_pretty_line_formatting=False
|
||||
if type(element) == str:
|
||||
html_block += element
|
||||
elif issubclass(type(element), etree._Element) or issubclass(type(element), etree._ElementTree):
|
||||
html_block += etree.tostring(element, pretty_print=True).decode('utf-8')
|
||||
# Use 'xml' method for RSS/XML content, 'html' for HTML content
|
||||
# parser will be XMLParser if we detected XML content
|
||||
method = 'xml' if (is_rss or isinstance(parser, etree.XMLParser)) else 'html'
|
||||
html_block += etree.tostring(element, pretty_print=True, method=method, encoding='unicode')
|
||||
else:
|
||||
html_block += elementpath_tostring(element)
|
||||
|
||||
@@ -226,13 +231,14 @@ def xpath_filter(xpath_filter, html_content, append_pretty_line_formatting=False
|
||||
# 'xpath1:'
|
||||
def xpath1_filter(xpath_filter, html_content, append_pretty_line_formatting=False, is_rss=False):
|
||||
from lxml import etree, html
|
||||
|
||||
parser = None
|
||||
if is_rss:
|
||||
# So that we can keep CDATA for cdata_in_document_to_text() to process
|
||||
parser = etree.XMLParser(strip_cdata=False)
|
||||
|
||||
tree = html.fromstring(bytes(html_content, encoding='utf-8'), parser=parser)
|
||||
# For XML/RSS content, use etree.fromstring to properly handle XML declarations
|
||||
tree = etree.fromstring(html_content.encode('utf-8') if isinstance(html_content, str) else html_content, parser=parser)
|
||||
else:
|
||||
tree = html.fromstring(html_content, parser=parser)
|
||||
html_block = ""
|
||||
|
||||
# Build namespace map for XPath queries
|
||||
@@ -261,8 +267,11 @@ def xpath1_filter(xpath_filter, html_content, append_pretty_line_formatting=Fals
|
||||
if isinstance(element, (str, bytes)):
|
||||
html_block += element
|
||||
else:
|
||||
# Return the HTML which will get parsed as text
|
||||
html_block += etree.tostring(element, pretty_print=True).decode('utf-8')
|
||||
# Return the HTML/XML which will get parsed as text
|
||||
# Use 'xml' method for RSS/XML content, 'html' for HTML content
|
||||
# parser will be XMLParser if we detected XML content
|
||||
method = 'xml' if (is_rss or isinstance(parser, etree.XMLParser)) else 'html'
|
||||
html_block += etree.tostring(element, pretty_print=True, method=method, encoding='unicode')
|
||||
|
||||
return html_block
|
||||
|
||||
|
||||
@@ -103,15 +103,15 @@ class guess_stream_type():
|
||||
self.is_json = True
|
||||
elif 'pdf' in magic_content_header:
|
||||
self.is_pdf = True
|
||||
elif has_html_patterns or http_content_header == 'text/html':
|
||||
self.is_html = True
|
||||
elif any(s in magic_content_header for s in JSON_CONTENT_TYPES):
|
||||
self.is_json = True
|
||||
# magic will call a rss document 'xml'
|
||||
# Rarely do endpoints give the right header, usually just text/xml, so we check also for <rss
|
||||
# This also triggers the automatic CDATA text parser so the RSS goes back a nice content list
|
||||
elif '<rss' in test_content_normalized or '<feed' in test_content_normalized or any(s in magic_content_header for s in RSS_XML_CONTENT_TYPES) or '<rdf:' in test_content_normalized:
|
||||
self.is_rss = True
|
||||
elif has_html_patterns or http_content_header == 'text/html':
|
||||
self.is_html = True
|
||||
elif any(s in magic_content_header for s in JSON_CONTENT_TYPES):
|
||||
self.is_json = True
|
||||
elif any(s in http_content_header for s in XML_CONTENT_TYPES):
|
||||
# Only mark as generic XML if not already detected as RSS
|
||||
if not self.is_rss:
|
||||
|
||||
@@ -201,3 +201,120 @@ def test_trips(html_content, xpath, answer):
|
||||
html_content = html_tools.xpath_filter(xpath, html_content, append_pretty_line_formatting=True)
|
||||
assert type(html_content) == str
|
||||
assert answer in html_content
|
||||
|
||||
|
||||
# Test for UTF-8 encoding bug fix (issue #3658)
|
||||
# Polish and other UTF-8 characters should be preserved correctly
|
||||
polish_html = """<!DOCTYPE html>
|
||||
<html>
|
||||
<head><meta charset="utf-8"></head>
|
||||
<body>
|
||||
<div class="index--s-headline-link">
|
||||
<a class="index--s-headline-link" href="#">
|
||||
Naukowcy potwierdzają: oglądanie krótkich filmików prowadzi do "zgnilizny mózgu"
|
||||
</a>
|
||||
</div>
|
||||
<div>
|
||||
<a class="other-class" href="#">
|
||||
Test with Polish chars: żółć ąę śń
|
||||
</a>
|
||||
</div>
|
||||
<div>
|
||||
<p class="unicode-test">Cyrillic: Привет мир</p>
|
||||
<p class="unicode-test">Greek: Γειά σου κόσμε</p>
|
||||
<p class="unicode-test">Arabic: مرحبا بالعالم</p>
|
||||
<p class="unicode-test">Chinese: 你好世界</p>
|
||||
<p class="unicode-test">Japanese: こんにちは世界</p>
|
||||
<p class="unicode-test">Emoji: 🌍🎉✨</p>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
@pytest.mark.parametrize("html_content", [polish_html])
|
||||
@pytest.mark.parametrize("xpath, expected_text", [
|
||||
# Test Polish characters in xpath_filter
|
||||
('//a[(contains(@class,"index--s-headline-link"))]', 'Naukowcy potwierdzają'),
|
||||
('//a[(contains(@class,"index--s-headline-link"))]', 'oglądanie krótkich filmików'),
|
||||
('//a[(contains(@class,"index--s-headline-link"))]', 'zgnilizny mózgu'),
|
||||
('//a[@class="other-class"]', 'żółć ąę śń'),
|
||||
|
||||
# Test various Unicode scripts
|
||||
('//p[@class="unicode-test"]', 'Привет мир'),
|
||||
('//p[@class="unicode-test"]', 'Γειά σου κόσμε'),
|
||||
('//p[@class="unicode-test"]', 'مرحبا بالعالم'),
|
||||
('//p[@class="unicode-test"]', '你好世界'),
|
||||
('//p[@class="unicode-test"]', 'こんにちは世界'),
|
||||
('//p[@class="unicode-test"]', '🌍🎉✨'),
|
||||
|
||||
# Test with text() extraction
|
||||
('//a[@class="other-class"]/text()', 'żółć'),
|
||||
])
|
||||
def test_xpath_utf8_encoding(html_content, xpath, expected_text):
|
||||
"""Test that XPath filters preserve UTF-8 characters correctly (issue #3658)"""
|
||||
result = html_tools.xpath_filter(xpath, html_content, append_pretty_line_formatting=False)
|
||||
assert type(result) == str
|
||||
assert expected_text in result
|
||||
# Ensure characters are NOT HTML-entity encoded
|
||||
# For example, 'ą' should NOT become 'ą'
|
||||
assert '&#' not in result or expected_text in result
|
||||
|
||||
|
||||
@pytest.mark.parametrize("html_content", [polish_html])
|
||||
@pytest.mark.parametrize("xpath, expected_text", [
|
||||
# Test Polish characters in xpath1_filter
|
||||
('//a[(contains(@class,"index--s-headline-link"))]', 'Naukowcy potwierdzają'),
|
||||
('//a[(contains(@class,"index--s-headline-link"))]', 'mózgu'),
|
||||
('//a[@class="other-class"]', 'żółć ąę śń'),
|
||||
|
||||
# Test various Unicode scripts with xpath1
|
||||
('//p[@class="unicode-test" and contains(text(), "Cyrillic")]', 'Привет мир'),
|
||||
('//p[@class="unicode-test" and contains(text(), "Greek")]', 'Γειά σου'),
|
||||
('//p[@class="unicode-test" and contains(text(), "Chinese")]', '你好世界'),
|
||||
])
|
||||
def test_xpath1_utf8_encoding(html_content, xpath, expected_text):
|
||||
"""Test that XPath1 filters preserve UTF-8 characters correctly"""
|
||||
result = html_tools.xpath1_filter(xpath, html_content, append_pretty_line_formatting=False)
|
||||
assert type(result) == str
|
||||
assert expected_text in result
|
||||
# Ensure characters are NOT HTML-entity encoded
|
||||
assert '&#' not in result or expected_text in result
|
||||
|
||||
|
||||
# Test with real-world example from wyborcza.pl (issue #3658)
|
||||
wyborcza_style_html = """<!DOCTYPE html>
|
||||
<html lang="pl">
|
||||
<head><meta charset="utf-8"></head>
|
||||
<body>
|
||||
<div class="article-list">
|
||||
<a class="index--s-headline-link" href="/article1">
|
||||
Naukowcy potwierdzają: oglądanie krótkich filmików prowadzi do "zgnilizny mózgu"
|
||||
</a>
|
||||
<a class="index--s-headline-link" href="/article2">
|
||||
Zmiany klimatyczne wpływają na życie w miastach
|
||||
</a>
|
||||
<a class="index--s-headline-link" href="/article3">
|
||||
Łódź: Nowe inwestycje w infrastrukturę miejską
|
||||
</a>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
def test_wyborcza_real_world_example():
|
||||
"""Test real-world case from wyborcza.pl that was failing (issue #3658)"""
|
||||
xpath = '//a[(contains(@class,"index--s-headline-link"))]'
|
||||
result = html_tools.xpath_filter(xpath, wyborcza_style_html, append_pretty_line_formatting=False)
|
||||
|
||||
# These exact strings should appear in the result
|
||||
assert 'Naukowcy potwierdzają' in result
|
||||
assert 'oglądanie krótkich filmików' in result
|
||||
assert 'zgnilizny mózgu' in result
|
||||
assert 'Łódź' in result
|
||||
|
||||
# Make sure they're NOT corrupted to mojibake like "potwierdzajÄ"
|
||||
assert 'potwierdzajÄ' not in result
|
||||
assert 'oglądanie' not in result
|
||||
assert 'mózgu' not in result
|
||||
|
||||
Reference in New Issue
Block a user