Decode JavaScript Unicode code-point escapes in extracted URLs (#772)

Signed-off-by: Shubham Padkonde <shubhampadkonde12@gmail.com>
This commit is contained in:
Shubham Padkonde authored and GitHub committed 2026-09-22 17:21:42 +09:00
1 parent ed2805f6b7
commit 16deaaf4eb
2 files changed
+85 -1

No files matched your search

@@ -22,12 +22,16 @@ import static org.archive.modules.extractor.Hop.SPECULATIVE;
import static org.archive.modules.extractor.LinkContext.JS_MISC;
import java.io.IOException;
import java.io.Writer;
import java.util.logging.Level;
import java.util.logging.Logger;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import org.archive.url.URIException;
import org.apache.commons.lang3.StringEscapeUtils;
import org.apache.commons.lang3.text.translate.AggregateTranslator;
import org.apache.commons.lang3.text.translate.CharSequenceTranslator;
import org.archive.io.ReplayCharSequence;
import org.archive.modules.CrawlURI;
import org.archive.net.UURI;
@@ -62,6 +66,30 @@ public class ExtractorJS extends ContentExtractor {
private static Logger LOGGER =
Logger.getLogger(ExtractorJS.class.getName());
private static final Pattern CODE_POINT_ESCAPE = Pattern.compile("\\\\u\\{([0-9a-fA-F]+)\\}");
// Translate in one pass so an escaped backslash cannot introduce a second escape.
static final CharSequenceTranslator UNESCAPE_JAVASCRIPT = new AggregateTranslator(
new CharSequenceTranslator() {
@Override
public int translate(CharSequence input, int index, Writer out) throws IOException {
if (input.charAt(index) != '\\') {
return 0;
}
Matcher matcher = CODE_POINT_ESCAPE.matcher(input).region(index, input.length());
if (!matcher.lookingAt()) {
return 0;
}
int codePoint = Integer.parseInt(matcher.group(1), 16);
if (!Character.isValidCodePoint(codePoint)) {
throw new IllegalArgumentException("Invalid JavaScript Unicode code point: "
+ matcher.group(1));
}
out.write(Character.toChars(codePoint));
return matcher.end() - index;
}
}, StringEscapeUtils.UNESCAPE_ECMASCRIPT);
// finds strings in Javascript
// (areas between paired ' or " characters, possibly backslash-quoted
// on the ends, but not in the middle)
@@ -170,7 +198,7 @@ public class ExtractorJS extends ContentExtractor {
protected boolean considerString(Extractor ext, CrawlURI curi,
boolean handlingJSFile, String candidate) {
try {
candidate = StringEscapeUtils.unescapeEcmaScript(candidate);
candidate = UNESCAPE_JAVASCRIPT.translate(candidate);
} catch (Exception e) {
LOGGER.log(Level.WARNING, "problem unescaping some javascript", e);
}
@@ -21,11 +21,20 @@ package org.archive.modules.extractor;
import java.util.ArrayList;
import java.util.Collection;
import java.util.List;
import java.util.Set;
import java.util.stream.Stream;
import org.archive.modules.CrawlURI;
import org.archive.net.UURI;
import org.archive.net.UURIFactory;
import org.archive.util.Recorder;
import org.junit.jupiter.params.ParameterizedTest;
import org.junit.jupiter.params.provider.Arguments;
import org.junit.jupiter.params.provider.MethodSource;
import org.junit.jupiter.params.provider.ValueSource;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertThrows;
/**
* Unit test for {@link ExtractorJS}.
@@ -35,6 +44,53 @@ import org.archive.util.Recorder;
*/
public class ExtractorJSTest extends StringExtractorTestBase {
static Stream<Arguments> unicodeEscapeUrls() {
return Stream.of(
Arguments.of("\\u{61}.html", "a.html"),
Arguments.of("\\u{79F}.html", "\u079f.html"),
Arguments.of("\\u{1f600}.html", "\ud83d\ude00.html"),
Arguments.of("\\u{000000000061}.html", "a.html"),
Arguments.of("\\u{61}\\u002f\\u{62}.html", "a/b.html"),
Arguments.of("\\u0061.html", "a.html"));
}
@ParameterizedTest
@MethodSource("unicodeEscapeUrls")
void extractsUnicodeEscapeUrls(String escaped, String decoded) throws Exception {
for (TestData data : makeData("var url = 'http://example.com/" + escaped + "';",
"http://example.com/" + decoded)) {
data.uri.setFetchStatus(200);
extractor.process(data.uri);
assertEquals(Set.of(data.expectedResult), data.uri.getOutLinks());
assertNoSideEffects(data.uri);
}
}
static Stream<Arguments> escapedBackslashes() {
return Stream.of(
Arguments.of("\\\\u{61}", "\\u{61}"),
Arguments.of("\\\\\\u{61}", "\\a"),
Arguments.of("\\u{5c}u{61}", "\\u{61}"),
Arguments.of("\\u005cu{61}", "\\u{61}"),
Arguments.of("\\u{0}", "\0"),
Arguments.of("\\u{10FFFF}", "\udbff\udfff"),
Arguments.of("\\u{D800}", "\ud800"));
}
@ParameterizedTest
@MethodSource("escapedBackslashes")
void preservesEscapeBoundaries(String escaped, String decoded) {
assertEquals(decoded, ExtractorJS.UNESCAPE_JAVASCRIPT.translate(escaped));
}
@ParameterizedTest
@ValueSource(strings = {"\\u{}", "\\u{61", "\\u{xyz}", "\\u{110000}",
"\\u{ffffffffffffffff}", "\\u{+61}", "\\u{6_1}"})
void rejectsInvalidCodePointEscapes(String escaped) {
assertThrows(IllegalArgumentException.class,
() -> ExtractorJS.UNESCAPE_JAVASCRIPT.translate(escaped));
}
final public static String[] VALID_TEST_DATA = new String[] {
"var foo = \"http://www.example.com/outlink\";",
"http://www.example.com/outlink",