mirror of
https://github.com/internetarchive/heritrix3.git
synced 2026-10-09 22:01:47 +00:00
Decode JavaScript Unicode code-point escapes in extracted URLs (#772)
Signed-off-by: Shubham Padkonde <shubhampadkonde12@gmail.com>
This commit is contained in:
1 parent
ed2805f6b7
commit
16deaaf4eb
2 files changed
+85
-1
No files matched your search
@@ -22,12 +22,16 @@ import static org.archive.modules.extractor.Hop.SPECULATIVE;
|
||||
import static org.archive.modules.extractor.LinkContext.JS_MISC;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.Writer;
|
||||
import java.util.logging.Level;
|
||||
import java.util.logging.Logger;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.archive.url.URIException;
|
||||
import org.apache.commons.lang3.StringEscapeUtils;
|
||||
import org.apache.commons.lang3.text.translate.AggregateTranslator;
|
||||
import org.apache.commons.lang3.text.translate.CharSequenceTranslator;
|
||||
import org.archive.io.ReplayCharSequence;
|
||||
import org.archive.modules.CrawlURI;
|
||||
import org.archive.net.UURI;
|
||||
@@ -62,6 +66,30 @@ public class ExtractorJS extends ContentExtractor {
|
||||
private static Logger LOGGER =
|
||||
Logger.getLogger(ExtractorJS.class.getName());
|
||||
|
||||
private static final Pattern CODE_POINT_ESCAPE = Pattern.compile("\\\\u\\{([0-9a-fA-F]+)\\}");
|
||||
|
||||
// Translate in one pass so an escaped backslash cannot introduce a second escape.
|
||||
static final CharSequenceTranslator UNESCAPE_JAVASCRIPT = new AggregateTranslator(
|
||||
new CharSequenceTranslator() {
|
||||
@Override
|
||||
public int translate(CharSequence input, int index, Writer out) throws IOException {
|
||||
if (input.charAt(index) != '\\') {
|
||||
return 0;
|
||||
}
|
||||
Matcher matcher = CODE_POINT_ESCAPE.matcher(input).region(index, input.length());
|
||||
if (!matcher.lookingAt()) {
|
||||
return 0;
|
||||
}
|
||||
int codePoint = Integer.parseInt(matcher.group(1), 16);
|
||||
if (!Character.isValidCodePoint(codePoint)) {
|
||||
throw new IllegalArgumentException("Invalid JavaScript Unicode code point: "
|
||||
+ matcher.group(1));
|
||||
}
|
||||
out.write(Character.toChars(codePoint));
|
||||
return matcher.end() - index;
|
||||
}
|
||||
}, StringEscapeUtils.UNESCAPE_ECMASCRIPT);
|
||||
|
||||
// finds strings in Javascript
|
||||
// (areas between paired ' or " characters, possibly backslash-quoted
|
||||
// on the ends, but not in the middle)
|
||||
@@ -170,7 +198,7 @@ public class ExtractorJS extends ContentExtractor {
|
||||
protected boolean considerString(Extractor ext, CrawlURI curi,
|
||||
boolean handlingJSFile, String candidate) {
|
||||
try {
|
||||
candidate = StringEscapeUtils.unescapeEcmaScript(candidate);
|
||||
candidate = UNESCAPE_JAVASCRIPT.translate(candidate);
|
||||
} catch (Exception e) {
|
||||
LOGGER.log(Level.WARNING, "problem unescaping some javascript", e);
|
||||
}
|
||||
|
||||
@@ -21,11 +21,20 @@ package org.archive.modules.extractor;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.archive.modules.CrawlURI;
|
||||
import org.archive.net.UURI;
|
||||
import org.archive.net.UURIFactory;
|
||||
import org.archive.util.Recorder;
|
||||
import org.junit.jupiter.params.ParameterizedTest;
|
||||
import org.junit.jupiter.params.provider.Arguments;
|
||||
import org.junit.jupiter.params.provider.MethodSource;
|
||||
import org.junit.jupiter.params.provider.ValueSource;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
|
||||
/**
|
||||
* Unit test for {@link ExtractorJS}.
|
||||
@@ -35,6 +44,53 @@ import org.archive.util.Recorder;
|
||||
*/
|
||||
public class ExtractorJSTest extends StringExtractorTestBase {
|
||||
|
||||
static Stream<Arguments> unicodeEscapeUrls() {
|
||||
return Stream.of(
|
||||
Arguments.of("\\u{61}.html", "a.html"),
|
||||
Arguments.of("\\u{79F}.html", "\u079f.html"),
|
||||
Arguments.of("\\u{1f600}.html", "\ud83d\ude00.html"),
|
||||
Arguments.of("\\u{000000000061}.html", "a.html"),
|
||||
Arguments.of("\\u{61}\\u002f\\u{62}.html", "a/b.html"),
|
||||
Arguments.of("\\u0061.html", "a.html"));
|
||||
}
|
||||
|
||||
@ParameterizedTest
|
||||
@MethodSource("unicodeEscapeUrls")
|
||||
void extractsUnicodeEscapeUrls(String escaped, String decoded) throws Exception {
|
||||
for (TestData data : makeData("var url = 'http://example.com/" + escaped + "';",
|
||||
"http://example.com/" + decoded)) {
|
||||
data.uri.setFetchStatus(200);
|
||||
extractor.process(data.uri);
|
||||
assertEquals(Set.of(data.expectedResult), data.uri.getOutLinks());
|
||||
assertNoSideEffects(data.uri);
|
||||
}
|
||||
}
|
||||
|
||||
static Stream<Arguments> escapedBackslashes() {
|
||||
return Stream.of(
|
||||
Arguments.of("\\\\u{61}", "\\u{61}"),
|
||||
Arguments.of("\\\\\\u{61}", "\\a"),
|
||||
Arguments.of("\\u{5c}u{61}", "\\u{61}"),
|
||||
Arguments.of("\\u005cu{61}", "\\u{61}"),
|
||||
Arguments.of("\\u{0}", "\0"),
|
||||
Arguments.of("\\u{10FFFF}", "\udbff\udfff"),
|
||||
Arguments.of("\\u{D800}", "\ud800"));
|
||||
}
|
||||
|
||||
@ParameterizedTest
|
||||
@MethodSource("escapedBackslashes")
|
||||
void preservesEscapeBoundaries(String escaped, String decoded) {
|
||||
assertEquals(decoded, ExtractorJS.UNESCAPE_JAVASCRIPT.translate(escaped));
|
||||
}
|
||||
|
||||
@ParameterizedTest
|
||||
@ValueSource(strings = {"\\u{}", "\\u{61", "\\u{xyz}", "\\u{110000}",
|
||||
"\\u{ffffffffffffffff}", "\\u{+61}", "\\u{6_1}"})
|
||||
void rejectsInvalidCodePointEscapes(String escaped) {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> ExtractorJS.UNESCAPE_JAVASCRIPT.translate(escaped));
|
||||
}
|
||||
|
||||
final public static String[] VALID_TEST_DATA = new String[] {
|
||||
"var foo = \"http://www.example.com/outlink\";",
|
||||
"http://www.example.com/outlink",
|
||||
|
||||
Reference in new issue
Block a user