diff --git a/commons/src/main/java/org/archive/util/UriUtils.java b/commons/src/main/java/org/archive/util/UriUtils.java
new file mode 100644
index 00000000..86ce43ff
--- /dev/null
+++ b/commons/src/main/java/org/archive/util/UriUtils.java
@@ -0,0 +1,217 @@
+/* UriUtils
+ *
+ * $Id: MimetypeUtils.java 3119 2005-02-17 20:39:21Z stack-sf $
+ *
+ * Created on April 15, 2010
+ *
+ * Copyright (C) 2010 Internet Archive.
+ *
+ * This file is part of the Heritrix web crawler (crawler.archive.org).
+ *
+ * Heritrix is free software; you can redistribute it and/or modify
+ * it under the terms of the GNU Lesser Public License as published by
+ * the Free Software Foundation; either version 2.1 of the License, or
+ * any later version.
+ *
+ * Heritrix is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+ * GNU Lesser Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser Public License
+ * along with Heritrix; if not, write to the Free Software
+ * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
+ */
+package org.archive.util;
+
+import java.util.logging.Level;
+import java.util.logging.Logger;
+import java.util.regex.Matcher;
+
+import org.apache.commons.codec.DecoderException;
+import org.apache.commons.httpclient.URIException;
+import org.archive.net.LaxURLCodec;
+import org.archive.net.UURI;
+
+
+/**
+ * URI-related utilities.
+ *
+ * Primarily, a place to centralize and better document and test certain URI-related heuristics
+ * that may be useful in many places.
+ *
+ * The choice of when to consider a string likely enough to be a URI that we try crawling it
+ * is, so far, based on rather arbitrary rules-of-thumb. We have not quantitatively tested
+ * how often the strings that pass these tests yield meaningful (not 404, non-soft-404,
+ * non-garbage) replies. We are willing to accept some level of mistaken requests, knowing
+ * that their cost is usually negligible, if that allows us to discover meaningful content
+ * that could be not be discovered via other heuristics.
+ *
+ * Our intuitive understanding so far is that: strings that appear to have ./.. relative-path
+ * prefixes, dot-extensions, or path-slashes are good candidates for trying as URIs, even
+ * though with some Javascript/HTML-VALUE-attributes, this yields a lot of false positives.
+ *
+ * We want to get strings like....
+ *
+ * photo.jpg
+ * /photos
+ * /photos/
+ * ./photos
+ * ../../photos
+ * photos/index.html
+ *
+ * ...but we will thus also sometimes try strings that were other kinds of variables/
+ * parameters, like...
+ *
+ * rectangle.x
+ * 11.2px
+ * text/xml
+ * width:6.33
+ *
+ * Until better rules, exception-blacklists or even site-sensitive dynamic adjustment of
+ * heuristics (eg: this site, guesses are yield 200s, keep guessing; this site, guesses are
+ * all 404s, stop guessing) are developed, crawl operators should monitor their crawls
+ * (and contact email) for cases where speculative crawling are generating many errors, and
+ * use settings like ExtractorHTML's 'extract-javascript' and 'extract-value-attributes' or
+ * disable of ExtractorJS entirely when they want to curtail those errors.
+ *
+ * The 'legacy' tests are those used in H1 at least through 1.14.4. They have
+ * some known problems, but are not yet being dropped until more experience
+ * with the 'new' isLikelyUri() test is collected (in H3). Enable the 'xest'
+ * methods of the UriUtilsTest class for details.
+ *
+ * @contributor gojomo
+ */
+public class UriUtils {
+ private static final Logger LOGGER = Logger.getLogger(UriUtils.class.getName());
+
+//
+// new combined test
+//
+ // naive likely-uri test:
+ // no whitespace or '<' or '>';
+ // at least one '.' or '/';
+ // not ending with '.'
+ static final String NAIVE_LIKELY_URI_PATTERN = "[^<>\\s]*[\\./][^<>\\s]*(?', has an internal dot or some slash,
+ // begins and ends with either '/' or a word-char)
+ static final String STRING_URI_DETECTOR =
+ "(?:\\w|[\\.]{0,2}/)[\\S&&[^<>]]*(?:\\.|/)[\\S&&[^<>]]*(?:\\w|/)";
+
+
+ // blacklist of strings that STRING_URI_DETECTOR picks up as URIs,
+ // which are known to be problematic, and NOT to be
+ // added to outLinks
+ protected final static String[] STRING_URI_DETECTOR_EXCEPTIONS = {
+ "text/javascript"
+ };
+
+ public static boolean isLikelyUriJavascriptContextLegacy(CharSequence candidate) {
+ if(!TextUtils.matches(STRING_URI_DETECTOR,candidate)) {
+ return false;
+ }
+ for (String s : STRING_URI_DETECTOR_EXCEPTIONS) {
+ if (s.contentEquals(candidate))
+ return false;
+ }
+ // matches detector and not an exception: so a likely URI
+ return true;
+
+ }
+
+//
+// legacy likely-URI test from ExtractorHTML
+//
+
+ // much like the javascript likely-URI extractor, but
+ // without requiring quotes -- this can indicate whether
+ // an HTML tag attribute that isn't definitionally a
+ // URI might be one anyway, as in form-tag VALUE attributes
+ static final String LIKELY_URI_PATH =
+ "(\\.{0,2}[^\\.\\n\\r\\s\"']*(\\.[^\\.\\n\\r\\s\"']+)+)";
+
+ public static boolean isLikelyUriHtmlContextLegacy(CharSequence candidate) {
+ return TextUtils.matches(LIKELY_URI_PATH, candidate);
+ }
+}
diff --git a/commons/src/main/java/org/archive/util/UriUtilsTest.java b/commons/src/main/java/org/archive/util/UriUtilsTest.java
new file mode 100644
index 00000000..6fe1aca2
--- /dev/null
+++ b/commons/src/main/java/org/archive/util/UriUtilsTest.java
@@ -0,0 +1,192 @@
+/* UriUtilsTest
+ *
+ * $Id: ArchiveUtilsTest.java 5052 2007-04-10 02:26:52Z gojomo $
+ *
+ * Copyright (C) 2010 Internet Archive.
+ *
+ * This file is part of the Heritrix web crawler (crawler.archive.org).
+ *
+ * Heritrix is free software; you can redistribute it and/or modify
+ * it under the terms of the GNU Lesser Public License as published by
+ * the Free Software Foundation; either version 2.1 of the License, or
+ * any later version.
+ *
+ * Heritrix is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+ * GNU Lesser Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser Public License
+ * along with Heritrix; if not, write to the Free Software
+ * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
+ */
+
+package org.archive.util;
+
+import junit.framework.Test;
+import junit.framework.TestCase;
+import junit.framework.TestSuite;
+
+/**
+ * JUnit test suite for UriUtils.
+ *
+ * Several of the tests for the 'legacy' (H1 through at least 1.14.4)
+ * heuristics are disabled by renaming, because those heuristics have known
+ * failures; however, until more experience with the new heuristics is
+ * collected, H1 still uses them for consistency.
+ *
+ * @contributor gojomo
+ * @version $Id: ArchiveUtilsTest.java 5052 2007-04-10 02:26:52Z gojomo $
+ */
+public class UriUtilsTest extends TestCase {
+
+ public UriUtilsTest(final String testName) {
+ super(testName);
+ }
+
+ /**
+ * run all the tests for ArchiveUtilsTest
+ *
+ * @param argv
+ * the command line arguments
+ */
+ public static void main(String argv[]) {
+ junit.textui.TestRunner.run(suite());
+ }
+
+ public static Test suite() {
+ return new TestSuite(UriUtilsTest.class);
+ }
+
+ /** image URIs that should be considered likely URIs **/
+ static String[] urisRelativeImages = {
+ "photo.jpg",
+ "./photo.jpg",
+ "../photo.jpg",
+ "images/photo.jpg",
+ "../../images/photo.jpg" };
+
+ /** check that plausible relative image URIs return true with legacy tests */
+ public void xestLegacySimpleImageRelatives() {
+ legacyTryAll(urisRelativeImages, true);
+ }
+
+ /** check that plausible relative image URIs return true with new tests */
+ public void testNewSimpleImageRelatives() {
+ tryAll(urisRelativeImages,true);
+ }
+
+ /** absolute URIs that should be considered likely URIs **/
+ static String[] urisAbsolute = {
+ "http://example.com",
+ "http://example.com/", "http://www.example.com",
+ "http://www.example.com/", "http://www.example.com/about",
+ "http://www.example.com/about/",
+ "http://www.example.com/about/index.html", "https://example.com",
+ "https://example.com/", "https://www.example.com",
+ "https://www.example.com/", "https://www.example.com/about",
+ "https://www.example.com/about/",
+ "https://www.example.com/about/index.html",
+ "ftp://example.com/public/report.pdf",
+ // TODO: other schemes? mailto?
+
+ };
+
+ /** check that absolute URIs return true with legacy tests */
+ public void testLegacyAbsolutes() {
+ legacyTryAll(urisAbsolute,true);
+ }
+
+ /** check that absolute URIs return true with new tests */
+ public void testAbsolutes() {
+ tryAll(urisAbsolute,true);
+ }
+
+ /** path-absolute images URIs that should be considered likely URIs **/
+ static String[] urisPathAbsoluteImages = {
+ "/photo.jpg",
+ "/images/photo.jpg",
+ };
+
+ /** check that path-absolute image URIs return true with legacy tests*/
+ public void testLegacySimpleImagePathAbsolutes() {
+ legacyTryAll(urisPathAbsoluteImages, true);
+ }
+
+ /** check that path-absolute image URIs return true with new tests*/
+ public void testSimpleImagePathAbsolutes() {
+ tryAll(urisPathAbsoluteImages, true);
+ }
+
+ /** URI-like strings risking false positives that should NOT be likely URIs **/
+ static String[] notUrisNaiveFalsePositives = {
+ "0.99",
+ "3.14157",
+ "text/javascript"
+ };
+
+ /** check that typical false-positives of the naive test are not deemed URIs */
+ public void xestLegacyNaiveFalsePositives() {
+ legacyTryAll(notUrisNaiveFalsePositives, false);
+ }
+
+ /** check that typical false-positives of the naive test are not deemed URIs */
+ public void testNaiveFalsePositives() {
+ tryAll(notUrisNaiveFalsePositives, false);
+ }
+
+ /** strings that should not be considered likely URIs **/
+ static String[] notUrisNaive = {
+ "foo bar",
+ "",
+ "item\t$0.99\tred",
+ };
+
+ /** check that strings that fail naive test are not deemed URIs legacy tests*/
+ public void testLegacyNaiveNotUris() {
+ legacyTryAll(notUrisNaive, false);
+ }
+
+ /** check that strings that fail naive test are not deemed URIs new tests*/
+ public void testNaiveNotUris() {
+ tryAll(notUrisNaive, false);
+ }
+
+
+ /**
+ * Test that all supplied candidates give the expected result, for each of
+ * the 'legacy' (H1) likely-URI-tests
+ *
+ * @param candidates String[] to test
+ * @param expected desired answer
+ */
+ protected void legacyTryAll(String[] candidates, boolean expected) {
+ for (String candidate : candidates) {
+ assertEquals("javascript context: " + candidate,
+ expected,
+ UriUtils.isLikelyUriJavascriptContextLegacy(candidate));
+ assertEquals("html context: " + candidate,
+ expected,
+ UriUtils.isLikelyUriHtmlContextLegacy(candidate));
+ }
+ }
+
+
+
+ /**
+ * Test that all supplied candidates give the expected results, for
+ * the 'new' heuristics now in this class.
+ * @param candidates String[] to test
+ * @param expected desired answer
+ */
+ protected void tryAll(String[] candidates, boolean expected) {
+ for (String candidate : candidates) {
+ assertEquals("new: " + candidate,
+ expected,
+ UriUtils.isLikelyUri(candidate));
+ assertEquals("html context: " + candidate,
+ expected,
+ UriUtils.isLikelyUri(candidate));
+ }
+ }
+}
diff --git a/modules/src/main/java/org/archive/modules/extractor/ExtractorHTML.java b/modules/src/main/java/org/archive/modules/extractor/ExtractorHTML.java
index 23b97f74..3a1555f7 100644
--- a/modules/src/main/java/org/archive/modules/extractor/ExtractorHTML.java
+++ b/modules/src/main/java/org/archive/modules/extractor/ExtractorHTML.java
@@ -38,6 +38,7 @@ import org.springframework.beans.factory.InitializingBean;
import org.archive.util.ArchiveUtils;
import org.archive.util.DevUtils;
import org.archive.util.TextUtils;
+import org.archive.util.UriUtils;
import org.springframework.beans.factory.annotation.Autowired;
/**
@@ -181,13 +182,7 @@ public class ExtractorHTML extends ContentExtractor implements InitializingBean
// 15: single-quote delimited attr value
// 16: space-delimited attr value
-
- // much like the javascript likely-URI extractor, but
- // without requiring quotes -- this can indicate whether
- // an HTML tag attribute that isn't definitionally a
- // URI might be one anyway, as in form-tag VALUE attributes
- static final String LIKELY_URI_PATH =
- "(\\.{0,2}[^\\.\\n\\r\\s\"']*(\\.[^\\.\\n\\r\\s\"']+)+)";
+
static final String WHITESPACE = "\\s";
static final String CLASSEXT =".class";
static final String APPLET = "applet";
@@ -435,10 +430,8 @@ public class ExtractorHTML extends ContentExtractor implements InitializingBean
}
} else if (attr.start(10) > -1) {
// VALUE, with possibility of URI
- if (extractValueAttributes
- && TextUtils.matches(LIKELY_URI_PATH, value)) {
- CharSequence context = elementContext(element,
- attr.group(10));
+ if (extractValueAttributes && UriUtils.isLikelyUri(value)) {
+ CharSequence context = elementContext(element, attr.group(10));
processLink(curi,value, context);
}
diff --git a/modules/src/main/java/org/archive/modules/extractor/ExtractorJS.java b/modules/src/main/java/org/archive/modules/extractor/ExtractorJS.java
index 9f8b4903..92050a9b 100644
--- a/modules/src/main/java/org/archive/modules/extractor/ExtractorJS.java
+++ b/modules/src/main/java/org/archive/modules/extractor/ExtractorJS.java
@@ -18,23 +18,21 @@
*/
package org.archive.modules.extractor;
+import static org.archive.modules.extractor.Hop.SPECULATIVE;
+import static org.archive.modules.extractor.LinkContext.JS_MISC;
+
import java.io.IOException;
-import java.util.logging.Level;
import java.util.logging.Logger;
import java.util.regex.Matcher;
-import org.apache.commons.codec.DecoderException;
import org.apache.commons.httpclient.URIException;
import org.archive.io.ReplayCharSequence;
import org.archive.modules.CrawlURI;
-import org.archive.net.LaxURLCodec;
import org.archive.net.UURI;
import org.archive.util.ArchiveUtils;
import org.archive.util.DevUtils;
import org.archive.util.TextUtils;
-
-import static org.archive.modules.extractor.Hop.SPECULATIVE;
-import static org.archive.modules.extractor.LinkContext.JS_MISC;
+import org.archive.util.UriUtils;
/**
* Processes Javascript files for strings that are likely to be
@@ -47,13 +45,10 @@ public class ExtractorJS extends ContentExtractor {
private static final long serialVersionUID = 2L;
+ @SuppressWarnings("unused")
private static Logger LOGGER =
Logger.getLogger("org.archive.crawler.extractor.ExtractorJS");
- static final String AMP = "&";
- static final String ESCAPED_AMP = "&";
- static final String WHITESPACE = "\\s";
-
// finds whitespace-free strings in Javascript
// (areas between paired ' or " characters, possibly backslash-quoted
// on the ends, but not in the middle)
@@ -63,22 +58,9 @@ public class ExtractorJS extends ContentExtractor {
// (G1) ' or " with optional leading backslashes
// (G2) whitespace-free string delimited on boths ends by G1
- // determines whether a string is likely URI
- // (no whitespace or '<' '>', has an internal dot or some slash,
- // begins and ends with either '/' or a word-char)
- static final String STRING_URI_DETECTOR =
- "(?:\\w|[\\.]{0,2}/)[\\S&&[^<>]]*(?:\\.|/)[\\S&&[^<>]]*(?:\\w|/)";
-
protected long numberOfCURIsHandled = 0;
protected static long numberOfLinksExtracted = 0;
- // strings that STRING_URI_DETECTOR picks up as URIs,
- // which are known to be problematic, and NOT to be
- // added to outLinks
- protected final static String[] STRING_URI_DETECTOR_EXCEPTIONS = {
- "text/javascript"
- };
-
// URIs known to produce false-positives with the current JS extractor.
// e.g. currently (2.0.3) the JS extractor produces 13 false-positive
// URIs from http://www.google-analytics.com/urchin.js and only 2
@@ -142,8 +124,7 @@ public class ExtractorJS extends ContentExtractor {
try {
cs = curi.getRecorder().getReplayCharSequence();
try {
- numberOfLinksExtracted += considerStrings(this, curi, cs,
- true);
+ numberOfLinksExtracted += considerStrings(this, curi, cs, true);
} catch (StackOverflowError e) {
DevUtils.warnHandle(e, "ExtractorJS StackOverflowError");
}
@@ -165,16 +146,9 @@ public class ExtractorJS extends ContentExtractor {
while(strings.find()) {
CharSequence subsequence =
cs.subSequence(strings.start(2), strings.end(2));
- Matcher uri =
- TextUtils.getMatcher(STRING_URI_DETECTOR, subsequence);
- if(uri.matches()) {
- String string = uri.group();
- // protect against adding outlinks for known problematic matches
- if (isUriMatchException(string,cs)) {
- TextUtils.recycleMatcher(uri);
- continue;
- }
- string = speculativeFixup(string, curi);
+ if(UriUtils.isLikelyUri(subsequence)) {
+ String string = subsequence.toString();
+ string = UriUtils.speculativeFixup(string, curi.getUURI());
foundLinks++;
try {
int max = ext.getExtractorParameters().getMaxOutlinks();
@@ -192,75 +166,8 @@ public class ExtractorJS extends ContentExtractor {
foundLinks += considerStrings(ext, curi, subsequence,
handlingJSFile);
}
- TextUtils.recycleMatcher(uri);
}
TextUtils.recycleMatcher(strings);
return foundLinks;
}
-
- /**
- * checks to see if URI match is a special case
- * @param string matched by STRING_URI_DETECTOR
- * @param cs
- * @return true if string is one of STRING_URI_EXCEPTIONS
- */
- private static boolean isUriMatchException(String string,CharSequence cs) {
- for (String s : STRING_URI_DETECTOR_EXCEPTIONS) {
- if (s.equals(string))
- return true;
- }
- return false;
- }
-
-
- /**
- * Perform additional fixup of likely-URI Strings
- *
- * @param string detected candidate String
- * @return String changed/decoded to increase liklihood it is a
- * meaningful non-404 URI
- */
- public static String speculativeFixup(String string, CrawlURI puri) {
- String retVal = string;
-
- // unescape ampersands
- retVal = TextUtils.replaceAll(ESCAPED_AMP, retVal, AMP);
-
- // uri-decode if begins with encoded 'http(s)?%3A'
- Matcher m = TextUtils.getMatcher("(?i)^https?%3A.*",retVal);
- if(m.matches()) {
- try {
- retVal = LaxURLCodec.DEFAULT.decode(retVal);
- } catch (DecoderException e) {
- LOGGER.log(Level.INFO,"unable to decode",e);
- }
- }
- TextUtils.recycleMatcher(m);
-
- // TODO: more URI-decoding if there are %-encoded parts?
-
- // detect scheme-less intended-absolute-URI
- // intent: "opens with what looks like a dotted-domain, and
- // last segment is a top-level-domain (eg "com", "org", etc)"
- m = TextUtils.getMatcher(
- "^[^\\./:\\s%]+\\.[^/:\\s%]+\\.([^\\./:\\s%]+)(/.*|)$",
- retVal);
- if(m.matches()) {
- if(ArchiveUtils.isTld(m.group(1))) {
- String schemePlus = "http://";
- // if on exact same host preserve scheme (eg https)
- try {
- if (retVal.startsWith(puri.getUURI().getHost())) {
- schemePlus = puri.getUURI().getScheme() + "://";
- }
- } catch (URIException e) {
- // error retrieving source host - ignore it
- }
- retVal = schemePlus + retVal;
- }
- }
- TextUtils.recycleMatcher(m);
-
- return retVal;
- }
}
diff --git a/modules/src/main/java/org/archive/modules/extractor/ExtractorSWF.java b/modules/src/main/java/org/archive/modules/extractor/ExtractorSWF.java
index 95da44a5..b98be287 100644
--- a/modules/src/main/java/org/archive/modules/extractor/ExtractorSWF.java
+++ b/modules/src/main/java/org/archive/modules/extractor/ExtractorSWF.java
@@ -21,10 +21,9 @@ package org.archive.modules.extractor;
import java.io.IOException;
import java.io.InputStream;
import java.util.logging.Logger;
-import java.util.regex.Matcher;
import org.archive.modules.CrawlURI;
-import org.archive.util.TextUtils;
+import org.archive.util.UriUtils;
import com.anotherbigidea.flash.interfaces.SWFActions;
import com.anotherbigidea.flash.interfaces.SWFTagTypes;
@@ -326,15 +325,12 @@ public class ExtractorSWF extends ContentExtractor {
}
public void considerStringAsUri(String str) throws IOException {
- Matcher uri = TextUtils.getMatcher(ExtractorJS.STRING_URI_DETECTOR,
- str);
-
- if (uri.matches()) {
+ if (UriUtils.isLikelyUri(str)) {
int max = ext.getExtractorParameters().getMaxOutlinks();
- Link.addRelativeToVia(curi, max, uri.group(), LinkContext.SPECULATIVE_MISC, Hop.SPECULATIVE);
+ Link.addRelativeToVia(curi, max, str,
+ LinkContext.SPECULATIVE_MISC, Hop.SPECULATIVE);
linkCount++;
}
- TextUtils.recycleMatcher(uri);
}
diff --git a/modules/src/main/java/org/archive/modules/extractor/JerichoExtractorHTML.java b/modules/src/main/java/org/archive/modules/extractor/JerichoExtractorHTML.java
index c001b137..8a07814f 100644
--- a/modules/src/main/java/org/archive/modules/extractor/JerichoExtractorHTML.java
+++ b/modules/src/main/java/org/archive/modules/extractor/JerichoExtractorHTML.java
@@ -42,6 +42,7 @@ import org.archive.net.UURI;
import org.archive.net.UURIFactory;
import org.archive.util.DevUtils;
import org.archive.util.TextUtils;
+import org.archive.util.UriUtils;
import au.id.jericho.lib.html.Attribute;
import au.id.jericho.lib.html.Attributes;
@@ -235,8 +236,7 @@ public class JerichoExtractorHTML extends ExtractorHTML {
// VALUE
if (((attr = attributes.get("value")) != null) &&
((attrValue = attr.getValue()) != null)) {
- if (TextUtils.matches(LIKELY_URI_PATH, attrValue)
- && overlyEagerLinkDetection) {
+ if (UriUtils.isLikelyUri(attrValue) && overlyEagerLinkDetection) {
CharSequence context = elementContext(elementName, attr
.getKey());
processLink(curi, attrValue, context);