diff --git a/contrib/pom.xml b/contrib/pom.xml
index 6e048084..a9d16633 100644
--- a/contrib/pom.xml
+++ b/contrib/pom.xml
@@ -59,6 +59,11 @@
amqp-client
3.2.1
+
+ com.itextpdf
+ itextpdf
+ 5.5.0
+
diff --git a/contrib/src/main/java/org/archive/modules/extractor/ExtractorPDFContent.java b/contrib/src/main/java/org/archive/modules/extractor/ExtractorPDFContent.java
new file mode 100644
index 00000000..77b3378c
--- /dev/null
+++ b/contrib/src/main/java/org/archive/modules/extractor/ExtractorPDFContent.java
@@ -0,0 +1,181 @@
+/*
+ * This file is part of the Heritrix web crawler (crawler.archive.org).
+ *
+ * Licensed to the Internet Archive (IA) by one or more individual
+ * contributors.
+ *
+ * The IA licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.archive.modules.extractor;
+
+import java.io.File;
+import java.io.IOException;
+import java.util.ArrayList;
+import java.util.logging.Logger;
+import java.util.regex.Matcher;
+import java.util.regex.Pattern;
+
+import org.apache.commons.httpclient.URIException;
+import org.archive.io.SinkHandlerLogThread;
+import org.archive.modules.CrawlURI;
+import org.archive.net.UURI;
+import org.archive.net.UURIFactory;
+import org.archive.util.FileUtils;
+
+import com.itextpdf.text.pdf.PdfReader;
+import com.itextpdf.text.pdf.parser.PdfReaderContentParser;
+import com.itextpdf.text.pdf.parser.SimpleTextExtractionStrategy;
+import com.itextpdf.text.pdf.parser.TextExtractionStrategy;
+
+/**
+ * PDF Content Extractor. This will parse the text content of a PDF and apply a
+ * regex to search for links within the body of the text.
+ *
+ * @contributor adam
+ */
+public class ExtractorPDFContent extends ContentExtractor {
+
+ @SuppressWarnings("unused")
+ private static final long serialVersionUID = 3L;
+
+ private static final Logger LOGGER =
+ Logger.getLogger(ExtractorPDF.class.getName());
+
+ public static final Pattern URLPattern = Pattern.compile(
+ "(https?):\\/\\/"+ // protocol
+ "(([a-z0-9$_\\.\\+!\\*\\'\\(\\),;\\?&=-]|%[0-9a-f]{2})+"+ // username
+ "(:([a-z0-9$_\\.\\+!\\*\\'\\(\\),;\\?&=-]|%[0-9a-f]{2})+)?"+ // password
+ "@)?(?"+ // auth requires @
+ ")((([a-z0-9]\\.|[a-z0-9][a-z0-9-]*[a-z0-9]\\.)*"+ // domain segments AND
+ "[a-z][a-z0-9-]*[a-z0-9]"+ // top level domain OR
+ "|((\\d|[1-9]\\d|1\\d{2}|2[0-4][0-9]|25[0-5])\\.){3}"+
+ "(\\d|[1-9]\\d|1\\d{2}|2[0-4][0-9]|25[0-5])"+ // IP address
+ ")(:\\d+)?)"+ // port
+ "(((\\/+([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)*"+ // path
+ "(\\?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)?)?)?"+ // query string
+ "(\\n"+ // possible newline (seems to happen in pdfs)
+ "((\\/)?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)*"+ // continue possible path
+ "(\\?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)?"+ // or possible query
+ ")?");
+
+ /**
+ * The maximum size of PDF files to consider. PDFs larger than this
+ * maximum will not be searched for links.
+ */
+ {
+ setMaxSizeToParse(10*1024*1024L); // 10MB
+ }
+ public long getMaxSizeToParse() {
+ return (Long) kp.get("maxSizeToParse");
+ }
+ public void setMaxSizeToParse(long threshold) {
+ kp.put("maxSizeToParse",threshold);
+ }
+
+
+ public ExtractorPDFContent() {
+ }
+
+ protected boolean innerExtract(CrawlURI curi){
+ File tempFile;
+ int sn;
+ Thread thread = Thread.currentThread();
+ if (thread instanceof SinkHandlerLogThread) {
+ sn = ((SinkHandlerLogThread)thread).getSerialNumber();
+ } else {
+ sn = System.identityHashCode(thread);
+ }
+ try {
+ tempFile = File.createTempFile("tt" + sn , "tmp.pdf");
+ } catch (IOException ioe) {
+ throw new RuntimeException(ioe);
+ }
+
+ PdfReader documentReader;
+ ArrayList uris = new ArrayList();
+ try {
+ documentReader = new PdfReader(curi.getRecorder().getContentReplayInputStream());
+ String content = extractContent(documentReader);
+ Matcher matcher = URLPattern.matcher(content);
+ while(matcher.find()) {
+ uris.add(content.substring(matcher.start(),matcher.end()));
+
+ //also add match without newline in case we are wrong
+ uris.add(matcher.group(1)+"://"+(matcher.group(2)!=null?matcher.group(2):"")+matcher.group(6)+matcher.group(13));
+ }
+
+
+
+ } catch (IOException e) {
+ curi.getNonFatalFailures().add(e);
+ return false;
+ } catch (RuntimeException e) {
+ // Truncated/corrupt PDFs may generate ClassCast exceptions, or
+ // other problems
+ curi.getNonFatalFailures().add(e);
+ return false;
+ } finally {
+ FileUtils.deleteSoonerOrLater(tempFile);
+ }
+
+ if (uris.size()<1) {
+ return true;
+ }
+
+ for (String uri: uris) {
+ try {
+ UURI src = curi.getUURI();
+ UURI dest = UURIFactory.getInstance(uri);
+ LinkContext lc = LinkContext.NAVLINK_MISC;
+ Hop hop = Hop.NAVLINK;
+ Link out = new Link(src, dest, lc, hop);
+ curi.getOutLinks().add(out);
+ } catch (URIException e1) {
+ logUriError(e1, curi.getUURI(), uri);
+ }
+ }
+
+ numberOfLinksExtracted.addAndGet(uris.size());
+
+ LOGGER.fine(curi+" has "+uris.size()+" links.");
+ // Set flag to indicate that link extraction is completed.
+ return true;
+ }
+
+ public String extractContent(PdfReader documentReader){
+ String content ="";
+ PdfReaderContentParser parser = new PdfReaderContentParser(documentReader);
+ TextExtractionStrategy strat;
+ for(int i=1; i< documentReader.getNumberOfPages(); i++) {
+ try {
+ strat = parser.processContent(i, new SimpleTextExtractionStrategy());
+ content += strat.getResultantText();
+
+ } catch (IOException e) {
+ // TODO Auto-generated catch block
+ e.printStackTrace();
+ }
+ }
+ return content;
+ }
+ @Override
+ protected boolean shouldExtract(CrawlURI uri) {
+ long max = getMaxSizeToParse();
+ if (uri.getRecorder().getRecordedInput().getSize() > max) {
+ return false;
+ }
+
+ String ct = uri.getContentType();
+ return (ct != null) && (ct.startsWith("application/pdf"));
+ }
+}
diff --git a/contrib/src/test/java/org/archive/modules/extractor/ExtractorPDFContentTest.java b/contrib/src/test/java/org/archive/modules/extractor/ExtractorPDFContentTest.java
new file mode 100644
index 00000000..850418ad
--- /dev/null
+++ b/contrib/src/test/java/org/archive/modules/extractor/ExtractorPDFContentTest.java
@@ -0,0 +1,118 @@
+/*
+ * This file is part of the Heritrix web crawler (crawler.archive.org).
+ *
+ * Licensed to the Internet Archive (IA) by one or more individual
+ * contributors.
+ *
+ * The IA licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.archive.modules.extractor;
+
+import java.io.BufferedInputStream;
+import java.io.BufferedReader;
+import java.io.ByteArrayInputStream;
+import java.io.File;
+import java.io.IOException;
+import java.io.InputStream;
+import java.io.InputStreamReader;
+import java.io.UnsupportedEncodingException;
+
+import java.util.HashSet;
+
+import java.util.Set;
+import java.util.regex.Matcher;
+
+import org.apache.commons.httpclient.URIException;
+import org.apache.commons.io.IOUtils;
+import org.archive.modules.CrawlURI;
+import org.archive.net.UURI;
+import org.archive.net.UURIFactory;
+import org.archive.util.Recorder;
+
+import com.itextpdf.text.pdf.PdfReader;
+
+public class ExtractorPDFContentTest extends ContentExtractorTestBase {
+
+ protected static final String TEST_RESOURCE_FILE_NAME = "ExtractorPDFContentTest1.pdf";
+
+
+ public void testA() throws URIException, UnsupportedEncodingException, IOException, InterruptedException{
+ CrawlURI testUri = createTestUri("http://www.example.com/fake.pdf", TEST_RESOURCE_FILE_NAME);
+ extractor.process(testUri);
+
+ Set expected = makeLinkSet(testUri, new String[]{"http://www.businessdictionary.com/definition/supervisor.html","http://management.about.com/od/policiesandprocedures/g/supervisor1.html"});
+ assertTrue(testUri.getOutLinks().containsAll(expected));
+ }
+
+ @Override
+ protected Extractor makeExtractor() {
+ ExtractorPDFContent result = new ExtractorPDFContent();
+ UriErrorLoggerModule ulm = new UnitTestUriLoggerModule();
+ result.setLoggerModule(ulm);
+ return (Extractor)result;
+ }
+ private Set makeLinkSet(CrawlURI sourceUri, String[] urlStrs) throws URIException {
+ HashSet linkSet = new HashSet();
+ for (String urlStr : urlStrs) {
+ linkSet.add(new Link(sourceUri.getUURI(),
+ UURIFactory.getInstance(urlStr),
+ HTMLLinkContext.NAVLINK_MISC, Hop.NAVLINK)
+ );
+ }
+ return linkSet;
+ }
+ private CrawlURI createTestUri(String urlStr, String resourceFileName) throws URIException,
+ UnsupportedEncodingException, IOException {
+ UURI testUuri = UURIFactory.getInstance(urlStr);
+ CrawlURI testUri = new CrawlURI(testUuri, null, null, LinkContext.NAVLINK_MISC);
+
+
+ //InputStream is = ExtractorPDFContentTest.class.getClassLoader().getResourceAsStream(resourceFileName);
+
+ //BufferedInputStream reader = new BufferedInputStream(new InputStreamReader(is));
+
+ File temp = File.createTempFile("test", ".tmp");
+ Recorder recorder = new Recorder(temp, 1024, 1024);
+ InputStream is = recorder.inputWrap(ExtractorPDFContentTest.class.getClassLoader().getResourceAsStream(resourceFileName));
+ recorder.markContentBegin();
+ for(int x = is.read(); x>=0; x=is.read());
+ is.close();
+
+//
+// byte[] b = content.getBytes(charset);
+// ByteArrayInputStream bais = new ByteArrayInputStream(b);
+// InputStream is = recorder.inputWrap(bais);
+// recorder.markContentBegin();
+// for (int x = is.read(); x >= 0; x = is.read());
+// is.close();
+// return recorder;
+
+
+
+// StringBuilder content = new StringBuilder();
+// String line = "";
+// while ((line = reader.readLine()) != null) {
+// content.append(line);
+// }
+// Recorder recorder = createRecorder(content.toString(), "UTF-8");
+// IOUtils.closeQuietly(is);
+
+ testUri.setContentType("application/pdf");
+ testUri.setFetchStatus(200);
+ testUri.setRecorder(recorder);
+ testUri.setContentSize(recorder.getResponseContentLength());
+ return testUri;
+ }
+
+
+}