diff --git a/contrib/pom.xml b/contrib/pom.xml index 6e048084..a9d16633 100644 --- a/contrib/pom.xml +++ b/contrib/pom.xml @@ -59,6 +59,11 @@ amqp-client 3.2.1 + + com.itextpdf + itextpdf + 5.5.0 + diff --git a/contrib/src/main/java/org/archive/modules/extractor/ExtractorPDFContent.java b/contrib/src/main/java/org/archive/modules/extractor/ExtractorPDFContent.java new file mode 100644 index 00000000..77b3378c --- /dev/null +++ b/contrib/src/main/java/org/archive/modules/extractor/ExtractorPDFContent.java @@ -0,0 +1,181 @@ +/* + * This file is part of the Heritrix web crawler (crawler.archive.org). + * + * Licensed to the Internet Archive (IA) by one or more individual + * contributors. + * + * The IA licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.archive.modules.extractor; + +import java.io.File; +import java.io.IOException; +import java.util.ArrayList; +import java.util.logging.Logger; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +import org.apache.commons.httpclient.URIException; +import org.archive.io.SinkHandlerLogThread; +import org.archive.modules.CrawlURI; +import org.archive.net.UURI; +import org.archive.net.UURIFactory; +import org.archive.util.FileUtils; + +import com.itextpdf.text.pdf.PdfReader; +import com.itextpdf.text.pdf.parser.PdfReaderContentParser; +import com.itextpdf.text.pdf.parser.SimpleTextExtractionStrategy; +import com.itextpdf.text.pdf.parser.TextExtractionStrategy; + +/** + * PDF Content Extractor. This will parse the text content of a PDF and apply a + * regex to search for links within the body of the text. + * + * @contributor adam + */ +public class ExtractorPDFContent extends ContentExtractor { + + @SuppressWarnings("unused") + private static final long serialVersionUID = 3L; + + private static final Logger LOGGER = + Logger.getLogger(ExtractorPDF.class.getName()); + + public static final Pattern URLPattern = Pattern.compile( + "(https?):\\/\\/"+ // protocol + "(([a-z0-9$_\\.\\+!\\*\\'\\(\\),;\\?&=-]|%[0-9a-f]{2})+"+ // username + "(:([a-z0-9$_\\.\\+!\\*\\'\\(\\),;\\?&=-]|%[0-9a-f]{2})+)?"+ // password + "@)?(?"+ // auth requires @ + ")((([a-z0-9]\\.|[a-z0-9][a-z0-9-]*[a-z0-9]\\.)*"+ // domain segments AND + "[a-z][a-z0-9-]*[a-z0-9]"+ // top level domain OR + "|((\\d|[1-9]\\d|1\\d{2}|2[0-4][0-9]|25[0-5])\\.){3}"+ + "(\\d|[1-9]\\d|1\\d{2}|2[0-4][0-9]|25[0-5])"+ // IP address + ")(:\\d+)?)"+ // port + "(((\\/+([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)*"+ // path + "(\\?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)?)?)?"+ // query string + "(\\n"+ // possible newline (seems to happen in pdfs) + "((\\/)?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)*"+ // continue possible path + "(\\?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)?"+ // or possible query + ")?"); + + /** + * The maximum size of PDF files to consider. PDFs larger than this + * maximum will not be searched for links. + */ + { + setMaxSizeToParse(10*1024*1024L); // 10MB + } + public long getMaxSizeToParse() { + return (Long) kp.get("maxSizeToParse"); + } + public void setMaxSizeToParse(long threshold) { + kp.put("maxSizeToParse",threshold); + } + + + public ExtractorPDFContent() { + } + + protected boolean innerExtract(CrawlURI curi){ + File tempFile; + int sn; + Thread thread = Thread.currentThread(); + if (thread instanceof SinkHandlerLogThread) { + sn = ((SinkHandlerLogThread)thread).getSerialNumber(); + } else { + sn = System.identityHashCode(thread); + } + try { + tempFile = File.createTempFile("tt" + sn , "tmp.pdf"); + } catch (IOException ioe) { + throw new RuntimeException(ioe); + } + + PdfReader documentReader; + ArrayList uris = new ArrayList(); + try { + documentReader = new PdfReader(curi.getRecorder().getContentReplayInputStream()); + String content = extractContent(documentReader); + Matcher matcher = URLPattern.matcher(content); + while(matcher.find()) { + uris.add(content.substring(matcher.start(),matcher.end())); + + //also add match without newline in case we are wrong + uris.add(matcher.group(1)+"://"+(matcher.group(2)!=null?matcher.group(2):"")+matcher.group(6)+matcher.group(13)); + } + + + + } catch (IOException e) { + curi.getNonFatalFailures().add(e); + return false; + } catch (RuntimeException e) { + // Truncated/corrupt PDFs may generate ClassCast exceptions, or + // other problems + curi.getNonFatalFailures().add(e); + return false; + } finally { + FileUtils.deleteSoonerOrLater(tempFile); + } + + if (uris.size()<1) { + return true; + } + + for (String uri: uris) { + try { + UURI src = curi.getUURI(); + UURI dest = UURIFactory.getInstance(uri); + LinkContext lc = LinkContext.NAVLINK_MISC; + Hop hop = Hop.NAVLINK; + Link out = new Link(src, dest, lc, hop); + curi.getOutLinks().add(out); + } catch (URIException e1) { + logUriError(e1, curi.getUURI(), uri); + } + } + + numberOfLinksExtracted.addAndGet(uris.size()); + + LOGGER.fine(curi+" has "+uris.size()+" links."); + // Set flag to indicate that link extraction is completed. + return true; + } + + public String extractContent(PdfReader documentReader){ + String content =""; + PdfReaderContentParser parser = new PdfReaderContentParser(documentReader); + TextExtractionStrategy strat; + for(int i=1; i< documentReader.getNumberOfPages(); i++) { + try { + strat = parser.processContent(i, new SimpleTextExtractionStrategy()); + content += strat.getResultantText(); + + } catch (IOException e) { + // TODO Auto-generated catch block + e.printStackTrace(); + } + } + return content; + } + @Override + protected boolean shouldExtract(CrawlURI uri) { + long max = getMaxSizeToParse(); + if (uri.getRecorder().getRecordedInput().getSize() > max) { + return false; + } + + String ct = uri.getContentType(); + return (ct != null) && (ct.startsWith("application/pdf")); + } +} diff --git a/contrib/src/test/java/org/archive/modules/extractor/ExtractorPDFContentTest.java b/contrib/src/test/java/org/archive/modules/extractor/ExtractorPDFContentTest.java new file mode 100644 index 00000000..850418ad --- /dev/null +++ b/contrib/src/test/java/org/archive/modules/extractor/ExtractorPDFContentTest.java @@ -0,0 +1,118 @@ +/* + * This file is part of the Heritrix web crawler (crawler.archive.org). + * + * Licensed to the Internet Archive (IA) by one or more individual + * contributors. + * + * The IA licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.archive.modules.extractor; + +import java.io.BufferedInputStream; +import java.io.BufferedReader; +import java.io.ByteArrayInputStream; +import java.io.File; +import java.io.IOException; +import java.io.InputStream; +import java.io.InputStreamReader; +import java.io.UnsupportedEncodingException; + +import java.util.HashSet; + +import java.util.Set; +import java.util.regex.Matcher; + +import org.apache.commons.httpclient.URIException; +import org.apache.commons.io.IOUtils; +import org.archive.modules.CrawlURI; +import org.archive.net.UURI; +import org.archive.net.UURIFactory; +import org.archive.util.Recorder; + +import com.itextpdf.text.pdf.PdfReader; + +public class ExtractorPDFContentTest extends ContentExtractorTestBase { + + protected static final String TEST_RESOURCE_FILE_NAME = "ExtractorPDFContentTest1.pdf"; + + + public void testA() throws URIException, UnsupportedEncodingException, IOException, InterruptedException{ + CrawlURI testUri = createTestUri("http://www.example.com/fake.pdf", TEST_RESOURCE_FILE_NAME); + extractor.process(testUri); + + Set expected = makeLinkSet(testUri, new String[]{"http://www.businessdictionary.com/definition/supervisor.html","http://management.about.com/od/policiesandprocedures/g/supervisor1.html"}); + assertTrue(testUri.getOutLinks().containsAll(expected)); + } + + @Override + protected Extractor makeExtractor() { + ExtractorPDFContent result = new ExtractorPDFContent(); + UriErrorLoggerModule ulm = new UnitTestUriLoggerModule(); + result.setLoggerModule(ulm); + return (Extractor)result; + } + private Set makeLinkSet(CrawlURI sourceUri, String[] urlStrs) throws URIException { + HashSet linkSet = new HashSet(); + for (String urlStr : urlStrs) { + linkSet.add(new Link(sourceUri.getUURI(), + UURIFactory.getInstance(urlStr), + HTMLLinkContext.NAVLINK_MISC, Hop.NAVLINK) + ); + } + return linkSet; + } + private CrawlURI createTestUri(String urlStr, String resourceFileName) throws URIException, + UnsupportedEncodingException, IOException { + UURI testUuri = UURIFactory.getInstance(urlStr); + CrawlURI testUri = new CrawlURI(testUuri, null, null, LinkContext.NAVLINK_MISC); + + + //InputStream is = ExtractorPDFContentTest.class.getClassLoader().getResourceAsStream(resourceFileName); + + //BufferedInputStream reader = new BufferedInputStream(new InputStreamReader(is)); + + File temp = File.createTempFile("test", ".tmp"); + Recorder recorder = new Recorder(temp, 1024, 1024); + InputStream is = recorder.inputWrap(ExtractorPDFContentTest.class.getClassLoader().getResourceAsStream(resourceFileName)); + recorder.markContentBegin(); + for(int x = is.read(); x>=0; x=is.read()); + is.close(); + +// +// byte[] b = content.getBytes(charset); +// ByteArrayInputStream bais = new ByteArrayInputStream(b); +// InputStream is = recorder.inputWrap(bais); +// recorder.markContentBegin(); +// for (int x = is.read(); x >= 0; x = is.read()); +// is.close(); +// return recorder; + + + +// StringBuilder content = new StringBuilder(); +// String line = ""; +// while ((line = reader.readLine()) != null) { +// content.append(line); +// } +// Recorder recorder = createRecorder(content.toString(), "UTF-8"); +// IOUtils.closeQuietly(is); + + testUri.setContentType("application/pdf"); + testUri.setFetchStatus(200); + testUri.setRecorder(recorder); + testUri.setContentSize(recorder.getResponseContentLength()); + return testUri; + } + + +}