Adding ExtractorPDFContent to find nav links from the text content of PDFs

This commit is contained in:
Adam Miller
2014-04-14 16:02:35 -07:00
parent 67a7a4fb6c
commit a48a29f9bd
3 changed files with 304 additions and 0 deletions
+5
View File
@@ -59,6 +59,11 @@
<artifactId>amqp-client</artifactId>
<version>3.2.1</version>
</dependency>
<dependency>
<groupId>com.itextpdf</groupId>
<artifactId>itextpdf</artifactId>
<version>5.5.0</version>
</dependency>
</dependencies>
<repositories>
<repository>
@@ -0,0 +1,181 @@
/*
* This file is part of the Heritrix web crawler (crawler.archive.org).
*
* Licensed to the Internet Archive (IA) by one or more individual
* contributors.
*
* The IA licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package org.archive.modules.extractor;
import java.io.File;
import java.io.IOException;
import java.util.ArrayList;
import java.util.logging.Logger;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import org.apache.commons.httpclient.URIException;
import org.archive.io.SinkHandlerLogThread;
import org.archive.modules.CrawlURI;
import org.archive.net.UURI;
import org.archive.net.UURIFactory;
import org.archive.util.FileUtils;
import com.itextpdf.text.pdf.PdfReader;
import com.itextpdf.text.pdf.parser.PdfReaderContentParser;
import com.itextpdf.text.pdf.parser.SimpleTextExtractionStrategy;
import com.itextpdf.text.pdf.parser.TextExtractionStrategy;
/**
* PDF Content Extractor. This will parse the text content of a PDF and apply a
* regex to search for links within the body of the text.
*
* @contributor adam
*/
public class ExtractorPDFContent extends ContentExtractor {
@SuppressWarnings("unused")
private static final long serialVersionUID = 3L;
private static final Logger LOGGER =
Logger.getLogger(ExtractorPDF.class.getName());
public static final Pattern URLPattern = Pattern.compile(
"(https?):\\/\\/"+ // protocol
"(([a-z0-9$_\\.\\+!\\*\\'\\(\\),;\\?&=-]|%[0-9a-f]{2})+"+ // username
"(:([a-z0-9$_\\.\\+!\\*\\'\\(\\),;\\?&=-]|%[0-9a-f]{2})+)?"+ // password
"@)?(?"+ // auth requires @
")((([a-z0-9]\\.|[a-z0-9][a-z0-9-]*[a-z0-9]\\.)*"+ // domain segments AND
"[a-z][a-z0-9-]*[a-z0-9]"+ // top level domain OR
"|((\\d|[1-9]\\d|1\\d{2}|2[0-4][0-9]|25[0-5])\\.){3}"+
"(\\d|[1-9]\\d|1\\d{2}|2[0-4][0-9]|25[0-5])"+ // IP address
")(:\\d+)?)"+ // port
"(((\\/+([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)*"+ // path
"(\\?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)?)?)?"+ // query string
"(\\n"+ // possible newline (seems to happen in pdfs)
"((\\/)?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)*"+ // continue possible path
"(\\?([a-z0-9$_\\.\\+!\\*\\'\\(\\),;:@&=-]|%[0-9a-f]{2})*)?"+ // or possible query
")?");
/**
* The maximum size of PDF files to consider. PDFs larger than this
* maximum will not be searched for links.
*/
{
setMaxSizeToParse(10*1024*1024L); // 10MB
}
public long getMaxSizeToParse() {
return (Long) kp.get("maxSizeToParse");
}
public void setMaxSizeToParse(long threshold) {
kp.put("maxSizeToParse",threshold);
}
public ExtractorPDFContent() {
}
protected boolean innerExtract(CrawlURI curi){
File tempFile;
int sn;
Thread thread = Thread.currentThread();
if (thread instanceof SinkHandlerLogThread) {
sn = ((SinkHandlerLogThread)thread).getSerialNumber();
} else {
sn = System.identityHashCode(thread);
}
try {
tempFile = File.createTempFile("tt" + sn , "tmp.pdf");
} catch (IOException ioe) {
throw new RuntimeException(ioe);
}
PdfReader documentReader;
ArrayList<String> uris = new ArrayList<String>();
try {
documentReader = new PdfReader(curi.getRecorder().getContentReplayInputStream());
String content = extractContent(documentReader);
Matcher matcher = URLPattern.matcher(content);
while(matcher.find()) {
uris.add(content.substring(matcher.start(),matcher.end()));
//also add match without newline in case we are wrong
uris.add(matcher.group(1)+"://"+(matcher.group(2)!=null?matcher.group(2):"")+matcher.group(6)+matcher.group(13));
}
} catch (IOException e) {
curi.getNonFatalFailures().add(e);
return false;
} catch (RuntimeException e) {
// Truncated/corrupt PDFs may generate ClassCast exceptions, or
// other problems
curi.getNonFatalFailures().add(e);
return false;
} finally {
FileUtils.deleteSoonerOrLater(tempFile);
}
if (uris.size()<1) {
return true;
}
for (String uri: uris) {
try {
UURI src = curi.getUURI();
UURI dest = UURIFactory.getInstance(uri);
LinkContext lc = LinkContext.NAVLINK_MISC;
Hop hop = Hop.NAVLINK;
Link out = new Link(src, dest, lc, hop);
curi.getOutLinks().add(out);
} catch (URIException e1) {
logUriError(e1, curi.getUURI(), uri);
}
}
numberOfLinksExtracted.addAndGet(uris.size());
LOGGER.fine(curi+" has "+uris.size()+" links.");
// Set flag to indicate that link extraction is completed.
return true;
}
public String extractContent(PdfReader documentReader){
String content ="";
PdfReaderContentParser parser = new PdfReaderContentParser(documentReader);
TextExtractionStrategy strat;
for(int i=1; i< documentReader.getNumberOfPages(); i++) {
try {
strat = parser.processContent(i, new SimpleTextExtractionStrategy());
content += strat.getResultantText();
} catch (IOException e) {
// TODO Auto-generated catch block
e.printStackTrace();
}
}
return content;
}
@Override
protected boolean shouldExtract(CrawlURI uri) {
long max = getMaxSizeToParse();
if (uri.getRecorder().getRecordedInput().getSize() > max) {
return false;
}
String ct = uri.getContentType();
return (ct != null) && (ct.startsWith("application/pdf"));
}
}
@@ -0,0 +1,118 @@
/*
* This file is part of the Heritrix web crawler (crawler.archive.org).
*
* Licensed to the Internet Archive (IA) by one or more individual
* contributors.
*
* The IA licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package org.archive.modules.extractor;
import java.io.BufferedInputStream;
import java.io.BufferedReader;
import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.io.UnsupportedEncodingException;
import java.util.HashSet;
import java.util.Set;
import java.util.regex.Matcher;
import org.apache.commons.httpclient.URIException;
import org.apache.commons.io.IOUtils;
import org.archive.modules.CrawlURI;
import org.archive.net.UURI;
import org.archive.net.UURIFactory;
import org.archive.util.Recorder;
import com.itextpdf.text.pdf.PdfReader;
public class ExtractorPDFContentTest extends ContentExtractorTestBase {
protected static final String TEST_RESOURCE_FILE_NAME = "ExtractorPDFContentTest1.pdf";
public void testA() throws URIException, UnsupportedEncodingException, IOException, InterruptedException{
CrawlURI testUri = createTestUri("http://www.example.com/fake.pdf", TEST_RESOURCE_FILE_NAME);
extractor.process(testUri);
Set<Link> expected = makeLinkSet(testUri, new String[]{"http://www.businessdictionary.com/definition/supervisor.html","http://management.about.com/od/policiesandprocedures/g/supervisor1.html"});
assertTrue(testUri.getOutLinks().containsAll(expected));
}
@Override
protected Extractor makeExtractor() {
ExtractorPDFContent result = new ExtractorPDFContent();
UriErrorLoggerModule ulm = new UnitTestUriLoggerModule();
result.setLoggerModule(ulm);
return (Extractor)result;
}
private Set<Link> makeLinkSet(CrawlURI sourceUri, String[] urlStrs) throws URIException {
HashSet<Link> linkSet = new HashSet<Link>();
for (String urlStr : urlStrs) {
linkSet.add(new Link(sourceUri.getUURI(),
UURIFactory.getInstance(urlStr),
HTMLLinkContext.NAVLINK_MISC, Hop.NAVLINK)
);
}
return linkSet;
}
private CrawlURI createTestUri(String urlStr, String resourceFileName) throws URIException,
UnsupportedEncodingException, IOException {
UURI testUuri = UURIFactory.getInstance(urlStr);
CrawlURI testUri = new CrawlURI(testUuri, null, null, LinkContext.NAVLINK_MISC);
//InputStream is = ExtractorPDFContentTest.class.getClassLoader().getResourceAsStream(resourceFileName);
//BufferedInputStream reader = new BufferedInputStream(new InputStreamReader(is));
File temp = File.createTempFile("test", ".tmp");
Recorder recorder = new Recorder(temp, 1024, 1024);
InputStream is = recorder.inputWrap(ExtractorPDFContentTest.class.getClassLoader().getResourceAsStream(resourceFileName));
recorder.markContentBegin();
for(int x = is.read(); x>=0; x=is.read());
is.close();
//
// byte[] b = content.getBytes(charset);
// ByteArrayInputStream bais = new ByteArrayInputStream(b);
// InputStream is = recorder.inputWrap(bais);
// recorder.markContentBegin();
// for (int x = is.read(); x >= 0; x = is.read());
// is.close();
// return recorder;
// StringBuilder content = new StringBuilder();
// String line = "";
// while ((line = reader.readLine()) != null) {
// content.append(line);
// }
// Recorder recorder = createRecorder(content.toString(), "UTF-8");
// IOUtils.closeQuietly(is);
testUri.setContentType("application/pdf");
testUri.setFetchStatus(200);
testUri.setRecorder(recorder);
testUri.setContentSize(recorder.getResponseContentLength());
return testUri;
}
}