mirror of
https://github.com/internetarchive/heritrix3.git
synced 2026-09-22 13:46:01 +00:00
Merge pull request #441 from netarchivesuite/iipc-master
Enabled configurable url-matching and extraction for sitemaps.
This commit is contained in:
@@ -30,6 +30,19 @@ public class ExtractorSitemap extends ContentExtractor {
|
||||
private static final Logger LOGGER = Logger
|
||||
.getLogger(ExtractorSitemap.class.getName());
|
||||
|
||||
/**
|
||||
* If urlPattern is not null then any url marked as a sitemap and matching the pattern is
|
||||
* assumed to be a sitemap. Otherwise the mime-type is checked (must be "text/xml" or "application/xml") and the
|
||||
* file is "sniffed" for the expected start of a sitemap file.
|
||||
*/
|
||||
private String urlPattern = null;
|
||||
|
||||
/**
|
||||
* If true, all urls in the sitemap file are extracted, regardless of whether or not they obey the scoping rules
|
||||
* specified in the sitemap protocol (https://www.sitemaps.org/protocol.html).
|
||||
*/
|
||||
private boolean enableLenientExtraction = false;
|
||||
|
||||
/* (non-Javadoc)
|
||||
* @see org.archive.modules.extractor.ContentExtractor#shouldExtract(org.archive.modules.CrawlURI)
|
||||
*/
|
||||
@@ -49,6 +62,10 @@ public class ExtractorSitemap extends ContentExtractor {
|
||||
}
|
||||
}
|
||||
|
||||
if (urlPattern != null && uri.getURI().matches(urlPattern)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Via content type:
|
||||
String mimeType = uri.getContentType();
|
||||
if (mimeType != null ) {
|
||||
@@ -120,9 +137,8 @@ public class ExtractorSitemap extends ContentExtractor {
|
||||
private AbstractSiteMap parseSiteMap(CrawlURI uri) {
|
||||
// The thing we will create:
|
||||
AbstractSiteMap sitemap = null;
|
||||
|
||||
// Be strict about URLs but allow partial extraction:
|
||||
SiteMapParser smp = new SiteMapParser(true, true);
|
||||
// allow partial extraction
|
||||
SiteMapParser smp = new SiteMapParser(!isEnableLenientExtraction(), true);
|
||||
// Parse it up:
|
||||
try {
|
||||
// Sitemaps are not supposed to be bigger than 50MB (according to
|
||||
@@ -184,4 +200,32 @@ public class ExtractorSitemap extends ContentExtractor {
|
||||
|
||||
}
|
||||
|
||||
public String getUrlPattern() {
|
||||
return urlPattern;
|
||||
}
|
||||
|
||||
/**
|
||||
* If urlPattern is not null then any url marked as a sitemap and matching the pattern is
|
||||
* assumed to be a sitemap. Otherwise the mime-type is checked (must be "text/xml" or "application/xml") and the
|
||||
* file is "sniffed" for the expected start of a sitemap file.
|
||||
* @param urlPattern the url pattern to match
|
||||
*/
|
||||
public void setUrlPattern(String urlPattern) {
|
||||
this.urlPattern = urlPattern;
|
||||
}
|
||||
|
||||
public boolean isEnableLenientExtraction() {
|
||||
return enableLenientExtraction;
|
||||
}
|
||||
|
||||
/**
|
||||
* If true, all urls in the sitemap file are extracted, regardless of whether or not they obey the scoping rules
|
||||
* specified in the sitemap protocol (https://www.sitemaps.org/protocol.html).
|
||||
* @param enableLenientExtraction whether to extract all urls
|
||||
*/
|
||||
public void setEnableLenientExtraction(boolean enableLenientExtraction) {
|
||||
this.enableLenientExtraction = enableLenientExtraction;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user