diff --git a/commons/src/test/resources/log4j.xml b/commons/src/test/resources/log4j.xml index eddf51e2..475b130e 100644 --- a/commons/src/test/resources/log4j.xml +++ b/commons/src/test/resources/log4j.xml @@ -11,6 +11,10 @@ + + + + diff --git a/contrib/src/main/java/org/archive/modules/extractor/ExtractorYoutubeDL.java b/contrib/src/main/java/org/archive/modules/extractor/ExtractorYoutubeDL.java new file mode 100644 index 00000000..a1f2df97 --- /dev/null +++ b/contrib/src/main/java/org/archive/modules/extractor/ExtractorYoutubeDL.java @@ -0,0 +1,404 @@ +/* + * This file is part of the Heritrix web crawler (crawler.archive.org). + * + * Licensed to the Internet Archive (IA) by one or more individual + * contributors. + * + * The IA licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.archive.modules.extractor; + +import java.io.IOException; +import java.io.InputStreamReader; +import java.io.Reader; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.Callable; +import java.util.concurrent.ExecutionException; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.Future; +import java.util.logging.Level; +import java.util.logging.Logger; + +import org.apache.commons.httpclient.URIException; +import org.archive.crawler.reporting.CrawlerLoggerModule; +import org.archive.modules.CoreAttributeConstants; +import org.archive.modules.CrawlURI; +import org.archive.net.UURI; +import org.archive.net.UURIFactory; +import org.archive.util.ArchiveUtils; +import org.archive.util.MimetypeUtils; +import org.springframework.beans.factory.annotation.Autowired; +import org.springframework.context.Lifecycle; + +import com.google.gson.JsonObject; +import com.google.gson.JsonParseException; +import com.google.gson.JsonStreamParser; + +/** + * Extracts links to media by running youtube-dl in a subprocess. Runs only on + * html. + * + *

+ * Keeps a log of containing pages and media captured as a result of youtube-dl + * extraction. The format of the log is as follows: + * + *

[timestamp] [media-http-status] [media-length] [media-mimetype] [media-digest] [media-timestamp] [media-url] [annotation] [containing-page-digest] [containing-page-timestamp] [containing-page-url] [seed-url]
+ * + *

+ * For containing pages, all of the {@code media-*} fields have the value + * {@code "-"}, and the annotation field looks like {@code "youtube-dl:3"}, + * meaning that ExtractorYoutubeDL extracted 3 media links from the page. + * + *

+ * For media, the annotation field looks like {@code "youtube-dl:1/3"}, meaning + * this is the first of three media links extracted from the containing page. + * The intention is to use this for playback. The rest of the fields included in + * the log were also chosen to support creation of an index of media by + * containing page, to be used for playback. + * + * @author nlevitt + */ +public class ExtractorYoutubeDL extends Extractor implements Lifecycle { + private static Logger logger = + Logger.getLogger(ExtractorYoutubeDL.class.getName()); + + protected static final String YDL_CONTAINING_PAGE_DIGEST = "ydl-containing-page-digest"; + protected static final String YDL_CONTAINING_PAGE_TIMESTAMP = "ydl-containing-page-timestamp"; + protected static final String YDL_CONTAINING_PAGE_URI = "ydl-containing-page-uri"; + + protected transient Logger ydlLogger = null; + + protected CrawlerLoggerModule crawlerLoggerModule; + public CrawlerLoggerModule getCrawlerLoggerModule() { + return this.crawlerLoggerModule; + } + @Autowired + public void setCrawlerLoggerModule(CrawlerLoggerModule crawlerLoggerModule) { + this.crawlerLoggerModule = crawlerLoggerModule; + } + + @Override + public void start() { + if (!isRunning) { + // loggerModule.start() creates the log directory, and it might be + // possible for this module to start before loggerModule, so we need + // to run this here to prevent an exception + getCrawlerLoggerModule().start(); + + ydlLogger = getCrawlerLoggerModule().setupSimpleLog(getBeanName()); + } + super.start(); + } + + protected String readToEnd(Reader r) throws IOException { + StringBuilder buf = new StringBuilder(); + char[] rbuf = new char[4096]; + while (true) { + int n = r.read(rbuf); + if (n < 0) { + return buf.toString(); + } + buf.append(rbuf, 0, n); + } + } + + /** + * - If {@code uri} is annotated "youtube-dl" and is a 3xx (redirect), + * find the redirect among the outlinks and add the "youtube-dl" + * annotation to it as well, and also make a note of the containing page + * inside the CrawlURI. {@link ExtractorHTTP} needs to have have run + * already. + * + * - If {@code uri} is annotated "youtube-dl" and is an actual video + * download, log a line to ExtractorYoutubeDL.log + * + * - If {@link #shouldExtract(CrawlURI)}, do youtube-dl extraction. + */ + @Override + protected void extract(CrawlURI uri) { + String ydlAnnotation = findYdlAnnotation(uri); + if (ydlAnnotation != null) { + if (uri.getFetchStatus() >= 300 && uri.getFetchStatus() < 400) { + doRedirectInheritance(uri, ydlAnnotation); + } else { + logCapturedVideo(uri, ydlAnnotation); + } + } else { + List ydlJsons = runYoutubeDL(uri); + if (ydlJsons != null && !ydlJsons.isEmpty()) { + for (JsonObject json: ydlJsons) { + if (json.get("url") != null) { + String videoUrl = json.get("url").getAsString(); + addVideoOutlink(uri, json, videoUrl); + } + } + String annotation = "youtube-dl:" + ydlJsons.size(); + uri.getAnnotations().add(annotation); + logContainingPage(uri, annotation); + } + } + } + + protected void addVideoOutlink(CrawlURI uri, JsonObject json, + String videoUrl) { + try { + UURI dest = UURIFactory.getInstance(uri.getUURI(), videoUrl); + CrawlURI link = uri.createCrawlURI(dest, LinkContext.EMBED_MISC, + Hop.EMBED); + + // annotation + String annotation = "youtube-dl:1/1"; + if (!json.get("playlist_index").isJsonNull()) { + annotation = "youtube-dl:" + json.get("playlist_index") + "/" + + json.get("n_entries"); + } + link.getAnnotations().add(annotation); + + // save info unambiguously identifying containing page capture + link.getData().put(YDL_CONTAINING_PAGE_URI, uri.toString()); + link.getData().put(YDL_CONTAINING_PAGE_TIMESTAMP, + ArchiveUtils.get17DigitDate(uri.getFetchBeginTime())); + link.getData().put(YDL_CONTAINING_PAGE_DIGEST, + uri.getContentDigestSchemeString()); + + uri.getOutLinks().add(link); + } catch (URIException e) { + logUriError(e, uri.getUURI(), videoUrl); + } + } + + protected String findYdlAnnotation(CrawlURI uri) { + for (String annotation: uri.getAnnotations()) { + if (annotation.startsWith("youtube-dl:")) { + return annotation; + } + } + return null; + } + + protected void logCapturedVideo(CrawlURI uri, String ydlAnnotation) { + // "length" logic copied from UriProcessingFormatter + String length = "-"; + if (uri.isHttpTransaction()) { + if(uri.getContentLength() >= 0) { + length = Long.toString(uri.getContentLength()); + } else if (uri.getContentSize() > 0) { + length = Long.toString(uri.getContentSize()); + } + } else { + if (uri.getContentSize() > 0) { + length = Long.toString(uri.getContentSize()); + } + } + + String seed = uri.containsDataKey(CoreAttributeConstants.A_SOURCE_TAG) + ? uri.getSourceTag() + : "-"; + + ydlLogger.info( + uri.getFetchStatus() + + " " + length + + " " + MimetypeUtils.truncate(uri.getContentType()) + + " " + uri.getContentDigestSchemeString() + + " " + ArchiveUtils.get17DigitDate(uri.getFetchBeginTime()) + + " " + uri + + " " + ydlAnnotation + + " " + uri.getData().get(YDL_CONTAINING_PAGE_DIGEST) + + " " + uri.getData().get(YDL_CONTAINING_PAGE_TIMESTAMP) + + " " + uri.getData().get(YDL_CONTAINING_PAGE_URI) + + " " + seed); + } + + protected void logContainingPage(CrawlURI uri, String annotation) { + String seed = uri.containsDataKey(CoreAttributeConstants.A_SOURCE_TAG) + ? uri.getSourceTag() + : "-"; + + ydlLogger.info( + "- - - - - -" + + " " + annotation + + " " + uri.getContentDigestSchemeString() + + " " + ArchiveUtils.get17DigitDate(uri.getFetchBeginTime()) + + " " + uri + + " " + seed); + } + + protected void doRedirectInheritance(CrawlURI uri, String ydlAnnotation) { + for (CrawlURI link: uri.getOutLinks()) { + if ("R".equals(link.getLastHop())) { + link.getAnnotations().add(ydlAnnotation); + link.getData().put(YDL_CONTAINING_PAGE_URI, + uri.getData().get(YDL_CONTAINING_PAGE_URI)); + link.getData().put(YDL_CONTAINING_PAGE_TIMESTAMP, + uri.getData().get(YDL_CONTAINING_PAGE_TIMESTAMP)); + link.getData().put(YDL_CONTAINING_PAGE_DIGEST, + uri.getData().get(YDL_CONTAINING_PAGE_DIGEST)); + } + } + } + + static protected class ProcessOutput { + public String stdout; + public String stderr; + } + + // read stdout in this thread, stderr in separate thread + // see https://github.com/internetarchive/heritrix3/pull/257/files#r279990349 + protected ProcessOutput readOutput(Process proc) throws IOException { + ProcessOutput output = new ProcessOutput(); + + Reader err = new InputStreamReader(proc.getErrorStream(), "UTF-8"); + InputStreamReader out = new InputStreamReader(proc.getInputStream(), "UTF-8"); + ExecutorService threadPool = Executors.newSingleThreadExecutor(); + + Future future = threadPool.submit(new Callable() { + @Override + public String call() throws IOException { + return readToEnd(err); + } + }); + + output.stdout = readToEnd(out); + + try { + output.stderr = future.get(); + } catch (InterruptedException e) { + throw new IOException(e); // :shrug: + } catch (ExecutionException e) { + if (e.getCause() instanceof IOException) { + throw (IOException) e.getCause(); + } else { + throw new IOException(e); + } + } finally { + threadPool.shutdown(); + } + + return output; + } + + /** + * + * @param uri + * @return list of json blobs returned by {@code youtube-dl --dump-json}, or + * empty list if no videos found, or failure + */ + protected List runYoutubeDL(CrawlURI uri) { + /* + * --format=best + * + * best: Select the best quality format represented by a single file + * with video and audio. + * https://github.com/ytdl-org/youtube-dl/blob/master/README.md#format-selection + */ + ProcessBuilder pb = new ProcessBuilder("youtube-dl", "--ignore-config", + "--simulate", "--dump-json", "--format=best", uri.toString()); + logger.fine("running " + pb.command()); + + Process proc = null; + try { + proc = pb.start(); + } catch (IOException e) { + logger.log(Level.WARNING, "youtube-dl failed " + pb.command(), e); + return null; + } + + ProcessOutput output; + try { + output = readOutput(proc); + } catch (IOException e) { + logger.log(Level.WARNING, + "problem reading output from youtube-dl " + pb.command(), + e); + return null; + } + + try { + if (proc.waitFor() != 0) { + /* + * youtube-dl is noisy when it fails to find a video. I guess + * the assumption is that you're running it on pages you know + * have videos. We could be hiding real errors in some cases + * but it's just too much noise to log this at WARNING level. + */ + logger.fine("youtube-dl exited with status " + + proc.waitFor() + " " + pb.command() + + "\n=== stdout ===\n" + output.stdout + + "\n=== stderr ===\n" + output.stderr); + return null; + } + } catch (InterruptedException e) { + proc.destroyForcibly(); + } + + List ydlJsons = new ArrayList(); + JsonStreamParser parser = new JsonStreamParser(output.stdout); + try { + while (parser.hasNext()) { + ydlJsons.add((JsonObject) parser.next()); + } + } catch (JsonParseException e) { + // sometimes we get no output at all from youtube-dl, which + // manifests as a JsonIOException + logger.log(Level.FINE, + "problem parsing json from youtube-dl " + pb.command() + + "\n=== stdout ===\n" + output.stdout + + "\n=== stderr ===\n" + output.stderr, + e); + return null; + } + + return ydlJsons; + } + + @Override + protected boolean shouldProcess(CrawlURI uri) { + // We have some special sauce (not actually extraction) to apply to + // "youtube-dl"-annotated urls. See extract(). + if (findYdlAnnotation(uri) != null) { + return true; + } + + // Otherwise, check if we want to run youtube-dl on the url. + return shouldExtract(uri); + } + + /** + * Returns {@code true} if we should run youtube-dl on this url. We run + * youtube-dl on html 200s that are not too huge. + */ + protected boolean shouldExtract(CrawlURI uri) { + if (uri.getFetchStatus() != 200) { + return false; + } + + // see https://github.com/internetarchive/brozzler/blob/65fad5e8b/brozzler/ydl.py#L48 + if (uri.getContentLength() <= 0 || uri.getContentLength() >= 200000000) { + return false; + } + + String mime = uri.getContentType().toLowerCase(); + if (mime.startsWith("text/html") + || mime.startsWith("application/xhtml") + || mime.startsWith("text/vnd.wap.wml") + || mime.startsWith("application/vnd.wap.wml") + || mime.startsWith("application/vnd.wap.xhtml")) { + return true; + } + + return false; + } +} diff --git a/contrib/src/main/resources/log4j.xml b/contrib/src/main/resources/log4j.xml index eddf51e2..475b130e 100644 --- a/contrib/src/main/resources/log4j.xml +++ b/contrib/src/main/resources/log4j.xml @@ -11,6 +11,10 @@ + + + + diff --git a/contrib/src/test/resources/log4j.xml b/contrib/src/test/resources/log4j.xml index eddf51e2..475b130e 100644 --- a/contrib/src/test/resources/log4j.xml +++ b/contrib/src/test/resources/log4j.xml @@ -11,6 +11,10 @@ + + + + diff --git a/dist/src/test/resources/log4j.xml b/dist/src/test/resources/log4j.xml index eddf51e2..475b130e 100644 --- a/dist/src/test/resources/log4j.xml +++ b/dist/src/test/resources/log4j.xml @@ -11,6 +11,10 @@ + + + + diff --git a/engine/src/main/java/org/archive/crawler/io/UriProcessingFormatter.java b/engine/src/main/java/org/archive/crawler/io/UriProcessingFormatter.java index 5c12e4ae..3af99d35 100644 --- a/engine/src/main/java/org/archive/crawler/io/UriProcessingFormatter.java +++ b/engine/src/main/java/org/archive/crawler/io/UriProcessingFormatter.java @@ -151,7 +151,6 @@ extends Formatter implements Preformatter, CoreAttributeConstants { } if (logExtraInfo) { - // XXX would we rather have "-" if info's empty? buffer.append(" ").append(curi.getExtraInfo()); } diff --git a/engine/src/test/resources/log4j.xml b/engine/src/test/resources/log4j.xml index eddf51e2..475b130e 100644 --- a/engine/src/test/resources/log4j.xml +++ b/engine/src/test/resources/log4j.xml @@ -11,6 +11,10 @@ + + + + diff --git a/modules/src/main/java/org/archive/modules/CrawlURI.java b/modules/src/main/java/org/archive/modules/CrawlURI.java index 46a935b2..92d2870f 100644 --- a/modules/src/main/java/org/archive/modules/CrawlURI.java +++ b/modules/src/main/java/org/archive/modules/CrawlURI.java @@ -31,9 +31,6 @@ import static org.archive.modules.CoreAttributeConstants.A_HTTP_RESPONSE_HEADERS import static org.archive.modules.CoreAttributeConstants.A_NONFATAL_ERRORS; import static org.archive.modules.CoreAttributeConstants.A_PREREQUISITE_URI; import static org.archive.modules.CoreAttributeConstants.A_SOURCE_TAG; -import static org.archive.modules.CoreAttributeConstants.A_SUBMIT_DATA; -import static org.archive.modules.CoreAttributeConstants.A_SUBMIT_ENCTYPE; -import static org.archive.modules.CoreAttributeConstants.A_WARC_RESPONSE_HEADERS; import static org.archive.modules.SchedulingConstants.NORMAL; import static org.archive.modules.fetcher.FetchStatusCodes.S_BLOCKED_BY_CUSTOM_PROCESSOR; import static org.archive.modules.fetcher.FetchStatusCodes.S_BLOCKED_BY_USER; @@ -76,7 +73,6 @@ import java.util.LinkedHashSet; import java.util.List; import java.util.Map; import java.util.Set; -import java.util.concurrent.CopyOnWriteArrayList; import java.util.logging.Level; import java.util.logging.Logger; @@ -248,15 +244,6 @@ implements Reporter, Serializable, OverlayContext, Comparable { * buggy */ protected long ordinal; - - /** - * Array to hold keys of data members that persist across URI processings. - * Any key mentioned in this list will not be cleared out at the end - * of a pass down the processing chain. - */ - private static final Collection persistentKeys - = new CopyOnWriteArrayList( - new String [] {A_CREDENTIALS_KEY, A_HTTP_AUTH_CHALLENGES, A_SUBMIT_DATA, A_WARC_RESPONSE_HEADERS, A_ANNOTATIONS, A_SUBMIT_ENCTYPE}); /** maximum length for pathFromSeed/hopsPath; longer truncated with leading counter **/ private static final int MAX_HOPS_DISPLAYED = 50; @@ -875,8 +862,6 @@ implements Reporter, Serializable, OverlayContext, Comparable { this.contentLength = UNCALCULATED; // Clear 'links extracted' flag. this.linkExtractorFinished = false; - // Clean the data map of all but registered permanent members. - this.data = getPersistentDataMap(); extraInfo = null; outLinks = null; @@ -886,23 +871,6 @@ implements Reporter, Serializable, OverlayContext, Comparable { // XXX er uh surprised this wasn't here before? fetchType = FetchType.UNKNOWN; } - - public Map getPersistentDataMap() { - if (data == null) { - return null; - } - Map result = new HashMap(getData()); - Set retain = new HashSet(persistentKeys); - - if (containsDataKey(A_HERITABLE_KEYS)) { - @SuppressWarnings("unchecked") - HashSet heritable = (HashSet)getData().get(A_HERITABLE_KEYS); - retain.addAll(heritable); - } - - result.keySet().retainAll(retain); - return result; - } /** * @return Credential avatars. Null if none set. @@ -1132,39 +1100,6 @@ implements Reporter, Serializable, OverlayContext, Comparable { } return (UURI)getData().get(A_HTML_BASE); } - - public static Collection getPersistentDataKeys() { - return persistentKeys; - } - - /** - * Add the key of items you want to persist across - * processings. - * @param s Key to add. - */ - public void addPersistentDataMapKey(String s) { - if (!persistentKeys.contains(s)) { - addDataPersistentMember(s); - } - } - - /** - * Add the key of data map items you want to persist across - * processings. - * @param key Key to add. - */ - public static void addDataPersistentMember(String key) { - persistentKeys.add(key); - } - - /** - * Remove the key from those data map members persisted. - * @param key Key to remove. - * @return True if list contained the element. - */ - public static boolean removeDataPersistentMember(String key) { - return persistentKeys.remove(key); - } private void writeObject(ObjectOutputStream stream) throws IOException { stream.defaultWriteObject(); @@ -1782,9 +1717,7 @@ implements Reporter, Serializable, OverlayContext, Comparable { return containsDataKey(A_FORCE_RETIRE) && (Boolean)getData().get(A_FORCE_RETIRE); } - - - + protected JSONObject extraInfo; public JSONObject getExtraInfo() { diff --git a/modules/src/main/java/org/archive/modules/recrawl/FetchHistoryProcessor.java b/modules/src/main/java/org/archive/modules/recrawl/FetchHistoryProcessor.java index 7bcd5ccf..889fa05c 100644 --- a/modules/src/main/java/org/archive/modules/recrawl/FetchHistoryProcessor.java +++ b/modules/src/main/java/org/archive/modules/recrawl/FetchHistoryProcessor.java @@ -67,7 +67,6 @@ public class FetchHistoryProcessor extends Processor { @Override protected void innerProcess(CrawlURI puri) throws InterruptedException { CrawlURI curi = (CrawlURI) puri; - curi.addPersistentDataMapKey(A_FETCH_HISTORY); HashMap latestFetch = new HashMap(); // save status diff --git a/modules/src/main/java/org/archive/modules/recrawl/PersistLogProcessor.java b/modules/src/main/java/org/archive/modules/recrawl/PersistLogProcessor.java index d03cb5af..f5dada4d 100644 --- a/modules/src/main/java/org/archive/modules/recrawl/PersistLogProcessor.java +++ b/modules/src/main/java/org/archive/modules/recrawl/PersistLogProcessor.java @@ -96,7 +96,7 @@ implements Checkpointable, Lifecycle { protected void innerProcess(CrawlURI curi) { log.writeLine(persistKeyFor(curi), " ", new String(Base64.encodeBase64( - SerializationUtils.serialize((Serializable)curi.getPersistentDataMap())))); + SerializationUtils.serialize((Serializable)curi.getData())))); } public void startCheckpoint(Checkpoint checkpointInProgress) {} diff --git a/modules/src/main/java/org/archive/modules/recrawl/PersistStoreProcessor.java b/modules/src/main/java/org/archive/modules/recrawl/PersistStoreProcessor.java index 9bb7da92..73468c3b 100644 --- a/modules/src/main/java/org/archive/modules/recrawl/PersistStoreProcessor.java +++ b/modules/src/main/java/org/archive/modules/recrawl/PersistStoreProcessor.java @@ -40,7 +40,7 @@ public class PersistStoreProcessor extends PersistOnlineProcessor @Override protected void innerProcess(CrawlURI curi) throws InterruptedException { - store.put(persistKeyFor(curi),curi.getPersistentDataMap()); + store.put(persistKeyFor(curi), curi.getData()); } @Override diff --git a/modules/src/test/resources/log4j.xml b/modules/src/test/resources/log4j.xml index eddf51e2..475b130e 100644 --- a/modules/src/test/resources/log4j.xml +++ b/modules/src/test/resources/log4j.xml @@ -11,6 +11,10 @@ + + + +