diff --git a/contrib/src/main/java/org/archive/modules/extractor/ExtractorChrome.java b/contrib/src/main/java/org/archive/modules/extractor/ExtractorChrome.java deleted file mode 100644 index ae8b48d2..00000000 --- a/contrib/src/main/java/org/archive/modules/extractor/ExtractorChrome.java +++ /dev/null @@ -1,435 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.modules.extractor; - -import org.apache.commons.io.IOUtils; -import org.archive.crawler.event.CrawlURIDispositionEvent; -import org.archive.crawler.framework.CrawlController; -import org.archive.crawler.framework.Frontier; -import org.archive.modules.CrawlURI; -import org.archive.modules.Processor; -import org.archive.modules.ProcessorChain; -import org.archive.net.chrome.*; -import org.archive.spring.KeyedProperties; -import org.archive.util.Recorder; -import org.archive.util.UriUtils; -import org.json.JSONArray; -import org.springframework.context.ApplicationEventPublisher; - -import java.io.*; -import java.util.ArrayList; -import java.util.Arrays; -import java.util.List; -import java.util.Map; -import java.util.concurrent.ExecutionException; -import java.util.concurrent.Semaphore; -import java.util.concurrent.TimeUnit; -import java.util.concurrent.TimeoutException; -import java.util.concurrent.atomic.AtomicInteger; -import java.util.concurrent.atomic.AtomicLong; -import java.util.logging.Logger; -import java.util.regex.Pattern; - -import static java.nio.charset.StandardCharsets.US_ASCII; -import static java.util.Collections.enumeration; -import static java.util.logging.Level.*; -import static java.util.regex.Pattern.CASE_INSENSITIVE; -import static org.archive.crawler.event.CrawlURIDispositionEvent.Disposition.FAILED; -import static org.archive.crawler.event.CrawlURIDispositionEvent.Disposition.SUCCEEDED; -import static org.archive.modules.CoreAttributeConstants.A_HTTP_RESPONSE_HEADERS; -import static org.archive.modules.CrawlURI.FetchType.*; - -/** - * Extracts links using a web browser via the Chrome Devtools Protocol. - *

- * To use, first define this as a top-level bean: - *

- * <bean id="extractorChrome" class="org.archive.modules.extractor.ExtractorChrome">
- *   <!-- <property name="captureRequests" value="true" /> -->
- *   <!-- <property name="devtoolsUrl" value="ws://127.0.0.1:1234/devtools/browser/2bc831e8-6c02-4c9b-affd-14c93b8579d7" /> -->
- *   <!-- <property name="executable" value="chromium-browser" /> -->
- *   <!-- <property name="loadTimeoutSeconds" value="30" /> -->
- *   <!-- <property name="maxOpenWindows" value="16" /> -->
- *   <!-- <property name="windowWidth" value="1366" /> -->
- *   <!-- <property name="windowWidth" value="768" /> -->
- * </bean>
- * 
- * Then add <ref bean="extractorChrome"/> to the fetch chain before extractorHTML. - *

- * By default an instance of the browser will be run as a subprocess for the duration of the crawl. Alternatively set - * devtoolsUrl to connect to an existing instance of the browser (run with - * --headless --remote-debugging-port=1234). - */ -public class ExtractorChrome extends ContentExtractor { - private static final Logger logger = Logger.getLogger(ExtractorChrome.class.getName()); - private static final AtomicLong nextRecorderId = new AtomicLong(); - - private static final Pattern TRANSFER_ENCODING_RE = Pattern.compile("\r\nTransfer-Encoding:[^\n\r]+", CASE_INSENSITIVE); - - /** - * The maximum number of browser windows that are allowed to be opened simultaneously. Feel free to increase this - * if you have lots of RAM available. - */ - private int maxOpenWindows = 16; - - /** - * URL of the devtools server to connect. If null a new browser process will be launched. - */ - private String devtoolsUrl = null; - - /** - * The name or path to the browser executable. If null common locations will be searched. Not used if devtoolsUrl - * is set. - */ - private String executable = null; - - /** - * Extra command-line options passed to the browser process. Not used if devtoolsUrl is null. - */ - private List commandLineOptions = new ArrayList<>(); - - /** - * Width of the browser window. - */ - private int windowWidth = 1366; - - /** - * Height of the browser window. - */ - private int windowHeight = 768; - - /** - * Number of seconds to wait for the page to load. - */ - private int loadTimeoutSeconds = 30; - - /** - * Capture requests made by the browser. - */ - private boolean captureRequests = true; - - /** - * The maximum size response body that can be replayed to the browser. Setting this to -1 will cause all requests by - * the browser to be made against the live web. - */ - private int maxReplayLength = 100 * 1024 * 1024; - - private Semaphore openWindowsSemaphore = null; - private ChromeProcess process = null; - private ChromeClient client = null; - - private final CrawlController controller; - private final ApplicationEventPublisher eventPublisher; - private ProcessorChain extractorChain; - - public ExtractorChrome(CrawlController controller, ApplicationEventPublisher eventPublisher) { - this.controller = controller; - this.eventPublisher = eventPublisher; - } - - @Override - protected boolean shouldExtract(CrawlURI uri) { - return uri.getContentType().startsWith("text/html") && uri.is2XXSuccess(); - } - - @Override - protected boolean innerExtract(CrawlURI uri) { - ensureConnected(); - try { - openWindowsSemaphore.acquire(); - try { - visit(uri); - } finally { - openWindowsSemaphore.release(); - } - } catch (InterruptedException e) { - Thread.currentThread().interrupt(); - } - return false; - } - - private void visit(CrawlURI curi) throws InterruptedException { - try (ChromeWindow window = client.createWindow(windowWidth, windowHeight)) { - window.interceptRequests(request -> handleInterceptedRequest(curi, request)); - - if (captureRequests) { - window.captureRequests(request -> handleCapturedRequest(curi, request)); - } - - try { - window.navigateAsync(curi.getURI()).get(loadTimeoutSeconds, TimeUnit.SECONDS); - } catch (ExecutionException e) { - throw new RuntimeException(e.getCause()); - } catch (TimeoutException e) { - throw new RuntimeException("Timed out navigating to " + curi.getURI()); - } - - JSONArray links = window.eval("Array.from(document.querySelectorAll('a[href], area[href]'))" + - ".map(link => link.protocol + '//' + link.host + link.pathname + link.search + link.hash)") - .getJSONArray("value"); - for (int i = 0; i < links.length(); i++) { - addOutlink(curi, links.getString(i), LinkContext.NAVLINK_MISC, Hop.NAVLINK); - } - } - } - - private void handleInterceptedRequest(CrawlURI curi, InterceptedRequest interceptedRequest) { - ChromeRequest request = interceptedRequest.getRequest(); - if (request.getMethod().equals("GET") && request.getUrl().equals(curi.getURI())) { - replayResponseToBrowser(curi, interceptedRequest); - } else { - interceptedRequest.continueNormally(); - } - } - - @SuppressWarnings("unchecked") - private void replayResponseToBrowser(CrawlURI curi, InterceptedRequest interceptedRequest) { - // There seems to be no easy way to stream the body to the browser so we slurp it into - // memory with a size limit. The one way I can see to achieve streaming is to have Heritrix - // serve the request over its HTTP server and pass a Heritrix URL to Fetch.fulfillRequest - // instead of the body directly. We might need to do that if memory pressure becomes a - // problem but for now just keep it simple. - - long bodyLength = curi.getRecorder().getResponseContentLength(); - if (bodyLength > maxReplayLength) { - logger.log(FINE, "Page body too large to replay: {0}", curi.getURI()); - interceptedRequest.continueNormally(); - return; - } - - byte[] body = new byte[(int)bodyLength]; - try (InputStream stream = curi.getRecorder().getContentReplayInputStream()) { - IOUtils.readFully(stream, body); - } catch (IOException e) { - logger.log(WARNING, "Error reading back page body: " + curi.getURI(), e); - interceptedRequest.continueNormally(); - return; - } - - Map headers = (Map) curi.getData().get(A_HTTP_RESPONSE_HEADERS); - if (headers == null) { - logger.log(WARNING, "Response headers unavailable in CrawlURI. Letting the browser " + - "refetch {0}", curi.getURI()); - interceptedRequest.continueNormally(); - return; - } - interceptedRequest.fulfill(curi.getFetchStatus(), headers.entrySet(), body); - } - - private void handleCapturedRequest(CrawlURI via, ChromeRequest request) { - if (request.isResponseFulfilledByInterception() || UriUtils.isDataUri(request.getUrl())) { - return; - } - - String recorderBaseName = "ExtractorChrome-" + nextRecorderId.getAndIncrement(); - Recorder recorder = new Recorder(new File(controller.getScratchDir().getFile(), recorderBaseName), - controller.getRecorderOutBufferBytes(), controller.getRecorderInBufferBytes()); - try { - String digestAlgorithm = "sha1"; - recorder.getRecordedInput().setDigest(digestAlgorithm); - recorder.getRecordedOutput().write(request.getRequestHeader().getBytes(US_ASCII)); - recorder.getRecordedOutput().write(request.getRequestBody()); - - // strip the Transfer-Encoding header since we only have access to the decoded body - String responseHeader = TRANSFER_ENCODING_RE.matcher(request.getResponseHeader()).replaceAll(""); - - recorder.inputWrap(new SequenceInputStream(enumeration(Arrays.asList( - new ByteArrayInputStream(responseHeader.getBytes(US_ASCII)), - new InputStream() { - public int read() { - recorder.markContentBegin(); - return -1; - } - }, - new ByteArrayInputStream(request.getResponseBody()))))); - recorder.getRecordedInput().readFully(); - recorder.closeRecorders(); - - CrawlURI curi = via.createCrawlURI(request.getUrl(), LinkContext.EMBED_MISC, Hop.EMBED); - curi.getAnnotations().add("browser"); - curi.setContentDigest(digestAlgorithm, recorder.getRecordedInput().getDigestValue()); - curi.setContentSize(recorder.getRecordedInput().getSize()); - curi.setContentType(request.getResponseContentType()); - curi.setFetchBeginTime(request.getBeginTime()); - curi.setFetchCompletedTime(System.currentTimeMillis()); - curi.setFetchStatus(request.getStatus()); - curi.setRecorder(recorder); - curi.setServerIP(request.getRemoteIPAddress()); - curi.setThreadNumber(via.getThreadNumber()); - - switch (request.getMethod()) { - case "GET": - curi.setFetchType(HTTP_GET); - break; - case "POST": - curi.setFetchType(HTTP_POST); - break; - default: - curi.setFetchType(UNKNOWN); - break; - } - - Frontier frontier = controller.getFrontier(); - curi.getOverlayNames(); // for side-effect of creating the overlayNames list - - // inform the frontier we've already seen this uri so it won't schedule it - // we only do this for GETs so a POST doesn't prevent scheduling a GET of the same URI - if (request.getMethod().equals("GET")) { - frontier.considerIncluded(curi); - } - - KeyedProperties.loadOverridesFrom(curi); - try { - // perform link extraction - extractorChain.process(curi, null); - - // send the result to the disposition chain to dispatch outlinks and write warcs - frontier.beginDisposition(curi); - controller.getDispositionChain().process(curi,null); - } finally { - KeyedProperties.clearOverridesFrom(curi); - } - - curi.aboutToLog(); - controller.getLoggerModule().getUriProcessing().log(INFO, curi.getUURI().toString(), curi); - - if (curi.isSuccess()) { - eventPublisher.publishEvent(new CrawlURIDispositionEvent(this, curi, SUCCEEDED)); - } else { - eventPublisher.publishEvent(new CrawlURIDispositionEvent(this, curi, FAILED)); - } - frontier.endDisposition(); - } catch (Exception e) { - logger.log(WARNING, "Exception handling subrequest " + request.getUrl(), e); - } finally { - recorder.cleanup(); - } - } - - @Override - public void start() { - if (isRunning) return; - super.start(); - openWindowsSemaphore = new Semaphore(maxOpenWindows); - - // If we're enabled by default launch the browser now to get early feedback if there's a connection error. - // Otherwise, we launch it on demand so that we don't create browser processes for jobs that don't need it. - if (getEnabled()) { - ensureConnected(); - } - - if (extractorChain == null) { - // The fetch chain normally includes some preprocessing, fetch and extractor processors, but we want just - // the extractors as we let the browser fetch subresources. So we construct a new chain consisting of the - // extractors only. - List extractors = new ArrayList<>(); - for (Processor processor : controller.getFetchChain().getProcessors()) { - if (processor instanceof Extractor) { - extractors.add(processor); - } - } - extractorChain = new ProcessorChain(); - extractorChain.setProcessors(extractors); - } - } - - private synchronized void ensureConnected() { - if (client != null) return; - if (devtoolsUrl != null) { - client = new ChromeClient(devtoolsUrl); - } else { - try { - process = new ChromeProcess(executable, commandLineOptions); - } catch (IOException e) { - throw new RuntimeException("Failed to launch browser process", e); - } - client = new ChromeClient(process.getDevtoolsUrl()); - } - } - - @Override - public void stop() { - super.stop(); - if (client != null) { - client.close(); - client = null; - } - if (process != null) { - process.close(); - process = null; - } - } - - public String getExecutable() { - return executable; - } - - public void setExecutable(String executable) { - this.executable = executable; - } - - public int getMaxOpenWindows() { - return maxOpenWindows; - } - - public void setMaxOpenWindows(int maxOpenWindows) { - this.maxOpenWindows = maxOpenWindows; - } - - public String getDevtoolsUrl() { - return devtoolsUrl; - } - - public void setDevtoolsUrl(String devtoolsUrl) { - this.devtoolsUrl = devtoolsUrl; - } - - public int getWindowWidth() { - return windowWidth; - } - - public void setWindowWidth(int windowWidth) { - this.windowWidth = windowWidth; - } - - public int getWindowHeight() { - return windowHeight; - } - - public void setWindowHeight(int windowHeight) { - this.windowHeight = windowHeight; - } - - public int getLoadTimeoutSeconds() { - return loadTimeoutSeconds; - } - - public void setLoadTimeoutSeconds(int loadTimeoutSeconds) { - this.loadTimeoutSeconds = loadTimeoutSeconds; - } - - public List getCommandLineOptions() { - return commandLineOptions; - } - - public void setCommandLineOptions(List commandLineOptions) { - this.commandLineOptions = commandLineOptions; - } -} \ No newline at end of file diff --git a/contrib/src/main/java/org/archive/net/chrome/ChromeClient.java b/contrib/src/main/java/org/archive/net/chrome/ChromeClient.java deleted file mode 100644 index 263acff6..00000000 --- a/contrib/src/main/java/org/archive/net/chrome/ChromeClient.java +++ /dev/null @@ -1,168 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -import org.java_websocket.client.WebSocketClient; -import org.java_websocket.handshake.ServerHandshake; -import org.json.JSONObject; - -import java.io.Closeable; -import java.net.URI; -import java.util.Map; -import java.util.concurrent.*; -import java.util.concurrent.atomic.AtomicLong; -import java.util.function.Consumer; -import java.util.logging.Level; -import java.util.logging.Logger; - -import static java.util.logging.Level.WARNING; - -/** - * A client for the Chrome Devtools Protocol. - */ -public class ChromeClient implements Closeable { - private static final Logger logger = Logger.getLogger(ChromeClient.class.getName()); - private static final int RPC_TIMEOUT_SECONDS = 60; - - private final DevtoolsSocket devtoolsSocket; - private final AtomicLong nextMessageId = new AtomicLong(0); - private final Map> responseFutures = new ConcurrentHashMap<>(); - final ConcurrentHashMap> sessionEventHandlers = new ConcurrentHashMap<>(); - - public ChromeClient(String devtoolsUrl) { - devtoolsSocket = new DevtoolsSocket(URI.create(devtoolsUrl)); - try { - devtoolsSocket.connectBlocking(); - } catch (InterruptedException e) { - throw new ChromeException("Interrupted while connecting", e); - } - } - - public JSONObject call(String method, Object... keysAndValues) { - return callInSession(null, method, keysAndValues); - } - - public JSONObject callInSession(String sessionId, String method, Object... keysAndValues) { - JSONObject params = new JSONObject(); - if (keysAndValues.length % 2 != 0) { - throw new IllegalArgumentException("keysAndValues.length must even"); - } - for (int i = 0; i < keysAndValues.length; i += 2) { - params.put((String)keysAndValues[i], keysAndValues[i + 1]); - } - return callInternal(sessionId, method, params); - } - - private JSONObject callInternal(String sessionId, String method, JSONObject params) { - long id = nextMessageId.getAndIncrement(); - JSONObject message = new JSONObject(); - message.put("id", id); - if (sessionId != null) { - message.put("sessionId", sessionId); - } - message.put("method", method); - message.put("params", params); - CompletableFuture future = new CompletableFuture<>(); - responseFutures.put(id, future); - devtoolsSocket.send(message.toString()); - try { - return future.get(RPC_TIMEOUT_SECONDS, TimeUnit.SECONDS); - } catch (InterruptedException e) { - throw new ChromeException("Call interrupted", e); - } catch (TimeoutException e) { - throw new ChromeException("Call timed out: " + message, e); - } catch (ExecutionException e) { - throw new ChromeException("Call failed: " + message + ": " + e.getMessage(), e.getCause()); - } - } - - private void handleResponse(JSONObject message) { - long id = message.getLong("id"); - CompletableFuture future = responseFutures.remove(id); - if (future == null) { - logger.log(WARNING, "Unexpected RPC response id {0}", id); - } else if (message.has("error")) { - future.completeExceptionally(new ChromeException(message.getJSONObject("error").getString("message"))); - } else { - future.complete(message.getJSONObject("result")); - } - } - - private void handleEvent(JSONObject message) { - if (message.has("sessionId")) { - String sessionId = message.getString("sessionId"); - Consumer handler = sessionEventHandlers.get(sessionId); - if (handler != null) { - handler.accept(message); - } else { - logger.log(WARNING, "Received event for unknown session {0}", sessionId); - } - } - } - - public ChromeWindow createWindow(int width, int height) { - String targetId = call("Target.createTarget", "url", "about:blank", - "width", width, "height", height).getString("targetId"); - return new ChromeWindow(this, targetId); - } - - @Override - public void close() { - devtoolsSocket.close(); - } - - private class DevtoolsSocket extends WebSocketClient { - public DevtoolsSocket(URI uri) { - super(uri); - setConnectionLostTimeout(-1); // disable pings - Chromium doesn't support them - } - - @Override - public void onOpen(ServerHandshake serverHandshake) { - - } - - @Override - public void onMessage(String messageString) { - try { - JSONObject message = new JSONObject(messageString); - if (message.has("method")) { - handleEvent(message); - } else { - handleResponse(message); - } - } catch (Throwable e) { - logger.log(WARNING, "Exception handling message from Chromium", e); - throw e; - } - } - - @Override - public void onClose(int code, String reason, boolean remote) { - if (!remote) return; - logger.log(WARNING, "Websocket closed by browser: " + reason); - } - - @Override - public void onError(Exception e) { - logger.log(Level.SEVERE, "Websocket error", e); - } - } -} diff --git a/contrib/src/main/java/org/archive/net/chrome/ChromeException.java b/contrib/src/main/java/org/archive/net/chrome/ChromeException.java deleted file mode 100644 index e65dabc0..00000000 --- a/contrib/src/main/java/org/archive/net/chrome/ChromeException.java +++ /dev/null @@ -1,38 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -public class ChromeException extends RuntimeException { - public ChromeException() { - super(); - } - - public ChromeException(String message) { - super(message); - } - - public ChromeException(String message, Throwable throwable) { - super(message, throwable); - } - - public ChromeException(Throwable throwable) { - super(throwable); - } -} diff --git a/contrib/src/main/java/org/archive/net/chrome/ChromeProcess.java b/contrib/src/main/java/org/archive/net/chrome/ChromeProcess.java deleted file mode 100644 index 743c8ada..00000000 --- a/contrib/src/main/java/org/archive/net/chrome/ChromeProcess.java +++ /dev/null @@ -1,197 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -import org.archive.modules.extractor.ExtractorChrome; - -import java.io.BufferedReader; -import java.io.Closeable; -import java.io.IOException; -import java.io.InputStreamReader; -import java.util.*; -import java.util.concurrent.*; -import java.util.logging.Logger; - -import static java.nio.charset.StandardCharsets.ISO_8859_1; -import static java.util.logging.Level.FINER; - -/** - * Manages starting and stopping a browser process. - */ -public class ChromeProcess implements Closeable { - private static final Logger logger = Logger.getLogger(ExtractorChrome.class.getName()); - - private static final String[] DEFAULT_EXECUTABLES = {"chromium-browser", "chromium", "google-chrome", - "C:\\Program Files (x86)\\Google\\Chrome\\Application\\chrome.exe", - "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", - "firefox"}; - private static final int SHUTDOWN_TIMEOUT_SECONDS = 2; - - private static final Set runningProcesses = Collections.newSetFromMap(new ConcurrentHashMap<>()); - private static Thread shutdownHook; - - private final Process process; - private final String devtoolsUrl; - - public ChromeProcess(String executable, List commandLineOptions) throws IOException { - process = executable == null ? launchAny(commandLineOptions) : launch(executable, commandLineOptions); - runningProcesses.add(process); - registerShutdownHook(); - devtoolsUrl = readDevtoolsUriFromStderr(process); - } - - private static Process launch(String executable, List commandLineOptions) throws IOException { - List command = new ArrayList<>(); - command.add(executable); - command.add("--headless"); - command.add("--remote-debugging-port=0"); - - // https://github.com/GoogleChrome/chrome-launcher/blob/master/docs/chrome-flags-for-tools.md - command.add("--disable-background-networking"); - command.add("--disable-background-timer-throttling"); - command.add("--disable-backgrounding-occluded-windows"); - command.add("--disable-breakpad"); - command.add("--disable-client-side-phishing-detection"); - command.add("--disable-component-extensions-with-background-pages"); - command.add("--disable-component-update"); - command.add("--disable-crash-reporter"); - command.add("--disable-default-apps"); - command.add("--disable-extensions"); - command.add("--disable-features=Translate"); - command.add("--disable-ipc-flooding-protection"); - command.add("--disable-popup-blocking"); - command.add("--disable-prompt-on-repost"); - command.add("--disable-renderer-backgrounding"); - command.add("--disable-sync"); - command.add("--metrics-recording-only"); - command.add("--mute-audio"); - command.add("--no-default-browser-check"); - command.add("--no-first-run"); - command.add("--password-store=basic"); - command.add("--use-mock-keychain"); - - command.addAll(commandLineOptions); - return new ProcessBuilder(command) - .inheritIO() - .redirectError(ProcessBuilder.Redirect.PIPE) - .start(); - } - - /** - * Try to launch the browser process using each of DEFAUSLT_EXECUTABLES in turn until one succeeds. - */ - private static Process launchAny(List extraCommandLineOptions) throws IOException { - IOException lastException = null; - for (String executable : DEFAULT_EXECUTABLES) { - try { - return launch(executable, extraCommandLineOptions); - } catch (IOException e) { - lastException = e; - } - } - throw new IOException("Failed to launch any of " + Arrays.asList(DEFAULT_EXECUTABLES), lastException); - } - - @Override - public void close() { - destroyProcess(process); - runningProcesses.remove(process); - } - - /** - * Register a shutdown hook that destroys all running browser processes before exiting in case stop() is never - * called. This can happen if the Heritrix exits abnormally. - */ - private static synchronized void registerShutdownHook() { - if (shutdownHook != null) return; - shutdownHook = new Thread(ChromeProcess::destroyAllRunningProcesses, "ChromiumClient shutdown hook"); - Runtime.getRuntime().addShutdownHook(shutdownHook); - } - - private static void destroyAllRunningProcesses() { - for (Process process : runningProcesses) { - process.destroy(); - } - for (Process process : runningProcesses) { - try { - if (!process.waitFor(SHUTDOWN_TIMEOUT_SECONDS, TimeUnit.SECONDS)) { - break; - } - } catch (InterruptedException e) { - break; - } - } - for (Process process : runningProcesses) { - process.destroyForcibly(); - } - } - - private static void destroyProcess(Process process) { - process.destroy(); - try { - process.waitFor(SHUTDOWN_TIMEOUT_SECONDS, TimeUnit.SECONDS); - } catch (InterruptedException e) { - Thread.currentThread().interrupt(); - } finally { - process.destroyForcibly(); - } - } - - /** - * Reads the stderr of a Chromium process and returns the DevTools URI. Once this method - * returns stderr will continue to be consumed and logged by a background thread. - */ - private static String readDevtoolsUriFromStderr(Process process) throws IOException { - BufferedReader stderr = new BufferedReader(new InputStreamReader(process.getErrorStream(), ISO_8859_1)); - CompletableFuture future = new CompletableFuture<>(); - Thread thread = new Thread(() -> { - String listenMsg = "DevTools listening on "; - try { - while (true) { - String line = stderr.readLine(); - if (line == null) break; - if (!future.isDone() && line.startsWith(listenMsg)) { - future.complete(line.substring(listenMsg.length())); - } - logger.log(FINER, "Chromium STDERR: {0}", line); - } - } catch (IOException e) { - future.completeExceptionally(e); - } - }); - thread.setName("Chromium stderr reader"); - thread.setDaemon(true); - thread.start(); - - try { - return future.get(10, TimeUnit.SECONDS); - } catch (InterruptedException | ExecutionException | TimeoutException e) { - // unwrap the exception if we can to cut down on log noise - if (e.getCause() instanceof IOException) { - throw (IOException) e.getCause(); - } - throw new IOException(e); - } - } - - public String getDevtoolsUrl() { - return devtoolsUrl; - } -} diff --git a/contrib/src/main/java/org/archive/net/chrome/ChromeRequest.java b/contrib/src/main/java/org/archive/net/chrome/ChromeRequest.java deleted file mode 100644 index 17530e9c..00000000 --- a/contrib/src/main/java/org/archive/net/chrome/ChromeRequest.java +++ /dev/null @@ -1,163 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -import org.json.JSONObject; - -import java.nio.charset.StandardCharsets; -import java.util.Base64; - -public class ChromeRequest { - private final ChromeWindow window; - private final String id; - private JSONObject requestJson; - private JSONObject rawRequestHeaders; - private JSONObject responseJson; - private JSONObject rawResponseHeaders; - private String responseHeadersText; - private final long beginTime = System.currentTimeMillis(); - private boolean responseFulfilledByInterception; - - public ChromeRequest(ChromeWindow window, String id) { - this.window = window; - this.id = id; - } - - void setRawRequestHeaders(JSONObject rawRequestHeaders) { - this.rawRequestHeaders = rawRequestHeaders; - } - - void setResponseJson(JSONObject responseJson) { - this.responseJson = responseJson; - } - - void setRawResponseHeaders(JSONObject headers) { - this.rawResponseHeaders = headers; - } - - void setResponseHeadersText(String responseHeadersText) { - this.responseHeadersText = responseHeadersText; - } - - public String getUrl() { - return requestJson.getString("url"); - } - - public byte[] getResponseBody() { - JSONObject reply = window.call("Network.getResponseBody", "requestId", id); - byte[] body; - if (reply.getBoolean("base64Encoded")) { - body = Base64.getDecoder().decode(reply.getString("body")); - } else { - body = reply.getString("body").getBytes(StandardCharsets.UTF_8); - } - return body; - } - - public String getRequestHeader() { - if (responseJson != null && responseJson.has("requestHeadersText")) { - return responseJson.getString("requestHeadersText"); - } - StringBuilder builder = new StringBuilder(); - builder.append(requestJson.getString("method")); - builder.append(' '); - builder.append(getUrl()); - builder.append(" HTTP/1.1\r\n"); - formatHeaders(builder, requestJson.getJSONObject("headers"), rawRequestHeaders); - return builder.toString(); - } - - public byte[] getRequestBody() { - if (requestJson.has("postData")) { - return requestJson.getString("postData").getBytes(StandardCharsets.UTF_8); - } else { - return new byte[0]; - } - } - - public String getResponseHeader() { - if (responseHeadersText != null) { - return responseHeadersText; - } else if (responseJson.has("headersText")) { - return responseJson.getString("headersText"); - } - StringBuilder builder = new StringBuilder(); - if (responseJson.getString("protocol").equals("http/1.0")) { - builder.append("HTTP/1.0"); - } else { - builder.append("HTTP/1.1"); - } - builder.append(getStatus()); - builder.append(" "); - builder.append(responseJson.getString("statusText")); - builder.append("\r\n"); - formatHeaders(builder, responseJson.getJSONObject("headers"), rawResponseHeaders); - return builder.toString(); - } - - private void formatHeaders(StringBuilder builder, JSONObject headers, JSONObject rawHeaders) { - if (rawHeaders != null) { - headers = rawHeaders; - } - for (Object key : headers.keySet()) { - builder.append(key); - builder.append(": "); - builder.append(headers.getString((String) key)); - builder.append("\r\n"); - } - builder.append("\r\n"); - } - - public String getMethod() { - return requestJson.getString("method"); - } - - public int getStatus() { - return responseJson.getInt("status"); - } - - public String getResponseContentType() { - return responseJson.getString("mimeType"); - } - - public long getBeginTime() { - return beginTime; - } - - public String getRemoteIPAddress() { - if (responseJson.has("remoteIPAddress")) { - return responseJson.getString("remoteIPAddress"); - } else { - return null; - } - } - - void setRequestJson(JSONObject requestJson) { - this.requestJson = requestJson; - } - - void setResponseFulfilledByInterception(boolean responseFulfilledByInterception) { - this.responseFulfilledByInterception = responseFulfilledByInterception; - } - - public boolean isResponseFulfilledByInterception() { - return responseFulfilledByInterception; - } -} diff --git a/contrib/src/main/java/org/archive/net/chrome/ChromeWindow.java b/contrib/src/main/java/org/archive/net/chrome/ChromeWindow.java deleted file mode 100644 index cb7b7b8e..00000000 --- a/contrib/src/main/java/org/archive/net/chrome/ChromeWindow.java +++ /dev/null @@ -1,233 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -import org.java_websocket.exceptions.WebsocketNotConnectedException; -import org.json.JSONObject; - -import java.io.Closeable; -import java.util.Map; -import java.util.concurrent.*; -import java.util.function.Consumer; -import java.util.logging.Logger; - -import static java.util.logging.Level.*; - -/** - * A browser window or tab. - */ -public class ChromeWindow implements Closeable { - private static final Logger logger = Logger.getLogger(ChromeWindow.class.getName()); - - private final ChromeClient client; - private final String targetId; - private final String sessionId; - private boolean closed; - private CompletableFuture loadEventFuture; - private final Map requestMap = new ConcurrentHashMap<>(); - private Consumer requestConsumer; - private Consumer requestInterceptor; - private final ExecutorService eventExecutor; - - public ChromeWindow(ChromeClient client, String targetId) { - this.client = client; - this.targetId = targetId; - this.sessionId = client.call("Target.attachToTarget", "targetId", targetId, - "flatten", true).getString("sessionId"); - eventExecutor = Executors.newSingleThreadExecutor(runnable -> - new Thread(runnable, "ChromeWindow (sessionId=" + sessionId +")")); - client.sessionEventHandlers.put(sessionId, this::handleEvent); - call("Page.enable"); // for loadEventFired - call("Page.setLifecycleEventsEnabled", "enabled", true); // for networkidle - call("Runtime.enable"); // required by Firefox for Runtime.evaluate to work - } - - /** - * Call a devtools method in the session of this window. - */ - public JSONObject call(String method, Object... keysAndValues) { - return client.callInSession(sessionId, method, keysAndValues); - } - - /** - * Evaluate a JavaScript expression. - */ - public JSONObject eval(String expression) { - return call("Runtime.evaluate", "expression", expression, - "returnByValue", true).getJSONObject("result"); - } - - /** - * Navigate this window to a new URL. Returns a future which will be fulfilled when the page finishes loading. - */ - public CompletableFuture navigateAsync(String url) { - if (loadEventFuture != null) { - loadEventFuture.cancel(false); - } - loadEventFuture = new CompletableFuture<>(); - call("Page.navigate", "url", url); - return loadEventFuture; - } - - private void handleEvent(JSONObject message) { - if (closed) return; - // Run event handlers on a different thread so we don't block the websocket receiving thread. - // That would cause a deadlock if an event handler itself made an RPC call as the response could - // never be processed. - // We use a single thread per window though as the order events are processed is important. - eventExecutor.submit(() -> { - try { - handleEventOnEventThread(message); - } catch (Throwable t) { - logger.log(WARNING, "Exception handling browser event " + message, t); - } - }); - } - - private void handleEventOnEventThread(JSONObject message) { - JSONObject params = message.getJSONObject("params"); - switch (message.getString("method")) { - case "Fetch.requestPaused": - handlePausedRequest(params); - break; - case "Network.requestWillBeSent": - handleRequestWillBeSent(params); - break; - case "Network.requestWillBeSentExtraInfo": - handleRequestWillBeSentExtraInfo(params); - break; - case "Network.responseReceived": - handleResponseReceived(params); - break; - case "Network.responseReceivedExtraInfo": - handleResponseReceivedExtraInfo(params); - break; - case "Network.loadingFinished": - handleLoadingFinished(params); - break; - case "Page.loadEventFired": - if (loadEventFuture != null) { - try { - Thread.sleep(1000); - } catch (InterruptedException e) { - Thread.currentThread().interrupt(); - } - loadEventFuture.complete(null); - } - break; - default: - logger.log(FINE, "Unhandled event {0}", message); - break; - } - } - - private void handlePausedRequest(JSONObject params) { - String networkId = params.getString("networkId"); - ChromeRequest request = requestMap.computeIfAbsent(networkId, id -> new ChromeRequest(this, id)); - request.setRequestJson(params.getJSONObject("request")); - String id = params.getString("requestId"); - InterceptedRequest interceptedRequest = new InterceptedRequest(this, id, request); - try { - requestInterceptor.accept(interceptedRequest); - } catch (Exception e) { - logger.log(SEVERE, "Request interceptor threw", e); - } - if (!interceptedRequest.isHandled()) { - interceptedRequest.continueNormally(); - } - } - - private void handleRequestWillBeSent(JSONObject params) { - String requestId = params.getString("requestId"); - ChromeRequest request = requestMap.computeIfAbsent(requestId, id -> new ChromeRequest(this, id)); - request.setRequestJson(params.getJSONObject("request")); - } - - private void handleRequestWillBeSentExtraInfo(JSONObject params) { - // it seems this event can arrive both before and after requestWillBeSent so we need to cope with that - String requestId = params.getString("requestId"); - ChromeRequest request = requestMap.computeIfAbsent(requestId, id -> new ChromeRequest(this, id)); - if (params.has("headers")) { - request.setRawRequestHeaders(params.getJSONObject("headers")); - } - } - - private void handleResponseReceived(JSONObject params) { - ChromeRequest request = requestMap.get(params.getString("requestId")); - if (request == null) { - logger.log(WARNING, "Got responseReceived event without corresponding requestWillBeSent"); - return; - } - request.setResponseJson(params.getJSONObject("response")); - } - - private void handleResponseReceivedExtraInfo(JSONObject params) { - ChromeRequest request = requestMap.get(params.getString("requestId")); - if (request == null) { - logger.log(WARNING, "Got responseReceivedExtraInfo event without corresponding requestWillBeSent"); - return; - } - if (params.has("headers")) { - request.setRawResponseHeaders(params.getJSONObject("headers")); - } - if (params.has("headersText")) { - request.setResponseHeadersText(params.getString("headersText")); - } - } - - private void handleLoadingFinished(JSONObject params) { - ChromeRequest request = requestMap.get(params.getString("requestId")); - if (request == null) { - logger.log(WARNING, "Got loadingFinished event without corresponding requestWillBeSent"); - return; - } - if (requestConsumer != null) { - requestConsumer.accept(request); - } - } - - @Override - public void close() { - if (closed) return; - closed = true; - eventExecutor.shutdown(); - try { - eventExecutor.awaitTermination(1, TimeUnit.SECONDS); - } catch (InterruptedException e) { - Thread.currentThread().interrupt(); - } - try { - client.call("Target.closeTarget", "targetId", targetId); - } catch (WebsocketNotConnectedException e) { - // no need to close the window if the browser has already exited - } - client.sessionEventHandlers.remove(sessionId); - } - - public void captureRequests(Consumer requestConsumer) { - this.requestConsumer = requestConsumer; - call("Network.enable"); - } - - public void interceptRequests(Consumer requestInterceptor) { - this.requestInterceptor = requestInterceptor; - call("Fetch.enable"); - } -} \ No newline at end of file diff --git a/contrib/src/main/java/org/archive/net/chrome/InterceptedRequest.java b/contrib/src/main/java/org/archive/net/chrome/InterceptedRequest.java deleted file mode 100644 index 39c69a5e..00000000 --- a/contrib/src/main/java/org/archive/net/chrome/InterceptedRequest.java +++ /dev/null @@ -1,78 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -import org.json.JSONArray; -import org.json.JSONObject; - -import java.util.Base64; -import java.util.Collection; -import java.util.Map; - -public class InterceptedRequest { - private final String id; - private final ChromeRequest request; - private final ChromeWindow window; - private boolean handled; - - public InterceptedRequest(ChromeWindow window, String id, ChromeRequest request) { - this.window = window; - this.id = id; - this.request = request; - } - - public ChromeRequest getRequest() { - return request; - } - - public void fulfill(int status, Collection> headers, byte[] body) { - setHandled(); - JSONArray headerArray = new JSONArray(); - for (Map.Entry entry : headers) { - JSONObject object = new JSONObject(); - object.put("name", entry.getKey()); - object.put("value", entry.getValue()); - headerArray.put(object); - } - String encodedBody = Base64.getEncoder().encodeToString(body); - request.setResponseFulfilledByInterception(true); - window.call("Fetch.fulfillRequest", - "requestId", id, - "responseCode", status, - "responseHeaders", headerArray, - "body", encodedBody); - } - - public void continueNormally() { - setHandled(); - window.call("Fetch.continueRequest", "requestId", id); - } - - public boolean isHandled() { - return handled; - } - - private void setHandled() { - if (handled) { - throw new IllegalStateException("intercepted request already handled"); - } - handled = true; - } -} diff --git a/contrib/src/test/java/org/archive/modules/extractor/ExtractorChromeTest.java b/contrib/src/test/java/org/archive/modules/extractor/ExtractorChromeTest.java deleted file mode 100644 index f3a39166..00000000 --- a/contrib/src/test/java/org/archive/modules/extractor/ExtractorChromeTest.java +++ /dev/null @@ -1,203 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.modules.extractor; - -import org.archive.crawler.framework.CrawlController; -import org.archive.crawler.framework.Frontier; -import org.archive.crawler.reporting.CrawlerLoggerModule; -import org.archive.modules.*; -import org.archive.modules.fetcher.DefaultServerCache; -import org.archive.modules.fetcher.FetchHTTP; -import org.archive.modules.fetcher.SimpleCookieStore; -import org.archive.net.UURIFactory; -import org.archive.util.Recorder; -import org.eclipse.jetty.server.Request; -import org.eclipse.jetty.server.Server; -import org.eclipse.jetty.server.handler.AbstractHandler; -import org.junit.AfterClass; -import org.junit.BeforeClass; -import org.junit.Rule; -import org.junit.Test; -import org.junit.rules.TemporaryFolder; - -import javax.servlet.ServletException; -import javax.servlet.http.HttpServletRequest; -import javax.servlet.http.HttpServletResponse; -import java.io.IOException; -import java.net.InetSocketAddress; -import java.util.*; -import java.util.logging.Level; -import java.util.logging.Logger; - -import static java.util.stream.Collectors.toList; -import static org.archive.modules.CrawlURI.FetchType.HTTP_GET; -import static org.archive.modules.CrawlURI.FetchType.HTTP_POST; -import static org.easymock.EasyMock.*; -import static org.junit.Assert.*; -import static org.junit.Assume.assumeNoException; - -public class ExtractorChromeTest { - private static Server server; - - @BeforeClass - public static void startServer() throws Exception { - server = new Server(InetSocketAddress.createUnresolved("127.0.0.1", 7778)); - server.setHandler(new AbstractHandler() { - @Override - public void handle(String target, Request baseRequest, HttpServletRequest request, HttpServletResponse response) throws IOException, ServletException { - switch (target) { - case "/": - response.setContentType("text/html"); - response.getWriter().write("link" + - "" + - "" + - "" + - ""); - baseRequest.setHandled(true); - break; - case "/style.css": - response.setContentType("text/css"); - response.getWriter().write("@media only print { body { background: url('printonly.png'); } }"); - baseRequest.setHandled(true); - break; - case "/blue.png": - case "/printonly.png": - response.setContentType("image/png"); - response.getWriter().write("bogus png"); - baseRequest.setHandled(true); - break; - case "/post": - response.setContentType("plain/text"); - response.getWriter().write("method=" + request.getMethod()); - baseRequest.setHandled(true); - break; - default: - System.err.println("Unhandled target: " + target); - break; - } - - } - }); - server.start(); - } - - @AfterClass - public static void stopServer() throws Exception { - server.stop(); - } - - @Rule - public TemporaryFolder tempFolder = new TemporaryFolder(); - - @Test - public void test() throws IOException, InterruptedException { - List processedURIs = Collections.synchronizedList(new ArrayList<>()); - - DispositionChain dispositionChain = new DispositionChain(); - dispositionChain.setProcessors(Arrays.asList(new Processor() { - @Override - protected boolean shouldProcess(CrawlURI uri) { - return true; - } - - @Override - protected void innerProcess(CrawlURI uri) { - processedURIs.add(uri); - } - })); - - FetchChain fetchChain = new FetchChain(); - fetchChain.setProcessors(Arrays.asList(new ExtractorCSS())); - - CrawlController controller = new CrawlController(); - controller.setDispositionChain(dispositionChain); - controller.setFetchChain(fetchChain); - controller.setLoggerModule(new CrawlerLoggerModule() { - @Override - public Logger getUriProcessing() { - Logger logger = Logger.getAnonymousLogger(); - logger.setLevel(Level.WARNING); - return logger; - } - }); - Frontier frontier = createMock(Frontier.class); - frontier.considerIncluded(anyObject()); - expectLastCall().anyTimes(); - frontier.beginDisposition(anyObject()); - expectLastCall().anyTimes(); - frontier.endDisposition(); - expectLastCall().anyTimes(); - frontier.finished(anyObject()); - expectLastCall().anyTimes(); - replay(frontier); - controller.setFrontier(frontier); - - FetchHTTP fetchHTTP = new FetchHTTP(); - fetchHTTP.setServerCache(new DefaultServerCache()); - fetchHTTP.setCookieStore(new SimpleCookieStore()); - fetchHTTP.setUserAgentProvider(new CrawlMetadata()); - fetchHTTP.start(); - - ExtractorChrome extractor = new ExtractorChrome(controller, event -> { /* ignored */ }); - try { - extractor.start(); - } catch (RuntimeException e) { - assumeNoException("Unable to start Chrome", e); - } - - Recorder recorder = new Recorder(tempFolder.newFile(), 1024, 1024); - try { - CrawlURI curi = new CrawlURI(UURIFactory.getInstance("http://127.0.0.1:7778/")); - - Recorder.setHttpRecorder(recorder); - curi.setRecorder(recorder); - fetchHTTP.process(curi); - - extractor.innerExtract(curi); - - List outLinks = curi.getOutLinks().stream().map(CrawlURI::toString).sorted().collect(toList()); - assertEquals(Collections.singletonList("http://example.org/page2.html"), outLinks); - } finally { - extractor.stop(); - fetchHTTP.stop(); - recorder.cleanup(); - } - - assertEquals(3, processedURIs.size()); - Set subresourceUrls = new HashSet<>(); - for (CrawlURI curi: processedURIs) { - assertEquals(200, curi.getFetchStatus()); - assertEquals(curi.getUURI().getPath().equals("/post") ? HTTP_POST : HTTP_GET, curi.getFetchType()); - assertNotNull(curi.getContentDigest()); - assertTrue(curi.getContentSize() > 0); - subresourceUrls.add(curi.getURI()); - - if (curi.getURI().equals("http://127.0.0.1:7778/style.css")) { - assertEquals("check link extraction ran on captured resources", - "http://127.0.0.1:7778/printonly.png", - new ArrayList<>(curi.getOutLinks()).get(0).getURI()); - } - } - assertEquals(new HashSet<>(Arrays.asList( - "http://127.0.0.1:7778/style.css", - "http://127.0.0.1:7778/blue.png", - "http://127.0.0.1:7778/post")), subresourceUrls); - } -} \ No newline at end of file diff --git a/contrib/src/test/java/org/archive/net/chrome/ChromeClientTest.java b/contrib/src/test/java/org/archive/net/chrome/ChromeClientTest.java deleted file mode 100644 index 47db21f3..00000000 --- a/contrib/src/test/java/org/archive/net/chrome/ChromeClientTest.java +++ /dev/null @@ -1,50 +0,0 @@ -/* - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Licensed to the Internet Archive (IA) by one or more individual - * contributors. - * - * The IA licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.archive.net.chrome; - -import org.json.JSONObject; -import org.junit.Test; - -import java.io.IOException; -import java.util.Collections; -import java.util.concurrent.TimeUnit; - -import static org.junit.Assert.assertEquals; -import static org.junit.Assume.assumeNoException; - -public class ChromeClientTest { - @Test - public void navigate() throws Exception { - ChromeProcess triedProcess; - try { - triedProcess = new ChromeProcess(null, Collections.emptyList()); - } catch (IOException e) { - assumeNoException("Chrome unavailable", e); - return; - } - try (ChromeProcess process = triedProcess; - ChromeClient client = new ChromeClient(process.getDevtoolsUrl()); - ChromeWindow window = client.createWindow(1024, 768)) { - window.navigateAsync("data:text/html,

hi

").get(10, TimeUnit.SECONDS); - JSONObject result = window.eval("document.getElementsByTagName('h1')[0].textContent"); - assertEquals("hi", result.getString("value")); - } - } -} \ No newline at end of file diff --git a/docs/bean-reference.rst b/docs/bean-reference.rst index 825d25ec..7c9e1157 100644 --- a/docs/bean-reference.rst +++ b/docs/bean-reference.rst @@ -378,11 +378,6 @@ FetchWhois Link Extractors --------------- -ExtractorChrome (contrib) -~~~~~~~~~~~~~~~~~~~~~~~~~ - -.. bean-doc:: ../contrib/src/main/java/org/archive/modules/extractor/ExtractorChrome.java - ExtractorCSS ~~~~~~~~~~~~