diff --git a/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java b/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java index b65a3574..bf0a2fc3 100644 --- a/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java +++ b/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java @@ -23,6 +23,7 @@ import java.io.BufferedReader; import java.io.BufferedWriter; import java.io.File; import java.io.FileInputStream; +import java.io.FileNotFoundException; import java.io.FileOutputStream; import java.io.IOException; import java.io.InputStreamReader; @@ -30,56 +31,31 @@ import java.io.OutputStreamWriter; import java.nio.ByteBuffer; import java.nio.CharBuffer; import java.nio.channels.FileChannel; +import java.nio.charset.CharacterCodingException; import java.nio.charset.Charset; +import java.nio.charset.CharsetDecoder; +import java.text.NumberFormat; +import java.util.logging.Level; import java.util.logging.Logger; -import org.archive.util.FileUtils; +import org.archive.util.DevUtils; /** - * Provides a (Replay)CharSequence view on recorded streams (a prefix - * buffer and overflow backing file) that can handle streams of multibyte - * characters. + * (Replay)CharSequence view on recorded streams. * - * For better performance on ISO-8859-1 text, use - * {@link Latin1ByteReplayCharSequence}. - * - *

Call close on this class when done so can clean up resources. + * For small streams, use {@link InMemoryReplayCharSequence}. * - *

Implementation currently works by checking to see if content to read - * all fits the in-memory buffer. If so, we decode into a CharBuffer and - * keep this around for CharSequence operations. This CharBuffer is - * discarded on close. - * - *

If content length is greater than in-memory buffer, we decode the - * buffer plus backing file into a new file named for the backing file w/ - * a suffix of the encoding we write the file as. We then run w/ a - * memory-mapped CharBuffer against this file to implement CharSequence. - * Reasons for this implemenation are that CharSequence wants to return the - * length of the CharSequence. - * - *

Obvious optimizations would keep around decodings whether the - * in-memory decoded buffer or the file of decodings written to disk but the - * general usage pattern processing URIs is that the decoding is used by one - * processor only. Also of note, files usually fit into the in-memory - * buffer. - * - *

We might also be able to keep up 3 windows that moved across the file - * decoding a window at a time trying to keep one of the buffers just in - * front of the regex processing returning it a length that would be only - * the length of current position to end of current block or else the length - * could be got by multipling the backing files length by the decoders' - * estimate of average character size. This would save us writing out the - * decoded file. We'd have to do the latter for files that are - * > Integer.MAX_VALUE. + *

Call {@link close()} on this class when done to clean up resources. * * @author stack + * @author nlevitt * @version $Revision$, $Date$ */ public class GenericReplayCharSequence implements ReplayCharSequence { - protected static Logger logger = - Logger.getLogger(GenericReplayCharSequence.class.getName()); - + protected static Logger logger = Logger + .getLogger(GenericReplayCharSequence.class.getName()); + /** * Name of the encoding we use writing out concatenated decoded prefix * buffer and decoded backing file. @@ -92,12 +68,47 @@ public class GenericReplayCharSequence implements ReplayCharSequence { */ private static final String WRITE_ENCODING = "UTF-16BE"; + private static final long MAP_MAX_BYTES = 64 * 1024 * 1024; // 64M + /** - * CharBuffer of decoded content. - * - * Content of this buffer is unicode. + * When the memory map moves away from the beginning of the file + * (to the "right") in order to reach a certain index, it will + * map up to this many bytes preceding (to the left of) the target character. + * Consequently it will map up to + * MAP_MAX_BYTES - MAP_TARGET_LEFT_PADDING + * bytes to the right of the target. */ - private CharBuffer content = null; + private static final long MAP_TARGET_LEFT_PADDING_BYTES = (long) (MAP_MAX_BYTES * 0.2); + + /** + * Total length of character stream to replay minus the HTTP headers + * if present. + * + * If the backing file is larger than Integer.MAX_VALUE (i.e. 2gb), + * only the first Integer.MAX_VALUE characters are available through this API. + * We're overriding java.lang.CharSequence so that we can use + * java.util.regex directly on the data, and the CharSequence + * API uses int for the length and index. + */ + protected int length; + + /** + * Byte offset into the file where the memory mapped portion begins. + */ + private long mapByteOffset; + + // XXX do we need to keep the input stream around? + private FileInputStream backingFileIn = null; + + private FileChannel backingFileChannel = null; + + private long bytesPerChar; + + private ByteBuffer mappedBuffer = null; + + private CharsetDecoder decoder = null; + + private ByteBuffer tempBuf = null; /** * File that has decoded content. @@ -106,9 +117,15 @@ public class GenericReplayCharSequence implements ReplayCharSequence { */ private File decodedFile = null; + /* + * This portion of the CharSequence precedes what's in the backing file. In + * cases where we decodeToFile(), this is always empty, because we decode + * the entire input stream. + */ + private CharBuffer prefixBuffer = null; /** - * Constructor for all in-memory operation. + * Constructor. * * @param buffer In-memory buffer of recordings prefix. We read from * here first and will only go to the backing file if size @@ -116,225 +133,257 @@ public class GenericReplayCharSequence implements ReplayCharSequence { * @param size Total size of stream to replay in bytes. Used to find * EOS. This is total length of content including HTTP headers if * present. - * @param responseBodyStart Where the response body starts in bytes. - * Used to skip over the HTTP headers if present. + * @param contentReplayInputStream inputStream of content + * @param charsetName Encoding to use reading the passed prefix + * buffer and backing file. For now, should be java canonical name for the + * encoding. Must not be null. * @param backingFilename Path to backing file with content in excess of * whats in buffer. - * @param encoding Encoding to use reading the passed prefix buffer and - * backing file. For now, should be java canonical name for the - * encoding. (If null is passed, we will default to - * ByteReplayCharSequence). - * - * @throws IOException - */ - public GenericReplayCharSequence(byte[] buffer, long size, - long responseBodyStart, String encoding) - throws IOException { - super(); - this.content = decodeInMemory(buffer, size, responseBodyStart, - encoding); - } - - /** - * Constructor for overflow-to-disk-file operation. - * - * @param contentReplayInputStream inputStream of content - * @param backingFilename hint for name of temp file - * @param characterEncoding Encoding to use reading the stream. - * For now, should be java canonical name for the - * encoding. * * @throws IOException */ public GenericReplayCharSequence( - ReplayInputStream contentReplayInputStream, - String backingFilename, - String characterEncoding) - throws IOException { + ReplayInputStream contentReplayInputStream, String backingFilename, + String charsetName) throws IOException { super(); - this.content = decodeToFile(contentReplayInputStream, - backingFilename, characterEncoding); + logger.info("new GenericReplayCharSequence() characterEncoding=" + + charsetName + " backingFilename=" + backingFilename); + + Charset charset = Charset.forName(charsetName); + if (charset.newEncoder().maxBytesPerChar() == 1.0) { + logger.info("charset=" + charsetName + + ": supports random access, using backing file directly"); + this.bytesPerChar = 1; + this.backingFileIn = new FileInputStream(backingFilename); + this.decoder = charset.newDecoder(); + this.prefixBuffer = this.decoder.decode( + ByteBuffer.wrap(contentReplayInputStream.getBuffer())); + } else { + logger.info("charset=" + charsetName + + ": may not support random access, decoding to separate file"); + + // decodes only up to Integer.MAX_VALUE characters + decodeToFile(contentReplayInputStream, backingFilename, charsetName); + + this.bytesPerChar = 2; + this.backingFileIn = new FileInputStream(decodedFile); + this.decoder = Charset.forName(WRITE_ENCODING).newDecoder(); + this.prefixBuffer = CharBuffer.wrap(""); + } + + this.tempBuf = ByteBuffer.wrap(new byte[(int) this.bytesPerChar]); + this.backingFileChannel = backingFileIn.getChannel(); + + // we only support the first Integer.MAX_VALUE characters + long wouldBeLength = prefixBuffer.limit() + backingFileChannel.size() / bytesPerChar; + if (wouldBeLength <= Integer.MAX_VALUE) { + this.length = (int) wouldBeLength; + } else { + logger.warning("input stream is longer than Integer.MAX_VALUE=" + + NumberFormat.getInstance().format(Integer.MAX_VALUE) + + " characters -- only first " + + NumberFormat.getInstance().format(Integer.MAX_VALUE) + + " are accessible through this GenericReplayCharSequence"); + this.length = Integer.MAX_VALUE; + } + + this.mapByteOffset = 0; + updateMemoryMappedBuffer(); + } + + private void updateMemoryMappedBuffer() { + long fileLength = (long) this.length() - (long) prefixBuffer.limit(); // in characters + long mapSize = Math.min(fileLength * bytesPerChar - mapByteOffset, MAP_MAX_BYTES); + logger.fine("updateMemoryMappedBuffer: mapOffset=" + + NumberFormat.getInstance().format(mapByteOffset) + + " mapSize=" + NumberFormat.getInstance().format(mapSize)); + try { + System.gc(); + System.runFinalization(); + // TODO: Confirm the READ_ONLY works. I recall it not working. + // The buffers seem to always say that the buffer is writable. + mappedBuffer = backingFileChannel.map( + FileChannel.MapMode.READ_ONLY, mapByteOffset, mapSize) + .asReadOnlyBuffer(); + } catch (IOException e) { + // TODO convert this to a runtime error? + DevUtils.logger.log(Level.SEVERE, + " backingFileChannel.map() mapByteOffset=" + mapByteOffset + + " mapSize=" + mapSize + "\n" + "decodedFile=" + + decodedFile + " length=" + length + "\n" + + DevUtils.extraInfo(), e); + throw new RuntimeException(e); + } } /** - * Decode passed buffer and backing file into a CharBuffer. - * - * This method writes a new file made of the decoded concatenation of - * the in-memory prefix buffer and the backing file. Returns a - * charSequence view onto this new file. - * - * @param buffer In-memory buffer of recordings prefix. We read from - * here first and will only go to the backing file if size - * requested is greater than buffer.length. - * @param size Total size of stream to replay in bytes. Used to find - * EOS. This is total length of content including HTTP headers if - * present. - * @param responseBodyStart Where the response body starts in bytes. - * Used to skip over the HTTP headers if present. - * @param backingFilename Path to backing file with content in excess of - * whats in buffer. - * @param encoding Encoding to use reading the passed prefix buffer and - * backing file. For now, should be java canonical name for the - * encoding. (If null is passed, we will default to - * ByteReplayCharSequence). - * - * @return A CharBuffer view on decodings of the contents of passed - * buffer. + * Converts the first Integer.MAX_VALUE characters from the + * file backingFilename from encoding encoding to + * encoding WRITE_ENCODING and saves as + * this.decodedFile, which is named backingFilename + * + "." + WRITE_ENCODING. + * * @throws IOException */ - private CharBuffer decodeToFile(ReplayInputStream inStream, - String backingFilename, String encoding) - throws IOException { + private void decodeToFile(ReplayInputStream inStream, + String backingFilename, String encoding) throws IOException { - CharBuffer charBuffer = null; + BufferedReader reader = new BufferedReader(new InputStreamReader( + inStream, encoding)); + + this.decodedFile = new File(backingFilename + "." + WRITE_ENCODING); + + logger.info("decodeToFile: backingFilename=" + backingFilename + + " encoding=" + encoding + " decodedFile=" + decodedFile); - BufferedReader reader = new BufferedReader( - new InputStreamReader(inStream,encoding)); - - File backingFile = new File(backingFilename); - this.decodedFile = File.createTempFile( - backingFile.getName(), WRITE_ENCODING, backingFile.getParentFile()); FileOutputStream fos; - fos = new FileOutputStream(this.decodedFile); - - BufferedWriter writer = new BufferedWriter( - new OutputStreamWriter( - fos, - WRITE_ENCODING)); + try { + fos = new FileOutputStream(this.decodedFile); + } catch (FileNotFoundException e) { + // Windows workaround attempt + System.gc(); + System.runFinalization(); + this.decodedFile = new File(decodedFile.getAbsolutePath()+".win"); + logger.info("Windows 'file with a user-mapped section open' " + + "workaround gc/finalization/name-extension performed."); + // try again + fos = new FileOutputStream(this.decodedFile); + } + BufferedWriter writer = new BufferedWriter(new OutputStreamWriter(fos, + WRITE_ENCODING)); int c; - while((c = reader.read())>=0) { + long count = 0; + while ((c = reader.read()) >= 0 && count < Integer.MAX_VALUE) { writer.write(c); + count++; + if (count % 100000000 == 0) { + logger.fine("wrote " + count + " characters so far..."); + } } writer.close(); - - charBuffer = getReadOnlyMemoryMappedBuffer(this.decodedFile). - asCharBuffer(); - return charBuffer; + logger.info("decodeToFile: wrote " + count + " characters to " + + decodedFile); } /** - * Decode passed buffer into a CharBuffer. - * - * This method decodes a memory buffer returning a memory buffer. - * - * @param buffer In-memory buffer of recordings prefix. We read from - * here first and will only go to the backing file if size - * requested is greater than buffer.length. - * @param size Total size of stream to replay in bytes. Used to find - * EOS. This is total length of content including HTTP headers if - * present. - * @param responseBodyStart Where the response body starts in bytes. - * Used to skip over the HTTP headers if present. - * @param encoding Encoding to use reading the passed prefix buffer and - * backing file. For now, should be java canonical name for the - * encoding. (If null is passed, we will default to - * ByteReplayCharSequence). - * - * @return A CharBuffer view on decodings of the contents of passed - * buffer. + * Get character at passed absolute position. + * @param index Index into content + * @return Character at offset index. */ - private CharBuffer decodeInMemory(byte[] buffer, long size, - long responseBodyStart, String encoding) - { - ByteBuffer bb = ByteBuffer.wrap(buffer); - // Move past the HTTP header if present. - bb.position((int)responseBodyStart); - // Set the end-of-buffer to be end-of-content. - bb.limit((int)size); - return (Charset.forName(encoding)).decode(bb).asReadOnlyBuffer(); - } + public char charAt(int index) { + if (index < 0 || index >= this.length()) { + throw new IndexOutOfBoundsException("index=" + index + + " - should be between 0 and length()=" + this.length()); + } - /** - * Create read-only memory-mapped buffer onto passed file. - * - * @param file File to get memory-mapped buffer on. - * @return Read-only memory-mapped ByteBuffer view on to passed file. - * @throws IOException - */ - private ByteBuffer getReadOnlyMemoryMappedBuffer(File file) - throws IOException { + // is it in the buffer + if (index < prefixBuffer.limit()) { + return prefixBuffer.get(index); + } - ByteBuffer bb = null; - FileInputStream in = null; - FileChannel c = null; - assert file.exists(): "No file " + file.getAbsolutePath(); + // otherwise we gotta get it from disk via memory map + long fileIndex = (long) index - (long) prefixBuffer.limit(); + long fileLength = (long) this.length() - (long) prefixBuffer.limit(); // in characters + if (fileIndex * bytesPerChar < mapByteOffset + || fileIndex * bytesPerChar - mapByteOffset >= mappedBuffer.limit()) { + // fault + /* + * mapByteOffset is bounded by 0 and file size +/- size of the map, + * and starts as close to fileIndex - + * MAP_TARGET_LEFT_PADDING_BYTES as it can while also not + * being smaller than it needs to be. + */ + mapByteOffset = Math.max(0, fileIndex * bytesPerChar - MAP_TARGET_LEFT_PADDING_BYTES); + mapByteOffset = Math.min(mapByteOffset, fileLength * bytesPerChar - MAP_MAX_BYTES); + updateMemoryMappedBuffer(); + } + + // CharsetDecoder always decodes up to the end of the ByteBuffer, so we + // create a new ByteBuffer with only the bytes we're interested in. + mappedBuffer.position((int) (fileIndex * bytesPerChar - mapByteOffset)); + mappedBuffer.get(tempBuf.array()); + tempBuf.position(0); // decoder starts at this position try { - in = new FileInputStream(file); - c = in.getChannel(); - // TODO: Confirm the READ_ONLY works. I recall it not working. - // The buffers seem to always say that the buffer is writeable. - bb = c.map(FileChannel.MapMode.READ_ONLY, 0, c.size()). - asReadOnlyBuffer(); + CharBuffer cbuf = decoder.decode(tempBuf); + return cbuf.get(); + } catch (CharacterCodingException e) { + logger.warning("unable to get character at index=" + index + " (fileIndex=" + fileIndex + "): " + e); + // U+FFFD REPLACEMENT CHARACTER -- + // "used to replace an incoming character whose value is unknown or unrepresentable in Unicode" + return (char) 0xfffd; } - - finally { - if (c != null && c.isOpen()) { - c.close(); - } - if (in != null) { - in.close(); - } - } - - return bb; - } - - private void deleteFile(File fileToDelete) { - deleteFile(fileToDelete, null); - } - - private void deleteFile(File fileToDelete, final Exception e) { - if (e != null) { - // Log why the delete to help with debug of java.io.FileNotFoundException: - // ....tt53http.ris.UTF-16BE. - logger.severe("Deleting " + fileToDelete + " because of " - + e.toString()); - } - if (fileToDelete != null && fileToDelete.exists()) { - FileUtils.deleteSoonerOrLater(fileToDelete); - } - } - - public void close() - { - this.content = null; - deleteFile(this.decodedFile); - // clear decodedFile -- so that double-close (as in - // finalize()) won't delete a later instance with same name - // see bug [ 1218961 ] "failed get of replay" in ExtractorHTML... usu: UTF-16BE - this.decodedFile = null; - } - - protected void finalize() throws Throwable - { - super.finalize(); - // Maybe TODO: eliminate close here, requiring explicit close instead - close(); - } - - public int length() - { - return this.content.limit(); - } - - public char charAt(int index) - { - return this.content.get(index); } public CharSequence subSequence(int start, int end) { return new CharSubSequence(this, start, end); } - - public String toString() { - StringBuffer sb = new StringBuffer(length()); - // could use StringBuffer.append(CharSequence) if willing to do 1.5 & up - for (int i = 0;iUses a wraparound rolling buffer of the last windowSize bytes read - * from disk in memory; as long as the 'random access' of a CharSequence - * user stays within this window, access should remain fairly efficient. - * (So design any regexps pointed at these CharSequences to work within - * that range!) - * - *

When rereading of a location is necessary, the whole window is - * recentered around the location requested. (TODO: More research - * into whether this is the best strategy.) - * - *

An implementation of a ReplayCharSequence done with ByteBuffers -- one - * to wrap the passed prefix buffer and the second, a memory-mapped - * ByteBuffer view into the backing file -- was consistently slower: ~10%. - * My tests did the following. Made a buffer filled w/ regular content. - * This buffer was used as the prefix buffer. The buffer content was - * written MULTIPLER times to a backing file. I then did accesses w/ the - * following pattern: Skip forward 32 bytes, then back 16 bytes, and then - * read forward from byte 16-32. Repeat. Though I varied the size of the - * buffer to the size of the backing file,from 3-10, the difference of 10% - * or so seemed to persist. Same if I tried to favor get() over get(index). - * I used a profiler, JMP, to study times taken (St.Ack did above comment). - * - *

TODO determine in memory mapped files is better way to do this; - * probably not -- they don't offer the level of control over - * total memory used that this approach does. - * - * @author Gordon Mohr - * @version $Revision$, $Date$ - */ -class Latin1ByteReplayCharSequence implements ReplayCharSequence { - - protected static Logger logger = - Logger.getLogger(Latin1ByteReplayCharSequence.class.getName()); - - /** - * Buffer that holds the first bit of content. - * - * Once this is exhausted we go to the backing file. - */ - private byte[] prefixBuffer; - - /** - * Total length of character stream to replay minus the HTTP headers - * if present. - * - * Used to find EOS. - */ - protected int length; - - /** - * Absolute length of the stream. - * - * Includes HTTP headers. Needed doing calc. in the below figuring - * how much to load into buffer. - */ - private int absoluteLength = -1; - - /** - * Buffer window on to backing file. - */ - private byte[] wraparoundBuffer; - - /** - * Absolute index into underlying bytestream where wrap starts. - */ - private int wrapOrigin; - - /** - * Index in wraparoundBuffer that corresponds to wrapOrigin - */ - private int wrapOffset; - - /** - * Name of backing file we go to when we've exhausted content from the - * prefix buffer. - */ - private String backingFilename; - - /** - * Random access to the backing file. - */ - private RandomAccessFile raFile; - - /** - * Offset into prefix buffer at which content beings. - */ - private int contentOffset; - - /** - * 8-bit encoding used reading single bytes from buffer and - * stream. - */ - @SuppressWarnings("unused") - private static final String DEFAULT_SINGLE_BYTE_ENCODING = - "ISO-8859-1"; - - - /** - * Constructor. - * - * @param buffer In-memory buffer of recordings prefix. We read from - * here first and will only go to the backing file if size - * requested is greater than buffer.length. - * @param size Total size of stream to replay in bytes. Used to find - * EOS. This is total length of content including HTTP headers if - * present. - * @param responseBodyStart Where the response body starts in bytes. - * Used to skip over the HTTP headers if present. - * @param backingFilename Path to backing file with content in excess of - * whats in buffer. - * - * @throws IOException - */ - public Latin1ByteReplayCharSequence(byte[] buffer, long size, - long responseBodyStart, String backingFilename) - throws IOException { - - this.length = (int)(size - responseBodyStart); - this.absoluteLength = (int)size; - this.prefixBuffer = buffer; - this.contentOffset = (int)responseBodyStart; - - // If amount to read is > than what is in our prefix buffer, then - // open the backing file. - if (size > buffer.length) { - this.backingFilename = backingFilename; - this.raFile = new RandomAccessFile(backingFilename, "r"); - this.wraparoundBuffer = new byte[this.prefixBuffer.length]; - this.wrapOrigin = this.prefixBuffer.length; - this.wrapOffset = 0; - loadBuffer(); - } - } - - /** - * @return Length of characters in stream to replay. Starts counting - * at the HTTP header/body boundary. - */ - public int length() { - return this.length; - } - - /** - * Get character at passed absolute position. - * - * Called by {@link #charAt(int)} which has a relative index into the - * content, one that doesn't account for HTTP header if present. - * - * @param index Index into content adjusted to accomodate initial offset - * to get us past the HTTP header if present (i.e. - * {@link #contentOffset}). - * - * @return Characater at offset index. - */ - public char charAt(int index) { - int c = -1; - // Add to index start-of-content offset to get us over HTTP header - // if present. - index += this.contentOffset; - if (index < this.prefixBuffer.length) { - // If index is into our prefix buffer. - c = this.prefixBuffer[index]; - } else if (index >= this.wrapOrigin && - (index - this.wrapOrigin) < this.wraparoundBuffer.length) { - // If index is into our buffer window on underlying backing file. - c = this.wraparoundBuffer[ - ((index - this.wrapOrigin) + this.wrapOffset) % - this.wraparoundBuffer.length]; - } else { - // Index is outside of both prefix buffer and our buffer window - // onto the underlying backing file. Fix the buffer window - // location. - c = faultCharAt(index); - } - // Stream is treated as single byte. Make sure characters returned - // are not negative. - return (char)(c & 0xff); - } - - /** - * Get a character that's outside the current buffers. - * - * will cause the wraparoundBuffer to be changed to - * cover a region including the index - * - * if index is higher than the highest index in the - * wraparound buffer, buffer is moved forward such - * that requested char is last item in buffer - * - * if index is lower than lowest index in the - * wraparound buffer, buffet is reset centered around - * index - * - * @param index Index of character to fetch. - * @return A character that's outside the current buffers - */ - private int faultCharAt(int index) { - if(Thread.interrupted()) { - throw new RuntimeException("thread interrupted"); - } - if(index >= this.wrapOrigin + this.wraparoundBuffer.length) { - // Moving forward - while (index >= this.wrapOrigin + this.wraparoundBuffer.length) - { - // TODO optimize this - advanceBuffer(); - } - return charAt(index - this.contentOffset); - } - // Moving backward - recenterBuffer(index); - return charAt(index - this.contentOffset); - } - - /** - * Move the buffer window on backing file back centering current access - * position in middle of window. - * - * @param index Index of character to access. - */ - private void recenterBuffer(int index) { - if (logger.isLoggable(Level.FINE)) { - logger.fine("Recentering around " + index + " in " + - this.backingFilename); - } - this.wrapOrigin = index - (this.wraparoundBuffer.length / 2); - if(this.wrapOrigin < this.prefixBuffer.length) { - this.wrapOrigin = this.prefixBuffer.length; - } - this.wrapOffset = 0; - loadBuffer(); - } - - /** - * Load from backing file into the wrapper buffer. - */ - private void loadBuffer() - { - long len = -1; - try { - len = this.raFile.length(); - this.raFile.seek(this.wrapOrigin - this.prefixBuffer.length); - this.raFile.readFully(this.wraparoundBuffer, 0, - Math.min(this.wraparoundBuffer.length, - this.absoluteLength - this.wrapOrigin)); - } - - catch (IOException e) { - // TODO convert this to a runtime error? - DevUtils.logger.log ( - Level.SEVERE, - "raFile.seek(" + - (this.wrapOrigin - this.prefixBuffer.length) + - ")\n" + - "raFile.readFully(wraparoundBuffer,0," + - (Math.min(this.wraparoundBuffer.length, - this.length - this.wrapOrigin )) + - ")\n"+ - "raFile.length()" + len + "\n" + - DevUtils.extraInfo(), - e); - throw new RuntimeException(e); - } - } - - /** - * Roll the wraparound buffer forward one position - */ - private void advanceBuffer() { - try { - this.wraparoundBuffer[this.wrapOffset] = - (byte)this.raFile.read(); - this.wrapOffset++; - this.wrapOffset %= this.wraparoundBuffer.length; - this.wrapOrigin++; - } catch (IOException e) { - DevUtils.logger.log(Level.SEVERE, "advanceBuffer()" + - DevUtils.extraInfo(), e); - throw new RuntimeException(e); - } - } - - public CharSequence subSequence(int start, int end) { - return new CharSubSequence(this, start, end); - } - - /** - * Cleanup resources. - * - * @exception IOException Failed close of random access file. - */ - public void close() throws IOException - { - this.prefixBuffer = null; - if (this.raFile != null) { - this.raFile.close(); - this.raFile = null; - } - } - - /* (non-Javadoc) - * @see java.lang.Object#finalize() - */ - protected void finalize() throws Throwable - { - super.finalize(); - close(); - } - - /** - * Convenience method for getting a substring. - * @deprecated please use subSequence() and then toString() directly - */ - public String substring(int offset, int len) { - return subSequence(offset, offset+len).toString(); - } - - /* (non-Javadoc) - * @see java.lang.Object#toString() - */ - public String toString() { - StringBuilder sb = new StringBuilder(this.length()); - sb.append(this); - return sb.toString(); - } -} \ No newline at end of file diff --git a/commons/src/main/java/org/archive/io/RecordingOutputStream.java b/commons/src/main/java/org/archive/io/RecordingOutputStream.java index 01f64a05..edc1c502 100644 --- a/commons/src/main/java/org/archive/io/RecordingOutputStream.java +++ b/commons/src/main/java/org/archive/io/RecordingOutputStream.java @@ -1,27 +1,22 @@ -/* ReplayableOutputStream +/* + * This file is part of the Heritrix web crawler (crawler.archive.org). * - * $Id$ + * Licensed to the Internet Archive (IA) by one or more individual + * contributors. * - * Created on Sep 23, 2003 + * The IA licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at * - * Copyright (C) 2003 Internet Archive. + * http://www.apache.org/licenses/LICENSE-2.0 * - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Heritrix is free software; you can redistribute it and/or modify - * it under the terms of the GNU Lesser Public License as published by - * the Free Software Foundation; either version 2.1 of the License, or - * any later version. - * - * Heritrix is distributed in the hope that it will be useful, - * but WITHOUT ANY WARRANTY; without even the implied warranty of - * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the - * GNU Lesser Public License for more details. - * - * You should have received a copy of the GNU Lesser Public License - * along with Heritrix; if not, write to the Free Software - * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. */ + package org.archive.io; import it.unimi.dsi.fastutil.io.FastBufferedOutputStream; @@ -513,9 +508,7 @@ public class RecordingOutputStream extends OutputStream { throws IOException { return getReplayCharSequence(characterEncoding, this.contentBeginMark); } - - private static final String canonicalLatin1 = Charset.forName("iso8859-1").name(); - + /** * @param characterEncoding Encoding of recorded stream. * @return A ReplayCharSequence Will return null if an IOException. Call @@ -524,39 +517,27 @@ public class RecordingOutputStream extends OutputStream { */ public ReplayCharSequence getReplayCharSequence(String characterEncoding, long startOffset) throws IOException { - if (characterEncoding == null) { + if (characterEncoding == null) characterEncoding = Charset.defaultCharset().name(); - } - // TODO: handled transfer-encoding: chunked content-bodies properly - if (canonicalLatin1.equals(Charset.forName(characterEncoding).name())) { - return new Latin1ByteReplayCharSequence( + logger.fine("this.size=" + this.size + " this.buffer.length=" + this.buffer.length); + if (this.size <= this.buffer.length) { + logger.fine("using InMemoryReplayCharSequence"); + // raw data is all in memory; do in memory + return new InMemoryReplayCharSequence( this.buffer, this.size, startOffset, - this.backingFilename); - } else { - // multibyte - if(this.size <= this.buffer.length) { - // raw data is all in memory; do in memory - return new GenericReplayCharSequence( - this.buffer, - this.size, - startOffset, - characterEncoding); - - } else { - // raw data overflows to disk; use temp file - ReplayInputStream ris = getReplayInputStream(startOffset); - ReplayCharSequence rcs = new GenericReplayCharSequence( - ris, - this.backingFilename, - characterEncoding); - ris.close(); - return rcs; - } - + characterEncoding); + } + else { + logger.fine("using GenericReplayCharSequence"); + // raw data overflows to disk; use temp file + ReplayInputStream ris = getReplayInputStream(startOffset); + return new GenericReplayCharSequence( + ris, + this.backingFilename, + characterEncoding); } - } public long getResponseContentLength() { @@ -624,3 +605,4 @@ public class RecordingOutputStream extends OutputStream { return maxLength - position; } } + diff --git a/commons/src/main/java/org/archive/io/ReplayInputStream.java b/commons/src/main/java/org/archive/io/ReplayInputStream.java index a590ccdb..04656d4b 100644 --- a/commons/src/main/java/org/archive/io/ReplayInputStream.java +++ b/commons/src/main/java/org/archive/io/ReplayInputStream.java @@ -1,27 +1,22 @@ -/* ReplayInputStream +/* + * This file is part of the Heritrix web crawler (crawler.archive.org). * - * $Id$ + * Licensed to the Internet Archive (IA) by one or more individual + * contributors. * - * Created on Sep 24, 2003 + * The IA licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at * - * Copyright (C) 2003 Internet Archive. + * http://www.apache.org/licenses/LICENSE-2.0 * - * This file is part of the Heritrix web crawler (crawler.archive.org). - * - * Heritrix is free software; you can redistribute it and/or modify - * it under the terms of the GNU Lesser Public License as published by - * the Free Software Foundation; either version 2.1 of the License, or - * any later version. - * - * Heritrix is distributed in the hope that it will be useful, - * but WITHOUT ANY WARRANTY; without even the implied warranty of - * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the - * GNU Lesser Public License for more details. - * - * You should have received a copy of the GNU Lesser Public License - * along with Heritrix; if not, write to the Free Software - * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. */ + package org.archive.io; import java.io.File; @@ -279,4 +274,9 @@ public class ReplayInputStream extends SeekInputStream public long position() throws IOException { return position; } + + // package private + byte[] getBuffer() { + return buffer; + } } diff --git a/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java b/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java index fb7d229f..11b84f56 100644 --- a/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java +++ b/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java @@ -20,7 +20,9 @@ package org.archive.io; import java.io.IOException; +import java.text.NumberFormat; import java.util.Date; +import java.util.Random; import java.util.logging.Logger; import org.archive.util.FileUtils; @@ -210,14 +212,24 @@ public class ReplayCharSequenceTest extends TmpDirTestCase } public void testReplayCharSequenceByteToStringOverflow() throws IOException { - String fileContent = "Some file content. "; + String fileContent = "Some file content. "; // ascii byte [] buffer = fileContent.getBytes(); RecordingOutputStream ros = writeTestStream( buffer,1, - "testReplayCharSequenceByteToString.txt",1); + "testReplayCharSequenceByteToStringOverflow.txt",1); String expectedContent = fileContent+fileContent; - ReplayCharSequence rcs = ros.getReplayCharSequence(); - String result = rcs.toString(); + + // The string is ascii which is a subset of both these encodings. Use + // both encodings because they exercise different code paths. UTF-8 is + // decoded to UTF-16 while windows-1252 is memory mapped directly. See + // GenericReplayCharSequence + ReplayCharSequence rcsUtf8 = ros.getReplayCharSequence("UTF-8"); + ReplayCharSequence rcs1252 = ros.getReplayCharSequence("windows-1252"); + + String result = rcsUtf8.toString(); + assertEquals("Strings don't match", expectedContent, result); + + result = rcs1252.toString(); assertEquals("Strings don't match", expectedContent, result); } @@ -244,6 +256,67 @@ public class ReplayCharSequenceTest extends TmpDirTestCase } } + public void xestHugeReplayCharSequence() throws IOException { + String fileContent = "01234567890123456789"; + String characterEncoding = "ascii"; + byte[] buffer = fileContent.getBytes(characterEncoding); + + long reps = (long) Integer.MAX_VALUE / (long) buffer.length + 1000000l; + + logger.info("writing " + (reps * buffer.length) + + " bytes to testHugeReplayCharSequence.txt"); + RecordingOutputStream ros = writeTestStream(buffer, 0, + "testHugeReplayCharSequence.txt", reps); + ReplayCharSequence rcs = ros.getReplayCharSequence(characterEncoding); + + if (reps * fileContent.length() > (long) Integer.MAX_VALUE) { + assertTrue("ReplayCharSequence has wrong length (length()=" + + rcs.length() + ") (should be " + Integer.MAX_VALUE + ")", + rcs.length() == Integer.MAX_VALUE); + } else { + assertEquals("ReplayCharSequence has wrong length (length()=" + + rcs.length() + ") (should be " + + (reps * fileContent.length()) + ")", (long) rcs.length(), + reps * (long) fileContent.length()); + } + + // boundary cases or something + for (int index : new int[] { 0, rcs.length() / 4, rcs.length() / 2, + rcs.length() - 1, rcs.length() / 4 }) { + // logger.info("testing char at index=" + + // NumberFormat.getInstance().format(index)); + assertEquals("Characters don't match (index=" + + NumberFormat.getInstance().format(index) + ")", + fileContent.charAt(index % fileContent.length()), rcs + .charAt(index)); + } + + // check that out of bounds indices throw exception + for (int n : new int[] { -1, Integer.MIN_VALUE, rcs.length() + 1 }) { + try { + String message = "rcs.charAt(" + n + ")=" + rcs.charAt(n) + + " ?!? -- expected IndexOutOfBoundsException"; + logger.severe(message); + fail(message); + } catch (IndexOutOfBoundsException e) { + logger.info("got expected exception: " + e); + } + } + + // check some characters at random spots & kinda stress test the + // system's memory mapping facility + Random rand = new Random(0); // seed so we get the same ones each time + for (int i = 0; i < 5000; i++) { + int index = rand.nextInt(rcs.length()); + // logger.info(i + ". testing char at index=" + + // NumberFormat.getInstance().format(index)); + assertEquals("Characters don't match (index=" + + NumberFormat.getInstance().format(index) + ")", + fileContent.charAt(index % fileContent.length()), rcs + .charAt(index)); + } + } + /** * Accessing characters test. * @@ -290,13 +363,14 @@ public class ReplayCharSequenceTest extends TmpDirTestCase * @throws IOException */ private RecordingOutputStream writeTestStream(byte[] content, - int memReps, String baseName, int fileReps) throws IOException { + int memReps, String baseName, long fileReps) throws IOException { String backingFilename = FileUtils.maybeRelative(getTmpDir(),baseName).getAbsolutePath(); RecordingOutputStream ros = new RecordingOutputStream( content.length * memReps, backingFilename); ros.open(); - for(int i = 0; i < (memReps+fileReps); i++) { + ros.markContentBegin(); + for(long i = 0; i < (memReps+fileReps); i++) { // fill buffer (repeat MULTIPLIER times) and // overflow to disk (also MULTIPLIER times) ros.write(content);