diff --git a/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java b/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java index b65a3574..bf0a2fc3 100644 --- a/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java +++ b/commons/src/main/java/org/archive/io/GenericReplayCharSequence.java @@ -23,6 +23,7 @@ import java.io.BufferedReader; import java.io.BufferedWriter; import java.io.File; import java.io.FileInputStream; +import java.io.FileNotFoundException; import java.io.FileOutputStream; import java.io.IOException; import java.io.InputStreamReader; @@ -30,56 +31,31 @@ import java.io.OutputStreamWriter; import java.nio.ByteBuffer; import java.nio.CharBuffer; import java.nio.channels.FileChannel; +import java.nio.charset.CharacterCodingException; import java.nio.charset.Charset; +import java.nio.charset.CharsetDecoder; +import java.text.NumberFormat; +import java.util.logging.Level; import java.util.logging.Logger; -import org.archive.util.FileUtils; +import org.archive.util.DevUtils; /** - * Provides a (Replay)CharSequence view on recorded streams (a prefix - * buffer and overflow backing file) that can handle streams of multibyte - * characters. + * (Replay)CharSequence view on recorded streams. * - * For better performance on ISO-8859-1 text, use - * {@link Latin1ByteReplayCharSequence}. - * - *
Call close on this class when done so can clean up resources. + * For small streams, use {@link InMemoryReplayCharSequence}. * - *
Implementation currently works by checking to see if content to read - * all fits the in-memory buffer. If so, we decode into a CharBuffer and - * keep this around for CharSequence operations. This CharBuffer is - * discarded on close. - * - *
If content length is greater than in-memory buffer, we decode the - * buffer plus backing file into a new file named for the backing file w/ - * a suffix of the encoding we write the file as. We then run w/ a - * memory-mapped CharBuffer against this file to implement CharSequence. - * Reasons for this implemenation are that CharSequence wants to return the - * length of the CharSequence. - * - *
Obvious optimizations would keep around decodings whether the - * in-memory decoded buffer or the file of decodings written to disk but the - * general usage pattern processing URIs is that the decoding is used by one - * processor only. Also of note, files usually fit into the in-memory - * buffer. - * - *
We might also be able to keep up 3 windows that moved across the file - * decoding a window at a time trying to keep one of the buffers just in - * front of the regex processing returning it a length that would be only - * the length of current position to end of current block or else the length - * could be got by multipling the backing files length by the decoders' - * estimate of average character size. This would save us writing out the - * decoded file. We'd have to do the latter for files that are - * > Integer.MAX_VALUE. + *
Call {@link close()} on this class when done to clean up resources.
*
* @author stack
+ * @author nlevitt
* @version $Revision$, $Date$
*/
public class GenericReplayCharSequence implements ReplayCharSequence {
- protected static Logger logger =
- Logger.getLogger(GenericReplayCharSequence.class.getName());
-
+ protected static Logger logger = Logger
+ .getLogger(GenericReplayCharSequence.class.getName());
+
/**
* Name of the encoding we use writing out concatenated decoded prefix
* buffer and decoded backing file.
@@ -92,12 +68,47 @@ public class GenericReplayCharSequence implements ReplayCharSequence {
*/
private static final String WRITE_ENCODING = "UTF-16BE";
+ private static final long MAP_MAX_BYTES = 64 * 1024 * 1024; // 64M
+
/**
- * CharBuffer of decoded content.
- *
- * Content of this buffer is unicode.
+ * When the memory map moves away from the beginning of the file
+ * (to the "right") in order to reach a certain index, it will
+ * map up to this many bytes preceding (to the left of) the target character.
+ * Consequently it will map up to
+ * When rereading of a location is necessary, the whole window is
- * recentered around the location requested. (TODO: More research
- * into whether this is the best strategy.)
- *
- * An implementation of a ReplayCharSequence done with ByteBuffers -- one
- * to wrap the passed prefix buffer and the second, a memory-mapped
- * ByteBuffer view into the backing file -- was consistently slower: ~10%.
- * My tests did the following. Made a buffer filled w/ regular content.
- * This buffer was used as the prefix buffer. The buffer content was
- * written MULTIPLER times to a backing file. I then did accesses w/ the
- * following pattern: Skip forward 32 bytes, then back 16 bytes, and then
- * read forward from byte 16-32. Repeat. Though I varied the size of the
- * buffer to the size of the backing file,from 3-10, the difference of 10%
- * or so seemed to persist. Same if I tried to favor get() over get(index).
- * I used a profiler, JMP, to study times taken (St.Ack did above comment).
- *
- * TODO determine in memory mapped files is better way to do this;
- * probably not -- they don't offer the level of control over
- * total memory used that this approach does.
- *
- * @author Gordon Mohr
- * @version $Revision$, $Date$
- */
-class Latin1ByteReplayCharSequence implements ReplayCharSequence {
-
- protected static Logger logger =
- Logger.getLogger(Latin1ByteReplayCharSequence.class.getName());
-
- /**
- * Buffer that holds the first bit of content.
- *
- * Once this is exhausted we go to the backing file.
- */
- private byte[] prefixBuffer;
-
- /**
- * Total length of character stream to replay minus the HTTP headers
- * if present.
- *
- * Used to find EOS.
- */
- protected int length;
-
- /**
- * Absolute length of the stream.
- *
- * Includes HTTP headers. Needed doing calc. in the below figuring
- * how much to load into buffer.
- */
- private int absoluteLength = -1;
-
- /**
- * Buffer window on to backing file.
- */
- private byte[] wraparoundBuffer;
-
- /**
- * Absolute index into underlying bytestream where wrap starts.
- */
- private int wrapOrigin;
-
- /**
- * Index in wraparoundBuffer that corresponds to wrapOrigin
- */
- private int wrapOffset;
-
- /**
- * Name of backing file we go to when we've exhausted content from the
- * prefix buffer.
- */
- private String backingFilename;
-
- /**
- * Random access to the backing file.
- */
- private RandomAccessFile raFile;
-
- /**
- * Offset into prefix buffer at which content beings.
- */
- private int contentOffset;
-
- /**
- * 8-bit encoding used reading single bytes from buffer and
- * stream.
- */
- @SuppressWarnings("unused")
- private static final String DEFAULT_SINGLE_BYTE_ENCODING =
- "ISO-8859-1";
-
-
- /**
- * Constructor.
- *
- * @param buffer In-memory buffer of recordings prefix. We read from
- * here first and will only go to the backing file if MAP_MAX_BYTES - MAP_TARGET_LEFT_PADDING
+ * bytes to the right of the target.
*/
- private CharBuffer content = null;
+ private static final long MAP_TARGET_LEFT_PADDING_BYTES = (long) (MAP_MAX_BYTES * 0.2);
+
+ /**
+ * Total length of character stream to replay minus the HTTP headers
+ * if present.
+ *
+ * If the backing file is larger than Integer.MAX_VALUE (i.e. 2gb),
+ * only the first Integer.MAX_VALUE characters are available through this API.
+ * We're overriding java.lang.CharSequence so that we can use
+ * java.util.regex directly on the data, and the CharSequence
+ * API uses int for the length and index.
+ */
+ protected int length;
+
+ /**
+ * Byte offset into the file where the memory mapped portion begins.
+ */
+ private long mapByteOffset;
+
+ // XXX do we need to keep the input stream around?
+ private FileInputStream backingFileIn = null;
+
+ private FileChannel backingFileChannel = null;
+
+ private long bytesPerChar;
+
+ private ByteBuffer mappedBuffer = null;
+
+ private CharsetDecoder decoder = null;
+
+ private ByteBuffer tempBuf = null;
/**
* File that has decoded content.
@@ -106,9 +117,15 @@ public class GenericReplayCharSequence implements ReplayCharSequence {
*/
private File decodedFile = null;
+ /*
+ * This portion of the CharSequence precedes what's in the backing file. In
+ * cases where we decodeToFile(), this is always empty, because we decode
+ * the entire input stream.
+ */
+ private CharBuffer prefixBuffer = null;
/**
- * Constructor for all in-memory operation.
+ * Constructor.
*
* @param buffer In-memory buffer of recordings prefix. We read from
* here first and will only go to the backing file if size
@@ -116,225 +133,257 @@ public class GenericReplayCharSequence implements ReplayCharSequence {
* @param size Total size of stream to replay in bytes. Used to find
* EOS. This is total length of content including HTTP headers if
* present.
- * @param responseBodyStart Where the response body starts in bytes.
- * Used to skip over the HTTP headers if present.
+ * @param contentReplayInputStream inputStream of content
+ * @param charsetName Encoding to use reading the passed prefix
+ * buffer and backing file. For now, should be java canonical name for the
+ * encoding. Must not be null.
* @param backingFilename Path to backing file with content in excess of
* whats in buffer.
- * @param encoding Encoding to use reading the passed prefix buffer and
- * backing file. For now, should be java canonical name for the
- * encoding. (If null is passed, we will default to
- * ByteReplayCharSequence).
- *
- * @throws IOException
- */
- public GenericReplayCharSequence(byte[] buffer, long size,
- long responseBodyStart, String encoding)
- throws IOException {
- super();
- this.content = decodeInMemory(buffer, size, responseBodyStart,
- encoding);
- }
-
- /**
- * Constructor for overflow-to-disk-file operation.
- *
- * @param contentReplayInputStream inputStream of content
- * @param backingFilename hint for name of temp file
- * @param characterEncoding Encoding to use reading the stream.
- * For now, should be java canonical name for the
- * encoding.
*
* @throws IOException
*/
public GenericReplayCharSequence(
- ReplayInputStream contentReplayInputStream,
- String backingFilename,
- String characterEncoding)
- throws IOException {
+ ReplayInputStream contentReplayInputStream, String backingFilename,
+ String charsetName) throws IOException {
super();
- this.content = decodeToFile(contentReplayInputStream,
- backingFilename, characterEncoding);
+ logger.info("new GenericReplayCharSequence() characterEncoding="
+ + charsetName + " backingFilename=" + backingFilename);
+
+ Charset charset = Charset.forName(charsetName);
+ if (charset.newEncoder().maxBytesPerChar() == 1.0) {
+ logger.info("charset=" + charsetName
+ + ": supports random access, using backing file directly");
+ this.bytesPerChar = 1;
+ this.backingFileIn = new FileInputStream(backingFilename);
+ this.decoder = charset.newDecoder();
+ this.prefixBuffer = this.decoder.decode(
+ ByteBuffer.wrap(contentReplayInputStream.getBuffer()));
+ } else {
+ logger.info("charset=" + charsetName
+ + ": may not support random access, decoding to separate file");
+
+ // decodes only up to Integer.MAX_VALUE characters
+ decodeToFile(contentReplayInputStream, backingFilename, charsetName);
+
+ this.bytesPerChar = 2;
+ this.backingFileIn = new FileInputStream(decodedFile);
+ this.decoder = Charset.forName(WRITE_ENCODING).newDecoder();
+ this.prefixBuffer = CharBuffer.wrap("");
+ }
+
+ this.tempBuf = ByteBuffer.wrap(new byte[(int) this.bytesPerChar]);
+ this.backingFileChannel = backingFileIn.getChannel();
+
+ // we only support the first Integer.MAX_VALUE characters
+ long wouldBeLength = prefixBuffer.limit() + backingFileChannel.size() / bytesPerChar;
+ if (wouldBeLength <= Integer.MAX_VALUE) {
+ this.length = (int) wouldBeLength;
+ } else {
+ logger.warning("input stream is longer than Integer.MAX_VALUE="
+ + NumberFormat.getInstance().format(Integer.MAX_VALUE)
+ + " characters -- only first "
+ + NumberFormat.getInstance().format(Integer.MAX_VALUE)
+ + " are accessible through this GenericReplayCharSequence");
+ this.length = Integer.MAX_VALUE;
+ }
+
+ this.mapByteOffset = 0;
+ updateMemoryMappedBuffer();
+ }
+
+ private void updateMemoryMappedBuffer() {
+ long fileLength = (long) this.length() - (long) prefixBuffer.limit(); // in characters
+ long mapSize = Math.min(fileLength * bytesPerChar - mapByteOffset, MAP_MAX_BYTES);
+ logger.fine("updateMemoryMappedBuffer: mapOffset="
+ + NumberFormat.getInstance().format(mapByteOffset)
+ + " mapSize=" + NumberFormat.getInstance().format(mapSize));
+ try {
+ System.gc();
+ System.runFinalization();
+ // TODO: Confirm the READ_ONLY works. I recall it not working.
+ // The buffers seem to always say that the buffer is writable.
+ mappedBuffer = backingFileChannel.map(
+ FileChannel.MapMode.READ_ONLY, mapByteOffset, mapSize)
+ .asReadOnlyBuffer();
+ } catch (IOException e) {
+ // TODO convert this to a runtime error?
+ DevUtils.logger.log(Level.SEVERE,
+ " backingFileChannel.map() mapByteOffset=" + mapByteOffset
+ + " mapSize=" + mapSize + "\n" + "decodedFile="
+ + decodedFile + " length=" + length + "\n"
+ + DevUtils.extraInfo(), e);
+ throw new RuntimeException(e);
+ }
}
/**
- * Decode passed buffer and backing file into a CharBuffer.
- *
- * This method writes a new file made of the decoded concatenation of
- * the in-memory prefix buffer and the backing file. Returns a
- * charSequence view onto this new file.
- *
- * @param buffer In-memory buffer of recordings prefix. We read from
- * here first and will only go to the backing file if size
- * requested is greater than buffer.length.
- * @param size Total size of stream to replay in bytes. Used to find
- * EOS. This is total length of content including HTTP headers if
- * present.
- * @param responseBodyStart Where the response body starts in bytes.
- * Used to skip over the HTTP headers if present.
- * @param backingFilename Path to backing file with content in excess of
- * whats in buffer.
- * @param encoding Encoding to use reading the passed prefix buffer and
- * backing file. For now, should be java canonical name for the
- * encoding. (If null is passed, we will default to
- * ByteReplayCharSequence).
- *
- * @return A CharBuffer view on decodings of the contents of passed
- * buffer.
+ * Converts the first Integer.MAX_VALUE characters from the
+ * file backingFilename from encoding encoding to
+ * encoding WRITE_ENCODING and saves as
+ * this.decodedFile, which is named backingFilename
+ * + "." + WRITE_ENCODING.
+ *
* @throws IOException
*/
- private CharBuffer decodeToFile(ReplayInputStream inStream,
- String backingFilename, String encoding)
- throws IOException {
+ private void decodeToFile(ReplayInputStream inStream,
+ String backingFilename, String encoding) throws IOException {
- CharBuffer charBuffer = null;
+ BufferedReader reader = new BufferedReader(new InputStreamReader(
+ inStream, encoding));
+
+ this.decodedFile = new File(backingFilename + "." + WRITE_ENCODING);
+
+ logger.info("decodeToFile: backingFilename=" + backingFilename
+ + " encoding=" + encoding + " decodedFile=" + decodedFile);
- BufferedReader reader = new BufferedReader(
- new InputStreamReader(inStream,encoding));
-
- File backingFile = new File(backingFilename);
- this.decodedFile = File.createTempFile(
- backingFile.getName(), WRITE_ENCODING, backingFile.getParentFile());
FileOutputStream fos;
- fos = new FileOutputStream(this.decodedFile);
-
- BufferedWriter writer = new BufferedWriter(
- new OutputStreamWriter(
- fos,
- WRITE_ENCODING));
+ try {
+ fos = new FileOutputStream(this.decodedFile);
+ } catch (FileNotFoundException e) {
+ // Windows workaround attempt
+ System.gc();
+ System.runFinalization();
+ this.decodedFile = new File(decodedFile.getAbsolutePath()+".win");
+ logger.info("Windows 'file with a user-mapped section open' "
+ + "workaround gc/finalization/name-extension performed.");
+ // try again
+ fos = new FileOutputStream(this.decodedFile);
+ }
+ BufferedWriter writer = new BufferedWriter(new OutputStreamWriter(fos,
+ WRITE_ENCODING));
int c;
- while((c = reader.read())>=0) {
+ long count = 0;
+ while ((c = reader.read()) >= 0 && count < Integer.MAX_VALUE) {
writer.write(c);
+ count++;
+ if (count % 100000000 == 0) {
+ logger.fine("wrote " + count + " characters so far...");
+ }
}
writer.close();
-
- charBuffer = getReadOnlyMemoryMappedBuffer(this.decodedFile).
- asCharBuffer();
- return charBuffer;
+ logger.info("decodeToFile: wrote " + count + " characters to "
+ + decodedFile);
}
/**
- * Decode passed buffer into a CharBuffer.
- *
- * This method decodes a memory buffer returning a memory buffer.
- *
- * @param buffer In-memory buffer of recordings prefix. We read from
- * here first and will only go to the backing file if size
- * requested is greater than buffer.length.
- * @param size Total size of stream to replay in bytes. Used to find
- * EOS. This is total length of content including HTTP headers if
- * present.
- * @param responseBodyStart Where the response body starts in bytes.
- * Used to skip over the HTTP headers if present.
- * @param encoding Encoding to use reading the passed prefix buffer and
- * backing file. For now, should be java canonical name for the
- * encoding. (If null is passed, we will default to
- * ByteReplayCharSequence).
- *
- * @return A CharBuffer view on decodings of the contents of passed
- * buffer.
+ * Get character at passed absolute position.
+ * @param index Index into content
+ * @return Character at offset index.
*/
- private CharBuffer decodeInMemory(byte[] buffer, long size,
- long responseBodyStart, String encoding)
- {
- ByteBuffer bb = ByteBuffer.wrap(buffer);
- // Move past the HTTP header if present.
- bb.position((int)responseBodyStart);
- // Set the end-of-buffer to be end-of-content.
- bb.limit((int)size);
- return (Charset.forName(encoding)).decode(bb).asReadOnlyBuffer();
- }
+ public char charAt(int index) {
+ if (index < 0 || index >= this.length()) {
+ throw new IndexOutOfBoundsException("index=" + index
+ + " - should be between 0 and length()=" + this.length());
+ }
- /**
- * Create read-only memory-mapped buffer onto passed file.
- *
- * @param file File to get memory-mapped buffer on.
- * @return Read-only memory-mapped ByteBuffer view on to passed file.
- * @throws IOException
- */
- private ByteBuffer getReadOnlyMemoryMappedBuffer(File file)
- throws IOException {
+ // is it in the buffer
+ if (index < prefixBuffer.limit()) {
+ return prefixBuffer.get(index);
+ }
- ByteBuffer bb = null;
- FileInputStream in = null;
- FileChannel c = null;
- assert file.exists(): "No file " + file.getAbsolutePath();
+ // otherwise we gotta get it from disk via memory map
+ long fileIndex = (long) index - (long) prefixBuffer.limit();
+ long fileLength = (long) this.length() - (long) prefixBuffer.limit(); // in characters
+ if (fileIndex * bytesPerChar < mapByteOffset
+ || fileIndex * bytesPerChar - mapByteOffset >= mappedBuffer.limit()) {
+ // fault
+ /*
+ * mapByteOffset is bounded by 0 and file size +/- size of the map,
+ * and starts as close to fileIndex -
+ * MAP_TARGET_LEFT_PADDING_BYTES as it can while also not
+ * being smaller than it needs to be.
+ */
+ mapByteOffset = Math.max(0, fileIndex * bytesPerChar - MAP_TARGET_LEFT_PADDING_BYTES);
+ mapByteOffset = Math.min(mapByteOffset, fileLength * bytesPerChar - MAP_MAX_BYTES);
+ updateMemoryMappedBuffer();
+ }
+
+ // CharsetDecoder always decodes up to the end of the ByteBuffer, so we
+ // create a new ByteBuffer with only the bytes we're interested in.
+ mappedBuffer.position((int) (fileIndex * bytesPerChar - mapByteOffset));
+ mappedBuffer.get(tempBuf.array());
+ tempBuf.position(0); // decoder starts at this position
try {
- in = new FileInputStream(file);
- c = in.getChannel();
- // TODO: Confirm the READ_ONLY works. I recall it not working.
- // The buffers seem to always say that the buffer is writeable.
- bb = c.map(FileChannel.MapMode.READ_ONLY, 0, c.size()).
- asReadOnlyBuffer();
+ CharBuffer cbuf = decoder.decode(tempBuf);
+ return cbuf.get();
+ } catch (CharacterCodingException e) {
+ logger.warning("unable to get character at index=" + index + " (fileIndex=" + fileIndex + "): " + e);
+ // U+FFFD REPLACEMENT CHARACTER --
+ // "used to replace an incoming character whose value is unknown or unrepresentable in Unicode"
+ return (char) 0xfffd;
}
-
- finally {
- if (c != null && c.isOpen()) {
- c.close();
- }
- if (in != null) {
- in.close();
- }
- }
-
- return bb;
- }
-
- private void deleteFile(File fileToDelete) {
- deleteFile(fileToDelete, null);
- }
-
- private void deleteFile(File fileToDelete, final Exception e) {
- if (e != null) {
- // Log why the delete to help with debug of java.io.FileNotFoundException:
- // ....tt53http.ris.UTF-16BE.
- logger.severe("Deleting " + fileToDelete + " because of "
- + e.toString());
- }
- if (fileToDelete != null && fileToDelete.exists()) {
- FileUtils.deleteSoonerOrLater(fileToDelete);
- }
- }
-
- public void close()
- {
- this.content = null;
- deleteFile(this.decodedFile);
- // clear decodedFile -- so that double-close (as in
- // finalize()) won't delete a later instance with same name
- // see bug [ 1218961 ] "failed get of replay" in ExtractorHTML... usu: UTF-16BE
- this.decodedFile = null;
- }
-
- protected void finalize() throws Throwable
- {
- super.finalize();
- // Maybe TODO: eliminate close here, requiring explicit close instead
- close();
- }
-
- public int length()
- {
- return this.content.limit();
- }
-
- public char charAt(int index)
- {
- return this.content.get(index);
}
public CharSequence subSequence(int start, int end) {
return new CharSubSequence(this, start, end);
}
-
- public String toString() {
- StringBuffer sb = new StringBuffer(length());
- // could use StringBuffer.append(CharSequence) if willing to do 1.5 & up
- for (int i = 0;isize
- * requested is greater than buffer.length.
- * @param size Total size of stream to replay in bytes. Used to find
- * EOS. This is total length of content including HTTP headers if
- * present.
- * @param responseBodyStart Where the response body starts in bytes.
- * Used to skip over the HTTP headers if present.
- * @param backingFilename Path to backing file with content in excess of
- * whats in buffer.
- *
- * @throws IOException
- */
- public Latin1ByteReplayCharSequence(byte[] buffer, long size,
- long responseBodyStart, String backingFilename)
- throws IOException {
-
- this.length = (int)(size - responseBodyStart);
- this.absoluteLength = (int)size;
- this.prefixBuffer = buffer;
- this.contentOffset = (int)responseBodyStart;
-
- // If amount to read is > than what is in our prefix buffer, then
- // open the backing file.
- if (size > buffer.length) {
- this.backingFilename = backingFilename;
- this.raFile = new RandomAccessFile(backingFilename, "r");
- this.wraparoundBuffer = new byte[this.prefixBuffer.length];
- this.wrapOrigin = this.prefixBuffer.length;
- this.wrapOffset = 0;
- loadBuffer();
- }
- }
-
- /**
- * @return Length of characters in stream to replay. Starts counting
- * at the HTTP header/body boundary.
- */
- public int length() {
- return this.length;
- }
-
- /**
- * Get character at passed absolute position.
- *
- * Called by {@link #charAt(int)} which has a relative index into the
- * content, one that doesn't account for HTTP header if present.
- *
- * @param index Index into content adjusted to accomodate initial offset
- * to get us past the HTTP header if present (i.e.
- * {@link #contentOffset}).
- *
- * @return Characater at offset index.
- */
- public char charAt(int index) {
- int c = -1;
- // Add to index start-of-content offset to get us over HTTP header
- // if present.
- index += this.contentOffset;
- if (index < this.prefixBuffer.length) {
- // If index is into our prefix buffer.
- c = this.prefixBuffer[index];
- } else if (index >= this.wrapOrigin &&
- (index - this.wrapOrigin) < this.wraparoundBuffer.length) {
- // If index is into our buffer window on underlying backing file.
- c = this.wraparoundBuffer[
- ((index - this.wrapOrigin) + this.wrapOffset) %
- this.wraparoundBuffer.length];
- } else {
- // Index is outside of both prefix buffer and our buffer window
- // onto the underlying backing file. Fix the buffer window
- // location.
- c = faultCharAt(index);
- }
- // Stream is treated as single byte. Make sure characters returned
- // are not negative.
- return (char)(c & 0xff);
- }
-
- /**
- * Get a character that's outside the current buffers.
- *
- * will cause the wraparoundBuffer to be changed to
- * cover a region including the index
- *
- * if index is higher than the highest index in the
- * wraparound buffer, buffer is moved forward such
- * that requested char is last item in buffer
- *
- * if index is lower than lowest index in the
- * wraparound buffer, buffet is reset centered around
- * index
- *
- * @param index Index of character to fetch.
- * @return A character that's outside the current buffers
- */
- private int faultCharAt(int index) {
- if(Thread.interrupted()) {
- throw new RuntimeException("thread interrupted");
- }
- if(index >= this.wrapOrigin + this.wraparoundBuffer.length) {
- // Moving forward
- while (index >= this.wrapOrigin + this.wraparoundBuffer.length)
- {
- // TODO optimize this
- advanceBuffer();
- }
- return charAt(index - this.contentOffset);
- }
- // Moving backward
- recenterBuffer(index);
- return charAt(index - this.contentOffset);
- }
-
- /**
- * Move the buffer window on backing file back centering current access
- * position in middle of window.
- *
- * @param index Index of character to access.
- */
- private void recenterBuffer(int index) {
- if (logger.isLoggable(Level.FINE)) {
- logger.fine("Recentering around " + index + " in " +
- this.backingFilename);
- }
- this.wrapOrigin = index - (this.wraparoundBuffer.length / 2);
- if(this.wrapOrigin < this.prefixBuffer.length) {
- this.wrapOrigin = this.prefixBuffer.length;
- }
- this.wrapOffset = 0;
- loadBuffer();
- }
-
- /**
- * Load from backing file into the wrapper buffer.
- */
- private void loadBuffer()
- {
- long len = -1;
- try {
- len = this.raFile.length();
- this.raFile.seek(this.wrapOrigin - this.prefixBuffer.length);
- this.raFile.readFully(this.wraparoundBuffer, 0,
- Math.min(this.wraparoundBuffer.length,
- this.absoluteLength - this.wrapOrigin));
- }
-
- catch (IOException e) {
- // TODO convert this to a runtime error?
- DevUtils.logger.log (
- Level.SEVERE,
- "raFile.seek(" +
- (this.wrapOrigin - this.prefixBuffer.length) +
- ")\n" +
- "raFile.readFully(wraparoundBuffer,0," +
- (Math.min(this.wraparoundBuffer.length,
- this.length - this.wrapOrigin )) +
- ")\n"+
- "raFile.length()" + len + "\n" +
- DevUtils.extraInfo(),
- e);
- throw new RuntimeException(e);
- }
- }
-
- /**
- * Roll the wraparound buffer forward one position
- */
- private void advanceBuffer() {
- try {
- this.wraparoundBuffer[this.wrapOffset] =
- (byte)this.raFile.read();
- this.wrapOffset++;
- this.wrapOffset %= this.wraparoundBuffer.length;
- this.wrapOrigin++;
- } catch (IOException e) {
- DevUtils.logger.log(Level.SEVERE, "advanceBuffer()" +
- DevUtils.extraInfo(), e);
- throw new RuntimeException(e);
- }
- }
-
- public CharSequence subSequence(int start, int end) {
- return new CharSubSequence(this, start, end);
- }
-
- /**
- * Cleanup resources.
- *
- * @exception IOException Failed close of random access file.
- */
- public void close() throws IOException
- {
- this.prefixBuffer = null;
- if (this.raFile != null) {
- this.raFile.close();
- this.raFile = null;
- }
- }
-
- /* (non-Javadoc)
- * @see java.lang.Object#finalize()
- */
- protected void finalize() throws Throwable
- {
- super.finalize();
- close();
- }
-
- /**
- * Convenience method for getting a substring.
- * @deprecated please use subSequence() and then toString() directly
- */
- public String substring(int offset, int len) {
- return subSequence(offset, offset+len).toString();
- }
-
- /* (non-Javadoc)
- * @see java.lang.Object#toString()
- */
- public String toString() {
- StringBuilder sb = new StringBuilder(this.length());
- sb.append(this);
- return sb.toString();
- }
-}
\ No newline at end of file
diff --git a/commons/src/main/java/org/archive/io/RecordingOutputStream.java b/commons/src/main/java/org/archive/io/RecordingOutputStream.java
index 01f64a05..edc1c502 100644
--- a/commons/src/main/java/org/archive/io/RecordingOutputStream.java
+++ b/commons/src/main/java/org/archive/io/RecordingOutputStream.java
@@ -1,27 +1,22 @@
-/* ReplayableOutputStream
+/*
+ * This file is part of the Heritrix web crawler (crawler.archive.org).
*
- * $Id$
+ * Licensed to the Internet Archive (IA) by one or more individual
+ * contributors.
*
- * Created on Sep 23, 2003
+ * The IA licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
*
- * Copyright (C) 2003 Internet Archive.
+ * http://www.apache.org/licenses/LICENSE-2.0
*
- * This file is part of the Heritrix web crawler (crawler.archive.org).
- *
- * Heritrix is free software; you can redistribute it and/or modify
- * it under the terms of the GNU Lesser Public License as published by
- * the Free Software Foundation; either version 2.1 of the License, or
- * any later version.
- *
- * Heritrix is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU Lesser Public License for more details.
- *
- * You should have received a copy of the GNU Lesser Public License
- * along with Heritrix; if not, write to the Free Software
- * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
*/
+
package org.archive.io;
import it.unimi.dsi.fastutil.io.FastBufferedOutputStream;
@@ -513,9 +508,7 @@ public class RecordingOutputStream extends OutputStream {
throws IOException {
return getReplayCharSequence(characterEncoding, this.contentBeginMark);
}
-
- private static final String canonicalLatin1 = Charset.forName("iso8859-1").name();
-
+
/**
* @param characterEncoding Encoding of recorded stream.
* @return A ReplayCharSequence Will return null if an IOException. Call
@@ -524,39 +517,27 @@ public class RecordingOutputStream extends OutputStream {
*/
public ReplayCharSequence getReplayCharSequence(String characterEncoding,
long startOffset) throws IOException {
- if (characterEncoding == null) {
+ if (characterEncoding == null)
characterEncoding = Charset.defaultCharset().name();
- }
- // TODO: handled transfer-encoding: chunked content-bodies properly
- if (canonicalLatin1.equals(Charset.forName(characterEncoding).name())) {
- return new Latin1ByteReplayCharSequence(
+ logger.fine("this.size=" + this.size + " this.buffer.length=" + this.buffer.length);
+ if (this.size <= this.buffer.length) {
+ logger.fine("using InMemoryReplayCharSequence");
+ // raw data is all in memory; do in memory
+ return new InMemoryReplayCharSequence(
this.buffer,
this.size,
startOffset,
- this.backingFilename);
- } else {
- // multibyte
- if(this.size <= this.buffer.length) {
- // raw data is all in memory; do in memory
- return new GenericReplayCharSequence(
- this.buffer,
- this.size,
- startOffset,
- characterEncoding);
-
- } else {
- // raw data overflows to disk; use temp file
- ReplayInputStream ris = getReplayInputStream(startOffset);
- ReplayCharSequence rcs = new GenericReplayCharSequence(
- ris,
- this.backingFilename,
- characterEncoding);
- ris.close();
- return rcs;
- }
-
+ characterEncoding);
+ }
+ else {
+ logger.fine("using GenericReplayCharSequence");
+ // raw data overflows to disk; use temp file
+ ReplayInputStream ris = getReplayInputStream(startOffset);
+ return new GenericReplayCharSequence(
+ ris,
+ this.backingFilename,
+ characterEncoding);
}
-
}
public long getResponseContentLength() {
@@ -624,3 +605,4 @@ public class RecordingOutputStream extends OutputStream {
return maxLength - position;
}
}
+
diff --git a/commons/src/main/java/org/archive/io/ReplayInputStream.java b/commons/src/main/java/org/archive/io/ReplayInputStream.java
index a590ccdb..04656d4b 100644
--- a/commons/src/main/java/org/archive/io/ReplayInputStream.java
+++ b/commons/src/main/java/org/archive/io/ReplayInputStream.java
@@ -1,27 +1,22 @@
-/* ReplayInputStream
+/*
+ * This file is part of the Heritrix web crawler (crawler.archive.org).
*
- * $Id$
+ * Licensed to the Internet Archive (IA) by one or more individual
+ * contributors.
*
- * Created on Sep 24, 2003
+ * The IA licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
*
- * Copyright (C) 2003 Internet Archive.
+ * http://www.apache.org/licenses/LICENSE-2.0
*
- * This file is part of the Heritrix web crawler (crawler.archive.org).
- *
- * Heritrix is free software; you can redistribute it and/or modify
- * it under the terms of the GNU Lesser Public License as published by
- * the Free Software Foundation; either version 2.1 of the License, or
- * any later version.
- *
- * Heritrix is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU Lesser Public License for more details.
- *
- * You should have received a copy of the GNU Lesser Public License
- * along with Heritrix; if not, write to the Free Software
- * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
*/
+
package org.archive.io;
import java.io.File;
@@ -279,4 +274,9 @@ public class ReplayInputStream extends SeekInputStream
public long position() throws IOException {
return position;
}
+
+ // package private
+ byte[] getBuffer() {
+ return buffer;
+ }
}
diff --git a/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java b/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java
index fb7d229f..11b84f56 100644
--- a/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java
+++ b/commons/src/test/java/org/archive/io/ReplayCharSequenceTest.java
@@ -20,7 +20,9 @@
package org.archive.io;
import java.io.IOException;
+import java.text.NumberFormat;
import java.util.Date;
+import java.util.Random;
import java.util.logging.Logger;
import org.archive.util.FileUtils;
@@ -210,14 +212,24 @@ public class ReplayCharSequenceTest extends TmpDirTestCase
}
public void testReplayCharSequenceByteToStringOverflow() throws IOException {
- String fileContent = "Some file content. ";
+ String fileContent = "Some file content. "; // ascii
byte [] buffer = fileContent.getBytes();
RecordingOutputStream ros = writeTestStream(
buffer,1,
- "testReplayCharSequenceByteToString.txt",1);
+ "testReplayCharSequenceByteToStringOverflow.txt",1);
String expectedContent = fileContent+fileContent;
- ReplayCharSequence rcs = ros.getReplayCharSequence();
- String result = rcs.toString();
+
+ // The string is ascii which is a subset of both these encodings. Use
+ // both encodings because they exercise different code paths. UTF-8 is
+ // decoded to UTF-16 while windows-1252 is memory mapped directly. See
+ // GenericReplayCharSequence
+ ReplayCharSequence rcsUtf8 = ros.getReplayCharSequence("UTF-8");
+ ReplayCharSequence rcs1252 = ros.getReplayCharSequence("windows-1252");
+
+ String result = rcsUtf8.toString();
+ assertEquals("Strings don't match", expectedContent, result);
+
+ result = rcs1252.toString();
assertEquals("Strings don't match", expectedContent, result);
}
@@ -244,6 +256,67 @@ public class ReplayCharSequenceTest extends TmpDirTestCase
}
}
+ public void xestHugeReplayCharSequence() throws IOException {
+ String fileContent = "01234567890123456789";
+ String characterEncoding = "ascii";
+ byte[] buffer = fileContent.getBytes(characterEncoding);
+
+ long reps = (long) Integer.MAX_VALUE / (long) buffer.length + 1000000l;
+
+ logger.info("writing " + (reps * buffer.length)
+ + " bytes to testHugeReplayCharSequence.txt");
+ RecordingOutputStream ros = writeTestStream(buffer, 0,
+ "testHugeReplayCharSequence.txt", reps);
+ ReplayCharSequence rcs = ros.getReplayCharSequence(characterEncoding);
+
+ if (reps * fileContent.length() > (long) Integer.MAX_VALUE) {
+ assertTrue("ReplayCharSequence has wrong length (length()="
+ + rcs.length() + ") (should be " + Integer.MAX_VALUE + ")",
+ rcs.length() == Integer.MAX_VALUE);
+ } else {
+ assertEquals("ReplayCharSequence has wrong length (length()="
+ + rcs.length() + ") (should be "
+ + (reps * fileContent.length()) + ")", (long) rcs.length(),
+ reps * (long) fileContent.length());
+ }
+
+ // boundary cases or something
+ for (int index : new int[] { 0, rcs.length() / 4, rcs.length() / 2,
+ rcs.length() - 1, rcs.length() / 4 }) {
+ // logger.info("testing char at index=" +
+ // NumberFormat.getInstance().format(index));
+ assertEquals("Characters don't match (index="
+ + NumberFormat.getInstance().format(index) + ")",
+ fileContent.charAt(index % fileContent.length()), rcs
+ .charAt(index));
+ }
+
+ // check that out of bounds indices throw exception
+ for (int n : new int[] { -1, Integer.MIN_VALUE, rcs.length() + 1 }) {
+ try {
+ String message = "rcs.charAt(" + n + ")=" + rcs.charAt(n)
+ + " ?!? -- expected IndexOutOfBoundsException";
+ logger.severe(message);
+ fail(message);
+ } catch (IndexOutOfBoundsException e) {
+ logger.info("got expected exception: " + e);
+ }
+ }
+
+ // check some characters at random spots & kinda stress test the
+ // system's memory mapping facility
+ Random rand = new Random(0); // seed so we get the same ones each time
+ for (int i = 0; i < 5000; i++) {
+ int index = rand.nextInt(rcs.length());
+ // logger.info(i + ". testing char at index=" +
+ // NumberFormat.getInstance().format(index));
+ assertEquals("Characters don't match (index="
+ + NumberFormat.getInstance().format(index) + ")",
+ fileContent.charAt(index % fileContent.length()), rcs
+ .charAt(index));
+ }
+ }
+
/**
* Accessing characters test.
*
@@ -290,13 +363,14 @@ public class ReplayCharSequenceTest extends TmpDirTestCase
* @throws IOException
*/
private RecordingOutputStream writeTestStream(byte[] content,
- int memReps, String baseName, int fileReps) throws IOException {
+ int memReps, String baseName, long fileReps) throws IOException {
String backingFilename = FileUtils.maybeRelative(getTmpDir(),baseName).getAbsolutePath();
RecordingOutputStream ros = new RecordingOutputStream(
content.length * memReps,
backingFilename);
ros.open();
- for(int i = 0; i < (memReps+fileReps); i++) {
+ ros.markContentBegin();
+ for(long i = 0; i < (memReps+fileReps); i++) {
// fill buffer (repeat MULTIPLIER times) and
// overflow to disk (also MULTIPLIER times)
ros.write(content);