mirror of https://github.com/apache/cassandra
add support for multiple mmapped index segments, and add mmap_index_only option
patch by jbellis; tested by Brandon Williams for CASSANDRA-669 git-svn-id: https://svn.apache.org/repos/asf/incubator/cassandra/trunk@896742 13f79535-47bb-0310-9956-ffa450edef68
This commit is contained in:
parent
d6bdce610d
commit
33acfcd657
|
|
@ -229,9 +229,11 @@
|
||||||
~ Access mode. mmapped i/o is substantially faster, but only practical on
|
~ Access mode. mmapped i/o is substantially faster, but only practical on
|
||||||
~ a 64bit machine (which notably does not include EC2 "small" instances)
|
~ a 64bit machine (which notably does not include EC2 "small" instances)
|
||||||
~ or relatively small datasets. "auto", the safe choice, will enable
|
~ or relatively small datasets. "auto", the safe choice, will enable
|
||||||
~ mmapping on a 64bit JVM. Other values are "mmap" and "standard" if you
|
~ mmapping on a 64bit JVM. Other values are "mmap", "mmap_index_only"
|
||||||
~ need to force those modes. (The buffer size settings that follow only
|
~ (which may allow you to get part of the benefits of mmap on a 32bit
|
||||||
~ apply to standard, non-mmapped i/o.)
|
~ machine by mmapping only index files) and "standard".
|
||||||
|
~ (The buffer size settings that follow only apply to standard,
|
||||||
|
~ non-mmapped i/o.)
|
||||||
-->
|
-->
|
||||||
<DiskAccessMode>auto</DiskAccessMode>
|
<DiskAccessMode>auto</DiskAccessMode>
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -53,6 +53,7 @@ public class DatabaseDescriptor
|
||||||
public static enum DiskAccessMode {
|
public static enum DiskAccessMode {
|
||||||
auto,
|
auto,
|
||||||
mmap,
|
mmap,
|
||||||
|
mmap_index_only,
|
||||||
standard,
|
standard,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -128,6 +129,7 @@ public class DatabaseDescriptor
|
||||||
private static int commitLogSyncPeriodMS_;
|
private static int commitLogSyncPeriodMS_;
|
||||||
|
|
||||||
private static DiskAccessMode diskAccessMode_;
|
private static DiskAccessMode diskAccessMode_;
|
||||||
|
private static DiskAccessMode indexAccessMode_;
|
||||||
|
|
||||||
private static boolean snapshotBeforeCompaction_;
|
private static boolean snapshotBeforeCompaction_;
|
||||||
private static boolean autoBootstrap_ = false;
|
private static boolean autoBootstrap_ = false;
|
||||||
|
|
@ -198,13 +200,23 @@ public class DatabaseDescriptor
|
||||||
}
|
}
|
||||||
catch (IllegalArgumentException e)
|
catch (IllegalArgumentException e)
|
||||||
{
|
{
|
||||||
throw new ConfigurationException("DiskAccessMode must be either 'auto', or 'mmap', or 'standard'");
|
throw new ConfigurationException("DiskAccessMode must be either 'auto', 'mmap', 'mmap_index_only', or 'standard'");
|
||||||
}
|
}
|
||||||
if (diskAccessMode_ == DiskAccessMode.auto)
|
if (diskAccessMode_ == DiskAccessMode.auto)
|
||||||
{
|
{
|
||||||
diskAccessMode_ = System.getProperty("os.arch").contains("64") ? DiskAccessMode.mmap : DiskAccessMode.standard;
|
diskAccessMode_ = System.getProperty("os.arch").contains("64") ? DiskAccessMode.mmap : DiskAccessMode.standard;
|
||||||
|
indexAccessMode_ = diskAccessMode_;
|
||||||
logger_.info("Auto DiskAccessMode determined to be " + diskAccessMode_);
|
logger_.info("Auto DiskAccessMode determined to be " + diskAccessMode_);
|
||||||
}
|
}
|
||||||
|
else if (diskAccessMode_ == DiskAccessMode.mmap_index_only)
|
||||||
|
{
|
||||||
|
diskAccessMode_ = DiskAccessMode.standard;
|
||||||
|
indexAccessMode_ = DiskAccessMode.mmap;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
indexAccessMode_ = diskAccessMode_;
|
||||||
|
}
|
||||||
|
|
||||||
/* Hashing strategy */
|
/* Hashing strategy */
|
||||||
String partitionerClassName = xmlUtils.getNodeValue("/Storage/Partitioner");
|
String partitionerClassName = xmlUtils.getNodeValue("/Storage/Partitioner");
|
||||||
|
|
@ -983,6 +995,11 @@ public class DatabaseDescriptor
|
||||||
return diskAccessMode_;
|
return diskAccessMode_;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static DiskAccessMode getIndexAccessMode()
|
||||||
|
{
|
||||||
|
return indexAccessMode_;
|
||||||
|
}
|
||||||
|
|
||||||
public static double getFlushDataBufferSizeInMB()
|
public static double getFlushDataBufferSizeInMB()
|
||||||
{
|
{
|
||||||
return flushDataBufferSizeInMB_;
|
return flushDataBufferSizeInMB_;
|
||||||
|
|
|
||||||
|
|
@ -90,7 +90,7 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
};
|
};
|
||||||
new Thread(runnable, "SSTABLE-DELETER").start();
|
new Thread(runnable, "SSTABLE-DELETER").start();
|
||||||
}};
|
}};
|
||||||
private static final int BUFFER_SIZE = Integer.MAX_VALUE;
|
private static final long BUFFER_SIZE = Integer.MAX_VALUE;
|
||||||
|
|
||||||
public static int indexInterval()
|
public static int indexInterval()
|
||||||
{
|
{
|
||||||
|
|
@ -185,8 +185,9 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
}
|
}
|
||||||
|
|
||||||
FileDeletingReference phantomReference;
|
FileDeletingReference phantomReference;
|
||||||
private final MappedByteBuffer indexBuffer;
|
// jvm can only map up to 2GB at a time, so we split index/data into segments of that size when using mmap i/o
|
||||||
private final MappedByteBuffer[] buffers; // jvm can only map up to 2GB at a time
|
private final MappedByteBuffer[] indexBuffers;
|
||||||
|
private final MappedByteBuffer[] buffers;
|
||||||
|
|
||||||
|
|
||||||
public static ConcurrentLinkedHashMap<DecoratedKey, PositionSize> createKeyCache(int size)
|
public static ConcurrentLinkedHashMap<DecoratedKey, PositionSize> createKeyCache(int size)
|
||||||
|
|
@ -204,7 +205,25 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
throws IOException
|
throws IOException
|
||||||
{
|
{
|
||||||
super(filename, partitioner);
|
super(filename, partitioner);
|
||||||
indexBuffer = mmap(indexFilename());
|
|
||||||
|
if (DatabaseDescriptor.getIndexAccessMode() == DatabaseDescriptor.DiskAccessMode.mmap)
|
||||||
|
{
|
||||||
|
long indexLength = new File(indexFilename()).length();
|
||||||
|
int bufferCount = 1 + (int) (indexLength / BUFFER_SIZE);
|
||||||
|
indexBuffers = new MappedByteBuffer[bufferCount];
|
||||||
|
long remaining = indexLength;
|
||||||
|
for (int i = 0; i < bufferCount; i++)
|
||||||
|
{
|
||||||
|
indexBuffers[i] = mmap(indexFilename(), i * BUFFER_SIZE, (int) Math.min(remaining, BUFFER_SIZE));
|
||||||
|
remaining -= BUFFER_SIZE;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
assert DatabaseDescriptor.getIndexAccessMode() == DatabaseDescriptor.DiskAccessMode.standard;
|
||||||
|
indexBuffers = null;
|
||||||
|
}
|
||||||
|
|
||||||
if (DatabaseDescriptor.getDiskAccessMode() == DatabaseDescriptor.DiskAccessMode.mmap)
|
if (DatabaseDescriptor.getDiskAccessMode() == DatabaseDescriptor.DiskAccessMode.mmap)
|
||||||
{
|
{
|
||||||
int bufferCount = 1 + (int) (new File(path).length() / BUFFER_SIZE);
|
int bufferCount = 1 + (int) (new File(path).length() / BUFFER_SIZE);
|
||||||
|
|
@ -231,7 +250,7 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
this.keyCache = keyCache;
|
this.keyCache = keyCache;
|
||||||
}
|
}
|
||||||
|
|
||||||
private static MappedByteBuffer mmap(String filename, int start, int size) throws IOException
|
private static MappedByteBuffer mmap(String filename, long start, int size) throws IOException
|
||||||
{
|
{
|
||||||
RandomAccessFile raf;
|
RandomAccessFile raf;
|
||||||
try
|
try
|
||||||
|
|
@ -243,12 +262,6 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
throw new IOError(e);
|
throw new IOError(e);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (size < 0)
|
|
||||||
{
|
|
||||||
if (raf.length() > Integer.MAX_VALUE)
|
|
||||||
throw new UnsupportedOperationException("File " + filename + " is too large to map in its entirety");
|
|
||||||
size = (int) raf.length();
|
|
||||||
}
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
return raf.getChannel().map(FileChannel.MapMode.READ_ONLY, start, size);
|
return raf.getChannel().map(FileChannel.MapMode.READ_ONLY, start, size);
|
||||||
|
|
@ -259,11 +272,6 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private static MappedByteBuffer mmap(String filename) throws IOException
|
|
||||||
{
|
|
||||||
return mmap(filename, 0, -1);
|
|
||||||
}
|
|
||||||
|
|
||||||
private SSTableReader(String filename, IPartitioner partitioner) throws IOException
|
private SSTableReader(String filename, IPartitioner partitioner) throws IOException
|
||||||
{
|
{
|
||||||
this(filename, partitioner, null, null, null, null);
|
this(filename, partitioner, null, null, null, null);
|
||||||
|
|
@ -383,42 +391,59 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
return info;
|
return info;
|
||||||
}
|
}
|
||||||
|
|
||||||
FileDataInput input = new MappedFileDataInput(indexBuffer, indexFilename());
|
long p = sampledPosition.position;
|
||||||
input.seek(sampledPosition.position);
|
FileDataInput input;
|
||||||
int i = 0;
|
if (indexBuffers == null)
|
||||||
do
|
|
||||||
{
|
{
|
||||||
DecoratedKey indexDecoratedKey;
|
input = new BufferedRandomAccessFile(indexFilename(), "r");
|
||||||
try
|
input.seek(p);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
input = new MappedFileDataInput(indexBuffers[bufferIndex(p)], indexFilename(), (int)(p % BUFFER_SIZE));
|
||||||
|
}
|
||||||
|
try
|
||||||
|
{
|
||||||
|
int i = 0;
|
||||||
|
do
|
||||||
{
|
{
|
||||||
indexDecoratedKey = partitioner.convertFromDiskFormat(input.readUTF());
|
DecoratedKey indexDecoratedKey;
|
||||||
}
|
try
|
||||||
catch (EOFException e)
|
|
||||||
{
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
long position = input.readLong();
|
|
||||||
int v = indexDecoratedKey.compareTo(decoratedKey);
|
|
||||||
if (v == 0)
|
|
||||||
{
|
|
||||||
PositionSize info;
|
|
||||||
if (input.getFilePointer() < input.length())
|
|
||||||
{
|
{
|
||||||
int utflen = input.readUnsignedShort();
|
indexDecoratedKey = partitioner.convertFromDiskFormat(input.readUTF());
|
||||||
input.skipBytes(utflen);
|
|
||||||
info = new PositionSize(position, input.readLong() - position);
|
|
||||||
}
|
}
|
||||||
else
|
catch (EOFException e)
|
||||||
{
|
{
|
||||||
info = new PositionSize(position, length() - position);
|
return null;
|
||||||
}
|
}
|
||||||
if (keyCache != null)
|
long position = input.readLong();
|
||||||
keyCache.put(decoratedKey, info);
|
int v = indexDecoratedKey.compareTo(decoratedKey);
|
||||||
return info;
|
if (v == 0)
|
||||||
}
|
{
|
||||||
if (v > 0)
|
PositionSize info;
|
||||||
return null;
|
if (input.getFilePointer() < input.length())
|
||||||
} while (++i < INDEX_INTERVAL);
|
{
|
||||||
|
int utflen = input.readUnsignedShort();
|
||||||
|
if (utflen != input.skipBytes(utflen))
|
||||||
|
throw new EOFException();
|
||||||
|
info = new PositionSize(position, input.readLong() - position);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
info = new PositionSize(position, length() - position);
|
||||||
|
}
|
||||||
|
if (keyCache != null)
|
||||||
|
keyCache.put(decoratedKey, info);
|
||||||
|
return info;
|
||||||
|
}
|
||||||
|
if (v > 0)
|
||||||
|
return null;
|
||||||
|
} while (++i < INDEX_INTERVAL);
|
||||||
|
}
|
||||||
|
finally
|
||||||
|
{
|
||||||
|
input.close();
|
||||||
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -431,25 +456,9 @@ public class SSTableReader extends SSTable implements Comparable<SSTableReader>
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
// by default, we plan to start scanning at the nearest bsearched index entry
|
// can't use a MappedFileDataInput here, since we might cross a segment boundary while scanning
|
||||||
long start = sampledPosition.position;
|
|
||||||
if (spannedIndexDataPositions != null)
|
|
||||||
{
|
|
||||||
// check if the index entry spans a mmap segment boundary
|
|
||||||
PositionSize info = spannedIndexDataPositions.get(sampledPosition);
|
|
||||||
if (info != null)
|
|
||||||
{
|
|
||||||
// if the key matches the index entry we don't have to scan the index after all
|
|
||||||
if (sampledPosition.key.compareTo(decoratedKey) == 0)
|
|
||||||
return info.position;
|
|
||||||
// otherwise, start scanning at the next entry (which won't span a boundary;
|
|
||||||
// if it did it would have been in the index sample and we would have started with that instead)
|
|
||||||
start = info.position + sampledPosition.key.serializedSize() + (Long.SIZE / 8);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
BufferedRandomAccessFile input = new BufferedRandomAccessFile(indexFilename(path), "r");
|
BufferedRandomAccessFile input = new BufferedRandomAccessFile(indexFilename(path), "r");
|
||||||
input.seek(start);
|
input.seek(sampledPosition.position);
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
while (true)
|
while (true)
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue