[performance optimizations] - option to compact chunks in persistent hash map via value deserialization instead of chunks's concatenation. Usage of the option in indices decreases disk usage in 10-50% depending on index.

- do not store key -> 0 mapping when removing nonexistent key
This commit is contained in:
Maxim.Mossienko
2016-07-14 19:28:33 +02:00
parent 660c89716c
commit 5bb0d30a08
3 changed files with 83 additions and 30 deletions
@@ -99,10 +99,12 @@ public final class MapIndexStorage<Key, Value> implements IndexStorage<Key, Valu
private void initMapAndCache() throws IOException {
final ValueContainerMap<Key, Value> map;
PersistentHashMapValueStorage.CreationTimeOptions.EXCEPTIONAL_IO_CANCELLATION.set(ourProgressManagerCheckCancelledIOCanceller);
PersistentHashMapValueStorage.CreationTimeOptions.COMPACT_CHUNKS_WITH_VALUE_DESERIALIZATION.set(Boolean.TRUE);
try {
map = new ValueContainerMap<Key, Value>(getStorageFile(), myKeyDescriptor, myDataExternalizer, myKeyIsUniqueForIndexedFile);
} finally {
PersistentHashMapValueStorage.CreationTimeOptions.EXCEPTIONAL_IO_CANCELLATION.set(null);
PersistentHashMapValueStorage.CreationTimeOptions.COMPACT_CHUNKS_WITH_VALUE_DESERIALIZATION.set(null);
}
myCache = new SLRUCache<Key, ChangeTrackingValueContainer<Value>>(myCacheSize, (int)(Math.ceil(myCacheSize * 0.25)) /* 25% from the main cache size*/) {
@Override
@@ -526,17 +526,33 @@ public class PersistentHashMap<Key, Value> extends PersistentEnumeratorDelegate<
myEnumerator.unlockStorage();
}
PersistentHashMapValueStorage.ReadResult readResult = myValueStorage.readBytes(valueOffset);
final PersistentHashMapValueStorage.ReadResult readResult = myValueStorage.readBytes(valueOffset);
DataInputStream input = new DataInputStream(new UnsyncByteArrayInputStream(readResult.buffer));
final Value valueRead;
try {
valueRead = myValueExternalizer.read(input);
}
finally {
input.close();
}
if (myValueStorage.performChunksCompaction(readResult.chunksCount, readResult.buffer.length)) {
long newValueOffset = myValueStorage.compactChunks(new ValueDataAppender() {
@Override
public void append(DataOutput out) throws IOException {
myValueExternalizer.save(out, valueRead);
}
}, readResult);
if (readResult.offset != valueOffset) { // compacted several chunks produced during append
myEnumerator.lockStorage();
try {
myEnumerator.markDirty(true);
if (myDirectlyStoreLongFileOffsetMode) {
((PersistentBTreeEnumerator<Key>)myEnumerator).putNonnegativeValue(key, readResult.offset);
((PersistentBTreeEnumerator<Key>)myEnumerator).putNonnegativeValue(key, newValueOffset);
} else {
updateValueId(id, readResult.offset, valueOffset, key, 0);
updateValueId(id, newValueOffset, valueOffset, key, 0);
}
myLiveAndGarbageKeysCounter++;
myReadCompactionGarbageSize += readResult.buffer.length;
@@ -544,14 +560,7 @@ public class PersistentHashMap<Key, Value> extends PersistentEnumeratorDelegate<
myEnumerator.unlockStorage();
}
}
final DataInputStream input = new DataInputStream(new UnsyncByteArrayInputStream(readResult.buffer));
try {
return myValueExternalizer.read(input);
}
finally {
input.close();
}
return valueRead;
}
public final boolean containsMapping(Key key) throws IOException {
@@ -596,10 +605,12 @@ public class PersistentHashMap<Key, Value> extends PersistentEnumeratorDelegate<
if (myDirectlyStoreLongFileOffsetMode) {
assert !myIntMapping; // removal isn't supported
record = ((PersistentBTreeEnumerator<Key>)myEnumerator).getNonnegativeValue(key);
((PersistentBTreeEnumerator<Key>)myEnumerator).putNonnegativeValue(key, NULL_ADDR);
if (record != NULL_ADDR) {
((PersistentBTreeEnumerator<Key>)myEnumerator).putNonnegativeValue(key, NULL_ADDR);
}
} else {
final int id = tryEnumerate(key);
if (id == PersistentEnumerator.NULL_ID) {
if (id == PersistentEnumeratorBase.NULL_ID) {
return;
}
assert !myIntMapping; // removal isn't supported
@@ -38,6 +38,7 @@ public class PersistentHashMapValueStorage {
private final File myFile;
private final String myPath;
private final boolean myReadOnly;
private final boolean myCompactChunksWithValueDeserialization;
private final ExceptionalIOCancellationCallback myExceptionalIOCancellationCallback;
private boolean myCompactionMode = false;
@@ -47,6 +48,7 @@ public class PersistentHashMapValueStorage {
public static class CreationTimeOptions {
public static final ThreadLocal<ExceptionalIOCancellationCallback> EXCEPTIONAL_IO_CANCELLATION = new ThreadLocal<ExceptionalIOCancellationCallback>();
public static final ThreadLocal<Boolean> READONLY = new ThreadLocal<Boolean>();
public static final ThreadLocal<Boolean> COMPACT_CHUNKS_WITH_VALUE_DESERIALIZATION = new ThreadLocal<Boolean>();
}
public interface ExceptionalIOCancellationCallback {
@@ -102,6 +104,7 @@ public class PersistentHashMapValueStorage {
public PersistentHashMapValueStorage(String path) throws IOException {
myExceptionalIOCancellationCallback = CreationTimeOptions.EXCEPTIONAL_IO_CANCELLATION.get();
myReadOnly = CreationTimeOptions.READONLY.get() == Boolean.TRUE;
myCompactChunksWithValueDeserialization = CreationTimeOptions.COMPACT_CHUNKS_WITH_VALUE_DESERIALIZATION.get() == Boolean.TRUE;
myPath = path;
myFile = new File(path);
@@ -143,14 +146,14 @@ public class PersistentHashMapValueStorage {
}
public long appendBytes(byte[] data, int offset, int dataLength, long prevChunkAddress) throws IOException {
assert !myCompactionMode && !myReadOnly;
assert allowedToCompactChunks();
long result = mySize; // volatile read
final FileAccessorCache.Handle<DataOutputStream> appender = myCompressedAppendableFile != null? null : ourAppendersCache.get(myPath);
DataOutputStream dataOutputStream;
try {
if (myCompressedAppendableFile != null) {
BufferExposingByteArrayOutputStream stream = new BufferExposingByteArrayOutputStream();
BufferExposingByteArrayOutputStream stream = new BufferExposingByteArrayOutputStream(dataLength + 15);
DataOutputStream testStream = new DataOutputStream(stream);
saveData(data, offset, dataLength, prevChunkAddress, result, testStream);
myCompressedAppendableFile.append(stream.getInternalBuffer(), stream.size());
@@ -339,17 +342,21 @@ public class PersistentHashMapValueStorage {
}
public static class ReadResult {
public final long offset;
public final byte[] buffer;
public final int chunksCount;
public ReadResult(long offset, byte[] buffer) {
this.offset = offset;
public ReadResult(byte[] buffer, int chunksCount) {
this.buffer = buffer;
this.chunksCount = chunksCount;
}
}
private long myChunksRemovalTime;
private long myChunksReadingTime;
private int myChunks;
private long myChunksOriginalBytes;
private long myChunksBytesAfterRemoval;
private int myLastReportedChunksCount;
/**
* Reads bytes pointed by tailChunkAddress into result passed, returns new address if linked list compactification have been performed
@@ -428,22 +435,55 @@ public class PersistentHashMapValueStorage {
}
}
if (chunkCount > 1 && !myCompactionMode && !myReadOnly) {
if (chunkCount > 1) {
checkCancellation();
long endCompactionTime = ourDumpChunkRemovalTime ? System.nanoTime() : 0;
long diff = endCompactionTime - startedTime;
myChunksRemovalTime += diff;
myChunksReadingTime += (ourDumpChunkRemovalTime ? System.nanoTime() : 0) - startedTime;
myChunks += chunkCount;
if (ourDumpChunkRemovalTime && chunkCount > 2) {
System.out.println("Removed " + chunkCount + " chunks for " + (diff / 1000000) + "ms, bytes: " + result.length + ", total: " +
(myChunksRemovalTime / 1000000) + "ms for " + myChunks + " chunks in " + myPath);
}
long l = appendBytes(new ByteSequence(result), 0);
return new ReadResult(l, result);
myChunksOriginalBytes += result.length;
}
return new ReadResult(tailChunkAddress, result);
return new ReadResult(result, chunkCount);
}
private boolean allowedToCompactChunks() {
return !myCompactionMode && !myReadOnly;
}
boolean performChunksCompaction(int chunksCount, int chunksBytesSize) {
return chunksCount > 1 && allowedToCompactChunks();
}
long compactChunks(PersistentHashMap.ValueDataAppender appender, ReadResult result) throws IOException {
checkCancellation();
long startedTime = ourDumpChunkRemovalTime ? System.nanoTime() : 0;
long newValueOffset;
if (myCompactChunksWithValueDeserialization) {
final BufferExposingByteArrayOutputStream stream = new BufferExposingByteArrayOutputStream(result.buffer.length);
DataOutputStream testStream = new DataOutputStream(stream);
appender.append(testStream);
newValueOffset = appendBytes(stream.getInternalBuffer(), 0, stream.size(), 0);
myChunksBytesAfterRemoval += stream.size();
} else {
newValueOffset = appendBytes(new ByteSequence(result.buffer), 0);
myChunksBytesAfterRemoval += result.buffer.length;
}
if (ourDumpChunkRemovalTime) {
myChunksRemovalTime += System.nanoTime() - startedTime;
if (myChunks - myLastReportedChunksCount > 1000) {
myLastReportedChunksCount = myChunks;
System.out.println(myChunks + " chunks were read " + (myChunksReadingTime / 1000000) +
"ms, bytes: " + myChunksOriginalBytes +
(myChunksOriginalBytes != myChunksBytesAfterRemoval ? "->" + myChunksBytesAfterRemoval : "") +
" compaction:" + (myChunksRemovalTime / 1000000) + "ms in " + myPath);
}
}
return newValueOffset;
}
private static final boolean ourDumpChunkRemovalTime = SystemProperties.getBooleanProperty("idea.phmp.dump.chunk.removal.time", false);