diff --git a/src/java/org/apache/cassandra/net/ManyToOneConcurrentLinkedQueue.java b/src/java/org/apache/cassandra/concurrent/ManyToOneConcurrentLinkedQueue.java similarity index 97% rename from src/java/org/apache/cassandra/net/ManyToOneConcurrentLinkedQueue.java rename to src/java/org/apache/cassandra/concurrent/ManyToOneConcurrentLinkedQueue.java index 4c73bdc9cd..8615e99c22 100644 --- a/src/java/org/apache/cassandra/net/ManyToOneConcurrentLinkedQueue.java +++ b/src/java/org/apache/cassandra/concurrent/ManyToOneConcurrentLinkedQueue.java @@ -15,7 +15,7 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.net; +package org.apache.cassandra.concurrent; import java.util.Collection; import java.util.Iterator; @@ -37,12 +37,12 @@ import java.util.function.Consumer; * In addition to that, provides a {@link #relaxedPeekLastAndOffer(Object)} method that we use to avoid a CAS when * putting message handlers onto the wait queue. */ -class ManyToOneConcurrentLinkedQueue extends ManyToOneConcurrentLinkedQueueHead implements Queue +public class ManyToOneConcurrentLinkedQueue extends ManyToOneConcurrentLinkedQueueHead implements Queue { @SuppressWarnings("unused") // pad two cache lines after the head to prevent false sharing protected long p31, p32, p33, p34, p35, p36, p37, p38, p39, p40, p41, p42, p43, p44, p45; - ManyToOneConcurrentLinkedQueue() + public ManyToOneConcurrentLinkedQueue() { head = tail = new Node<>(null); } @@ -63,7 +63,7 @@ class ManyToOneConcurrentLinkedQueue extends ManyToOneConcurrentLinkedQueueHe * - {@code false} result indicates that the queue MIGHT BE non-empty - the value of {@code head} might * not yet have been made externally visible by the consumer thread. */ - boolean relaxedIsEmpty() + public boolean relaxedIsEmpty() { return null == head.next; } @@ -156,7 +156,7 @@ class ManyToOneConcurrentLinkedQueue extends ManyToOneConcurrentLinkedQueueHe * Yields no performance benefit over invoking {@link #poll()} manually - there just isn't * anything to meaningfully amortise on the consumer side of this queue. */ - void drain(Consumer consumer) + public void drain(Consumer consumer) { E item; while ((item = poll()) != null) @@ -181,7 +181,7 @@ class ManyToOneConcurrentLinkedQueue extends ManyToOneConcurrentLinkedQueueHe * * @return previously last tail item in the queue, potentially stale */ - E relaxedPeekLastAndOffer(E e) + public E relaxedPeekLastAndOffer(E e) { return internalOffer(e); } diff --git a/src/java/org/apache/cassandra/config/Config.java b/src/java/org/apache/cassandra/config/Config.java index 63e81481be..704dda8aa9 100644 --- a/src/java/org/apache/cassandra/config/Config.java +++ b/src/java/org/apache/cassandra/config/Config.java @@ -397,6 +397,10 @@ public class Config @Replaces(oldName = "commitlog_total_space_in_mb", converter = Converters.MEBIBYTES_DATA_STORAGE_INT, deprecated = true) public DataStorageSpec.IntMebibytesBound commitlog_total_space; public CommitLogSync commitlog_sync; + + // Accord Journal + public String accord_journal_directory; + @Replaces(oldName = "commitlog_sync_group_window_in_ms", converter = Converters.MILLIS_DURATION_DOUBLE, deprecated = true) public DurationSpec.IntMillisecondsBound commitlog_sync_group_window = new DurationSpec.IntMillisecondsBound("0ms"); @Replaces(oldName = "commitlog_sync_period_in_ms", converter = Converters.MILLIS_DURATION_INT, deprecated = true) diff --git a/src/java/org/apache/cassandra/config/DatabaseDescriptor.java b/src/java/org/apache/cassandra/config/DatabaseDescriptor.java index 0c15ba1227..d251d3ef63 100644 --- a/src/java/org/apache/cassandra/config/DatabaseDescriptor.java +++ b/src/java/org/apache/cassandra/config/DatabaseDescriptor.java @@ -386,6 +386,14 @@ public class DatabaseDescriptor clientInitialization(failIfDaemonOrTool, Config::new); } + // For simulator tests + public static void clientWithDaemonConfig() + { + clientInitialization(true, DatabaseDescriptor::loadConfig); + applyAll(); + AuthConfig.applyAuth(); + } + /** * Initializes this class as a client, which means that just an empty configuration will * be used. @@ -714,6 +722,11 @@ public class DatabaseDescriptor if (commitLogWriteDiskAccessMode != conf.commitlog_disk_access_mode) logger.info("commitlog_disk_access_mode resolved to: {}", commitLogWriteDiskAccessMode); + if (conf.accord_journal_directory == null) + { + conf.accord_journal_directory = storagedirFor("accord_journal"); + } + if (conf.hints_directory == null) { conf.hints_directory = storagedirFor("hints"); @@ -789,6 +802,8 @@ public class DatabaseDescriptor throw new ConfigurationException("local_system_data_file_directory must not be the same as any data_file_directories", false); if (datadir.equals(conf.commitlog_directory)) throw new ConfigurationException("commitlog_directory must not be the same as any data_file_directories", false); + if (datadir.equals(conf.accord_journal_directory)) + throw new ConfigurationException("accord_journal_directory must not be the same as any data_file_directories", false); if (datadir.equals(conf.hints_directory)) throw new ConfigurationException("hints_directory must not be the same as any data_file_directories", false); if (datadir.equals(conf.saved_caches_directory)) @@ -804,6 +819,8 @@ public class DatabaseDescriptor { if (conf.local_system_data_file_directory.equals(conf.commitlog_directory)) throw new ConfigurationException("local_system_data_file_directory must not be the same as the commitlog_directory", false); + if (conf.local_system_data_file_directory.equals(conf.accord_journal_directory)) + throw new ConfigurationException("local_system_data_file_directory must not be the same as the accord_journal_directory", false); if (conf.local_system_data_file_directory.equals(conf.saved_caches_directory)) throw new ConfigurationException("local_system_data_file_directory must not be the same as the saved_caches_directory", false); if (conf.local_system_data_file_directory.equals(conf.hints_directory)) @@ -816,10 +833,18 @@ public class DatabaseDescriptor FBUtilities.prettyPrintMemory(freeBytes)); } - if (conf.commitlog_directory.equals(conf.saved_caches_directory)) - throw new ConfigurationException("saved_caches_directory must not be the same as the commitlog_directory", false); + if (conf.commitlog_directory.equals(conf.accord_journal_directory)) + throw new ConfigurationException("accord_journal_directory must not be the same as the commitlog_directory", false); if (conf.commitlog_directory.equals(conf.hints_directory)) throw new ConfigurationException("hints_directory must not be the same as the commitlog_directory", false); + if (conf.commitlog_directory.equals(conf.saved_caches_directory)) + throw new ConfigurationException("saved_caches_directory must not be the same as the commitlog_directory", false); + + if (conf.accord_journal_directory.equals(conf.hints_directory)) + throw new ConfigurationException("hints_directory must not be the same as the accord_journal_directory", false); + if (conf.accord_journal_directory.equals(conf.saved_caches_directory)) + throw new ConfigurationException("saved_caches_directory must not be the same as the accord_journal_directory", false); + if (conf.hints_directory.equals(conf.saved_caches_directory)) throw new ConfigurationException("saved_caches_directory must not be the same as the hints_directory", false); @@ -2115,6 +2140,10 @@ public class DatabaseDescriptor throw new ConfigurationException("commitlog_directory must be specified", false); FileUtils.createDirectory(conf.commitlog_directory); + if (conf.accord_journal_directory == null) + throw new ConfigurationException("accord_journal_directory must be specified", false); + FileUtils.createDirectory(conf.accord_journal_directory); + if (conf.hints_directory == null) throw new ConfigurationException("hints_directory must be specified", false); FileUtils.createDirectory(conf.hints_directory); @@ -3018,6 +3047,16 @@ public class DatabaseDescriptor conf.commitlog_compression = compressor; } + public static String getAccordJournalDirectory() + { + return conf.accord_journal_directory; + } + + public static void setAccordJournalDirectory(String path) + { + conf.accord_journal_directory = path; + } + public static Config.FlushCompression getFlushCompression() { return conf.flush_compression; diff --git a/src/java/org/apache/cassandra/io/util/TrackedDataOutputPlus.java b/src/java/org/apache/cassandra/io/util/TrackedDataOutputPlus.java new file mode 100644 index 0000000000..34363e03b8 --- /dev/null +++ b/src/java/org/apache/cassandra/io/util/TrackedDataOutputPlus.java @@ -0,0 +1,181 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.io.util; + +import java.io.IOException; +import java.nio.ByteBuffer; + +import org.apache.cassandra.utils.vint.VIntCoding; + +public class TrackedDataOutputPlus implements DataOutputPlus +{ + private final DataOutputPlus out; + private int position = 0; + + private TrackedDataOutputPlus(DataOutputPlus out) + { + this.out = out; + } + + public static TrackedDataOutputPlus wrap(DataOutputPlus out) + { + return new TrackedDataOutputPlus(out); + } + + @Override + public void write(int b) throws IOException + { + out.write(b); + position += 1; + } + + @Override + public void write(byte[] b) throws IOException + { + out.write(b); + position += b.length; + } + + @Override + public void write(byte[] b, int off, int len) throws IOException + { + out.write(b, off, len); + position += len; + } + + @Override + public void writeBoolean(boolean v) throws IOException + { + out.writeBoolean(v); + position += 1; + } + + @Override + public void writeByte(int v) throws IOException + { + out.writeByte(v); + position += 1; + } + + @Override + public void writeShort(int v) throws IOException + { + out.writeShort(v); + position += 2; + } + + @Override + public void writeChar(int v) throws IOException + { + out.writeChar(v); + position += 2; + } + + @Override + public void writeInt(int v) throws IOException + { + out.writeInt(v); + position += 4; + } + + @Override + public void writeLong(long v) throws IOException + { + out.writeLong(v); + position += 8; + } + + @Override + public void writeFloat(float v) throws IOException + { + out.writeFloat(v); + position += 4; + } + + @Override + public void writeDouble(double v) throws IOException + { + out.writeDouble(v); + position += 8; + } + + @Override + public void writeBytes(String s) throws IOException + { + out.writeBytes(s); + position += s.length(); + } + + @Override + public void writeChars(String s) throws IOException + { + out.writeChars(s); + position += s.length() * 2; + } + + @Override + public void writeUTF(String s) throws IOException + { + UnbufferedDataOutputStreamPlus.writeUTF(s, this); + } + + @Override + public void write(ByteBuffer buffer) throws IOException + { + out.write(buffer); + position += buffer.remaining(); + } + + @Override + public void write(ReadableMemory memory, long offset, long length) throws IOException + { + out.write(memory, offset, length); + position += length; + } + + @Override + public void writeVInt(long i) throws IOException + { + VIntCoding.writeVInt(i, this); + } + + @Override + public void writeUnsignedVInt(long i) throws IOException + { + VIntCoding.writeUnsignedVInt(i, this); + } + + @Override + public void writeMostSignificantBytes(long register, int bytes) throws IOException + { + out.writeMostSignificantBytes(register, bytes); + position += bytes; + } + + @Override + public long position() + { + return position; + } + + @Override + public boolean hasPosition() + { + return true; + } +} diff --git a/src/java/org/apache/cassandra/journal/ActiveSegment.java b/src/java/org/apache/cassandra/journal/ActiveSegment.java new file mode 100644 index 0000000000..0f4d0dc09c --- /dev/null +++ b/src/java/org/apache/cassandra/journal/ActiveSegment.java @@ -0,0 +1,517 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.nio.MappedByteBuffer; +import java.nio.channels.FileChannel; +import java.nio.file.StandardOpenOption; +import java.util.*; +import java.util.concurrent.Executor; +import java.util.concurrent.atomic.AtomicInteger; +import java.util.concurrent.locks.LockSupport; + +import com.codahale.metrics.Timer; +import org.apache.cassandra.concurrent.ExecutionFailure; +import org.apache.cassandra.concurrent.ManyToOneConcurrentLinkedQueue; +import org.apache.cassandra.io.util.*; +import org.apache.cassandra.utils.*; +import org.apache.cassandra.utils.concurrent.OpOrder; +import org.apache.cassandra.utils.concurrent.Ref; +import org.apache.cassandra.utils.concurrent.WaitQueue; + +final class ActiveSegment extends Segment +{ + final FileChannel channel; + + // OpOrder used to order appends wrt flush + private final OpOrder appendOrder = new OpOrder(); + + // position in the buffer we are allocating from + private final AtomicInteger allocatePosition = new AtomicInteger(0); + + /* + * Everything before this offset has been written and flushed. + */ + private volatile int lastFlushedOffset = 0; + + /* + * End position of the buffer; initially set to its capacity and + * updated to point to the last written position as the segment is being closed + * no need to be volatile as writes are protected by appendOrder barrier. + */ + private int endOfBuffer; + + // a signal that writers can wait on to be notified of a completed flush in BATCH and GROUP FlushMode + private final WaitQueue flushComplete = WaitQueue.newWaitQueue(); + + private final Ref> selfRef; + + final InMemoryIndex index; + + private ActiveSegment( + Descriptor descriptor, Params params, SyncedOffsets syncedOffsets, InMemoryIndex index, Metadata metadata, KeySupport keySupport) + { + super(descriptor, syncedOffsets, metadata, keySupport); + this.index = index; + try + { + channel = FileChannel.open(file.toPath(), StandardOpenOption.WRITE, StandardOpenOption.READ, StandardOpenOption.CREATE); + buffer = channel.map(FileChannel.MapMode.READ_WRITE, 0, params.segmentSize()); + endOfBuffer = buffer.capacity(); + selfRef = new Ref<>(this, new Tidier(descriptor, channel, buffer, syncedOffsets)); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, file, e); + } + } + + @SuppressWarnings("resource") + static ActiveSegment create(Descriptor descriptor, Params params, KeySupport keySupport) + { + SyncedOffsets syncedOffsets = SyncedOffsets.active(descriptor, true); + InMemoryIndex index = InMemoryIndex.create(keySupport); + Metadata metadata = Metadata.create(); + return new ActiveSegment<>(descriptor, params, syncedOffsets, index, metadata, keySupport); + } + + @Override + InMemoryIndex index() + { + return index; + } + + /** + * Read the entry and specified offset into the entry holder. + * Expects the caller to acquire the ref to the segment and the record to exist. + */ + @Override + boolean read(int offset, EntrySerializer.EntryHolder into) + { + ByteBuffer duplicate = (ByteBuffer) buffer.duplicate().position(offset).limit(buffer.capacity()); + try + { + EntrySerializer.read(into, keySupport, duplicate, descriptor.userVersion); + } + catch (IOException e) + { + throw new JournalReadError(descriptor, file, e); + } + return true; + } + + /** + * Stop writing to this file, flush and close it. Does nothing if the file is already closed. + */ + @Override + public synchronized void close() + { + close(true); + } + + /** + * @return true if the closed segment was definitely empty, false otherwise + */ + private synchronized boolean close(boolean persistComponents) + { + boolean isEmpty = discardUnusedTail(); + if (!isEmpty) + { + flush(); + if (persistComponents) persistComponents(); + } + release(); + return isEmpty; + } + + /** + * Close and discard a pre-allocated, available segment, that's never been exposed + */ + void closeAndDiscard() + { + boolean isEmpty = close(false); + if (!isEmpty) throw new IllegalStateException(); + discard(); + } + + void closeAndIfEmptyDiscard() + { + boolean isEmpty = close(true); + if (isEmpty) discard(); + } + + void persistComponents() + { + index.persist(descriptor); + metadata.persist(descriptor); + SyncUtil.trySyncDir(descriptor.directory); + } + + private void discard() + { + selfRef.ensureReleased(); + + descriptor.fileFor(Component.DATA).deleteIfExists(); + descriptor.fileFor(Component.INDEX).deleteIfExists(); + descriptor.fileFor(Component.METADATA).deleteIfExists(); + descriptor.fileFor(Component.SYNCED_OFFSETS).deleteIfExists(); + } + + void release() + { + selfRef.release(); + } + + @Override + public Ref> tryRef() + { + return selfRef.tryRef(); + } + + @Override + public Ref> ref() + { + return selfRef.ref(); + } + + private static final class Tidier implements Tidy + { + private final Descriptor descriptor; + private final FileChannel channel; + private final ByteBuffer buffer; + private final SyncedOffsets syncedOffsets; + + Tidier(Descriptor descriptor, FileChannel channel, ByteBuffer buffer, SyncedOffsets syncedOffsets) + { + this.descriptor = descriptor; + this.channel = channel; + this.buffer = buffer; + this.syncedOffsets = syncedOffsets; + } + + @Override + public void tidy() + { + FileUtils.clean(buffer); + try + { + channel.close(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, Component.DATA, e); + } + syncedOffsets.close(); + } + + @Override + public String name() + { + return descriptor.toString(); + } + } + + /* + * Flush logic; closing and component flushing + */ + + /** + * Possibly force a disk flush for this segment file. + * TODO FIXME: calls from outside Flusher + callbacks + * @return last synced offset + */ + synchronized int flush() + { + int allocatePosition = this.allocatePosition.get(); + if (lastFlushedOffset >= allocatePosition) + return lastFlushedOffset; + + waitForModifications(); + flushInternal(); + lastFlushedOffset = allocatePosition; + int syncedOffset = Math.min(allocatePosition, endOfBuffer); + syncedOffsets.mark(syncedOffset); + flushComplete.signalAll(); + return syncedOffset; + } + + private void waitForFlush(int position) + { + while (lastFlushedOffset < position) + { + WaitQueue.Signal signal = flushComplete.register(); + if (lastFlushedOffset < position) + signal.awaitThrowUncheckedOnInterrupt(); + else + signal.cancel(); + } + } + + /** + * Wait for any appends or discardUnusedTail() operations started before this method was called + */ + private void waitForModifications() + { + // issue a barrier and wait for it + appendOrder.awaitNewBarrier(); + } + + private void flushInternal() + { + try + { + SyncUtil.force((MappedByteBuffer) buffer); + } + catch (Exception e) // MappedByteBuffer.force() does not declare IOException but can actually throw it + { + throw new JournalWriteError(descriptor, file, e); + } + } + + boolean isFullyFlushed(int syncedOffset) + { + return syncedOffset >= endOfBuffer; + } + + /** + * Ensures no more of this segment is writeable, by allocating any unused section at the end + * and marking it discarded void discartUnusedTail() + * + * @return true if the segment was empty, false otherwise + */ + boolean discardUnusedTail() + { + try (OpOrder.Group ignored = appendOrder.start()) + { + while (true) + { + int prev = allocatePosition.get(); + int next = endOfBuffer + 1; + + if (prev >= next) + { + // already stopped allocating, might also be closed + assert buffer == null || prev == buffer.capacity() + 1; + return false; + } + + if (allocatePosition.compareAndSet(prev, next)) + { + // stopped allocating now; can only succeed once, no further allocation or discardUnusedTail can succeed + endOfBuffer = prev; + assert buffer != null && next == buffer.capacity() + 1; + return prev == 0; + } + } + } + } + + /* + * Entry/bytes allocation logic + */ + + @SuppressWarnings({ "resource", "RedundantSuppression" }) // op group will be closed by Allocation#write() + Allocation allocate(int entrySize, Set hosts) + { + int totalSize = totalEntrySize(hosts, entrySize); + OpOrder.Group opGroup = appendOrder.start(); + try + { + int position = allocateBytes(totalSize); + if (position < 0) + { + opGroup.close(); + return null; + } + return new Allocation(opGroup, (ByteBuffer) buffer.duplicate().position(position).limit(position + totalSize)); + } + catch (Throwable t) + { + opGroup.close(); + throw t; + } + } + + private int totalEntrySize(Set hosts, int recordSize) + { + return EntrySerializer.fixedEntrySize(keySupport, descriptor.userVersion) + + EntrySerializer.variableEntrySize(hosts.size(), recordSize); + } + + // allocate bytes in the segment, or return -1 if not enough space + private int allocateBytes(int size) + { + while (true) + { + int prev = allocatePosition.get(); + int next = prev + size; + if (next >= endOfBuffer) + return -1; + if (allocatePosition.compareAndSet(prev, next)) + { + assert buffer != null; + return prev; + } + LockSupport.parkNanos(1); // ConstantBackoffCAS Algorithm from https://arxiv.org/pdf/1305.5800.pdf + } + } + + final class Allocation + { + private final OpOrder.Group appendOp; + private final ByteBuffer buffer; + private final int position; + + Allocation(OpOrder.Group appendOp, ByteBuffer buffer) + { + this.appendOp = appendOp; + this.buffer = buffer; + this.position = buffer.position(); + } + + void write(K id, ByteBuffer record, Set hosts) + { + try (BufferedDataOutputStreamPlus out = new DataOutputBufferFixed(buffer)) + { + EntrySerializer.write(id, record, hosts, keySupport, out, descriptor.userVersion); + index.update(id, position); + metadata.update(hosts); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, file, e); + } + finally + { + appendOp.close(); + } + } + + void asyncWrite(K id, ByteBuffer record, Set hosts, Executor executor, AsyncWriteCallback callback) + { + try (BufferedDataOutputStreamPlus out = new DataOutputBufferFixed(buffer)) + { + int entrySize = totalEntrySize(hosts, record.remaining()); + EntrySerializer.write(id, record, hosts, keySupport, out, descriptor.userVersion); + index.update(id, position); + metadata.update(hosts); + writeCallbacksExternal.offer(new QueuedWriteCallback(position + entrySize, executor, callback)); + } + catch (Throwable t) + { + executor.execute(() -> callback.onFailure(t)); + } + finally + { + appendOp.close(); + } + } + + void awaitFlush(Timer waitingOnFlush) + { + try (Timer.Context ignored = waitingOnFlush.time()) + { + waitForFlush(position); + } + } + } + + // (external) MPSC queue for async write (flush) callbacks, to be executed in *write position order* + private final ManyToOneConcurrentLinkedQueue writeCallbacksExternal = + new ManyToOneConcurrentLinkedQueue<>(); + // (internal) single writer / single reader list of callbacks used to drain the callbacks into for sorting + private final ArrayList writeCallbacksInternal = + new ArrayList<>(); + + static final class QueuedWriteCallback implements Comparable + { + final long recordLimit; + final Executor executor; + final AsyncWriteCallback callback; + + QueuedWriteCallback(long recordLimit, Executor executor, AsyncWriteCallback callback) + { + this.recordLimit = recordLimit; + this.executor = executor; + this.callback = callback; + } + + @Override + public int compareTo(QueuedWriteCallback other) + { + // sort more recent callbacks first to simplify callback execution order later + return -Long.compare(this.recordLimit, other.recordLimit); + } + + void scheduleOnSuccess() + { + try + { + executor.execute(callback); + } + catch (Throwable t) + { + ExecutionFailure.handle(t); + } + } + + void scheduleOnFailure(Throwable error) + { + try + { + executor.execute(() -> callback.onFailure(error)); + } + catch (Throwable t) + { + ExecutionFailure.handle(t); + } + } + } + + void scheduleOnSuccessCallbacks(long syncedOffset) + { + // sort and execute callbacks in write position order, up until the furtherst synced offset + writeCallbacksExternal.drain(writeCallbacksInternal::add); + writeCallbacksInternal.sort(null); + + for (int i = writeCallbacksInternal.size() - 1; i >= 0; i--) + { + QueuedWriteCallback callback = writeCallbacksInternal.get(i); + if (callback.recordLimit > syncedOffset) + break; + callback.scheduleOnSuccess(); + writeCallbacksInternal.remove(i); + } + } + + void scheduleOnFailureCallbacks(Throwable t) + { + writeCallbacksExternal.drain(writeCallbacksInternal::add); + writeCallbacksInternal.sort(null); + + for (int i = writeCallbacksInternal.size() - 1; i >= 0; i--) + { + QueuedWriteCallback callback = writeCallbacksInternal.get(i); + callback.scheduleOnFailure(t); + } + + writeCallbacksInternal.clear(); + } +} diff --git a/src/java/org/apache/cassandra/journal/AsyncWriteCallback.java b/src/java/org/apache/cassandra/journal/AsyncWriteCallback.java new file mode 100644 index 0000000000..161b972b14 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/AsyncWriteCallback.java @@ -0,0 +1,23 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +public interface AsyncWriteCallback extends Runnable +{ + void onFailure(Throwable error); +} diff --git a/src/java/org/apache/cassandra/journal/Component.java b/src/java/org/apache/cassandra/journal/Component.java new file mode 100644 index 0000000000..c9de451adb --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Component.java @@ -0,0 +1,43 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +enum Component +{ + DATA ("data"), + INDEX ("indx"), + METADATA ("meta"), + SYNCED_OFFSETS ("sync"); + //OFFSET_MAP (".offs"), + //INVLALIDATIONS (".invl"); + + final String extension; + + Component(String extension) + { + this.extension = extension; + } + + /** + * @return if this component for the provided descrtiptor exists on disk + */ + boolean existsFor(Descriptor descriptor) + { + return descriptor.fileFor(this).exists(); + } +} diff --git a/src/java/org/apache/cassandra/journal/Descriptor.java b/src/java/org/apache/cassandra/journal/Descriptor.java new file mode 100644 index 0000000000..ea7ce7a1c2 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Descriptor.java @@ -0,0 +1,210 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.util.Arrays; +import java.util.List; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +import org.apache.cassandra.io.util.File; + +import static java.lang.String.format; +import static java.util.stream.Collectors.toList; + +/** + * Timestamp and version encoded in the file name, e.g. + * log-1637159888484-2-1-1.data + * log-1637159888484-2-1-1.indx + * log-1637159888484-2-1-1.meta + * log-1637159888484-2-1-1.sync + */ +final class Descriptor implements Comparable +{ + private static final String SEPARATOR = "-"; + private static final String PREFIX = "log" + SEPARATOR; + private static final String TMP_SUFFIX = "tmp"; + + private static final Pattern DATA_FILE_PATTERN = + Pattern.compile( PREFIX + "(\\d+)" // timestamp + + SEPARATOR + "(\\d+)" // generation + + SEPARATOR + "(\\d+)" // journal version + + SEPARATOR + "(\\d+)" // user version + + "\\." + Component.DATA.extension); + + private static final Pattern TMP_FILE_PATTERN = + Pattern.compile( PREFIX + "\\d+" // timestamp + + SEPARATOR + "\\d+" // generation + + SEPARATOR + "\\d+" // journal version + + SEPARATOR + "\\d+" // user version + + "\\." + "[a-z]+" // component extension + + "\\." + TMP_SUFFIX); + + + /* + * NOTE: If and when another journal version is introduced, have implementations + * expose the version used via yaml. This way operators can force previous journal + * version on upgrade, temporarily, to allow easier downgrades if something goes wrong. + */ + static final int JOURNAL_VERSION_1 = 1; + static final int CURRENT_JOURNAL_VERSION = JOURNAL_VERSION_1; + + final File directory; + final long timestamp; + final int generation; + + /** + * Serialization version for journal components; bumped as journal + * implementation evolves over time. + */ + final int journalVersion; + + /** + * Serialization version for user content - specifically journal keys + * and journal values; bumped when user logic evolves. + */ + final int userVersion; + + Descriptor(File directory, long timestamp, int generation, int journalVersion, int userVersion) + { + this.directory = directory; + this.timestamp = timestamp; + this.generation = generation; + this.journalVersion = journalVersion; + this.userVersion = userVersion; + } + + static Descriptor create(File directory, long timestamp, int userVersion) + { + return new Descriptor(directory, timestamp, 1, CURRENT_JOURNAL_VERSION, userVersion); + } + + static Descriptor fromName(File directory, String name) + { + Matcher matcher = DATA_FILE_PATTERN.matcher(name); + if (!matcher.matches()) + throw new IllegalArgumentException("Provided filename " + new File(directory, name) + " is not valid for a data segment file"); + + long timestamp = Long.parseLong(matcher.group(1)); + int generation = Integer.parseInt(matcher.group(2)); + int journalVersion = Integer.parseInt(matcher.group(3)); + int userVersion = Integer.parseInt(matcher.group(4)); + + return new Descriptor(directory, timestamp, generation, journalVersion, userVersion); + } + + static Descriptor fromFile(File file) + { + return fromName(file.parent(), file.name()); + } + + Descriptor withIncrementedGeneration() + { + return new Descriptor(directory, timestamp, generation + 1, journalVersion, userVersion); + } + + File fileFor(Component component) + { + return new File(directory, formatFileName(component)); + } + + File tmpFileFor(Component component) + { + return new File(directory, formatFileName(component) + '.' + TMP_SUFFIX); + } + + static boolean isTmpFile(File file) + { + return TMP_FILE_PATTERN.matcher(file.name()).matches(); + } + + private String formatFileName(Component component) + { + return format("%s%d%s%d%s%d%s%d.%s", + PREFIX, timestamp, + SEPARATOR, generation, + SEPARATOR, journalVersion, + SEPARATOR, userVersion, + component.extension); + } + + static List list(File directory) + { + try + { + return Arrays.stream(directory.listNames((file, name) -> DATA_FILE_PATTERN.matcher(name).matches())) + .map(name -> fromName(directory, name)) + .collect(toList()); + } + catch (IOException e) + { + throw new RuntimeException(e); + } + } + + @Override + public int compareTo(Descriptor other) + { + assert this.directory.equals(other.directory) + : format("Descriptors have mismatching directories: %s and %s", this.directory, other.directory); + + int cmp = Long.compare(this.timestamp, other.timestamp); + if (cmp == 0) cmp = Integer.compare(this.generation, other.generation); + if (cmp == 0) cmp = Integer.compare(this.journalVersion, other.journalVersion); + if (cmp == 0) cmp = Integer.compare(this.userVersion, other.userVersion); + return cmp; + } + + @Override + public boolean equals(Object other) + { + if (this == other) + return true; + return (other instanceof Descriptor) && equals((Descriptor) other); + } + + boolean equals(Descriptor other) + { + assert this.directory.equals(other.directory) + : format("Descriptors have mismatching directories: %s and %s", this.directory, other.directory); + + return this.timestamp == other.timestamp + && this.generation == other.generation + && this.journalVersion == other.journalVersion + && this.userVersion == other.userVersion; + } + + @Override + public int hashCode() + { + int result = directory.hashCode(); + result = 31 * result + Long.hashCode(timestamp); + result = 31 * result + generation; + result = 31 * result + journalVersion; + result = 31 * result + userVersion; + return result; + } + + @Override + public String toString() + { + return format("dir: %s, ts: %d, gen: %d, journal ver: %d, user ver: %d", + directory, timestamp, generation, journalVersion, userVersion); + } +} diff --git a/src/java/org/apache/cassandra/journal/EntrySerializer.java b/src/java/org/apache/cassandra/journal/EntrySerializer.java new file mode 100644 index 0000000000..ab1d026499 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/EntrySerializer.java @@ -0,0 +1,226 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.EOFException; +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.Set; +import java.util.zip.CRC32; + +import org.agrona.collections.IntHashSet; +import org.apache.cassandra.db.TypeSizes; +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.Crc; + +import static org.apache.cassandra.journal.Journal.validateCRC; +import static org.apache.cassandra.utils.FBUtilities.updateChecksum; +import static org.apache.cassandra.utils.FBUtilities.updateChecksumInt; +import static org.apache.cassandra.utils.FBUtilities.updateChecksumShort; + +final class EntrySerializer +{ + static void write(K key, + ByteBuffer record, + Set hosts, + KeySupport keySupport, + DataOutputPlus out, + int userVersion) + throws IOException + { + CRC32 crc = Crc.crc32(); + + keySupport.serialize(key, out, userVersion); + keySupport.updateChecksum(crc, key, userVersion); + + out.writeShort(hosts.size()); + updateChecksumShort(crc, (short) hosts.size()); + + int recordSize = record.remaining(); + out.writeInt(recordSize); + updateChecksumInt(crc, recordSize); + + out.writeInt((int) crc.getValue()); + + for (int host : hosts) + { + out.writeInt(host); + updateChecksumInt(crc, host); + } + + out.write(record); + Crc.updateCrc32(crc, record, record.position(), record.limit()); + + out.writeInt((int) crc.getValue()); + } + + static void read(EntryHolder into, + KeySupport keySupport, + ByteBuffer buffer, + int userVersion) + throws IOException + { + CRC32 crc = Crc.crc32(); + into.clear(); + + try (DataInputBuffer in = new DataInputBuffer(buffer, false)) + { + K key = keySupport.deserialize(in, userVersion); + keySupport.updateChecksum(crc, key, userVersion); + into.key = key; + + int hostCount = in.readShort(); + updateChecksumShort(crc, (short) hostCount); + + int entrySize = in.readInt(); + updateChecksumInt(crc, entrySize); + + validateCRC(crc, in.readInt()); + + for (int i = 0; i < hostCount; i++) + { + int hostId = in.readInt(); + updateChecksumInt(crc, hostId); + into.hosts.add(hostId); + } + + ByteBuffer entry = ByteBufferUtil.read(in, entrySize); + updateChecksum(crc, entry); + into.value = entry; + + validateCRC(crc, in.readInt()); + } + } + + static boolean tryRead(EntryHolder into, + KeySupport keySupport, + ByteBuffer buffer, + DataInputBuffer in, + int syncedOffset, + int userVersion) + throws IOException + { + CRC32 crc = Crc.crc32(); + into.clear(); + + int fixedSize = EntrySerializer.fixedEntrySize(keySupport, userVersion); + if (buffer.remaining() < fixedSize) + return handleReadException(new EOFException(), buffer.limit(), syncedOffset); + + updateChecksum(crc, buffer, buffer.position(), fixedSize - TypeSizes.INT_SIZE); + int fixedCrc = buffer.getInt(buffer.position() + fixedSize - TypeSizes.INT_SIZE); + + try + { + validateCRC(crc, fixedCrc); + } + catch (IOException e) + { + return handleReadException(e, buffer.position() + fixedSize, syncedOffset); + } + + int hostCount, recordSize; + try + { + into.key = keySupport.deserialize(in, userVersion); + hostCount = in.readShort(); + recordSize = in.readInt(); + in.skipBytesFully(TypeSizes.INT_SIZE); + } + catch (IOException e) + { + throw new RuntimeException(); // can't happen unless deserializer is buggy + } + + int variableSize = EntrySerializer.variableEntrySize(hostCount, recordSize); + if (buffer.remaining() < variableSize) + return handleReadException(new EOFException(), buffer.limit(), syncedOffset); + + updateChecksum(crc, buffer, buffer.position(), variableSize - TypeSizes.INT_SIZE); + int variableCrc = buffer.getInt(buffer.position() + variableSize - TypeSizes.INT_SIZE); + + try + { + validateCRC(crc, variableCrc); + } + catch (IOException e) + { + return handleReadException(e, buffer.position() + variableSize, syncedOffset); + } + + for (int i = 0; i < hostCount; i++) + { + into.hosts.add(in.readInt()); + } + + try + { + in.skipBytesFully(recordSize); + } + catch (IOException e) + { + throw new AssertionError(); // can't happen + } + + into.value = (ByteBuffer) buffer.duplicate() + .position(buffer.position() - recordSize) + .limit(buffer.position()); + + in.skipBytesFully(TypeSizes.INT_SIZE); + return true; + } + + private static boolean handleReadException(IOException e, int bufferPosition, int fsyncedLimit) throws IOException + { + if (bufferPosition <= fsyncedLimit) + throw e; + else + return false; + } + + static int fixedEntrySize(KeySupport keySupport, int userVersion) + { + return keySupport.serializedSize(userVersion) // key/id + + TypeSizes.SHORT_SIZE // host count + + TypeSizes.INT_SIZE // record size + + TypeSizes.INT_SIZE; // CRC + } + + static int variableEntrySize(int hostCount, int recordSize) + { + return TypeSizes.INT_SIZE * hostCount // hosts + + recordSize // record + + TypeSizes.INT_SIZE; // CRC + } + + static final class EntryHolder + { + K key; + ByteBuffer value; + IntHashSet hosts = new IntHashSet(); + + void clear() + { + key = null; + value = null; + hosts.clear(); + } + } +} diff --git a/src/java/org/apache/cassandra/journal/Flusher.java b/src/java/org/apache/cassandra/journal/Flusher.java new file mode 100644 index 0000000000..b6d5a29141 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Flusher.java @@ -0,0 +1,368 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.util.ArrayList; +import java.util.concurrent.atomic.AtomicLong; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import com.codahale.metrics.Timer; +import org.apache.cassandra.concurrent.Interruptible; +import org.apache.cassandra.concurrent.Interruptible.TerminateException; +import org.apache.cassandra.utils.MonotonicClock; +import org.apache.cassandra.utils.NoSpamLogger; +import org.apache.cassandra.utils.concurrent.Semaphore; +import org.apache.cassandra.utils.concurrent.WaitQueue; + +import static java.lang.String.format; +import static java.util.concurrent.TimeUnit.MINUTES; +import static java.util.concurrent.TimeUnit.NANOSECONDS; +import static org.apache.cassandra.concurrent.ExecutorFactory.Global.executorFactory; +import static org.apache.cassandra.concurrent.InfiniteLoopExecutor.Daemon.NON_DAEMON; +import static org.apache.cassandra.concurrent.InfiniteLoopExecutor.Interrupts.SYNCHRONIZED; +import static org.apache.cassandra.concurrent.InfiniteLoopExecutor.SimulatorSafe.SAFE; +import static org.apache.cassandra.concurrent.Interruptible.State.NORMAL; +import static org.apache.cassandra.concurrent.Interruptible.State.SHUTTING_DOWN; +import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis; +import static org.apache.cassandra.utils.Clock.Global.nanoTime; +import static org.apache.cassandra.utils.LocalizeString.toLowerCaseLocalized; +import static org.apache.cassandra.utils.MonotonicClock.Global.preciseTime; +import static org.apache.cassandra.utils.concurrent.Semaphore.newSemaphore; +import static org.apache.cassandra.utils.concurrent.WaitQueue.newWaitQueue; + +final class Flusher +{ + private static final Logger logger = LoggerFactory.getLogger(Flusher.class); + + private final Journal journal; + private final Params params; + + private volatile Interruptible flushExecutor; + + // counts of total pending write and written entries + private final AtomicLong pending = new AtomicLong(0); + private final AtomicLong written = new AtomicLong(0); + + // all Allocations written before this time will be flushed + volatile long lastFlushedAt = currentTimeMillis(); + + // a signal that writers can wait on to be notified of a completed flush in PERIODIC FlushMode + private final WaitQueue flushComplete = newWaitQueue(); + + // a signal and flag that callers outside the flusher thread can use + // to signal they want the journal segments to be flushed to disk + private final Semaphore haveWork = newSemaphore(1); + private volatile boolean flushRequested; + + private final FlushMethod syncFlushMethod; + private final FlushMethod asyncFlushMethod; + + Flusher(Journal journal) + { + this.journal = journal; + this.params = journal.params; + this.syncFlushMethod = syncFlushMethod(params); + this.asyncFlushMethod = asyncFlushMethod(params); + } + + void start() + { + String flushExecutorName = journal.name + "-disk-flusher-" + toLowerCaseLocalized(params.flushMode().toString()); + flushExecutor = executorFactory().infiniteLoop(flushExecutorName, new FlushRunnable(preciseTime), SAFE, NON_DAEMON, SYNCHRONIZED); + } + + void shutdown() + { + flushExecutor.shutdown(); + } + + private class FlushRunnable implements Interruptible.Task + { + private final MonotonicClock clock; + private final NoSpamLogger noSpamLogger; + + private final ArrayList> segmentsToFlush = new ArrayList<>(); + + FlushRunnable(MonotonicClock clock) + { + this.clock = clock; + this.noSpamLogger = NoSpamLogger.wrap(logger, 5, MINUTES); + } + + @Override + public void run(Interruptible.State state) throws InterruptedException + { + try + { + doRun(state); + } + catch (Throwable t) + { + if (!journal.handleError("Failed to flush segments to disk", t)) + throw new TerminateException(); + else // sleep for full poll-interval after an error, so we don't spam the log file + haveWork.tryAcquire(1, flushPeriodNanos(), NANOSECONDS); + } + } + + public void doRun(Interruptible.State state) throws InterruptedException + { + long startedRunAt = clock.now(); + boolean flushToDisk = lastFlushedAt + flushPeriodNanos() <= startedRunAt || state != NORMAL || flushRequested; + + // synchronized to prevent thread interrupts while performing IO operations and also + // clear interrupted status to prevent ClosedByInterruptException in ActiveSegment::flush + synchronized (this) + { + boolean ignore = Thread.interrupted(); + if (flushToDisk) + { + flushRequested = false; + doFlush(); + lastFlushedAt = startedRunAt; + flushComplete.signalAll(); + } + } + + long now = clock.now(); + if (flushToDisk) + processFlushDuration(startedRunAt, now); + + if (state == SHUTTING_DOWN) + return; + + long wakeUpAt = startedRunAt + flushPeriodNanos(); + if (wakeUpAt > now) + haveWork.tryAcquireUntil(1, wakeUpAt); + } + + private void doFlush() + { + journal.selectSegmentToFlush(segmentsToFlush); + // only schedule onSuccess callbacks for a segment if the preceding segments + // have been fully flushed, to preserve 1:1 mapping between record's position + // in the journal and onSuccess callback scheduling order + boolean scheduleOnSuccessCallbacks = true; + try + { + for (ActiveSegment segment : segmentsToFlush) + { + try + { + scheduleOnSuccessCallbacks = doFlush(segment, scheduleOnSuccessCallbacks) && scheduleOnSuccessCallbacks; + } + catch (Throwable t) + { + segmentsToFlush.forEach(s -> s.scheduleOnFailureCallbacks(t)); + throw t; + } + } + } + finally + { + segmentsToFlush.clear(); + } + } + + // flush the segment, schedule write callbacks if requested, return whether the segment has been flushed fully + private boolean doFlush(ActiveSegment segment, boolean scheduleCallbacks) + { + int syncedOffset = segment.flush(); + if (scheduleCallbacks) + segment.scheduleOnSuccessCallbacks(syncedOffset); + return segment.isFullyFlushed(syncedOffset); + } + + private long firstLaggedAt = Long.MIN_VALUE; // first lag ever or since last logged warning + private int flushCount = 0; // flush count since firstLaggedAt + private int lagCount = 0; // lag count since firstLaggedAt + private long flushDuration = 0; // time spent flushing since firstLaggedAt + private long lagDuration = 0; // cumulative lag since firstLaggedAt + + private void processFlushDuration(long startedFlushAt, long finishedFlushAt) + { + flushCount++; + flushDuration += (finishedFlushAt - startedFlushAt); + + long lag = finishedFlushAt - (startedFlushAt + flushPeriodNanos()); + if (lag <= 0) + return; + + lagCount++; + lagDuration += lag; + + if (firstLaggedAt == Long.MIN_VALUE) + firstLaggedAt = finishedFlushAt; + + boolean logged = + noSpamLogger.warn(finishedFlushAt, + "Out of {} {} journal flushes over the past {}s with average duration of {}ms, " + + "{} have exceeded the configured flush period by an average of {}ms", + flushCount, + journal.name, + format("%.2f", (finishedFlushAt - firstLaggedAt) * 1e-9d), + format("%.2f", flushDuration * 1e-6d / flushCount), + lagCount, + format("%.2f", lagDuration * 1e-6d / lagCount)); + + if (logged) // reset metrics for next log statement + { + firstLaggedAt = Long.MIN_VALUE; + flushCount = lagCount = 0; + flushDuration = lagDuration = 0; + } + } + } + + @FunctionalInterface + private interface FlushMethod + { + void flush(ActiveSegment.Allocation allocation); + } + + private FlushMethod syncFlushMethod(Params params) + { + switch (params.flushMode()) + { + default: throw new IllegalArgumentException(); + case BATCH: return this::waitForFlushBatch; + case GROUP: return this::waitForFlushGroup; + case PERIODIC: return this::waitForFlushPeriodic; + } + } + + private FlushMethod asyncFlushMethod(Params params) + { + switch (params.flushMode()) + { + default: throw new IllegalArgumentException(); + case BATCH: return this::asyncFlushBatch; + case GROUP: return this::asyncFlushGroup; + case PERIODIC: return this::asyncFlushPeriodic; + } + } + + void waitForFlush(ActiveSegment.Allocation alloc) + { + syncFlushMethod.flush(alloc); + } + + void asyncFlush(ActiveSegment.Allocation alloc) + { + asyncFlushMethod.flush(alloc); + } + + private void waitForFlushBatch(ActiveSegment.Allocation alloc) + { + pending.incrementAndGet(); + requestExtraFlush(); + alloc.awaitFlush(journal.metrics.waitingOnFlush); + pending.decrementAndGet(); + written.incrementAndGet(); + } + + private void asyncFlushBatch(ActiveSegment.Allocation alloc) + { + pending.incrementAndGet(); + requestExtraFlush(); + // alloc.awaitFlush(journal.metrics.waitingOnFlush); // TODO FIXME + pending.decrementAndGet(); + written.incrementAndGet(); + } + + private void waitForFlushGroup(ActiveSegment.Allocation alloc) + { + pending.incrementAndGet(); + alloc.awaitFlush(journal.metrics.waitingOnFlush); + pending.decrementAndGet(); + written.incrementAndGet(); + } + + private void asyncFlushGroup(ActiveSegment.Allocation alloc) + { + pending.incrementAndGet(); + // alloc.awaitFlush(journal.metrics.waitingOnFlush); // TODO FIXME + pending.decrementAndGet(); + written.incrementAndGet(); + } + + private void waitForFlushPeriodic(ActiveSegment.Allocation alloc) + { + long expectedFlushTime = nanoTime() - periodicFlushLagBlockNanos(); + if (lastFlushedAt < expectedFlushTime) + { + pending.incrementAndGet(); + awaitFlushAt(expectedFlushTime, journal.metrics.waitingOnFlush.time()); + pending.decrementAndGet(); + } + written.incrementAndGet(); + } + + private void asyncFlushPeriodic(ActiveSegment.Allocation ignore) + { + pending.incrementAndGet(); + // awaitFlushAt(expectedFlushTime, journal.metrics.waitingOnFlush.time()); // TODO FIXME + pending.decrementAndGet(); + written.incrementAndGet(); + } + + /** + * Request an additional flush cycle without blocking + */ + void requestExtraFlush() + { + // note: cannot simply invoke executor.interrupt() as some filesystems don't like it (jimfs, at least) + flushRequested = true; + haveWork.release(1); + } + + private void awaitFlushAt(long flushTime, Timer.Context context) + { + do + { + WaitQueue.Signal signal = flushComplete.register(context, Timer.Context::stop); + if (lastFlushedAt < flushTime) + signal.awaitUninterruptibly(); + else + signal.cancel(); + } + while (lastFlushedAt < flushTime); + } + + private long flushPeriodNanos() + { + return 1_000_000L * params.flushPeriod(); + } + + private long periodicFlushLagBlockNanos() + { + return 1_000_000L * params.periodicFlushLagBlock(); + } + + long pendingEntries() + { + return pending.get(); + } + + long writtenEntries() + { + return written.get(); + } +} diff --git a/src/java/org/apache/cassandra/journal/InMemoryIndex.java b/src/java/org/apache/cassandra/journal/InMemoryIndex.java new file mode 100644 index 0000000000..e4ec73a679 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/InMemoryIndex.java @@ -0,0 +1,137 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.util.Arrays; +import java.util.NavigableMap; +import java.util.TreeMap; +import java.util.concurrent.ConcurrentSkipListMap; +import java.util.concurrent.atomic.AtomicReference; + +import javax.annotation.Nullable; + +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.FileOutputStreamPlus; + +/** + * An index for a segment that's still being updated by journal writers concurrently. + */ +final class InMemoryIndex extends Index +{ + private static final int[] EMPTY = new int[0]; + + private final NavigableMap index; + + // CSLM#lastKey() can be costly, so track lastId separately; + // TODO: this could easily be premature and misguided; + // benchmark to ensure it's not acitevly harmful + private final AtomicReference lastId; + + static InMemoryIndex create(KeySupport keySupport) + { + return new InMemoryIndex<>(keySupport, new ConcurrentSkipListMap<>(keySupport)); + } + + private InMemoryIndex(KeySupport keySupport, NavigableMap index) + { + super(keySupport); + this.index = index; + this.lastId = new AtomicReference<>(); + } + + public void update(K id, int offset) + { + index.merge(id, new int[] { offset }, (current, value) -> + { + int idx = Arrays.binarySearch(current, offset); + if (idx >= 0) // repeat update() call; shouldn't occur, but we might as well allow this NOOP + return current; + + /* Merge the new offset with existing values */ + int pos = -idx - 1; + int[] merged = new int[current.length + 1]; + System.arraycopy(current, 0, merged, 0, pos); + merged[pos] = offset; + System.arraycopy(current, pos, merged, pos + 1, current.length - pos); + return merged; + }); + + lastId.accumulateAndGet(id, (current, update) -> (null == current || keySupport.compare(current, update) < 0) ? update : current); + } + + @Override + @Nullable + public K firstId() + { + return index.isEmpty() ? null : index.firstKey(); + } + + @Override + @Nullable + public K lastId() + { + return lastId.get(); + } + + @Override + public int[] lookUp(K id) + { + return mayContainId(id) ? index.getOrDefault(id, EMPTY) : EMPTY; + } + + @Override + public int lookUpFirst(K id) + { + int[] offests = lookUp(id); + return offests.length == 0 ? -1 : offests[0]; + } + + public void persist(Descriptor descriptor) + { + File tmpFile = descriptor.tmpFileFor(Component.INDEX); + try (FileOutputStreamPlus out = new FileOutputStreamPlus(tmpFile)) + { + OnDiskIndex.write(index, keySupport, out, descriptor.userVersion); + + out.flush(); + out.sync(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, tmpFile, e); + } + tmpFile.move(descriptor.fileFor(Component.INDEX)); + } + + static InMemoryIndex rebuild(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + InMemoryIndex index = new InMemoryIndex<>(keySupport, new TreeMap<>(keySupport)); + try (StaticSegment.SequentialReader reader = StaticSegment.reader(descriptor, keySupport, fsyncedLimit)) + { + while (reader.advance()) + index.update(reader.id(), reader.offset()); + } + return index; + } + + @Override + public void close() + { + } +} diff --git a/src/java/org/apache/cassandra/journal/Index.java b/src/java/org/apache/cassandra/journal/Index.java new file mode 100644 index 0000000000..cfce4407e9 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Index.java @@ -0,0 +1,76 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import javax.annotation.Nullable; + +import org.apache.cassandra.utils.Closeable; + +/** + * Mapping of client supplied ids to in-segment offsets + */ +abstract class Index implements Closeable +{ + final KeySupport keySupport; + + Index(KeySupport keySupport) + { + this.keySupport = keySupport; + } + + /** + * Look up offsets by id. It's possible, due to retries, for a segment + * to contain the same record with the same id more than once, at + * different offsets. + * + * @return the found offsets into the segment, if any; can be empty + */ + abstract int[] lookUp(K id); + + /** + * Look up offsets by id. It's possible, due to retries, for a segment + * to contain the same record with the same id more than once, at + * different offsets. Return the first offset for provided record id, or -1 if none. + * + * @return the first offset into the segment, or -1 is none were found + */ + abstract int lookUpFirst(K id); + + /** + * @return the first (smallest) id in the index + */ + @Nullable + abstract K firstId(); + + /** + * @return the last (largest) id in the index + */ + @Nullable + abstract K lastId(); + + /** + * @return whether the id falls within lower/upper bounds of the index + */ + boolean mayContainId(K id) + { + K firstId = firstId(); + K lastId = lastId(); + + return null != firstId && null != lastId && keySupport.compare(id, firstId) >= 0 && keySupport.compare(id, lastId) <= 0; + } +} diff --git a/src/java/org/apache/cassandra/journal/Journal.java b/src/java/org/apache/cassandra/journal/Journal.java new file mode 100644 index 0000000000..ff90c58f7b --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Journal.java @@ -0,0 +1,638 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.file.FileStore; +import java.util.ArrayList; +import java.util.Collection; +import java.util.List; +import java.util.Set; +import java.util.concurrent.Executor; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicLong; +import java.util.concurrent.atomic.AtomicReference; +import java.util.function.BooleanSupplier; +import java.util.function.Function; +import java.util.zip.CRC32; + +import javax.annotation.Nonnull; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import com.codahale.metrics.Timer.Context; +import org.apache.cassandra.concurrent.Interruptible; +import org.apache.cassandra.concurrent.Interruptible.TerminateException; +import org.apache.cassandra.concurrent.SequentialExecutorPlus; +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.io.util.DataOutputBuffer; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.PathUtils; +import org.apache.cassandra.journal.Segments.ReferencedSegments; +import org.apache.cassandra.service.StorageService; +import org.apache.cassandra.utils.Crc; +import org.apache.cassandra.utils.JVMStabilityInspector; +import org.apache.cassandra.utils.concurrent.WaitQueue; + +import static java.lang.String.format; +import static java.util.Comparator.comparing; +import static org.apache.cassandra.concurrent.ExecutorFactory.Global.executorFactory; +import static org.apache.cassandra.concurrent.InfiniteLoopExecutor.Daemon.NON_DAEMON; +import static org.apache.cassandra.concurrent.InfiniteLoopExecutor.Interrupts.SYNCHRONIZED; +import static org.apache.cassandra.concurrent.InfiniteLoopExecutor.SimulatorSafe.SAFE; +import static org.apache.cassandra.concurrent.Interruptible.State.NORMAL; +import static org.apache.cassandra.concurrent.Interruptible.State.SHUTTING_DOWN; +import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis; +import static org.apache.cassandra.utils.concurrent.WaitQueue.newWaitQueue; + +/** + * A generic append-only journal with some special features: + *

    + *
  • Records can be looked up by key + *
  • Records can be tagged with multiple owner node ids + *
  • Records can be invalidated by their owner ids + *
  • Fully invalidated records get purged during segment compaction + *

+ * + * Type parameters: + * @param the type of records stored in the journal + * @param the type of keys used to address the records; + must be fixed-size and byte-order comparable + */ +public class Journal +{ + private static final Logger logger = LoggerFactory.getLogger(Journal.class); + + final String name; + final File directory; + final Params params; + + final KeySupport keySupport; + final ValueSerializer valueSerializer; + + final Metrics metrics; + final Flusher flusher; + //final Invalidator invalidator; + //final Compactor compactor; + + volatile long replayLimit; + final AtomicLong nextSegmentId = new AtomicLong(); + + private volatile ActiveSegment currentSegment = null; + + // segment that is ready to be used; allocator thread fills this and blocks until consumed + private volatile ActiveSegment availableSegment = null; + + private final AtomicReference> segments = new AtomicReference<>(); + + Interruptible allocator; + private final WaitQueue segmentPrepared = newWaitQueue(); + private final WaitQueue allocatorThreadWaitQueue = newWaitQueue(); + private final BooleanSupplier allocatorThreadWaitCondition = () -> (availableSegment == null); + + SequentialExecutorPlus closer; + //private final Set invalidations = Collections.newSetFromMap(new ConcurrentHashMap<>()); + + public Journal(String name, + File directory, + Params params, + KeySupport keySupport, + ValueSerializer valueSerializer) + { + this.name = name; + this.directory = directory; + this.params = params; + + this.keySupport = keySupport; + this.valueSerializer = valueSerializer; + + this.metrics = new Metrics<>(name); + this.flusher = new Flusher<>(this); + //this.invalidator = new Invalidator<>(this); + //this.compactor = new Compactor<>(this); + } + + public void start() + { + metrics.register(flusher); + + deleteTmpFiles(); + + List descriptors = Descriptor.list(directory); + // find the largest existing timestamp + descriptors.sort(null); + long maxTimestamp = descriptors.isEmpty() + ? Long.MIN_VALUE + : descriptors.get(descriptors.size() - 1).timestamp; + nextSegmentId.set(replayLimit = Math.max(currentTimeMillis(), maxTimestamp + 1)); + + segments.set(Segments.ofStatic(StaticSegment.open(descriptors, keySupport))); + closer = executorFactory().sequential(name + "-closer"); + allocator = executorFactory().infiniteLoop(name + "-allocator", new AllocateRunnable(), SAFE, NON_DAEMON, SYNCHRONIZED); + advanceSegment(null); + flusher.start(); + //invalidator.start(); + //compactor.start(); + } + + /** + * Cleans up unfinished component files from previous run (metadata and index) + */ + private void deleteTmpFiles() + { + for (File tmpFile : directory.listUnchecked(Descriptor::isTmpFile)) + tmpFile.delete(); + } + + public void shutdown() + { + allocator.shutdown(); + //compactor.stop(); + //invalidator.stop(); + flusher.shutdown(); + closer.shutdown(); + closeAllSegments(); + metrics.deregister(); + } + + /** + * Looks up a record by the provided id. + *

+ * Looking up an invalidated record may or may not return a record, depending on + * compaction progress. + *

+ * In case multiple copies of the record exist in the log (e.g. because of user retries), + * only the first found record will be consumed. + * + * @param id user-provided record id, expected to roughly correlate with time and go up + * @param consumer function to consume the raw record (bytes and invalidation set) if found + * @return true if the record was found, false otherwise + */ + public boolean read(K id, RecordConsumer consumer) + { + try (ReferencedSegments segments = selectAndReference(id)) + { + for (Segment segment : segments.all()) + if (segment.read(id, consumer)) + return true; + } + + return false; + } + + /** + * Looks up a record by the provided id. + *

+ * Looking up an invalidated record may or may not return a record, depending on + * compaction progress. + *

+ * In case multiple copies of the record exist in the log (e.g. because of user retries), + * the first one found will be returned. + * + * @param id user-provided record id, expected to roughly correlate with time and go up + * @return deserialized record if found, null otherwise + */ + public V read(K id) + { + EntrySerializer.EntryHolder holder = new EntrySerializer.EntryHolder<>(); + + try (ReferencedSegments segments = selectAndReference(id)) + { + for (Segment segment : segments.all()) + { + if (segment.read(id, holder)) + { + try (DataInputBuffer in = new DataInputBuffer(holder.value, false)) + { + return valueSerializer.deserialize(holder.key, in, segment.descriptor.userVersion); + + } + catch (IOException e) + { + // can only throw if serializer is buggy + throw new RuntimeException(e); + } + } + } + } + + return null; + } + + /** + * Synchronously write a record to the journal. + *

+ * Blocks until the record has been deemed durable according to the journal flush mode. + * + * @param id user-provided record id, expected to roughly correlate with time and go up + * @param record the record to store + * @param hosts hosts expected to invalidate the record + */ + public void write(K id, V record, Set hosts) + { + try (DataOutputBuffer dob = DataOutputBuffer.scratchBuffer.get()) + { + valueSerializer.serialize(record, dob, params.userVersion()); + ActiveSegment.Allocation alloc = allocate(dob.getLength(), hosts); + alloc.write(id, dob.unsafeGetBufferAndFlip(), hosts); + flusher.waitForFlush(alloc); + } + catch (IOException e) + { + // exception during record serialization into the scratch buffer + throw new RuntimeException(e); + } + } + + /** + * Asynchronously write a record to the journal. Writes to the journal in the calling thread, + * but doesn't wait for flush. + *

+ * Executes the supplied callback on the executor provided once the record has been durably written to disk + * + * @param id user-provided record id, expected to roughly correlate with time and go up + * @param record the record to store + * @param hosts hosts expected to invalidate the record + * @param executor executor to run the callback on + * @param callback the callback to run on + */ + public void asyncWrite(K id, V record, Set hosts, @Nonnull Executor executor, @Nonnull AsyncWriteCallback callback) + { + try (DataOutputBuffer dob = DataOutputBuffer.scratchBuffer.get()) + { + valueSerializer.serialize(record, dob, params.userVersion()); + ActiveSegment.Allocation alloc = allocate(dob.getLength(), hosts); + alloc.asyncWrite(id, dob.unsafeGetBufferAndFlip(), hosts, executor, callback); + flusher.asyncFlush(alloc); + } + catch (IOException e) + { + // exception during record serialization into the scratch buffer + executor.execute(() -> callback.onFailure(e)); + } + } + + private ActiveSegment.Allocation allocate(int entrySize, Set hosts) + { + ActiveSegment segment = currentSegment; + + ActiveSegment.Allocation alloc; + while (null == (alloc = segment.allocate(entrySize, hosts))) + { + // failed to allocate; move to a new segment with enough room + advanceSegment(segment); + segment = currentSegment; + } + return alloc; + } + + /* + * Segment allocation logic. + */ + + private void advanceSegment(ActiveSegment oldSegment) + { + while (true) + { + synchronized (this) + { + // do this in a critical section, so we can maintain the order of + // segment construction when moving to allocatingFrom/activeSegments + if (currentSegment != oldSegment) + return; + + // if a segment is ready, take it now, otherwise wait for the allocator thread to construct it + if (availableSegment != null) + { + // success - change allocatingFrom and activeSegments (which must be kept in order) before leaving the critical section + addNewActiveSegment(currentSegment = availableSegment); + availableSegment = null; + break; + } + } + + awaitAvailableSegment(oldSegment); + } + + // signal the allocator thread to prepare a new segment + wakeAllocator(); + + if (null != oldSegment) + closeActiveSegmentAndOpenAsStatic(oldSegment); + + // request that the journal be flushed out-of-band, as we've finished a segment + flusher.requestExtraFlush(); + } + + private void awaitAvailableSegment(ActiveSegment currentActiveSegment) + { + do + { + WaitQueue.Signal prepared = segmentPrepared.register(metrics.waitingOnSegmentAllocation.time(), Context::stop); + if (availableSegment == null && currentSegment == currentActiveSegment) + prepared.awaitUninterruptibly(); + else + prepared.cancel(); + } + while (availableSegment == null && currentSegment == currentActiveSegment); + } + + private void wakeAllocator() + { + allocatorThreadWaitQueue.signalAll(); + } + + private void discardAvailableSegment() + { + ActiveSegment next; + synchronized (this) + { + next = availableSegment; + availableSegment = null; + } + if (next != null) + next.closeAndDiscard(); + } + + private class AllocateRunnable implements Interruptible.Task + { + @Override + public void run(Interruptible.State state) throws InterruptedException + { + if (state == NORMAL) + runNormal(); + else if (state == SHUTTING_DOWN) + shutDown(); + } + + private void runNormal() throws InterruptedException + { + boolean interrupted = false; + try + { + if (availableSegment != null) + throw new IllegalStateException("availableSegment is not null"); + + // synchronized to prevent thread interrupts while performing IO operations and also + // clear interrupted status to prevent ClosedByInterruptException in createSegment() + synchronized (this) + { + interrupted = Thread.interrupted(); + availableSegment = createSegment(); + + segmentPrepared.signalAll(); + Thread.yield(); + } + } + catch (Throwable t) + { + if (!handleError("Failed allocating journal segments", t)) + { + discardAvailableSegment(); + throw new TerminateException(); + } + TimeUnit.SECONDS.sleep(1L); // sleep for a second to avoid log spam + } + + interrupted = interrupted || Thread.interrupted(); + if (!interrupted) + { + try + { + // If we offered a segment, wait for it to be taken before reentering the loop. + // There could be a new segment in next not offered, but only on failure to discard it while + // shutting down-- nothing more can or needs to be done in that case. + WaitQueue.waitOnCondition(allocatorThreadWaitCondition, allocatorThreadWaitQueue); + } + catch (InterruptedException e) + { + interrupted = true; + } + } + + if (interrupted) + { + discardAvailableSegment(); + throw new InterruptedException(); + } + } + + private void shutDown() throws InterruptedException + { + try + { + // if shutdown() started and finished during segment creation, we'll be left with a + // segment that no one will consume; discard it + discardAvailableSegment(); + } + catch (Throwable t) + { + handleError("Failed shutting down segment allocator", t); + throw new TerminateException(); + } + } + } + + private ActiveSegment createSegment() + { + Descriptor descriptor = Descriptor.create(directory, nextSegmentId.getAndIncrement(), params.userVersion()); + return ActiveSegment.create(descriptor, params, keySupport); + } + + private void closeAllSegments() + { + Segments segments = swapSegments(ignore -> Segments.none()); + + for (ActiveSegment segment : segments.onlyActive()) + segment.closeAndIfEmptyDiscard(); + for (StaticSegment segment : segments.onlyStatic()) + segment.close(); + } + + /** + * Select segments that could potentially have an entry with the specified id and + * attempt to grab references to them all. + * + * @return a subset of segments with references to them + */ + ReferencedSegments selectAndReference(K id) + { + while (true) + { + ReferencedSegments referenced = segments().selectAndReference(id); + if (null != referenced) + return referenced; + } + } + + private Segments segments() + { + return segments.get(); + } + + private Segments swapSegments(Function, Segments> transformation) + { + Segments currentSegments, newSegments; + do + { + currentSegments = segments(); + newSegments = transformation.apply(currentSegments); + } + while (!segments.compareAndSet(currentSegments, newSegments)); + return currentSegments; + } + + private void addNewActiveSegment(ActiveSegment activeSegment) + { + swapSegments(current -> current.withNewActiveSegment(activeSegment)); + } + + private void replaceCompletedSegment(ActiveSegment activeSegment, StaticSegment staticSegment) + { + swapSegments(current -> current.withCompletedSegment(activeSegment, staticSegment)); + } + + private void replaceCompactedSegment(StaticSegment oldSegment, StaticSegment newSegment) + { + swapSegments(current -> current.withCompactedSegment(oldSegment, newSegment)); + } + + void selectSegmentToFlush(Collection> into) + { + ActiveSegment current = currentSegment; + for (ActiveSegment segment : segments().onlyActive()) + { + // do not sync segments that became active after flush started + if (segment.descriptor.timestamp <= current.descriptor.timestamp) + into.add(segment); + } + } + + /** + * Take care of a finished active segment: + * 1. discard tail + * 2. flush to disk + * 3. persist index and metadata + * 4. open the segment as static + * 5. replace the finished active segment with the opened static one in Segments view + * 6. release the Ref so the active segment will be cleaned up by its Tidy instance + */ + private class CloseActiveSegmentRunnable implements Runnable + { + private final ActiveSegment activeSegment; + + CloseActiveSegmentRunnable(ActiveSegment activeSegment) + { + this.activeSegment = activeSegment; + } + + @Override + public void run() + { + activeSegment.discardUnusedTail(); + activeSegment.flush(); + activeSegment.persistComponents(); + replaceCompletedSegment(activeSegment, StaticSegment.open(activeSegment.descriptor, keySupport)); + activeSegment.release(); + } + } + + void closeActiveSegmentAndOpenAsStatic(ActiveSegment activeSegment) + { + closer.execute(new CloseActiveSegmentRunnable(activeSegment)); + } + + /* + * Replay logic + */ + + /** + * Iterate over and invoke the supplied callback on every record, + * with segments iterated in segment timestamp order. Only visits + * finished, on-disk segments. + */ + public void replayStaticSegments(RecordConsumer consumer) + { + List> staticSegments = new ArrayList<>(segments().onlyStatic()); + staticSegments.sort(comparing(segment -> segment.descriptor)); + + for (StaticSegment segment : staticSegments) + segment.forEachRecord(consumer); + } + + /* + * Static helper methods used by journal components + */ + + static void validateCRC(CRC32 crc, int readCRC) throws Crc.InvalidCrc + { + if (readCRC != (int) crc.getValue()) + throw new Crc.InvalidCrc(readCRC, (int) crc.getValue()); + } + + /* + * Error handling + */ + + /** + * @return true if the invoking thread should continue, or false if it should terminate itself + */ + boolean handleError(String message, Throwable t) + { + Params.FailurePolicy policy = params.failurePolicy(); + JVMStabilityInspector.inspectJournalThrowable(t, name, policy); + + switch (policy) + { + default: + throw new AssertionError(policy); + case DIE: + case STOP: + StorageService.instance.stopTransports(); + //$FALL-THROUGH$ + case STOP_JOURNAL: + message = format("%s. Journal %s failure policy is %s; terminating thread.", message, name, policy); + logger.error(maybeAddDiskSpaceContext(message), t); + return false; + case IGNORE: + message = format("%s. Journal %s failure policy is %s; ignoring excepton.", message, name, policy); + logger.error(maybeAddDiskSpaceContext(message), t); + return true; + } + } + + /** + * Add additional information to the error message if the journal directory does not have enough free space. + * + * @param message the original error message + * @return the message with additional information if possible + */ + private String maybeAddDiskSpaceContext(String message) + { + long availableDiskSpace = PathUtils.tryGetSpace(directory.toPath(), FileStore::getTotalSpace); + int segmentSize = params.segmentSize(); + + if (availableDiskSpace >= segmentSize) + return message; + + return format("%s. %d bytes required for next journal segment but only %d bytes available. " + + "Check %s to see if not enough free space is the reason for this error.", + message, segmentSize, availableDiskSpace, directory); + } +} diff --git a/src/java/org/apache/cassandra/journal/JournalReadError.java b/src/java/org/apache/cassandra/journal/JournalReadError.java new file mode 100644 index 0000000000..87366c8d7c --- /dev/null +++ b/src/java/org/apache/cassandra/journal/JournalReadError.java @@ -0,0 +1,38 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import org.apache.cassandra.io.FSReadError; +import org.apache.cassandra.io.util.File; + +public class JournalReadError extends FSReadError +{ + public final Descriptor descriptor; + + JournalReadError(Descriptor descriptor, File file, Throwable throwable) + { + super(throwable, file); + this.descriptor = descriptor; + } + + JournalReadError(Descriptor descriptor, Component component, Throwable throwable) + { + super(throwable, descriptor.fileFor(component)); + this.descriptor = descriptor; + } +} diff --git a/src/java/org/apache/cassandra/journal/JournalWriteError.java b/src/java/org/apache/cassandra/journal/JournalWriteError.java new file mode 100644 index 0000000000..03193af545 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/JournalWriteError.java @@ -0,0 +1,38 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import org.apache.cassandra.io.FSWriteError; +import org.apache.cassandra.io.util.File; + +public class JournalWriteError extends FSWriteError +{ + public final Descriptor descriptor; + + JournalWriteError(Descriptor descriptor, File file, Throwable throwable) + { + super(throwable, file); + this.descriptor = descriptor; + } + + JournalWriteError(Descriptor descriptor, Component component, Throwable throwable) + { + super(throwable, descriptor.fileFor(component)); + this.descriptor = descriptor; + } +} diff --git a/src/java/org/apache/cassandra/journal/KeySupport.java b/src/java/org/apache/cassandra/journal/KeySupport.java new file mode 100644 index 0000000000..13cb902ddf --- /dev/null +++ b/src/java/org/apache/cassandra/journal/KeySupport.java @@ -0,0 +1,47 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.Comparator; +import java.util.zip.Checksum; + +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; + +/** + * Record keys must satisfy two properties: + *

+ * 1. Must have a fixed serialized size + * 2. Must be byte-order comparable + */ +public interface KeySupport extends Comparator +{ + int serializedSize(int userVersion); + + void serialize(K key, DataOutputPlus out, int userVersion) throws IOException; + + K deserialize(DataInputPlus in, int userVersion) throws IOException; + + K deserialize(ByteBuffer buffer, int position, int userVersion); + + void updateChecksum(Checksum crc, K key, int userVersion); + + int compareWithKeyAt(K key, ByteBuffer buffer, int position, int userVersion); +} diff --git a/src/java/org/apache/cassandra/journal/Metadata.java b/src/java/org/apache/cassandra/journal/Metadata.java new file mode 100644 index 0000000000..bc521cc83c --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Metadata.java @@ -0,0 +1,216 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.util.Collections; +import java.util.Map; +import java.util.Set; +import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.atomic.AtomicIntegerFieldUpdater; +import java.util.zip.CRC32; + +import org.agrona.collections.Int2IntHashMap; +import org.agrona.collections.IntHashSet; +import org.apache.cassandra.io.util.*; +import org.apache.cassandra.utils.Crc; + +import static org.apache.cassandra.journal.Journal.validateCRC; +import static org.apache.cassandra.utils.FBUtilities.updateChecksumInt; + +/** + * Tracks and serializes the following information: + * - all the hosts with entries in the data segment and #of records each is tagged in; + * used for compaction prioritisation and to act in response to topology changes + * - total count of records in this segment file + * used for compaction prioritisation + */ +final class Metadata +{ + private final Set unmodifiableHosts; + private final Map recordsPerHost; + + private volatile int recordsCount; + private static final AtomicIntegerFieldUpdater recordsCountUpdater = + AtomicIntegerFieldUpdater.newUpdater(Metadata.class, "recordsCount"); + + static Metadata create() + { + return new Metadata(new ConcurrentHashMap<>(), 0); + } + + private Metadata(Map recordsPerHost, int recordsCount) + { + this.recordsPerHost = recordsPerHost; + this.recordsCount = recordsCount; + this.unmodifiableHosts = Collections.unmodifiableSet(recordsPerHost.keySet()); + } + + void update(Set hosts) + { + updateHosts(hosts); + incrementRecordsCount(); + } + + private void updateHosts(Set hosts) + { + for (int host : hosts) + recordsPerHost.compute(host, (k, v) -> null == v ? 1 : v + 1); + } + + private void incrementRecordsCount() + { + recordsCountUpdater.incrementAndGet(this); + } + + Set hosts() + { + return unmodifiableHosts; + } + + int count(int host) + { + return recordsPerHost.getOrDefault(host, 0); + } + + int totalCount() + { + return recordsCount; + } + + void write(DataOutputPlus out) throws IOException + { + CRC32 crc = Crc.crc32(); + + /* Write records count per host */ + + int size = recordsPerHost.size(); + out.writeInt(size); + updateChecksumInt(crc, size); + + out.writeInt((int) crc.getValue()); + + for (Map.Entry entry : recordsPerHost.entrySet()) + { + int host = entry.getKey(); + int count = entry.getValue(); + + out.writeInt(host); + out.writeInt(count); + + updateChecksumInt(crc, host); + updateChecksumInt(crc, count); + } + + /* Write records count */ + + out.writeInt(recordsCount); + updateChecksumInt(crc, recordsCount); + + out.writeInt((int) crc.getValue()); + } + + static Metadata read(DataInputPlus in) throws IOException + { + CRC32 crc = Crc.crc32(); + + /* Read records count per host */ + + int size = in.readInt(); + updateChecksumInt(crc, size); + validateCRC(crc, in.readInt()); + + Int2IntHashMap recordsPerHost = new Int2IntHashMap(Integer.MIN_VALUE); + for (int i = 0; i < size; i++) + { + int host = in.readInt(); + int count = in.readInt(); + + updateChecksumInt(crc, host); + updateChecksumInt(crc, count); + + recordsPerHost.put(host, count); + } + + /* Read records count */ + + int recordsCount = in.readInt(); + updateChecksumInt(crc, recordsCount); + + validateCRC(crc, in.readInt()); + return new Metadata(recordsPerHost, recordsCount); + } + + void persist(Descriptor descriptor) + { + File tmpFile = descriptor.tmpFileFor(Component.METADATA); + try (FileOutputStreamPlus out = new FileOutputStreamPlus(tmpFile)) + { + write(out); + + out.flush(); + out.sync(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, tmpFile, e); + } + tmpFile.move(descriptor.fileFor(Component.METADATA)); + } + + static Metadata load(Descriptor descriptor) + { + File file = descriptor.fileFor(Component.METADATA); + try (FileInputStreamPlus in = new FileInputStreamPlus(file)) + { + return read(in); + } + catch (IOException e) + { + throw new JournalReadError(descriptor, file, e); + } + } + + static Metadata rebuild(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + Int2IntHashMap recordsPerHost = new Int2IntHashMap(Integer.MIN_VALUE); + int recordsCount = 0; + + try (StaticSegment.SequentialReader reader = StaticSegment.reader(descriptor, keySupport, fsyncedLimit)) + { + while (reader.advance()) + { + // iterator is cached and reused by IntHashSet + IntHashSet.IntIterator hosts = reader.hosts().iterator(); + while (hosts.hasNext()) + recordsPerHost.merge(hosts.nextValue(), 1, Integer::sum); + + ++recordsCount; + } + } + + return new Metadata(recordsPerHost, recordsCount); + } + + static Metadata rebuildAndPersist(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + Metadata metadata = rebuild(descriptor, keySupport, fsyncedLimit); + metadata.persist(descriptor); + return metadata; + } +} diff --git a/src/java/org/apache/cassandra/journal/Metrics.java b/src/java/org/apache/cassandra/journal/Metrics.java new file mode 100644 index 0000000000..befc3c2ddb --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Metrics.java @@ -0,0 +1,75 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import com.codahale.metrics.Gauge; +import com.codahale.metrics.Timer; +import org.apache.cassandra.metrics.CassandraMetricsRegistry; +import org.apache.cassandra.metrics.DefaultNameFactory; +import org.apache.cassandra.metrics.MetricNameFactory; + +final class Metrics +{ + private static final String WAITING_ON_FLUSH = "WaitingOnFlush"; + private static final String WAITING_ON_ALLOCATION = "WaitingOnSegmentAllocation"; + private static final String WRITTEN_ENTRIES = "WrittenEntries"; + private static final String PENDING_ENTRIES = "PendingEntries"; + + /** + * The time spent waiting on journal flush; for {@link org.apache.cassandra.journal.Params.FlushMode#PERIODIC} + * this is only occurs when the flush is lagging its flush interval. + */ + Timer waitingOnFlush; + + /** Time spent waiting for a segment to be allocated - under normal conditions this should be zero */ + Timer waitingOnSegmentAllocation; + + /** Number of pending (flush) entries */ + Gauge pendingEntries; + + /** Number of written (flushed) entries */ + Gauge writtenEntries; + + private final MetricNameFactory factory; + + Metrics(String name) + { + this.factory = new DefaultNameFactory("Journal", name); + } + + void register(Flusher flusher) + { + waitingOnFlush = CassandraMetricsRegistry.Metrics.timer(createName(WAITING_ON_FLUSH)); + waitingOnSegmentAllocation = CassandraMetricsRegistry.Metrics.timer(createName(WAITING_ON_ALLOCATION)); + pendingEntries = CassandraMetricsRegistry.Metrics.register(createName(PENDING_ENTRIES), flusher::pendingEntries); + writtenEntries = CassandraMetricsRegistry.Metrics.register(createName(WRITTEN_ENTRIES), flusher::writtenEntries); + } + + void deregister() + { + CassandraMetricsRegistry.Metrics.remove(createName(WAITING_ON_FLUSH)); + CassandraMetricsRegistry.Metrics.remove(createName(WAITING_ON_ALLOCATION)); + CassandraMetricsRegistry.Metrics.remove(createName(PENDING_ENTRIES)); + CassandraMetricsRegistry.Metrics.remove(createName(WRITTEN_ENTRIES)); + } + + private CassandraMetricsRegistry.MetricName createName(String metricName) + { + return factory.createMetricName(metricName); + } +} diff --git a/src/java/org/apache/cassandra/journal/OnDiskIndex.java b/src/java/org/apache/cassandra/journal/OnDiskIndex.java new file mode 100644 index 0000000000..2bc40c6a5e --- /dev/null +++ b/src/java/org/apache/cassandra/journal/OnDiskIndex.java @@ -0,0 +1,295 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.MappedByteBuffer; +import java.nio.channels.FileChannel; +import java.nio.file.StandardOpenOption; +import java.util.Arrays; +import java.util.Map; +import java.util.NavigableMap; +import java.util.zip.CRC32; + +import javax.annotation.Nullable; + +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.FileUtils; +import org.apache.cassandra.utils.Crc; + +import static org.apache.cassandra.journal.Journal.validateCRC; +import static org.apache.cassandra.utils.FBUtilities.updateChecksumInt; + +/** + * An on-disk (memory-mapped) index for a completed flushed segment. + *

+ * TODO (expected): block-level CRC + */ +final class OnDiskIndex extends Index +{ + private static final int[] EMPTY = new int[0]; + + private static final int FILE_PREFIX_SIZE = 4 + 4; // count of entries, CRC + private static final int VALUE_SIZE = 4; // int offset + + private final int KEY_SIZE; + private final int ENTRY_SIZE; + + private final Descriptor descriptor; + + private final FileChannel channel; + private volatile MappedByteBuffer buffer; + private final int entryCount; + + private volatile K firstId, lastId; + + private OnDiskIndex( + Descriptor descriptor, KeySupport keySupport, FileChannel channel, MappedByteBuffer buffer, int entryCount) + { + super(keySupport); + + this.descriptor = descriptor; + this.channel = channel; + this.buffer = buffer; + this.entryCount = entryCount; + + KEY_SIZE = keySupport.serializedSize(descriptor.userVersion); + ENTRY_SIZE = KEY_SIZE + VALUE_SIZE; + } + + /** + * Open the index for reading, validate CRC + */ + @SuppressWarnings({ "resource", "RedundantSuppression" }) + static OnDiskIndex open(Descriptor descriptor, KeySupport keySupport) + { + File file = descriptor.fileFor(Component.INDEX); + FileChannel channel = null; + MappedByteBuffer buffer = null; + try + { + channel = FileChannel.open(file.toPath(), StandardOpenOption.READ); + buffer = channel.map(FileChannel.MapMode.READ_ONLY, 0, channel.size()); + + int entryCount = buffer.getInt(0); + OnDiskIndex index = new OnDiskIndex<>(descriptor, keySupport, channel, buffer, entryCount); + index.validate(); + index.init(); + return index; + } + catch (Throwable e) + { + FileUtils.clean(buffer); + FileUtils.closeQuietly(channel); + throw new JournalReadError(descriptor, file, e); + } + } + + private void init() + { + if (entryCount > 0) + { + firstId = keyAtIndex(0); + lastId = keyAtIndex(entryCount - 1); + } + } + + @Override + public void close() + { + try + { + FileUtils.clean(buffer); + buffer = null; + channel.close(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, Component.INDEX, e); + } + } + + void validate() throws IOException + { + CRC32 crc = Crc.crc32(); + + try (DataInputBuffer in = new DataInputBuffer(buffer, true)) + { + int entryCount = in.readInt(); + updateChecksumInt(crc, entryCount); + validateCRC(crc, in.readInt()); + + Crc.updateCrc32(crc, buffer, FILE_PREFIX_SIZE, FILE_PREFIX_SIZE + entryCount * ENTRY_SIZE); + in.skipBytesFully(entryCount * ENTRY_SIZE); + validateCRC(crc, in.readInt()); + + if (in.available() != 0) + throw new IOException("Trailing data encountered in segment index " + descriptor.fileFor(Component.INDEX)); + } + } + + static void write( + NavigableMap entries, KeySupport keySupport, DataOutputPlus out, int userVersion) throws IOException + { + CRC32 crc = Crc.crc32(); + + int size = entries.values() + .stream() + .mapToInt(offsets -> offsets.length) + .sum(); + out.writeInt(size); + updateChecksumInt(crc, size); + out.writeInt((int) crc.getValue()); + + for (Map.Entry entry : entries.entrySet()) + { + for (int offset : entry.getValue()) + { + K key = entry.getKey(); + keySupport.serialize(key, out, userVersion); + keySupport.updateChecksum(crc, key, userVersion); + + out.writeInt(offset); + updateChecksumInt(crc, offset); + } + } + + out.writeInt((int) crc.getValue()); + } + + @Override + @Nullable + public K firstId() + { + return firstId; + } + + @Override + @Nullable + public K lastId() + { + return lastId; + } + + @Override + public int[] lookUp(K id) + { + if (!mayContainId(id)) + return EMPTY; + + int keyIndex = binarySearch(id); + if (keyIndex < 0) + return EMPTY; + + int[] offsets = new int[] { offsetAtIndex(keyIndex) }; + + /* + * Duplicate entries are possible within one segment (but should be rare). + * Check and add entries before and after the found result (not guaranteed to be first). + */ + + for (int i = keyIndex - 1; i >= 0 && id.equals(keyAtIndex(i)); i--) + { + int length = offsets.length; + offsets = Arrays.copyOf(offsets, length + 1); + offsets[length] = offsetAtIndex(i); + } + + for (int i = keyIndex + 1; i < entryCount && id.equals(keyAtIndex(i)); i++) + { + int length = offsets.length; + offsets = Arrays.copyOf(offsets, length + 1); + offsets[length] = offsetAtIndex(i); + } + + Arrays.sort(offsets); + return offsets; + } + + @Override + public int lookUpFirst(K id) + { + if (!mayContainId(id)) + return -1; + + int keyIndex = binarySearch(id); + + /* + * Duplicate entries are possible within one segment (but should be rare). + * Check and add entries before until we find the first occurrence of key. + */ + for (int i = keyIndex - 1; i >= 0 && id.equals(keyAtIndex(i)); i--) + keyIndex = i; + + return keyIndex < 0 ? -1 : offsetAtIndex(keyIndex); + } + + private K keyAtIndex(int index) + { + return keySupport.deserialize(buffer, FILE_PREFIX_SIZE + index * ENTRY_SIZE, descriptor.userVersion); + } + + private int offsetAtIndex(int index) + { + return buffer.getInt(FILE_PREFIX_SIZE + index * ENTRY_SIZE + KEY_SIZE); + } + + /* + * This has been lifted from {@see IndexSummary}'s implementation, + * which itself was lifted from Harmony's Collections implementation. + */ + private int binarySearch(K key) + { + int low = 0, mid = entryCount, high = mid - 1, result = -1; + while (low <= high) + { + mid = (low + high) >> 1; + result = compareWithKeyAt(key, mid); + if (result > 0) + { + low = mid + 1; + } + else if (result == 0) + { + return mid; + } + else + { + high = mid - 1; + } + } + return -mid - (result < 0 ? 1 : 2); + } + + private int compareWithKeyAt(K key, int keyIndex) + { + int offset = FILE_PREFIX_SIZE + ENTRY_SIZE * keyIndex; + return keySupport.compareWithKeyAt(key, buffer, offset, descriptor.userVersion); + } + + static OnDiskIndex rebuildAndPersist(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + try (InMemoryIndex index = InMemoryIndex.rebuild(descriptor, keySupport, fsyncedLimit)) + { + index.persist(descriptor); + } + return open(descriptor, keySupport); + } +} diff --git a/src/java/org/apache/cassandra/journal/Params.java b/src/java/org/apache/cassandra/journal/Params.java new file mode 100644 index 0000000000..f462f450ac --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Params.java @@ -0,0 +1,56 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +public interface Params +{ + enum FlushMode { BATCH, GROUP, PERIODIC } + + enum FailurePolicy { STOP, STOP_JOURNAL, IGNORE, DIE } + + /** + * @return maximum segment size + */ + int segmentSize(); + + /** + * @return this journal's {@link FailurePolicy} + */ + FailurePolicy failurePolicy(); + + /** + * @return journal flush (sync) mode + */ + FlushMode flushMode(); + + /** + * @return milliseconds between journal flushes + */ + int flushPeriod(); + + /** + * @return milliseconds to block writes for while waiting for a slow disk flush to complete + * when in {@link FlushMode#PERIODIC} mode + */ + int periodicFlushLagBlock(); + + /** + * @return user provided version to use for key and value serialization + */ + int userVersion(); +} diff --git a/src/java/org/apache/cassandra/exceptions/ChecksumMismatchException.java b/src/java/org/apache/cassandra/journal/RecordConsumer.java similarity index 74% rename from src/java/org/apache/cassandra/exceptions/ChecksumMismatchException.java rename to src/java/org/apache/cassandra/journal/RecordConsumer.java index a76c46c782..2a1adb9fa4 100644 --- a/src/java/org/apache/cassandra/exceptions/ChecksumMismatchException.java +++ b/src/java/org/apache/cassandra/journal/RecordConsumer.java @@ -15,20 +15,14 @@ * See the License for the specific language governing permissions and * limitations under the License. */ +package org.apache.cassandra.journal; -package org.apache.cassandra.exceptions; +import java.nio.ByteBuffer; -import java.io.IOException; +import org.agrona.collections.IntHashSet; -public class ChecksumMismatchException extends IOException +@FunctionalInterface +public interface RecordConsumer { - public ChecksumMismatchException() - { - super(); - } - - public ChecksumMismatchException(String s) - { - super(s); - } + void accept(K key, ByteBuffer buffer, IntHashSet hosts, int userVersion); } diff --git a/src/java/org/apache/cassandra/journal/Segment.java b/src/java/org/apache/cassandra/journal/Segment.java new file mode 100644 index 0000000000..e63f6f2cad --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Segment.java @@ -0,0 +1,78 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.nio.ByteBuffer; + +import accord.utils.Invariants; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.utils.*; +import org.apache.cassandra.utils.concurrent.RefCounted; + +abstract class Segment implements Closeable, RefCounted> +{ + final File file; + final Descriptor descriptor; + final SyncedOffsets syncedOffsets; + final Metadata metadata; + final KeySupport keySupport; + + ByteBuffer buffer; + + Segment(Descriptor descriptor, SyncedOffsets syncedOffsets, Metadata metadata, KeySupport keySupport) + { + this.file = descriptor.fileFor(Component.DATA); + this.descriptor = descriptor; + this.syncedOffsets = syncedOffsets; + this.metadata = metadata; + this.keySupport = keySupport; + } + + abstract Index index(); + + /* + * Reading entries (by id, by offset, iterate) + */ + + boolean read(K id, RecordConsumer consumer) + { + int offset = index().lookUpFirst(id); + if (offset == -1) + return false; + + EntrySerializer.EntryHolder into = new EntrySerializer.EntryHolder<>(); + if (read(offset, into)) + { + Invariants.checkState(id.equals(into.key), "Index for %s read incorrect key: expected %s but read %s", descriptor, id, into.key); + consumer.accept(id, into.value, into.hosts, descriptor.userVersion); + return true; + } + return false; + } + + boolean read(K id, EntrySerializer.EntryHolder into) + { + int offset = index().lookUpFirst(id); + if (offset == -1 || !read(offset, into)) + return false; + Invariants.checkState(id.equals(into.key), "Index for %s read incorrect key: expected %s but read %s", descriptor, id, into.key); + return true; + } + + abstract boolean read(int offset, EntrySerializer.EntryHolder into); +} diff --git a/src/java/org/apache/cassandra/journal/SegmentWriter.java b/src/java/org/apache/cassandra/journal/SegmentWriter.java new file mode 100644 index 0000000000..852e955b21 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/SegmentWriter.java @@ -0,0 +1,115 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.Closeable; +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.Set; + +import com.google.common.primitives.Ints; + +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.FileOutputStreamPlus; +import org.apache.cassandra.io.util.TrackedDataOutputPlus; + +final class SegmentWriter implements Closeable +{ + private final Descriptor descriptor; + private final KeySupport keySupport; + + private final InMemoryIndex index; + private final Metadata metadata; + + private final File file; + private FileOutputStreamPlus untrackedOut; + private TrackedDataOutputPlus trackedOut; + + private SegmentWriter(Descriptor descriptor, KeySupport keySupport) + { + this.descriptor = descriptor; + this.keySupport = keySupport; + + index = InMemoryIndex.create(keySupport); + metadata = Metadata.create(); + + file = descriptor.fileFor(Component.DATA); + try + { + untrackedOut = new FileOutputStreamPlus(file); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, file, e); + } + trackedOut = TrackedDataOutputPlus.wrap(untrackedOut); + } + + static SegmentWriter create(Descriptor descriptor, KeySupport keySupport) + { + return new SegmentWriter<>(descriptor, keySupport); + } + + int write(K key, ByteBuffer record, Set hosts) + { + int position = position(); + try + { + index.update(key, position); + metadata.update(hosts); + + EntrySerializer.write(key, record, hosts, keySupport, trackedOut, descriptor.userVersion); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, file, e); + } + return position; + } + + int position() + { + return Ints.checkedCast(trackedOut.position()); + } + + @Override + public void close() + { + try + { + untrackedOut.flush(); + untrackedOut.sync(); + untrackedOut.close(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, file, e); + } + + try (SyncedOffsets syncedOffsets = SyncedOffsets.active(descriptor, true)) + { + syncedOffsets.mark(position()); + } + + index.persist(descriptor); + metadata.persist(descriptor); + + untrackedOut = null; + trackedOut = null; + } +} diff --git a/src/java/org/apache/cassandra/journal/Segments.java b/src/java/org/apache/cassandra/journal/Segments.java new file mode 100644 index 0000000000..8873ea7b3d --- /dev/null +++ b/src/java/org/apache/cassandra/journal/Segments.java @@ -0,0 +1,200 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.util.ArrayList; +import java.util.Collection; +import java.util.Collections; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import com.google.common.collect.Iterables; +import com.google.common.collect.Maps; + +import accord.utils.Invariants; +import org.apache.cassandra.utils.concurrent.Refs; + +import static java.util.Collections.emptyList; +import static java.util.Collections.emptyMap; + +/** + * Consistent, immutable view of active + static segments + *

+ * TODO: an interval/range structure for StaticSegment lookup based on min/max key bounds + */ +class Segments +{ + // active segments, containing unflushed data; the tail of this queue is the one we allocate writes from + private final List> activeSegments; + + // finalised segments, no longer written to + private final Map> staticSegments; + + // cached Iterable of concatenated active and static segments + private final Iterable> allSegments; + + Segments(List> activeSegments, Map> staticSegments) + { + this.activeSegments = activeSegments; + this.staticSegments = staticSegments; + this.allSegments = Iterables.concat(onlyActive(), onlyStatic()); + } + + static Segments ofStatic(Collection> segments) + { + HashMap> staticSegments = + Maps.newHashMapWithExpectedSize(segments.size()); + for (StaticSegment segment : segments) + staticSegments.put(segment.descriptor, segment); + return new Segments<>(new ArrayList<>(), staticSegments); + } + + static Segments none() + { + return new Segments<>(Collections.emptyList(), Collections.emptyMap()); + } + + Segments withNewActiveSegment(ActiveSegment activeSegment) + { + ArrayList> newActiveSegments = + new ArrayList<>(activeSegments.size() + 1); + newActiveSegments.addAll(activeSegments); + newActiveSegments.add(activeSegment); + return new Segments<>(newActiveSegments, staticSegments); + } + + Segments withCompletedSegment(ActiveSegment activeSegment, StaticSegment staticSegment) + { + Invariants.checkArgument(activeSegment.descriptor.equals(staticSegment.descriptor)); + + ArrayList> newActiveSegments = + new ArrayList<>(activeSegments.size() - 1); + for (ActiveSegment segment : activeSegments) + if (segment != activeSegment) + newActiveSegments.add(segment); + Invariants.checkState(newActiveSegments.size() == activeSegments.size() - 1); + + HashMap> newStaticSegments = + Maps.newHashMapWithExpectedSize(staticSegments.size() + 1); + newStaticSegments.putAll(staticSegments); + if (newStaticSegments.put(staticSegment.descriptor, staticSegment) != null) + throw new IllegalStateException(); + + return new Segments<>(newActiveSegments, newStaticSegments); + } + + Segments withCompactedSegment(StaticSegment oldSegment, StaticSegment newSegment) + { + Invariants.checkArgument(oldSegment.descriptor.timestamp == newSegment.descriptor.timestamp); + Invariants.checkArgument(oldSegment.descriptor.generation < newSegment.descriptor.generation); + + HashMap> newStaticSegments = new HashMap<>(staticSegments); + if (!newStaticSegments.remove(oldSegment.descriptor, oldSegment)) + throw new IllegalStateException(); + if (null != newStaticSegments.put(newSegment.descriptor, newSegment)) + throw new IllegalStateException(); + + return new Segments<>(activeSegments, newStaticSegments); + } + + Segments withoutInvalidatedSegment(StaticSegment staticSegment) + { + HashMap> newStaticSegments = new HashMap<>(staticSegments); + if (!newStaticSegments.remove(staticSegment.descriptor, staticSegment)) + throw new IllegalStateException(); + return new Segments<>(activeSegments, newStaticSegments); + } + + Iterable> all() + { + return allSegments; + } + + Collection> onlyActive() + { + return activeSegments; + } + + Collection> onlyStatic() + { + return staticSegments.values(); + } + + /** + * Select segments that could potentially have an entry with the specified id and + * attempt to grab references to them all. + * + * @return a subset of segments with references to them, or {@code null} if failed to grab the refs + */ + @SuppressWarnings("resource") + ReferencedSegments selectAndReference(K id) + { + List> selectedActive = null; + for (ActiveSegment segment : onlyActive()) + { + if (segment.index.mayContainId(id)) + { + if (null == selectedActive) + selectedActive = new ArrayList<>(); + selectedActive.add(segment); + } + } + if (null == selectedActive) selectedActive = emptyList(); + + Map> selectedStatic = null; + for (StaticSegment segment : onlyStatic()) + { + if (segment.index().mayContainId(id)) + { + if (null == selectedStatic) + selectedStatic = new HashMap<>(); + selectedStatic.put(segment.descriptor, segment); + } + } + if (null == selectedStatic) selectedStatic = emptyMap(); + + Refs> refs = null; + if (!selectedActive.isEmpty() || !selectedStatic.isEmpty()) + { + refs = Refs.tryRef(Iterables.concat(selectedActive, selectedStatic.values())); + if (null == refs) + return null; + } + return new ReferencedSegments<>(selectedActive, selectedStatic, refs); + } + + static class ReferencedSegments extends Segments implements AutoCloseable + { + public final Refs> refs; + + ReferencedSegments( + List> activeSegments, Map> staticSegments, Refs> refs) + { + super(activeSegments, staticSegments); + this.refs = refs; + } + + @Override + public void close() + { + if (null != refs) + refs.release(); + } + } +} diff --git a/src/java/org/apache/cassandra/journal/StaticSegment.java b/src/java/org/apache/cassandra/journal/StaticSegment.java new file mode 100644 index 0000000000..52b8d954e1 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/StaticSegment.java @@ -0,0 +1,349 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.nio.MappedByteBuffer; +import java.nio.channels.FileChannel; +import java.nio.file.NoSuchFileException; +import java.nio.file.StandardOpenOption; +import java.util.*; + +import org.agrona.collections.IntHashSet; +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.FileUtils; +import org.apache.cassandra.utils.Closeable; +import org.apache.cassandra.utils.concurrent.Ref; + +/** + * An immutable data segment that is no longer written to. + *

+ * Can be compacted with input from {@code PersistedInvalidations} into a new smaller segment, + * with invalidated entries removed. + */ +final class StaticSegment extends Segment +{ + final FileChannel channel; + + private final Ref> selfRef; + + private final OnDiskIndex index; + + private StaticSegment(Descriptor descriptor, + FileChannel channel, + MappedByteBuffer buffer, + SyncedOffsets syncedOffsets, + OnDiskIndex index, + Metadata metadata, + KeySupport keySupport) + { + super(descriptor, syncedOffsets, metadata, keySupport); + this.index = index; + + this.channel = channel; + this.buffer = buffer; + + selfRef = new Ref<>(this, new Tidier<>(descriptor, channel, buffer, index)); + } + + /** + * Loads all segments matching the supplied desctiptors + * + * @param descriptors descriptors of the segments to load + * @return list of the loaded segments + */ + static List> open(Collection descriptors, KeySupport keySupport) + { + List> segments = new ArrayList<>(descriptors.size()); + for (Descriptor descriptor : descriptors) + segments.add(open(descriptor, keySupport)); + return segments; + } + + /** + * Load the segment corresponding to the provided desrciptor + * + * @param descriptor descriptor of the segment to load + * @return the loaded segment + */ + @SuppressWarnings({ "resource", "RedundantSuppression" }) + static StaticSegment open(Descriptor descriptor, KeySupport keySupport) + { + if (!Component.DATA.existsFor(descriptor)) + throw new IllegalArgumentException("Data file for segment " + descriptor + " doesn't exist"); + + SyncedOffsets syncedOffsets = Component.SYNCED_OFFSETS.existsFor(descriptor) + ? SyncedOffsets.load(descriptor) + : SyncedOffsets.absent(); + + Metadata metadata = Component.METADATA.existsFor(descriptor) + ? Metadata.load(descriptor) + : Metadata.rebuildAndPersist(descriptor, keySupport, syncedOffsets.syncedOffset()); + + OnDiskIndex index = Component.INDEX.existsFor(descriptor) + ? OnDiskIndex.open(descriptor, keySupport) + : OnDiskIndex.rebuildAndPersist(descriptor, keySupport, syncedOffsets.syncedOffset()); + + try + { + return internalOpen(descriptor, syncedOffsets, index, metadata, keySupport); + } + catch (IOException e) + { + throw new JournalReadError(descriptor, Component.DATA, e); + } + } + + @SuppressWarnings("resource") + private static StaticSegment internalOpen( + Descriptor descriptor, SyncedOffsets syncedOffsets, OnDiskIndex index, Metadata metadata, KeySupport keySupport) + throws IOException + { + File file = descriptor.fileFor(Component.DATA); + FileChannel channel = FileChannel.open(file.toPath(), StandardOpenOption.READ); + MappedByteBuffer buffer = channel.map(FileChannel.MapMode.READ_ONLY, 0, channel.size()); + return new StaticSegment<>(descriptor, channel, buffer, syncedOffsets, index, metadata, keySupport); + } + + @Override + public void close() + { + selfRef.release(); + } + + @Override + public Ref> tryRef() + { + return selfRef.tryRef(); + } + + @Override + public Ref> ref() + { + return selfRef.ref(); + } + + private static final class Tidier implements Tidy + { + private final Descriptor descriptor; + private final FileChannel channel; + private final ByteBuffer buffer; + private final Index index; + + Tidier(Descriptor descriptor, FileChannel channel, ByteBuffer buffer, Index index) + { + this.descriptor = descriptor; + this.channel = channel; + this.buffer = buffer; + this.index = index; + } + + @Override + public void tidy() + { + FileUtils.clean(buffer); + FileUtils.closeQuietly(channel); + index.close(); + } + + @Override + public String name() + { + return descriptor.toString(); + } + } + + @Override + OnDiskIndex index() + { + return index; + } + + /** + * Read the entry and specified offset into the entry holder. + * Expects the record to have been written at this offset, but potentially not flushed and lost. + */ + @Override + boolean read(int offset, EntrySerializer.EntryHolder into) + { + ByteBuffer duplicate = (ByteBuffer) buffer.duplicate().position(offset); + try (DataInputBuffer in = new DataInputBuffer(duplicate, false)) + { + return EntrySerializer.tryRead(into, keySupport, duplicate, in, syncedOffsets.syncedOffset(), descriptor.userVersion); + } + catch (IOException e) + { + throw new JournalReadError(descriptor, file, e); + } + } + + /** + * Iterate over and invoke the supplied callback on every record. + */ + void forEachRecord(RecordConsumer consumer) + { + try (SequentialReader reader = reader(descriptor, keySupport, syncedOffsets.syncedOffset())) + { + while (reader.advance()) + { + consumer.accept(reader.id(), reader.record(), reader.hosts(), descriptor.userVersion); + } + } + } + + /* + * Sequential reading (replay and components rebuild) + */ + + static SequentialReader reader(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + return SequentialReader.open(descriptor, keySupport, fsyncedLimit); + } + + /** + * A sequential data segment reader to use for journal replay and rebuilding + * missing auxilirary components (index and metadata). + *

+ * Unexpected EOF and CRC mismatches in synced portions of segments are treated + * strictly, throwing {@link JournalReadError}. Errors encountered in unsynced portions + * of segments are treated as segment EOF. + */ + static final class SequentialReader implements Closeable + { + private final Descriptor descriptor; + private final KeySupport keySupport; + private final int fsyncedLimit; // exclusive + + private final File file; + private final FileChannel channel; + private final MappedByteBuffer buffer; + private final DataInputBuffer in; + + private int offset = -1; + private final EntrySerializer.EntryHolder holder = new EntrySerializer.EntryHolder<>(); + private State state = State.RESET; + + static SequentialReader open(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + return new SequentialReader<>(descriptor, keySupport, fsyncedLimit); + } + + SequentialReader(Descriptor descriptor, KeySupport keySupport, int fsyncedLimit) + { + this.descriptor = descriptor; + this.keySupport = keySupport; + this.fsyncedLimit = fsyncedLimit; + + file = descriptor.fileFor(Component.DATA); + try + { + channel = file.newReadChannel(); + buffer = channel.map(FileChannel.MapMode.READ_ONLY, 0, channel.size()); + } + catch (NoSuchFileException e) + { + throw new IllegalArgumentException("Data file for segment " + descriptor + " doesn't exist"); + } + catch (IOException e) + { + throw new JournalReadError(descriptor, file, e); + } + in = new DataInputBuffer(buffer, false); + } + + @Override + public void close() + { + FileUtils.closeQuietly(channel); + FileUtils.clean(buffer); + } + + int offset() + { + ensureHasAdvanced(); + return offset; + } + + K id() + { + ensureHasAdvanced(); + return holder.key; + } + + IntHashSet hosts() + { + ensureHasAdvanced(); + return holder.hosts; + } + + ByteBuffer record() + { + ensureHasAdvanced(); + return holder.value; + } + + private void ensureHasAdvanced() + { + if (state != State.ADVANCED) + throw new IllegalStateException("Must call advance() before accessing entry content"); + } + + boolean advance() + { + if (state == State.EOF) + return false; + + reset(); + return buffer.hasRemaining() ? doAdvance() : eof(); + } + + private boolean doAdvance() + { + offset = buffer.position(); + try + { + if (!EntrySerializer.tryRead(holder, keySupport, buffer, in, fsyncedLimit, descriptor.userVersion)) + return eof(); + } + catch (IOException e) + { + throw new JournalReadError(descriptor, file, e); + } + + state = State.ADVANCED; + return true; + } + + private void reset() + { + offset = -1; + holder.clear(); + state = State.RESET; + } + + private boolean eof() + { + state = State.EOF; + return false; + } + + enum State { RESET, ADVANCED, EOF } + } +} \ No newline at end of file diff --git a/src/java/org/apache/cassandra/journal/SyncedOffsets.java b/src/java/org/apache/cassandra/journal/SyncedOffsets.java new file mode 100644 index 0000000000..bee302d6d8 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/SyncedOffsets.java @@ -0,0 +1,243 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.io.UncheckedIOException; +import java.nio.file.NoSuchFileException; +import java.util.zip.CRC32; + +import org.apache.cassandra.io.FSWriteError; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.FileOutputStreamPlus; +import org.apache.cassandra.io.util.RandomAccessReader; +import org.apache.cassandra.utils.Closeable; +import org.apache.cassandra.utils.Crc; + +import static org.apache.cassandra.utils.FBUtilities.updateChecksumInt; + +/** + * Keeps track of fsynced limits of a data file. Enables us to treat invalid + * records that are known to have been fsynced to disk differently from those + * that aren't. + *

+ * On disk representation is a sequence of 2-int tuples of {synced offset, CRC32(synced offset)} + */ +interface SyncedOffsets extends Closeable +{ + /** + * @return furthest known synced offset + */ + int syncedOffset(); + + /** + * Record an offset as synced to disk. + * + * @param offset the offset into datafile, up to which contents have been fsynced (exclusive) + */ + void mark(int offset); + + @Override + default void close() + { + } + + /** + * @return a disk-backed synced offset tracker for a new {@link ActiveSegment} + */ + static Active active(Descriptor descriptor, boolean syncOnMark) + { + return new Active(descriptor, syncOnMark); + } + + /** + * Load an existing log of synced offsets from disk into an immutable instance. + */ + static Static load(Descriptor descriptor) + { + return Static.load(descriptor); + } + + /** + * @return a placeholder instance in case this component is missing + */ + static Absent absent() + { + return Absent.INSTANCE; + } + + /** + * Single-threaded, file-based list of synced offsets. + */ + final class Active implements SyncedOffsets + { + private final Descriptor descriptor; + private final boolean syncOnMark; + + private final FileOutputStreamPlus output; + private volatile int syncedOffset; + + private Active(Descriptor descriptor, boolean syncOnMark) + { + this.descriptor = descriptor; + this.syncOnMark = syncOnMark; + + File file = descriptor.fileFor(Component.SYNCED_OFFSETS); + if (file.exists()) + throw new IllegalArgumentException("Synced offsets file " + file + " already exists"); + + try + { + output = file.newOutputStream(File.WriteMode.OVERWRITE); + } + catch (UncheckedIOException | FSWriteError e) + { + // extract original cause and throw as JournalWriteError + throw new JournalWriteError(descriptor, file, e.getCause()); + } + catch (NoSuchFileException e) + { + throw new AssertionError(); // unreachable + } + } + + @Override + public int syncedOffset() + { + return syncedOffset; + } + + @Override + public void mark(int offset) + { + if (offset < syncedOffset) + throw new IllegalArgumentException("offset " + offset + " is smaller than previous mark " + offset); + + CRC32 crc = Crc.crc32(); + updateChecksumInt(crc, offset); + + try + { + output.writeInt(offset); + output.writeInt((int) crc.getValue()); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, Component.SYNCED_OFFSETS, e); + } + + syncedOffset = offset; + if (syncOnMark) sync(); + } + + private void sync() + { + try + { + output.sync(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, Component.SYNCED_OFFSETS, e); + } + } + + @Override + public void close() + { + if (!syncOnMark) sync(); + + try + { + output.close(); + } + catch (IOException e) + { + throw new JournalWriteError(descriptor, Component.SYNCED_OFFSETS, e); + } + } + } + + final class Static implements SyncedOffsets + { + private final int syncedOffset; + + static Static load(Descriptor descriptor) + { + File file = descriptor.fileFor(Component.SYNCED_OFFSETS); + if (!file.exists()) + throw new IllegalArgumentException("Synced offsets file " + file + " doesn't exist"); + + int syncedOffset = 0; + try (RandomAccessReader reader = RandomAccessReader.open(file)) + { + CRC32 crc = Crc.crc32(); + while (reader.bytesRemaining() >= 8) + { + int offset = reader.readInt(); + updateChecksumInt(crc, offset); + int readCrc = reader.readInt(); + if (readCrc != (int) crc.getValue()) + break; + syncedOffset = offset; + Crc.initialize(crc); + } + } + catch (Throwable t) + { + throw new JournalReadError(descriptor, file, t); + } + + return new Static(syncedOffset); + } + + Static(int offset) + { + this.syncedOffset = offset; + } + + @Override + public int syncedOffset() + { + return syncedOffset; + } + + @Override + public void mark(int offset) + { + throw new UnsupportedOperationException(); + } + } + + final class Absent implements SyncedOffsets + { + static final Absent INSTANCE = new Absent(); + + @Override + public int syncedOffset() + { + return 0; + } + + @Override + public void mark(int offset) + { + throw new UnsupportedOperationException(); + } + } +} diff --git a/src/java/org/apache/cassandra/journal/ValueSerializer.java b/src/java/org/apache/cassandra/journal/ValueSerializer.java new file mode 100644 index 0000000000..f004c3a37a --- /dev/null +++ b/src/java/org/apache/cassandra/journal/ValueSerializer.java @@ -0,0 +1,36 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; + +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; + +public interface ValueSerializer +{ + int serializedSize(V value, int userVersion); + + void serialize(V value, DataOutputPlus out, int userVersion) throws IOException; + + /** + * Deserialize the value given the key is known. Allows to avoid serializing + * redundant information in values, if it can be derived from keys. + */ + V deserialize(K key, DataInputPlus in, int userVersion) throws IOException; +} diff --git a/src/java/org/apache/cassandra/journal/package-info.java b/src/java/org/apache/cassandra/journal/package-info.java new file mode 100644 index 0000000000..5ae20b9274 --- /dev/null +++ b/src/java/org/apache/cassandra/journal/package-info.java @@ -0,0 +1,22 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/** + * TODO + */ +package org.apache.cassandra.journal; diff --git a/src/java/org/apache/cassandra/net/AbstractMessageHandler.java b/src/java/org/apache/cassandra/net/AbstractMessageHandler.java index 5b5b8b7f1a..895b277ad5 100644 --- a/src/java/org/apache/cassandra/net/AbstractMessageHandler.java +++ b/src/java/org/apache/cassandra/net/AbstractMessageHandler.java @@ -33,6 +33,7 @@ import io.netty.channel.Channel; import io.netty.channel.ChannelHandlerContext; import io.netty.channel.ChannelInboundHandlerAdapter; import io.netty.channel.EventLoop; +import org.apache.cassandra.concurrent.ManyToOneConcurrentLinkedQueue; import org.apache.cassandra.metrics.ClientMetrics; import org.apache.cassandra.net.FrameDecoder.CorruptFrame; import org.apache.cassandra.net.FrameDecoder.Frame; @@ -43,7 +44,7 @@ import org.apache.cassandra.net.ResourceLimits.Limit; import static java.lang.Math.max; import static java.lang.Math.min; -import static org.apache.cassandra.net.Crc.InvalidCrc; +import static org.apache.cassandra.utils.Crc.InvalidCrc; import static org.apache.cassandra.utils.MonotonicClock.Global.approxTime; /** diff --git a/src/java/org/apache/cassandra/net/FrameDecoderCrc.java b/src/java/org/apache/cassandra/net/FrameDecoderCrc.java index 2a54f5f663..86f4442453 100644 --- a/src/java/org/apache/cassandra/net/FrameDecoderCrc.java +++ b/src/java/org/apache/cassandra/net/FrameDecoderCrc.java @@ -24,7 +24,7 @@ import java.util.zip.CRC32; import io.netty.channel.ChannelPipeline; -import static org.apache.cassandra.net.Crc.*; +import static org.apache.cassandra.utils.Crc.*; /** * Framing format that protects integrity of data in movement with CRCs (of both header and payload). diff --git a/src/java/org/apache/cassandra/net/FrameDecoderLZ4.java b/src/java/org/apache/cassandra/net/FrameDecoderLZ4.java index 9cc100586e..7dafb7fedf 100644 --- a/src/java/org/apache/cassandra/net/FrameDecoderLZ4.java +++ b/src/java/org/apache/cassandra/net/FrameDecoderLZ4.java @@ -26,7 +26,7 @@ import io.netty.channel.ChannelPipeline; import net.jpountz.lz4.LZ4Factory; import net.jpountz.lz4.LZ4SafeDecompressor; -import static org.apache.cassandra.net.Crc.*; +import static org.apache.cassandra.utils.Crc.*; /** * Framing format that compresses payloads with LZ4, and protects integrity of data in movement with CRCs diff --git a/src/java/org/apache/cassandra/net/FrameEncoderCrc.java b/src/java/org/apache/cassandra/net/FrameEncoderCrc.java index 3646248165..75b84aa14f 100644 --- a/src/java/org/apache/cassandra/net/FrameEncoderCrc.java +++ b/src/java/org/apache/cassandra/net/FrameEncoderCrc.java @@ -24,7 +24,7 @@ import java.util.zip.CRC32; import io.netty.buffer.ByteBuf; import io.netty.channel.ChannelHandler; -import static org.apache.cassandra.net.Crc.*; +import static org.apache.cassandra.utils.Crc.*; /** * Please see {@link FrameDecoderCrc} for description of the framing produced by this encoder. diff --git a/src/java/org/apache/cassandra/net/FrameEncoderLZ4.java b/src/java/org/apache/cassandra/net/FrameEncoderLZ4.java index 75f15c726b..423e377a27 100644 --- a/src/java/org/apache/cassandra/net/FrameEncoderLZ4.java +++ b/src/java/org/apache/cassandra/net/FrameEncoderLZ4.java @@ -28,7 +28,7 @@ import net.jpountz.lz4.LZ4Factory; import org.apache.cassandra.io.compress.BufferType; import org.apache.cassandra.utils.ByteBufferUtil; -import static org.apache.cassandra.net.Crc.*; +import static org.apache.cassandra.utils.Crc.*; /** * Please see {@link FrameDecoderLZ4} for description of the framing produced by this encoder. diff --git a/src/java/org/apache/cassandra/net/HandshakeProtocol.java b/src/java/org/apache/cassandra/net/HandshakeProtocol.java index 3217aeae8a..085c8b966c 100644 --- a/src/java/org/apache/cassandra/net/HandshakeProtocol.java +++ b/src/java/org/apache/cassandra/net/HandshakeProtocol.java @@ -35,7 +35,7 @@ import static java.util.concurrent.TimeUnit.MILLISECONDS; import static org.apache.cassandra.locator.InetAddressAndPort.Serializer.inetAddressAndPortSerializer; import static org.apache.cassandra.net.MessagingService.VERSION_40; import static org.apache.cassandra.net.Message.validateLegacyProtocolMagic; -import static org.apache.cassandra.net.Crc.*; +import static org.apache.cassandra.utils.Crc.*; import static org.apache.cassandra.net.OutboundConnectionSettings.*; /** diff --git a/src/java/org/apache/cassandra/net/InboundMessageHandler.java b/src/java/org/apache/cassandra/net/InboundMessageHandler.java index edee108bb0..6c697d4848 100644 --- a/src/java/org/apache/cassandra/net/InboundMessageHandler.java +++ b/src/java/org/apache/cassandra/net/InboundMessageHandler.java @@ -40,6 +40,7 @@ import org.apache.cassandra.net.ResourceLimits.Limit; import org.apache.cassandra.tcm.ClusterMetadataService; import org.apache.cassandra.tracing.TraceState; import org.apache.cassandra.tracing.Tracing; +import org.apache.cassandra.utils.Crc; import org.apache.cassandra.utils.JVMStabilityInspector; import org.apache.cassandra.utils.NoSpamLogger; diff --git a/src/java/org/apache/cassandra/net/InboundMessageHandlers.java b/src/java/org/apache/cassandra/net/InboundMessageHandlers.java index c7b946350d..f1b98bdbbe 100644 --- a/src/java/org/apache/cassandra/net/InboundMessageHandlers.java +++ b/src/java/org/apache/cassandra/net/InboundMessageHandlers.java @@ -30,6 +30,7 @@ import io.netty.channel.Channel; import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.metrics.InternodeInboundMetrics; import org.apache.cassandra.net.Message.Header; +import org.apache.cassandra.utils.Crc; import static java.util.concurrent.TimeUnit.NANOSECONDS; import static org.apache.cassandra.utils.MonotonicClock.Global.approxTime; diff --git a/src/java/org/apache/cassandra/net/OutboundMessageQueue.java b/src/java/org/apache/cassandra/net/OutboundMessageQueue.java index 8280055e68..5098f32bc9 100644 --- a/src/java/org/apache/cassandra/net/OutboundMessageQueue.java +++ b/src/java/org/apache/cassandra/net/OutboundMessageQueue.java @@ -27,6 +27,7 @@ import java.util.function.Consumer; import com.google.common.annotations.VisibleForTesting; +import org.apache.cassandra.concurrent.ManyToOneConcurrentLinkedQueue; import org.apache.cassandra.utils.concurrent.CountDownLatch; import org.slf4j.Logger; import org.slf4j.LoggerFactory; diff --git a/src/java/org/apache/cassandra/net/SocketFactory.java b/src/java/org/apache/cassandra/net/SocketFactory.java index 78256267ac..bb4013dee5 100644 --- a/src/java/org/apache/cassandra/net/SocketFactory.java +++ b/src/java/org/apache/cassandra/net/SocketFactory.java @@ -60,6 +60,7 @@ import io.netty.util.concurrent.RejectedExecutionHandlers; import io.netty.util.concurrent.ThreadPerTaskExecutor; import io.netty.util.internal.logging.InternalLoggerFactory; import io.netty.util.internal.logging.Slf4JLoggerFactory; +import org.apache.cassandra.concurrent.ManyToOneConcurrentLinkedQueue; import org.apache.cassandra.concurrent.NamedThreadFactory; import org.apache.cassandra.config.EncryptionOptions; import org.apache.cassandra.locator.InetAddressAndPort; diff --git a/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java b/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java index 45208e5c34..235cc1be57 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java @@ -15,7 +15,6 @@ * See the License for the specific language governing permissions and * limitations under the License. */ - package org.apache.cassandra.service.accord; import accord.api.Agent; @@ -23,20 +22,93 @@ import accord.api.DataStore; import accord.api.ProgressLog; import accord.local.CommandStores; import accord.local.NodeTimeService; +import accord.local.PreLoadContext; +import accord.local.SafeCommandStore; import accord.local.ShardDistributor; +import accord.primitives.Routables; import accord.topology.Topology; +import accord.utils.MapReduceConsume; import accord.utils.RandomSource; +import org.apache.cassandra.concurrent.ImmediateExecutor; +import org.apache.cassandra.journal.AsyncWriteCallback; public class AccordCommandStores extends CommandStores { - private long cacheSize; + private final AccordJournal journal; + AccordCommandStores(NodeTimeService time, Agent agent, DataStore store, RandomSource random, - ShardDistributor shardDistributor, ProgressLog.Factory progressLogFactory) + ShardDistributor shardDistributor, ProgressLog.Factory progressLogFactory, AccordJournal journal) { super(time, agent, store, random, shardDistributor, progressLogFactory, AccordCommandStore::new); + this.journal = journal; setCacheSize(maxCacheSize()); } + static Factory factory(AccordJournal journal) + { + return (time, agent, store, random, shardDistributor, progressLogFactory) -> + new AccordCommandStores(time, agent, store, random, shardDistributor, progressLogFactory, journal); + } + + @Override + public synchronized void shutdown() + { + super.shutdown(); + journal.shutdown(); + //TODO shutdown isn't useful by itself, we need a way to "wait" as well. Should be AutoCloseable or offer awaitTermination as well (think Shutdownable interface) + } + + @Override + protected void mapReduceConsume( + PreLoadContext context, + Routables keys, + long minEpoch, + long maxEpoch, + MapReduceConsume mapReduceConsume) + { + // append PreAccept, Accept, Commit, and Apply messages durably to AccordJournal before processing + if (journal.mustMakeDurable(context)) + mapReduceConsumeDurable(context, keys, minEpoch, maxEpoch, mapReduceConsume); + else + super.mapReduceConsume(context, keys, minEpoch, maxEpoch, mapReduceConsume); + } + + private void mapReduceConsumeDurable( + PreLoadContext context, + Routables keys, + long minEpoch, + long maxEpoch, + MapReduceConsume mapReduceConsume) + { + journal.append(context, ImmediateExecutor.INSTANCE, new AsyncWriteCallback() + { + @Override + public void run() + { + // TODO (performance, expected): do not retain references to messages beyond a certain total + // cache threshold; in case of flush lagging behind, read the messages from journal and + // deserialize instead before processing, to prevent memory pressure buildup from messages + // pending flush to disk. + AccordCommandStores.super.mapReduceConsume(context, keys, minEpoch, maxEpoch, mapReduceConsume); + } + + @Override + public void onFailure(Throwable error) + { + mapReduceConsume.accept(null, error); + } + }); + } + + @Override + public synchronized void updateTopology(Topology newTopology) + { + super.updateTopology(newTopology); + refreshCacheSizes(); + } + + private long cacheSize; + synchronized void setCacheSize(long bytes) { cacheSize = bytes; @@ -56,18 +128,4 @@ public class AccordCommandStores extends CommandStores { return 5 << 20; // TODO (required): make configurable } - - @Override - public synchronized void updateTopology(Topology newTopology) - { - super.updateTopology(newTopology); - refreshCacheSizes(); - } - - @Override - public synchronized void shutdown() - { - super.shutdown(); - //TODO shutdown isn't useful by itself, we need a way to "wait" as well. Should be AutoCloseable or offer awaitTermination as well (think Shutdownable interface) - } } diff --git a/src/java/org/apache/cassandra/service/accord/AccordJournal.java b/src/java/org/apache/cassandra/service/accord/AccordJournal.java new file mode 100644 index 0000000000..382ab1e659 --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/AccordJournal.java @@ -0,0 +1,499 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.service.accord; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.*; +import java.util.concurrent.Executor; +import java.util.zip.Checksum; + +import com.google.common.annotations.VisibleForTesting; +import com.google.common.primitives.Ints; + +import accord.local.Node.Id; +import accord.local.PreLoadContext; +import accord.messages.Accept; +import accord.messages.Apply; +import accord.messages.Commit; +import accord.messages.MessageType; +import accord.messages.PreAccept; +import accord.messages.TxnRequest; +import accord.primitives.*; +import accord.utils.Invariants; +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.io.IVersionedSerializer; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.journal.AsyncWriteCallback; +import org.apache.cassandra.journal.Journal; +import org.apache.cassandra.journal.KeySupport; +import org.apache.cassandra.journal.Params; +import org.apache.cassandra.journal.ValueSerializer; +import org.apache.cassandra.net.MessagingService; +import org.apache.cassandra.service.accord.serializers.AcceptSerializers; +import org.apache.cassandra.service.accord.serializers.ApplySerializers; +import org.apache.cassandra.service.accord.serializers.CommitSerializers; +import org.apache.cassandra.service.accord.serializers.EnumSerializer; +import org.apache.cassandra.service.accord.serializers.PreacceptSerializers; + +import static org.apache.cassandra.db.TypeSizes.BYTE_SIZE; +import static org.apache.cassandra.db.TypeSizes.INT_SIZE; +import static org.apache.cassandra.db.TypeSizes.LONG_SIZE; +import static org.apache.cassandra.utils.FBUtilities.updateChecksumInt; +import static org.apache.cassandra.utils.FBUtilities.updateChecksumLong; + +public class AccordJournal +{ + private static final Set SENTINEL_HOSTS = Collections.singleton(0); + + static final Params PARAMS = new Params() + { + @Override + public int segmentSize() + { + return 32 << 20; + } + + @Override + public FailurePolicy failurePolicy() + { + return FailurePolicy.STOP; + } + + @Override + public FlushMode flushMode() + { + return FlushMode.GROUP; + } + + @Override + public int flushPeriod() + { + return 1000; + } + + @Override + public int periodicFlushLagBlock() + { + return 1500; + } + + @Override + public int userVersion() + { + /* + * NOTE: when accord journal version gets bumped, expose it via yaml. + * This way operators can force previous version on upgrade, temporarily, + * to allow easier downgrades if something goes wrong. + */ + return 1; + } + }; + + final File directory; + final Journal> journal; + + @VisibleForTesting + public AccordJournal() + { + directory = new File(DatabaseDescriptor.getAccordJournalDirectory()); + journal = new Journal<>("AccordJournal", directory, PARAMS, Key.SUPPORT, MESSAGE_SERIALIZER); + } + + public AccordJournal start() + { + // journal.start(); TODO: re-enable + return this; + } + + public void shutdown() + { + // journal.shutdown(); TODO: re-enable + } + + boolean mustMakeDurable(PreLoadContext context) + { + return false; + // return context instanceof TxnRequest && Type.mustMakeDurable((TxnRequest) context); TODO: re-enable + } + + public void append(PreLoadContext context, Executor executor, AsyncWriteCallback callback) + { + append((TxnRequest) context, executor, callback); + } + + public void append(TxnRequest message, Executor executor, AsyncWriteCallback callback) + { + Key key = new Key(message.txnId, Type.fromMsgType(message.type())); + journal.asyncWrite(key, message, SENTINEL_HOSTS, executor, callback); + } + + public TxnRequest read(TxnId txnId, Type type) + { + Key key = new Key(txnId, type); + return journal.read(key); + } + + PreAccept readPreAccept(TxnId txnId) + { + return (PreAccept) read(txnId, Type.PREACCEPT_REQ); + } + + Accept readAccept(TxnId txnId) + { + return (Accept) read(txnId, Type.ACCEPT_REQ); + } + + Commit readCommit(TxnId txnId) + { + return (Commit) read(txnId, Type.COMMIT_REQ); + } + + Apply readApply(TxnId txnId) + { + return (Apply) read(txnId, Type.APPLY_REQ); + } + + static class Key + { + final TxnId txnId; + final Type type; + + Key(TxnId txnId, Type type) + { + this.txnId = txnId; + this.type = type; + } + + /** + * Support for (de)serializing and comparing record keys. + *

+ * Implements its own serialization and comparison for {@link TxnId} to satisty + * {@link KeySupport} contract - puts hybrid logical clock ahead of epoch + * when ordering txn ids. This is done for more precise elimination of candidate + * segments by min/max record key in segment. + */ + static final KeySupport SUPPORT = new KeySupport() + { + private static final int HLC_OFFSET = 0; + private static final int EPOCH_AND_FLAGS_OFFSET = HLC_OFFSET + LONG_SIZE; + private static final int NODE_OFFSET = EPOCH_AND_FLAGS_OFFSET + LONG_SIZE; + private static final int TYPE_OFFSET = NODE_OFFSET + INT_SIZE; + + @Override + public int serializedSize(int version) + { + return LONG_SIZE // txnId.hlc() + + 6 // txnId.epoch() + + 2 // txnId.flags() + + INT_SIZE // txnId.node + + BYTE_SIZE; // type + } + + @Override + public void serialize(Key key, DataOutputPlus out, int version) throws IOException + { + serializeTxnId(key.txnId, out); + out.writeByte(key.type.id); + } + + private void serializeTxnId(TxnId txnId, DataOutputPlus out) throws IOException + { + out.writeLong(txnId.hlc()); + out.writeLong(epochAndFlags(txnId)); + out.writeInt(txnId.node.id); + } + + @Override + public Key deserialize(DataInputPlus in, int version) throws IOException + { + TxnId txnId = deserializeTxnId(in); + int type = in.readByte(); + return new Key(txnId, Type.fromId(type)); + } + + private TxnId deserializeTxnId(DataInputPlus in) throws IOException + { + long hlc = in.readLong(); + long epochAndFlags = in.readLong(); + int nodeId = in.readInt(); + return TxnId.fromValues(epoch(epochAndFlags), hlc, flags(epochAndFlags), new Id(nodeId)); + } + + @Override + public Key deserialize(ByteBuffer buffer, int position, int version) + { + TxnId txnId = deserializeTxnId(buffer, position); + int type = buffer.get(position + TYPE_OFFSET); + return new Key(txnId, Type.fromId(type)); + } + + private TxnId deserializeTxnId(ByteBuffer buffer, int position) + { + long hlc = buffer.getLong(position + HLC_OFFSET); + long epochAndFlags = buffer.getLong(position + EPOCH_AND_FLAGS_OFFSET); + int nodeId = buffer.getInt(position + NODE_OFFSET); + return TxnId.fromValues(epoch(epochAndFlags), hlc, flags(epochAndFlags), new Id(nodeId)); + } + + @Override + public void updateChecksum(Checksum crc, Key key, int version) + { + updateChecksum(crc, key.txnId); + crc.update(key.type.id & 0xFF); + } + + private void updateChecksum(Checksum crc, TxnId txnId) + { + updateChecksumLong(crc, txnId.hlc()); + updateChecksumLong(crc, epochAndFlags(txnId)); + updateChecksumInt(crc, txnId.node.id); + } + + @Override + public int compareWithKeyAt(Key k, ByteBuffer buffer, int position, int version) + { + int cmp = compareWithTxnIdAt(k.txnId, buffer, position); + if (cmp != 0) return cmp; + + byte type = buffer.get(position + TYPE_OFFSET); + cmp = Byte.compare((byte) k.type.id, type); + return cmp; + } + + private int compareWithTxnIdAt(TxnId txnId, ByteBuffer buffer, int position) + { + long hlc = buffer.getLong(position + HLC_OFFSET); + int cmp = Long.compareUnsigned(txnId.hlc(), hlc); + if (cmp != 0) return cmp; + + long epochAndFlags = buffer.getLong(position + EPOCH_AND_FLAGS_OFFSET); + cmp = Long.compareUnsigned(epochAndFlags(txnId), epochAndFlags); + if (cmp != 0) return cmp; + + int nodeId = buffer.getInt(position + NODE_OFFSET); + cmp = Integer.compareUnsigned(txnId.node.id, nodeId); + return cmp; + } + + @Override + public int compare(Key k1, Key k2) + { + int cmp = compare(k1.txnId, k2.txnId); + if (cmp == 0) cmp = Byte.compare((byte) k1.type.id, (byte) k2.type.id); + return cmp; + } + + private int compare(TxnId txnId1, TxnId txnId2) + { + int cmp = Long.compareUnsigned(txnId1.hlc(), txnId2.hlc()); + if (cmp == 0) cmp = Long.compareUnsigned(epochAndFlags(txnId1), epochAndFlags(txnId2)); + if (cmp == 0) cmp = Integer.compareUnsigned(txnId1.node.id, txnId2.node.id); + return cmp; + } + + private long epochAndFlags(TxnId txnId) + { + return (txnId.epoch() << 16) | (long) txnId.flags(); + } + + private long epoch(long epochAndFlags) + { + return epochAndFlags >>> 16; + } + + private int flags(long epochAndFlags) + { + return (int) (epochAndFlags & ((1 << 16) - 1)); + } + }; + + @Override + public boolean equals(Object other) + { + if (this == other) + return true; + return (other instanceof Key) && equals((Key) other); + } + + boolean equals(Key other) + { + return this.type == other.type && this.txnId.equals(other.txnId); + } + + @Override + public int hashCode() + { + return type.hashCode() + 31 * txnId.hashCode(); + } + + @Override + public String toString() + { + return "Key{" + txnId + ", " + type + '}'; + } + } + + static final ValueSerializer> MESSAGE_SERIALIZER = new ValueSerializer>() + { + @Override + public int serializedSize(TxnRequest message, int version) + { + return Ints.checkedCast(Type.ofMessage(message).serializedSize(message, version)); + } + + @Override + public void serialize(TxnRequest message, DataOutputPlus out, int version) throws IOException + { + Type.ofMessage(message).serialize(message, out, version); + } + + @Override + public TxnRequest deserialize(Key key, DataInputPlus in, int version) throws IOException + { + return key.type.deserialize(in, version); + } + }; + + /** + * Accord Message type - consequently the kind of persisted record. + *

+ * Note: {@link EnumSerializer} is intentionally not being reused here, for two reasons: + * 1. This is an internal enum, fully under our control, not part of an external library + * 2. It's persisted in the record key, so has the additional constraint of being fixed size and + * shouldn't be using varint encoding + */ + public enum Type implements IVersionedSerializer> + { + PREACCEPT_REQ (0, MessageType.PREACCEPT_REQ, PreacceptSerializers.request), + ACCEPT_REQ (1, MessageType.ACCEPT_REQ, AcceptSerializers.request ), + COMMIT_REQ (2, MessageType.COMMIT_REQ, CommitSerializers.request ), + APPLY_REQ (3, MessageType.APPLY_REQ, ApplySerializers.request ); + + final int id; + final MessageType msgType; + final IVersionedSerializer> serializer; + + Type(int id, MessageType msgType, IVersionedSerializer> serializer) + { + if (id < 0) + throw new IllegalArgumentException("Negative Type id " + id); + if (id > Byte.MAX_VALUE) + throw new IllegalArgumentException("Type id doesn't fit in a single byte: " + id); + + this.id = id; + this.msgType = msgType; + + //noinspection unchecked + this.serializer = (IVersionedSerializer>) serializer; + } + + private static final Type[] idToTypeMapping; + private static final Map msgTypeToTypeMap; + + static + { + Type[] types = values(); + + int maxId = -1; + for (Type type : types) + maxId = Math.max(type.id, maxId); + + + Type[] idToType = new Type[maxId + 1]; + for (Type type : types) + { + if (null != idToType[type.id]) + throw new IllegalStateException("Duplicate Type id " + type.id); + idToType[type.id] = type; + } + idToTypeMapping = idToType; + + EnumMap msgTypeToType = new EnumMap<>(MessageType.class); + for (Type type : types) + { + if (null != msgTypeToType.put(type.msgType, type)) + throw new IllegalStateException("Duplicate MessageType " + type.msgType); + } + msgTypeToTypeMap = msgTypeToType; + } + + static Type fromId(int id) + { + if (id < 0 || id >= idToTypeMapping.length) + throw new IllegalArgumentException("Out or range Type id " + id); + Type type = idToTypeMapping[id]; + if (null == type) + throw new IllegalArgumentException("Unknown Type id " + id); + return type; + } + + static Type fromMsgType(MessageType msgType) + { + Type type = msgTypeToTypeMap.get(msgType); + if (null == type) + throw new IllegalArgumentException("Unsupported MessageType " + msgType); + return type; + } + + static Type ofMessage(TxnRequest request) + { + return fromMsgType(request.type()); + } + + static boolean mustMakeDurable(TxnRequest message) + { + return msgTypeToTypeMap.containsKey(message.type()); + } + + @Override + public void serialize(TxnRequest request, DataOutputPlus out, int version) throws IOException + { + serializer.serialize(request, out, msVersion(version)); + } + + @Override + public TxnRequest deserialize(DataInputPlus in, int version) throws IOException + { + return serializer.deserialize(in, msVersion(version)); + } + + @Override + public long serializedSize(TxnRequest request, int version) + { + return serializer.serializedSize(request, msVersion(version)); + } + + static + { + // make noise early if we forget to update our version mappings + Invariants.checkState(MessagingService.current_version == MessagingService.VERSION_50); + } + + private static int msVersion(int version) + { + switch (version) + { + default: throw new IllegalArgumentException(); + case 1: return MessagingService.VERSION_50; + } + } + } +} diff --git a/src/java/org/apache/cassandra/service/accord/AccordService.java b/src/java/org/apache/cassandra/service/accord/AccordService.java index d8b9f4d89b..4de78375bd 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordService.java +++ b/src/java/org/apache/cassandra/service/accord/AccordService.java @@ -77,7 +77,7 @@ public class AccordService implements IAccordService, Shutdownable private final AccordConfigurationService configService; private final AccordScheduler scheduler; private final AccordVerbHandler verbHandler; - + private static final IAccordService NOOP_SERVICE = new IAccordService() { @Override @@ -148,7 +148,7 @@ public class AccordService implements IAccordService, Shutdownable scheduler, SizeOfIntersectionSorter.SUPPLIER, SimpleProgressLog::new, - AccordCommandStores::new); + AccordCommandStores.factory(new AccordJournal().start())); this.nodeShutdown = toShutdownable(node); this.verbHandler = new AccordVerbHandler<>(this.node); } diff --git a/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java b/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java index 41314a2eaf..06e4881fb4 100644 --- a/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java +++ b/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java @@ -38,7 +38,7 @@ import org.apache.cassandra.service.paxos.Ballot; import org.apache.cassandra.service.paxos.Commit; import static org.apache.cassandra.io.util.SequentialWriterOption.FINISH_ON_CLOSE; -import static org.apache.cassandra.net.Crc.crc32; +import static org.apache.cassandra.utils.Crc.crc32; /** * Tracks the highest paxos ballot we've seen, and the lowest ballot we can accept. diff --git a/src/java/org/apache/cassandra/net/Crc.java b/src/java/org/apache/cassandra/utils/Crc.java similarity index 90% rename from src/java/org/apache/cassandra/net/Crc.java rename to src/java/org/apache/cassandra/utils/Crc.java index 8f63e51a93..f1a31584f3 100644 --- a/src/java/org/apache/cassandra/net/Crc.java +++ b/src/java/org/apache/cassandra/utils/Crc.java @@ -15,7 +15,7 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.net; +package org.apache.cassandra.utils; import java.io.IOException; import java.nio.ByteBuffer; @@ -48,26 +48,31 @@ public class Crc public static CRC32 crc32() { CRC32 crc = crc32.get(); + return initialize(crc); + } + + public static CRC32 initialize(CRC32 crc) + { crc.reset(); - crc.update(initialBytes); + crc.update(initialBytes, 0, initialBytes.length); return crc; } - static int computeCrc32(ByteBuf buffer, int startReaderIndex, int endReaderIndex) + public static int computeCrc32(ByteBuf buffer, int startReaderIndex, int endReaderIndex) { CRC32 crc = crc32(); crc.update(buffer.internalNioBuffer(startReaderIndex, endReaderIndex - startReaderIndex)); return (int) crc.getValue(); } - static int computeCrc32(ByteBuffer buffer, int start, int end) + public static int computeCrc32(ByteBuffer buffer, int start, int end) { CRC32 crc = crc32(); updateCrc32(crc, buffer, start, end); return (int) crc.getValue(); } - static void updateCrc32(CRC32 crc, ByteBuffer buffer, int start, int end) + public static void updateCrc32(CRC32 crc, ByteBuffer buffer, int start, int end) { int savePosition = buffer.position(); int saveLimit = buffer.limit(); @@ -116,7 +121,7 @@ public class Crc * @param len the number of bytes, greater than 0 and fewer than 9, to be read from bytes * @return the least-significant bit AND byte order crc24 using the CRC24_POLY polynomial */ - static int crc24(long bytes, int len) + public static int crc24(long bytes, int len) { int crc = CRC24_INIT; while (len-- > 0) diff --git a/src/java/org/apache/cassandra/utils/FBUtilities.java b/src/java/org/apache/cassandra/utils/FBUtilities.java index 79b5c621c6..6892777422 100644 --- a/src/java/org/apache/cassandra/utils/FBUtilities.java +++ b/src/java/org/apache/cassandra/utils/FBUtilities.java @@ -131,7 +131,7 @@ public class FBUtilities private static volatile String previousReleaseVersionString; - private static int availableProcessors = CASSANDRA_AVAILABLE_PROCESSORS.getInt(DatabaseDescriptor.getAvailableProcessors()); + private static final int availableProcessors = CASSANDRA_AVAILABLE_PROCESSORS.getInt(DatabaseDescriptor.getAvailableProcessors()); private static volatile Supplier systemInfoSupplier = Suppliers.memoize(SystemInfo::new); @@ -1165,20 +1165,26 @@ public class FBUtilities { process.destroyForcibly(); logger.error("Command {} did not complete in {}, killed forcibly:\noutput:\n{}\n(truncated {} bytes)\nerror:\n{}\n(truncated {} bytes)", - Arrays.toString(cmd), timeout, out.asString(), outOverflow, err.asString(), errOverflow); + Arrays.toString(cmd), timeout, out.asString(), outOverflow, err.asString(), errOverflow); throw new TimeoutException("Command " + Arrays.toString(cmd) + " did not complete in " + timeout); } int r = process.exitValue(); if (r != 0) { logger.error("Command {} failed with exit code {}:\noutput:\n{}\n(truncated {} bytes)\nerror:\n{}\n(truncated {} bytes)", - Arrays.toString(cmd), r, out.asString(), outOverflow, err.asString(), errOverflow); + Arrays.toString(cmd), r, out.asString(), outOverflow, err.asString(), errOverflow); throw new IOException("Command " + Arrays.toString(cmd) + " failed with exit code " + r); } return out.asString(); } } + public static void updateChecksumShort(Checksum checksum, short v) + { + checksum.update((v >>> 8) & 0xFF); + checksum.update((v >>> 0) & 0xFF); + } + public static void updateChecksumInt(Checksum checksum, int v) { checksum.update((v >>> 24) & 0xFF); @@ -1187,6 +1193,12 @@ public class FBUtilities checksum.update((v >>> 0) & 0xFF); } + public static void updateChecksumLong(Checksum checksum, long v) + { + updateChecksumInt(checksum, (int) (v >>> 32)); + updateChecksumInt(checksum, (int) (v & 0xFFFFFFFFL)); + } + /** * Updates checksum with the provided ByteBuffer at the given offset + length. * Resets position and limit back to their original values on return. @@ -1432,4 +1444,21 @@ public class FBUtilities { return systemInfoSupplier.get(); } + + public enum Order { LT, EQ, GT } + public static Order compare(T a, T b, Comparator comparator) + { + int rc = comparator.compare(a, b); + if (rc < 0) return Order.LT; + if (rc == 0) return Order.EQ; + return Order.GT; + } + + public static Order compare(A a, B b, AsymmetricOrdering comparator) + { + int rc = comparator.compareAsymmetric(a, b); + if (rc < 0) return Order.LT; + if (rc == 0) return Order.EQ; + return Order.GT; + } } \ No newline at end of file diff --git a/src/java/org/apache/cassandra/utils/JVMStabilityInspector.java b/src/java/org/apache/cassandra/utils/JVMStabilityInspector.java index fdd678efd0..220c7ff9fd 100644 --- a/src/java/org/apache/cassandra/utils/JVMStabilityInspector.java +++ b/src/java/org/apache/cassandra/utils/JVMStabilityInspector.java @@ -30,10 +30,6 @@ import java.util.function.Consumer; import com.google.common.annotations.VisibleForTesting; import com.google.common.collect.ImmutableSet; -import org.apache.cassandra.exceptions.UnrecoverableIllegalStateException; -import org.apache.cassandra.metrics.StorageMetrics; -import org.apache.cassandra.service.DiskErrorsHandlerService; -import org.apache.cassandra.tracing.Tracing; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -41,9 +37,14 @@ import net.nicoulaj.compilecommand.annotations.Exclude; import org.apache.cassandra.concurrent.ScheduledExecutors; import org.apache.cassandra.config.Config; import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.exceptions.UnrecoverableIllegalStateException; import org.apache.cassandra.io.FSError; import org.apache.cassandra.io.sstable.CorruptSSTableException; +import org.apache.cassandra.journal.Params.FailurePolicy; +import org.apache.cassandra.service.DiskErrorsHandlerService; +import org.apache.cassandra.metrics.StorageMetrics; import org.apache.cassandra.service.StorageService; +import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException; import static org.apache.cassandra.config.CassandraRelevantProperties.PRINT_HEAP_HISTOGRAM_ON_OUT_OF_MEMORY_ERROR; @@ -94,6 +95,11 @@ public final class JVMStabilityInspector inspectThrowable(t, ex -> DiskErrorsHandlerService.get().inspectCommitLogError(ex)); } + public static void inspectJournalThrowable(Throwable t, String journalName, FailurePolicy failurePolicy) + { + inspectThrowable(t, th -> inspectJournalError(th, journalName, failurePolicy)); + } + public static void inspectThrowable(Throwable t, Consumer fn) throws OutOfMemoryError { boolean isUnstable = false; @@ -128,7 +134,14 @@ public final class JVMStabilityInspector } // Anything other than an OOM, we should try and heap dump to capture what's going on if configured to do so - HeapUtils.maybeCreateHeapDump(); + try + { + HeapUtils.maybeCreateHeapDump(); + } + catch (Throwable sub) + { + t.addSuppressed(sub); + } if (t instanceof InterruptedException) throw new UncheckedInterruptedException((InterruptedException) t); @@ -189,6 +202,19 @@ public final class JVMStabilityInspector } } + private static void inspectJournalError(Throwable t, String journalName, FailurePolicy failurePolicy) + { + if (!StorageService.instance.isDaemonSetupCompleted()) + { + logger.error("Exiting due to error while processing journal {} during initialization.", journalName, t); + killer.killCurrentJVM(t, true); + } + else if (failurePolicy == FailurePolicy.DIE) + { + killer.killCurrentJVM(t); + } + } + public static void killCurrentJVM(Throwable t, boolean quiet) { killer.killCurrentJVM(t, quiet); diff --git a/src/java/org/apache/cassandra/utils/NoSpamLogger.java b/src/java/org/apache/cassandra/utils/NoSpamLogger.java index 0a13f6b2a5..5c37a043d8 100644 --- a/src/java/org/apache/cassandra/utils/NoSpamLogger.java +++ b/src/java/org/apache/cassandra/utils/NoSpamLogger.java @@ -217,6 +217,11 @@ public class NoSpamLogger minIntervalNanos = timeUnit.toNanos(minInterval); } + public static NoSpamLogger wrap(Logger wrapped, long minInterval, TimeUnit timeUnit) + { + return new NoSpamLogger(wrapped, minInterval, timeUnit); + } + public boolean info(long nowNanos, String s, Object... objects) { return NoSpamLogger.this.log( Level.INFO, s, nowNanos, objects); diff --git a/src/java/org/apache/cassandra/utils/TimeUUID.java b/src/java/org/apache/cassandra/utils/TimeUUID.java index 6f4281c7ca..b7930ed582 100644 --- a/src/java/org/apache/cassandra/utils/TimeUUID.java +++ b/src/java/org/apache/cassandra/utils/TimeUUID.java @@ -143,7 +143,12 @@ public class TimeUUID implements Serializable, Comparable public static TimeUUID deserialize(ByteBuffer buffer) { - return fromBytes(buffer.getLong(buffer.position()), buffer.getLong(buffer.position() + 8)); + return deserialize(buffer, buffer.position()); + } + + public static TimeUUID deserialize(ByteBuffer buffer, int position) + { + return fromBytes(buffer.getLong(position), buffer.getLong(position + 8)); } public static TimeUUID deserialize(DataInput in) throws IOException diff --git a/test/conf/cassandra-mtls.yaml b/test/conf/cassandra-mtls.yaml index d6f1b3e52c..356d3e918d 100644 --- a/test/conf/cassandra-mtls.yaml +++ b/test/conf/cassandra-mtls.yaml @@ -25,6 +25,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal cdc_raw_directory: build/test/cassandra/cdc_raw cdc_enabled: false hints_directory: build/test/cassandra/hints diff --git a/test/conf/cassandra-murmur.yaml b/test/conf/cassandra-murmur.yaml index 2e5828fb56..c3f7442aa6 100644 --- a/test/conf/cassandra-murmur.yaml +++ b/test/conf/cassandra-murmur.yaml @@ -8,6 +8,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal cdc_raw_directory: build/test/cassandra/cdc_raw cdc_enabled: false hints_directory: build/test/cassandra/hints diff --git a/test/conf/cassandra-old.yaml b/test/conf/cassandra-old.yaml index b8c3b028c5..0cda4138d7 100644 --- a/test/conf/cassandra-old.yaml +++ b/test/conf/cassandra-old.yaml @@ -9,6 +9,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size_in_mb: 5 commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/conf/cassandra-pem-jks-sslcontextfactory.yaml b/test/conf/cassandra-pem-jks-sslcontextfactory.yaml index 0bd034d5ae..0e8a83f0aa 100644 --- a/test/conf/cassandra-pem-jks-sslcontextfactory.yaml +++ b/test/conf/cassandra-pem-jks-sslcontextfactory.yaml @@ -27,6 +27,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/conf/cassandra-pem-sslcontextfactory-invalidconfiguration.yaml b/test/conf/cassandra-pem-sslcontextfactory-invalidconfiguration.yaml index 16cffb58bd..a3146b79ab 100644 --- a/test/conf/cassandra-pem-sslcontextfactory-invalidconfiguration.yaml +++ b/test/conf/cassandra-pem-sslcontextfactory-invalidconfiguration.yaml @@ -27,6 +27,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/conf/cassandra-pem-sslcontextfactory.yaml b/test/conf/cassandra-pem-sslcontextfactory.yaml index 229a0b03fd..d382d9d64b 100644 --- a/test/conf/cassandra-pem-sslcontextfactory.yaml +++ b/test/conf/cassandra-pem-sslcontextfactory.yaml @@ -27,6 +27,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/conf/cassandra-seeds.yaml b/test/conf/cassandra-seeds.yaml index 53f82dd6ec..3a0fbf1831 100644 --- a/test/conf/cassandra-seeds.yaml +++ b/test/conf/cassandra-seeds.yaml @@ -9,6 +9,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal cdc_raw_directory: build/test/cassandra/cdc_raw cdc_enabled: false hints_directory: build/test/cassandra/hints diff --git a/test/conf/cassandra-sslcontextfactory-invalidconfiguration.yaml b/test/conf/cassandra-sslcontextfactory-invalidconfiguration.yaml index 6b74883360..6f83ec3341 100644 --- a/test/conf/cassandra-sslcontextfactory-invalidconfiguration.yaml +++ b/test/conf/cassandra-sslcontextfactory-invalidconfiguration.yaml @@ -27,6 +27,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/conf/cassandra-sslcontextfactory.yaml b/test/conf/cassandra-sslcontextfactory.yaml index a20d26e59b..a570bdf5f6 100644 --- a/test/conf/cassandra-sslcontextfactory.yaml +++ b/test/conf/cassandra-sslcontextfactory.yaml @@ -27,6 +27,7 @@ commitlog_sync: periodic commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/conf/cassandra.yaml b/test/conf/cassandra.yaml index ff017b3afc..c0e732cec2 100644 --- a/test/conf/cassandra.yaml +++ b/test/conf/cassandra.yaml @@ -10,6 +10,7 @@ commitlog_sync_period: 10s commitlog_segment_size: 5MiB commitlog_directory: build/test/cassandra/commitlog commitlog_disk_access_mode: legacy +accord_journal_directory: build/test/cassandra/accord_journal # commitlog_compression: # - class_name: LZ4Compressor cdc_raw_directory: build/test/cassandra/cdc_raw diff --git a/test/distributed/org/apache/cassandra/distributed/impl/InstanceConfig.java b/test/distributed/org/apache/cassandra/distributed/impl/InstanceConfig.java index 19194188ee..66bd85689e 100644 --- a/test/distributed/org/apache/cassandra/distributed/impl/InstanceConfig.java +++ b/test/distributed/org/apache/cassandra/distributed/impl/InstanceConfig.java @@ -72,6 +72,7 @@ public class InstanceConfig implements IInstanceConfig String commitlog_directory, String hints_directory, String cdc_raw_directory, + String accord_journal_directory, Collection initial_token, int storage_port, int native_transport_port, @@ -92,6 +93,7 @@ public class InstanceConfig implements IInstanceConfig .set("commitlog_directory", commitlog_directory) .set("hints_directory", hints_directory) .set("cdc_raw_directory", cdc_raw_directory) + .set("accord_journal_directory", accord_journal_directory) .set("partitioner", "org.apache.cassandra.dht.Murmur3Partitioner") .set("start_native_transport", true) .set("concurrent_writes", 2) @@ -329,6 +331,7 @@ public class InstanceConfig implements IInstanceConfig String.format("%s/node%d/commitlog", root, nodeNum), String.format("%s/node%d/hints", root, nodeNum), String.format("%s/node%d/cdc", root, nodeNum), + String.format("%s/node%d/accord_journal", root, nodeNum), tokens, provisionStrategy.storagePort(nodeNum), provisionStrategy.nativeTransportPort(nodeNum), diff --git a/test/simulator/asm/org/apache/cassandra/simulator/asm/GlobalMethodTransformer.java b/test/simulator/asm/org/apache/cassandra/simulator/asm/GlobalMethodTransformer.java index 883b7a66a3..064d205646 100644 --- a/test/simulator/asm/org/apache/cassandra/simulator/asm/GlobalMethodTransformer.java +++ b/test/simulator/asm/org/apache/cassandra/simulator/asm/GlobalMethodTransformer.java @@ -85,12 +85,17 @@ class GlobalMethodTransformer extends MethodVisitor || !deterministic && owner.equals("java/lang/System") && name.equals("identityHashCode") || owner.equals("java/util/UUID") && name.equals("randomUUID") || owner.equals("com/google/common/util/concurrent/Uninterruptibles") && name.equals("sleepUninterruptibly") - || owner.equals("sun/misc/Unsafe") && name.equals("getUnsafe"))) - )) + || owner.equals("sun/misc/Unsafe") && name.equals("getUnsafe")))) + ) { transformer.witness(GLOBAL_METHOD); super.visitMethodInsn(Opcodes.INVOKESTATIC, "org/apache/cassandra/simulator/systems/InterceptorOfSystemMethods$Global", name, descriptor, false); } + else if (owner.equals("java/util/concurrent/TimeUnit") && name.equals("sleep")) + { + transformer.witness(GLOBAL_METHOD); + super.visitMethodInsn(Opcodes.INVOKESTATIC, "org/apache/cassandra/simulator/systems/InterceptorOfSystemMethods$Global", "sleep", "(Ljava/util/concurrent/TimeUnit;J)V", false); + } else if ((globalMethods || deterministic) && opcode == Opcodes.INVOKESTATIC && ((owner.equals("java/util/concurrent/ThreadLocalRandom") && (name.equals("getProbe") || name.equals("advanceProbe") || name.equals("localInit"))) || (owner.equals("java/util/concurrent/atomic/Striped64") && (name.equals("getProbe") || name.equals("advanceProbe")))) diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java new file mode 100644 index 0000000000..2f9da8910c --- /dev/null +++ b/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java @@ -0,0 +1,264 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.simulator.test; + +import java.io.IOException; +import java.util.List; +import java.util.UUID; +import java.util.concurrent.CopyOnWriteArrayList; +import javax.annotation.Nullable; + +import com.google.common.collect.ImmutableMap; +import org.apache.cassandra.schema.*; +import org.junit.Ignore; +import org.junit.Test; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import accord.Utils; +import accord.api.Data; +import accord.api.RoutingKey; +import accord.api.Update; +import accord.api.Write; +import accord.impl.TopologyUtils; +import accord.local.Node; +import accord.messages.PreAccept; +import accord.messages.TxnRequest; +import accord.primitives.FullKeyRoute; +import accord.primitives.FullRoute; +import accord.primitives.Keys; +import accord.primitives.Ranges; +import accord.primitives.Seekables; +import accord.primitives.Txn; +import accord.primitives.TxnId; +import accord.topology.Topologies; +import org.apache.cassandra.concurrent.ExecutorFactory; +import org.apache.cassandra.concurrent.ExecutorPlus; +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.config.ParameterizedClass; +import org.apache.cassandra.db.ColumnFamilyStore; +import org.apache.cassandra.db.Keyspace; +import org.apache.cassandra.db.commitlog.CommitLog; +import org.apache.cassandra.db.marshal.Int32Type; +import org.apache.cassandra.dht.Murmur3Partitioner; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.Files; +import org.apache.cassandra.journal.AsyncWriteCallback; +import org.apache.cassandra.service.accord.AccordJournal; +import org.apache.cassandra.service.accord.TokenRange; +import org.apache.cassandra.service.accord.api.AccordRoutingKey; +import org.apache.cassandra.service.accord.api.PartitionKey; +import org.apache.cassandra.service.accord.txn.TxnNamedRead; +import org.apache.cassandra.service.accord.txn.TxnQuery; +import org.apache.cassandra.service.accord.txn.TxnRead; +import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.Isolated; +import org.apache.cassandra.utils.concurrent.CountDownLatch; + +public class AccordJournalSimulationTest extends SimulationTestBase +{ + @Test + @Ignore // TODO: re-enable + public void test() throws IOException + { + simulate(arr(() -> run()), + () -> check()); + } + + private static void run() + { + for (int i = 0; i < State.events; i++) + { + int finalI = i; + State.executor.execute(() -> State.append(finalI)); + } + + try + { + State.eventsDurable.await(); + State.logger.info("All events are durable done!"); + } + catch (InterruptedException e) + { + throw new AssertionError(e); + } + + if (!State.exceptions.isEmpty()) + { + AssertionError error = new AssertionError("Exceptions found during test"); + State.exceptions.forEach(error::addSuppressed); + throw error; + } + + State.journal.shutdown(); + State.logger.info("Run complete"); + } + + private static void check() + { + State.logger.info("Check starting"); + State.journal.start(); // to avoid a while true deadlock + try + { + for (int i = 0; i < State.events; i++) + { + TxnRequest event = State.journal.read(State.toTxnId(i), AccordJournal.Type.PREACCEPT_REQ); + State.logger.info("Event {} -> {}", i, event); + if (event == null) + throw new AssertionError(String.format("Unable to read event %d", i)); + } + State.logger.info("Check complete"); + } + finally + { + State.journal.shutdown(); + } + } + + @Isolated + public static class State + { + private static final Logger logger = LoggerFactory.getLogger(State.class); + private static final String KEYSPACE = "test"; + + static + { + Files.newGlobalInMemoryFileSystem(); + DatabaseDescriptor.clientWithDaemonConfig(); + DatabaseDescriptor.setPartitionerUnsafe(Murmur3Partitioner.instance); + DatabaseDescriptor.setAccordJournalDirectory("/journal"); + new File("/journal").createDirectoriesIfNotExists(); + DatabaseDescriptor.setCommitLogCompression(new ParameterizedClass("LZ4Compressor", ImmutableMap.of())); + DatabaseDescriptor.setDumpHeapOnUncaughtException(false); + + // in order to do journal.read, we need all this setup first! + Keyspace.setInitialized(); + Schema.instance.submit(SchemaTransformations.addKeyspace(KeyspaceMetadata.create(State.KEYSPACE, KeyspaceParams.simple(1)), true)); + Keyspace ks = Keyspace.open(State.KEYSPACE); + ks.initCfCustom(ColumnFamilyStore.createColumnFamilyStore(ks, TableMetadataRef.forOfflineTools(TableMetadata.builder(State.KEYSPACE, State.KEYSPACE) + .addPartitionKeyColumn("pk", Int32Type.instance) + .build()).get(), false)); + + try + { + CommitLog.instance.shutdownBlocking(); + } + catch (InterruptedException e) + { + // ignore + } + } + private static final ExecutorPlus executor = ExecutorFactory.Global.executorFactory().pooled("name", 10); + private static final AccordJournal journal = new AccordJournal(); + private static final int events = 100; + private static final CountDownLatch eventsWritten = CountDownLatch.newCountDownLatch(events); + private static final CountDownLatch eventsDurable = CountDownLatch.newCountDownLatch(events); + private static final List exceptions = new CopyOnWriteArrayList<>(); + + static + { + journal.start(); + } + + public static void append(int event) + { + TxnRequest request = toRequest(event); + journal.append(request, executor, new AsyncWriteCallback() + { + @Override + public void run() + { + durable(event); + } + + @Override + public void onFailure(Throwable error) + { + eventsDurable.decrement(); // to make sure we don't block forever + exceptions.add(error); + } + }); + eventsWritten.decrement(); + logger.info("append({}); remaining {}", event, eventsWritten.count()); + } + + private static void durable(int event) + { + eventsDurable.decrement(); + logger.info("durable({}); remaining {}", event, eventsDurable.count()); + } + + private static TxnRequest toRequest(int event) + { + TxnId id = toTxnId(event); + Ranges ranges = Ranges.of(new TokenRange(AccordRoutingKey.SentinelKey.min("system"), AccordRoutingKey.SentinelKey.max("system"))); + Topologies topologies = Utils.topologies(TopologyUtils.initialTopology(new Node.Id[] {node}, ranges, 3)); + Keys keys = Keys.of(toKey(0)); + Txn txn = new Txn.InMemory(keys, new TxnRead(new TxnNamedRead[0], keys), TxnQuery.ALL, new NoopUpdate()); + FullRoute route = route(); + return new PreAccept(node, topologies, id, txn, route); + } + + private static TxnId toTxnId(int event) + { + return TxnId.fromValues(1, event, 0, node); + } + + private static PartitionKey toKey(int a) + { + return new PartitionKey(KEYSPACE, tableId, Murmur3Partitioner.instance.decorateKey(ByteBufferUtil.bytes(a))); + } + + private static final TableId tableId = TableId.fromUUID(new UUID(0, 0)); + private static final Node.Id node = new Node.Id(0); + + private static FullRoute route() + { + return new FullKeyRoute(key, new RoutingKey[]{ key }); + } + + private static final RoutingKey key = new AccordRoutingKey.TokenKey("system", new Murmur3Partitioner.LongToken(42)); + } + + public static class NoopUpdate implements Update + { + @Override + public Seekables keys() + { + return null; + } + + @Override + public Write apply(@Nullable Data data) + { + return null; + } + + @Override + public Update slice(Ranges ranges) + { + return null; + } + + @Override + public Update merge(Update other) + { + return null; + } + } +} diff --git a/test/unit/org/apache/cassandra/ServerTestUtils.java b/test/unit/org/apache/cassandra/ServerTestUtils.java index ead4a1a558..79b9efaa89 100644 --- a/test/unit/org/apache/cassandra/ServerTestUtils.java +++ b/test/unit/org/apache/cassandra/ServerTestUtils.java @@ -208,6 +208,7 @@ public final class ServerTestUtils if (cdcDir != null) cleanupDirectory(cdcDir); cleanupDirectory(DatabaseDescriptor.getHintsDirectory()); + cleanupDirectory(DatabaseDescriptor.getAccordJournalDirectory()); cleanupSavedCaches(); // clean up data directory which are stored as data directory/keyspace/data files diff --git a/test/unit/org/apache/cassandra/net/ManyToOneConcurrentLinkedQueueTest.java b/test/unit/org/apache/cassandra/concurrent/ManyToOneConcurrentLinkedQueueTest.java similarity index 99% rename from test/unit/org/apache/cassandra/net/ManyToOneConcurrentLinkedQueueTest.java rename to test/unit/org/apache/cassandra/concurrent/ManyToOneConcurrentLinkedQueueTest.java index 4f60d01f24..d1baae7e10 100644 --- a/test/unit/org/apache/cassandra/net/ManyToOneConcurrentLinkedQueueTest.java +++ b/test/unit/org/apache/cassandra/concurrent/ManyToOneConcurrentLinkedQueueTest.java @@ -15,7 +15,7 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.net; +package org.apache.cassandra.concurrent; import java.util.BitSet; import java.util.NoSuchElementException; diff --git a/test/unit/org/apache/cassandra/io/util/Files.java b/test/unit/org/apache/cassandra/io/util/Files.java new file mode 100644 index 0000000000..4a0ec3285d --- /dev/null +++ b/test/unit/org/apache/cassandra/io/util/Files.java @@ -0,0 +1,33 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.io.util; + +import java.nio.file.FileSystem; + +import com.google.common.jimfs.Jimfs; + +public class Files +{ + public static FileSystem newGlobalInMemoryFileSystem() + { + FileSystem fs = Jimfs.newFileSystem(); + File.unsafeSetFilesystem(fs); + return fs; + } +} diff --git a/test/unit/org/apache/cassandra/journal/DescriptorTest.java b/test/unit/org/apache/cassandra/journal/DescriptorTest.java new file mode 100644 index 0000000000..f0f2975be9 --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/DescriptorTest.java @@ -0,0 +1,166 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.nio.file.FileSystem; +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; +import java.util.Set; + +import com.google.common.collect.Sets; +import org.junit.Test; + +import accord.utils.Gen; +import accord.utils.Gens; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.io.util.Files; +import org.apache.cassandra.io.util.PathUtils; +import org.apache.cassandra.utils.Pair; +import org.assertj.core.api.Condition; + +import static accord.utils.Property.qt; +import static org.assertj.core.api.Assertions.assertThat; + +public class DescriptorTest +{ + private static final FileSystem FS = Files.newGlobalInMemoryFileSystem(); + + static + { + PathUtils.setDeletionListener(ignore -> {}); + } + + @Test + public void serde() + { + qt().forAll(descriptors()) + .check(desc -> + assertThat(Descriptor.fromFile(desc.fileFor(Component.DATA))).isEqualTo(desc)); + } + + @Test + public void isTmp() + { + Condition isTmp = new Condition("isTmpFile") + { + @Override + public boolean matches(File value) + { + return Descriptor.isTmpFile(value); + } + }; + qt().forAll(descriptors()).check(desc -> { + for (Component comp : Component.values()) + { + assertThat(desc.tmpFileFor(comp)).is(isTmp); + assertThat(desc.fileFor(comp)).isNot(isTmp); + } + }); + } + + @Test + public void list() + { + qt().withPure(false) + .forAll(children()) + .check(pair -> + assertThat(Descriptor.list(pair.left)).containsExactlyInAnyOrderElementsOf(pair.right)); + } + + @Test + public void order() + { + qt().withPure(false).forAll(children().filter(p -> p.right.size() >= 2)).check(pair -> + { + List list = new ArrayList<>(pair.right); + Collections.sort(list); + + Descriptor last = list.get(0); + for (int i = 1; i < list.size(); i++) + { + Descriptor current = list.get(i); + assertThat(current.directory).isEqualTo(last.directory); + assertThat(current.timestamp).isGreaterThanOrEqualTo(last.timestamp); + if (current.timestamp == last.timestamp) + assertThat(current.generation).isGreaterThanOrEqualTo(last.generation); + if (current.timestamp == last.timestamp + && current.generation == last.generation) + assertThat(current.journalVersion).isGreaterThanOrEqualTo(last.journalVersion); + if (current.timestamp == last.timestamp + && current.generation == last.generation + && current.journalVersion == last.journalVersion) + assertThat(current.userVersion).isGreaterThanOrEqualTo(last.userVersion); + last = current; + } + }); + } + + private static Gen>> children() + { + Gen dirs = dirs(); + return rs -> + { + File dir = dirs.next(rs); + if (dir.exists()) + dir.deleteRecursive(); + if (!dir.createDirectoriesIfNotExists()) + throw new AssertionError("Directory " + dir + " exists"); + int size = rs.nextInt(0, 10); + if (size == 0) + return Pair.create(dir, Collections.emptySet()); + Set uniq = Sets.newHashSetWithExpectedSize(size); + Gen descriptors = descriptors(Gens.constant(dir)); + for (int i = 0; i < size; i++) + { + Descriptor d = descriptors.next(rs); + while (!uniq.add(d)) + d = descriptors.next(rs); + } + for (Descriptor d : uniq) + d.fileFor(Component.DATA).createFileIfNotExists(); + return Pair.create(dir, uniq); + }; + } + + private static Gen descriptors() + { + Gen dir = dirs(); + return descriptors(dir); + } + + private static Gen descriptors(Gen dir) + { + Gen.LongGen longs = Gens.longs().between(0, 10); + Gen.IntGen ints = Gens.ints().between(0, 10); + return rs -> new Descriptor(dir.next(rs), longs.nextLong(rs), ints.next(rs), ints.next(rs), ints.next(rs)); + } + + private static Gen dirs() + { + Gen names = asciiVisible().ofLengthBetween(1, 100); + Gen gen = rs -> new File(FS.getPath('/' + names.next(rs))); + return gen.filter(f -> f.toCanonical().parent() != null); + } + + // TODO: replace with Gens.strings().asciiVisible() + public static Gens.SizeBuilder asciiVisible() + { + return new Gens.SizeBuilder<>(sizes -> Gens.strings().betweenCodePoints(sizes, 33, 127)); + } +} diff --git a/test/unit/org/apache/cassandra/journal/IndexTest.java b/test/unit/org/apache/cassandra/journal/IndexTest.java new file mode 100644 index 0000000000..8301ea988e --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/IndexTest.java @@ -0,0 +1,251 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.file.Files; +import java.util.Arrays; +import java.util.List; +import java.util.Map; +import java.util.stream.Collectors; + +import com.google.common.collect.Maps; +import org.junit.Test; + +import org.agrona.collections.IntHashSet; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.utils.Generators; +import org.apache.cassandra.utils.TimeUUID; +import org.quicktheories.core.Gen; +import org.quicktheories.impl.Constraint; + +import static org.apache.cassandra.utils.TimeUUID.Generator.nextTimeUUID; +import static org.assertj.core.api.Assertions.assertThat; +import static org.junit.Assert.assertArrayEquals; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertTrue; +import static org.quicktheories.QuickTheory.qt; + +public class IndexTest +{ + private static final int[] EMPTY = new int[0]; + + @Test + public void testInMemoryIndexBasics() + { + InMemoryIndex index = InMemoryIndex.create(TimeUUIDKeySupport.INSTANCE); + + TimeUUID key0 = nextTimeUUID(); + TimeUUID key1 = nextTimeUUID(); + TimeUUID key2 = nextTimeUUID(); + TimeUUID key3 = nextTimeUUID(); + TimeUUID key4 = nextTimeUUID(); + + assertArrayEquals(EMPTY, index.lookUp(key0)); + assertArrayEquals(EMPTY, index.lookUp(key1)); + assertArrayEquals(EMPTY, index.lookUp(key2)); + assertArrayEquals(EMPTY, index.lookUp(key3)); + assertArrayEquals(EMPTY, index.lookUp(key4)); + + int val11 = 1100; + int val21 = 2100; + int val22 = 2200; + int val31 = 3100; + int val32 = 3200; + int val33 = 3300; + + index.update(key1, val11); + index.update(key2, val21); + index.update(key2, val22); + index.update(key3, val31); + index.update(key3, val32); + index.update(key3, val33); + + assertArrayEquals(EMPTY, index.lookUp(key0)); + assertArrayEquals(new int[] { val11 }, index.lookUp(key1)); + assertArrayEquals(new int[] { val21, val22 }, index.lookUp(key2)); + assertArrayEquals(new int[] { val31, val32, val33 }, index.lookUp(key3)); + assertArrayEquals(EMPTY, index.lookUp(key4)); + + assertEquals(key1, index.firstId()); + assertEquals(key3, index.lastId()); + + assertFalse(index.mayContainId(key0)); + assertTrue(index.mayContainId(key1)); + assertTrue(index.mayContainId(key2)); + assertTrue(index.mayContainId(key3)); + assertFalse(index.mayContainId(key4)); + } + + @Test + public void testInMemoryIndexPersists() throws IOException + { + InMemoryIndex inMemory = InMemoryIndex.create(TimeUUIDKeySupport.INSTANCE); + + TimeUUID key0 = nextTimeUUID(); + TimeUUID key1 = nextTimeUUID(); + TimeUUID key2 = nextTimeUUID(); + TimeUUID key3 = nextTimeUUID(); + TimeUUID key4 = nextTimeUUID(); + + int val11 = 1100; + int val21 = 2100; + int val22 = 2200; + int val31 = 3100; + int val32 = 3200; + int val33 = 3300; + + inMemory.update(key1, val11); + inMemory.update(key2, val21); + inMemory.update(key2, val22); + inMemory.update(key3, val31); + inMemory.update(key3, val32); + inMemory.update(key3, val33); + + File directory = new File(Files.createTempDirectory(null)); + directory.deleteOnExit(); + Descriptor descriptor = Descriptor.create(directory, System.currentTimeMillis(), 1); + inMemory.persist(descriptor); + + try (OnDiskIndex onDisk = OnDiskIndex.open(descriptor, TimeUUIDKeySupport.INSTANCE)) + { + assertArrayEquals(EMPTY, onDisk.lookUp(key0)); + assertArrayEquals(new int[] { val11 }, onDisk.lookUp(key1)); + assertArrayEquals(new int[] { val21, val22 }, onDisk.lookUp(key2)); + assertArrayEquals(new int[] { val31, val32, val33 }, onDisk.lookUp(key3)); + assertArrayEquals(EMPTY, onDisk.lookUp(key4)); + + assertEquals(key1, onDisk.firstId()); + assertEquals(key3, onDisk.lastId()); + + assertFalse(onDisk.mayContainId(key0)); + assertTrue(onDisk.mayContainId(key1)); + assertTrue(onDisk.mayContainId(key2)); + assertTrue(onDisk.mayContainId(key3)); + assertFalse(onDisk.mayContainId(key4)); + } + } + + @Test + public void prop() throws IOException + { + Constraint sizeConstraint = Constraint.between(1, 10); + Constraint valueSizeConstraint = Constraint.between(0, 10); + Constraint positionConstraint = Constraint.between(0, Integer.MAX_VALUE); + Gen keyGen = Generators.timeUUID(); + Gen valueGen = rs -> { + int[] array = new int[(int) rs.next(valueSizeConstraint)]; + IntHashSet uniq = new IntHashSet(); + for (int i = 0; i < array.length; i++) + { + int value = (int) rs.next(positionConstraint); + while (!uniq.add(value)) + value = (int) rs.next(positionConstraint); + array[i] = value; + } + return array; + }; + Gen> gen = rs -> { + int size = (int) rs.next(sizeConstraint); + Map map = Maps.newHashMapWithExpectedSize(size); + for (int i = 0; i < size; i++) + { + TimeUUID key = keyGen.generate(rs); + while (map.containsKey(key)) + key = keyGen.generate(rs); + int[] value = valueGen.generate(rs); + map.put(key, value); + } + return map; + }; + gen = gen.describedAs(map -> { + StringBuilder sb = new StringBuilder(); + for (Map.Entry entry : map.entrySet()) + sb.append('\n').append(entry.getKey()).append('\t').append(Arrays.toString(entry.getValue())); + return sb.toString(); + }); + File directory = new File(Files.createTempDirectory(null)); + directory.deleteOnExit(); + qt().forAll(gen).checkAssert(map -> test(directory, map)); + } + + private static void test(File directory, Map map) + { + InMemoryIndex inMemory = InMemoryIndex.create(TimeUUIDKeySupport.INSTANCE); + for (Map.Entry e : map.entrySet()) + { + TimeUUID key = e.getKey(); + assertThat(inMemory.lookUp(key)).isEmpty(); + + int[] value = e.getValue(); + if (value.length == 0) + continue; + for (int i : value) + inMemory.update(key, i); + Arrays.sort(value); + } + assertIndex(map, inMemory); + + Descriptor descriptor = Descriptor.create(directory, System.nanoTime(), 1); + inMemory.persist(descriptor); + + try (OnDiskIndex onDisk = OnDiskIndex.open(descriptor, TimeUUIDKeySupport.INSTANCE)) + { + assertIndex(map, onDisk); + } + } + + private static void assertIndex(Map expected, Index actual) + { + List keys = expected.entrySet() + .stream() + .filter(e -> e.getValue().length != 0) + .map(Map.Entry::getKey) + .sorted() + .collect(Collectors.toList()); + + if (keys.isEmpty()) + { + assertThat(actual.firstId()).describedAs("Index %s had wrong firstId", actual).isNull(); + assertThat(actual.lastId()).describedAs("Index %s had wrong lastId", actual).isNull(); + } + else + { + assertThat(actual.firstId()).describedAs("Index %s had wrong firstId", actual).isEqualTo(keys.get(0)); + assertThat(actual.lastId()).describedAs("Index %s had wrong lastId", actual).isEqualTo(keys.get(keys.size() - 1)); + } + + for (Map.Entry e : expected.entrySet()) + { + TimeUUID key = e.getKey(); + int[] value = e.getValue(); + int[] read = actual.lookUp(key); + + if (value.length == 0) + { + assertThat(read).describedAs("Index %s returned wrong values for %s", actual, key).isEmpty(); + } + else + { + assertThat(read).describedAs("Index %s returned wrong values for %s", actual, key).isEqualTo(value); + assertThat(actual.mayContainId(key)).describedAs("Index %s expected %s to exist", key, actual).isTrue(); + } + } + } +} \ No newline at end of file diff --git a/test/unit/org/apache/cassandra/journal/JournalTest.java b/test/unit/org/apache/cassandra/journal/JournalTest.java new file mode 100644 index 0000000000..cb2dd33911 --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/JournalTest.java @@ -0,0 +1,102 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.file.Files; +import java.util.Collections; + +import org.junit.BeforeClass; +import org.junit.Test; + +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.utils.TimeUUID; + +import static org.junit.Assert.assertEquals; +import static org.apache.cassandra.utils.TimeUUID.Generator.nextTimeUUID; + +public class JournalTest +{ + @BeforeClass + public static void setUp() + { + DatabaseDescriptor.daemonInitialization(); + } + + @Test + public void testSimpleReadWrite() throws IOException + { + File directory = new File(Files.createTempDirectory("JournalTest")); + directory.deleteRecursiveOnExit(); + + Journal journal = + new Journal<>("TestJournal", directory, TestParams.INSTANCE, TimeUUIDKeySupport.INSTANCE, LongSerializer.INSTANCE); + + journal.start(); + + TimeUUID id1 = nextTimeUUID(); + TimeUUID id2 = nextTimeUUID(); + TimeUUID id3 = nextTimeUUID(); + TimeUUID id4 = nextTimeUUID(); + + journal.write(id1, 1L, Collections.singleton(1)); + journal.write(id2, 2L, Collections.singleton(1)); + journal.write(id3, 3L, Collections.singleton(1)); + journal.write(id4, 4L, Collections.singleton(1)); + + assertEquals(1L, (long) journal.read(id1)); + assertEquals(2L, (long) journal.read(id2)); + assertEquals(3L, (long) journal.read(id3)); + assertEquals(4L, (long) journal.read(id4)); + + journal.shutdown(); + + journal = new Journal<>("TestJournal", directory, TestParams.INSTANCE, TimeUUIDKeySupport.INSTANCE, LongSerializer.INSTANCE); + journal.start(); + + assertEquals(1L, (long) journal.read(id1)); + assertEquals(2L, (long) journal.read(id2)); + assertEquals(3L, (long) journal.read(id3)); + assertEquals(4L, (long) journal.read(id4)); + + journal.shutdown(); + } + + static class LongSerializer implements ValueSerializer + { + static final LongSerializer INSTANCE = new LongSerializer(); + + public int serializedSize(Long value, int userVersion) + { + return Long.BYTES; + } + + public void serialize(Long value, DataOutputPlus out, int userVersion) throws IOException + { + out.writeLong(value); + } + + public Long deserialize(TimeUUID key, DataInputPlus in, int userVersion) throws IOException + { + return in.readLong(); + } + } +} diff --git a/test/unit/org/apache/cassandra/journal/MetadataTest.java b/test/unit/org/apache/cassandra/journal/MetadataTest.java new file mode 100644 index 0000000000..a10aaff82c --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/MetadataTest.java @@ -0,0 +1,105 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.Arrays; +import java.util.HashSet; +import java.util.Random; +import java.util.Set; + +import org.junit.Test; + +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.io.util.DataOutputBuffer; + +import static org.junit.Assert.assertEquals; + +public class MetadataTest +{ + @Test + public void testUpdate() + { + Random rng = new Random(); + + int host1 = rng.nextInt(); + int host2 = host1 + 1; + int host3 = host2 + 1; + int host4 = host3 + 1; + int host5 = host4 + 1; + + Metadata metadata = Metadata.create(); + + metadata.update(set(host1)); + metadata.update(set(host2, host3)); + metadata.update(set(host1, host4)); + metadata.update(set(host1, host2, host3, host4)); + + assertEquals(set(host1, host2, host3, host4), metadata.hosts()); + assertEquals(3, metadata.count(host1)); + assertEquals(2, metadata.count(host2)); + assertEquals(2, metadata.count(host3)); + assertEquals(2, metadata.count(host4)); + assertEquals(0, metadata.count(host5)); + assertEquals(4, metadata.totalCount()); + } + + @Test + public void testWriteRead() throws IOException + { + Random rng = new Random(); + + int host1 = rng.nextInt(); + int host2 = host1 + 1; + int host3 = host2 + 1; + int host4 = host3 + 1; + int host5 = host4 + 1; + + Metadata metadata = Metadata.create(); + + metadata.update(set(host1)); + metadata.update(set(host2, host3)); + metadata.update(set(host1, host4)); + metadata.update(set(host1, host2, host3, host4)); + + try (DataOutputBuffer out = DataOutputBuffer.scratchBuffer.get()) + { + metadata.write(out); + ByteBuffer serialized = out.buffer(); + + try (DataInputBuffer in = new DataInputBuffer(serialized, false)) + { + Metadata deserialized = Metadata.read(in); + + assertEquals(set(host1, host2, host3, host4), deserialized.hosts()); + assertEquals(3, deserialized.count(host1)); + assertEquals(2, deserialized.count(host2)); + assertEquals(2, deserialized.count(host3)); + assertEquals(2, deserialized.count(host4)); + assertEquals(0, deserialized.count(host5)); + assertEquals(4, deserialized.totalCount()); + } + } + } + + private static Set set(Integer... ids) + { + return new HashSet<>(Arrays.asList(ids)); + } +} diff --git a/test/unit/org/apache/cassandra/journal/SegmentTest.java b/test/unit/org/apache/cassandra/journal/SegmentTest.java new file mode 100644 index 0000000000..0294cfef09 --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/SegmentTest.java @@ -0,0 +1,241 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.nio.file.Files; +import java.util.*; + +import org.junit.Test; + +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.io.util.File; +import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.TimeUUID; + +import static org.apache.cassandra.utils.TimeUUID.Generator.nextTimeUUID; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertTrue; + +public class SegmentTest +{ + @Test + public void testWriteReadActiveSegment() throws IOException + { + TimeUUID id1 = nextTimeUUID(); + TimeUUID id2 = nextTimeUUID(); + TimeUUID id3 = nextTimeUUID(); + TimeUUID id4 = nextTimeUUID(); + + ByteBuffer record1 = ByteBufferUtil.bytes("sample record 1"); + ByteBuffer record2 = ByteBufferUtil.bytes("sample record 2"); + ByteBuffer record3 = ByteBufferUtil.bytes("sample record 3"); + ByteBuffer record4 = ByteBufferUtil.bytes("sample record 4"); + + Random rng = new Random(); + + int host1 = rng.nextInt(); + int host2 = rng.nextInt(); + int host3 = rng.nextInt(); + int host4 = rng.nextInt(); + + Set hosts1 = set(host1); + Set hosts2 = set(host1, host2); + Set hosts3 = set(host1, host2, host3); + Set hosts4 = set(host4); + + File directory = new File(Files.createTempDirectory(null)); + directory.deleteRecursiveOnExit(); + + Descriptor descriptor = Descriptor.create(directory, System.currentTimeMillis(), 1); + + ActiveSegment segment = ActiveSegment.create(descriptor, params(), TimeUUIDKeySupport.INSTANCE); + + segment.allocate(record1.remaining(), hosts1).write(id1, record1, hosts1); + segment.allocate(record2.remaining(), hosts2).write(id2, record2, hosts2); + segment.allocate(record3.remaining(), hosts3).write(id3, record3, hosts3); + segment.allocate(record4.remaining(), hosts4).write(id4, record4, hosts4); + + // read all 4 entries by id and compare with originals + EntrySerializer.EntryHolder holder = new EntrySerializer.EntryHolder<>(); + + segment.read(id1, holder); + assertEquals(id1, holder.key); + assertEquals(hosts1, holder.hosts); + assertEquals(record1, holder.value); + + segment.read(id2, holder); + assertEquals(id2, holder.key); + assertEquals(hosts2, holder.hosts); + assertEquals(record2, holder.value); + + segment.read(id3, holder); + assertEquals(id3, holder.key); + assertEquals(hosts3, holder.hosts); + assertEquals(record3, holder.value); + + segment.read(id4, holder); + assertEquals(id4, holder.key); + assertEquals(hosts4, holder.hosts); + assertEquals(record4, holder.value); + } + + @Test + public void testReadClosedSegmentByID() throws IOException + { + DatabaseDescriptor.daemonInitialization(); + + TimeUUID id1 = nextTimeUUID(); + TimeUUID id2 = nextTimeUUID(); + TimeUUID id3 = nextTimeUUID(); + TimeUUID id4 = nextTimeUUID(); + + ByteBuffer record1 = ByteBufferUtil.bytes("sample record 1"); + ByteBuffer record2 = ByteBufferUtil.bytes("sample record 2"); + ByteBuffer record3 = ByteBufferUtil.bytes("sample record 3"); + ByteBuffer record4 = ByteBufferUtil.bytes("sample record 4"); + + Random rng = new Random(); + + int host1 = rng.nextInt(); + int host2 = rng.nextInt(); + int host3 = rng.nextInt(); + int host4 = rng.nextInt(); + + Set hosts1 = set(host1); + Set hosts2 = set(host1, host2); + Set hosts3 = set(host1, host2, host3); + Set hosts4 = set(host4); + + File directory = new File(Files.createTempDirectory(null)); + directory.deleteRecursiveOnExit(); + + Descriptor descriptor = Descriptor.create(directory, System.currentTimeMillis(), 1); + + ActiveSegment activeSegment = ActiveSegment.create(descriptor, params(), TimeUUIDKeySupport.INSTANCE); + + activeSegment.allocate(record1.remaining(), hosts1).write(id1, record1, hosts1); + activeSegment.allocate(record2.remaining(), hosts2).write(id2, record2, hosts2); + activeSegment.allocate(record3.remaining(), hosts3).write(id3, record3, hosts3); + activeSegment.allocate(record4.remaining(), hosts4).write(id4, record4, hosts4); + + activeSegment.close(); + + StaticSegment staticSegment = StaticSegment.open(descriptor, TimeUUIDKeySupport.INSTANCE); + + // read all 4 entries by id and compare with originals + EntrySerializer.EntryHolder holder = new EntrySerializer.EntryHolder<>(); + + staticSegment.read(id1, holder); + assertEquals(id1, holder.key); + assertEquals(hosts1, holder.hosts); + assertEquals(record1, holder.value); + + staticSegment.read(id2, holder); + assertEquals(id2, holder.key); + assertEquals(hosts2, holder.hosts); + assertEquals(record2, holder.value); + + staticSegment.read(id3, holder); + assertEquals(id3, holder.key); + assertEquals(hosts3, holder.hosts); + assertEquals(record3, holder.value); + + staticSegment.read(id4, holder); + assertEquals(id4, holder.key); + assertEquals(hosts4, holder.hosts); + assertEquals(record4, holder.value); + } + + @Test + public void testReadClosedSegmentSequentially() throws IOException + { + TimeUUID id1 = nextTimeUUID(); + TimeUUID id2 = nextTimeUUID(); + TimeUUID id3 = nextTimeUUID(); + TimeUUID id4 = nextTimeUUID(); + + ByteBuffer record1 = ByteBufferUtil.bytes("sample record 1"); + ByteBuffer record2 = ByteBufferUtil.bytes("sample record 2"); + ByteBuffer record3 = ByteBufferUtil.bytes("sample record 3"); + ByteBuffer record4 = ByteBufferUtil.bytes("sample record 4"); + + Random rng = new Random(); + + int host1 = rng.nextInt(); + int host2 = rng.nextInt(); + int host3 = rng.nextInt(); + int host4 = rng.nextInt(); + + Set hosts1 = set(host1); + Set hosts2 = set(host1, host2); + Set hosts3 = set(host1, host2, host3); + Set hosts4 = set(host4); + + File directory = new File(Files.createTempDirectory(null)); + directory.deleteRecursiveOnExit(); + + Descriptor descriptor = Descriptor.create(directory, System.currentTimeMillis(), 1); + + ActiveSegment activeSegment = ActiveSegment.create(descriptor, params(), TimeUUIDKeySupport.INSTANCE); + + activeSegment.allocate(record1.remaining(), hosts1).write(id1, record1, hosts1); + activeSegment.allocate(record2.remaining(), hosts2).write(id2, record2, hosts2); + activeSegment.allocate(record3.remaining(), hosts3).write(id3, record3, hosts3); + activeSegment.allocate(record4.remaining(), hosts4).write(id4, record4, hosts4); + + activeSegment.close(); + + StaticSegment.SequentialReader reader = StaticSegment.reader(descriptor, TimeUUIDKeySupport.INSTANCE, 0); + + // read all 4 entries sequentially and compare with originals + assertTrue(reader.advance()); + assertEquals(id1, reader.id()); + assertEquals(hosts1, reader.hosts()); + assertEquals(record1, reader.record()); + + assertTrue(reader.advance()); + assertEquals(id2, reader.id()); + assertEquals(hosts2, reader.hosts()); + assertEquals(record2, reader.record()); + + assertTrue(reader.advance()); + assertEquals(id3, reader.id()); + assertEquals(hosts3, reader.hosts()); + assertEquals(record3, reader.record()); + + assertTrue(reader.advance()); + assertEquals(id4, reader.id()); + assertEquals(hosts4, reader.hosts()); + assertEquals(record4, reader.record()); + + assertFalse(reader.advance()); + } + + private static Set set(Integer... ids) + { + return new HashSet<>(Arrays.asList(ids)); + } + + private static Params params() + { + return TestParams.INSTANCE; + } +} diff --git a/test/unit/org/apache/cassandra/journal/SyncedOffsetsTest.java b/test/unit/org/apache/cassandra/journal/SyncedOffsetsTest.java new file mode 100644 index 0000000000..5b83ee88f1 --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/SyncedOffsetsTest.java @@ -0,0 +1,70 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.file.Files; + +import org.junit.BeforeClass; +import org.junit.Test; + +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.io.util.File; + +import static org.junit.Assert.assertEquals; + +public class SyncedOffsetsTest +{ + @BeforeClass + public static void setUp() + { + DatabaseDescriptor.clientInitialization(); + } + + @Test + public void testCommonCase() throws IOException + { + testReadWrite(512, true); + testReadWrite(512, false); + } + + @Test + public void testResize() throws IOException + { + testReadWrite(2048, true); + testReadWrite(2048, false); + } + + private void testReadWrite(int n, boolean syncOnMark) throws IOException + { + File directory = new File(Files.createTempDirectory(null)); + directory.deleteOnExit(); + + Descriptor descriptor = Descriptor.create(directory, System.currentTimeMillis(), 1); + + SyncedOffsets active = SyncedOffsets.active(descriptor, syncOnMark); + for (int i = 0; i < n; i++) + active.mark(i); + assertEquals(n - 1, active.syncedOffset()); + active.close(); + + SyncedOffsets loaded = SyncedOffsets.load(descriptor); + assertEquals(n - 1, loaded.syncedOffset()); + loaded.close(); + } +} diff --git a/test/unit/org/apache/cassandra/journal/TestParams.java b/test/unit/org/apache/cassandra/journal/TestParams.java new file mode 100644 index 0000000000..3beb378536 --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/TestParams.java @@ -0,0 +1,61 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import org.apache.cassandra.net.MessagingService; + +public class TestParams implements Params +{ + static final TestParams INSTANCE = new TestParams(); + + @Override + public int segmentSize() + { + return 32 << 20; + } + + @Override + public FailurePolicy failurePolicy() + { + return FailurePolicy.STOP; + } + + @Override + public FlushMode flushMode() + { + return FlushMode.GROUP; + } + + @Override + public int flushPeriod() + { + return 1000; + } + + @Override + public int periodicFlushLagBlock() + { + return 1500; + } + + @Override + public int userVersion() + { + return MessagingService.current_version; + } +} diff --git a/test/unit/org/apache/cassandra/journal/TimeUUIDKeySupport.java b/test/unit/org/apache/cassandra/journal/TimeUUIDKeySupport.java new file mode 100644 index 0000000000..3d04fad89b --- /dev/null +++ b/test/unit/org/apache/cassandra/journal/TimeUUIDKeySupport.java @@ -0,0 +1,85 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.journal; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.zip.Checksum; + +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.utils.TimeUUID; + +import static org.apache.cassandra.utils.FBUtilities.updateChecksumLong; + +class TimeUUIDKeySupport implements KeySupport +{ + static final TimeUUIDKeySupport INSTANCE = new TimeUUIDKeySupport(); + + @Override + public int serializedSize(int userVersion) + { + return 16; + } + + @Override + public void serialize(TimeUUID key, DataOutputPlus out, int userVersion) throws IOException + { + out.writeLong(key.uuidTimestamp()); + out.writeLong(key.lsb()); + } + + @Override + public TimeUUID deserialize(DataInputPlus in, int userVersion) throws IOException + { + long uuidTimestamp = in.readLong(); + long lsb = in.readLong(); + return new TimeUUID(uuidTimestamp, lsb); + } + + @Override + public TimeUUID deserialize(ByteBuffer buffer, int position, int userVersion) + { + long uuidTimestamp = buffer.getLong(position); + long lsb = buffer.getLong(position + 8); + return new TimeUUID(uuidTimestamp, lsb); + } + + @Override + public void updateChecksum(Checksum crc, TimeUUID key, int userVersion) + { + updateChecksumLong(crc, key.uuidTimestamp()); + updateChecksumLong(crc, key.lsb()); + } + + @Override + public int compareWithKeyAt(TimeUUID key, ByteBuffer buffer, int position, int userVersion) + { + long uuidTimestamp = buffer.getLong(position); + long lsb = buffer.getLong(position + 8); + return key.uuidTimestamp() != uuidTimestamp + ? Long.compare(key.uuidTimestamp(), uuidTimestamp) + : Long.compare(key.lsb(), lsb); + } + + @Override + public int compare(TimeUUID o1, TimeUUID o2) + { + return o1.compareTo(o2); + } +} diff --git a/test/unit/org/apache/cassandra/service/accord/AccordJournalTest.java b/test/unit/org/apache/cassandra/service/accord/AccordJournalTest.java new file mode 100644 index 0000000000..7059a12f2a --- /dev/null +++ b/test/unit/org/apache/cassandra/service/accord/AccordJournalTest.java @@ -0,0 +1,120 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.service.accord; + +import java.io.IOException; +import java.io.UncheckedIOException; +import java.nio.ByteBuffer; +import java.util.ArrayList; +import java.util.List; + +import org.junit.Test; + +import accord.primitives.TxnId; +import accord.utils.AccordGens; +import accord.utils.Gen; +import accord.utils.Gens; +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.io.util.DataOutputBuffer; +import org.apache.cassandra.service.accord.AccordJournal.Key; +import org.apache.cassandra.utils.AsymmetricOrdering; +import org.apache.cassandra.utils.FBUtilities; +import org.apache.cassandra.utils.FBUtilities.Order; +import org.checkerframework.checker.nullness.qual.Nullable; + +import static accord.utils.Property.qt; +import static org.assertj.core.api.Assertions.assertThat; + +public class AccordJournalTest +{ + @Test + public void keySerde() + { + DataOutputBuffer buffer = new DataOutputBuffer(); + qt().forAll(keyGen()).check(key -> + { + buffer.clear(); + int expectedSize = Key.SUPPORT.serializedSize(1); + Key.SUPPORT.serialize(key, buffer, 1); + assertThat(buffer.getLength()).isEqualTo(expectedSize); + try (DataInputBuffer input = new DataInputBuffer(buffer.unsafeGetBufferAndFlip(), false)) + { + Key read = Key.SUPPORT.deserialize(input, 1); + assertThat(read).isEqualTo(key); + } + }); + } + + @Test + public void compareKeys() + { + qt().forAll(Gens.lists(keyGen()).ofSizeBetween(2, 100)).check(keys -> + { + keys.sort(Key.SUPPORT); + + List buffers = new ArrayList<>(keys.size()); + for (Key k : keys) buffers.add(toBuffer(k)); + + for (int i = 0; i < keys.size(); i++) + { + Key outerKey = keys.get(i); + for (int j = 0; j < keys.size(); j++) + { + Key innerKey = keys.get(j); + ByteBuffer innerBuffer = buffers.get(j); + Order expected = FBUtilities.compare(outerKey, innerKey, Key.SUPPORT); + Order actual = FBUtilities.compare(outerKey, innerBuffer, new AsymmetricOrdering() + { + @Override + public int compareAsymmetric(Key left, ByteBuffer right) + { + return Key.SUPPORT.compareWithKeyAt(left, right, 0, 1); + } + + @Override + public int compare(@Nullable Key left, @Nullable Key right) + { + throw new UnsupportedOperationException(); + } + }); + assertThat(actual).isEqualTo(expected); + } + } + }); + } + + private static ByteBuffer toBuffer(Key k) + { + try (DataOutputBuffer buffer = new DataOutputBuffer(Key.SUPPORT.serializedSize(1))) + { + Key.SUPPORT.serialize(k, buffer, 1); + return buffer.unsafeGetBufferAndFlip(); + } + catch (IOException e) + { + throw new UncheckedIOException(e); + } + } + + private Gen keyGen() + { + Gen txnIdGen = AccordGens.txnIds(); + Gen typeGen = Gens.enums().all(AccordJournal.Type.class); + return rs -> new Key(txnIdGen.next(rs), typeGen.next(rs)); + } +}