From a813a51f7c4bc7379af3621e7386e6a6ca11884f Mon Sep 17 00:00:00 2001 From: Benedict Elliott Smith Date: Sat, 2 Aug 2025 16:44:55 +0100 Subject: [PATCH] Improve Observability: - Track all active Coordinations - Refactor Replica/Coordinator metrics and report Coordinator exhausted/preempted/timeout - DurabilityQueue metrics and visibility Also Fix: - WaitingState can get cause distributed stall when asked to wait for CanApply if not yet PreCommitted; track separate querying state and advance this to the next achievable state rather than the desired final state - Stalled coordinators should not prevent recovery - Edge case with fetch unable to make progress when pre-bootstrap and all peers have GC'd - Dependency initialisation for sync points across certain ownership changes - SyncPoint propagation may not include all of the epochs required on the receiving node for ranges they have lost but not closed, and receiving node does not validate them - Stable tracker accounting with LocalExecute - Do not prune non-durable APPLIED as must be reported in dependencies until durably applied (so as not to break recovery) - Ensure we cannot race with replies when initiating Coordination - ProgressLog does not guarantee to clear home or waiting states when erased or invalidated by compaction - WaitingState on non-home shard cannot guarantee progress once home shard is Erased - WaitingOnSync handles retired ranges incorrectly Also Improve: - Standardise failure accounting, use null to represent single reply timeouts - BurnTest record/replay to/from file patch by Benedict; reviewed by Alex Petrov for CASSANDRA-20878 --- modules/accord | 2 +- .../db/virtual/AccordDebugKeyspace.java | 128 +++++++- .../cassandra/metrics/AccordCacheMetrics.java | 99 ++++-- .../metrics/AccordClientRequestMetrics.java | 1 - ...ics.java => AccordCoordinatorMetrics.java} | 226 ++++++-------- .../cassandra/metrics/AccordMetricUtils.java | 63 ++++ .../metrics/AccordReplicaMetrics.java | 175 +++++++++++ .../metrics/AccordSystemMetrics.java | 233 ++++++++++++++ ...Reservoir.java => CassandraHistogram.java} | 22 +- .../metrics/CassandraMetricsRegistry.java | 76 +++-- .../cassandra/metrics/CassandraReservoir.java | 121 ++++++++ .../cassandra/metrics/ClearableHistogram.java | 6 +- .../cassandra/metrics/ClientMetrics.java | 6 +- .../DecayingEstimatedHistogramReservoir.java | 16 +- .../cassandra/metrics/LogLinearHistogram.java | 288 ++++++++++++++++++ .../cassandra/metrics/OnDemandHistogram.java | 59 ++++ .../cassandra/metrics/OverrideHistogram.java | 50 +++ .../cassandra/metrics/ScalingReservoir.java | 29 +- .../cassandra/metrics/ShardedHistogram.java | 146 +++++++++ .../cassandra/metrics/ShardedHitRate.java | 248 +++++++++++++++ .../cassandra/metrics/SnapshottingTimer.java | 6 +- .../cassandra/service/RetryStrategy.java | 5 + .../cassandra/service/TimeoutStrategy.java | 17 +- .../cassandra/service/accord/AccordCache.java | 81 ++--- .../service/accord/AccordCacheEntry.java | 1 + .../service/accord/AccordCommandStore.java | 2 +- .../service/accord/AccordCommandStores.java | 16 +- .../accord/AccordConfigurationService.java | 9 + .../service/accord/AccordExecutor.java | 138 ++++++++- .../AccordExecutorAbstractLockLoop.java | 5 +- .../AccordExecutorAbstractSemiSyncSubmit.java | 5 +- .../accord/AccordExecutorAsyncSubmit.java | 9 +- .../accord/AccordExecutorSemiSyncSubmit.java | 9 +- .../service/accord/AccordExecutorSimple.java | 17 +- .../accord/AccordExecutorSyncSubmit.java | 13 +- .../service/accord/AccordJournal.java | 2 +- .../service/accord/AccordMessageSink.java | 6 +- .../accord/AccordResponseVerbHandler.java | 3 +- .../service/accord/AccordResult.java | 158 +++++----- .../service/accord/AccordService.java | 13 +- .../service/accord/IAccordService.java | 15 +- .../service/accord/api/AccordAgent.java | 81 +++-- .../service/accord/events/CacheEvents.java | 10 +- .../accord/interop/AccordInteropAdapter.java | 4 +- .../accord/interop/AccordInteropApply.java | 8 - .../interop/AccordInteropExecution.java | 22 +- .../accord/interop/AccordInteropPersist.java | 22 +- .../repair/RequiredResponseTracker.java | 6 + .../accord/serializers/ApplySerializers.java | 62 ++-- .../accord/serializers/AwaitSerializers.java | 17 +- .../accord/serializers/DepsSerializers.java | 31 +- .../accord/serializers/FetchSerializers.java | 23 +- .../serializers/ReadDataSerializer.java | 19 +- .../serializers/TopologySerializers.java | 24 +- .../serializers/WaitingOnSerializer.java | 10 +- .../service/accord/txn/TxnWrite.java | 9 +- .../service/paxos/ContentionStrategy.java | 5 + .../apache/cassandra/tcm/RemoteProcessor.java | 3 +- .../transport/ClientResourceLimits.java | 26 +- .../cassandra/utils/EstimatedHistogram.java | 15 + ...alizer.java => LargeBitSetSerializer.java} | 18 +- .../test/accord/AccordBootstrapTest.java | 8 +- .../test/accord/AccordLoadTest.java | 3 +- .../test/accord/AccordMetricsTest.java | 58 ++-- .../AccordMigrationReadRaceTestBase.java | 11 +- .../AccordMigrationWriteRaceTestBase.java | 30 +- .../distributed/test/jmx/JMXTestsUtil.java | 3 +- .../service/accord/AccordJournalBurnTest.java | 2 +- .../apache/cassandra/metrics/HitRateTest.java | 60 ++++ .../metrics/JmxVirtualTableMetricsTest.java | 2 +- .../service/accord/AccordCacheTest.java | 147 +++++---- .../accord/AccordCommandStoreTest.java | 4 +- .../service/accord/AccordTestUtils.java | 3 +- .../accord/SimulatedAccordCommandStore.java | 3 +- .../CommandsForKeySerializerTest.java | 7 +- .../serializers/WaitingOnSerializerTest.java | 6 +- ...st.java => LargeBitSetSerializerTest.java} | 10 +- 77 files changed, 2572 insertions(+), 724 deletions(-) rename src/java/org/apache/cassandra/metrics/{AccordMetrics.java => AccordCoordinatorMetrics.java} (55%) create mode 100644 src/java/org/apache/cassandra/metrics/AccordMetricUtils.java create mode 100644 src/java/org/apache/cassandra/metrics/AccordReplicaMetrics.java create mode 100644 src/java/org/apache/cassandra/metrics/AccordSystemMetrics.java rename src/java/org/apache/cassandra/metrics/{SnapshottingReservoir.java => CassandraHistogram.java} (63%) create mode 100644 src/java/org/apache/cassandra/metrics/CassandraReservoir.java create mode 100644 src/java/org/apache/cassandra/metrics/LogLinearHistogram.java create mode 100644 src/java/org/apache/cassandra/metrics/OnDemandHistogram.java create mode 100644 src/java/org/apache/cassandra/metrics/OverrideHistogram.java create mode 100644 src/java/org/apache/cassandra/metrics/ShardedHistogram.java create mode 100644 src/java/org/apache/cassandra/metrics/ShardedHitRate.java rename src/java/org/apache/cassandra/utils/{SimpleBitSetSerializer.java => LargeBitSetSerializer.java} (78%) create mode 100644 test/unit/org/apache/cassandra/metrics/HitRateTest.java rename test/unit/org/apache/cassandra/utils/{SimpleBitSetSerializerTest.java => LargeBitSetSerializerTest.java} (84%) diff --git a/modules/accord b/modules/accord index 5cbe8d62f1..48061a521c 160000 --- a/modules/accord +++ b/modules/accord @@ -1 +1 @@ -Subproject commit 5cbe8d62f15cc7d66af2c7686f14b8fa52b1d35b +Subproject commit 48061a521cde7a085fe7a5503023c8ef8b687a89 diff --git a/src/java/org/apache/cassandra/db/virtual/AccordDebugKeyspace.java b/src/java/org/apache/cassandra/db/virtual/AccordDebugKeyspace.java index f7c89754e3..423a7b3f6c 100644 --- a/src/java/org/apache/cassandra/db/virtual/AccordDebugKeyspace.java +++ b/src/java/org/apache/cassandra/db/virtual/AccordDebugKeyspace.java @@ -41,6 +41,13 @@ import javax.annotation.Nullable; import com.google.common.collect.BoundType; import com.google.common.collect.Range; import com.google.common.collect.Sets; + +import accord.coordinate.AbstractCoordination; +import accord.coordinate.Coordination; +import accord.coordinate.Coordinations; +import accord.coordinate.PrepareRecovery; +import accord.coordinate.tracking.AbstractTracker; +import accord.utils.SortedListMap; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.db.EmptyIterators; import org.apache.cassandra.db.filter.ClusteringIndexFilter; @@ -57,7 +64,6 @@ import accord.api.TraceEventType; import accord.coordinate.FetchData; import accord.coordinate.FetchRoute; import accord.coordinate.MaybeRecover; -import accord.coordinate.RecoverWithRoute; import accord.impl.CommandChange; import accord.impl.progresslog.DefaultProgressLog; import accord.impl.progresslog.TxnStateKind; @@ -151,6 +157,8 @@ import static org.apache.cassandra.utils.MonotonicClock.Global.approxTime; public class AccordDebugKeyspace extends VirtualKeyspace { + public static final String COORDINATIONS = "coordinations"; + public static final String EXECUTORS = "executors"; public static final String COMMANDS_FOR_KEY = "commands_for_key"; public static final String COMMANDS_FOR_KEY_UNMANAGED = "commands_for_key_unmanaged"; public static final String DURABILITY_SERVICE = "durability_service"; @@ -173,6 +181,8 @@ public class AccordDebugKeyspace extends VirtualKeyspace private AccordDebugKeyspace() { super(VIRTUAL_ACCORD_DEBUG, List.of( + new ExecutorsTable(), + new CoordinationsTable(), new CommandsForKeyTable(), new CommandsForKeyUnmanagedTable(), new DurabilityServiceTable(), @@ -192,6 +202,118 @@ public class AccordDebugKeyspace extends VirtualKeyspace )); } + // TODO (desired): human readable packed key tracker (but requires loading Txn, so might be preferable to only do conditionally) + public static final class ExecutorsTable extends AbstractVirtualTable + { + private ExecutorsTable() + { + super(parse(VIRTUAL_ACCORD_DEBUG, EXECUTORS, + "Accord Executor State", + "CREATE TABLE %s (\n" + + " executor_id int,\n" + + " status text,\n" + + " position int,\n" + + " unique_position int,\n" + + " description text,\n" + + " command_store_id int,\n" + + " txn_id 'TxnIdUtf8Type',\n" + + " txn_id_additional 'TxnIdUtf8Type',\n" + + " keys text,\n" + + " keysLoad text,\n" + + " keysLoadFor text,\n" + + " PRIMARY KEY (executor_id, status, position, unique_position)" + + ')', UTF8Type.instance)); + } + + @Override + public DataSet data() + { + AccordCommandStores commandStores = (AccordCommandStores) AccordService.instance().node().commandStores(); + SimpleDataSet ds = new SimpleDataSet(metadata()); + + for (AccordExecutor executor : commandStores.executors()) + { + int uniquePos = 0; + int executorId = executor.executorId(); + AccordExecutor.TaskInfo prev = null; + for (AccordExecutor.TaskInfo info : executor.taskSnapshot()) + { + if (prev != null && info.status() == prev.status() && info.position() == prev.position()) ++uniquePos; + else uniquePos = 0; + prev = info; + PreLoadContext preLoadContext = info.preLoadContext(); + ds.row(executorId, info.status(), info.position(), uniquePos) + .column("description", info.describe()) + .column("command_store_id", info.commandStoreId()) + .column("txn_id", preLoadContext == null ? null : preLoadContext.primaryTxnId()) + .column("txn_id_additional", preLoadContext == null ? null : preLoadContext.additionalTxnId()) + .column("keys", preLoadContext == null ? null : preLoadContext.keys()) + .column("keysLoad", preLoadContext == null ? null : preLoadContext.loadKeys()) + .column("keysLoadFor", preLoadContext == null ? null : preLoadContext.loadKeysFor()) + ; + } + } + return ds; + } + } + + // TODO (desired): human readable packed key tracker (but requires loading Txn, so might be preferable to only do conditionally) + public static final class CoordinationsTable extends AbstractVirtualTable + { + private CoordinationsTable() + { + super(parse(VIRTUAL_ACCORD_DEBUG, COORDINATIONS, + "Accord Coordination State", + "CREATE TABLE %s (\n" + + " txn_id int,\n" + + " kind text,\n" + + " coordination_id int,\n" + + " description text,\n" + + " nodes text,\n" + + " nodes_inflight text,\n" + + " nodes_contacted text,\n" + + " participants text,\n" + + " replies text,\n" + + " tracker text,\n" + + " PRIMARY KEY (txn_id, kind, coordination_id)" + + ')', UTF8Type.instance)); + } + + @Override + public DataSet data() + { + Coordinations coordinations = AccordService.instance().node().coordinations(); + SimpleDataSet ds = new SimpleDataSet(metadata()); + for (Coordination c : coordinations) + { + ds.row(c.txnId(), c.kind().toString(), c.coordinationId()) + .column("nodes", toStringOrNull(c.nodes())) + .column("nodes_inflight", toStringOrNull(c.inflight())) + .column("nodes_contacted", toStringOrNull(c.contacted())) + .column("description", c.describe()) + .column("participants", toStringOrNull(c.scope())) + .column("replies", summarise(c.replies())) + .column("tracker", summarise(c.tracker())); + } + return ds; + } + + private static String summarise(@Nullable SortedListMap replies) + { + if (replies == null) + return null; + return AbstractCoordination.summariseReplies(replies, 60); + } + + private static String summarise(@Nullable AbstractTracker tracker) + { + if (tracker == null) + return null; + return tracker.summariseTracker(); + } + } + + // TODO (desired): don't report null as "null" public static final class CommandsForKeyTable extends AbstractVirtualTable implements AbstractVirtualTable.DataSet { @@ -529,7 +651,7 @@ public class AccordDebugKeyspace extends VirtualKeyspace private static void addRow(SimpleDataSet ds, int executorId, String scope, AccordCache.ImmutableStats stats) { ds.row(executorId, scope) - .column("queries", stats.queries) + .column("queries", stats.hits + stats.misses) .column("hits", stats.hits) .column("misses", stats.misses); } @@ -1365,7 +1487,7 @@ public class AccordDebugKeyspace extends VirtualKeyspace Node node = AccordService.instance().node(); if (Route.isFullRoute(route)) { - RecoverWithRoute.recover(node, node.someSequentialExecutor(), txnId, NotKnownToBeInvalid, (FullRoute) route, null, LatentStoreSelector.standard(), (success, fail) -> { + PrepareRecovery.recover(node, node.someSequentialExecutor(), txnId, NotKnownToBeInvalid, (FullRoute) route, null, LatentStoreSelector.standard(), (success, fail) -> { if (fail != null) result.setFailure(fail); else result.setSuccess(null); }, node.agent().trace(txnId, RECOVER)); diff --git a/src/java/org/apache/cassandra/metrics/AccordCacheMetrics.java b/src/java/org/apache/cassandra/metrics/AccordCacheMetrics.java index 1250c34751..5285bd914e 100644 --- a/src/java/org/apache/cassandra/metrics/AccordCacheMetrics.java +++ b/src/java/org/apache/cassandra/metrics/AccordCacheMetrics.java @@ -18,35 +18,94 @@ package org.apache.cassandra.metrics; -import java.util.Map; -import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.locks.Lock; +import java.util.function.Function; +import java.util.function.ToLongFunction; -import com.codahale.metrics.Histogram; +import com.codahale.metrics.Gauge; +import org.apache.cassandra.service.accord.AccordExecutor; +import org.apache.cassandra.service.accord.IAccordService; -import static org.apache.cassandra.metrics.CacheMetrics.TYPE_NAME; +import static org.apache.cassandra.metrics.AccordMetricUtils.fromAccordService; import static org.apache.cassandra.metrics.CassandraMetricsRegistry.Metrics; -public class AccordCacheMetrics extends CacheAccessMetrics +public class AccordCacheMetrics { - public static final String OBJECT_SIZE = "ObjectSize"; + public static final String ACCORD_CACHE = "AccordCache"; + public static final AccordCacheMetrics CommandsCacheMetrics = new AccordCacheMetrics("Commands"); + public static final AccordCacheMetrics CommandsForKeyCacheMetrics = new AccordCacheMetrics("CommandsForKey"); + public static final AccordCacheGlobalMetrics Global = new AccordCacheGlobalMetrics(); - public final Histogram objectSize; - - private final Map instanceMetrics = new ConcurrentHashMap<>(2); - - private final String scope; - - public AccordCacheMetrics(String scope) + // not sure why we create these wrapper objects that can only be instantiated once, but it's a pattern in this package so... + public static class AccordCacheGlobalMetrics { - super(new DefaultNameFactory(TYPE_NAME, scope)); - objectSize = Metrics.histogram(factory.createMetricName(OBJECT_SIZE), false); - this.scope = scope; + final Gauge usedBytes; + final Gauge unreferencedBytes; + + public AccordCacheGlobalMetrics() + { + DefaultNameFactory factory = new DefaultNameFactory("AccordCache"); + this.usedBytes = Metrics.gauge(factory.createMetricName("UsedBytes"), fromAccordService(sumExecutors(executor -> executor.cacheUnsafe().weightedSize()), 0L)); + this.unreferencedBytes = Metrics.gauge(factory.createMetricName("UnreferencedBytes"), fromAccordService(sumExecutors(executor -> executor.cacheUnsafe().unreferencedBytes()), 0L)); + } + + private static Function sumExecutors(ToLongFunction f) + { + return service -> { + long sum = 0; + for (AccordExecutor executor : service.executors()) + sum += f.applyAsLong(executor); + return sum; + }; + } } - public CacheAccessMetrics forInstance(Class klass) + public static class Shard { - // cannot make Class hashCode deterministic, as cannot rewrite - so cannot safely use as Map key if want deterministic simulation - // (or we need to create extra hoops to catch this specific case in method rewriting) - return instanceMetrics.computeIfAbsent(klass.getSimpleName(), k -> new CacheAccessMetrics(new DefaultNameFactory(TYPE_NAME, String.format("%s-%s", scope, k)))); + public final ShardedHitRate.HitRateShard hitRate; + public final LogLinearHistogram objectSize; + + public Shard(ShardedHitRate.HitRateShard hitRate, LogLinearHistogram objectSize) + { + this.hitRate = hitRate; + this.objectSize = objectSize; + } + } + + public final ShardedHitRate hitRate = new ShardedHitRate(); + public final ShardedHistogram objectSize; + + public final Gauge hits; + public final Gauge misses; + public final Gauge requests; + public final Gauge requestRate1m; + public final Gauge requestRate5m; + public final Gauge requestRate15m; + public final Gauge hitRateAllTime; + public final Gauge hitRate1m; + public final Gauge hitRate5m; + public final Gauge hitRate15m; + private final String subTypeName; + + public AccordCacheMetrics(String subTypeName) + { + DefaultNameFactory factory = new DefaultNameFactory("AccordCache", subTypeName); + this.objectSize = Metrics.shardedHistogram(factory.createMetricName("EntrySize")); + this.hits = Metrics.gauge(factory.createMetricName("Hits"), hitRate::totalHits); + this.misses = Metrics.gauge(factory.createMetricName("Misses"), hitRate::totalMisses); + this.requests = Metrics.gauge(factory.createMetricName("Requests"), hitRate::totalRequests); + this.requestRate1m = Metrics.gauge(factory.createMetricName("Requests"), () -> hitRate.requestsPerSecond(1)); + this.requestRate5m = Metrics.gauge(factory.createMetricName("Requests"), () -> hitRate.requestsPerSecond(5)); + this.requestRate15m = Metrics.gauge(factory.createMetricName(RatioGaugeSet.FIFTEEN_MINUTE + "RequestRate"), () -> hitRate.requestsPerSecond(15)); + this.hitRate1m = Metrics.gauge(factory.createMetricName(RatioGaugeSet.ONE_MINUTE + "HitRate"), () -> hitRate.hitRate(1)); + this.hitRate5m = Metrics.gauge(factory.createMetricName(RatioGaugeSet.FIVE_MINUTE + "HitRate"), () -> hitRate.hitRate(5)); + this.hitRate15m = Metrics.gauge(factory.createMetricName(RatioGaugeSet.FIFTEEN_MINUTE + "HitRate"), () -> hitRate.hitRate(15)); + this.hitRateAllTime = Metrics.gauge(factory.createMetricName("Misses"), hitRate::hitRateAllTime); + this.subTypeName = subTypeName; + } + + public Shard newShard(Lock guardedBy) + { + return new Shard(hitRate.newShard(guardedBy), objectSize.newShard(guardedBy)); } } diff --git a/src/java/org/apache/cassandra/metrics/AccordClientRequestMetrics.java b/src/java/org/apache/cassandra/metrics/AccordClientRequestMetrics.java index 3dc3509dff..5acbec3655 100644 --- a/src/java/org/apache/cassandra/metrics/AccordClientRequestMetrics.java +++ b/src/java/org/apache/cassandra/metrics/AccordClientRequestMetrics.java @@ -72,5 +72,4 @@ public class AccordClientRequestMetrics extends ClientRequestMetrics Metrics.remove(factory.createMetricName("Preempted")); Metrics.remove(factory.createMetricName("TopologyMismatches")); } - } diff --git a/src/java/org/apache/cassandra/metrics/AccordMetrics.java b/src/java/org/apache/cassandra/metrics/AccordCoordinatorMetrics.java similarity index 55% rename from src/java/org/apache/cassandra/metrics/AccordMetrics.java rename to src/java/org/apache/cassandra/metrics/AccordCoordinatorMetrics.java index 4a5d245312..f062e95076 100644 --- a/src/java/org/apache/cassandra/metrics/AccordMetrics.java +++ b/src/java/org/apache/cassandra/metrics/AccordCoordinatorMetrics.java @@ -19,50 +19,34 @@ package org.apache.cassandra.metrics; import java.lang.reflect.Field; -import java.util.concurrent.TimeUnit; -import java.util.concurrent.atomic.AtomicLong; - import javax.annotation.Nullable; import accord.api.CoordinatorEventListener; -import accord.api.LocalEventListener; import accord.api.Result; import accord.coordinate.ExecutePath; -import accord.impl.progresslog.DefaultProgressLog; -import accord.local.Command; import accord.local.Node; -import accord.local.SafeCommandStore; import accord.primitives.Ballot; import accord.primitives.Deps; -import accord.primitives.PartialDeps; -import accord.primitives.Timestamp; import accord.primitives.TxnId; import com.codahale.metrics.Counting; -import com.codahale.metrics.Gauge; import com.codahale.metrics.Histogram; import com.codahale.metrics.Meter; import com.codahale.metrics.Timer; -import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.api.AccordTimeService; import org.apache.cassandra.tracing.Tracing; import static java.util.concurrent.TimeUnit.MICROSECONDS; import static org.apache.cassandra.metrics.CassandraMetricsRegistry.Metrics; -public class AccordMetrics +public class AccordCoordinatorMetrics { - public final static AccordMetrics readMetrics = new AccordMetrics("ro"); - public final static AccordMetrics writeMetrics = new AccordMetrics("rw"); - - public static final String ACCORD_REPLICA = "AccordReplica"; - public static final String REPLICA_STABLE_LATENCY = "StableLatency"; - public static final String REPLICA_PREAPPLY_LATENCY = "PreApplyLatency"; - public static final String REPLICA_APPLY_LATENCY = "ApplyLatency"; - public static final String REPLICA_APPLY_DURATION = "ApplyDuration"; - public static final String REPLICA_DEPENDENCIES = "Dependencies"; - public static final String PROGRESS_LOG_SIZE = "ProgressLogSize"; + public final static AccordCoordinatorMetrics readMetrics = new AccordCoordinatorMetrics("ro"); + public final static AccordCoordinatorMetrics writeMetrics = new AccordCoordinatorMetrics("rw"); public static final String ACCORD_COORDINATOR = "AccordCoordinator"; + public static final String COORDINATOR_EPOCHS = "Epochs"; + public static final String COORDINATOR_KEYS = "Keys"; + public static final String COORDINATOR_TABLES = "Tables"; public static final String COORDINATOR_DEPENDENCIES = "Dependencies"; public static final String COORDINATOR_PREACCEPT_LATENCY = "PreAcceptLatency"; public static final String COORDINATOR_EXECUTE_LATENCY = "ExecuteLatency"; @@ -71,60 +55,47 @@ public class AccordMetrics public static final String MEDIUM_PATHS = "MediumPaths"; public static final String SLOW_PATHS = "SlowPaths"; public static final String PREEMPTED = "Preempted"; + public static final String REJECTED = "Rejected"; public static final String TIMEOUTS = "Timeouts"; public static final String INVALIDATIONS = "Invalidations"; public static final String RECOVERY_DELAY = "RecoveryDelay"; public static final String RECOVERY_TIME = "RecoveryTime"; public static final String FAST_PATH_TO_TOTAL = "FastPathToTotal"; - /** - * The time between start on the coordinator and commit on this replica. - */ - public final Timer replicaStableLatency; - - /** - * The time between start on the coordinator and arrival of the result on this replica. - */ - public final Timer replicaPreapplyLatency; - - /** - * The time between start on the coordinator and application on this replica. - */ - public final Timer replicaApplyLatency; - - /** - * TODO (expected): probably more interesting is latency from preapplied to apply; - * we already track local write latencies, whch this effectively duplicates (but including queueing latencies) - * Duration of applying changes. - */ - public final Timer replicaApplyDuration; - - /** - * A histogram of the number of dependencies per transaction at this replica. - */ - public final Histogram replicaDependencies; - - public final Gauge progressLogSize; - /** * A histogram of the number of dependencies per transaction at this coordinator. */ - public final Histogram coordinatorDependencies; + public final Histogram dependencies; /** * A histogram of the time to preaccept on this coordinator */ - public final Histogram coordinatorPreacceptLatency; + public final Histogram preacceptLatency; /** * A histogram of the time to begin execution on this coordinator */ - public final Histogram coordinatorExecuteLatency; + public final Histogram executeLatency; /** * A histogram of the time to complete execution on this coordinator */ - public final Histogram coordinatorApplyLatency; + public final Histogram applyLatency; + + /** + * The number of epochs used to coordinate the transaction + */ + public final Histogram epochs; + + /** + * The number of keys involved in a transaction + */ + public final Histogram keys; + + /** + * The number of tables involved in a transaction + */ + public final Histogram tables; /** * The number of fast path transactions executed on this coordinator. @@ -146,6 +117,11 @@ public class AccordMetrics */ public final Meter preempted; + /** + * The number of preempted transactions on this coordinator. + */ + public final Meter rejected; + /** * The number of timed out transactions on this coordinator. */ @@ -171,31 +147,22 @@ public class AccordMetrics */ public final RatioGaugeSet fastPathToTotal; - private AccordMetrics(String scope) + private AccordCoordinatorMetrics(String scope) { - DefaultNameFactory replica = new DefaultNameFactory(ACCORD_REPLICA, scope); - replicaStableLatency = Metrics.timer(replica.createMetricName(REPLICA_STABLE_LATENCY)); - replicaPreapplyLatency = Metrics.timer(replica.createMetricName(REPLICA_PREAPPLY_LATENCY)); - replicaApplyLatency = Metrics.timer(replica.createMetricName(REPLICA_APPLY_LATENCY)); - replicaApplyDuration = Metrics.timer(replica.createMetricName(REPLICA_APPLY_DURATION)); - replicaDependencies = Metrics.histogram(replica.createMetricName(REPLICA_DEPENDENCIES), true); - progressLogSize = Metrics.gauge(replica.createMetricName(PROGRESS_LOG_SIZE), () -> { - AtomicLong i = new AtomicLong(); - AccordService.instance().node().commandStores().forEachCommandStore(store -> { - i.addAndGet(((DefaultProgressLog)store.unsafeProgressLog()).size()); - }); - return i.get(); - }); - DefaultNameFactory coordinator = new DefaultNameFactory(ACCORD_COORDINATOR, scope); - coordinatorDependencies = Metrics.histogram(coordinator.createMetricName(COORDINATOR_DEPENDENCIES), true); - coordinatorPreacceptLatency = Metrics.histogram(coordinator.createMetricName(COORDINATOR_PREACCEPT_LATENCY), true); - coordinatorExecuteLatency = Metrics.histogram(coordinator.createMetricName(COORDINATOR_EXECUTE_LATENCY), true); - coordinatorApplyLatency = Metrics.histogram(coordinator.createMetricName(COORDINATOR_APPLY_LATENCY), true); + dependencies = Metrics.histogram(coordinator.createMetricName(COORDINATOR_DEPENDENCIES), true); + preacceptLatency = Metrics.histogram(coordinator.createMetricName(COORDINATOR_PREACCEPT_LATENCY), true); + executeLatency = Metrics.histogram(coordinator.createMetricName(COORDINATOR_EXECUTE_LATENCY), true); + applyLatency = Metrics.histogram(coordinator.createMetricName(COORDINATOR_APPLY_LATENCY), true); + epochs = Metrics.histogram(coordinator.createMetricName(COORDINATOR_EPOCHS), true); + keys = Metrics.histogram(coordinator.createMetricName(COORDINATOR_KEYS), true); + tables = Metrics.histogram(coordinator.createMetricName(COORDINATOR_TABLES), true); + fastPaths = Metrics.meter(coordinator.createMetricName(FAST_PATHS)); mediumPaths = Metrics.meter(coordinator.createMetricName(MEDIUM_PATHS)); slowPaths = Metrics.meter(coordinator.createMetricName(SLOW_PATHS)); preempted = Metrics.meter(coordinator.createMetricName(PREEMPTED)); + rejected = Metrics.meter(coordinator.createMetricName(REJECTED)); timeouts = Metrics.meter(coordinator.createMetricName(TIMEOUTS)); invalidations = Metrics.meter(coordinator.createMetricName(INVALIDATIONS)); recoveryDelay = Metrics.timer(coordinator.createMetricName(RECOVERY_DELAY)); @@ -207,7 +174,7 @@ public class AccordMetrics public String toString() { StringBuilder builder = new StringBuilder(); - builder.append("AccordMetrics ["); + builder.append("AccordCoordinatorMetrics ["); try { @@ -229,20 +196,11 @@ public class AccordMetrics return builder.toString(); } - public static class Listener implements CoordinatorEventListener, LocalEventListener + public static class Listener implements CoordinatorEventListener { - public final static Listener instance = new Listener(AccordMetrics.readMetrics, AccordMetrics.writeMetrics); + public static final Listener instance = new Listener(); - private final AccordMetrics readMetrics; - private final AccordMetrics writeMetrics; - - public Listener(AccordMetrics readMetrics, AccordMetrics writeMetrics) - { - this.readMetrics = readMetrics; - this.writeMetrics = writeMetrics; - } - - private AccordMetrics forTransaction(TxnId txnId) + private AccordCoordinatorMetrics forTransaction(TxnId txnId) { if (txnId != null) { @@ -254,57 +212,14 @@ public class AccordMetrics return null; } - @Override - public void onStable(SafeCommandStore safeStore, Command cmd) - { - Tracing.trace("Stable {} on {}", cmd.txnId(), safeStore.commandStore()); - long now = AccordTimeService.nowMicros(); - AccordMetrics metrics = forTransaction(cmd.txnId()); - if (metrics != null) - { - long trxTimestamp = cmd.txnId().hlc(); - metrics.replicaStableLatency.update(now - trxTimestamp, TimeUnit.MICROSECONDS); - } - } - - @Override - public void onPreApplied(SafeCommandStore safeStore, Command cmd) - { - Tracing.trace("Preapplied {} on {}", cmd.txnId(), safeStore.commandStore()); - long now = AccordTimeService.nowMicros(); - AccordMetrics metrics = forTransaction(cmd.txnId()); - if (metrics != null) - { - Timestamp trxTimestamp = cmd.txnId(); - metrics.replicaPreapplyLatency.update(now - trxTimestamp.hlc(), TimeUnit.MICROSECONDS); - PartialDeps deps = cmd.partialDeps(); - metrics.replicaDependencies.update(deps != null ? deps.txnIdCount() : 0); - } - } - - @Override - public void onApplied(SafeCommandStore safeStore, Command cmd, long applyStartedAt) - { - Tracing.trace("Applied {} on {}", cmd.txnId(), safeStore.commandStore()); - long now = AccordTimeService.nowMicros(); - AccordMetrics metrics = forTransaction(cmd.txnId()); - if (metrics != null) - { - Timestamp trxTimestamp = cmd.txnId(); - metrics.replicaApplyLatency.update(now - trxTimestamp.hlc(), TimeUnit.MICROSECONDS); - if (applyStartedAt > 0) - metrics.replicaApplyDuration.update(now - applyStartedAt, TimeUnit.MICROSECONDS); - } - } - @Override public void onPreAccepted(TxnId txnId) { - AccordMetrics metrics = forTransaction(txnId); + AccordCoordinatorMetrics metrics = forTransaction(txnId); if (metrics != null) { long now = AccordTimeService.nowMicros(); - metrics.coordinatorPreacceptLatency.update(Math.max(0, now - txnId.hlc())); + metrics.preacceptLatency.update(Math.max(0, now - txnId.hlc())); } } @@ -312,12 +227,12 @@ public class AccordMetrics public void onExecuting(TxnId txnId, @Nullable Ballot ballot, Deps deps, @Nullable ExecutePath path) { Tracing.trace("{} agreed {}", path, txnId); - AccordMetrics metrics = forTransaction(txnId); + AccordCoordinatorMetrics metrics = forTransaction(txnId); if (metrics != null) { - metrics.coordinatorDependencies.update(deps.txnIdCount()); + metrics.dependencies.update(deps.txnIdCount()); long now = AccordTimeService.nowMicros(); - metrics.coordinatorExecuteLatency.update(Math.max(0, now - txnId.hlc())); + metrics.executeLatency.update(Math.max(0, now - txnId.hlc())); if (path != null) { switch (path) @@ -333,18 +248,18 @@ public class AccordMetrics @Override public void onExecuted(TxnId txnId, Ballot ballot) { - AccordMetrics metrics = forTransaction(txnId); + AccordCoordinatorMetrics metrics = forTransaction(txnId); if (metrics != null) { long now = AccordTimeService.nowMicros(); - metrics.coordinatorApplyLatency.update(Math.max(0, now - txnId.hlc())); + metrics.applyLatency.update(Math.max(0, now - txnId.hlc())); } } @Override public void onRecoveryStopped(Node node, TxnId txnId, Ballot ballot, Result result, Throwable failure) { - AccordMetrics metrics = forTransaction(txnId); + AccordCoordinatorMetrics metrics = forTransaction(txnId); if (metrics != null) { long now = AccordTimeService.nowMicros(); @@ -358,9 +273,42 @@ public class AccordMetrics public void onInvalidated(TxnId txnId) { Tracing.trace("Invalidated {}", txnId); - AccordMetrics metrics = forTransaction(txnId); + AccordCoordinatorMetrics metrics = forTransaction(txnId); if (metrics != null) metrics.invalidations.mark(); } + + @Override + public void onTimeout(@Nullable TxnId txnId) + { + AccordCoordinatorMetrics metrics = forTransaction(txnId); + if (metrics != null) + metrics.timeouts.mark(); + } + + @Override + public void onRejected(TxnId txnId) + { + AccordCoordinatorMetrics metrics = forTransaction(txnId); + if (metrics != null) + metrics.rejected.mark(); + } + + @Override + public void onExhausted(@Nullable TxnId txnId) + { + AccordCoordinatorMetrics metrics = forTransaction(txnId); + if (metrics != null) + metrics.timeouts.mark(); + CoordinatorEventListener.super.onExhausted(txnId); + } + + @Override + public void onPreempted(@Nullable TxnId txnId) + { + AccordCoordinatorMetrics metrics = forTransaction(txnId); + if (metrics != null) + metrics.preempted.mark(); + } } } diff --git a/src/java/org/apache/cassandra/metrics/AccordMetricUtils.java b/src/java/org/apache/cassandra/metrics/AccordMetricUtils.java new file mode 100644 index 0000000000..190664018e --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/AccordMetricUtils.java @@ -0,0 +1,63 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.util.function.Function; + +import accord.local.durability.DurabilityService; +import accord.topology.TopologyManager; +import com.codahale.metrics.Gauge; +import org.apache.cassandra.service.accord.AccordService; +import org.apache.cassandra.service.accord.IAccordService; + +public class AccordMetricUtils +{ + static Gauge fromAccordService(Function ifSetup, T ifNotSetup) + { + return fromAccordService(ifSetup, Function.identity(), ifNotSetup); + } + + static Gauge fromTopologyManager(Function ifSetup) + { + return fromAccordService(service -> service.node().topology(), ifSetup, 0L); + } + + static Gauge fromDurabilityService(Function ifSetup) + { + return fromAccordService(service -> service.node().durability(), ifSetup, 0L); + } + + static Gauge fromAccordService(Function initialStep, Function finalStep, T ifNotSetup) + { + if (AccordService.isSetup()) + { + IAccordService service = AccordService.instance(); + if (!service.isEnabled()) + return () -> ifNotSetup; + I intermediate = initialStep.apply(service); + return () -> finalStep.apply(intermediate); + } + return () -> { + IAccordService service = AccordService.instance(); + if (!service.isEnabled()) + return ifNotSetup; + return finalStep.apply(initialStep.apply(service)); + }; + } +} diff --git a/src/java/org/apache/cassandra/metrics/AccordReplicaMetrics.java b/src/java/org/apache/cassandra/metrics/AccordReplicaMetrics.java new file mode 100644 index 0000000000..551a7ee37d --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/AccordReplicaMetrics.java @@ -0,0 +1,175 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.lang.reflect.Field; +import java.util.concurrent.TimeUnit; + +import accord.api.ReplicaEventListener; +import accord.local.Command; +import accord.local.SafeCommandStore; +import accord.primitives.PartialDeps; +import accord.primitives.Timestamp; +import accord.primitives.TxnId; +import com.codahale.metrics.Counting; +import com.codahale.metrics.Histogram; +import com.codahale.metrics.Timer; +import org.apache.cassandra.service.accord.api.AccordTimeService; +import org.apache.cassandra.tracing.Tracing; + +import static org.apache.cassandra.metrics.CassandraMetricsRegistry.Metrics; + +public class AccordReplicaMetrics +{ + public final static AccordReplicaMetrics readMetrics = new AccordReplicaMetrics("ro"); + public final static AccordReplicaMetrics writeMetrics = new AccordReplicaMetrics("rw"); + + public static final String ACCORD_REPLICA = "AccordReplica"; + public static final String REPLICA_STABLE_LATENCY = "StableLatency"; + public static final String REPLICA_PREAPPLY_LATENCY = "PreApplyLatency"; + public static final String REPLICA_APPLY_LATENCY = "ApplyLatency"; + public static final String REPLICA_APPLY_DURATION = "ApplyDuration"; + public static final String REPLICA_DEPENDENCIES = "Dependencies"; + + /** + * The time between start on the coordinator and commit on this replica. + */ + public final Timer stableLatency; + + /** + * The time between start on the coordinator and arrival of the result on this replica. + */ + public final Timer preapplyLatency; + + /** + * The time between start on the coordinator and application on this replica. + */ + public final Timer applyLatency; + + /** + * TODO (expected): probably more interesting is latency from preapplied to apply; + * we already track local write latencies, whch this effectively duplicates (but including queueing latencies) + * Duration of applying changes. + */ + public final Timer applyDuration; + + /** + * A histogram of the number of dependencies per transaction at this replica. + */ + public final Histogram dependencies; + + private AccordReplicaMetrics(String scope) + { + DefaultNameFactory replica = new DefaultNameFactory(ACCORD_REPLICA, scope); + stableLatency = Metrics.timer(replica.createMetricName(REPLICA_STABLE_LATENCY)); + preapplyLatency = Metrics.timer(replica.createMetricName(REPLICA_PREAPPLY_LATENCY)); + applyLatency = Metrics.timer(replica.createMetricName(REPLICA_APPLY_LATENCY)); + applyDuration = Metrics.timer(replica.createMetricName(REPLICA_APPLY_DURATION)); + dependencies = Metrics.histogram(replica.createMetricName(REPLICA_DEPENDENCIES), true); + } + + @Override + public String toString() + { + StringBuilder builder = new StringBuilder(); + builder.append("AccordReplicaMetrics ["); + + try + { + for (Field f : getClass().getDeclaredFields()) + { + f.setAccessible(true); + if (Counting.class.isAssignableFrom(f.getType())) + { + Counting metric = (Counting) f.get(this); + builder.append(String.format("%s: count=%d, ", f.getName(), metric.getCount())); + } + } + } + catch (IllegalAccessException e) + { + throw new RuntimeException(e); + } + builder.append("]"); + return builder.toString(); + } + + public static class Listener implements ReplicaEventListener + { + public static final Listener instance = new Listener(); + + private AccordReplicaMetrics forTransaction(TxnId txnId) + { + if (txnId != null) + { + if (txnId.isWrite()) + return writeMetrics; + else if (txnId.isSomeRead()) + return readMetrics; + } + return null; + } + + @Override + public void onStable(SafeCommandStore safeStore, Command cmd) + { + Tracing.trace("Stable {} on {}", cmd.txnId(), safeStore.commandStore()); + long now = AccordTimeService.nowMicros(); + AccordReplicaMetrics metrics = forTransaction(cmd.txnId()); + if (metrics != null) + { + long trxTimestamp = cmd.txnId().hlc(); + metrics.stableLatency.update(now - trxTimestamp, TimeUnit.MICROSECONDS); + } + } + + @Override + public void onPreApplied(SafeCommandStore safeStore, Command cmd) + { + Tracing.trace("Preapplied {} on {}", cmd.txnId(), safeStore.commandStore()); + long now = AccordTimeService.nowMicros(); + AccordReplicaMetrics metrics = forTransaction(cmd.txnId()); + if (metrics != null) + { + Timestamp trxTimestamp = cmd.txnId(); + metrics.preapplyLatency.update(now - trxTimestamp.hlc(), TimeUnit.MICROSECONDS); + PartialDeps deps = cmd.partialDeps(); + metrics.dependencies.update(deps != null ? deps.txnIdCount() : 0); + } + } + + @Override + public void onApplied(SafeCommandStore safeStore, Command cmd, long applyStartedAt) + { + Tracing.trace("Applied {} on {}", cmd.txnId(), safeStore.commandStore()); + long now = AccordTimeService.nowMicros(); + AccordReplicaMetrics metrics = forTransaction(cmd.txnId()); + if (metrics != null) + { + Timestamp trxTimestamp = cmd.txnId(); + metrics.applyLatency.update(now - trxTimestamp.hlc(), TimeUnit.MICROSECONDS); + if (applyStartedAt > 0) + metrics.applyDuration.update(now - applyStartedAt, TimeUnit.MICROSECONDS); + } + } + } + + // to ensure initialised + public static void touch() {} +} diff --git a/src/java/org/apache/cassandra/metrics/AccordSystemMetrics.java b/src/java/org/apache/cassandra/metrics/AccordSystemMetrics.java new file mode 100644 index 0000000000..a47fff2302 --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/AccordSystemMetrics.java @@ -0,0 +1,233 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.lang.reflect.Field; +import java.util.concurrent.TimeUnit; + +import accord.impl.progresslog.DefaultProgressLog; +import accord.local.MaxDecidedRX; +import accord.local.RedundantBefore; +import accord.primitives.TxnId; +import accord.topology.TopologyManager; +import accord.utils.Invariants; +import com.codahale.metrics.Counting; +import com.codahale.metrics.Gauge; +import org.apache.cassandra.metrics.LogLinearHistogram.LogLinearSnapshot; +import org.apache.cassandra.service.accord.AccordService; +import org.apache.cassandra.service.accord.IAccordService; +import org.apache.cassandra.utils.Clock; + +import static accord.local.RedundantStatus.Property.GC_BEFORE; +import static accord.local.RedundantStatus.Property.LOCALLY_APPLIED; +import static accord.local.RedundantStatus.Property.LOCALLY_DURABLE_TO_COMMAND_STORE; +import static accord.local.RedundantStatus.Property.LOCALLY_DURABLE_TO_DATA_STORE; +import static accord.local.RedundantStatus.Property.QUORUM_APPLIED; +import static accord.local.RedundantStatus.Property.SHARD_APPLIED; +import static org.apache.cassandra.metrics.AccordMetricUtils.fromDurabilityService; +import static org.apache.cassandra.metrics.AccordMetricUtils.fromTopologyManager; +import static org.apache.cassandra.metrics.CassandraMetricsRegistry.Metrics; + +public class AccordSystemMetrics +{ + public final static AccordSystemMetrics metrics = new AccordSystemMetrics(); + private static final long REFRESH_RATE = TimeUnit.SECONDS.toNanos(30); + + public static final String ACCORD_SYSTEM = "AccordSystem"; + public static final String MIN_EPOCH = "MinEpoch"; + public static final String MAX_EPOCH = "MaxEpoch"; + public static final String PROGRESS_LOG_ACTIVE = "ProgressLogActive"; + public static final String PROGRESS_LOG_SIZE = "ProgressLogSize"; + public static final String PROGRESS_LOG_AGE = "ProgressLogAge"; + public static final String DURABILITY_QUEUE_ACTIVE = "DurabilityQueueActive"; + public static final String DURABILITY_QUEUE_PENDING = "DurabilityQueuePending"; + public static final String SYNCPOINT_AGREED_LAG = "SyncPointAgreedLag"; + public static final String LOCALLY_APPLIED_LAG = "LocallyAppliedLag"; + public static final String LOCALLY_DURABLE_LAG = "LocallyDurableLag"; + public static final String QUORUM_APPLIED_LAG = "QuorumAppliedLag"; + public static final String SHARD_APPLIED_LAG = "ShardAppliedLag"; + public static final String GC_LAG = "GCLag"; + + public final Gauge minEpoch; + public final Gauge maxEpoch; + public final Gauge progressLogActive; + public final Gauge durabilityQueueActive; + public final Gauge durabilityQueuePending; + public final OnDemandHistogram progressLogSize; + public final OnDemandHistogram progressLogAge; + public final OnDemandHistogram syncPointAgreedLag; + public final OnDemandHistogram locallyAppliedLag; + public final OnDemandHistogram locallyDurableLag; + public final OnDemandHistogram quorumAppliedLag; + public final OnDemandHistogram shardAppliedLag; + public final OnDemandHistogram gcLag; + + private Snapshot snapshot; + + static class Snapshot + { + final long at; + final long progressLogActive; + final LogLinearSnapshot progressLogSize; + final LogLinearSnapshot progressLogAge; + final LogLinearSnapshot syncPointAgreedLag; + final LogLinearSnapshot locallyAppliedLag; + final LogLinearSnapshot locallyDurableLag; + final LogLinearSnapshot quorumAppliedLag; + final LogLinearSnapshot shardAppliedLag; + final LogLinearSnapshot gcLag; + + Snapshot() + { + this.at = Clock.Global.nanoTime(); + progressLogActive = 0; + progressLogSize = progressLogAge = syncPointAgreedLag = locallyAppliedLag = locallyDurableLag = gcLag = quorumAppliedLag = shardAppliedLag = new LogLinearSnapshot(0); + } + + Snapshot(SnapshotBuilder builder) + { + this.at = Clock.Global.nanoTime(); + this.progressLogActive = builder.progressLogActive; + this.progressLogSize = builder.progressLogSize.destroyToSnapshot(); + this.progressLogAge = builder.progressLogAge.destroyToSnapshot(); + this.syncPointAgreedLag = builder.syncPointAgreedLag.destroyToSnapshot(); + this.locallyAppliedLag = builder.locallyAppliedLag.destroyToSnapshot(); + this.locallyDurableLag = builder.locallyDurableLag.destroyToSnapshot(); + this.quorumAppliedLag = builder.quorumAppliedLag.destroyToSnapshot(); + this.shardAppliedLag = builder.shardAppliedLag.destroyToSnapshot(); + this.gcLag = builder.gcLag.destroyToSnapshot(); + } + } + + static class SnapshotBuilder + { + long progressLogActive; + final LogLinearHistogram progressLogSize = new LogLinearHistogram(100); + final LogLinearHistogram progressLogAge = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + final LogLinearHistogram syncPointAgreedLag = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + final LogLinearHistogram locallyAppliedLag = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + final LogLinearHistogram locallyDurableLag = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + final LogLinearHistogram quorumAppliedLag = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + final LogLinearHistogram shardAppliedLag = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + final LogLinearHistogram gcLag = new LogLinearHistogram(TimeUnit.HOURS.toSeconds(1L)); + + SnapshotBuilder() + { + } + } + + private AccordSystemMetrics() + { + Invariants.expect(AccordService.isSetup()); + DefaultNameFactory factory = new DefaultNameFactory(ACCORD_SYSTEM); + minEpoch = Metrics.gauge(factory.createMetricName(MIN_EPOCH), fromTopologyManager(TopologyManager::minEpoch)); + maxEpoch = Metrics.gauge(factory.createMetricName(MAX_EPOCH), fromTopologyManager(TopologyManager::epoch)); + durabilityQueueActive = Metrics.gauge(factory.createMetricName(DURABILITY_QUEUE_ACTIVE), fromDurabilityService(durability -> (long)durability.queue().activeCount())); + durabilityQueuePending = Metrics.gauge(factory.createMetricName(DURABILITY_QUEUE_PENDING), fromDurabilityService(durability -> (long)durability.queue().pendingCount())); + progressLogActive = Metrics.gauge(factory.createMetricName(PROGRESS_LOG_ACTIVE), fromDurabilityService(durability -> (long)durability.queue().activeCount())); + progressLogSize = Metrics.onDemandHistogram(factory.createMetricName(PROGRESS_LOG_SIZE), () -> maybeRefreshHistograms().progressLogSize); + progressLogAge = Metrics.onDemandHistogram(factory.createMetricName(PROGRESS_LOG_AGE), () -> maybeRefreshHistograms().progressLogAge); + syncPointAgreedLag = Metrics.onDemandHistogram(factory.createMetricName(SYNCPOINT_AGREED_LAG), () -> maybeRefreshHistograms().syncPointAgreedLag); + locallyAppliedLag = Metrics.onDemandHistogram(factory.createMetricName(LOCALLY_APPLIED_LAG), () -> maybeRefreshHistograms().locallyAppliedLag); + locallyDurableLag = Metrics.onDemandHistogram(factory.createMetricName(LOCALLY_DURABLE_LAG), () -> maybeRefreshHistograms().locallyDurableLag); + quorumAppliedLag = Metrics.onDemandHistogram(factory.createMetricName(QUORUM_APPLIED_LAG), () -> maybeRefreshHistograms().quorumAppliedLag); + shardAppliedLag = Metrics.onDemandHistogram(factory.createMetricName(SHARD_APPLIED_LAG), () -> maybeRefreshHistograms().shardAppliedLag); + gcLag = Metrics.onDemandHistogram(factory.createMetricName(GC_LAG), () -> maybeRefreshHistograms().gcLag); + } + + private synchronized Snapshot maybeRefreshHistograms() + { + if (snapshot == null || Clock.Global.nanoTime() >= snapshot.at + REFRESH_RATE) + refreshHistograms(); + return snapshot; + } + + private synchronized void refreshHistograms() + { + if (!AccordService.isSetup()) + { + snapshot = new Snapshot(); + return; + } + + IAccordService service = AccordService.instance(); + if (!service.isEnabled()) + { + snapshot = new Snapshot(); + return; + } + + int nowSeconds = (int) (Clock.Global.currentTimeMillis() / 1000); + SnapshotBuilder builder = new SnapshotBuilder(); + service.node().commandStores().forEachCommandStore(commandStore -> { + DefaultProgressLog.ImmutableView view = ((DefaultProgressLog)commandStore.unsafeProgressLog()).immutableView(); + builder.progressLogActive += view.activeCount(); + builder.progressLogSize.increment(view.size()); + while (view.advance()) + builder.progressLogAge.increment(ageSeconds(nowSeconds, view.txnId())); + + RedundantBefore redundantBefore = commandStore.unsafeGetRedundantBefore(); + for (int i = 0 ; i < redundantBefore.size() ; ++i) + { + RedundantBefore.Bounds bounds = redundantBefore.valueAt(i); + builder.locallyAppliedLag.increment(ageSeconds(nowSeconds, bounds.maxBound(LOCALLY_APPLIED))); + builder.locallyDurableLag.increment(ageSeconds(nowSeconds, bounds.maxBoundBoth(LOCALLY_DURABLE_TO_DATA_STORE, LOCALLY_DURABLE_TO_COMMAND_STORE))); + builder.quorumAppliedLag.increment(ageSeconds(nowSeconds, bounds.maxBound(QUORUM_APPLIED))); + builder.shardAppliedLag.increment(ageSeconds(nowSeconds, bounds.maxBound(SHARD_APPLIED))); + builder.gcLag.increment(ageSeconds(nowSeconds, bounds.maxBound(GC_BEFORE))); + } + + MaxDecidedRX maxDecidedRX = commandStore.unsafeGetMaxDecidedRX(); + for (int i = 0 ; i < maxDecidedRX.size() ; ++i) + builder.syncPointAgreedLag.increment(ageSeconds(nowSeconds, maxDecidedRX.valueAt(i).any)); + }); + this.snapshot = new Snapshot(builder); + } + + static int ageSeconds(int nowSeconds, TxnId txnId) + { + return Math.max(0, nowSeconds - (int)TimeUnit.MICROSECONDS.toSeconds(txnId.hlc())); + } + + @Override + public String toString() + { + StringBuilder builder = new StringBuilder(); + builder.append("AccordSystemMetrics ["); + + try + { + for (Field f : getClass().getDeclaredFields()) + { + f.setAccessible(true); + if (Counting.class.isAssignableFrom(f.getType())) + { + Counting metric = (Counting) f.get(this); + builder.append(String.format("%s: count=%d, ", f.getName(), metric.getCount())); + } + } + } + catch (IllegalAccessException e) + { + throw new RuntimeException(e); + } + builder.append(']'); + return builder.toString(); + } +} diff --git a/src/java/org/apache/cassandra/metrics/SnapshottingReservoir.java b/src/java/org/apache/cassandra/metrics/CassandraHistogram.java similarity index 63% rename from src/java/org/apache/cassandra/metrics/SnapshottingReservoir.java rename to src/java/org/apache/cassandra/metrics/CassandraHistogram.java index 1b6c9684ae..84d3df5a59 100644 --- a/src/java/org/apache/cassandra/metrics/SnapshottingReservoir.java +++ b/src/java/org/apache/cassandra/metrics/CassandraHistogram.java @@ -18,10 +18,24 @@ package org.apache.cassandra.metrics; -import com.codahale.metrics.Reservoir; -import com.codahale.metrics.Snapshot; +import com.codahale.metrics.Histogram; -public interface SnapshottingReservoir extends Reservoir +public class CassandraHistogram extends Histogram { - Snapshot getPercentileSnapshot(); + final CassandraReservoir reservoir; + public CassandraHistogram(CassandraReservoir reservoir) + { + super(reservoir); + this.reservoir = reservoir; + } + + public CassandraReservoir.BucketStrategy bucketStrategy() + { + return reservoir.bucketStrategy(); + } + + public long[] bucketStarts(int length) + { + return reservoir.buckets(length); + } } diff --git a/src/java/org/apache/cassandra/metrics/CassandraMetricsRegistry.java b/src/java/org/apache/cassandra/metrics/CassandraMetricsRegistry.java index 44bcf2d6cf..4eecb75eac 100644 --- a/src/java/org/apache/cassandra/metrics/CassandraMetricsRegistry.java +++ b/src/java/org/apache/cassandra/metrics/CassandraMetricsRegistry.java @@ -29,6 +29,7 @@ import java.util.concurrent.ConcurrentHashMap; import java.util.concurrent.TimeUnit; import java.util.function.Consumer; import java.util.function.Function; +import java.util.function.Supplier; import java.util.function.UnaryOperator; import java.util.stream.Stream; import javax.annotation.Nullable; @@ -114,8 +115,10 @@ public class CassandraMetricsRegistry extends MetricRegistry // for virtual tables. metricGroups = ImmutableSet.builder() .add(AbstractMetrics.TYPE) - .add(AccordMetrics.ACCORD_COORDINATOR) - .add(AccordMetrics.ACCORD_REPLICA) + .add(AccordCoordinatorMetrics.ACCORD_COORDINATOR) + .add(AccordCacheMetrics.ACCORD_CACHE) + .add(AccordReplicaMetrics.ACCORD_REPLICA) + .add(AccordSystemMetrics.ACCORD_SYSTEM) .add(BatchMetrics.TYPE_NAME) .add(BufferPoolMetrics.TYPE_NAME) .add(CIDRAuthorizerMetrics.TYPE_NAME) @@ -169,8 +172,8 @@ public class CassandraMetricsRegistry extends MetricRegistry return Long.toString(((Counter) metric).getCount()); else if (metric instanceof Gauge) return getGaugeValue((Gauge) metric); - else if (metric instanceof Histogram) - return Double.toString(((Histogram) metric).getSnapshot().getMedian()); + else if (metric instanceof CassandraHistogram) + return Double.toString(((CassandraHistogram) metric).getSnapshot().getMedian()); else if (metric instanceof Meter) return Long.toString(((Meter) metric).getCount()); else if (metric instanceof Timer) @@ -245,7 +248,7 @@ public class CassandraMetricsRegistry extends MetricRegistry "All metrics with type \"Histogram\"", new HistogramMetricRowWalker(), Metrics.getMetrics(), - Histogram.class::isInstance, + CassandraHistogram.class::isInstance, HistogramMetricRow::new)) .add(createSinglePartitionedValueFiltered(VIRTUAL_METRICS, "type_meter", @@ -316,18 +319,28 @@ public class CassandraMetricsRegistry extends MetricRegistry return meter; } - public Histogram histogram(MetricName name, boolean considerZeroes) + public CassandraHistogram histogram(MetricName name, boolean considerZeroes) { return register(name, new ClearableHistogram(new DecayingEstimatedHistogramReservoir(considerZeroes))); } - public Histogram histogram(MetricName name, MetricName alias, boolean considerZeroes) + public CassandraHistogram histogram(MetricName name, MetricName alias, boolean considerZeroes) { - Histogram histogram = histogram(name, considerZeroes); + CassandraHistogram histogram = histogram(name, considerZeroes); register(alias, histogram); return histogram; } + public ShardedHistogram shardedHistogram(MetricName name) + { + return register(name, new ShardedHistogram()); + } + + public OnDemandHistogram onDemandHistogram(MetricName name, Supplier snapshot) + { + return register(name, new OnDemandHistogram(snapshot)); + } + public > T gauge(MetricName name, T gauge) { return register(name, gauge); @@ -362,14 +375,14 @@ public class CassandraMetricsRegistry extends MetricRegistry return timer; } - public static SnapshottingReservoir createReservoir(TimeUnit durationUnit) + public static CassandraReservoir createReservoir(TimeUnit durationUnit) { - SnapshottingReservoir reservoir; + CassandraReservoir reservoir; if (durationUnit != TimeUnit.NANOSECONDS) { - SnapshottingReservoir underlying = new DecayingEstimatedHistogramReservoir(DecayingEstimatedHistogramReservoir.DEFAULT_ZERO_CONSIDERATION, - DecayingEstimatedHistogramReservoir.LOW_BUCKET_COUNT, - DecayingEstimatedHistogramReservoir.DEFAULT_STRIPE_COUNT); + CassandraReservoir underlying = new DecayingEstimatedHistogramReservoir(DecayingEstimatedHistogramReservoir.DEFAULT_ZERO_CONSIDERATION, + DecayingEstimatedHistogramReservoir.LOW_BUCKET_COUNT, + DecayingEstimatedHistogramReservoir.DEFAULT_STRIPE_COUNT); // fewer buckets should suffice if timer is not based on nanos reservoir = new ScalingReservoir(underlying, // timer update values in nanos. @@ -518,8 +531,10 @@ public class CassandraMetricsRegistry extends MetricRegistry mbean = new JmxGauge((Gauge) metric, name); else if (metric instanceof Counter) mbean = new JmxCounter((Counter) metric, name); + else if (metric instanceof CassandraHistogram) + mbean = new JmxHistogram((CassandraHistogram) metric, name); else if (metric instanceof Histogram) - mbean = new JmxHistogram((Histogram) metric, name); + throw new UnsupportedOperationException("Must supply a CassandraHistogram"); else if (metric instanceof Timer) mbean = new JmxTimer((Timer) metric, name, TimeUnit.SECONDS, DEFAULT_TIMER_UNIT); else if (metric instanceof Metered) @@ -637,14 +652,20 @@ public class CassandraMetricsRegistry extends MetricRegistry long[] values(); long[] getRecentValues(); + + String bucketsId(); + + long[] rawBuckets(int count); + + long[] rawValues(); } private static class JmxHistogram extends AbstractBean implements JmxHistogramMBean { - private final Histogram metric; + final CassandraHistogram metric; private long[] last = null; - private JmxHistogram(Histogram metric, ObjectName objectName) + private JmxHistogram(CassandraHistogram metric, ObjectName objectName) { super(objectName); this.metric = metric; @@ -719,7 +740,10 @@ public class CassandraMetricsRegistry extends MetricRegistry @Override public long[] values() { - return metric.getSnapshot().getValues(); + long[] values = metric.getSnapshot().getValues(); + if (metric.bucketStrategy() == CassandraReservoir.BucketStrategy.log_linear) + values = metric.bucketStrategy().translateTo(CassandraReservoir.BucketStrategy.exp_12_nozero, values); + return values; } /** @@ -738,6 +762,24 @@ public class CassandraMetricsRegistry extends MetricRegistry last = now; return delta; } + + @Override + public String bucketsId() + { + return metric.bucketStrategy().name(); + } + + @Override + public long[] rawBuckets(int count) + { + return metric.bucketStarts(count); + } + + @Override + public long[] rawValues() + { + return metric.getSnapshot().getValues(); + } } /** diff --git a/src/java/org/apache/cassandra/metrics/CassandraReservoir.java b/src/java/org/apache/cassandra/metrics/CassandraReservoir.java new file mode 100644 index 0000000000..424735f088 --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/CassandraReservoir.java @@ -0,0 +1,121 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.util.Arrays; +import java.util.function.IntFunction; +import java.util.function.LongFunction; + +import accord.utils.Invariants; +import com.codahale.metrics.Reservoir; +import com.codahale.metrics.Snapshot; +import org.apache.cassandra.utils.EstimatedHistogram; + +public interface CassandraReservoir extends Reservoir +{ + enum BucketStrategy + { + none(false, length -> { throw new UnsupportedOperationException(); }, scale -> { throw new UnsupportedOperationException(); }), + exp_12(true, length -> EstimatedHistogram.newOffsets(length, true), scale -> EstimatedHistogram.newOffsetsWithScale(scale, true)), + exp_12_nozero(true, length -> EstimatedHistogram.newOffsets(length, false), scale -> EstimatedHistogram.newOffsetsWithScale(scale, false)), + log_linear(false, LogLinearHistogram::bucketsWithLength, LogLinearHistogram::bucketsWithScale); + + final boolean overflows; + final IntFunction bucketsWithLength; + final LongFunction bucketsWithScale; + volatile long[] cachedBuckets; + + BucketStrategy(boolean overflows, IntFunction bucketsWithLength, LongFunction bucketsWithScale) + { + this.overflows = overflows; + this.bucketsWithLength = bucketsWithLength; + this.bucketsWithScale = bucketsWithScale; + } + + private long[] sharedBucketsWithLength(int length) + { + long[] buckets = cachedBuckets; + if (buckets == null || buckets.length < length) + cachedBuckets = buckets = this.bucketsWithLength.apply(length); + return buckets; + } + + private long[] sharedBucketsWithScale(long scale) + { + long[] buckets = cachedBuckets; + if (buckets == null || buckets.length == 0 || buckets[buckets.length - 1] < scale) + cachedBuckets = buckets = this.bucketsWithScale.apply(scale); + return buckets; + } + + public long[] translateTo(BucketStrategy toStrategy, long[] srcValues) + { + if (srcValues.length == 0) + return srcValues; + + long[] srcBuckets = sharedBucketsWithLength(srcValues.length + 1); + long[] trgBuckets = toStrategy.sharedBucketsWithScale(srcBuckets[srcValues.length]); + return translate(srcBuckets, srcValues, overflows ? srcValues.length - 1 : srcValues.length, trgBuckets); + } + + public static long[] translate(long[] srcBuckets, long[] srcValues, int srcLength, long[] trgBuckets) + { + Invariants.requireArgument(srcLength < srcBuckets.length); + int trgLength = Arrays.binarySearch(trgBuckets, srcBuckets[srcLength]); + if (trgLength < 0) trgLength = -1 - trgLength; + else trgLength++; + long[] trgValues = new long[trgLength]; + long srcStart = srcBuckets[0], srcRemainder = srcValues[0], srcEnd = srcBuckets[1]; + int si = 0, ti = 0; + while (true) + { + long trgEnd = trgBuckets[ti + 1]; + if (trgEnd <= srcStart) + { + if (++ti == trgValues.length) break; + continue; + } + + long end = Math.min(trgEnd, srcEnd); + if (end == srcEnd) + { + trgValues[ti] += srcRemainder; + if (++si == srcLength) break; + srcStart = srcEnd; + srcEnd = srcBuckets[si + 1]; + srcRemainder = srcValues[si]; + } + else + { + double ratio = (end - srcStart) / (double)(srcEnd - srcStart); + long incr = Math.round(ratio * srcRemainder); + trgValues[ti] += incr; + srcRemainder -= incr; + srcStart = end; + } + } + return trgValues; + } + } + + Snapshot getPercentileSnapshot(); + long[] buckets(int length); + BucketStrategy bucketStrategy(); + +} diff --git a/src/java/org/apache/cassandra/metrics/ClearableHistogram.java b/src/java/org/apache/cassandra/metrics/ClearableHistogram.java index f83ef8276a..66f3f9fd38 100644 --- a/src/java/org/apache/cassandra/metrics/ClearableHistogram.java +++ b/src/java/org/apache/cassandra/metrics/ClearableHistogram.java @@ -23,12 +23,10 @@ import java.util.concurrent.atomic.LongAdder; import com.google.common.annotations.VisibleForTesting; -import com.codahale.metrics.Histogram; - /** * Adds ability to reset a histogram */ -public class ClearableHistogram extends Histogram +public class ClearableHistogram extends CassandraHistogram { private final DecayingEstimatedHistogramReservoir reservoirRef; @@ -58,7 +56,7 @@ public class ClearableHistogram extends Histogram // clearly if we start using an incompatible version of the metrics. try { - Field countField = Histogram.class.getDeclaredField("count"); + Field countField = com.codahale.metrics.Histogram.class.getDeclaredField("count"); countField.setAccessible(true); // in 3.1 the counter object is a LongAdderAdapter which is a package private interface // from com.codahale.metrics. In 4.0, it is a java LongAdder so the code will be simpler. diff --git a/src/java/org/apache/cassandra/metrics/ClientMetrics.java b/src/java/org/apache/cassandra/metrics/ClientMetrics.java index 1df5a23b21..2ac00d293c 100644 --- a/src/java/org/apache/cassandra/metrics/ClientMetrics.java +++ b/src/java/org/apache/cassandra/metrics/ClientMetrics.java @@ -31,9 +31,7 @@ import java.util.function.Predicate; import com.google.common.annotations.VisibleForTesting; import com.codahale.metrics.Gauge; -import com.codahale.metrics.Histogram; import com.codahale.metrics.Meter; -import com.codahale.metrics.Reservoir; import com.codahale.metrics.Timer; import org.apache.cassandra.auth.AuthenticatedUser; import org.apache.cassandra.auth.IAuthenticator; @@ -199,9 +197,9 @@ public final class ClientMetrics registerGauge("ClientsByProtocolVersion", "clientsByProtocolVersion", this::recentClientStats); registerGauge("RequestsSize", ClientResourceLimits::getCurrentGlobalUsage); - Reservoir ipUsageReservoir = ClientResourceLimits.ipUsageReservoir(); + CassandraReservoir ipUsageReservoir = ClientResourceLimits.ipUsageReservoir(); Metrics.register(factory.createMetricName("RequestsSizeByIpDistribution"), - new Histogram(ipUsageReservoir) + new CassandraHistogram(ipUsageReservoir) { public long getCount() { diff --git a/src/java/org/apache/cassandra/metrics/DecayingEstimatedHistogramReservoir.java b/src/java/org/apache/cassandra/metrics/DecayingEstimatedHistogramReservoir.java index cfca646042..b6f2bde23a 100644 --- a/src/java/org/apache/cassandra/metrics/DecayingEstimatedHistogramReservoir.java +++ b/src/java/org/apache/cassandra/metrics/DecayingEstimatedHistogramReservoir.java @@ -84,7 +84,7 @@ import static org.apache.cassandra.config.CassandraRelevantProperties.DECAYING_E * * @see ExponentiallyDecayingReservoir */ -public class DecayingEstimatedHistogramReservoir implements SnapshottingReservoir +public class DecayingEstimatedHistogramReservoir implements CassandraReservoir { private static final Logger logger = LoggerFactory.getLogger(DecayingEstimatedHistogramReservoir.class); private static final NoSpamLogger noSpamLogger = NoSpamLogger.getLogger(logger, 5L, TimeUnit.MINUTES); @@ -343,6 +343,20 @@ public class DecayingEstimatedHistogramReservoir implements SnapshottingReservoi return rescaleIfNeeded(clock.now()); } + @Override + public long[] buckets(int length) + { + if (length == bucketOffsets.length) + return bucketOffsets; + return EstimatedHistogram.newOffsets(length, bucketOffsets[0] == 0); + } + + @Override + public BucketStrategy bucketStrategy() + { + return bucketOffsets[0] == 0 ? BucketStrategy.exp_12 : BucketStrategy.exp_12_nozero; + } + /** * @return true if this histogram has overflowed -- that is, a value larger than our largest bucket could bound was added */ diff --git a/src/java/org/apache/cassandra/metrics/LogLinearHistogram.java b/src/java/org/apache/cassandra/metrics/LogLinearHistogram.java new file mode 100644 index 0000000000..43420cf7f6 --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/LogLinearHistogram.java @@ -0,0 +1,288 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.io.OutputStream; +import java.io.OutputStreamWriter; +import java.io.PrintWriter; +import java.nio.charset.StandardCharsets; +import java.util.Arrays; + +import com.google.common.annotations.VisibleForTesting; +import com.google.common.primitives.Ints; + +import accord.utils.Invariants; +import com.codahale.metrics.Snapshot; + +/** + * A simple single-threaded histogram with log-linear buckets. + * This has approximately the same accuracy as the lg 1.2 growth of EstimatedHistogram, but is simpler and faster. + * This histogram also importantly supports decrements and replace (decrement/increment pairs), and auto grows if necessary, + * so no overflow and can be initialised to a small size to avoid wasting memory + * + * TODO (desired): improve performance and memory locality by using a small buffer for collecting updates with e.g. 4 bits per counter, + */ +public class LogLinearHistogram +{ + private static final int MAX_INDEX = 247; + private long[] buckets; + long totalCount; + + public LogLinearHistogram(long expectedMaxValue) + { + buckets = new long[buckets(expectedMaxValue)]; + } + + public void increment(long value) + { + int index = index(value); + buckets(index)[index]++; + ++totalCount; + } + + public void decrement(long value) + { + int index = index(value); + buckets(index)[index]--; + --totalCount; + } + + public void replace(long decrement, long increment) + { + int decrementIndex = index(decrement); + int incrementIndex = index(increment); + long[] buckets = buckets(Math.max(decrementIndex, incrementIndex)); + if (decrementIndex != incrementIndex) + { + --buckets[decrementIndex]; + ++buckets[incrementIndex]; + } + } + + public static long[] bucketsWithLength(int length) + { + Invariants.require(length <= MAX_INDEX + 1); + long[] buckets = new long[length]; + for (int i = 0 ; i < length ; ++i) + buckets[i] = invertIndex(i); + return buckets; + } + + public static long[] bucketsWithScale(long maxScale) + { + return bucketsWithLength(1 + index(maxScale)); + } + + private long[] buckets(int withIndexAtLeast) + { + if (buckets.length <= withIndexAtLeast) + { + Invariants.require(withIndexAtLeast <= MAX_INDEX); + buckets = Arrays.copyOf(buckets, (withIndexAtLeast | 0x3) + 1); + } + return buckets; + } + + private static int buckets(long maxValue) + { + return 1 + (index(maxValue) | 0x3); + } + + private static int index(long value) + { + if (value < 4) + return (int) value; + int log = 61 - Long.numberOfLeadingZeros(value); + int linear = (int) (value >>> log) & 0x3; + return (log + 1) * 4 + linear; + } + + private static long invertIndex(int index) + { + if (index < 4) + return index; + + int log = index / 4; + int linear = index & 0x3; + return (4L | linear) << (log - 1); + } + + static + { + for (int i = 0 ; i < MAX_INDEX ; ++i) + { + Invariants.require(index(invertIndex(i)) == i); + Invariants.require(i == 0 || index(invertIndex(i) - 1) == i - 1); + } + Invariants.require(index(invertIndex(MAX_INDEX + 1) - 1) == MAX_INDEX); + } + + public static class LogLinearSnapshot extends Snapshot + { + long totalCount; + long[] raw; + long[] cumulative; + + public static LogLinearSnapshot emptyForMax(long maxValue) + { + return new LogLinearSnapshot(buckets(maxValue)); + } + + LogLinearSnapshot(int size) + { + this.raw = new long[size]; + } + + @VisibleForTesting + LogLinearSnapshot(long[] raw, long totalCount) + { + this.raw = raw; + this.totalCount = totalCount; + } + + private long[] cumulative() + { + if (cumulative == null) + { + cumulative = new long[raw.length]; + long sum = 0; + for (int i = 0 ; i < cumulative.length ; ++i) + cumulative[i] = sum += cumulative[i]; + } + return cumulative; + } + + private long get(long tot) + { + if (totalCount == 0) + return 0; + + long[] cumulative = cumulative(); + int i = Arrays.binarySearch(cumulative, tot); + if (i >= 0) while (i > 0 && cumulative[i-1] == tot) --i; + else i = Math.max(0, -2 - i); + long prevValue = invertIndex(i); + long prevCount = cumulative[i]; + long nextValue = invertIndex(i + 1); + long nextCount = i + 1 == cumulative.length ? totalCount : cumulative[i + 1]; + long gap = tot - prevCount; + // should we use double arithmetic here to avoid overflow? + return prevValue + ((nextValue - prevValue) * gap) / (nextCount - prevCount); + } + + @Override + public double getValue(double quantile) + { + return get(Math.round(totalCount * quantile)); + } + + @Override + public int size() + { + return Ints.saturatedCast(totalCount); + } + + @Override + public long getMax() + { + return get(totalCount); + } + + @Override + public double getMean() + { + if (totalCount <= 1) + return 0.0D; + return get(totalCount / 2); + } + + @Override + public long getMin() + { + return get(1); + } + + /** + * Get the estimated standard deviation of the values added to this reservoir. + * + * As values are collected in variable sized buckets, the actual deviation may be more or less than the value + * returned. + * + * @return an estimate of the standard deviation + */ + @Override + public double getStdDev() + { + if (totalCount <= 1) + { + return 0.0D; + } + else + { + double mean = this.getMean(); + double sum = 0.0D; + + for(int i = 0; i < raw.length; ++i) + { + long value = invertIndex(i); + double diff = value - mean; + sum += diff * diff * raw[i]; + } + + return Math.sqrt(sum / (totalCount - 1)); + } + } + + @Override + public void dump(OutputStream output) + { + try (PrintWriter out = new PrintWriter(new OutputStreamWriter(output, StandardCharsets.UTF_8))) + { + for(int i = 0; i < raw.length; ++i) + out.printf("%d%n", raw[i]); + } + } + + @Override + public long[] getValues() + { + return raw; + } + } + + public LogLinearSnapshot destroyToSnapshot() + { + LogLinearSnapshot result = new LogLinearSnapshot(buckets, totalCount); + buckets = null; + this.totalCount = 0; + return result; + } + + public void updateSnapshot(LogLinearSnapshot snapshot) + { + if (snapshot.raw.length < buckets.length) + snapshot.raw = Arrays.copyOf(snapshot.raw, buckets.length); + + long[] raw = snapshot.raw; + for (int i = 0 ; i < raw.length ; ++i) + raw[i] = raw[i] + buckets[i]; + snapshot.totalCount += totalCount; + snapshot.cumulative = null; + } +} diff --git a/src/java/org/apache/cassandra/metrics/OnDemandHistogram.java b/src/java/org/apache/cassandra/metrics/OnDemandHistogram.java new file mode 100644 index 0000000000..4ad890e3c0 --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/OnDemandHistogram.java @@ -0,0 +1,59 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.util.function.Supplier; + +import com.codahale.metrics.Snapshot; +import org.apache.cassandra.metrics.LogLinearHistogram.LogLinearSnapshot; + +import static org.apache.cassandra.metrics.CassandraReservoir.BucketStrategy.log_linear; + +public class OnDemandHistogram extends OverrideHistogram +{ + final Supplier snapshot; + protected OnDemandHistogram(Supplier snapshot) + { + this.snapshot = snapshot; + } + + @Override + public synchronized long getCount() + { + return snapshot.get().totalCount; + } + + @Override + public Snapshot getSnapshot() + { + return snapshot.get(); + } + + @Override + public CassandraReservoir.BucketStrategy bucketStrategy() + { + return log_linear; + } + + @Override + public long[] bucketStarts(int length) + { + return LogLinearHistogram.bucketsWithLength(length); + } +} diff --git a/src/java/org/apache/cassandra/metrics/OverrideHistogram.java b/src/java/org/apache/cassandra/metrics/OverrideHistogram.java new file mode 100644 index 0000000000..36085433b7 --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/OverrideHistogram.java @@ -0,0 +1,50 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import com.codahale.metrics.Snapshot; +import org.agrona.UnsafeAccess; + +public abstract class OverrideHistogram extends CassandraHistogram +{ + private static final CassandraReservoir NO_RESERVOIR = new CassandraReservoir() { + @Override public Snapshot getPercentileSnapshot() { return null; } + @Override public long[] buckets(int length) { return new long[0]; } + @Override public BucketStrategy bucketStrategy() { return BucketStrategy.none; } + @Override public int size() { return 0; } + @Override public void update(long value) {} + @Override public Snapshot getSnapshot() { return null; } + }; + + protected OverrideHistogram() + { + super(NO_RESERVOIR); + try + { + UnsafeAccess.UNSAFE.putObject(this, UnsafeAccess.UNSAFE.objectFieldOffset(CassandraHistogram.class.getDeclaredField("count")), null); + } + catch (Throwable t) + { + // best effort, don't worry if this fails + } + } + + public abstract CassandraReservoir.BucketStrategy bucketStrategy(); + public abstract long[] bucketStarts(int length); +} diff --git a/src/java/org/apache/cassandra/metrics/ScalingReservoir.java b/src/java/org/apache/cassandra/metrics/ScalingReservoir.java index 582b7a714a..3b6b2ec1a5 100644 --- a/src/java/org/apache/cassandra/metrics/ScalingReservoir.java +++ b/src/java/org/apache/cassandra/metrics/ScalingReservoir.java @@ -17,17 +17,19 @@ */ package org.apache.cassandra.metrics; +import java.util.function.LongUnaryOperator; + import com.codahale.metrics.Snapshot; /** * A reservoir that scales the values before updating. */ -public class ScalingReservoir implements SnapshottingReservoir +public class ScalingReservoir implements CassandraReservoir { - private final SnapshottingReservoir delegate; - private final ScaleFunction scaleFunc; + private final CassandraReservoir delegate; + private final LongUnaryOperator scaleFunc; - public ScalingReservoir(SnapshottingReservoir reservoir, ScaleFunction scaleFunc) + public ScalingReservoir(CassandraReservoir reservoir, LongUnaryOperator scaleFunc) { this.delegate = reservoir; this.scaleFunc = scaleFunc; @@ -42,7 +44,7 @@ public class ScalingReservoir implements SnapshottingReservoir @Override public void update(long value) { - delegate.update(scaleFunc.apply(value)); + delegate.update(scaleFunc.applyAsLong(value)); } @Override @@ -57,14 +59,15 @@ public class ScalingReservoir implements SnapshottingReservoir return delegate.getPercentileSnapshot(); } - /** - * Scale the input value. - * - * Not using {@code java.util.function.Function} to avoid auto-boxing. - */ - @FunctionalInterface - public static interface ScaleFunction + @Override + public BucketStrategy bucketStrategy() { - long apply(long value); + return delegate.bucketStrategy(); + } + + @Override + public long[] buckets(int length) + { + return delegate.buckets(length); } } diff --git a/src/java/org/apache/cassandra/metrics/ShardedHistogram.java b/src/java/org/apache/cassandra/metrics/ShardedHistogram.java new file mode 100644 index 0000000000..1677edad81 --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/ShardedHistogram.java @@ -0,0 +1,146 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.util.concurrent.CopyOnWriteArrayList; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.locks.Lock; + +import com.codahale.metrics.Snapshot; +import org.apache.cassandra.metrics.LogLinearHistogram.LogLinearSnapshot; +import org.apache.cassandra.utils.Clock; + +import static org.apache.cassandra.metrics.CassandraReservoir.BucketStrategy.log_linear; + +public class ShardedHistogram extends OverrideHistogram +{ + private static final long REFRESH_RATE = TimeUnit.SECONDS.toNanos(15); + + static class HistogramShard + { + final Lock lock; + final LogLinearHistogram histogram; + + HistogramShard(Lock lock, LogLinearHistogram histogram) + { + this.lock = lock; + this.histogram = histogram; + } + + long total() + { + lock.lock(); + try + { + return histogram.totalCount; + } + finally + { + lock.unlock(); + } + } + + public void updateSnapshot(LogLinearSnapshot snapshot) + { + lock.lock(); + try + { + histogram.updateSnapshot(snapshot); + } + finally + { + lock.unlock(); + } + } + } + + final CopyOnWriteArrayList shards = new CopyOnWriteArrayList<>(); + final long initialMaxValue; + + public ShardedHistogram() + { + this(1 << 16); + } + + @Override + public CassandraReservoir.BucketStrategy bucketStrategy() + { + return log_linear; + } + + @Override + public long[] bucketStarts(int length) + { + return LogLinearHistogram.bucketsWithLength(length); + } + + public ShardedHistogram(long initialMaxValue) + { + this.initialMaxValue = initialMaxValue; + } + + public LogLinearHistogram newShard(Lock guardedBy) + { + HistogramShard shard = new HistogramShard(guardedBy, new LogLinearHistogram(initialMaxValue)); + shards.add(shard); + return shard.histogram; + } + + private LogLinearSnapshot snapshot; + private long snapshotAt; + + public synchronized void refresh() + { + refresh(Clock.Global.nanoTime()); + } + + private synchronized LogLinearSnapshot refresh(long now) + { + LogLinearSnapshot snapshot = LogLinearSnapshot.emptyForMax(initialMaxValue); + for (HistogramShard shard : shards) + shard.updateSnapshot(snapshot); + this.snapshot = snapshot; + this.snapshotAt = now; + return snapshot; + } + + private synchronized LogLinearSnapshot maybeRefresh() + { + if (shards.isEmpty()) + return snapshot = new LogLinearSnapshot(0); + + long now = Clock.Global.nanoTime(); + if (snapshot != null && snapshotAt + REFRESH_RATE >= now) + return snapshot; + + return refresh(now); + } + + @Override + public synchronized long getCount() + { + return maybeRefresh().totalCount; + } + + @Override + public Snapshot getSnapshot() + { + return maybeRefresh(); + } +} diff --git a/src/java/org/apache/cassandra/metrics/ShardedHitRate.java b/src/java/org/apache/cassandra/metrics/ShardedHitRate.java new file mode 100644 index 0000000000..1a9feea7ac --- /dev/null +++ b/src/java/org/apache/cassandra/metrics/ShardedHitRate.java @@ -0,0 +1,248 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import java.util.Arrays; +import java.util.concurrent.CopyOnWriteArrayList; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.locks.Lock; + +import accord.utils.Invariants; +import org.apache.cassandra.utils.Clock; + +public class ShardedHitRate +{ + private static final long REFRESH_RATE = TimeUnit.SECONDS.toNanos(15L); + private static final int BUCKETS = 16; + private static final int BUCKET_MASK = 0xF; + private static final long CLOCK_SLICE_SHIFT = 10; + private static final long CLOCK_SLICE = 1 << CLOCK_SLICE_SHIFT; + private static final long CLOCK_DIVISOR = 60_000_000_000L >>> CLOCK_SLICE_SHIFT; + + private static long now() + { + return Clock.Global.nanoTime() / CLOCK_DIVISOR; + } + + private static int atToMinutes(long at) + { + return (int) (at >>> CLOCK_SLICE_SHIFT); + } + + public static class HitRateShard + { + final Lock lock; + + long lastAt; + final long[] hits = new long[BUCKETS]; + final long[] misses = new long[BUCKETS]; + + long totalHits, totalMisses; + + public HitRateShard(Lock lock) + { + this.lock = lock; + } + + void tick(long at) + { + if (lastAt >= at) + return; + + int count = Math.min(BUCKETS, atToMinutes(at - lastAt)); + for (int i = 1 ; i <= count ; ++i) + { + int index = (atToMinutes(lastAt) + i) & BUCKET_MASK; + hits[index] = misses[index] = 0; + } + lastAt = at; + } + + // assumes the lock is held + public void markHitExclusive() + { + markHitExclusive(now()); + } + + public void markMissExclusive() + { + markMissExclusive(now()); + } + + void markHitExclusive(long at) + { + tick(at); + hits[atToMinutes(at) & BUCKET_MASK]++; + ++totalHits; + } + + public void markMissExclusive(long at) + { + tick(at); + misses[atToMinutes(at) & BUCKET_MASK]++; + ++totalMisses; + } + + void updateSnapshot(ShardedHitRate.Snapshot snapshot, long at) + { + lock.lock(); + try + { + if (at > lastAt) + tick(at); + + int srcIndex = atToMinutes(lastAt) & BUCKET_MASK; + double split = (at % CLOCK_SLICE) / (double)CLOCK_SLICE; + { + snapshot.hits[0] += hits[srcIndex]; + snapshot.misses[0] += misses[srcIndex]; + } + for (int i = 1 ; i < BUCKETS - 1 ; ++i) + { + srcIndex = (srcIndex - 1) & BUCKET_MASK; + long h = hits[srcIndex], m = misses[srcIndex]; + long hSplit = Math.round(h * split), mSplit = Math.round(m * split); + snapshot.hits[i] += hSplit; + snapshot.misses[i] += mSplit; + snapshot.hits[i - 1] += h - hSplit; + snapshot.misses[i - 1] += m - mSplit; + } + { + srcIndex = (srcIndex - 1) & BUCKET_MASK; + long h = hits[srcIndex], m = misses[srcIndex]; + long hSplit = (long) (h * split), mSplit = (long) (m * split); + snapshot.hits[BUCKETS - 2] += h - hSplit; + snapshot.misses[BUCKETS - 2] += m - mSplit; + } + snapshot.totalHits += totalHits; + snapshot.totalMisses += totalMisses; + } + finally + { + lock.unlock(); + } + } + } + + static class Snapshot + { + final long[] hits = new long[BUCKETS - 1]; + final long[] misses = new long[BUCKETS - 1]; + long totalHits, totalMisses; + long lastUpdated = Clock.Global.nanoTime() - REFRESH_RATE; + + double hitRate(int minutes) + { + Invariants.requireArgument(minutes <= BUCKETS); + long misses = 0, hits = 0; + for (int i = 0 ; i < minutes; ++i) + { + hits += this.hits[i]; + misses += this.misses[i]; + } + long total = hits + misses; + if (total == 0) + return 0.0D; + return hits / (double)(total); + } + + long requests(int minutes) + { + Invariants.requireArgument(minutes <= BUCKETS); + long requests = 0; + for (int i = 0 ; i < minutes; ++i) + requests += hits[i] + misses[i]; + return requests; + } + + void reset() + { + Arrays.fill(hits, 0); + Arrays.fill(misses, 0); + totalMisses = totalHits = 0; + } + } + + final CopyOnWriteArrayList shards = new CopyOnWriteArrayList<>(); + final Snapshot snapshot = new Snapshot(); + + public ShardedHitRate() + { + } + + public HitRateShard newShard(Lock guardedBy) + { + HitRateShard shard = new HitRateShard(guardedBy); + shards.add(shard); + return shard; + } + + private void maybeRefresh() + { + if (Clock.Global.nanoTime() - snapshot.lastUpdated >= REFRESH_RATE) + refresh(); + } + + public synchronized void refresh() + { + long at = now(); + snapshot.reset(); + for (HitRateShard shard : shards) + shard.updateSnapshot(snapshot, at); + } + + public synchronized long totalHits() + { + maybeRefresh(); + return snapshot.totalHits; + } + + public synchronized long totalMisses() + { + maybeRefresh(); + return snapshot.totalMisses; + } + + public synchronized long totalRequests() + { + maybeRefresh(); + return snapshot.totalHits + snapshot.totalMisses; + } + + public double hitRateAllTime() + { + maybeRefresh(); + long totalRequests = snapshot.totalHits + snapshot.totalMisses; + if (totalRequests == 0) + return 0.0D; + return snapshot.totalHits / (double)(snapshot.totalHits + snapshot.totalMisses); + } + + public double hitRate(int minutes) + { + maybeRefresh(); + return snapshot.hitRate(minutes); + } + + public double requestsPerSecond(int minutes) + { + maybeRefresh(); + return snapshot.requests(minutes) / (minutes * 60d); + } +} diff --git a/src/java/org/apache/cassandra/metrics/SnapshottingTimer.java b/src/java/org/apache/cassandra/metrics/SnapshottingTimer.java index 3536492a51..b15b9c8b07 100644 --- a/src/java/org/apache/cassandra/metrics/SnapshottingTimer.java +++ b/src/java/org/apache/cassandra/metrics/SnapshottingTimer.java @@ -24,14 +24,14 @@ import com.codahale.metrics.Timer; public class SnapshottingTimer extends Timer { - private final SnapshottingReservoir reservoir; + private final CassandraReservoir reservoir; - public SnapshottingTimer(SnapshottingReservoir reservoir) + public SnapshottingTimer(CassandraReservoir reservoir) { this(reservoir, Clock.defaultClock()); } - public SnapshottingTimer(SnapshottingReservoir reservoir, Clock clock) + public SnapshottingTimer(CassandraReservoir reservoir, Clock clock) { super(reservoir, clock); this.reservoir = reservoir; diff --git a/src/java/org/apache/cassandra/service/RetryStrategy.java b/src/java/org/apache/cassandra/service/RetryStrategy.java index 91c87c520d..120768ad68 100644 --- a/src/java/org/apache/cassandra/service/RetryStrategy.java +++ b/src/java/org/apache/cassandra/service/RetryStrategy.java @@ -247,6 +247,11 @@ public class RetryStrategy implements WaitStrategy return units.convert(result, MICROSECONDS); } + public long computeMaxWait(int attempt, TimeUnit units) + { + return units.convert(max.getMaxMicros(attempt), MICROSECONDS); + } + public static RetryStrategy parse(String spec, LatencySourceFactory latencies) { return parse(spec, latencies, null); diff --git a/src/java/org/apache/cassandra/service/TimeoutStrategy.java b/src/java/org/apache/cassandra/service/TimeoutStrategy.java index 7ca39db222..aebbf3d506 100644 --- a/src/java/org/apache/cassandra/service/TimeoutStrategy.java +++ b/src/java/org/apache/cassandra/service/TimeoutStrategy.java @@ -103,12 +103,14 @@ public class TimeoutStrategy implements WaitStrategy public interface Wait { long getMicros(int attempts); + long getMaxMicros(int attempts); class Constant implements Wait { final long micros; public Constant(long micros) { this.micros = micros; } @Override public long getMicros(int attempts) { return micros; } + @Override public long getMaxMicros(int attempts) { return micros; } } class Modifying implements Wait @@ -127,18 +129,26 @@ public class TimeoutStrategy implements WaitStrategy { return modifier.modify(supplier.getMicros(), attempts); } + + @Override + public long getMaxMicros(int attempts) + { + return modifier.modify(supplier.getMaxMicros(), attempts); + } } } public interface LatencySupplier { long getMicros(); + long getMaxMicros(); class Constant implements LatencySupplier { final long micros; public Constant(long micros) {this.micros = micros; } @Override public long getMicros() { return micros; } + @Override public long getMaxMicros() { return micros; } } class Percentile implements LatencySupplier @@ -152,11 +162,8 @@ public class TimeoutStrategy implements WaitStrategy this.percentile = percentile; } - @Override - public long getMicros() - { - return latencies.get(percentile); - } + @Override public long getMicros() { return latencies.get(percentile); } + @Override public long getMaxMicros() { return latencies.get(1.0); } } } diff --git a/src/java/org/apache/cassandra/service/accord/AccordCache.java b/src/java/org/apache/cassandra/service/accord/AccordCache.java index 991c610cc1..20ce3d3b1f 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCache.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCache.java @@ -59,7 +59,8 @@ import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.exceptions.UnknownTableException; import org.apache.cassandra.io.util.DataInputBuffer; import org.apache.cassandra.metrics.AccordCacheMetrics; -import org.apache.cassandra.metrics.CacheAccessMetrics; +import org.apache.cassandra.metrics.LogLinearHistogram; +import org.apache.cassandra.metrics.ShardedHitRate; import org.apache.cassandra.service.accord.AccordCacheEntry.LoadExecutor; import org.apache.cassandra.service.accord.AccordCacheEntry.Status; import org.apache.cassandra.service.accord.events.CacheEvents; @@ -73,8 +74,6 @@ import static accord.utils.Invariants.require; import static org.apache.cassandra.service.accord.AccordCacheEntry.Status.EVICTED; import static org.apache.cassandra.service.accord.AccordCacheEntry.Status.LOADED; import static org.apache.cassandra.service.accord.AccordCacheEntry.Status.MODIFIED; -import static org.apache.cassandra.service.accord.AccordCacheEntry.Status.SAVING; -import static org.apache.cassandra.service.accord.AccordCacheEntry.Status.WAITING_TO_SAVE; /** * Cache for AccordCommand and AccordCommandsForKey, available memory is shared between the two object types. @@ -122,20 +121,17 @@ public class AccordCache implements CacheSize static class Stats { - long queries; long hits; long misses; } public static final class ImmutableStats { - public final long queries; public final long hits; public final long misses; public ImmutableStats(Stats stats) { - queries = stats.queries; hits = stats.hits; misses = stats.misses; } @@ -146,22 +142,17 @@ public class AccordCache implements CacheSize private final IntrusiveLinkedList> evictQueue = new IntrusiveLinkedList<>(); private final IntrusiveLinkedList> noEvictQueue = new IntrusiveLinkedList<>(); - private int unreferencedBytes; + private long unreferencedBytes; private int unreferenced; private long maxSizeInBytes; private long bytesCached; private int noEvictGeneration; private boolean shrinkingOn = true; - @VisibleForTesting - final AccordCacheMetrics metrics; - final Stats stats = new Stats(); - - public AccordCache(AccordCacheEntry.SaveExecutor saveExecutor, long maxSizeInBytes, AccordCacheMetrics metrics) + public AccordCache(AccordCacheEntry.SaveExecutor saveExecutor, long maxSizeInBytes) { this.saveExecutor = saveExecutor; this.maxSizeInBytes = maxSizeInBytes; - this.metrics = metrics; } @Override @@ -299,6 +290,7 @@ public class AccordCache implements CacheSize Type.Instance owner = node.owner; Type parent = owner.parent(); parent.bytesCached -= node.sizeOnHeap; + parent.objectSize.decrement(node.sizeOnHeap); --parent.size; // TODO (expected): use listeners @@ -348,14 +340,9 @@ public class AccordCache implements CacheSize safeRef.global().owner.release(safeRef, owner); } - public ImmutableStats stats() + public > Type newType(Class keyClass, Adapter adapter, AccordCacheMetrics.Shard metrics) { - return new ImmutableStats(stats); - } - - public > Type newType(Class keyClass, Adapter adapter) - { - Type instance = new Type<>(keyClass, adapter); + Type instance = new Type<>(keyClass, adapter, metrics); types.add(instance); return instance; } @@ -367,9 +354,10 @@ public class AccordCache implements CacheSize Function quickShrink, TriFunction validateFunction, ToLongFunction heapEstimator, - Function, S> safeRefFactory) + Function, S> safeRefFactory, + AccordCacheMetrics.Shard metrics) { - return newType(keyClass, loadFunction, saveFunction, quickShrink, (i, j) -> j, (c, i, j) -> (V)j, validateFunction, heapEstimator, i -> 0, safeRefFactory); + return newType(keyClass, loadFunction, saveFunction, quickShrink, (i, j) -> j, (c, i, j) -> (V)j, validateFunction, heapEstimator, i -> 0, safeRefFactory, metrics); } public > Type newType( @@ -382,12 +370,14 @@ public class AccordCache implements CacheSize TriFunction validateFunction, ToLongFunction heapEstimator, ToLongFunction shrunkHeapEstimator, - Function, S> safeRefFactory) + Function, S> safeRefFactory, + AccordCacheMetrics.Shard metrics) { return newType(keyClass, new FunctionalAdapter<>(loadFunction, saveFunction, quickShrink, fullShrink, inflate, validateFunction, heapEstimator, shrunkHeapEstimator, - safeRefFactory, AccordCacheEntry::createReadyToLoad)); + safeRefFactory, AccordCacheEntry::createReadyToLoad), + metrics); } public Collection> types() @@ -445,7 +435,6 @@ public class AccordCache implements CacheSize private AccordCacheEntry acquire(K key, boolean onlyIfLoaded) { - incrementCacheQueries(); @SuppressWarnings("unchecked") AccordCacheEntry node = cache.get(key); return node == null @@ -691,25 +680,24 @@ public class AccordCache implements CacheSize private int size; @VisibleForTesting - final CacheAccessMetrics typeMetrics; + final LogLinearHistogram objectSize; + final ShardedHitRate.HitRateShard hitRate; private final Stats stats = new Stats(); private List> typeListeners = null; - public Type( - Class keyClass, - Adapter adapter) + public Type(Class keyClass, Adapter adapter, AccordCacheMetrics.Shard metrics) { this.keyClass = keyClass; this.adapter = adapter; - this.typeMetrics = metrics.forInstance(keyClass); + this.objectSize = metrics.objectSize; + this.hitRate = metrics.hitRate; } void updateSize(long newSize, long delta, boolean isUnreferenced, boolean updateHistogram) { - // TODO (expected): deprecate this in favour of a histogram snapshot of any point in time bytesCached += delta; AccordCache.this.bytesCached += delta; - if (updateHistogram) metrics.objectSize.update(newSize); + if (updateHistogram) objectSize.replace(newSize - delta, newSize); if (isUnreferenced) AccordCache.this.unreferencedBytes += delta; } @@ -719,28 +707,16 @@ public class AccordCache implements CacheSize return new Instance(commandStore); } - private void incrementCacheQueries() - { - typeMetrics.requests.mark(); - metrics.requests.mark(); - stats.queries++; - AccordCache.this.stats.queries++; - } - private void incrementCacheHits() { - typeMetrics.hits.mark(); - metrics.hits.mark(); + hitRate.markHitExclusive(); stats.hits++; - AccordCache.this.stats.hits++; } private void incrementCacheMisses() { - typeMetrics.misses.mark(); - metrics.misses.mark(); + hitRate.markMissExclusive(); stats.misses++; - AccordCache.this.stats.misses++; } final AccordCache parent() @@ -758,11 +734,6 @@ public class AccordCache implements CacheSize return new ImmutableStats(stats); } - public Stats globalStats() - { - return AccordCache.this.stats; - } - @VisibleForTesting public void unsafeSetLoadFunction(BiFunction loadFunction) { @@ -896,7 +867,7 @@ public class AccordCache implements CacheSize } @VisibleForTesting - int unreferencedBytes() + public long unreferencedBytes() { return unreferencedBytes; } @@ -957,7 +928,6 @@ public class AccordCache implements CacheSize event.instanceAllocated = type.weightedSize(); AccordCache.Stats stats = type.stats(); - event.instanceStatsQueries = stats.queries; event.instanceStatsHits = stats.hits; event.instanceStatsMisses = stats.misses; @@ -967,11 +937,6 @@ public class AccordCache implements CacheSize event.globalCapacity = type.capacity(); event.globalAllocated = type.globalAllocated(); - stats = type.globalStats(); - event.globalStatsQueries = stats.queries; - event.globalStatsHits = stats.hits; - event.globalStatsMisses = stats.misses; - event.update(); } diff --git a/src/java/org/apache/cassandra/service/accord/AccordCacheEntry.java b/src/java/org/apache/cassandra/service/accord/AccordCacheEntry.java index 2da70603bd..13cc1a4bac 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCacheEntry.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCacheEntry.java @@ -258,6 +258,7 @@ public class AccordCacheEntry extends IntrusiveLinkedListNode // TODO (expected): we aren't weighing the keys sizeOnHeap = Ints.saturatedCast(EMPTY_SIZE); parent.updateSize(sizeOnHeap, sizeOnHeap, false, false); + parent.objectSize.increment(EMPTY_SIZE); } @Override diff --git a/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java b/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java index e283d9f5e0..20c86281dd 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java @@ -190,7 +190,7 @@ public class AccordCommandStore extends CommandStore this.caches = new ExclusiveCaches(sharedExecutor.lock, exclusive.global, commands, commandsForKey); } - this.exclusiveExecutor = sharedExecutor.executor(); + this.exclusiveExecutor = sharedExecutor.executor(id); this.commandsForRanges = new CommandsForRanges.Manager(this); maybeLoadRedundantBefore(journal.loadRedundantBefore(id())); diff --git a/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java b/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java index 18b9bccee2..cbf1fe49dc 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCommandStores.java @@ -37,8 +37,6 @@ import accord.utils.RandomSource; import org.apache.cassandra.cache.CacheSize; import org.apache.cassandra.config.AccordSpec.QueueShardModel; import org.apache.cassandra.config.DatabaseDescriptor; -import org.apache.cassandra.metrics.AccordCacheMetrics; -import org.apache.cassandra.metrics.CacheSizeMetrics; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.AccordExecutor.AccordExecutorFactory; import org.apache.cassandra.service.accord.api.TokenKey; @@ -52,9 +50,6 @@ import static org.apache.cassandra.service.accord.AccordExecutor.constant; public class AccordCommandStores extends CommandStores implements CacheSize { - public static final String ACCORD_STATE_CACHE = "AccordStateCache"; - - private final CacheSizeMetrics cacheSizeMetrics; private final AccordExecutor[] executors; private final int mask; @@ -69,7 +64,6 @@ public class AccordCommandStores extends CommandStores implements CacheSize super(node, agent, store, random, journal, shardDistributor, progressLogFactory, listenerFactory, AccordCommandStore.factory(id -> executors[id % executors.length])); this.executors = executors; - this.cacheSizeMetrics = new CacheSizeMetrics(ACCORD_STATE_CACHE, this); this.mask = Integer.highestOneBit(executors.length) - 1; cacheSize = DatabaseDescriptor.getAccordCacheSizeInMiB() << 20; workingSetSize = DatabaseDescriptor.getAccordWorkingSetSizeInMiB() << 20; @@ -99,7 +93,6 @@ public class AccordCommandStores extends CommandStores implements CacheSize for (int id = 0; id < executors.length; id++) { - AccordCacheMetrics metrics = new AccordCacheMetrics(ACCORD_STATE_CACHE); QueueShardModel shardModel = DatabaseDescriptor.getAccordQueueShardModel(); String baseName = AccordExecutor.class.getSimpleName() + '[' + id; int threads = Math.min(maxThreads, Math.max(DatabaseDescriptor.getAccordConcurrentOps() / getAccordQueueShardCount(), 1)); @@ -107,10 +100,10 @@ public class AccordCommandStores extends CommandStores implements CacheSize { case THREAD_PER_SHARD: case THREAD_PER_SHARD_SYNC_QUEUE: - executors[id] = factory.get(id, shardModel == THREAD_PER_SHARD ? RUN_WITHOUT_LOCK : RUN_WITH_LOCK, 1, constant(baseName + ']'), metrics, agent); + executors[id] = factory.get(id, shardModel == THREAD_PER_SHARD ? RUN_WITHOUT_LOCK : RUN_WITH_LOCK, 1, constant(baseName + ']'), agent); break; case THREAD_POOL_PER_SHARD: - executors[id] = factory.get(id, RUN_WITHOUT_LOCK, threads, num -> baseName + ',' + num + ']', metrics, agent); + executors[id] = factory.get(id, RUN_WITHOUT_LOCK, threads, num -> baseName + ',' + num + ']', agent); break; } } @@ -172,6 +165,11 @@ public class AccordCommandStores extends CommandStores implements CacheSize refreshCapacities(); } + public synchronized void setShrinking(boolean on) + { + shrinkingOn = on; + } + @Override public long capacity() { diff --git a/src/java/org/apache/cassandra/service/accord/AccordConfigurationService.java b/src/java/org/apache/cassandra/service/accord/AccordConfigurationService.java index 37f8d74594..0e09cd2234 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordConfigurationService.java +++ b/src/java/org/apache/cassandra/service/accord/AccordConfigurationService.java @@ -120,6 +120,15 @@ public class AccordConfigurationService extends AbstractConfigurationService diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutor.java b/src/java/org/apache/cassandra/service/accord/AccordExecutor.java index 2cd465c612..b02616816c 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutor.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutor.java @@ -43,6 +43,7 @@ import accord.api.Agent; import accord.api.AsyncExecutor; import accord.api.RoutingKey; import accord.local.Command; +import accord.local.PreLoadContext; import accord.local.SequentialAsyncExecutor; import accord.local.cfk.CommandsForKey; import accord.primitives.TxnId; @@ -88,13 +89,14 @@ import static org.apache.cassandra.service.accord.AccordTask.State.WAITING_TO_RU /** * NOTE: We assume that NO BLOCKING TASKS are submitted to this executor AND WAITED ON by another task executing on this executor. + * (as we do not immediately schedule additional threads for submitted tasks, but schedule new threads only if necessary when the submitting execution completes) */ public abstract class AccordExecutor implements CacheSize, LoadExecutor, Boolean>, SaveExecutor, Shutdownable, AsyncExecutor { private static final Logger logger = LoggerFactory.getLogger(AccordExecutor.class); public interface AccordExecutorFactory { - AccordExecutor get(int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent); + AccordExecutor get(int executorId, Mode mode, int threads, IntFunction name, Agent agent); } public enum Mode { RUN_WITH_LOCK, RUN_WITHOUT_LOCK } @@ -190,19 +192,19 @@ public abstract class AccordExecutor implements CacheSize, LoadExecutor commands; final AccordCache.Type commandsForKey; - commands = cache.newType(TxnId.class, COMMAND_ADAPTER); + commands = cache.newType(TxnId.class, COMMAND_ADAPTER, AccordCacheMetrics.CommandsCacheMetrics.newShard(lock)); registerJfrListener(executorId, commands, "Command"); - commandsForKey = cache.newType(RoutingKey.class, CFK_ADAPTER); + commandsForKey = cache.newType(RoutingKey.class, CFK_ADAPTER, AccordCacheMetrics.CommandsForKeyCacheMetrics.newShard(lock)); registerJfrListener(executorId, commandsForKey, "CommandsForKey"); this.caches = new ExclusiveGlobalCaches(this, cache, commands, commandsForKey); @@ -511,6 +513,11 @@ public abstract class AccordExecutor implements CacheSize, LoadExecutor ownerUpdater = AtomicReferenceFieldUpdater.newUpdater(SequentialExecutor.class, Thread.class, "owner"); public class SequentialExecutor extends TaskQueue implements SequentialAsyncExecutor { + final int commandStoreId; final SequentialQueueTask selfTask; private Task task; private volatile Thread owner, waiting; private boolean running; SequentialExecutor() + { + this(-1); + } + + SequentialExecutor(int commandStoreId) { super(WAITING_TO_RUN); + this.commandStoreId = commandStoreId; this.selfTask = new SequentialQueueTask(this); } @@ -1180,6 +1194,11 @@ public abstract class AccordExecutor implements CacheSize, LoadExecutor + { + public enum Status { WAITING_TO_LOAD, SCANNING_RANGES, LOADING, WAITING_TO_RUN, RUNNING } + + final Status status; + final int commandStoreId; + + final Task task; + + public TaskInfo(Status status, int commandStoreId, Task task) + { + this.status = status; + this.commandStoreId = commandStoreId; + this.task = task; + } + + public Status status() + { + return status; + } + + public Integer commandStoreId() + { + return commandStoreId >= 0 ? commandStoreId : null; + } + + public int position() + { + return task.queuePosition; + } + + public @Nullable String describe() + { + if (task instanceof AccordTask) + return ((AccordTask) task).preLoadContext().reason(); + + if (task instanceof DebuggableTask) + return ((DebuggableTask) task).description(); + + return null; + } + + public @Nullable PreLoadContext preLoadContext() + { + if (task instanceof AccordTask) + return ((AccordTask) task).preLoadContext(); + return null; + } + + @Override + public int compareTo(TaskInfo that) + { + int c = this.status.compareTo(that.status); + if (c == 0) c = this.position() - that.position(); + return c; + } + } + + public List taskSnapshot() + { + List result = new ArrayList<>(); + lock.lock(); + try + { + addToSnapshot(result, waitingToLoad, TaskInfo.Status.WAITING_TO_LOAD, TaskInfo.Status.WAITING_TO_LOAD); + addToSnapshot(result, waitingToLoadRangeTxns, TaskInfo.Status.WAITING_TO_LOAD, TaskInfo.Status.WAITING_TO_LOAD); + addToSnapshot(result, scanningRanges, TaskInfo.Status.SCANNING_RANGES, TaskInfo.Status.SCANNING_RANGES); + addToSnapshot(result, loading, TaskInfo.Status.LOADING, TaskInfo.Status.LOADING); + addToSnapshot(result, waitingToRun, TaskInfo.Status.WAITING_TO_RUN, TaskInfo.Status.WAITING_TO_RUN); + addToSnapshot(result, running, TaskInfo.Status.RUNNING, TaskInfo.Status.WAITING_TO_RUN); + } + finally + { + lock.unlock(); + } + result.sort(TaskInfo::compareTo); + return result; + } + + private static List toSimpleSnapshotList(TaskQueue queue) + { + List list = new ArrayList<>(); + for (int i = 0 ; i < queue.size() ; i++) + list.add(queue.get(i)); + return list; + } + + private static void addToSnapshot(List snapshot, TaskQueue queue, TaskInfo.Status ifCurrent, TaskInfo.Status ifQueued) + { + for (int i = 0 ; i < queue.size() ; ++i) + { + Task t = queue.get(i); + if (t instanceof SequentialQueueTask) + { + SequentialExecutor q = ((SequentialQueueTask) t).queue; + snapshot.add(new TaskInfo(ifCurrent, q.commandStoreId, q.task)); + for (int j = 0 ; j < q.size() ; ++j) + snapshot.add(new TaskInfo(ifQueued, q.commandStoreId, q.get(j))); + } + else + { + int commmandStoreId = t instanceof AccordTask ? ((AccordTask) t).commandStore.id() : -1; + snapshot.add(new TaskInfo(ifCurrent, commmandStoreId, t)); + } + } + } + } diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractLockLoop.java b/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractLockLoop.java index 2b80cbac8d..a23e47787f 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractLockLoop.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractLockLoop.java @@ -25,7 +25,6 @@ import accord.api.Agent; import accord.utils.QuadFunction; import accord.utils.QuintConsumer; import org.apache.cassandra.concurrent.DebuggableTask.DebuggableTaskRunner; -import org.apache.cassandra.metrics.AccordCacheMetrics; import org.apache.cassandra.service.accord.AccordExecutorLoops.LoopTask; import org.apache.cassandra.utils.concurrent.ConcurrentLinkedStack; @@ -37,9 +36,9 @@ abstract class AccordExecutorAbstractLockLoop extends AccordExecutor boolean isHeldByExecutor; boolean shutdown; - AccordExecutorAbstractLockLoop(Lock lock, int executorId, AccordCacheMetrics metrics, Agent agent) + AccordExecutorAbstractLockLoop(Lock lock, int executorId, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); } abstract void notifyWork(); diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractSemiSyncSubmit.java b/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractSemiSyncSubmit.java index eb0f56fe66..428270fc6b 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractSemiSyncSubmit.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutorAbstractSemiSyncSubmit.java @@ -23,13 +23,12 @@ import java.util.concurrent.locks.Lock; import accord.api.Agent; import accord.utils.QuadFunction; import accord.utils.QuintConsumer; -import org.apache.cassandra.metrics.AccordCacheMetrics; abstract class AccordExecutorAbstractSemiSyncSubmit extends AccordExecutorAbstractLockLoop { - AccordExecutorAbstractSemiSyncSubmit(Lock lock, int executorId, AccordCacheMetrics metrics, Agent agent) + AccordExecutorAbstractSemiSyncSubmit(Lock lock, int executorId, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); } abstract void awaitExclusive() throws InterruptedException; diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutorAsyncSubmit.java b/src/java/org/apache/cassandra/service/accord/AccordExecutorAsyncSubmit.java index cefdb46b59..06ebd26964 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutorAsyncSubmit.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutorAsyncSubmit.java @@ -22,7 +22,6 @@ import java.util.concurrent.TimeUnit; import java.util.function.IntFunction; import accord.api.Agent; -import org.apache.cassandra.metrics.AccordCacheMetrics; import org.apache.cassandra.utils.concurrent.LockWithAsyncSignal; // WARNING: experimental - needs more testing @@ -31,14 +30,14 @@ class AccordExecutorAsyncSubmit extends AccordExecutorAbstractSemiSyncSubmit private final AccordExecutorLoops loops; private final LockWithAsyncSignal lock; - public AccordExecutorAsyncSubmit(int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorAsyncSubmit(int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - this(new LockWithAsyncSignal(), executorId, mode, threads, name, metrics, agent); + this(new LockWithAsyncSignal(), executorId, mode, threads, name, agent); } - private AccordExecutorAsyncSubmit(LockWithAsyncSignal lock, int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + private AccordExecutorAsyncSubmit(LockWithAsyncSignal lock, int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); this.lock = lock; this.loops = new AccordExecutorLoops(mode, threads, name, this::task); } diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutorSemiSyncSubmit.java b/src/java/org/apache/cassandra/service/accord/AccordExecutorSemiSyncSubmit.java index c7677f8424..3ccbb5933a 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutorSemiSyncSubmit.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutorSemiSyncSubmit.java @@ -24,7 +24,6 @@ import java.util.concurrent.locks.ReentrantLock; import java.util.function.IntFunction; import accord.api.Agent; -import org.apache.cassandra.metrics.AccordCacheMetrics; // WARNING: experimental - needs more testing class AccordExecutorSemiSyncSubmit extends AccordExecutorAbstractSemiSyncSubmit @@ -33,14 +32,14 @@ class AccordExecutorSemiSyncSubmit extends AccordExecutorAbstractSemiSyncSubmit private final ReentrantLock lock; private final Condition hasWork; - public AccordExecutorSemiSyncSubmit(int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorSemiSyncSubmit(int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - this(new ReentrantLock(), executorId, mode, threads, name, metrics, agent); + this(new ReentrantLock(), executorId, mode, threads, name, agent); } - private AccordExecutorSemiSyncSubmit(ReentrantLock lock, int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + private AccordExecutorSemiSyncSubmit(ReentrantLock lock, int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); this.lock = lock; this.hasWork = lock.newCondition(); this.loops = new AccordExecutorLoops(mode, threads, name, this::task); diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutorSimple.java b/src/java/org/apache/cassandra/service/accord/AccordExecutorSimple.java index 11a1d04ca0..d5712ef53b 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutorSimple.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutorSimple.java @@ -27,7 +27,6 @@ import accord.utils.Invariants; import accord.utils.QuadFunction; import accord.utils.QuintConsumer; import org.apache.cassandra.concurrent.ExecutorPlus; -import org.apache.cassandra.metrics.AccordCacheMetrics; import static org.apache.cassandra.concurrent.ExecutorFactory.Global.executorFactory; @@ -37,26 +36,26 @@ class AccordExecutorSimple extends AccordExecutor final ReentrantLock lock; private Task active; - public AccordExecutorSimple(int executorId, String name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorSimple(int executorId, String name, Agent agent) { - this(new ReentrantLock(), executorId, name, metrics, agent); + this(new ReentrantLock(), executorId, name, agent); } - public AccordExecutorSimple(int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorSimple(int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - this(new ReentrantLock(), executorId, mode, threads, name, metrics, agent); + this(new ReentrantLock(), executorId, mode, threads, name, agent); } - private AccordExecutorSimple(ReentrantLock lock, int executorId, String name, AccordCacheMetrics metrics, Agent agent) + private AccordExecutorSimple(ReentrantLock lock, int executorId, String name, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); this.lock = lock; this.executor = executorFactory().sequential(name); } - public AccordExecutorSimple(ReentrantLock lock, int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorSimple(ReentrantLock lock, int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); Invariants.requireArgument(threads == 1); this.lock = lock; this.executor = executorFactory().sequential(name.apply(0)); diff --git a/src/java/org/apache/cassandra/service/accord/AccordExecutorSyncSubmit.java b/src/java/org/apache/cassandra/service/accord/AccordExecutorSyncSubmit.java index 89f836cd2c..8243e81e52 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordExecutorSyncSubmit.java +++ b/src/java/org/apache/cassandra/service/accord/AccordExecutorSyncSubmit.java @@ -26,7 +26,6 @@ import java.util.function.IntFunction; import accord.api.Agent; import accord.utils.QuadFunction; import accord.utils.QuintConsumer; -import org.apache.cassandra.metrics.AccordCacheMetrics; class AccordExecutorSyncSubmit extends AccordExecutorAbstractLockLoop { @@ -34,19 +33,19 @@ class AccordExecutorSyncSubmit extends AccordExecutorAbstractLockLoop private final ReentrantLock lock; private final Condition hasWork; - public AccordExecutorSyncSubmit(int executorId, Mode mode, String name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorSyncSubmit(int executorId, Mode mode, String name, Agent agent) { - this(executorId, mode, 1, constant(name), metrics, agent); + this(executorId, mode, 1, constant(name), agent); } - public AccordExecutorSyncSubmit(int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + public AccordExecutorSyncSubmit(int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - this(new ReentrantLock(), executorId, mode, threads, name, metrics, agent); + this(new ReentrantLock(), executorId, mode, threads, name, agent); } - private AccordExecutorSyncSubmit(ReentrantLock lock, int executorId, Mode mode, int threads, IntFunction name, AccordCacheMetrics metrics, Agent agent) + private AccordExecutorSyncSubmit(ReentrantLock lock, int executorId, Mode mode, int threads, IntFunction name, Agent agent) { - super(lock, executorId, metrics, agent); + super(lock, executorId, agent); this.lock = lock; this.hasWork = lock.newCondition(); this.loops = new AccordExecutorLoops(mode, threads, name, this::task); diff --git a/src/java/org/apache/cassandra/service/accord/AccordJournal.java b/src/java/org/apache/cassandra/service/accord/AccordJournal.java index 20bc467c15..8e0d1b6509 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordJournal.java +++ b/src/java/org/apache/cassandra/service/accord/AccordJournal.java @@ -1049,7 +1049,7 @@ public class AccordJournal implements accord.api.Journal, RangeSearcher.Supplier { // Since we are iterating in reverse order, we skip the fields that were // set by entries written later (i.e. already read ones). - if (isChanged(field, flags) && field != CLEANUP) + if (isChanged(field, flags | mask) && field != CLEANUP) skip(txnId, field, in, userVersion); else deserialize(field, in, userVersion); diff --git a/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java b/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java index 19e37f06c3..40305006cd 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java +++ b/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java @@ -38,6 +38,7 @@ import accord.messages.Reply; import accord.messages.ReplyContext; import accord.messages.Request; import accord.primitives.TxnId; +import accord.utils.async.Cancellable; import org.apache.cassandra.exceptions.RequestFailureReason; import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.net.Message; @@ -180,7 +181,7 @@ public class AccordMessageSink implements MessageSink // TODO (expected): permit bulk send to save esp. on callback registration (and combine records) @Override - public void send(Node.Id to, Request request, int attempt, AsyncExecutor executor, Callback callback) + public Cancellable send(Node.Id to, Request request, int attempt, AsyncExecutor executor, Callback callback) { Verb verb = VerbMapping.getVerb(request); Preconditions.checkNotNull(verb, "Verb is null for type %s", request.type()); @@ -214,8 +215,9 @@ public class AccordMessageSink implements MessageSink Message message = Message.out(verb, request, expiresAtNanos); InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(to); logger.trace("Sending {} {} to {}", verb, message.payload, endpoint); - callbacks.registerAt(message.id(), executor, callback, to, nowNanos, slowAtNanos, expiresAtNanos, NANOSECONDS); + Cancellable cancellable = callbacks.registerAt(message.id(), executor, callback, to, nowNanos, slowAtNanos, expiresAtNanos, NANOSECONDS); messaging.send(message, endpoint); + return cancellable; } @Override diff --git a/src/java/org/apache/cassandra/service/accord/AccordResponseVerbHandler.java b/src/java/org/apache/cassandra/service/accord/AccordResponseVerbHandler.java index 330043d351..6c36d4a1fd 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordResponseVerbHandler.java +++ b/src/java/org/apache/cassandra/service/accord/AccordResponseVerbHandler.java @@ -22,7 +22,6 @@ import java.util.concurrent.TimeUnit; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import accord.coordinate.Timeout; import accord.impl.RequestCallbacks; import accord.local.Node; import accord.messages.Reply; @@ -84,7 +83,7 @@ class AccordResponseVerbHandler implements IVerbHandler private static Throwable convertFailureMessage(RequestFailure failure) { return failure.reason == RequestFailureReason.TIMEOUT ? - new Timeout(null, null) : + null : new RuntimeException(failure.failure); } diff --git a/src/java/org/apache/cassandra/service/accord/AccordResult.java b/src/java/org/apache/cassandra/service/accord/AccordResult.java index df1f17eb7d..8a59ba4304 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordResult.java +++ b/src/java/org/apache/cassandra/service/accord/AccordResult.java @@ -21,6 +21,7 @@ package org.apache.cassandra.service.accord; import java.util.HashSet; import java.util.Iterator; import java.util.Set; +import java.util.concurrent.TimeoutException; import java.util.function.BiConsumer; import javax.annotation.Nullable; @@ -41,6 +42,7 @@ import org.apache.cassandra.exceptions.ReadFailureException; import org.apache.cassandra.exceptions.ReadTimeoutException; import org.apache.cassandra.exceptions.RequestExecutionException; import org.apache.cassandra.exceptions.RequestFailureException; +import org.apache.cassandra.exceptions.RequestTimeoutException; import org.apache.cassandra.schema.Schema; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.api.AccordAgent; @@ -78,11 +80,11 @@ public class AccordResult extends AsyncFuture implements BiConsumer extends AsyncFuture implements BiConsumer txnDroppedTables = txnDroppedTables(keysOrRanges); - Tracing.trace("Accord returned topology mismatch: " + coordinationFailed.getMessage()); - logger.debug("Accord returned topology mismatch", coordinationFailed); - bookkeeping.markTopologyMismatch(); - // Throw IRE in case the caller fails to check if the table still exists - if (!txnDroppedTables.isEmpty()) - { - Tracing.trace("Accord txn uses dropped tables {}", txnDroppedTables); - logger.debug("Accord txn uses dropped tables {}", txnDroppedTables); - throw new InvalidRequestException("Accord transaction uses dropped tables"); - } - trySuccess((V) RetryWithNewProtocolResult.instance); - return; + bookkeeping.newTimeout(txnId, keysOrRanges); + report = (RequestFailureException)maybeWrappedInRequestFailureException; } else { - report = bookkeeping.newFailed(txnId, keysOrRanges); + report = bookkeeping.newTimeout(txnId, keysOrRanges); } - // this case happens when a non-timeout exception is seen, and we are unable to move forward - if (txnId != null && txnId.isSyncPoint()) - AccordAgent.onFailedBarrier(txnId, fail); + } + else if (coordinationFailed instanceof Preempted) + { + report = bookkeeping.newPreempted(txnId, keysOrRanges); + } + else if (coordinationFailed instanceof Exhausted) + { + report = bookkeeping.newExhausted(txnId, keysOrRanges); + } + else if (isTxnRequest && coordinationFailed instanceof TopologyMismatch) + { + // Excluding bugs topology mismatch can occur because a table was dropped in between creating the txn + // and executing it. + // It could also race with the table stopping/starting being managed by Accord. + // The caller can retry if the table indeed exists and is managed by Accord. + Set txnDroppedTables = txnDroppedTables(keysOrRanges); + Tracing.trace("Accord returned topology mismatch: " + coordinationFailed.getMessage()); + logger.debug("Accord returned topology mismatch", coordinationFailed); + bookkeeping.markTopologyMismatch(); + // Throw IRE in case the caller fails to check if the table still exists + if (!txnDroppedTables.isEmpty()) + { + Tracing.trace("Accord txn uses dropped tables {}", txnDroppedTables); + logger.debug("Accord txn uses dropped tables {}", txnDroppedTables); + throw new InvalidRequestException("Accord transaction uses dropped tables"); + } + trySuccess((V) RetryWithNewProtocolResult.instance); + return false; } else { - logger.error("Unexpected exception", fail); - JVMStabilityInspector.inspectThrowable(fail); report = bookkeeping.newFailed(txnId, keysOrRanges); } - report.addSuppressed(fail); - tryFailure(report); + // this case happens when a non-timeout exception is seen, and we are unable to move forward + if (txnId != null && txnId.isSyncPoint()) + AccordAgent.onFailedBarrier(txnId, fail); + } + else if (fail instanceof RequestTimeoutException || fail instanceof TimeoutException) + { + report = bookkeeping.newTimeout(txnId, keysOrRanges); } else { - if (success == RetryWithNewProtocolResult.instance) - { - bookkeeping.markRetryDifferentSystem(); - Tracing.trace("Got retry different system error from Accord, will retry"); - } - trySuccess(success); + logger.error("Unexpected exception", fail); + JVMStabilityInspector.inspectThrowable(fail); + report = bookkeeping.newFailed(txnId, keysOrRanges); } + report.addSuppressed(fail); + if (!super.tryFailure(report)) + return false; + bookkeeping.markElapsedNanos(nanoTime() - startedAtNanos); + return true; + } + + @Override + protected boolean trySuccess(V success) + { + if (!super.trySuccess(success)) + return false; + + bookkeeping.markElapsedNanos(nanoTime() - startedAtNanos); + if (success == RetryWithNewProtocolResult.instance) + { + bookkeeping.markRetryDifferentSystem(); + Tracing.trace("Got retry different system error from Accord, will retry"); + } + return true; + } + + @Override + public void accept(V success, Throwable fail) + { + if (fail == null) trySuccess(success); + else tryFailure(fail); } @Override @@ -204,7 +226,7 @@ public class AccordResult extends AsyncFuture implements BiConsumer coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull Dispatcher.RequestTime requestTime, long minHlc) { diff --git a/src/java/org/apache/cassandra/service/accord/IAccordService.java b/src/java/org/apache/cassandra/service/accord/IAccordService.java index 1c5d624aab..0ad5f0c6dd 100644 --- a/src/java/org/apache/cassandra/service/accord/IAccordService.java +++ b/src/java/org/apache/cassandra/service/accord/IAccordService.java @@ -106,6 +106,7 @@ public interface IAccordService IAccordResult addCallback(BiConsumer callback); } + boolean isEnabled(); long currentEpoch(); void setCacheSize(long kb); @@ -248,6 +249,12 @@ public interface IAccordService throw new UnsupportedOperationException("No accord transaction should be executed when accord.enabled = false in cassandra.yaml"); } + @Override + public boolean isEnabled() + { + return false; + } + @Override public long currentEpoch() { @@ -383,7 +390,7 @@ public interface IAccordService class DelegatingAccordService implements IAccordService { - protected final IAccordService delegate; + protected IAccordService delegate; public DelegatingAccordService(IAccordService delegate) { @@ -439,6 +446,12 @@ public interface IAccordService return delegate.executors(); } + @Override + public boolean isEnabled() + { + return delegate.isEnabled(); + } + @Nonnull @Override public IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime, long minHlc) diff --git a/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java b/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java index 93ef9457bd..092801fcba 100644 --- a/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java +++ b/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java @@ -30,7 +30,7 @@ import org.slf4j.LoggerFactory; import accord.api.Agent; import accord.api.CoordinatorEventListener; -import accord.api.LocalEventListener; +import accord.api.ReplicaEventListener; import accord.api.ProgressLog.BlockedUntil; import accord.api.RoutingKey; import accord.api.Tracing; @@ -62,7 +62,8 @@ import accord.utils.async.AsyncResult; import accord.utils.async.AsyncResults; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.exceptions.RequestTimeoutException; -import org.apache.cassandra.metrics.AccordMetrics; +import org.apache.cassandra.metrics.AccordCoordinatorMetrics; +import org.apache.cassandra.metrics.AccordReplicaMetrics; import org.apache.cassandra.net.ResponseContext; import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.AccordTracing; @@ -71,11 +72,13 @@ import org.apache.cassandra.service.accord.txn.TxnQuery; import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.utils.Clock; import org.apache.cassandra.utils.JVMStabilityInspector; +import org.apache.cassandra.utils.NoSpamLogger; import static accord.primitives.Routable.Domain.Key; import static accord.utils.SortedArrays.SortedArrayList.ofSorted; import static java.util.concurrent.TimeUnit.MICROSECONDS; import static java.util.concurrent.TimeUnit.MILLISECONDS; +import static java.util.concurrent.TimeUnit.MINUTES; import static java.util.concurrent.TimeUnit.NANOSECONDS; import static java.util.concurrent.TimeUnit.SECONDS; import static org.apache.cassandra.config.DatabaseDescriptor.getAccordScheduleDurabilityTxnIdLag; @@ -94,6 +97,7 @@ import static org.apache.cassandra.utils.Clock.Global.nanoTime; public class AccordAgent implements Agent { private static final Logger logger = LoggerFactory.getLogger(AccordAgent.class); + private static final NoSpamLogger noSpamLogger = NoSpamLogger.getLogger(logger, 1L, MINUTES); private static BiConsumer onFailedBarrier; public static void setOnFailedBarrier(BiConsumer newOnFailedBarrier) { onFailedBarrier = newOnFailedBarrier; } @@ -224,17 +228,20 @@ public class AccordAgent implements Agent @Override public CoordinatorEventListener coordinatorEvents() { - return AccordMetrics.Listener.instance; + return AccordCoordinatorMetrics.Listener.instance; } @Override - public LocalEventListener localEvents() + public ReplicaEventListener replicaEvents() { - return AccordMetrics.Listener.instance; + return AccordReplicaMetrics.Listener.instance; } + private static final long ONE_SECOND = SECONDS.toMicros(1L); + private static final long ONE_MINUTE = MINUTES.toMicros(1L); + @Override - public long slowCoordinatorDelay(Node node, SafeCommandStore safeStore, TxnId txnId, TimeUnit units, int retryCount) + public long slowCoordinatorDelay(Node node, SafeCommandStore safeStore, TxnId txnId, TimeUnit units, int attempt) { SafeCommand safeCommand = safeStore.unsafeGetNoCleanup(txnId); Invariants.nonNull(safeCommand); @@ -242,29 +249,48 @@ public class AccordAgent implements Agent Command command = safeCommand.current(); Invariants.nonNull(command); + // TODO (expected): make this a configurable calculation on normal request latencies (like ContentionStrategy) + long nowMicros = MILLISECONDS.toMicros(Clock.Global.currentTimeMillis()); + long mostRecentStart = mostRecentStart(command, nowMicros); + long waitMicros = recover(txnId).computeWait(attempt, MICROSECONDS); + long startTime = mostRecentStart + waitMicros; + if (startTime < nowMicros) + { + // TODO (expected): support no waiting here + if (attempt == 1) + return 1; + + startTime = nowMicros + waitMicros/2; + } + RoutingKey homeKey = command.route().homeKey(); Shard shard = node.topology().forEpochIfKnown(homeKey, command.txnId().epoch()); - // TODO (expected): make this a configurable calculation on normal request latencies (like ContentionStrategy) - long nowMicros = MILLISECONDS.toMicros(Clock.Global.currentTimeMillis()); - long oneSecond = SECONDS.toMicros(1L); - long promisedHlc = command.promised().hlc(); - if (promisedHlc > nowMicros + TimeUnit.MINUTES.toMicros(1)) - promisedHlc = 0; - long mostRecentStart = Math.max(command.txnId().hlc(), promisedHlc); - long waitMicros = recover(txnId).computeWait(retryCount, MICROSECONDS); - if (mostRecentStart > nowMicros + SECONDS.toMicros(1L)) - logger.warn("max({},{})>{}", command.txnId(), command.promised(), nowMicros); - long startTime = mostRecentStart + waitMicros; - if (startTime < nowMicros) - startTime = nowMicros + waitMicros/2; - - startTime = nonClashingStartTime(startTime, shard == null ? null : shard.nodes, node.id(), oneSecond, random); + startTime = nonClashingStartTime(startTime, shard == null ? null : shard.nodes, node.id(), ONE_SECOND, random); long delayMicros = Math.max(1, startTime - nowMicros); Invariants.require(delayMicros < TimeUnit.HOURS.toMicros(1L), "unexpectedly long coordination recovery delay proposed: %d (start %d, now %d)", delayMicros, startTime, nowMicros, command.txnId(), command.promised()); return units.convert(delayMicros, MICROSECONDS); } + private static long mostRecentStart(Command command, long nowMicros) + { + // TODO (expected): make this a configurable calculation on normal request latencies (like ContentionStrategy) + long promisedHlc = command.promised().hlc(); + if (promisedHlc > nowMicros + ONE_MINUTE) + promisedHlc = 0; + long result = Math.max(command.txnId().hlc(), promisedHlc); + if (result > nowMicros + ONE_SECOND) + noSpamLogger.warn("max({},{})>{}", command.txnId(), command.promised(), nowMicros); + return result; + } + + @Override + public boolean isSlowCoordinator(long elapsed, TimeUnit units, TxnId txnId, int attempt) + { + long maxWait = recover(txnId).computeMaxWait(attempt, units); + return elapsed >= maxWait; + } + @VisibleForTesting public static long nonClashingStartTime(long startTime, SortedList nodes, Node.Id id, long granularity, RandomSource random) { @@ -291,7 +317,18 @@ public class AccordAgent implements Agent @Override public long slowReplicaDelay(Node node, SafeCommandStore safeStore, TxnId txnId, int attempt, BlockedUntil blockedUntil, TimeUnit units) { - return fetch(txnId).computeWait(attempt, units); + Command command = Invariants.nonNull(safeStore.unsafeGetNoCleanup(txnId).current()); + long nowMicros = MILLISECONDS.toMicros(Clock.Global.currentTimeMillis()); + long mostRecentStart = mostRecentStart(command, nowMicros); + long waitMicros = fetch(txnId).computeWait(attempt, units); + long startTime = mostRecentStart + waitMicros; + if (startTime < nowMicros) + { + // TODO (expected): support no waiting here + if (attempt == 1) return 1; + else return waitMicros/2; + } + return waitMicros; } @Override diff --git a/src/java/org/apache/cassandra/service/accord/events/CacheEvents.java b/src/java/org/apache/cassandra/service/accord/events/CacheEvents.java index b37e418509..68a806eaa6 100644 --- a/src/java/org/apache/cassandra/service/accord/events/CacheEvents.java +++ b/src/java/org/apache/cassandra/service/accord/events/CacheEvents.java @@ -40,7 +40,7 @@ public abstract class CacheEvents extends Event // instance @DataAmount(DataAmount.BYTES) public long instanceAllocated; - public long instanceStatsQueries, instanceStatsHits, instanceStatsMisses; + public long instanceStatsHits, instanceStatsMisses; @Percentage public double instanceStatsHitRate; @@ -50,17 +50,11 @@ public abstract class CacheEvents extends Event public long globalCapacity, globalAllocated; public int globalSize, globalReferenced, globalUnreferenced; - public long globalStatsQueries, globalStatsHits, globalStatsMisses; - - @Percentage - public double globalStatsHitRate; - @Percentage public double globalFree; public void update() { - instanceStatsHitRate = 1D - (instanceStatsHits / (double) instanceStatsQueries); - globalStatsHitRate = 1D - (globalStatsHits / (double) globalStatsQueries); + instanceStatsHitRate = 1D - (instanceStatsHits / (double) (instanceStatsHits + instanceStatsMisses)); globalFree = 1.0D - (globalAllocated / (double) globalCapacity); } diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropAdapter.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropAdapter.java index e33a2072c0..90df4950b1 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropAdapter.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropAdapter.java @@ -119,8 +119,8 @@ public class AccordInteropAdapter extends TxnAdapter return false; Topologies all = execution(node, any, sendTo, selectSendTo, fullRoute, txnId, executeAt); - new AccordInteropPersist(node, executor, all, txnId, require, ballot, txn, executeAt, deps, writes, result, fullRoute, consistencyLevel, CoordinationFlags.none(), informDurableOnDone, callback) - .start(Minimal, any, writes, result); + new AccordInteropPersist(node, executor, all, txnId, require, ballot, txn, executeAt, deps, writes, result, fullRoute, consistencyLevel, CoordinationFlags.none(), informDurableOnDone, Minimal, callback) + .start(); return true; } } diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropApply.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropApply.java index 69427e3ef6..cd83f3ee7d 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropApply.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropApply.java @@ -166,14 +166,6 @@ public class AccordInteropApply extends Apply implements LocalListeners.ComplexL node.reply(replyTo, replyContext, ApplyReply.Applied, null); } - @Override - public ApplyReply reduce(ApplyReply r1, ApplyReply r2) - { - return r1 == null || r2 == null - ? r1 == null ? r2 : r1 - : r1.compareTo(r2) >= 0 ? r1 : r2; - } - @Override protected void acceptInternal(ApplyReply reply, Throwable failure) { diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java index 864752d9c6..72ac8d75e9 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java @@ -31,7 +31,6 @@ import accord.api.Data; import accord.api.Result; import accord.coordinate.CoordinationAdapter; import accord.coordinate.ExecuteFlag.CoordinationFlags; -import accord.coordinate.Timeout; import accord.local.Node; import accord.local.Node.Id; import accord.local.SequentialAsyncExecutor; @@ -67,8 +66,6 @@ import org.apache.cassandra.db.partitions.PartitionIterator; import org.apache.cassandra.db.rows.RowIterator; import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; -import org.apache.cassandra.exceptions.ReadFailureException; -import org.apache.cassandra.exceptions.ReadTimeoutException; import org.apache.cassandra.locator.EndpointsForToken; import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.locator.Replica; @@ -391,28 +388,11 @@ public class AccordInteropExecution implements ReadCoordinator } else { - callback.accept(null, maybeWrapRequestFailureException(failure)); + callback.accept(null, failure); } }); } - /** - * Interop should expose these exceptions as the appropriate Accord types so AccordService - * knows how to handle them - */ - private Throwable maybeWrapRequestFailureException(Throwable failure) - { - Throwable toCheck = failure; - do - { - // TODO (required): There are probably more exceptions that will have this issue of wanting - // to be turned into the top level exception sent back to the client - if (toCheck instanceof ReadTimeoutException || toCheck instanceof ReadFailureException) - return new Timeout(txnId, route.homeKey(), failure); - } while ((toCheck = toCheck.getCause()) != null); - return failure; - } - private AsyncChain executeUnrecoverableRepairUpdate() { return AsyncChains.ofCallable(Stage.ACCORD_MIGRATION.executor(), () -> { diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropPersist.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropPersist.java index 5c494fc386..d567d2a175 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropPersist.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropPersist.java @@ -40,6 +40,7 @@ import accord.primitives.TxnId; import accord.primitives.Writes; import accord.topology.Topologies; import accord.utils.Invariants; +import accord.utils.UnhandledEnum; import org.apache.cassandra.db.ConsistencyLevel; import org.apache.cassandra.utils.Throwables; @@ -111,9 +112,9 @@ public class AccordInteropPersist extends Persist private final ConsistencyLevel consistencyLevel; private CallbackHolder callback; - public AccordInteropPersist(Node node, SequentialAsyncExecutor executor, Topologies topologies, TxnId txnId, Route sendTo, Ballot ballot, Txn txn, Timestamp executeAt, Deps deps, Writes writes, Result result, FullRoute fullRoute, ConsistencyLevel consistencyLevel, ExecuteFlag.CoordinationFlags flags, boolean informDurableOnDone, BiConsumer clientCallback) + public AccordInteropPersist(Node node, SequentialAsyncExecutor executor, Topologies topologies, TxnId txnId, Route sendTo, Ballot ballot, Txn txn, Timestamp executeAt, Deps deps, Writes writes, Result result, FullRoute fullRoute, ConsistencyLevel consistencyLevel, ExecuteFlag.CoordinationFlags flags, boolean informDurableOnDone, Apply.Kind applyKind, BiConsumer clientCallback) { - super(node, executor, topologies, txnId, ballot, sendTo, txn, executeAt, deps, writes, result, fullRoute, flags, informDurableOnDone, AccordInteropApply.FACTORY); + super(node, executor, topologies, txnId, ballot, sendTo, txn, executeAt, deps, writes, result, fullRoute, flags, informDurableOnDone, AccordInteropApply.FACTORY, applyKind); Invariants.requireArgument(consistencyLevel == ConsistencyLevel.QUORUM || consistencyLevel == ConsistencyLevel.ALL || consistencyLevel == ConsistencyLevel.SERIAL || consistencyLevel == ConsistencyLevel.ONE); this.consistencyLevel = consistencyLevel; registerClientCallback(result, clientCallback); @@ -127,10 +128,10 @@ public class AccordInteropPersist extends Persist case ONE: // Can safely upgrade ONE to QUORUM/SERIAL to get a synchronous commit case SERIAL: case QUORUM: - callback = new CallbackHolder(new QuorumTracker(topologies), result, clientCallback); + callback = new CallbackHolder(new QuorumTracker(tracker.topologies()), result, clientCallback); break; case ALL: - callback = new CallbackHolder(new AllTracker(topologies), result, clientCallback); + callback = new CallbackHolder(new AllTracker(tracker.topologies()), result, clientCallback); break; default: throw new IllegalArgumentException("Unhandled consistency level: " + consistencyLevel); @@ -141,8 +142,9 @@ public class AccordInteropPersist extends Persist public void onSuccess(Node.Id from, Apply.ApplyReply reply) { super.onSuccess(from, reply); - switch (reply) + switch (reply.kind) { + case InsufficientEpochs: throw UnhandledEnum.invalid(reply.kind); case Redundant: case Applied: callback.recordSuccess(from); @@ -151,19 +153,27 @@ public class AccordInteropPersist extends Persist // On insufficient Persist will send a commit with the missing information // which will allow a final response to be returned later that could be successful return; - default: throw new IllegalArgumentException("Unhandled apply response " + reply); + default: throw UnhandledEnum.unknown(reply.kind); } } + @Override + public void start() + { + super.start(); + } + @Override public void onFailure(Node.Id from, Throwable failure) { callback.recordFailure(from, failure); + super.onFailure(from, failure); } @Override public boolean onCallbackFailure(Node.Id from, Throwable failure) { + super.onCallbackFailure(from, failure); return callback.recordCallbackFailure(failure); } } diff --git a/src/java/org/apache/cassandra/service/accord/repair/RequiredResponseTracker.java b/src/java/org/apache/cassandra/service/accord/repair/RequiredResponseTracker.java index ac2651dcd3..1e6fd62b09 100644 --- a/src/java/org/apache/cassandra/service/accord/repair/RequiredResponseTracker.java +++ b/src/java/org/apache/cassandra/service/accord/repair/RequiredResponseTracker.java @@ -58,6 +58,12 @@ public class RequiredResponseTracker extends SimpleTracker requiredResponses, Topologies topologies) diff --git a/src/java/org/apache/cassandra/service/accord/serializers/ApplySerializers.java b/src/java/org/apache/cassandra/service/accord/serializers/ApplySerializers.java index e3c009d17e..8fdb1e0d80 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/ApplySerializers.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/ApplySerializers.java @@ -23,6 +23,7 @@ import java.io.IOException; import accord.api.Result; import accord.coordinate.ExecuteFlag.ExecuteFlags; import accord.messages.Apply; +import accord.messages.Apply.ApplyReply; import accord.primitives.Ballot; import accord.primitives.FullRoute; import accord.primitives.PartialDeps; @@ -31,47 +32,27 @@ import accord.primitives.Route; import accord.primitives.Timestamp; import accord.primitives.TxnId; import accord.primitives.Writes; -import accord.utils.Invariants; import accord.utils.VIntCoding; import org.apache.cassandra.db.TypeSizes; -import org.apache.cassandra.io.UnversionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.service.accord.serializers.CommandSerializers.ExecuteAtSerializer; import static accord.primitives.Txn.Kind.Write; - public class ApplySerializers { - private static final UnversionedSerializer kind = new UnversionedSerializer<>() - { - public void serialize(Apply.Kind kind, DataOutputPlus out) throws IOException - { - Invariants.requireArgument(kind == Apply.Kind.Maximal || kind == Apply.Kind.Minimal); - out.writeBoolean(kind == Apply.Kind.Maximal); - } - - public Apply.Kind deserialize(DataInputPlus in) throws IOException - { - return in.readBoolean() ? Apply.Kind.Maximal : Apply.Kind.Minimal; - } - - public long serializedSize(Apply.Kind t) - { - return TypeSizes.BOOL_SIZE; - } - }; - public abstract static class ApplySerializer extends TxnRequestSerializer { + private static final EncodeAsVInt32 kinds = EncodeAsVInt32.of(Apply.Kind.class); + @Override public void serializeBody(A apply, DataOutputPlus out, Version version) throws IOException { CommandSerializers.ballot.serialize(apply.ballot, out); out.writeVInt(apply.minEpoch - apply.waitForEpoch); out.writeUnsignedVInt(apply.maxEpoch - apply.minEpoch); - kind.serialize(apply.kind, out); + kinds.serialize(apply.kind, out); ExecuteAtSerializer.serialize(apply.txnId, apply.executeAt, out); DepsSerializers.partialDeps.serialize(apply.deps(), out); CommandSerializers.nullablePartialTxn.serialize(apply.txn(), out, version); @@ -91,7 +72,7 @@ public class ApplySerializers long minEpoch = waitForEpoch + in.readVInt(); long maxEpoch = minEpoch + in.readUnsignedVInt(); return deserializeApply(txnId, ballot, scope, minEpoch, waitForEpoch, maxEpoch, - kind.deserialize(in), + kinds.deserialize(in), ExecuteAtSerializer.deserialize(txnId, in), DepsSerializers.partialDeps.deserialize(in), CommandSerializers.nullablePartialTxn.deserialize(in, version), @@ -107,7 +88,7 @@ public class ApplySerializers return CommandSerializers.ballot.serializedSize(apply.ballot) + TypeSizes.sizeofVInt(apply.minEpoch - apply.waitForEpoch) + TypeSizes.sizeofUnsignedVInt(apply.maxEpoch - apply.minEpoch) - + kind.serializedSize(apply.kind) + + kinds.serializedSize(apply.kind) + ExecuteAtSerializer.serializedSize(apply.txnId, apply.executeAt) + DepsSerializers.partialDeps.serializedSize(apply.deps()) + CommandSerializers.nullablePartialTxn.serializedSize(apply.txn(), version) @@ -128,26 +109,37 @@ public class ApplySerializers } }; - public static final UnversionedSerializer reply = new UnversionedSerializer<>() + public static final IVersionedSerializer reply = new ReplySerializer(); + + public static final class ReplySerializer implements IVersionedSerializer { - private final Apply.ApplyReply[] replies = Apply.ApplyReply.values(); - + private static final EncodeAsVInt32 kinds = EncodeAsVInt32.of(ApplyReply.Kind.class); @Override - public void serialize(Apply.ApplyReply t, DataOutputPlus out) throws IOException + public void serialize(ApplyReply t, DataOutputPlus out, Version version) throws IOException { - out.writeByte(t.ordinal()); + kinds.serialize(t.kind, out); + if (t.kind == ApplyReply.Kind.InsufficientEpochs) + out.writeUnsignedVInt(t.minEpoch()); } @Override - public Apply.ApplyReply deserialize(DataInputPlus in) throws IOException + public ApplyReply deserialize(DataInputPlus in, Version version) throws IOException { - return replies[in.readByte()]; + ApplyReply.Kind kind = kinds.deserialize(in); + if (kind != ApplyReply.Kind.InsufficientEpochs) + return ApplyReply.lookupByKind(kind); + + long minEpoch = in.readUnsignedVInt(); + return new ApplyReply(kind, minEpoch); } @Override - public long serializedSize(Apply.ApplyReply t) + public long serializedSize(ApplyReply t, Version version) { - return 1; + long size = kinds.serializedSize(t.kind); + if (t.kind == ApplyReply.Kind.InsufficientEpochs) + size += VIntCoding.sizeOfUnsignedVInt(t.minEpoch()); + return size; } - }; + } } diff --git a/src/java/org/apache/cassandra/service/accord/serializers/AwaitSerializers.java b/src/java/org/apache/cassandra/service/accord/serializers/AwaitSerializers.java index d08d1bb822..af67c05b04 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/AwaitSerializers.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/AwaitSerializers.java @@ -20,7 +20,6 @@ package org.apache.cassandra.service.accord.serializers; import java.io.IOException; -import accord.api.ProgressLog.BlockedUntil; import accord.messages.Await; import accord.messages.Await.AsyncAwaitComplete; import accord.messages.Await.AwaitOk; @@ -42,19 +41,19 @@ public class AwaitSerializers public static final UnversionedSerializer request = new RequestSerializer<>() { @Override - public Await deserialize(TxnId txnId, Participants scope, BlockedUntil blockedUntil, boolean notifyProgressLog, long minAwaitEpoch, long maxAwaitEpoch, int callbackId, DataInputPlus in) + public Await deserialize(TxnId txnId, Participants scope, Await.Until awaitUntil, boolean notifyProgressLog, long minAwaitEpoch, long maxAwaitEpoch, int callbackId, DataInputPlus in) { - return Await.SerializerSupport.create(txnId, scope, blockedUntil, notifyProgressLog, minAwaitEpoch, maxAwaitEpoch, callbackId); + return Await.SerializerSupport.create(txnId, scope, awaitUntil, notifyProgressLog, minAwaitEpoch, maxAwaitEpoch, callbackId); } }; public static final UnversionedSerializer recoverRequest = new RequestSerializer<>() { @Override - public RecoverAwait deserialize(TxnId txnId, Participants scope, BlockedUntil blockedUntil, boolean notifyProgressLog, long minAwaitEpoch, long maxAwaitEpoch, int callbackId, DataInputPlus in) throws IOException + public RecoverAwait deserialize(TxnId txnId, Participants scope, Await.Until awaitUntil, boolean notifyProgressLog, long minAwaitEpoch, long maxAwaitEpoch, int callbackId, DataInputPlus in) throws IOException { TxnId recoverId = CommandSerializers.txnId.deserialize(in); - return RecoverAwait.SerializerSupport.create(txnId, scope, blockedUntil, notifyProgressLog, minAwaitEpoch, maxAwaitEpoch, callbackId, recoverId); + return RecoverAwait.SerializerSupport.create(txnId, scope, awaitUntil, notifyProgressLog, minAwaitEpoch, maxAwaitEpoch, callbackId, recoverId); } @Override @@ -73,14 +72,14 @@ public class AwaitSerializers static abstract class RequestSerializer implements UnversionedSerializer { - abstract A deserialize(TxnId txnId, Participants scope, BlockedUntil blockedUntil, boolean notifyProgressLog, long minAwaitEpoch, long maxAwaitEpoch, int callbackId, DataInputPlus in) throws IOException; + abstract A deserialize(TxnId txnId, Participants scope, Await.Until awaitUntil, boolean notifyProgressLog, long minAwaitEpoch, long maxAwaitEpoch, int callbackId, DataInputPlus in) throws IOException; @Override public void serialize(A await, DataOutputPlus out) throws IOException { CommandSerializers.txnId.serialize(await.txnId, out); KeySerializers.participants.serialize(await.scope, out); - out.writeByte((await.blockedUntil.ordinal() << 1) | (await.notifyProgressLog ? 1 : 0)); + out.writeByte((await.until.ordinal() << 1) | (await.notifyProgressLog ? 1 : 0)); out.writeUnsignedVInt(await.maxAwaitEpoch - await.txnId.epoch()); out.writeUnsignedVInt(await.maxAwaitEpoch - await.minAwaitEpoch); out.writeUnsignedVInt32(await.callbackId + 1); @@ -93,13 +92,13 @@ public class AwaitSerializers TxnId txnId = CommandSerializers.txnId.deserialize(in); Participants scope = KeySerializers.participants.deserialize(in); int blockedAndNotify = in.readByte(); - BlockedUntil blockedUntil = BlockedUntil.forOrdinal(blockedAndNotify >>> 1); + Await.Until awaitUntil = Await.Until.forOrdinal(blockedAndNotify >>> 1); boolean notifyProgressLog = (blockedAndNotify & 1) == 1; long maxAwaitEpoch = in.readUnsignedVInt() + txnId.epoch(); long minAwaitEpoch = maxAwaitEpoch - in.readUnsignedVInt(); int callbackId = in.readUnsignedVInt32() - 1; Invariants.require(callbackId >= -1); - return deserialize(txnId, scope, blockedUntil, notifyProgressLog, minAwaitEpoch, maxAwaitEpoch, callbackId, in); + return deserialize(txnId, scope, awaitUntil, notifyProgressLog, minAwaitEpoch, maxAwaitEpoch, callbackId, in); } @Override diff --git a/src/java/org/apache/cassandra/service/accord/serializers/DepsSerializers.java b/src/java/org/apache/cassandra/service/accord/serializers/DepsSerializers.java index f400d66d5e..c77ce05993 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/DepsSerializers.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/DepsSerializers.java @@ -72,12 +72,12 @@ public class DepsSerializers static final int KEYS_BY_TXNID = 0x1; static final int RANGES_BY_TXNID = 0x2; static final int FLAGS_SIZE = VIntCoding.sizeOfUnsignedVInt(KEYS_BY_TXNID | RANGES_BY_TXNID); - final boolean forceByTxnId; + final boolean byTxnId; protected final UnversionedSerializer tokenRange; - public AbstractDepsSerializer(boolean forceByTxnId, UnversionedSerializer tokenRange) + public AbstractDepsSerializer(boolean byTxnId, UnversionedSerializer tokenRange) { - this.forceByTxnId = forceByTxnId; + this.byTxnId = byTxnId; this.tokenRange = tokenRange; } @@ -86,8 +86,8 @@ public class DepsSerializers @Override public void serialize(D deps, DataOutputPlus out) throws IOException { - boolean keysByTxnId = forceByTxnId || deps.keyDeps.hasByTxnId(); - boolean rangesByTxnId = forceByTxnId || deps.rangeDeps.hasByTxnId(); + boolean keysByTxnId = byTxnId; + boolean rangesByTxnId = byTxnId; out.writeUnsignedVInt32((keysByTxnId ? KEYS_BY_TXNID : 0) | (rangesByTxnId ? RANGES_BY_TXNID : 0)); { KeyDeps keyDeps = deps.keyDeps; @@ -109,7 +109,7 @@ public class DepsSerializers { out.writeUnsignedVInt32(xtoy.length); - if ((xCount <= 1 || yCount <= 1) && (xtoy.length == xCount + yCount || xCount == 0 || yCount == 0)) + if ((xCount <= 1 || yCount <= 1) && (xtoy.length == (xCount == 1 ? 1 + yCount : 2 * xCount) || xCount == 0 || yCount == 0)) { // no point serializing as can be directly inferred if (Invariants.isParanoid()) @@ -172,7 +172,7 @@ public class DepsSerializers int length = in.readUnsignedVInt32(); int[] xtoy = new int[length]; - if ((xCount <= 1 || yCount <= 1) && (xtoy.length == xCount + yCount || xCount == 0 || yCount == 0)) + if ((xCount <= 1 || yCount <= 1) && (xtoy.length == (xCount == 1 ? 1 + yCount : 2 * xCount) || xCount == 0 || yCount == 0)) { // no point serializing as can be directly inferred if (xCount == 1) @@ -207,8 +207,8 @@ public class DepsSerializers @Override public long serializedSize(D deps) { - boolean keysByTxnId = forceByTxnId || deps.keyDeps.hasByTxnId(); - boolean rangesByTxnId = forceByTxnId || deps.rangeDeps.hasByTxnId(); + boolean keysByTxnId = byTxnId; + boolean rangesByTxnId = byTxnId; long size = FLAGS_SIZE; { KeyDeps keyDeps = deps.keyDeps; @@ -230,7 +230,7 @@ public class DepsSerializers private static long serializedPackedXtoYSize(int[] xtoy, int xCount, int yCount) { long size = VIntCoding.sizeOfUnsignedVInt(xtoy.length); - if ((xCount <= 1 || yCount <= 1) && (xtoy.length == xCount + yCount || xCount == 0 || yCount == 0)) + if ((xCount <= 1 || yCount <= 1) && (xtoy.length == (xCount == 1 ? 1 + yCount : 2 * xCount) || xCount == 0 || yCount == 0)) { // no point serializing as can be directly inferred } @@ -274,9 +274,9 @@ public class DepsSerializers static class DepsSerializer extends AbstractDepsSerializer { - public DepsSerializer(boolean preferByTxnId, UnversionedSerializer tokenRange) + public DepsSerializer(boolean byTxnId, UnversionedSerializer tokenRange) { - super(preferByTxnId, tokenRange); + super(byTxnId, tokenRange); } @Override @@ -292,7 +292,7 @@ public class DepsSerializers final UnversionedSerializer tokenRange; final DepsSerializer deps; final UnversionedSerializer nullableDeps; - final PartialDepsSerializer partialDeps; + final PartialDepsSerializer partialDeps; // currently happens to be byId but doesn't rely upon it final PartialDepsSerializer partialDepsById; final UnversionedSerializer nullablePartialDeps; @@ -301,10 +301,9 @@ public class DepsSerializers this.tokenRange = tokenRange; this.deps = new DepsSerializer(false, tokenRange); this.nullableDeps = NullableSerializer.wrap(deps); - this.partialDeps = new PartialDepsSerializer(false, tokenRange); - this.partialDepsById = new PartialDepsSerializer(true, tokenRange); + this.partialDeps = new PartialDepsSerializer(true, tokenRange); + this.partialDepsById = partialDeps; this.nullablePartialDeps = NullableSerializer.wrap(partialDeps); } } - } \ No newline at end of file diff --git a/src/java/org/apache/cassandra/service/accord/serializers/FetchSerializers.java b/src/java/org/apache/cassandra/service/accord/serializers/FetchSerializers.java index 7d45a6006d..e5da16bb07 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/FetchSerializers.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/FetchSerializers.java @@ -26,6 +26,7 @@ import accord.impl.AbstractFetchCoordinator.FetchResponse; import accord.messages.ReadData.CommitOrReadNack; import accord.messages.ReadData.ReadReply; import accord.primitives.Ranges; +import accord.utils.VIntCoding; import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.io.UnversionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; @@ -76,7 +77,6 @@ public class FetchSerializers public static final UnversionedSerializer reply = new UnversionedSerializer<>() { - final CommitOrReadNack[] nacks = CommitOrReadNack.values(); final UnversionedSerializer streamDataSerializer = CastingSerializer.create(StreamData.class, StreamData.serializer); @Override @@ -84,7 +84,10 @@ public class FetchSerializers { if (!reply.isOk()) { - out.writeByte(1 + ((CommitOrReadNack) reply).ordinal()); + CommitOrReadNack nack = (CommitOrReadNack) reply; + out.writeByte(1 + nack.kind.ordinal()); + if (nack.kind == CommitOrReadNack.Kind.InsufficientEpochs) + out.writeUnsignedVInt(nack.minEpoch()); return; } @@ -100,7 +103,13 @@ public class FetchSerializers { int id = in.readByte(); if (id != 0) - return nacks[id - 1]; + { + CommitOrReadNack.Kind kind = CommitOrReadNack.Kind.lookupByOrdinal(id - 1); + CommitOrReadNack nack = CommitOrReadNack.lookupByKind(kind); + if (nack != null) + return nack; + return new CommitOrReadNack(kind, in.readUnsignedVInt()); + } return new FetchResponse(deserializeNullable(in, KeySerializers.ranges), deserializeNullable(in, streamDataSerializer), @@ -111,7 +120,13 @@ public class FetchSerializers public long serializedSize(ReadReply reply) { if (!reply.isOk()) - return TypeSizes.BYTE_SIZE; + { + long size = 1; + CommitOrReadNack nack = (CommitOrReadNack) reply; + if (nack.kind == CommitOrReadNack.Kind.InsufficientEpochs) + size += VIntCoding.sizeOfUnsignedVInt(nack.minEpoch()); + return size; + } FetchResponse response = (FetchResponse) reply; return TypeSizes.BYTE_SIZE diff --git a/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializer.java b/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializer.java index 4311fb0efa..34ab52fc2d 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializer.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializer.java @@ -293,7 +293,6 @@ public class ReadDataSerializer implements IVersionedSerializer public static final class ReplySerializer implements IVersionedSerializer { - final CommitOrReadNack[] nacks = CommitOrReadNack.values(); private final VersionedSerializer dataSerializer; public ReplySerializer(VersionedSerializer dataSerializer) @@ -306,7 +305,10 @@ public class ReadDataSerializer implements IVersionedSerializer { if (!reply.isOk()) { - out.writeByte(3 + ((CommitOrReadNack) reply).ordinal()); + CommitOrReadNack nack = (CommitOrReadNack) reply; + out.writeByte(3 + nack.kind.ordinal()); + if (nack.kind == CommitOrReadNack.Kind.InsufficientEpochs) + out.writeUnsignedVInt(nack.minEpoch()); return; } @@ -327,7 +329,13 @@ public class ReadDataSerializer implements IVersionedSerializer { int flags = in.readByte(); if (flags > 2) - return nacks[flags - 3]; + { + CommitOrReadNack.Kind kind = CommitOrReadNack.Kind.lookupByOrdinal(flags - 3); + CommitOrReadNack nack = CommitOrReadNack.lookupByKind(kind); + if (nack != null) + return nack; + return new CommitOrReadNack(kind, in.readUnsignedVInt()); + } Ranges unavailable = deserializeNullable(in, KeySerializers.ranges); D data = dataSerializer.deserialize(in, version); @@ -342,7 +350,12 @@ public class ReadDataSerializer implements IVersionedSerializer public long serializedSize(ReadReply reply, Version version) { if (!reply.isOk()) + { + CommitOrReadNack nack = (CommitOrReadNack) reply; + if (nack.kind == CommitOrReadNack.Kind.InsufficientEpochs) + return TypeSizes.BYTE_SIZE + VIntCoding.computeUnsignedVIntSize(nack.minEpoch()); return TypeSizes.BYTE_SIZE; + } ReadOk readOk = (ReadOk) reply; long size = TypeSizes.BYTE_SIZE diff --git a/src/java/org/apache/cassandra/service/accord/serializers/TopologySerializers.java b/src/java/org/apache/cassandra/service/accord/serializers/TopologySerializers.java index 3076feca66..067c179ac4 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/TopologySerializers.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/TopologySerializers.java @@ -27,7 +27,7 @@ import accord.local.Node; import accord.primitives.Range; import accord.topology.Shard; import accord.topology.Topology; -import accord.utils.SimpleBitSet; +import accord.utils.LargeBitSet; import accord.utils.SortedArrays; import accord.utils.SortedArrays.SortedArrayList; import accord.utils.TinyEnumSet; @@ -41,7 +41,7 @@ import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.utils.ArraySerializers; import org.apache.cassandra.utils.CollectionSerializers; -import org.apache.cassandra.utils.SimpleBitSetSerializer; +import org.apache.cassandra.utils.LargeBitSetSerializer; import static accord.utils.SortedArrays.fromSimpleBitSet; @@ -220,10 +220,10 @@ public class TopologySerializers out.writeUnsignedVInt32(rangeIdx); CollectionSerializers.serializeList(shard.nodes, out, TopologySerializers.nodeId); - SimpleBitSet notInFastPath = SortedArrays.toSimpleBitSet(shard.nodes, shard.notInFastPath); - SimpleBitSetSerializer.instance.serialize(notInFastPath, out); - SimpleBitSet joining = SortedArrays.toSimpleBitSet(shard.nodes, shard.joining); - SimpleBitSetSerializer.instance.serialize(joining, out); + LargeBitSet notInFastPath = SortedArrays.toLargeBitSet(shard.nodes, shard.notInFastPath); + LargeBitSetSerializer.instance.serialize(notInFastPath, out); + LargeBitSet joining = SortedArrays.toLargeBitSet(shard.nodes, shard.joining); + LargeBitSetSerializer.instance.serialize(joining, out); out.writeUnsignedVInt32(shard.flags().bitset()); } } @@ -271,10 +271,10 @@ public class TopologySerializers size += TypeSizes.sizeofUnsignedVInt(rangeIdx); size += CollectionSerializers.serializedListSize(shard.nodes, TopologySerializers.nodeId); - SimpleBitSet notInFastPath = SortedArrays.toSimpleBitSet(shard.nodes, shard.notInFastPath); - size += SimpleBitSetSerializer.instance.serializedSize(notInFastPath); - SimpleBitSet joining = SortedArrays.toSimpleBitSet(shard.nodes, shard.joining); - size += SimpleBitSetSerializer.instance.serializedSize(joining); + LargeBitSet notInFastPath = SortedArrays.toLargeBitSet(shard.nodes, shard.notInFastPath); + size += LargeBitSetSerializer.instance.serializedSize(notInFastPath); + LargeBitSet joining = SortedArrays.toLargeBitSet(shard.nodes, shard.joining); + size += LargeBitSetSerializer.instance.serializedSize(joining); size += TypeSizes.sizeofUnsignedVInt(shard.flags().bitset()); } return size; @@ -301,8 +301,8 @@ public class TopologySerializers TokenRange range = ranges.get(rangeIndex).withTable(activeTableId); SortedArrays.SortedArrayList nodes = CollectionSerializers.deserializeSortedArrayList(in, TopologySerializers.nodeId, Node.Id[]::new); - SimpleBitSet notInFastPath = SimpleBitSetSerializer.instance.deserialize(in); - SimpleBitSet joining = SimpleBitSetSerializer.instance.deserialize(in); + LargeBitSet notInFastPath = LargeBitSetSerializer.instance.deserialize(in); + LargeBitSet joining = LargeBitSetSerializer.instance.deserialize(in); int flags = in.readUnsignedVInt32(); shards[i] = Shard.SerializerSupport.create(range, nodes, fromSimpleBitSet(nodes, notInFastPath, Node.Id[]::new), fromSimpleBitSet(nodes, joining, Node.Id[]::new), new TinyEnumSet<>(flags)); } diff --git a/src/java/org/apache/cassandra/service/accord/serializers/WaitingOnSerializer.java b/src/java/org/apache/cassandra/service/accord/serializers/WaitingOnSerializer.java index 5770125a0d..16a71e1bdd 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/WaitingOnSerializer.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/WaitingOnSerializer.java @@ -30,7 +30,7 @@ import accord.primitives.Timestamp; import accord.primitives.TxnId; import accord.utils.ImmutableBitSet; import accord.utils.Invariants; -import accord.utils.SimpleBitSet; +import accord.utils.LargeBitSet; import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; @@ -121,9 +121,9 @@ public class WaitingOnSerializer } } - private static void serialize(int length, SimpleBitSet write, DataOutputPlus out) throws IOException + private static void serialize(int length, LargeBitSet write, DataOutputPlus out) throws IOException { - long[] bits = SimpleBitSet.SerializationSupport.getArray(write); + long[] bits = LargeBitSet.SerializationSupport.getArray(write); Invariants.require(length == bits.length); for (int i = 0; i < length; i++) out.writeLong(bits[i]); @@ -137,9 +137,9 @@ public class WaitingOnSerializer return ImmutableBitSet.SerializationSupport.construct(bits); } - public static long serializedSize(int length, SimpleBitSet write) + public static long serializedSize(int length, LargeBitSet write) { - long[] bits = SimpleBitSet.SerializationSupport.getArray(write); + long[] bits = LargeBitSet.SerializationSupport.getArray(write); Invariants.require(length == bits.length, "Expected length %d != %d", length, bits.length); return (long) TypeSizes.LONG_SIZE * length; } diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java b/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java index c5890b1cd0..83a287b79c 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java @@ -42,7 +42,6 @@ import accord.primitives.Seekable; import accord.primitives.Seekables; import accord.primitives.Timestamp; import accord.primitives.TxnId; -import accord.primitives.Writes; import accord.utils.async.AsyncChain; import accord.utils.async.AsyncChains; import org.apache.cassandra.cql3.UpdateParameters; @@ -456,7 +455,7 @@ public class TxnWrite extends AbstractKeySorted implements Writ // Accord should skip the Update for a read transaction, but handle it here anyways TxnUpdate txnUpdate = ((TxnUpdate)txn.update()); if (txnUpdate == null) - return Writes.SUCCESS; + return AsyncChains.success(null); long timestamp = executeAt.uniqueHlc(); @@ -477,12 +476,12 @@ public class TxnWrite extends AbstractKeySorted implements Writ } if (results.isEmpty()) - return Writes.SUCCESS; + return AsyncChains.success(null); if (results.size() == 1) - return results.get(0).flatMap(o -> Writes.SUCCESS); + return results.get(0).map(o -> null); - return AsyncChains.reduce(results, (i1, i2) -> null, (Void)null).flatMap(ignore -> Writes.SUCCESS); + return AsyncChains.reduce(results, (i1, i2) -> null, (Void)null).map(ignore -> null); } public long estimatedSizeOnHeap() diff --git a/src/java/org/apache/cassandra/service/paxos/ContentionStrategy.java b/src/java/org/apache/cassandra/service/paxos/ContentionStrategy.java index 63dba970be..b5107016b6 100644 --- a/src/java/org/apache/cassandra/service/paxos/ContentionStrategy.java +++ b/src/java/org/apache/cassandra/service/paxos/ContentionStrategy.java @@ -333,6 +333,11 @@ public class ContentionStrategy extends RetryStrategy } } + public long getMaxMicros(int attempts) + { + return max(min, min(max, delegate.getMaxMicros(attempts))); + } + public String toString() { return "Bound{" + diff --git a/src/java/org/apache/cassandra/tcm/RemoteProcessor.java b/src/java/org/apache/cassandra/tcm/RemoteProcessor.java index 46de475f72..a8fe8193f0 100644 --- a/src/java/org/apache/cassandra/tcm/RemoteProcessor.java +++ b/src/java/org/apache/cassandra/tcm/RemoteProcessor.java @@ -133,12 +133,13 @@ public final class RemoteProcessor implements Processor if (waitFor == null) return fetchLogAndWait(new CandidateIterator(candidates(true), false), log); + Future cmFuture = null; try { Supplier> fetchFunction = () -> fetchLogAndWaitInternal(new CandidateIterator(candidates(true), false), log); - Future cmFuture = EpochAwareDebounce.instance.getAsync(fetchFunction, waitFor); + cmFuture = EpochAwareDebounce.instance.getAsync(fetchFunction, waitFor); return cmFuture.get(retryPolicy.remainingNanos(), TimeUnit.NANOSECONDS); } catch (InterruptedException e) diff --git a/src/java/org/apache/cassandra/transport/ClientResourceLimits.java b/src/java/org/apache/cassandra/transport/ClientResourceLimits.java index 4bc029c339..a6191d189c 100644 --- a/src/java/org/apache/cassandra/transport/ClientResourceLimits.java +++ b/src/java/org/apache/cassandra/transport/ClientResourceLimits.java @@ -24,11 +24,12 @@ import java.util.concurrent.ConcurrentMap; import java.util.concurrent.atomic.AtomicInteger; import com.google.common.annotations.VisibleForTesting; + +import org.apache.cassandra.metrics.CassandraReservoir; import org.apache.cassandra.utils.concurrent.NonBlockingRateLimiter; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import com.codahale.metrics.Reservoir; import com.codahale.metrics.Snapshot; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.metrics.DecayingEstimatedHistogramReservoir; @@ -117,24 +118,17 @@ public class ClientResourceLimits * This will recompute the ip usage histo on each query of the snapshot when requested instead of trying to keep * a histogram up to date with each request */ - public static Reservoir ipUsageReservoir() + public static CassandraReservoir ipUsageReservoir() { - return new Reservoir() + return new CassandraReservoir() { - public int size() - { - return PER_ENDPOINT_ALLOCATORS.size(); - } + @Override public Snapshot getPercentileSnapshot() { throw new UnsupportedOperationException(); } + @Override public long[] buckets(int length) { throw new UnsupportedOperationException(); } + @Override public BucketStrategy bucketStrategy() { return BucketStrategy.none; } + @Override public void update(long l) { throw new UnsupportedOperationException(); } - public void update(long l) - { - throw new IllegalStateException(); - } - - public Snapshot getSnapshot() - { - return getCurrentIpUsage(); - } + @Override public int size() { return PER_ENDPOINT_ALLOCATORS.size(); } + @Override public Snapshot getSnapshot() { return getCurrentIpUsage(); } }; } diff --git a/src/java/org/apache/cassandra/utils/EstimatedHistogram.java b/src/java/org/apache/cassandra/utils/EstimatedHistogram.java index 7cd8dc4a20..d366c855a7 100644 --- a/src/java/org/apache/cassandra/utils/EstimatedHistogram.java +++ b/src/java/org/apache/cassandra/utils/EstimatedHistogram.java @@ -88,6 +88,19 @@ public class EstimatedHistogram implements DoubleToLongFunction buckets = new AtomicLongArray(bucketData); } + public static long[] newOffsetsWithScale(long maxScale, boolean considerZeroes) + { + int i = 2; + long next = 1; + while (true) + { + next = Math.max(next + 1, Math.round(next * 1.2)); + if (next >= maxScale) + return newOffsets(i, considerZeroes); + ++i; + } + } + public static long[] newOffsets(int size, boolean considerZeroes) { long[] result = new long[size + (considerZeroes ? 1 : 0)]; @@ -256,6 +269,8 @@ public class EstimatedHistogram implements DoubleToLongFunction sum += bCount * bucketOffsets[i]; } + if (elements == 0) + return 0.0D; return (double) sum / elements; } diff --git a/src/java/org/apache/cassandra/utils/SimpleBitSetSerializer.java b/src/java/org/apache/cassandra/utils/LargeBitSetSerializer.java similarity index 78% rename from src/java/org/apache/cassandra/utils/SimpleBitSetSerializer.java rename to src/java/org/apache/cassandra/utils/LargeBitSetSerializer.java index c336f9565d..69a13948f7 100644 --- a/src/java/org/apache/cassandra/utils/SimpleBitSetSerializer.java +++ b/src/java/org/apache/cassandra/utils/LargeBitSetSerializer.java @@ -20,20 +20,20 @@ package org.apache.cassandra.utils; import java.io.IOException; -import accord.utils.SimpleBitSet; +import accord.utils.LargeBitSet; import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.io.UnversionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; -public class SimpleBitSetSerializer implements UnversionedSerializer +public class LargeBitSetSerializer implements UnversionedSerializer { - public static final SimpleBitSetSerializer instance = new SimpleBitSetSerializer(); + public static final LargeBitSetSerializer instance = new LargeBitSetSerializer(); @Override - public void serialize(SimpleBitSet t, DataOutputPlus out) throws IOException + public void serialize(LargeBitSet t, DataOutputPlus out) throws IOException { - long[] raw = SimpleBitSet.SerializationSupport.getArray(t); + long[] raw = LargeBitSet.SerializationSupport.getArray(t); // find the first word written int wordsInUse = wordsInUse(raw); out.writeUnsignedVInt32(raw.length); @@ -43,20 +43,20 @@ public class SimpleBitSetSerializer implements UnversionedSerializer { Assert.assertEquals(initialMax, ClusterMetadata.current().epoch.getEpoch()); - System.out.println("Awaiting " + initialMax); awaitEpoch(initialMax); AccordConfigurationService configService = service().configService(); long minEpoch = configService.minEpoch(); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordLoadTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordLoadTest.java index cec6ec9cb7..be260c352f 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordLoadTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordLoadTest.java @@ -117,7 +117,8 @@ public class AccordLoadTest extends AccordTestBase final long batchTime = TimeUnit.SECONDS.toNanos(10); final int concurrency = 100; final int ratePerSecond = 1000; - final int keyCount = 10_000; +// final int keyCount = 10_000; + final int keyCount = 10; final float readChance = 0.33f; long nextRepairAt = repairInterval; long nextCompactionAt = compactionInterval; diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMetricsTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMetricsTest.java index b7a39afe46..5c78547432 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMetricsTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMetricsTest.java @@ -43,7 +43,8 @@ import org.apache.cassandra.distributed.api.Row; import org.apache.cassandra.distributed.api.SimpleQueryResult; import org.apache.cassandra.exceptions.ReadTimeoutException; import org.apache.cassandra.exceptions.WriteTimeoutException; -import org.apache.cassandra.metrics.AccordMetrics; +import org.apache.cassandra.metrics.AccordCoordinatorMetrics; +import org.apache.cassandra.metrics.AccordReplicaMetrics; import org.apache.cassandra.metrics.DefaultNameFactory; import org.apache.cassandra.metrics.RatioGaugeSet; import org.apache.cassandra.net.Verb; @@ -80,7 +81,7 @@ public class AccordMetricsTest extends AccordTestBase ProtocolModifiers.Toggles.setPermittedFastPaths(new FastPaths(FastPath.Unoptimised)); })); for (int i = 0; i < SHARED_CLUSTER.size(); i++) // initialize metrics - logger.trace(SHARED_CLUSTER.get(i + 1).callOnInstance(() -> AccordMetrics.readMetrics.toString() + AccordMetrics.writeMetrics.toString())); + logger.trace(SHARED_CLUSTER.get(i + 1).callOnInstance(() -> AccordCoordinatorMetrics.readMetrics.toString() + AccordCoordinatorMetrics.writeMetrics.toString())); } String writeCql() @@ -173,9 +174,8 @@ public class AccordMetricsTest extends AccordTestBase assertClientMetrics(0, "AccordWrite", 1, 0); assertClientMetrics(1, "AccordWrite", 0, 0); - // TODO (required): reenable or remove metric -// assertCoordinatorMetrics(0, "rw", 0, 0, 1, 0, 0); - assertCoordinatorMetrics(1, "rw", 0, 0, 0, 0, 0); + assertCoordinatorMetrics(0, "rw", 0, 0, 1, 0, 0); + assertCoordinatorMetrics(1, "rw", 0, 0, 0, 0, 1); assertReplicaMetrics(0, "rw", 0, 0, 0); assertReplicaMetrics(1, "rw", 0, 0, 0); @@ -195,8 +195,8 @@ public class AccordMetricsTest extends AccordTestBase assertClientMetrics(0, "AccordRead", 1, 0); assertClientMetrics(1, "AccordRead", 0, 0); // TODO (required): reenable or remove metric -// assertCoordinatorMetrics(0, "ro", 0, 0, 1, 0, 0); - assertCoordinatorMetrics(1, "ro", 0, 0, 0, 0, 0); + assertCoordinatorMetrics(0, "ro", 0, 0, 1, 0, 0); + assertCoordinatorMetrics(1, "ro" , 0, 0, 0, 0, 1); assertReplicaMetrics(0, "ro", 0, 0, 0); assertReplicaMetrics(1, "ro", 0, 0, 0); @@ -231,7 +231,7 @@ public class AccordMetricsTest extends AccordTestBase assertClientMetrics(0, "AccordRead", 0, 1); assertClientMetrics(1, "AccordRead", 0, 0); - // TODO (required): reenable or remove metric + // TODO (required): rework tests: internal accord timeout triggers later than external C* client one // assertCoordinatorMetrics(0, "ro", 0, 0, 0, 1, 0); assertCoordinatorMetrics(1, "ro", 0, 0, 0, 0, 0); assertReplicaMetrics(0, "ro", 0, 0, 0); @@ -252,13 +252,15 @@ public class AccordMetricsTest extends AccordTestBase assertClientMetrics(0, "AccordWrite", 0, 1); assertClientMetrics(1, "AccordWrite", 0, 0); - // TODO (required): reenable or remove metric -// assertCoordinatorMetrics(0, "rw", 0, 0, 0, 1, 0); + assertCoordinatorMetrics(0, "rw", 0, 0, 0, 1, 0); assertCoordinatorMetrics(1, "rw", 0, 0, 0, 0, 0); assertReplicaMetrics(0, "rw", 0, 0, 0); assertReplicaMetrics(1, "rw", 0, 0, 0); - assertZeroMetrics("ro","AccordRead"); + assertCoordinatorMetrics(0, "ro", 0, 0, 0, 0, 1); + assertCoordinatorMetrics(1, "ro", 0, 0, 0, 0, 0); + assertReplicaMetrics(0, "ro", 0, 0, 0); + assertReplicaMetrics(1, "ro", 0, 0, 0); } private void assertZeroMetrics(String scope, String clientScope) @@ -293,17 +295,17 @@ public class AccordMetricsTest extends AccordTestBase private void assertCoordinatorMetrics(int node, String scope, long fastPaths, long slowPaths, long preempts, long timeouts, long recoveries) { - DefaultNameFactory nameFactory = new DefaultNameFactory(AccordMetrics.ACCORD_COORDINATOR, scope); + DefaultNameFactory nameFactory = new DefaultNameFactory(AccordCoordinatorMetrics.ACCORD_COORDINATOR, scope); Map metrics = diff(countingMetrics0).get(node); logger.info("Metrics for node {} / {}: {}", node, scope, metrics); Function metric = n -> metrics.get(nameFactory.createMetricName(n).getMetricName()); - assertThat(metric.apply(AccordMetrics.FAST_PATHS)).isEqualTo(fastPaths); - assertThat(metric.apply(AccordMetrics.SLOW_PATHS)).isEqualTo(slowPaths); - assertThat(metric.apply(AccordMetrics.PREEMPTED)).isEqualTo(preempts); - assertThat(metric.apply(AccordMetrics.TIMEOUTS)).isEqualTo(timeouts); - assertThat(metric.apply(AccordMetrics.RECOVERY_DELAY)).isEqualTo(recoveries); - assertThat(metric.apply(AccordMetrics.RECOVERY_TIME)).isEqualTo(recoveries); - assertThat(metric.apply(AccordMetrics.COORDINATOR_DEPENDENCIES)).isEqualTo(fastPaths + slowPaths); + assertThat(metric.apply(AccordCoordinatorMetrics.FAST_PATHS)).isEqualTo(fastPaths); + assertThat(metric.apply(AccordCoordinatorMetrics.SLOW_PATHS)).isEqualTo(slowPaths); + assertThat(metric.apply(AccordCoordinatorMetrics.PREEMPTED)).isEqualTo(preempts); + assertThat(metric.apply(AccordCoordinatorMetrics.TIMEOUTS)).isEqualTo(timeouts); + assertThat(metric.apply(AccordCoordinatorMetrics.RECOVERY_DELAY)).isEqualTo(recoveries); + assertThat(metric.apply(AccordCoordinatorMetrics.RECOVERY_TIME)).isEqualTo(recoveries); + assertThat(metric.apply(AccordCoordinatorMetrics.COORDINATOR_DEPENDENCIES)).isEqualTo(fastPaths + slowPaths); // Verify that coordinator metrics are published to the appropriate virtual table: SimpleQueryResult res = SHARED_CLUSTER.get(node + 1) @@ -317,21 +319,21 @@ public class AccordMetricsTest extends AccordTestBase if ((fastPaths + slowPaths) > 0) { - String fastPathToTotalName = nameFactory.createMetricName(AccordMetrics.FAST_PATH_TO_TOTAL + "." + RatioGaugeSet.MEAN_RATIO).getMetricName(); + String fastPathToTotalName = nameFactory.createMetricName(AccordCoordinatorMetrics.FAST_PATH_TO_TOTAL + "." + RatioGaugeSet.MEAN_RATIO).getMetricName(); assertThat((double) SHARED_CLUSTER.get(1).metrics().getGauge(fastPathToTotalName)).isEqualTo((double) fastPaths / (double) (fastPaths + slowPaths), Offset.offset(0.01d)); } } private void assertReplicaMetrics(int node, String scope, long stable, long executions, long applications) { - DefaultNameFactory nameFactory = new DefaultNameFactory(AccordMetrics.ACCORD_REPLICA, scope); + DefaultNameFactory nameFactory = new DefaultNameFactory(AccordReplicaMetrics.ACCORD_REPLICA, scope); Map metrics = diff(countingMetrics0).get(node); Function metric = n -> metrics.get(nameFactory.createMetricName(n).getMetricName()); - assertThat(metric.apply(AccordMetrics.REPLICA_STABLE_LATENCY)).isLessThanOrEqualTo(stable); - assertThat(metric.apply(AccordMetrics.REPLICA_PREAPPLY_LATENCY)).isEqualTo(executions); - assertThat(metric.apply(AccordMetrics.REPLICA_APPLY_LATENCY)).isEqualTo(applications); - assertThat(metric.apply(AccordMetrics.REPLICA_APPLY_DURATION)).isEqualTo(scope.equals("rw") ? applications : 0); - assertThat(metric.apply(AccordMetrics.REPLICA_DEPENDENCIES)).isEqualTo(executions); + assertThat(metric.apply(AccordReplicaMetrics.REPLICA_STABLE_LATENCY)).isLessThanOrEqualTo(stable); + assertThat(metric.apply(AccordReplicaMetrics.REPLICA_PREAPPLY_LATENCY)).isEqualTo(executions); + assertThat(metric.apply(AccordReplicaMetrics.REPLICA_APPLY_LATENCY)).isEqualTo(applications); + assertThat(metric.apply(AccordReplicaMetrics.REPLICA_APPLY_DURATION)).isEqualTo(scope.equals("rw") ? applications : 0); + assertThat(metric.apply(AccordReplicaMetrics.REPLICA_DEPENDENCIES)).isEqualTo(executions); // Verify that replica metrics are published to the appropriate virtual table: SimpleQueryResult vtableResults = SHARED_CLUSTER.get(node + 1) @@ -356,8 +358,8 @@ public class AccordMetricsTest extends AccordTestBase for (int i = 0; i < SHARED_CLUSTER.size(); i++) { Map map = SHARED_CLUSTER.get(i + 1).metrics().getCounters(name -> name.startsWith("org.apache.cassandra.metrics.Accord") || (name.startsWith("org.apache.cassandra.metrics.ClientRequest") && (name.endsWith("AccordRead") || name.endsWith("AccordWrite")))); - SHARED_CLUSTER.get(i + 1).metrics().getGauges(name -> name.startsWith("org.apache.cassandra.metrics.AccordReplica")) - .forEach((key, value) -> map.put(key, (Long)value)); + SHARED_CLUSTER.get(i + 1).metrics().getGauges(name -> name.startsWith("org.apache.cassandra.metrics.Accord")) + .forEach((key, value) -> map.put(key, ((Number)value).longValue())); metrics.put(i, map); } return metrics; diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java index 5cb835c205..d6a1fe3adf 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java @@ -36,6 +36,7 @@ import java.util.concurrent.ExecutionException; import java.util.concurrent.Future; import java.util.concurrent.TimeUnit; import java.util.function.Consumer; +import java.util.stream.Collectors; import com.google.common.base.Stopwatch; import com.google.common.collect.ImmutableList; @@ -47,7 +48,10 @@ import org.junit.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import accord.coordinate.Coordination; +import accord.coordinate.Coordination.CoordinationKind; import accord.primitives.Ranges; +import accord.utils.TinyEnumSet; import org.apache.cassandra.ServerTestUtils; import org.apache.cassandra.Util; import org.apache.cassandra.config.CassandraRelevantProperties; @@ -493,8 +497,11 @@ public abstract class AccordMigrationReadRaceTestBase extends AccordTestBase // Accord will block until we unpause enactment so to test the routing we wait until the transaction // has started so the epoch it is created in is the old one Util.spinUntilTrue(() -> outOfSyncInstance.callOnInstance(() -> { - logger.info("Coordinating {}", AccordService.instance().node().coordinating()); - return AccordService.instance().node().coordinating().size() == expectedTransactions; + logger.info("Fetching coordinations..."); + TinyEnumSet txnKinds = TinyEnumSet.of(CoordinationKind.PreAccept, CoordinationKind.Propose, CoordinationKind.Stabilise, CoordinationKind.Execute, CoordinationKind.Persist, CoordinationKind.BeginRecovery); + List coordinations = AccordService.instance().node().coordinations().stream().filter(c -> txnKinds.test(c.kind())).collect(Collectors.toList()); + logger.info("Coordinating {}", coordinations); + return coordinations.size() == expectedTransactions; })); logger.info("Accord node is now coordinating something, unpausing so it can continue to execute"); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java index b286e71de1..a5038ceec3 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java @@ -21,7 +21,6 @@ package org.apache.cassandra.distributed.test.accord; import java.io.IOException; import java.net.InetSocketAddress; import java.util.List; -import java.util.Map; import java.util.Queue; import java.util.Set; import java.util.concurrent.Callable; @@ -30,6 +29,7 @@ import java.util.concurrent.ExecutionException; import java.util.concurrent.Future; import java.util.concurrent.TimeUnit; import java.util.function.Consumer; +import java.util.stream.Collectors; import com.google.common.base.Stopwatch; import com.google.common.collect.ImmutableList; @@ -43,14 +43,14 @@ import org.slf4j.Logger; import org.slf4j.LoggerFactory; import accord.api.RoutingKey; -import accord.coordinate.Outcome; +import accord.coordinate.Coordination; +import accord.coordinate.Coordination.CoordinationKind; import accord.messages.PreAccept; import accord.primitives.KeyRoute; import accord.primitives.Ranges; import accord.primitives.Routable.Domain; import accord.primitives.Route; -import accord.primitives.TxnId; -import accord.utils.async.AsyncResult; +import accord.utils.TinyEnumSet; import org.apache.cassandra.ServerTestUtils; import org.apache.cassandra.Util; import org.apache.cassandra.batchlog.BatchlogManager; @@ -104,6 +104,7 @@ import org.apache.cassandra.utils.FBUtilities; import org.apache.cassandra.utils.Pair; import org.apache.cassandra.utils.concurrent.AsyncPromise; import org.apache.cassandra.utils.concurrent.Promise; +import org.assertj.core.api.Assertions; import org.eclipse.jetty.util.ConcurrentHashSet; import static java.lang.String.format; @@ -564,10 +565,11 @@ public abstract class AccordMigrationWriteRaceTestBase extends AccordTestBase // otherwise it will just be routed straight to non-Accord. logger.info("Spinning waiting on a transaction"); Util.spinUntilTrue(() -> { - Map> txns = AccordService.instance().node().coordinating(); - if (!txns.isEmpty()) + TinyEnumSet txnKinds = TinyEnumSet.of(CoordinationKind.PreAccept, CoordinationKind.Propose, CoordinationKind.Stabilise, CoordinationKind.Execute, CoordinationKind.BeginRecovery); + List coordinations = AccordService.instance().node().coordinations().stream().filter(c -> txnKinds.test(c.kind())).collect(Collectors.toList()); + if (!coordinations.isEmpty()) { - logger.info("Found txns {}", txns); + logger.info("Found txns {}", coordinations); return true; } return false; @@ -605,20 +607,22 @@ public abstract class AccordMigrationWriteRaceTestBase extends AccordTestBase // Accord will block until we unpause enactment so to test the routing we wait until the transaction // has started so the epoch it is created in is the old one Util.spinUntilTrue(() -> outOfSyncInstance.callOnInstance(() -> { - Map> coordinating = AccordService.instance().node().coordinating(); - if (!coordinating.isEmpty()) - logger.info("Accord coordinating: " + coordinating); - return !coordinating.isEmpty(); + TinyEnumSet txnKinds = TinyEnumSet.of(CoordinationKind.PreAccept, CoordinationKind.Propose, CoordinationKind.Stabilise, CoordinationKind.Execute, CoordinationKind.BeginRecovery); + List coordinations = AccordService.instance().node().coordinations().stream().filter(c -> txnKinds.test(c.kind())).collect(Collectors.toList()); + if (!coordinations.isEmpty()) + logger.info("Accord coordinating: " + coordinations); + return !coordinations.isEmpty(); }), 20); + boolean failed = false; try { validation.accept(cluster); - throw new AssertionError("Expected validation to fail"); } catch (AssertionError e) { - //ignored + failed = true; } + Assertions.assertThat(failed).isTrue(); } else { diff --git a/test/distributed/org/apache/cassandra/distributed/test/jmx/JMXTestsUtil.java b/test/distributed/org/apache/cassandra/distributed/test/jmx/JMXTestsUtil.java index ce6c301b39..9c7514353c 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/jmx/JMXTestsUtil.java +++ b/test/distributed/org/apache/cassandra/distributed/test/jmx/JMXTestsUtil.java @@ -66,7 +66,8 @@ public class JMXTestsUtil "org.apache.cassandra.db:type=StorageService:abortMove", // throws since there is no move in progress "org.apache.cassandra.db:type=CIDRGroupsMappingManager:loadCidrGroupsCache", // AllowAllCIDRAuthorizer doesn't support this operation, as feature is disabled by default "org.apache.cassandra.db:type=StorageService:forceRemoveCompletion", // deprecated (TCM) - "org.apache.cassandra.db:type=StorageService:createEpochUnsafe" // for Accord testing, but will likely be removed + "org.apache.cassandra.db:type=StorageService:createEpochUnsafe", // for Accord testing, but will likely be removed + "org.apache.cassandra.metrics:type=Client,name=RequestsSizeByIpDistribution:bucketsId" // for Accord testing, but will likely be removed ); // This set of mbeans are registered early enough during the startup of a // Cassandra instance for in-jvm dtests to avoid missing registration of mbeans. diff --git a/test/distributed/org/apache/cassandra/service/accord/AccordJournalBurnTest.java b/test/distributed/org/apache/cassandra/service/accord/AccordJournalBurnTest.java index bd20c03cb3..25b7b952c8 100644 --- a/test/distributed/org/apache/cassandra/service/accord/AccordJournalBurnTest.java +++ b/test/distributed/org/apache/cassandra/service/accord/AccordJournalBurnTest.java @@ -180,7 +180,7 @@ public class AccordJournalBurnTest extends BurnTestBase 5 + random.nextInt(15), operations, 10 + random.nextInt(30), - new RandomDelayQueue.Factory(random).get(), + RandomDelayQueue::new, (nodeId, randomSource) -> { try { diff --git a/test/unit/org/apache/cassandra/metrics/HitRateTest.java b/test/unit/org/apache/cassandra/metrics/HitRateTest.java new file mode 100644 index 0000000000..519aaa4f5d --- /dev/null +++ b/test/unit/org/apache/cassandra/metrics/HitRateTest.java @@ -0,0 +1,60 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.metrics; + +import org.junit.Assert; +import org.junit.Test; + +import org.agrona.concurrent.NoOpLock; + +public class HitRateTest +{ + static long M = 1024; + @Test + public void testRollingCounters() + { + ShardedHitRate.HitRateShard shard = new ShardedHitRate.HitRateShard(new NoOpLock()); + ShardedHitRate.Snapshot snapshot = new ShardedHitRate.Snapshot(); + + long now = M/2; + shard.markHitExclusive(now); + shard.markHitExclusive(now); + shard.markHitExclusive(now += M); + shard.markHitExclusive(now); + shard.markHitExclusive(now += M); + shard.markHitExclusive(now); + shard.markHitExclusive(now += M); + shard.markHitExclusive(now); + shard.markHitExclusive(now += M); + shard.markHitExclusive(now); + shard.updateSnapshot(snapshot, now); + Assert.assertEquals(10, snapshot.totalHits); + Assert.assertEquals(3, snapshot.requests(1)); + Assert.assertEquals(9, snapshot.requests(4)); + Assert.assertEquals(10, snapshot.requests(5)); + Assert.assertEquals(10, snapshot.requests(6)); + snapshot.reset(); + shard.updateSnapshot(snapshot, now + M/2); + Assert.assertEquals(10, snapshot.totalHits); + Assert.assertEquals(2, snapshot.requests(1)); + Assert.assertEquals(8, snapshot.requests(4)); + Assert.assertEquals(10, snapshot.requests(5)); + Assert.assertEquals(10, snapshot.requests(6)); + } +} diff --git a/test/unit/org/apache/cassandra/metrics/JmxVirtualTableMetricsTest.java b/test/unit/org/apache/cassandra/metrics/JmxVirtualTableMetricsTest.java index 6cf433e433..cd63f03a91 100644 --- a/test/unit/org/apache/cassandra/metrics/JmxVirtualTableMetricsTest.java +++ b/test/unit/org/apache/cassandra/metrics/JmxVirtualTableMetricsTest.java @@ -67,7 +67,7 @@ public class JmxVirtualTableMetricsTest extends CQLTester metricToNameMap.put(MetricType.METER, registry.meter("meter")); metricToNameMap.put(MetricType.COUNTER, registry.counter("counter")); - metricToNameMap.put(MetricType.HISTOGRAM, registry.histogram("histogram")); + metricToNameMap.put(MetricType.HISTOGRAM, registry.histogram("histogram", () -> new CassandraHistogram(new DecayingEstimatedHistogramReservoir(true)))); metricToNameMap.put(MetricType.TIMER, registry.timer("timer")); metricToNameMap.put(MetricType.GAUGE, registry.gauge("gauge", () -> gaugeValue::get)); diff --git a/test/unit/org/apache/cassandra/service/accord/AccordCacheTest.java b/test/unit/org/apache/cassandra/service/accord/AccordCacheTest.java index c8805ccc25..c23c0c0601 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordCacheTest.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordCacheTest.java @@ -17,18 +17,16 @@ */ package org.apache.cassandra.service.accord; -import java.util.UUID; import java.util.function.Function; import org.junit.Assert; -import org.junit.Before; import org.junit.Test; +import org.agrona.concurrent.NoOpLock; import org.apache.cassandra.cache.CacheSize; import org.apache.cassandra.concurrent.ExecutorPlus; import org.apache.cassandra.concurrent.ManualExecutor; import org.apache.cassandra.metrics.AccordCacheMetrics; -import org.apache.cassandra.metrics.CacheAccessMetrics; import org.apache.cassandra.service.accord.AccordCacheEntry.SaveExecutor; import org.apache.cassandra.service.accord.AccordCacheEntry.Status; @@ -41,7 +39,6 @@ import static org.mockito.Mockito.when; public class AccordCacheTest { private static final long DEFAULT_NODE_SIZE = nodeSize(0); - private AccordCacheMetrics cacheMetrics; private static abstract class TestSafeState implements AccordSafeState { @@ -134,6 +131,12 @@ public class AccordCacheTest return itemSize + emptyNodeSize(); } + private static int nextMetricId; + private static String nextMetricId() + { + return Integer.toString(++nextMetricId); + } + private static void assertCacheState(AccordCache cache, int referenced, int total, long bytes) { Assert.assertEquals(referenced, cache.numReferencedEntries()); @@ -141,33 +144,25 @@ public class AccordCacheTest Assert.assertEquals(bytes, cache.weightedSize()); } - private void assertCacheMetrics(CacheAccessMetrics metrics, int hits, int misses, int requests, int sizes) + private void assertCacheMetrics(AccordCacheMetrics metrics, int hits, int misses, int requests, int sizes) { - Assert.assertEquals(hits, metrics.hits.getCount()); - Assert.assertEquals(misses, metrics.misses.getCount()); - Assert.assertEquals(requests, metrics.requests.getCount()); - if (metrics instanceof AccordCacheMetrics) - { - AccordCacheMetrics ascMetrics = (AccordCacheMetrics) metrics; - Assert.assertEquals(sizes, ascMetrics.objectSize.getCount()); - assertThat(ascMetrics.objectSize.getSnapshot().getMax()).isGreaterThanOrEqualTo(DEFAULT_NODE_SIZE); - } + metrics.hitRate.refresh(); + Assert.assertEquals(hits, metrics.hits.getValue().intValue()); + Assert.assertEquals(misses, metrics.misses.getValue().intValue()); + Assert.assertEquals(requests, metrics.requests.getValue().intValue()); + Assert.assertEquals(sizes, metrics.objectSize.getCount()); } - @Before - public void before() - { - String type = String.format("%s-%s", AccordCommandStores.ACCORD_STATE_CACHE, UUID.randomUUID()); - cacheMetrics = new AccordCacheMetrics(type); - } - @Test public void testAcquisitionAndRelease() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), 500, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), 500); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance instance = type.newInstance(null); assertCacheState(cache, 0, 0, 0); @@ -190,20 +185,23 @@ public class AccordCacheTest Assert.assertSame(safeString1.global, cache.head()); Assert.assertSame(safeString2.global, cache.tail()); - assertCacheMetrics(cache.metrics, 0, 2, 2, 2); - assertCacheMetrics(type.typeMetrics, 0, 2, 2, 2); + assertCacheMetrics(cacheMetrics, 0, 2, 2, 2); } @Test public void testCachingMetricsWithTwoInstances() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + AccordCacheMetrics.Shard shard2 = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), 500, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), 500); AccordCache.Type stringType = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance stringInstance = stringType.newInstance(null); AccordCache.Type intType = - cache.newType(Integer.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, ignore -> Integer.BYTES, SafeInt::new); + cache.newType(Integer.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, ignore -> Integer.BYTES, SafeInt::new, shard2); assertCacheState(cache, 0, 0, 0); AccordCache.Type.Instance intInstance = intType.newInstance(null); @@ -239,10 +237,13 @@ public class AccordCacheTest @Test public void testRotation() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), DEFAULT_NODE_SIZE * 5, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), DEFAULT_NODE_SIZE * 5); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); assertCacheState(cache, 0, 0, 0); AccordCache.Type.Instance instance = type.newInstance(null); @@ -260,15 +261,13 @@ public class AccordCacheTest Assert.assertSame(items[0].global, cache.head()); Assert.assertSame(items[2].global, cache.tail()); assertCacheState(cache, 0, 3, nodeSize(1) * 3); - assertCacheMetrics(cache.metrics, 0, 3, 3, 3); - assertCacheMetrics(type.typeMetrics, 0, 3, 3, 3); + assertCacheMetrics(cacheMetrics, 0, 3, 3, 3); SafeString safeString = instance.acquire("1"); Assert.assertEquals(Status.LOADED, safeString.global.status()); assertCacheState(cache, 1, 3, nodeSize(1) * 3); - assertCacheMetrics(cache.metrics, 1, 3, 4, 3); - assertCacheMetrics(type.typeMetrics, 1, 3, 4, 3); + assertCacheMetrics(cacheMetrics, 1, 3, 4, 3); // releasing item should return it to the tail instance.release(safeString, null); @@ -280,10 +279,13 @@ public class AccordCacheTest @Test public void testEvictionOnAcquire() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), nodeSize(1) * 5, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), nodeSize(1) * 5); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance instance = type.newInstance(null); assertCacheState(cache, 0, 0, 0); @@ -300,8 +302,7 @@ public class AccordCacheTest assertCacheState(cache, 0, 5, nodeSize(1) * 5); Assert.assertSame(items[0].global, cache.head()); Assert.assertSame(items[4].global, cache.tail()); - assertCacheMetrics(cache.metrics, 0, 5, 5, 5); - assertCacheMetrics(type.typeMetrics, 0, 5, 5, 5); + assertCacheMetrics(cacheMetrics, 0, 5, 5, 5); SafeString safeString = instance.acquire("5"); Assert.assertTrue(instance.isReferenced(safeString.key())); @@ -312,25 +313,26 @@ public class AccordCacheTest Assert.assertSame(items[4].global, cache.tail()); Assert.assertFalse(instance.keyIsCached("0", SafeString.class)); Assert.assertFalse(instance.keyIsReferenced("0", SafeString.class)); - assertCacheMetrics(cache.metrics, 0, 6, 6, 5); - assertCacheMetrics(type.typeMetrics, 0, 6, 6, 5); + assertCacheMetrics(cacheMetrics, 0, 6, 6, 5); testLoad(executor, safeString, "5"); instance.release(safeString, null); assertCacheState(cache, 0, 5, nodeSize(1) * 5); Assert.assertSame(items[1].global, cache.head()); Assert.assertSame(safeString.global, cache.tail()); - assertCacheMetrics(cache.metrics, 0, 6, 6, 6); - assertCacheMetrics(type.typeMetrics, 0, 6, 6, 6); + assertCacheMetrics(cacheMetrics, 0, 6, 6, 5); } @Test public void testEvictionOnRelease() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), nodeSize(1) * 4, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), nodeSize(1) * 4); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance instance = type.newInstance(null); assertCacheState(cache, 0, 0, 0); @@ -344,22 +346,19 @@ public class AccordCacheTest } assertCacheState(cache, 5, 5, nodeSize(1) * 5); - assertCacheMetrics(cache.metrics, 0, 5, 5, 5); - assertCacheMetrics(type.typeMetrics, 0, 5, 5, 5); + assertCacheMetrics(cacheMetrics, 0, 5, 5, 5); Assert.assertNull(cache.head()); Assert.assertNull(cache.tail()); instance.release(items[2], null); assertCacheState(cache, 4, 4, nodeSize(1) * 4); - assertCacheMetrics(cache.metrics, 0, 5, 5, 5); - assertCacheMetrics(type.typeMetrics, 0, 5, 5, 5); + assertCacheMetrics(cacheMetrics, 0, 5, 5, 5); Assert.assertNull(cache.head()); Assert.assertNull(cache.tail()); instance.release(items[4], null); assertCacheState(cache, 3, 4, nodeSize(1) * 4); - assertCacheMetrics(cache.metrics, 0, 5, 5, 5); - assertCacheMetrics(type.typeMetrics, 0, 5, 5, 5); + assertCacheMetrics(cacheMetrics, 0, 5, 5, 5); Assert.assertSame(items[4].global, cache.head()); Assert.assertSame(items[4].global, cache.tail()); } @@ -367,18 +366,20 @@ public class AccordCacheTest @Test public void testMultiAcquireRelease() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), DEFAULT_NODE_SIZE * 4, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), DEFAULT_NODE_SIZE * 4); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance instance = type.newInstance(null); assertCacheState(cache, 0, 0, 0); SafeString safeString1 = instance.acquire("0"); testLoad(executor, safeString1, "0"); Assert.assertEquals(Status.LOADED, safeString1.global.status()); - assertCacheMetrics(cache.metrics, 0, 1, 1, 1); - assertCacheMetrics(type.typeMetrics, 0, 1, 1, 1); + assertCacheMetrics(cacheMetrics, 0, 1, 1, 1); Assert.assertEquals(1, instance.references("0", SafeString.class)); assertCacheState(cache, 1, 1, nodeSize(1)); @@ -388,8 +389,7 @@ public class AccordCacheTest Assert.assertEquals(Status.LOADED, safeString1.global.status()); Assert.assertEquals(2, instance.references("0", SafeString.class)); assertCacheState(cache, 1, 1, nodeSize(1)); - assertCacheMetrics(cache.metrics, 1, 1, 2, 1); - assertCacheMetrics(type.typeMetrics, 1, 1, 2, 1); + assertCacheMetrics(cacheMetrics, 1, 1, 2, 1); instance.release(safeString1, null); assertCacheState(cache, 1, 1, nodeSize(1)); @@ -400,10 +400,13 @@ public class AccordCacheTest @Test public void evictionBlockedOnSaving() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), nodeSize(1) * 3 + nodeSize(3), cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), nodeSize(1) * 3 + nodeSize(3)); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance instance = type.newInstance(null); assertCacheState(cache, 0, 0, 0); @@ -422,8 +425,7 @@ public class AccordCacheTest } assertCacheState(cache, 0, 4, nodeSize(1) * 3 + nodeSize(2)); - assertCacheMetrics(cache.metrics, 0, 4, 4, 5); - assertCacheMetrics(type.typeMetrics, 0, 4, 4, 5); + assertCacheMetrics(cacheMetrics, 0, 4, 4, 4); // force cache eviction instance.acquire(Integer.toString(0)); @@ -441,10 +443,13 @@ public class AccordCacheTest @Test public void testUpdates() { + AccordCacheMetrics cacheMetrics = new AccordCacheMetrics(nextMetricId()); + AccordCacheMetrics.Shard shard = cacheMetrics.newShard(new NoOpLock()); + ManualExecutor executor = new ManualExecutor(); - AccordCache cache = new AccordCache(saveExecutor(executor), 500, cacheMetrics); + AccordCache cache = new AccordCache(saveExecutor(executor), 500); AccordCache.Type type = - cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new); + cache.newType(String.class, (s, k) -> k, (s, k, c, o) -> null, Function.identity(), (s, k, v) -> true, String::length, SafeString::new, shard); AccordCache.Type.Instance instance = type.newInstance(null); assertCacheState(cache, 0, 0, 0); @@ -463,8 +468,7 @@ public class AccordCacheTest Assert.assertSame(safeString.global, cache.head()); Assert.assertSame(safeString.global, cache.tail()); - assertCacheMetrics(cache.metrics, 0, 1, 1, 2); - assertCacheMetrics(type.typeMetrics, 0, 1, 1, 2); + assertCacheMetrics(cacheMetrics, 0, 1, 1, 1); } private CacheSize mockCacheSize(long capacity, long size, int entries) @@ -476,21 +480,6 @@ public class AccordCacheTest return cacheSize; } - @Test - public void testAccorStateCacheMetrics() - { - CacheAccessMetrics stringInstance1 = cacheMetrics.forInstance(String.class); - CacheAccessMetrics stringInstance1Dup = cacheMetrics.forInstance(String.class); - CacheAccessMetrics stringInstance2 = cacheMetrics.forInstance(String.class); - CacheAccessMetrics integerInstance1 = cacheMetrics.forInstance(Integer.class); - CacheAccessMetrics integerInstance2 = cacheMetrics.forInstance(Integer.class); - - assertThat(stringInstance1).isSameAs(stringInstance1Dup); - assertThat(stringInstance1).isSameAs(stringInstance2); - assertThat(integerInstance1).isSameAs(integerInstance2); - assertThat(stringInstance1).isNotSameAs(integerInstance1); - } - private static SaveExecutor saveExecutor(ExecutorPlus executor) { return (saving, identity, save) -> { diff --git a/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java b/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java index 5532a21b23..3fe68ae195 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java @@ -49,7 +49,7 @@ import accord.primitives.Txn; import accord.primitives.TxnId; import accord.primitives.Writes; import accord.utils.ImmutableBitSet; -import accord.utils.SimpleBitSet; +import accord.utils.LargeBitSet; import accord.utils.async.AsyncChains; import org.apache.cassandra.SchemaLoader; import org.apache.cassandra.cql3.QueryProcessor; @@ -130,7 +130,7 @@ public class AccordCommandStoreTest Ballot promised = ballot(1, clock.incrementAndGet(), 1); Ballot accepted = ballot(1, clock.incrementAndGet(), 1); Timestamp executeAt = timestamp(1, clock.incrementAndGet(), 1); - SimpleBitSet waitingOnApply = new SimpleBitSet(3); + LargeBitSet waitingOnApply = new LargeBitSet(3); waitingOnApply.set(1); Command.WaitingOn waitingOn = new Command.WaitingOn(dependencies.keyDeps.keys(), dependencies.rangeDeps, new ImmutableBitSet(waitingOnApply), new ImmutableBitSet(2)); Pair result = AsyncChains.getBlocking(AccordTestUtils.processTxnResult(commandStore, txnId, txn, executeAt)); diff --git a/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java b/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java index 0cfab8b3d6..e96cf8a6c5 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java @@ -94,7 +94,6 @@ import org.apache.cassandra.dht.Murmur3Partitioner; import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; import org.apache.cassandra.io.util.File; -import org.apache.cassandra.metrics.AccordCacheMetrics; import org.apache.cassandra.schema.Schema; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.schema.TableMetadata; @@ -348,7 +347,7 @@ public class AccordTestUtils public static AccordCommandStore createAccordCommandStore( Node.Id node, LongSupplier now, Topology topology) { - AccordExecutor executor = new AccordExecutorSyncSubmit(0, RUN_WITH_LOCK, CommandStore.class.getSimpleName() + '[' + 0 + ']', new AccordCacheMetrics("test"), new AccordAgent()); + AccordExecutor executor = new AccordExecutorSyncSubmit(0, RUN_WITH_LOCK, CommandStore.class.getSimpleName() + '[' + 0 + ']', new AccordAgent()); return createAccordCommandStore(node, now, topology, executor); } diff --git a/test/unit/org/apache/cassandra/service/accord/SimulatedAccordCommandStore.java b/test/unit/org/apache/cassandra/service/accord/SimulatedAccordCommandStore.java index 868134292c..e0d40cd731 100644 --- a/test/unit/org/apache/cassandra/service/accord/SimulatedAccordCommandStore.java +++ b/test/unit/org/apache/cassandra/service/accord/SimulatedAccordCommandStore.java @@ -89,7 +89,6 @@ import org.apache.cassandra.db.Keyspace; import org.apache.cassandra.db.compaction.CompactionManager; import org.apache.cassandra.db.filter.ColumnFilter; import org.apache.cassandra.db.memtable.Memtable; -import org.apache.cassandra.metrics.AccordCacheMetrics; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.tcm.ClusterMetadata; @@ -273,7 +272,7 @@ public class SimulatedAccordCommandStore implements AutoCloseable }), updateHolder, journal, - new AccordExecutorSimple(0, CommandStore.class.getSimpleName() + '[' + 0 + ']', new AccordCacheMetrics("test"), agent)); + new AccordExecutorSimple(0, CommandStore.class.getSimpleName() + '[' + 0 + ']', agent)); this.commandStore.executor().executeDirectlyWithLock(() -> { commandStore.executor().setCapacity(8 << 20); commandStore.executor().setWorkingSetSize(4 << 20); diff --git a/test/unit/org/apache/cassandra/service/accord/serializers/CommandsForKeySerializerTest.java b/test/unit/org/apache/cassandra/service/accord/serializers/CommandsForKeySerializerTest.java index b806433dc1..fb62e7fd59 100644 --- a/test/unit/org/apache/cassandra/service/accord/serializers/CommandsForKeySerializerTest.java +++ b/test/unit/org/apache/cassandra/service/accord/serializers/CommandsForKeySerializerTest.java @@ -108,7 +108,6 @@ import org.apache.cassandra.dht.Token; import org.apache.cassandra.schema.KeyspaceParams; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.StorageService; -import org.apache.cassandra.service.accord.AccordTestUtils; import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.service.accord.txn.TxnData; import org.apache.cassandra.service.accord.txn.TxnWrite; @@ -181,7 +180,8 @@ public class CommandsForKeySerializerTest if (saveStatus.known.isDefinitionKnown()) builder.partialTxn(txn); - builder.setParticipants(StoreParticipants.all(txn.keys().toRoute(txn.keys().get(0).someIntersectingRoutingKey(null)))); + StoreParticipants participants = StoreParticipants.all(txn.keys().toRoute(txn.keys().get(0).someIntersectingRoutingKey(null))); + builder.setParticipants(participants); builder.durability(isDurable ? AllQuorums : NotDurable); if (saveStatus.known.deps().hasPreAcceptedOrProposedOrDecidedDeps()) { @@ -189,7 +189,7 @@ public class CommandsForKeySerializerTest { for (TxnId id : deps) keyBuilder.add(((Key)txn.keys().get(0)).toUnseekable(), id); - builder.partialDeps(new PartialDeps(AccordTestUtils.fullRange(txn), keyBuilder.build(), RangeDeps.NONE)); + builder.partialDeps(new PartialDeps(participants.touches(), keyBuilder.build(), RangeDeps.NONE)); } } @@ -669,6 +669,7 @@ public class CommandsForKeySerializerTest @Override public long maxConflictsPruneInterval() { return 0; } @Override public Txn emptySystemTxn(Kind kind, Routable.Domain domain) { throw new UnsupportedOperationException(); } @Override public long slowCoordinatorDelay(Node node, SafeCommandStore safeStore, TxnId txnId, TimeUnit units, int retryCount) { return 0; } + @Override public boolean isSlowCoordinator(long elapsed, TimeUnit units, TxnId txnId, int attempt) { return false; } @Override public long slowReplicaDelay(Node node, SafeCommandStore safeStore, TxnId txnId, int retryCount, ProgressLog.BlockedUntil blockedUntil, TimeUnit units) { return 0; } @Override public long slowAwaitDelay(Node node, SafeCommandStore safeStore, TxnId txnId, int retryCount, ProgressLog.BlockedUntil retrying, TimeUnit units) { return 0; } @Override public long retrySyncPointDelay(Node node, int attempt, TimeUnit units) { return 0; } diff --git a/test/unit/org/apache/cassandra/service/accord/serializers/WaitingOnSerializerTest.java b/test/unit/org/apache/cassandra/service/accord/serializers/WaitingOnSerializerTest.java index 85d2618f6e..0e209ce3e8 100644 --- a/test/unit/org/apache/cassandra/service/accord/serializers/WaitingOnSerializerTest.java +++ b/test/unit/org/apache/cassandra/service/accord/serializers/WaitingOnSerializerTest.java @@ -33,7 +33,7 @@ import accord.primitives.RoutingKeys; import accord.primitives.TxnId; import accord.utils.Gen; import accord.utils.Gens; -import accord.utils.SimpleBitSet; +import accord.utils.LargeBitSet; import accord.utils.Utils; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.dht.Murmur3Partitioner; @@ -95,8 +95,8 @@ public class WaitingOnSerializerTest int txnIdCount = deps.rangeDeps.txnIdCount(); int keyCount = deps.keyDeps.keys().size(); int[] selected = Gens.arrays(Gens.ints().between(0, txnIdCount + keyCount - 1)).unique().ofSizeBetween(0, txnIdCount + keyCount).next(rs); - SimpleBitSet waitingOn = new SimpleBitSet(txnIdCount + keyCount, false); - SimpleBitSet appliedOrInvalidated = rs.nextBoolean() ? null : new SimpleBitSet(txnIdCount, false); + LargeBitSet waitingOn = new LargeBitSet(txnIdCount + keyCount, false); + LargeBitSet appliedOrInvalidated = rs.nextBoolean() ? null : new LargeBitSet(txnIdCount, false); for (int i : selected) { WaitingOnSets set = appliedOrInvalidated == null || i >= txnIdCount ? WaitingOnSets.APPLY : sets.next(rs); diff --git a/test/unit/org/apache/cassandra/utils/SimpleBitSetSerializerTest.java b/test/unit/org/apache/cassandra/utils/LargeBitSetSerializerTest.java similarity index 84% rename from test/unit/org/apache/cassandra/utils/SimpleBitSetSerializerTest.java rename to test/unit/org/apache/cassandra/utils/LargeBitSetSerializerTest.java index f402be9c7c..7019835229 100644 --- a/test/unit/org/apache/cassandra/utils/SimpleBitSetSerializerTest.java +++ b/test/unit/org/apache/cassandra/utils/LargeBitSetSerializerTest.java @@ -19,27 +19,27 @@ package org.apache.cassandra.utils; import accord.utils.Gen; -import accord.utils.SimpleBitSet; +import accord.utils.LargeBitSet; import org.apache.cassandra.io.Serializers; import org.apache.cassandra.io.util.DataOutputBuffer; import org.junit.Test; import static accord.utils.Property.qt; -public class SimpleBitSetSerializerTest +public class LargeBitSetSerializerTest { @Test public void test() { @SuppressWarnings({ "resource", "IOResourceOpenedButNotSafelyClosed" }) DataOutputBuffer output = new DataOutputBuffer(); - qt().forAll(simpleBitSetGen()).check(bits -> Serializers.testSerde(output, SimpleBitSetSerializer.instance, bits)); + qt().forAll(largeBitSetGen()).check(bits -> Serializers.testSerde(output, LargeBitSetSerializer.instance, bits)); } - private static Gen simpleBitSetGen() + private static Gen largeBitSetGen() { return rs -> { int size = rs.nextInt(0, 1 << 10); - SimpleBitSet bitSet = new SimpleBitSet(size); + LargeBitSet bitSet = new LargeBitSet(size); if (size == 0 || rs.decide(0.2)) return bitSet; // empty if (rs.decide(0.2))