From e6cf2132ab33b2d6e68e79cb2e6d5c3a31103088 Mon Sep 17 00:00:00 2001 From: Ariel Weisberg Date: Wed, 9 Apr 2025 13:24:56 -0400 Subject: [PATCH] Add testing of consensus live migration to simulator Patch by Ariel Weisberg; Reviewed by Benedict Elliott Smith for CASSANDRA-20587 --- .gitmodules | 4 +- ide/idea/workspace.xml | 1 + modules/accord | 2 +- .../cql3/statements/CQL3CasRequest.java | 14 +- .../org/apache/cassandra/db/ReadCommand.java | 8 +- .../db/SinglePartitionReadCommand.java | 1 + .../apache/cassandra/db/SystemKeyspace.java | 14 +- .../exceptions/ReadFailureException.java | 12 ++ .../exceptions/ReadTimeoutException.java | 12 ++ .../exceptions/RequestFailureException.java | 9 + .../exceptions/RequestTimeoutException.java | 9 + .../metrics/CASClientRequestMetrics.java | 4 - .../cassandra/metrics/KeyspaceMetrics.java | 2 + src/java/org/apache/cassandra/net/Verb.java | 5 +- .../apache/cassandra/repair/RepairJob.java | 21 ++- .../cassandra/service/StorageProxy.java | 18 +- .../service/accord/AccordCommandStore.java | 8 + .../service/accord/AccordResult.java | 38 +++- .../service/accord/AccordService.java | 28 ++- .../service/accord/IAccordService.java | 47 ++++- .../service/accord/api/AccordAgent.java | 2 +- .../service/accord/api/PartitionKey.java | 8 + .../interop/AccordInteropExecution.java | 52 ++++- .../accord/interop/AccordInteropRead.java | 4 +- .../interop/AccordInteropReadCallback.java | 19 +- .../interop/AccordInteropReadRepair.java | 6 +- .../service/accord/repair/AccordRepair.java | 52 +++-- .../service/accord/txn/AccordUpdate.java | 5 + .../accord/txn/AccordUpdateParameters.java | 4 +- .../service/accord/txn/TxnNamedRead.java | 4 +- .../service/accord/txn/TxnQuery.java | 3 + .../service/accord/txn/TxnUpdate.java | 17 ++ .../service/consensus/TransactionalMode.java | 84 +++++++-- .../migration/ConsensusKeyMigrationState.java | 47 +++-- .../migration/ConsensusMigratedAt.java | 12 +- .../ConsensusMigrationRepairResult.java | 18 +- .../migration/ConsensusRequestRouter.java | 106 +++++++---- .../migration/ConsensusTableMigration.java | 15 +- .../migration/TableMigrationState.java | 18 ++ .../apache/cassandra/service/paxos/Paxos.java | 42 ++--- .../service/paxos/PaxosCommitAndPrepare.java | 18 +- .../cassandra/service/paxos/PaxosPrepare.java | 121 +++++++----- .../service/paxos/PaxosPrepareRefresh.java | 28 ++- .../cassandra/service/paxos/PaxosPropose.java | 101 ++++------ .../cassandra/service/paxos/PaxosRepair.java | 20 +- .../service/paxos/PaxosRequestCallback.java | 11 ++ .../cassandra/service/paxos/PaxosState.java | 31 +-- .../paxos/cleanup/PaxosUpdateLowBallot.java | 126 +++++++++++++ .../paxos/uncommitted/PaxosBallotTracker.java | 29 ++- .../cassandra/tcm/EpochAwareDebounce.java | 2 + .../apache/cassandra/tools/RepairRunner.java | 25 ++- .../cassandra/distributed/api/IMessage.java | 6 +- .../distributed/impl/MessageImpl.java | 15 +- .../distributed/test/TestBaseImpl.java | 58 +++++- .../AccordMigrationReadRaceTestBase.java | 9 +- .../test/accord/AccordMigrationTest.java | 82 +++----- .../AccordMigrationWriteRaceTestBase.java | 5 +- .../AccordReadInteroperabilityTest.java | 4 +- .../AccordWriteInteroperabilityTest.java | 154 ++++++++------- .../test/accord/InteropAccordCQLTest.java | 11 -- .../simulator/asm/InterceptClasses.java | 1 + .../simulator/AccordNetworkScheduler.java | 177 ++++++++++++++++++ .../AlwaysDeliverNetworkScheduler.java | 16 +- .../simulator/ClusterSimulation.java | 7 +- .../simulator/FixedLossNetworkScheduler.java | 18 +- .../simulator/FutureActionScheduler.java | 34 +++- .../cassandra/simulator/SimulationRunner.java | 2 +- .../cassandra/simulator/SimulatorUtils.java | 24 ++- .../cluster/ClusterActionListener.java | 6 +- .../simulator/cluster/KeyspaceActions.java | 13 +- .../OnClusterAssertMigrationComplete.java | 67 +++++++ .../simulator/cluster/OnClusterChangeRf.java | 3 +- .../cluster/OnClusterConsensusMigrations.java | 69 +++++++ .../cluster/OnClusterFullRepair.java | 13 +- .../cluster/OnClusterMigrateConsensus.java | 130 +++++++++---- .../OnClusterMigrateConsensusOneRange.java | 45 ++++- .../cluster/OnClusterRepairRanges.java | 5 +- .../simulator/cluster/OnClusterReplace.java | 3 +- .../OnInstanceFinishConsensusMigration.java | 83 ++++++++ .../simulator/cluster/OnInstanceRepair.java | 68 +++++-- .../OnInstanceStartConsensusMigration.java | 14 +- .../simulator/debug/SelfReconcile.java | 46 +++-- .../paxos/AccordClusterSimulation.java | 1 + .../paxos/AccordSimulationRunner.java | 7 +- .../PairOfSequencesAccordSimulation.java | 2 +- .../paxos/PaxosClusterSimulation.java | 60 ++++++ .../simulator/paxos/PaxosRepairValidator.java | 7 +- .../paxos/PaxosSimulationRunner.java | 15 +- .../simulator/systems/SimulatedAction.java | 14 +- .../SimulatedFutureActionScheduler.java | 27 ++- .../simulator/utils/KindOfSequence.java | 11 ++ .../cassandra/simulator/utils/LongRange.java | 8 + .../simulator/test/HarrySimulatorTest.java | 2 +- .../test/ShortAccordSimulationTest.java | 1 - ...axosMigrationFromAccordSimulationTest.java | 113 +++++++++++ ...tPaxosMigrationToAccordSimulationTest.java | 111 +++++++++++ .../test/ShortPaxosSimulationTest.java | 15 -- .../simulator/test/SimulationTestBase.java | 31 ++- .../cassandra/audit/AuditLoggerTest.java | 10 +- .../apache/cassandra/auth/TxnAuthTest.java | 10 +- .../cql3/PreparedStatementsTest.java | 3 + .../accord/AccordCommandStoreTest.java | 7 +- .../service/accord/AccordReadRepairTest.java | 2 +- .../uncommitted/PaxosBallotTrackerTest.java | 7 +- .../uncommitted/PaxosStateTrackerTest.java | 3 + ...axosUncommittedTrackerIntegrationTest.java | 4 +- 106 files changed, 2227 insertions(+), 719 deletions(-) create mode 100644 src/java/org/apache/cassandra/service/paxos/cleanup/PaxosUpdateLowBallot.java create mode 100644 test/simulator/main/org/apache/cassandra/simulator/AccordNetworkScheduler.java create mode 100644 test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterAssertMigrationComplete.java create mode 100644 test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterConsensusMigrations.java create mode 100644 test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceFinishConsensusMigration.java create mode 100644 test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationFromAccordSimulationTest.java create mode 100644 test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationToAccordSimulationTest.java diff --git a/.gitmodules b/.gitmodules index 616dacf610..4d7128c5f9 100644 --- a/.gitmodules +++ b/.gitmodules @@ -1,4 +1,4 @@ [submodule "modules/accord"] path = modules/accord - url = https://github.com/apache/cassandra-accord.git - branch = trunk + url = https://github.com/aweisberg/cassandra-accord.git + branch = 20587 diff --git a/ide/idea/workspace.xml b/ide/idea/workspace.xml index 5eb1a70b78..611a93735e 100644 --- a/ide/idea/workspace.xml +++ b/ide/idea/workspace.xml @@ -196,6 +196,7 @@ -Dcassandra.test.messagingService.nonGracefulShutdown=true -Dcassandra.test.simulator.determinismcheck=strict -Dcassandra.test.sstableformatdevelopment=true + -Dcassandra.test.accord.allow_test_modes=true -Dcassandra.tolerate_sstable_size=true -Dcassandra.use_nix_recursive_delete=true -Dinvalid-legacy-sstable-root=$PROJECT_DIR$/test/data/invalid-legacy-sstables diff --git a/modules/accord b/modules/accord index 3aba47c29a..ad26bb9df8 160000 --- a/modules/accord +++ b/modules/accord @@ -1 +1 @@ -Subproject commit 3aba47c29a1cb13dba05814fd6b007165193fd7e +Subproject commit ad26bb9df8f774300668a7099858e951e3f013cb diff --git a/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java b/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java index e7a561c76c..eeadd10456 100644 --- a/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java +++ b/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java @@ -41,6 +41,7 @@ import org.apache.cassandra.db.Clustering; import org.apache.cassandra.db.Columns; import org.apache.cassandra.db.ConsistencyLevel; import org.apache.cassandra.db.DecoratedKey; +import org.apache.cassandra.db.ReadCommand.PotentialTxnConflicts; import org.apache.cassandra.db.RegularAndStaticColumns; import org.apache.cassandra.db.SinglePartitionReadCommand; import org.apache.cassandra.db.Slice; @@ -248,12 +249,13 @@ public class CQL3CasRequest implements CASRequest // that exists (has some live data) but has not static content. So we query the first live row of the partition. if (conditions.isEmpty()) return SinglePartitionReadCommand.create(metadata, - nowInSec, - columnFilter, - RowFilter.none(), - DataLimits.cqlLimits(1), - key, - new ClusteringIndexSliceFilter(Slices.ALL, false)); + nowInSec, + columnFilter, + RowFilter.none(), + DataLimits.cqlLimits(1), + key, + new ClusteringIndexSliceFilter(Slices.ALL, false), + PotentialTxnConflicts.ALLOW); ClusteringIndexNamesFilter filter = new ClusteringIndexNamesFilter(conditions.navigableKeySet(), false); return SinglePartitionReadCommand.create(metadata, nowInSec, key, columnFilter, filter); diff --git a/src/java/org/apache/cassandra/db/ReadCommand.java b/src/java/org/apache/cassandra/db/ReadCommand.java index ae53eb292c..90ad710c33 100644 --- a/src/java/org/apache/cassandra/db/ReadCommand.java +++ b/src/java/org/apache/cassandra/db/ReadCommand.java @@ -488,6 +488,12 @@ public abstract class ReadCommand extends AbstractReadQuery */ // iterators created inside the try as long as we do close the original resultIterator), or by closing the result. public UnfilteredPartitionIterator executeLocally(ReadExecutionController executionController) + { + return executeLocally(executionController, null); + } + + // ClusterMetadata is null on startup when there are local reads from system tables before it's initialized + public UnfilteredPartitionIterator executeLocally(ReadExecutionController executionController, @Nullable ClusterMetadata cm) { long startTimeNanos = nanoTime(); @@ -496,7 +502,7 @@ public abstract class ReadCommand extends AbstractReadQuery { ColumnFamilyStore cfs = Keyspace.openAndGetStore(metadata()); if (!potentialTxnConflicts.allowed) - ConsensusRequestRouter.validateSafeToReadNonTransactionally(this); + ConsensusRequestRouter.validateSafeToReadNonTransactionally(this, cm); Index.QueryPlan indexQueryPlan = indexQueryPlan(); Index.Searcher searcher = null; diff --git a/src/java/org/apache/cassandra/db/SinglePartitionReadCommand.java b/src/java/org/apache/cassandra/db/SinglePartitionReadCommand.java index ec3b383bb9..a0a83f728b 100644 --- a/src/java/org/apache/cassandra/db/SinglePartitionReadCommand.java +++ b/src/java/org/apache/cassandra/db/SinglePartitionReadCommand.java @@ -221,6 +221,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar * @param limits the limits to use for the query. * @param partitionKey the partition key for the partition to query. * @param clusteringIndexFilter the clustering index filter to use for the query. + * @param potentialTxnConflicts Whether to generate an error if this read could potentially conflict with a txn * * @return a newly created read command. */ diff --git a/src/java/org/apache/cassandra/db/SystemKeyspace.java b/src/java/org/apache/cassandra/db/SystemKeyspace.java index dd38684b2d..a15b46665f 100644 --- a/src/java/org/apache/cassandra/db/SystemKeyspace.java +++ b/src/java/org/apache/cassandra/db/SystemKeyspace.java @@ -102,6 +102,8 @@ import org.apache.cassandra.schema.Types; import org.apache.cassandra.schema.UserFunctions; import org.apache.cassandra.schema.Views; import org.apache.cassandra.service.StorageService; +import org.apache.cassandra.service.consensus.migration.ConsensusMigratedAt; +import org.apache.cassandra.service.consensus.migration.ConsensusMigrationTarget; import org.apache.cassandra.service.paxos.Ballot; import org.apache.cassandra.service.paxos.Commit; import org.apache.cassandra.service.paxos.Commit.Accepted; @@ -148,8 +150,6 @@ import static org.apache.cassandra.gms.ApplicationState.RACK; import static org.apache.cassandra.gms.ApplicationState.RELEASE_VERSION; import static org.apache.cassandra.gms.ApplicationState.STATUS_WITH_PORT; import static org.apache.cassandra.gms.ApplicationState.TOKENS; -import org.apache.cassandra.service.consensus.migration.ConsensusMigratedAt; -import org.apache.cassandra.service.consensus.migration.ConsensusMigrationTarget; import static org.apache.cassandra.service.paxos.Commit.latest; import static org.apache.cassandra.service.snapshot.SnapshotOptions.systemSnapshot; import static org.apache.cassandra.utils.CassandraVersion.NULL_VERSION; @@ -270,6 +270,7 @@ public final class SystemKeyspace + "row_key blob, " + "cf_id UUID, " + "consensus_migrated_at_epoch bigint, " + + "consensus_max_hlc bigint, " + "consensus_target tinyint, " + "PRIMARY KEY ((row_key), cf_id, consensus_migrated_at_epoch)) " + "WITH CLUSTERING ORDER BY (cf_id ASC, consensus_migrated_at_epoch DESC)") @@ -1628,13 +1629,13 @@ public final class SystemKeyspace public static void saveConsensusKeyMigrationState(ByteBuffer partitionKey, UUID cfId, ConsensusMigratedAt consensusMigratedAt) { - String cql = "UPDATE system." + CONSENSUS_MIGRATION_STATE + " SET consensus_target = ? WHERE row_key = ? AND cf_id = ? AND consensus_migrated_at_epoch = ?"; - executeInternal(cql, consensusMigratedAt.migratedAtTarget.value, partitionKey, cfId, consensusMigratedAt.migratedAtEpoch.getEpoch()); + String cql = "UPDATE system." + CONSENSUS_MIGRATION_STATE + " SET consensus_target = ?, consensus_max_hlc = ? WHERE row_key = ? AND cf_id = ? AND consensus_migrated_at_epoch = ?"; + executeInternal(cql, consensusMigratedAt.migratedAtTarget.value, consensusMigratedAt.maxHLC, partitionKey, cfId, consensusMigratedAt.migratedAtEpoch.getEpoch()); } public static ConsensusMigratedAt loadConsensusKeyMigrationState(ByteBuffer partitionKey, UUID cfId) { - String cql = "SELECT consensus_migrated_at_epoch, consensus_target FROM system." + CONSENSUS_MIGRATION_STATE + " WHERE row_key = ? AND cf_id = ? LIMIT 1"; + String cql = "SELECT consensus_migrated_at_epoch, consensus_target, consensus_max_hlc FROM system." + CONSENSUS_MIGRATION_STATE + " WHERE row_key = ? AND cf_id = ? LIMIT 1"; UntypedResultSet results = executeInternal(cql, partitionKey, cfId); if (results.isEmpty()) @@ -1644,7 +1645,8 @@ public final class SystemKeyspace // TODO Period won't be necessary eventually Epoch migratedAtEpoch = Epoch.create(row.getLong("consensus_migrated_at_epoch")); ConsensusMigrationTarget target = ConsensusMigrationTarget.fromValue(row.getByte("consensus_target")); - return new ConsensusMigratedAt(migratedAtEpoch, target); + long maxHLC = row.getLong("consensus_max_hlc"); + return new ConsensusMigratedAt(migratedAtEpoch, maxHLC, target); } /** diff --git a/src/java/org/apache/cassandra/exceptions/ReadFailureException.java b/src/java/org/apache/cassandra/exceptions/ReadFailureException.java index 698c8ae9c4..98f0e7e20b 100644 --- a/src/java/org/apache/cassandra/exceptions/ReadFailureException.java +++ b/src/java/org/apache/cassandra/exceptions/ReadFailureException.java @@ -39,4 +39,16 @@ public class ReadFailureException extends RequestFailureException super(ExceptionCode.READ_FAILURE, msg, consistency, received, blockFor, failureReasonByEndpoint); this.dataPresent = dataPresent; } + + public ReadFailureException(String msg, ConsistencyLevel consistency, int received, int blockFor, boolean dataPresent, Map failureReasonByEndpoint, Throwable cause) + { + super(ExceptionCode.READ_FAILURE, msg, consistency, received, blockFor, failureReasonByEndpoint, cause); + this.dataPresent = dataPresent; + } + + public ReadFailureException(ReadFailureException rfe) + { + super(ExceptionCode.READ_FAILURE, rfe.getMessage(), rfe.consistency, rfe.received, rfe.blockFor, rfe.failureReasonByEndpoint, rfe); + this.dataPresent = rfe.dataPresent; + } } diff --git a/src/java/org/apache/cassandra/exceptions/ReadTimeoutException.java b/src/java/org/apache/cassandra/exceptions/ReadTimeoutException.java index 809f0a1780..199f165d31 100644 --- a/src/java/org/apache/cassandra/exceptions/ReadTimeoutException.java +++ b/src/java/org/apache/cassandra/exceptions/ReadTimeoutException.java @@ -34,4 +34,16 @@ public class ReadTimeoutException extends RequestTimeoutException super(ExceptionCode.READ_TIMEOUT, consistency, received, blockFor, msg); this.dataPresent = dataPresent; } + + public ReadTimeoutException(ConsistencyLevel consistency, int received, int blockFor, boolean dataPresent, Throwable cause) + { + super(ExceptionCode.READ_TIMEOUT, consistency, received, blockFor, cause); + this.dataPresent = dataPresent; + } + + public ReadTimeoutException(ReadFailureException rfe) + { + super(ExceptionCode.READ_TIMEOUT, rfe.consistency, rfe.received, rfe.blockFor, rfe); + this.dataPresent = rfe.dataPresent; + } } diff --git a/src/java/org/apache/cassandra/exceptions/RequestFailureException.java b/src/java/org/apache/cassandra/exceptions/RequestFailureException.java index a71534720b..2998032716 100644 --- a/src/java/org/apache/cassandra/exceptions/RequestFailureException.java +++ b/src/java/org/apache/cassandra/exceptions/RequestFailureException.java @@ -44,6 +44,15 @@ public class RequestFailureException extends RequestExecutionException this.failureReasonByEndpoint = failureReasonByEndpoint; } + public RequestFailureException(ExceptionCode code, String msg, ConsistencyLevel consistency, int received, int blockFor, Map failureReasonByEndpoint, Throwable cause) + { + super(code, buildErrorMessage(msg, failureReasonByEndpoint), cause); + this.consistency = consistency; + this.received = received; + this.blockFor = blockFor; + this.failureReasonByEndpoint = failureReasonByEndpoint; + } + private static String buildErrorMessage(int received, Map failures) { return String.format("received %d responses and %d failures", received, failures.size()); diff --git a/src/java/org/apache/cassandra/exceptions/RequestTimeoutException.java b/src/java/org/apache/cassandra/exceptions/RequestTimeoutException.java index 853ba2fd1c..992dfba791 100644 --- a/src/java/org/apache/cassandra/exceptions/RequestTimeoutException.java +++ b/src/java/org/apache/cassandra/exceptions/RequestTimeoutException.java @@ -40,4 +40,13 @@ public class RequestTimeoutException extends RequestExecutionException this.received = received; this.blockFor = blockFor; } + + // TODO (review): Was there an intentional choice here that RequestTimeoutException doesn't wrap its causes for brevity? + protected RequestTimeoutException(ExceptionCode code, ConsistencyLevel consistency, int received, int blockFor, Throwable cause) + { + super(code, String.format("Operation timed out - received only %d responses.", received), cause); + this.consistency = consistency; + this.received = received; + this.blockFor = blockFor; + } } diff --git a/src/java/org/apache/cassandra/metrics/CASClientRequestMetrics.java b/src/java/org/apache/cassandra/metrics/CASClientRequestMetrics.java index f3ae6c8924..4445ef9171 100644 --- a/src/java/org/apache/cassandra/metrics/CASClientRequestMetrics.java +++ b/src/java/org/apache/cassandra/metrics/CASClientRequestMetrics.java @@ -31,8 +31,6 @@ public class CASClientRequestMetrics extends ClientRequestMetrics public final Meter unknownResult; // CAS request rejected after Prepare/Promise due to migration from Paxos to Accord public final Meter beginMigrationRejects; - // Number of times a CAS request was rejected after Propose/Accept due to migration from Paxos to Accord - public final Meter acceptMigrationRejects; // Number of times a key was migrated from Accord to Paxos public final Meter accordKeyMigrations; @@ -43,7 +41,6 @@ public class CASClientRequestMetrics extends ClientRequestMetrics unfinishedCommit = Metrics.counter(factory.createMetricName("UnfinishedCommit")); unknownResult = Metrics.meter(factory.createMetricName("UnknownResult")); beginMigrationRejects = Metrics.meter(factory.createMetricName("PaxosBeginMigrationRejects")); - acceptMigrationRejects = Metrics.meter(factory.createMetricName("PaxosAcceptMigrationRejects")); accordKeyMigrations = Metrics.meter(factory.createMetricName("AccordKeyMigrations")); } @@ -54,7 +51,6 @@ public class CASClientRequestMetrics extends ClientRequestMetrics Metrics.remove(factory.createMetricName("UnfinishedCommit")); Metrics.remove(factory.createMetricName("UnknownResult")); Metrics.remove(factory.createMetricName("PaxosBeginMigrationRejects")); - Metrics.remove(factory.createMetricName("PaxosAcceptMigrationRejects")); Metrics.remove(factory.createMetricName("AccordKeyMigrations")); } } diff --git a/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java b/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java index e205230bd2..aca570c098 100644 --- a/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java +++ b/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java @@ -103,6 +103,7 @@ public class KeyspaceMetrics public final LatencyMetrics casCommit; /** Latency for locally run key migrations **/ public final LatencyMetrics keyMigration; + public final LatencyMetrics accordGetMaxConflicts; /** Latency for range migrations run by locally coordinated Accord repairs **/ public final LatencyMetrics accordRepair; public final LatencyMetrics accordPostStreamRepair; @@ -261,6 +262,7 @@ public class KeyspaceMetrics casPropose = createLatencyMetrics("CasPropose"); casCommit = createLatencyMetrics("CasCommit"); keyMigration = createLatencyMetrics("KeyMigration"); + accordGetMaxConflicts = createLatencyMetrics("AccordGetMaxConflicts"); accordRepair = createLatencyMetrics("AccordRepair"); accordPostStreamRepair = createLatencyMetrics("AccordPostStreamRepair"); rangeMigrationUnexpectedFailures = createKeyspaceMeter("RangeMigrationUnexpectedFailures"); diff --git a/src/java/org/apache/cassandra/net/Verb.java b/src/java/org/apache/cassandra/net/Verb.java index 3103b2058b..f157bf41bd 100644 --- a/src/java/org/apache/cassandra/net/Verb.java +++ b/src/java/org/apache/cassandra/net/Verb.java @@ -98,12 +98,12 @@ import org.apache.cassandra.service.accord.serializers.CheckStatusSerializers; import org.apache.cassandra.service.accord.serializers.CommitSerializers; import org.apache.cassandra.service.accord.serializers.EnumSerializer; import org.apache.cassandra.service.accord.serializers.FetchSerializers; +import org.apache.cassandra.service.accord.serializers.GetDurableBeforeSerializers; import org.apache.cassandra.service.accord.serializers.GetEphmrlReadDepsSerializers; import org.apache.cassandra.service.accord.serializers.GetMaxConflictSerializers; import org.apache.cassandra.service.accord.serializers.InformDurableSerializers; import org.apache.cassandra.service.accord.serializers.LatestDepsSerializers; import org.apache.cassandra.service.accord.serializers.PreacceptSerializers; -import org.apache.cassandra.service.accord.serializers.GetDurableBeforeSerializers; import org.apache.cassandra.service.accord.serializers.ReadDataSerializer; import org.apache.cassandra.service.accord.serializers.RecoverySerializers; import org.apache.cassandra.service.accord.serializers.SetDurableSerializers; @@ -125,6 +125,7 @@ import org.apache.cassandra.service.paxos.cleanup.PaxosCleanupRequest; import org.apache.cassandra.service.paxos.cleanup.PaxosCleanupResponse; import org.apache.cassandra.service.paxos.cleanup.PaxosFinishPrepareCleanup; import org.apache.cassandra.service.paxos.cleanup.PaxosStartPrepareCleanup; +import org.apache.cassandra.service.paxos.cleanup.PaxosUpdateLowBallot; import org.apache.cassandra.service.paxos.v1.PrepareVerbHandler; import org.apache.cassandra.service.paxos.v1.ProposeVerbHandler; import org.apache.cassandra.streaming.DataMovement; @@ -288,6 +289,8 @@ public enum Verb PAXOS2_CLEANUP_FINISH_PREPARE_REQ(47, P2, repairTimeout, IMMEDIATE, () -> PaxosCleanupHistory.serializer, () -> PaxosFinishPrepareCleanup.verbHandler, PAXOS2_CLEANUP_FINISH_PREPARE_RSP), PAXOS2_CLEANUP_COMPLETE_RSP (59, P2, repairTimeout, PAXOS_REPAIR, () -> NoPayload.serializer, RESPONSE_HANDLER ), PAXOS2_CLEANUP_COMPLETE_REQ (48, P2, repairTimeout, PAXOS_REPAIR, () -> PaxosCleanupComplete.serializer, () -> PaxosCleanupComplete.verbHandler, PAXOS2_CLEANUP_COMPLETE_RSP ), + PAXOS2_UPDATE_LOW_BALLOT_RSP (67, P2, repairTimeout, PAXOS_REPAIR, () -> NoPayload.serializer, RESPONSE_HANDLER ), + PAXOS2_UPDATE_LOW_BALLOT_REQ (64, P2, repairTimeout, PAXOS_REPAIR, () -> PaxosUpdateLowBallot.serializer, () -> PaxosUpdateLowBallot.verbHandler, PAXOS2_UPDATE_LOW_BALLOT_RSP ), // transactional cluster metadata TCM_COMMIT_RSP (801, P0, rpcTimeout, INTERNAL_METADATA, MessageSerializers::commitResultSerializer, RESPONSE_HANDLER ), diff --git a/src/java/org/apache/cassandra/repair/RepairJob.java b/src/java/org/apache/cassandra/repair/RepairJob.java index f20fe189f2..4bebca17e4 100644 --- a/src/java/org/apache/cassandra/repair/RepairJob.java +++ b/src/java/org/apache/cassandra/repair/RepairJob.java @@ -39,7 +39,6 @@ import com.google.common.util.concurrent.FutureCallback; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import accord.primitives.Ranges; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.Keyspace; @@ -53,9 +52,12 @@ import org.apache.cassandra.repair.asymmetric.ReduceHelper; import org.apache.cassandra.repair.state.JobState; import org.apache.cassandra.schema.SystemDistributedKeyspace; import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.service.accord.IAccordService; import org.apache.cassandra.service.accord.repair.AccordRepair; +import org.apache.cassandra.service.accord.repair.AccordRepair.AccordRepairResult; import org.apache.cassandra.service.consensus.migration.ConsensusMigrationRepairResult; import org.apache.cassandra.service.paxos.cleanup.PaxosCleanup; +import org.apache.cassandra.service.paxos.cleanup.PaxosUpdateLowBallot; import org.apache.cassandra.streaming.PreviewKind; import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tcm.Epoch; @@ -173,7 +175,7 @@ public class RepairJob extends AsyncFuture implements Runnable paxosRepair = ImmediateFuture.success(null); } - Future accordRepair; + Future accordRepair; if (doAccordRepair) { accordRepair = paxosRepair.flatMap(unused -> { @@ -191,7 +193,16 @@ public class RepairJob extends AsyncFuture implements Runnable } logger.info("{} {}.{} starting accord repair, require all endpoints {}", session.previewKind.logPrefix(session.getId()), desc.keyspace, desc.columnFamily, requireAllEndpoints); AccordRepair repair = new AccordRepair(ctx, cfs, desc.sessionId, desc.keyspace, desc.ranges, requireAllEndpoints, allEndpoints); - return repair.repair(taskExecutor); + return repair.repair(taskExecutor).flatMap(accordRepairResult -> { + // Propagate the HLC discovered during Accord repair to Paxos so Paxos doesn't use ballots < Accord has already used + if (accordRepairResult.maxHlc != IAccordService.NO_HLC) + { + PaxosUpdateLowBallot paxosLowBallot = new PaxosUpdateLowBallot(ctx, allEndpoints, accordRepairResult.maxHlc); + paxosLowBallot.start(); + return paxosLowBallot.map(ignored -> accordRepairResult); + } + return ImmediateFuture.success(accordRepairResult); + }, taskExecutor); }, taskExecutor); } else @@ -266,7 +277,7 @@ public class RepairJob extends AsyncFuture implements Runnable } cfs.metric.repairsCompleted.inc(); logger.info("Completing repair with excludedDeadNodes {}", session.excludedDeadNodes); - ConsensusMigrationRepairResult cmrs = ConsensusMigrationRepairResult.fromRepair(repairStartingEpoch, getUnchecked(accordRepair), session.repairData, doPaxosRepair, doAccordRepair, session.excludedDeadNodes); + ConsensusMigrationRepairResult cmrs = ConsensusMigrationRepairResult.fromRepair(repairStartingEpoch, getUnchecked(accordRepair), session.repairData, doPaxosRepair, doAccordRepair, session.excludedDeadNodes, session.isIncremental); trySuccess(new RepairResult(desc, stats, cmrs)); } @@ -293,7 +304,7 @@ public class RepairJob extends AsyncFuture implements Runnable }, taskExecutor); } - private Future> createSyncTasks(Future accordRepair, Future allSnapshotTasks, List allEndpoints) + private Future> createSyncTasks(Future accordRepair, Future allSnapshotTasks, List allEndpoints) { Future> treeResponses; if (allSnapshotTasks != null) diff --git a/src/java/org/apache/cassandra/service/StorageProxy.java b/src/java/org/apache/cassandra/service/StorageProxy.java index 4d0857b40c..1c91a1886d 100644 --- a/src/java/org/apache/cassandra/service/StorageProxy.java +++ b/src/java/org/apache/cassandra/service/StorageProxy.java @@ -151,6 +151,7 @@ import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper.SplitConsumer; import org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper.SplitMutations; import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter; +import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision; import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.SplitReads; import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; import org.apache.cassandra.service.paxos.Ballot; @@ -214,7 +215,6 @@ import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.range_read; import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol; import static org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper.mutateWithAccordAsync; import static org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper.splitMutationsIntoAccordAndNormal; -import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision; import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.getTableMetadata; import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.shouldReadEphemerally; import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.splitReadsIntoAccordAndNormal; @@ -372,13 +372,13 @@ public class StorageProxy implements StorageProxyMBean key, keyspaceName, cfName)); } - ConsensusAttemptResult lastAttemptResult; + ConsensusAttemptResult lastAttemptResult = null; do { ClusterMetadata cm = ClusterMetadata.current(); TableMetadata metadata = Schema.instance.validateTable(keyspaceName, cfName); ConsensusRoutingDecision decision = consensusRouting(cm, metadata, key, consistencyForPaxos, requestTime, true); - switch (decision) + switch (decision.target) { case paxosV2: lastAttemptResult = Paxos.cas(key, @@ -386,7 +386,8 @@ public class StorageProxy implements StorageProxyMBean consistencyForPaxos, consistencyForCommit, clientState, - requestTime); + requestTime, + decision.minHLC); break; case paxosV1: lastAttemptResult = legacyCas(metadata, @@ -408,7 +409,8 @@ public class StorageProxy implements StorageProxyMBean TxnResult txnResult = accordService.coordinate(metadata.epoch.getEpoch(), txn, consistencyForPaxos, - requestTime); + requestTime, + decision.minHLC); lastAttemptResult = request.toCasResult(txnResult); break; default: @@ -2146,7 +2148,7 @@ public class StorageProxy implements StorageProxyMBean private static ConsensusRoutingDecision consensusRouting(ClusterMetadata cm, TableMetadata metadata, DecoratedKey partitionKey, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, boolean isForWrite) { if (metadata.keyspace.equals(SchemaConstants.METADATA_KEYSPACE_NAME)) - return ConsensusRoutingDecision.paxosV2; + return ConsensusRoutingDecision.PAXOSV2; return ConsensusRequestRouter.instance.routeAndMaybeMigrate(cm, partitionKey, metadata.id, @@ -2165,10 +2167,10 @@ public class StorageProxy implements StorageProxyMBean ClusterMetadata cm = ClusterMetadata.current(); SinglePartitionReadCommand command = group.queries.get(0); ConsensusRoutingDecision decision = consensusRouting(cm, group.metadata(), command.partitionKey(), consistencyLevel, requestTime, false); - switch (decision) + switch (decision.target) { case paxosV2: - lastResult = Paxos.read(group, consistencyLevel, requestTime); + lastResult = Paxos.read(group, consistencyLevel, requestTime, decision.minHLC); break; case paxosV1: lastResult = legacyReadWithPaxos(group, consistencyLevel, requestTime); diff --git a/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java b/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java index f7e8c3004d..c8340a6b3c 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java @@ -66,6 +66,7 @@ import org.apache.cassandra.service.accord.AccordKeyspace.CommandsForKeyAccessor import org.apache.cassandra.service.accord.IAccordService.AccordCompactionInfo; import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.service.accord.txn.TxnRead; +import org.apache.cassandra.service.paxos.PaxosState; import org.apache.cassandra.utils.Clock; import static accord.api.Journal.CommandUpdate; @@ -546,6 +547,13 @@ public class AccordCommandStore extends CommandStore loadRangesForEpoch(rangesForEpoch); } + @Override + public void updateRangesForEpoch(SafeCommandStore safeStore) + { + super.updateRangesForEpoch(safeStore); + updateMinHlc(PaxosState.ballotTracker().getLowBound().unixMicros() + 1); + } + // TODO (expected): handle journal failures, and consider how we handle partial failures. // Very likely we will not be able to safely or cleanly handle partial failures of this logic, but decide and document. // TODO (desired): consider merging with PersistentField? This version is cheaper to manage which may be preferable at the CommandStore level. diff --git a/src/java/org/apache/cassandra/service/accord/AccordResult.java b/src/java/org/apache/cassandra/service/accord/AccordResult.java index 9d97d3108d..df1f17eb7d 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordResult.java +++ b/src/java/org/apache/cassandra/service/accord/AccordResult.java @@ -37,7 +37,10 @@ import accord.primitives.Seekables; import accord.primitives.TxnId; import accord.utils.UnhandledEnum; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.exceptions.ReadFailureException; +import org.apache.cassandra.exceptions.ReadTimeoutException; import org.apache.cassandra.exceptions.RequestExecutionException; +import org.apache.cassandra.exceptions.RequestFailureException; import org.apache.cassandra.schema.Schema; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.api.AccordAgent; @@ -88,6 +91,29 @@ public class AccordResult extends AsyncFuture implements BiConsumer extends AsyncFuture implements BiConsumer route = node.computeRoute(txnId, keys); - Txn txn = new Txn.InMemory(Write, keys, TxnRead.createNoOpRead(keys), TxnQuery.NONE, TxnUpdate.empty(), new TableMetadatasAndKeys(TableMetadatas.none(), keys)); - return CoordinateTransaction.coordinate(node, route, txnId, txn) - .map(ignore -> (Void)null).beginAsResult(); + return node.withEpochAtLeast(txnId.epoch(), null, () -> { + Txn txn = new Txn.InMemory(Write, keys, TxnRead.createNoOpRead(keys), TxnQuery.UNSAFE_EMPTY, TxnUpdate.empty(), new TableMetadatasAndKeys(TableMetadatas.none(), keys)); + return CoordinateTransaction.coordinate(node, route, txnId, txn) + .map(ignore -> (Void) null).beginAsResult(); + }).beginAsResult(); } @Override @@ -565,10 +568,9 @@ public class AccordService implements IAccordService, Shutdownable async.begin(result); return result.awaitAndGet(); } - - public static void getBlocking(AsyncChain async, Seekables keysOrRanges, RequestBookkeeping bookkeeping, long startedAt, long deadline) + public static V getBlocking(AsyncChain async, Seekables keysOrRanges, RequestBookkeeping bookkeeping, long startedAt, long deadline) { - getBlocking(async, keysOrRanges, bookkeeping, startedAt, deadline, false); + return getBlocking(async, keysOrRanges, bookkeeping, startedAt, deadline, false); } public static Keys intersecting(Keys keys) @@ -642,9 +644,9 @@ public class AccordService implements IAccordService, Shutdownable * with non-Accord operations. */ @Override - public @Nonnull TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull Dispatcher.RequestTime requestTime) throws RequestExecutionException + public @Nonnull TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull Dispatcher.RequestTime requestTime, long minHlc) throws RequestExecutionException { - return coordinateAsync(minEpoch, txn, consistencyLevel, requestTime).awaitAndGet(); + return coordinateAsync(minEpoch, txn, consistencyLevel, requestTime, minHlc).awaitAndGet(); } @Override @@ -654,9 +656,9 @@ public class AccordService implements IAccordService, Shutdownable } @Override - public @Nonnull IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull Dispatcher.RequestTime requestTime) + public @Nonnull IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull Dispatcher.RequestTime requestTime, long minHlc) { - TxnId txnId = node.nextTxnId(txn); + TxnId txnId = node.nextTxnId(minHlc >= 0 ? minHlc : 0, txn.kind(), txn.keys().domain(), cardinality(txn.keys())); long timeout = txnId.isWrite() ? DatabaseDescriptor.getWriteRpcTimeout(NANOSECONDS) : DatabaseDescriptor.getReadRpcTimeout(NANOSECONDS); ClientRequestBookkeeping bookkeeping = txn.isWrite() ? accordWriteBookkeeping : accordReadBookkeeping; bookkeeping.metrics.keySize.update(txn.keys().size()); @@ -900,6 +902,12 @@ public class AccordService implements IAccordService, Shutdownable return node; } + @Override + public void ensureMinHlc(long minHlc) + { + node.updateMinHlc(minHlc >= 0 ? minHlc : 0); + } + public AccordJournal journal() { return journal; diff --git a/src/java/org/apache/cassandra/service/accord/IAccordService.java b/src/java/org/apache/cassandra/service/accord/IAccordService.java index 136bd5290e..2e742fd934 100644 --- a/src/java/org/apache/cassandra/service/accord/IAccordService.java +++ b/src/java/org/apache/cassandra/service/accord/IAccordService.java @@ -32,13 +32,13 @@ import org.slf4j.Logger; import org.slf4j.LoggerFactory; import accord.api.Agent; -import accord.local.durability.DurabilityService.SyncLocal; -import accord.local.durability.DurabilityService.SyncRemote; import accord.local.CommandStores.RangesForEpoch; import accord.local.DurableBefore; import accord.local.Node; import accord.local.Node.Id; import accord.local.RedundantBefore; +import accord.local.durability.DurabilityService.SyncLocal; +import accord.local.durability.DurabilityService.SyncRemote; import accord.messages.Reply; import accord.messages.Request; import accord.primitives.Keys; @@ -74,6 +74,7 @@ public interface IAccordService EnumSet SUPPORTED_COMMIT_CONSISTENCY_LEVELS = EnumSet.of(ConsistencyLevel.ANY, ConsistencyLevel.ONE, ConsistencyLevel.QUORUM, ConsistencyLevel.SERIAL, ConsistencyLevel.ALL); EnumSet SUPPORTED_READ_CONSISTENCY_LEVELS = EnumSet.of(ConsistencyLevel.ONE, ConsistencyLevel.QUORUM, ConsistencyLevel.SERIAL, ConsistencyLevel.ALL); + long NO_HLC = Long.MIN_VALUE; IVerbHandler requestHandler(); IVerbHandler responseHandler(); @@ -82,8 +83,17 @@ public interface IAccordService AsyncChain sync(@Nullable Timestamp minBound, Keys keys, SyncLocal syncLocal, SyncRemote syncRemote); AsyncChain maxConflict(Ranges ranges); - @Nonnull IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime); - @Nonnull TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime) throws RequestExecutionException; + @Nonnull + default IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime) + { + return coordinateAsync(minEpoch, txn, consistencyLevel, requestTime, IAccordService.NO_HLC); + } + @Nonnull IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime, long minHlc); + @Nonnull default TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime) + { + return coordinate(minEpoch, txn, consistencyLevel, requestTime, NO_HLC); + } + @Nonnull TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime, long minHlc) throws RequestExecutionException; List executors(); @@ -174,6 +184,11 @@ public interface IAccordService Node node(); + /** + * Ensure Accord's hlc is at least larger than this for anything accepted at this node + */ + void ensureMinHlc(long minHlc); + // Implementation for the NO_OP service that also has what used to be the default implementations // that had to be overridden by the real AccordService anyways class NoOpAccordService implements IAccordService @@ -211,7 +226,7 @@ public interface IAccordService } @Override - public @Nonnull TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull RequestTime requestTime) + public @Nonnull TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull RequestTime requestTime, long minHlc) { throw new UnsupportedOperationException("No accord transaction should be executed when accord.enabled = false in cassandra.yaml"); } @@ -223,7 +238,7 @@ public interface IAccordService } @Override - public @Nonnull IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime) + public @Nonnull IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime, long minHlc) { throw new UnsupportedOperationException("No accord transaction should be executed when accord.enabled = false in cassandra.yaml"); } @@ -337,6 +352,12 @@ public interface IAccordService { return null; } + + @Override + public void ensureMinHlc(long minHlc) + { + + } } class DelegatingAccordService implements IAccordService @@ -386,9 +407,9 @@ public interface IAccordService @Nonnull @Override - public TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime) + public TxnResult coordinate(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime, long minHlc) { - return delegate.coordinate(minEpoch, txn, consistencyLevel, requestTime); + return delegate.coordinate(minEpoch, txn, consistencyLevel, requestTime, minHlc); } @Override @@ -399,9 +420,9 @@ public interface IAccordService @Nonnull @Override - public IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime) + public IAccordResult coordinateAsync(long minEpoch, @Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, RequestTime requestTime, long minHlc) { - return delegate.coordinateAsync(minEpoch, txn, consistencyLevel, requestTime); + return delegate.coordinateAsync(minEpoch, txn, consistencyLevel, requestTime, minHlc); } @Override @@ -512,5 +533,11 @@ public interface IAccordService { return delegate.node(); } + + @Override + public void ensureMinHlc(long minHlc) + { + delegate.ensureMinHlc(minHlc); + } } } diff --git a/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java b/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java index dd57b0e358..78da816fd7 100644 --- a/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java +++ b/src/java/org/apache/cassandra/service/accord/api/AccordAgent.java @@ -359,7 +359,7 @@ public class AccordAgent implements Agent logger.debug("Waiting {} micros for {} to be stale", waitMicros, staleId); AsyncResult.Settable result = AsyncResults.settable(); - node.scheduler().selfRecurring(() -> result.setSuccess(staleId), waitMicros, MICROSECONDS); + node.scheduler().once(() -> result.setSuccess(staleId), waitMicros, MICROSECONDS); return result; } diff --git a/src/java/org/apache/cassandra/service/accord/api/PartitionKey.java b/src/java/org/apache/cassandra/service/accord/api/PartitionKey.java index c98ff3e29a..5301d42446 100644 --- a/src/java/org/apache/cassandra/service/accord/api/PartitionKey.java +++ b/src/java/org/apache/cassandra/service/accord/api/PartitionKey.java @@ -22,6 +22,7 @@ import java.io.IOException; import java.nio.ByteBuffer; import accord.api.Key; +import accord.primitives.Range; import accord.utils.Invariants; import org.apache.cassandra.db.DecoratedKey; import org.apache.cassandra.db.SinglePartitionReadCommand; @@ -33,6 +34,7 @@ import org.apache.cassandra.dht.Token; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.schema.TableId; +import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.ObjectSizes; import org.apache.cassandra.utils.vint.VIntCoding; @@ -95,6 +97,12 @@ public final class PartitionKey extends AccordRoutableKey implements Key return new org.apache.cassandra.service.accord.api.TokenKey(table, token()); } + @Override + public Range asRange() + { + return TokenRange.create(TokenKey.before(table, key.getToken()), new TokenKey(table, key.getToken())); + } + @Override byte sentinel() { diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java index 39d1a0eb20..af1782db2a 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropExecution.java @@ -24,12 +24,14 @@ import java.util.List; import java.util.Set; import java.util.concurrent.ConcurrentHashMap; import java.util.concurrent.atomic.AtomicInteger; +import java.util.concurrent.atomic.AtomicLongFieldUpdater; import java.util.function.BiConsumer; import accord.api.Data; import accord.api.Result; import accord.coordinate.CoordinationAdapter; import accord.coordinate.ExecuteFlag.CoordinationFlags; +import accord.coordinate.Timeout; import accord.local.Node; import accord.local.Node.Id; import accord.local.SequentialAsyncExecutor; @@ -42,11 +44,13 @@ import accord.primitives.FullRoute; import accord.primitives.Keys; import accord.primitives.Participants; import accord.primitives.Timestamp; +import accord.primitives.TimestampWithUniqueHlc; import accord.primitives.Txn; import accord.primitives.TxnId; import accord.topology.Shard; import accord.topology.Topologies; import accord.topology.Topology; +import accord.utils.Invariants; import accord.utils.UnhandledEnum; import accord.utils.async.AsyncChain; import accord.utils.async.AsyncChains; @@ -63,6 +67,8 @@ import org.apache.cassandra.db.partitions.PartitionIterator; import org.apache.cassandra.db.rows.RowIterator; import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; +import org.apache.cassandra.exceptions.ReadFailureException; +import org.apache.cassandra.exceptions.ReadTimeoutException; import org.apache.cassandra.locator.EndpointsForToken; import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.locator.Replica; @@ -75,7 +81,6 @@ import org.apache.cassandra.service.StorageProxy; import org.apache.cassandra.service.accord.AccordEndpointMapper; import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.service.accord.api.TokenKey; -import org.apache.cassandra.service.accord.interop.AccordInteropReadCallback.MaximalCommitSender; import org.apache.cassandra.service.accord.serializers.TableMetadatasAndKeys; import org.apache.cassandra.service.accord.txn.AccordUpdate; import org.apache.cassandra.service.accord.txn.TxnData; @@ -106,8 +111,10 @@ import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.accordWri * on its inputs. * */ -public class AccordInteropExecution implements ReadCoordinator, MaximalCommitSender +public class AccordInteropExecution implements ReadCoordinator { + private static final AtomicLongFieldUpdater UNIQUE_HLC_UPDATER = AtomicLongFieldUpdater.newUpdater(AccordInteropExecution.class, "uniqueHlc"); + private final Node node; private final TxnId txnId; private final Txn txn; @@ -129,6 +136,7 @@ public class AccordInteropExecution implements ReadCoordinator, MaximalCommitSen private final Set contacted; private final AccordUpdate.Kind updateKind; + private volatile long uniqueHlc; public AccordInteropExecution(Node node, TxnId txnId, Txn txn, AccordUpdate.Kind updateKind, FullRoute route, Ballot ballot, Timestamp executeAt, Deps deps, BiConsumer callback, SequentialAsyncExecutor executor, ConsistencyLevel consistencyLevel, AccordEndpointMapper endpointMapper) @@ -371,12 +379,40 @@ public class AccordInteropExecution implements ReadCoordinator, MaximalCommitSen result.begin((data, failure) -> { if (failure == null) + { + long uniqueHlc = this.uniqueHlc; + Timestamp executeAt = this.executeAt; + if (txnId.is(Txn.Kind.Write) && uniqueHlc != 0) + { + Invariants.require(uniqueHlc > executeAt.hlc()); + executeAt = new TimestampWithUniqueHlc(executeAt, uniqueHlc); + } ((CoordinationAdapter)node.coordinationAdapter(txnId, Standard)).persist(node, executor, executes, route, ballot, CoordinationFlags.none(), txnId, txn, executeAt, deps, txnId.is(Write) ? txn.execute(txnId, executeAt, data) : null, txn.result(txnId, executeAt, data), callback); + } else - callback.accept(null, failure); + { + callback.accept(null, maybeWrapRequestFailureException(failure)); + } }); } + /** + * Interop should expose these exceptions as the appropriate Accord types so AccordService + * knows how to handle them + */ + private Throwable maybeWrapRequestFailureException(Throwable failure) + { + Throwable toCheck = failure; + do + { + // TODO (required): There are probably more exceptions that will have this issue of wanting + // to be turned into the top level exception sent back to the client + if (toCheck instanceof ReadTimeoutException || toCheck instanceof ReadFailureException) + return new Timeout(txnId, route.homeKey(), failure); + } while ((toCheck = toCheck.getCause()) != null); + return failure; + } + private AsyncChain executeUnrecoverableRepairUpdate() { return AsyncChains.ofCallable(Stage.ACCORD_MIGRATION.executor(), () -> { @@ -415,9 +451,17 @@ public class AccordInteropExecution implements ReadCoordinator, MaximalCommitSen } // Provide request callbacks with a way to send maximal commits on Insufficient responses - @Override public void sendMaximalCommit(Id to) { Commit.stableMaximal(node, to, txn, txnId, executeAt, route, deps); } + + public void maybeUpdateUniqueHlc(long uniqueHlc) + { + if (txnId.is(Txn.Kind.Write) && uniqueHlc > 0) + { + Invariants.require(uniqueHlc > executeAt.hlc()); + UNIQUE_HLC_UPDATER.accumulateAndGet(this, uniqueHlc, Math::max); + } + } } diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropRead.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropRead.java index 5f474cb22b..4f94cf2cd4 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropRead.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropRead.java @@ -235,9 +235,9 @@ public class AccordInteropRead extends ReadData static class ReadCallback extends AccordInteropReadCallback { - public ReadCallback(Node.Id id, InetAddressAndPort endpoint, Message message, RequestCallback wrapped, MaximalCommitSender maximalCommitSender) + public ReadCallback(Node.Id id, InetAddressAndPort endpoint, Message message, RequestCallback wrapped, AccordInteropExecution interopExecution) { - super(id, endpoint, message, wrapped, maximalCommitSender); + super(id, endpoint, message, wrapped, interopExecution); } @Override diff --git a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropReadCallback.java b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropReadCallback.java index ea9ffabe11..20aee31d95 100644 --- a/src/java/org/apache/cassandra/service/accord/interop/AccordInteropReadCallback.java +++ b/src/java/org/apache/cassandra/service/accord/interop/AccordInteropReadCallback.java @@ -18,8 +18,6 @@ package org.apache.cassandra.service.accord.interop; -import javax.annotation.Nonnull; - import accord.coordinate.Timeout; import accord.local.Node; import accord.messages.Callback; @@ -36,24 +34,19 @@ import static accord.messages.ReadData.CommitOrReadNack.Insufficient; public abstract class AccordInteropReadCallback implements Callback { - interface MaximalCommitSender - { - void sendMaximalCommit(@Nonnull Node.Id to); - } - private final Node.Id id; private final InetAddressAndPort endpoint; private final Message message; private final RequestCallback wrapped; - private final MaximalCommitSender maximalCommitSender; + private final AccordInteropExecution interopExecution; - public AccordInteropReadCallback(Node.Id id, InetAddressAndPort endpoint, Message message, RequestCallback wrapped, MaximalCommitSender maximalCommitSender) + public AccordInteropReadCallback(Node.Id id, InetAddressAndPort endpoint, Message message, RequestCallback wrapped, AccordInteropExecution interopExecution) { this.id = id; this.message = message; this.endpoint = endpoint; this.wrapped = wrapped; - this.maximalCommitSender = maximalCommitSender; + this.interopExecution = interopExecution; } abstract T convertResponse(ReadOk ok); @@ -64,14 +57,16 @@ public abstract class AccordInteropReadCallback implements Callback { - public ReadRepairCallback(Node.Id id, InetAddressAndPort endpoint, Message message, RequestCallback wrapped, MaximalCommitSender maximalCommitSender) + public ReadRepairCallback(Node.Id id, InetAddressAndPort endpoint, Message message, RequestCallback wrapped, AccordInteropExecution interopExecution) { - super(id, endpoint, message, wrapped, maximalCommitSender); + super(id, endpoint, message, wrapped, interopExecution); } @Override diff --git a/src/java/org/apache/cassandra/service/accord/repair/AccordRepair.java b/src/java/org/apache/cassandra/service/accord/repair/AccordRepair.java index 59b5abd4c7..e6020359b8 100644 --- a/src/java/org/apache/cassandra/service/accord/repair/AccordRepair.java +++ b/src/java/org/apache/cassandra/service/accord/repair/AccordRepair.java @@ -25,9 +25,13 @@ import java.util.concurrent.Executor; import java.util.stream.Collectors; import javax.annotation.Nullable; -import accord.local.durability.DurabilityService; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + import accord.local.Node; +import accord.local.durability.DurabilityService; import accord.primitives.Ranges; +import accord.primitives.Timestamp; import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.Keyspace; import org.apache.cassandra.dht.Range; @@ -45,6 +49,7 @@ import org.apache.cassandra.service.accord.TimeOnlyRequestBookkeeping.LatencyReq import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tcm.Epoch; +import org.apache.cassandra.utils.Pair; import org.apache.cassandra.utils.TimeUUID; import org.apache.cassandra.utils.concurrent.AsyncPromise; import org.apache.cassandra.utils.concurrent.Future; @@ -63,6 +68,8 @@ import static org.apache.cassandra.config.DatabaseDescriptor.getAccordRepairTime */ public class AccordRepair { + private static final Logger logger = LoggerFactory.getLogger(AccordRepair.class); + private final SharedContext ctx; private final ColumnFamilyStore cfs; private final TimeUUID repairId; @@ -92,17 +99,34 @@ public class AccordRepair return minEpoch; } - public Ranges repair() throws Throwable + public static class AccordRepairResult { - List repairedRanges = new ArrayList<>(); - for (accord.primitives.Range range : ranges) - repairedRanges.addAll(repairRange((TokenRange)range)); - return Ranges.of(repairedRanges.toArray(new accord.primitives.Range[0])); + public final Ranges repairedRanges; + public final long maxHlc; + + public AccordRepairResult(Ranges ranges, long maxHlc) + { + this.repairedRanges = ranges; + this.maxHlc = maxHlc; + } } - public Future repair(Executor executor) + public AccordRepairResult repair() throws Throwable { - AsyncPromise future = new AsyncPromise<>(); + List repairedRanges = new ArrayList<>(); + long maxHLC = Long.MIN_VALUE; + for (accord.primitives.Range range : ranges) + { + Pair, Long> rangesAndMaxHLC = repairRange((TokenRange) range); + repairedRanges.addAll(rangesAndMaxHLC.left); + maxHLC = Math.max(maxHLC, rangesAndMaxHLC.right); + } + return new AccordRepairResult(Ranges.of(repairedRanges.toArray(new accord.primitives.Range[0])), maxHLC); + } + + public Future repair(Executor executor) + { + AsyncPromise future = new AsyncPromise<>(); executor.execute(() -> { try { @@ -124,7 +148,7 @@ public class AccordRepair thread.interrupt(); } - private List repairRange(TokenRange range) throws Throwable + private Pair, Long> repairRange(TokenRange range) throws Throwable { List repairedRanges = new ArrayList<>(); List ids = endpoints == null ? null : endpoints.stream().map(AccordService.instance().configService()::mappedId).collect(Collectors.toList()); @@ -151,9 +175,9 @@ public class AccordRepair waiting = Thread.currentThread(); RequestBookkeeping bookkeeping = new LatencyRequestBookkeeping(latency); long timeoutNanos = getAccordRepairTimeoutNanos(); - AccordService.getBlocking(service.maxConflict(ranges).flatMap(conflict -> { - conflict = mergeMax(conflict, minForEpoch(this.minEpoch.getEpoch())); - return service.sync("[repairId #" + repairId + ']', conflict, Ranges.of(range), ids, NoLocal, syncRemote, timeoutNanos, NANOSECONDS); + long maxHlc = AccordService.getBlocking(service.maxConflict(ranges).flatMap(conflict -> { + Timestamp conflictMax = mergeMax(conflict, minForEpoch(this.minEpoch.getEpoch())); + return service.sync("[repairId #" + repairId + ']', conflictMax, Ranges.of(range), ids, NoLocal, syncRemote, timeoutNanos, NANOSECONDS).map(ignored -> conflictMax.hlc()); }), ranges, bookkeeping, start, start + timeoutNanos); waiting = null; @@ -162,6 +186,8 @@ public class AccordRepair for (accord.primitives.Range r : ranges) repairedRanges.add(r); + + return Pair.create(repairedRanges, maxHlc); } catch (Throwable t) { @@ -178,7 +204,5 @@ public class AccordRepair long end = ctx.clock().nanoTime(); cfs.metric.accordRepair.addNano(end - start); } - - return repairedRanges; } } diff --git a/src/java/org/apache/cassandra/service/accord/txn/AccordUpdate.java b/src/java/org/apache/cassandra/service/accord/txn/AccordUpdate.java index 3a9d1bbe7b..453ae24cc9 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/AccordUpdate.java +++ b/src/java/org/apache/cassandra/service/accord/txn/AccordUpdate.java @@ -67,6 +67,11 @@ public abstract class AccordUpdate implements Update return ((AccordUpdate)update).kind(); } + public void failCondition() + { + + } + public boolean checkCondition(Data data) { throw new UnsupportedOperationException(); diff --git a/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java b/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java index efb222b655..2e5baf296b 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java +++ b/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java @@ -85,10 +85,10 @@ public class AccordUpdateParameters timestamp, MICROSECONDS.toSeconds(timestamp), ttl, - prefetchRow(metadata, dk, rowIndex)); + prefetchRow(dk, rowIndex)); } - private Map prefetchRow(TableMetadata metadata, DecoratedKey dk, int index) + private Map prefetchRow(DecoratedKey dk, int index) { if (data != null) { diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java b/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java index 87a37a355a..17f69aefae 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java @@ -59,10 +59,10 @@ import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.AccordExecutor; import org.apache.cassandra.service.accord.TokenRange; -import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.service.accord.api.PartitionKey; import org.apache.cassandra.service.accord.serializers.TableMetadatas; import org.apache.cassandra.service.accord.serializers.TableMetadatasAndKeys; +import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.service.accord.serializers.Version; import org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind; import org.apache.cassandra.utils.ByteBufferUtil; @@ -230,7 +230,7 @@ public class TxnNamedRead extends AbstractParameterisedVersionedSerialized range) { + if (!IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS.contains(consistencyLevel)) + throw new UnsupportedOperationException("Consistency level " + consistencyLevel + " is unsupported with Accord for read, supported are ONE, QUORUM, and SERIAL"); + + checkState(range.unwrap().size() == 1); if (ignoresSuppliedReadCL()) { TableMigrationState tms = cm.consensusMigrationState.tableStates.get(tableId); @@ -199,12 +227,40 @@ public enum TransactionalMode // otherwise honor it because we might read through Accord for non-SERIAL reads before repair is run // this is OK to do because BRR still works and Accord isn't computing a write so recovery // determinism isn't an issue - if (tms == null || range.intersects(tms.migratedRangesAsPartitionPosition())) + if (tms == null) return null; + + // Fully migrated, Paxos key repair only repairs at QUORUM so need full migration to ignore read CL + for (Range migrated : tms.migratedRangesAsPartitionPosition()) + { + if (migrated.contains(range.left) && migrated.contains(range.right)) + return null; + } + + } + else if (!accordIsEnabled) + { + TableMigrationState tms = cm.consensusMigrationState.tableStates.get(tableId); + checkState(tms != null || fromMode == TransactionalMigrationFromMode.none); + + // Fully migrated + String illegalStateMsg = "Attempting to route transaction to Accord when there is no migration away for this range (" + range + ") and the current transactional mode (" + this + ") doesn't enable Accord"; + if (tms == null) + throw new IllegalStateException(illegalStateMsg); + + if (ConsensusRequestRouter.instance.getClass() == ConsensusRequestRouter.class) + { + // Migrated or migrating, should not be running on Accord + for (Range migratingAndMigrated : tms.migratingAndMigratedRangesAsPartitionPosition()) + { + if (migratingAndMigrated.contains(range.left) && migratingAndMigrated.contains(range.right)) + throw new IllegalStateException(illegalStateMsg); + } + } + + return fromMode.from.ignoresSuppliedReadCL() ? null : consistencyLevel; } - if (!IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS.contains(consistencyLevel)) - throw new UnsupportedOperationException("Consistency level " + consistencyLevel + " is unsupported with Accord for read, supported are ONE, QUORUM, and SERIAL"); return consistencyLevel; } diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusKeyMigrationState.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusKeyMigrationState.java index 4078b379b2..196c250e04 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusKeyMigrationState.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusKeyMigrationState.java @@ -33,6 +33,8 @@ import org.slf4j.Logger; import org.slf4j.LoggerFactory; import accord.primitives.Keys; +import accord.primitives.Range; +import accord.primitives.Ranges; import accord.primitives.Timestamp; import com.github.benmanes.caffeine.cache.CacheLoader; import com.github.benmanes.caffeine.cache.Caffeine; @@ -176,14 +178,14 @@ public abstract class ConsensusKeyMigrationState * This will trigger a distributed migration for the key, but will only block on local completion * so Paxos reads can return a result as soon as the local state is ready */ - public void maybePerformAccordToPaxosKeyMigration(boolean isForWrite) + public long maybePerformAccordToPaxosKeyMigration(boolean isForWrite) { if (paxosReadSatisfiedByKeyMigration()) - return; + return IAccordService.NO_HLC; // TODO (desired): Better query start time TableMigrationState tms = tableMigrationState; - repairKeyAccord(key, tms.tableId, tms.minMigrationEpoch(key.getToken()).getEpoch(), Dispatcher.RequestTime.forImmediateExecution(), false, isForWrite); + return repairKeyAccord(key, tms.tableId, tms.minMigrationEpoch(key.getToken()).getEpoch(), Dispatcher.RequestTime.forImmediateExecution(), false, isForWrite); } boolean paxosReadSatisfiedByKeyMigration() @@ -197,7 +199,7 @@ public abstract class ConsensusKeyMigrationState } private static final int EMPTY_KEY_SIZE = Ints.checkedCast(ObjectSizes.measureDeep(Pair.create(null, UUID.randomUUID()))); - private static final int VALUE_SIZE = Ints.checkedCast(ObjectSizes.measureDeep(new ConsensusMigratedAt(Epoch.EMPTY, ConsensusMigrationTarget.accord))); + private static final int VALUE_SIZE = Ints.checkedCast(ObjectSizes.measureDeep(new ConsensusMigratedAt(Epoch.EMPTY, IAccordService.NO_HLC, ConsensusMigrationTarget.accord))); private static final CacheLoader, ConsensusMigratedAt> LOADING_FUNCTION = k -> SystemKeyspace.loadConsensusKeyMigrationState(k.left, k.right); private static final Weigher, ConsensusMigratedAt> WEIGHER_FUNCTION = (k, v) -> EMPTY_KEY_SIZE + Ints.checkedCast(ByteBufferUtil.estimatedSizeOnHeap(k.left)) + VALUE_SIZE; @@ -214,7 +216,9 @@ public abstract class ConsensusKeyMigrationState saveConsensusKeyMigrationLocally(message.payload.partitionKey, message.payload.tableId, message.payload.consensusMigratedAt); }; - private ConsensusKeyMigrationState() {} + private ConsensusKeyMigrationState() + { + } @VisibleForTesting public static void reset() @@ -222,8 +226,10 @@ public abstract class ConsensusKeyMigrationState MIGRATION_STATE_CACHE.invalidateAll(); } - public static void maybeSaveAccordKeyMigrationLocally(PartitionKey partitionKey, Epoch epoch) + public static void maybeSaveAccordKeyMigrationLocally(PartitionKey partitionKey, Epoch epoch, long maxHLC) { + if (maxHLC == IAccordService.NO_HLC) + return; TableId tableId = partitionKey.table(); UUID tableUUID = tableId.asUUID(); DecoratedKey dk = partitionKey.partitionKey(); @@ -233,7 +239,7 @@ public abstract class ConsensusKeyMigrationState if (tms == null) return; - ConsensusMigratedAt migratedAt = new ConsensusMigratedAt(epoch, paxos); + ConsensusMigratedAt migratedAt = new ConsensusMigratedAt(epoch, maxHLC, paxos); if (!tms.paxosReadSatisfiedByKeyMigrationAtEpoch(dk, migratedAt)) return; @@ -242,7 +248,11 @@ public abstract class ConsensusKeyMigrationState public static KeyMigrationState getKeyMigrationState(TableId tableId, DecoratedKey key) { - ClusterMetadata cm = ClusterMetadata.current(); + return getKeyMigrationState(ClusterMetadata.current(), tableId, key); + } + + public static KeyMigrationState getKeyMigrationState(ClusterMetadata cm, TableId tableId, DecoratedKey key) + { TableMigrationState tms = cm.consensusMigrationState.tableStates.get(tableId); // No state means no migration for this table if (tms == null) @@ -277,17 +287,17 @@ public abstract class ConsensusKeyMigrationState /* * Trigger a distributed repair of Accord state for this key. */ - static void repairKeyAccord(DecoratedKey key, + static long repairKeyAccord(DecoratedKey key, TableId tableId, long minEpoch, Dispatcher.RequestTime requestTime, boolean global, boolean isForWrite) { - repairKeysAccord(ImmutableList.of(key), tableId, minEpoch, requestTime, global, isForWrite); + return repairKeysAccord(ImmutableList.of(key), tableId, minEpoch, requestTime, global, isForWrite); } - static void repairKeysAccord(List keys, + static long repairKeysAccord(List keys, TableId tableId, long minEpoch, Dispatcher.RequestTime requestTime, @@ -311,10 +321,17 @@ public abstract class ConsensusKeyMigrationState RequestBookkeeping bookkeeping = new LatencyRequestBookkeeping(cfs == null ? null : cfs.metric.keyMigration); AccordService.getBlocking(accord.sync(Timestamp.minForEpoch(minEpoch), partitionKeys, Self, global ? Quorum : NoRemote), partitionKeys, bookkeeping, start, deadline); - maybeSaveAccordKeyMigrationLocally((PartitionKey) partitionKeys.get(0), Epoch.create(minEpoch)); + Range[] asRanges = new Range[partitionKeys.size()]; + for (int i = 0; i < partitionKeys.size(); i++) + asRanges[i] = partitionKeys.get(i).asRange(); + Ranges ranges = Ranges.of(asRanges); + RequestBookkeeping maxConflictBookkeeping = new LatencyRequestBookkeeping(cfs == null ? null : cfs.keyspace.metric.accordGetMaxConflicts); + long maxHlc = AccordService.getBlocking(AccordService.instance().maxConflict(ranges), null, ranges, maxConflictBookkeeping, start, deadline, false).hlc(); + maybeSaveAccordKeyMigrationLocally((PartitionKey) partitionKeys.get(0), Epoch.create(minEpoch), maxHlc); + return maxHlc; } - static void repairKeyPaxos(EndpointsForToken naturalReplicas, + static long repairKeyPaxos(EndpointsForToken naturalReplicas, Epoch currentEpoch, DecoratedKey key, ColumnFamilyStore cfs, @@ -347,8 +364,8 @@ public abstract class ConsensusKeyMigrationState saveConsensusKeyMigration(naturalReplicas, new ConsensusKeyMigrationFinished(tableMetadata.id.asUUID(), key.getKey(), - new ConsensusMigratedAt(currentEpoch, ConsensusMigrationTarget.accord))); - return; + new ConsensusMigratedAt(currentEpoch, repair.maxHlc(), ConsensusMigrationTarget.accord))); + return repair.maxHlc(); case FAILURE: Failure failure = (Failure)result; if (failure.failure == null) diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigratedAt.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigratedAt.java index bf3e798466..cc64f3f582 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigratedAt.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigratedAt.java @@ -36,6 +36,7 @@ public class ConsensusMigratedAt public void serialize(ConsensusMigratedAt t, DataOutputPlus out) throws IOException { Epoch.serializer.serialize(t.migratedAtEpoch, out); + out.writeUnsignedVInt(t.maxHLC); out.writeByte(t.migratedAtTarget.value); } @@ -43,15 +44,17 @@ public class ConsensusMigratedAt public ConsensusMigratedAt deserialize(DataInputPlus in) throws IOException { Epoch migratedAtEpoch = Epoch.serializer.deserialize(in); + long maxHLC = in.readUnsignedVInt(); ConsensusMigrationTarget target = ConsensusMigrationTarget.fromValue(in.readByte()); - return new ConsensusMigratedAt(migratedAtEpoch, target); + return new ConsensusMigratedAt(migratedAtEpoch, maxHLC, target); } @Override public long serializedSize(ConsensusMigratedAt t) { return TypeSizes.sizeof(ConsensusMigrationTarget.accord.value) - + Epoch.serializer.serializedSize(t.migratedAtEpoch); + + Epoch.serializer.serializedSize(t.migratedAtEpoch) + + TypeSizes.sizeofUnsignedVInt(t.maxHLC); } }); @@ -59,12 +62,15 @@ public class ConsensusMigratedAt @Nullable public final Epoch migratedAtEpoch; + public final long maxHLC; + @Nullable public final ConsensusMigrationTarget migratedAtTarget; - public ConsensusMigratedAt(Epoch migratedAtEpoch, ConsensusMigrationTarget migratedAtTarget) + public ConsensusMigratedAt(Epoch migratedAtEpoch, long maxHLC, ConsensusMigrationTarget migratedAtTarget) { this.migratedAtEpoch = migratedAtEpoch; + this.maxHLC = maxHLC; this.migratedAtTarget = migratedAtTarget; } } diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationRepairResult.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationRepairResult.java index 551ee1f1c7..2efe2a6686 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationRepairResult.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationRepairResult.java @@ -21,6 +21,8 @@ package org.apache.cassandra.service.consensus.migration; import javax.annotation.Nullable; import accord.primitives.Ranges; +import org.apache.cassandra.service.accord.IAccordService; +import org.apache.cassandra.service.accord.repair.AccordRepair.AccordRepairResult; import org.apache.cassandra.tcm.Epoch; import static com.google.common.base.Preconditions.checkArgument; @@ -31,19 +33,25 @@ public class ConsensusMigrationRepairResult public final ConsensusMigrationRepairType type; public final Epoch minEpoch; @Nullable - public final Ranges barrieredRanges; + public final Ranges accordRepairedRanges; - private ConsensusMigrationRepairResult(ConsensusMigrationRepairType type, Epoch minEpoch, @Nullable Ranges barrieredRanges) + private ConsensusMigrationRepairResult(ConsensusMigrationRepairType type, Epoch minEpoch, @Nullable Ranges accordRepairedRanges) { this.type = type; this.minEpoch = minEpoch; - this.barrieredRanges = barrieredRanges; + this.accordRepairedRanges = accordRepairedRanges; } - public static ConsensusMigrationRepairResult fromRepair(Epoch minEpoch, Ranges barrieredRanges, boolean dataRepaired, boolean paxosRepaired, boolean accordRepaired, boolean deadNodesExcluded) + public static ConsensusMigrationRepairResult fromRepair(Epoch minEpoch, AccordRepairResult accordRepairResult, boolean dataRepaired, boolean paxosRepaired, boolean accordRepaired, boolean deadNodesExcluded, boolean incremental) { checkArgument(!accordRepaired || minEpoch.isAfter(Epoch.EMPTY), "Epoch should not be empty if Accord repairs was performed"); if (deadNodesExcluded) return INELIGIBLE; - return new ConsensusMigrationRepairResult(new ConsensusMigrationRepairType(dataRepaired, paxosRepaired, accordRepaired), minEpoch, barrieredRanges); + boolean eligibleAccordRepair = accordRepaired && accordRepairResult != null && accordRepairResult.maxHlc != IAccordService.NO_HLC; + Ranges accordRepairedRanges = eligibleAccordRepair ? accordRepairResult.repairedRanges : null; + // Incremental repair won't flush after Paxos repair (which is at QUORUM) and then be picked up in the incremental repair + // and thus won't be repaired at ALL which is what Accord needs + if (incremental) + paxosRepaired = false; + return new ConsensusMigrationRepairResult(new ConsensusMigrationRepairType(dataRepaired, paxosRepaired, eligibleAccordRepair), minEpoch, accordRepairedRanges); } } diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java index c508fbc279..2988b8563e 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java @@ -56,6 +56,7 @@ import org.apache.cassandra.schema.SchemaConstants; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.schema.TableMetadata; import org.apache.cassandra.schema.TableParams; +import org.apache.cassandra.service.accord.IAccordService; import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.KeyMigrationState; @@ -71,9 +72,9 @@ import static com.google.common.base.Preconditions.checkState; import static org.apache.cassandra.dht.Range.compareRightToken; import static org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.getConsensusMigratedAt; import static org.apache.cassandra.service.consensus.migration.ConsensusMigrationTarget.paxos; -import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision.accord; -import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision.paxosV1; -import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision.paxosV2; +import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingTarget.accord; +import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingTarget.paxosV1; +import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingTarget.paxosV2; /** * Helper class to decide where to route a request that requires consensus, migrating a key if necessary @@ -85,13 +86,52 @@ import static org.apache.cassandra.service.consensus.migration.ConsensusRequestR */ public class ConsensusRequestRouter { - public enum ConsensusRoutingDecision + public enum ConsensusRoutingTarget { paxosV1, paxosV2, accord, } + public static class ConsensusRoutingDecision + { + public static final ConsensusRoutingDecision PAXOSV1 = new ConsensusRoutingDecision(paxosV1, IAccordService.NO_HLC); + public static final ConsensusRoutingDecision PAXOSV2 = new ConsensusRoutingDecision(paxosV2, IAccordService.NO_HLC); + public static final ConsensusRoutingDecision ACCORD = new ConsensusRoutingDecision(accord, IAccordService.NO_HLC); + + public final ConsensusRoutingTarget target; + public final long minHLC; + public ConsensusRoutingDecision(ConsensusRoutingTarget target, long maxHLC) + { + this.target = target; + this.minHLC = maxHLC + 1; + } + + public static final ConsensusRoutingDecision decisionForTarget(ConsensusRoutingTarget target) + { + switch (target) + { + default: + throw new IllegalArgumentException("Unhandled target " + target); + case paxosV1: + return PAXOSV1; + case paxosV2: + return PAXOSV2; + case accord: + return ACCORD; + } + } + + @Override + public String toString() + { + return "ConsensusRoutingDecision{" + + "target=" + target + + ", minHLC=" + minHLC + + '}'; + } + } + public static volatile ConsensusRequestRouter instance = new ConsensusRequestRouter(); @VisibleForTesting @@ -111,7 +151,7 @@ public class ConsensusRequestRouter ConsensusRoutingDecision decisionFor(TransactionalMode transactionalMode) { if (transactionalMode.accordIsEnabled) - return accord; + return ConsensusRoutingDecision.decisionForTarget(accord); return pickPaxos(); } @@ -144,17 +184,7 @@ public class ConsensusRequestRouter return tbm; } - public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull ClusterMetadata cm, @Nonnull DecoratedKey key, @Nonnull String keyspace, @Nonnull String table, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) - { - TableMetadata metadata = metadata(cm, keyspace, table); - - // Non-distributed tables always take the Paxos path - if (metadata == null) - return pickPaxos(); - return routeAndMaybeMigrate(cm, metadata, key, consistencyLevel, requestTime, timeoutNanos, isForWrite); - } - - public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull ClusterMetadata cm, @Nonnull DecoratedKey key, @Nonnull TableId tableId, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) + public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull ClusterMetadata cm, @Nonnull DecoratedKey key, @Nonnull TableId tableId, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) { TableMetadata metadata = getTableMetadata(cm, tableId); // Non-distributed tables always take the Paxos path @@ -181,7 +211,6 @@ public class ConsensusRequestRouter protected ConsensusRoutingDecision routeAndMaybeMigrate(ClusterMetadata cm, @Nonnull TableMetadata tmd, @Nonnull DecoratedKey key, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) { - if (!tmd.params.transactionalMigrationFrom.isMigrating()) return decisionFor(tmd.params.transactionalMode); @@ -191,7 +220,7 @@ public class ConsensusRequestRouter Token token = key.getToken(); if (tms.migratedRanges.intersects(token)) - return pickMigrated(tms.targetProtocol); + return pickMigrated(tms.targetProtocol, IAccordService.NO_HLC); if (tms.migratingRanges.intersects(token)) return pickBasedOnKeyMigrationStatus(cm, tmd, tms, key, consistencyLevel, requestTime, timeoutNanos, isForWrite); @@ -206,7 +235,7 @@ public class ConsensusRequestRouter */ private static ConsensusRoutingDecision pickBasedOnKeyMigrationStatus(ClusterMetadata cm, TableMetadata tmd, TableMigrationState tms, DecoratedKey key, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) { - checkState(pickPaxos() != paxosV1, "Can't migrate from PaxosV1 to anything"); + checkState(pickPaxos() != ConsensusRoutingDecision.PAXOSV1, "Can't migrate from PaxosV1 to anything"); ColumnFamilyStore cfs = ColumnFamilyStore.getIfExists(tmd.id); if (cfs == null) @@ -231,7 +260,7 @@ public class ConsensusRequestRouter // This ends up still being safe because every single Paxos read (in a migrating range) during migration will check // locally to see if repair is necessary if (consensusMigratedAt != null && tms.satisfiedByKeyMigrationAtEpoch(key, consensusMigratedAt)) - return pickMigrated(tms.targetProtocol); + return pickMigrated(tms.targetProtocol, consensusMigratedAt.maxHLC); if (tms.targetProtocol == paxos) { @@ -239,8 +268,8 @@ public class ConsensusRequestRouter // barrier transactions to accomplish the migration // They still might need to go through the fast local path for barrier txns // at each replica, but they won't create their own txn since we created it here - ConsensusKeyMigrationState.repairKeyAccord(key, tms.tableId, tms.minMigrationEpoch(token).getEpoch(), requestTime, true, isForWrite); - return paxosV2; + long maxHLC = ConsensusKeyMigrationState.repairKeyAccord(key, tms.tableId, tms.minMigrationEpoch(token).getEpoch(), requestTime, true, isForWrite); + return pickPaxos(maxHLC); } // Fall through for repairKeyPaxos } @@ -249,20 +278,20 @@ public class ConsensusRequestRouter // Accord -> Paxos - Paxos will ask Accord to migrate in the read at each replica if necessary // Paxos -> Accord - Paxos needs to be repaired before Accord runs so do it here if (tms.targetProtocol == paxos) - // TODO (important): Why are these two cases paxosV2 instead of `pickPaxos`? - // Because we only supported PaxosV2 for migration? - // Eventually we want to support both so just use pickPaxos and error out on migration from paxosV1 elsewhere? - return paxosV2; + return pickPaxos(); else { if (tms.accordSafeToReadRanges.intersects(key.getToken())) + { // Should exit exceptionally if the repair is not done - ConsensusKeyMigrationState.repairKeyPaxos(naturalReplicas, cm.epoch, key, cfs, consistencyLevel, requestTime, timeoutNanos, isLocallyReplicated, isForWrite); + long maxHLC = ConsensusKeyMigrationState.repairKeyPaxos(naturalReplicas, cm.epoch, key, cfs, consistencyLevel, requestTime, timeoutNanos, isLocallyReplicated, isForWrite); + return new ConsensusRoutingDecision(accord, maxHLC); + } else + { return pickPaxos(); + } } - - return pickMigrated(tms.targetProtocol); } // Allows tests to inject specific responses @@ -489,12 +518,12 @@ public class ConsensusRequestRouter return Kind.EphemeralRead; } - private static ConsensusRoutingDecision pickMigrated(ConsensusMigrationTarget targetProtocol) + private static ConsensusRoutingDecision pickMigrated(ConsensusMigrationTarget targetProtocol, long maxHlc) { if (targetProtocol.equals(ConsensusMigrationTarget.accord)) - return accord; + return new ConsensusRoutingDecision(accord, maxHlc); else - return pickPaxos(); + return pickPaxos(maxHlc); } private static ConsensusRoutingDecision pickNotMigrated(ConsensusMigrationTarget targetProtocol) @@ -502,15 +531,20 @@ public class ConsensusRequestRouter if (targetProtocol.equals(ConsensusMigrationTarget.accord)) return pickPaxos(); else - return accord; + return ConsensusRoutingDecision.decisionForTarget(accord); } private static ConsensusRoutingDecision pickPaxos() { - return Paxos.useV2() ? paxosV2 : paxosV1; + return ConsensusRoutingDecision.decisionForTarget(Paxos.useV2() ? paxosV2 : paxosV1); } - public static void validateSafeToReadNonTransactionally(ReadCommand command) + private static ConsensusRoutingDecision pickPaxos(long maxHLC) + { + return new ConsensusRoutingDecision(Paxos.useV2() ? paxosV2 : paxosV1, maxHLC); + } + + public static void validateSafeToReadNonTransactionally(ReadCommand command, @Nullable ClusterMetadata cm) { if (command.potentialTxnConflicts().allowed) return; @@ -524,7 +558,7 @@ public class ConsensusRequestRouter if (Schema.instance.localKeyspaces().containsKeyspace(keyspace)) return; - ClusterMetadata cm = ClusterMetadata.current(); + cm = cm == null ? ClusterMetadata.current() : cm; TableId tableId = command.metadata().id; TableMetadata tableMetadata = getTableMetadata(cm, tableId); // Null for local or dropped tables diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java index 00ac2589a3..07b64aa57f 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java @@ -121,8 +121,9 @@ public abstract class ConsensusTableMigration TableMetadata tm = Schema.instance.getTableMetadata(desc.keyspace, desc.columnFamily); if (tm == null) return; + TableMigrationState tms = ClusterMetadata.current().consensusMigrationState.tableStates.get(tm.id); - if (tms == null || !Range.intersects(tms.migratingRanges, desc.ranges)) + if (tms == null || tms.normalizedIntersectionWithMigratingAtEpoch(repairResult.consensusMigrationRepairResult.minEpoch, desc.ranges).isEmpty()) return; if (!tms.targetProtocol.isMigratedBy(repairResult.consensusMigrationRepairResult.type)) @@ -134,18 +135,18 @@ public abstract class ConsensusTableMigration // repaired in the migrated and Accord managed ranges paxosRepairedRanges = normalizedRanges(desc.ranges); - NormalizedRanges accordBarrieredRanges = NormalizedRanges.empty(); + NormalizedRanges accordRepairedRanges = NormalizedRanges.empty(); if (repairType.migrationToPaxosEligible()) // Accord only barriers ranges it thinks it manages and repair collects which it barriered // precisely which doesn't have to match what the entire repair covers - accordBarrieredRanges = normalizedRanges(migrationResult.barrieredRanges.stream() + accordRepairedRanges = normalizedRanges(migrationResult.accordRepairedRanges.stream() .map(range -> ((TokenRange)range).toKeyspaceRange()) .collect(toImmutableList())); - accordBarrieredRanges = normalizedRanges(accordBarrieredRanges); + accordRepairedRanges = normalizedRanges(accordRepairedRanges); ClusterMetadataService.instance().commit( new MaybeFinishConsensusMigrationForTableAndRange( - desc.keyspace, desc.columnFamily, paxosRepairedRanges, accordBarrieredRanges, + desc.keyspace, desc.columnFamily, paxosRepairedRanges, accordRepairedRanges, migrationResult.minEpoch, repairType.repairedData, repairType.repairedPaxos, repairType.repairedAccord)); } @@ -370,8 +371,8 @@ public abstract class ConsensusTableMigration private static RepairOption getRepairOption(Collection tables, List> intersectingRanges, boolean repairData, boolean repairPaxos, boolean repairAccord) { boolean primaryRange = false; - // TODO (review): Should disabling incremental repair be exposed for the Paxos repair in case someone explicitly does not do incremental repair? - boolean incremental = repairData; + // No need for incremental if data isn't repaired, paxos requires non-incremental for the Paxos repair to propagate at all + boolean incremental = repairData && !repairPaxos; boolean trace = false; int numJobThreads = 1; boolean pullRepair = false; diff --git a/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java b/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java index 7fb435b2b2..01fb936565 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java @@ -119,6 +119,7 @@ public class TableMigrationState */ @Nonnull public final NormalizedRanges repairPendingRanges; + public final NormalizedRanges repairedRanges; /** * Ranges that are migrating could be in either phase when migrating to Accord. Paxos only has one phase. @@ -154,6 +155,7 @@ public class TableMigrationState .map(entry -> new AbstractMap.SimpleEntry<>(entry.getKey(), normalizedRanges(entry.getValue()))) .collect(Collectors.toList())); this.migratingRanges = normalizedRanges(migratingRangesByEpoch.values().stream().flatMap(Collection::stream).collect(Collectors.toList())); + this.repairedRanges = migratingRanges.subtract(this.repairPendingRanges); this.migratingAndMigratedRanges = normalizedRanges(ImmutableList.>builder().addAll(migratedRanges).addAll(migratingRanges).build()); this.accordSafeToReadRanges = !repairPendingRanges.isEmpty() ? migratingAndMigratedRanges.subtract(this.repairPendingRanges) : migratingAndMigratedRanges; } @@ -247,6 +249,17 @@ public class TableMigrationState return this; } + public NormalizedRanges normalizedIntersectionWithMigratingAtEpoch(Epoch epoch, Collection> ranges) + { + // This should be inclusive because the epoch we store in the map is the epoch in which the range has been marked migrating + // in startMigrationToConsensusProtocol + NormalizedRanges normalizedRepairedRanges = normalizedRanges(ranges); + NavigableMap> coveredEpochs = migratingRangesByEpoch.headMap(epoch, true); + NormalizedRanges normalizedMigratingRanges = normalizedRanges(coveredEpochs.values().stream().flatMap(Collection::stream).collect(Collectors.toList())); + // These are the ranges that are impacted by this repair based on the epoch filtering needed to make sure this repair applies to this migration + return normalizedRepairedRanges.intersection(normalizedMigratingRanges); + } + public TableMigrationState withRangesRepairedAtEpoch(@Nonnull Collection> ranges, @Nonnull Epoch epoch, @Nonnull ConsensusMigrationRepairType repairType) @@ -431,6 +444,11 @@ public class TableMigrationState return Iterables.transform(migratedRanges, range -> new Range<>(range.left.maxKeyBound(), range.right.maxKeyBound())); } + public Iterable> migratingAndMigratedRangesAsPartitionPosition() + { + return Iterables.transform(migratingAndMigratedRanges, range -> new Range<>(range.left.maxKeyBound(), range.right.maxKeyBound())); + } + @Override public String toString() { diff --git a/src/java/org/apache/cassandra/service/paxos/Paxos.java b/src/java/org/apache/cassandra/service/paxos/Paxos.java index 24a0ee7317..57b6da8d77 100644 --- a/src/java/org/apache/cassandra/service/paxos/Paxos.java +++ b/src/java/org/apache/cassandra/service/paxos/Paxos.java @@ -107,7 +107,6 @@ import org.apache.cassandra.utils.CassandraVersion; import org.apache.cassandra.utils.FBUtilities; import org.apache.cassandra.utils.NoSpamLogger; -import static com.google.common.base.Preconditions.checkState; import static java.util.Collections.emptyMap; import static java.util.concurrent.TimeUnit.NANOSECONDS; import static java.util.concurrent.TimeUnit.SECONDS; @@ -709,7 +708,8 @@ public class Paxos ConsistencyLevel consistencyForConsensus, ConsistencyLevel consistencyForCommit, ClientState clientState, - Dispatcher.RequestTime requestTime) + Dispatcher.RequestTime requestTime, + long minHLC) throws UnavailableException, IsBootstrappingException, RequestFailureException, RequestTimeoutException, InvalidRequestException { final long proposeDeadline = requestTime.startedAtNanos() + getCasContentionTimeout(NANOSECONDS); @@ -730,7 +730,7 @@ public class Paxos Tracing.trace("Reading existing values for CAS precondition"); BeginResult begin = begin(proposeDeadline, readCommand, consistencyForConsensus, - true, null, failedAttemptsDueToContention, request.requestTime()); + true, Ballot.atUnixMicrosWithLsb(minHLC, 0, GLOBAL), failedAttemptsDueToContention, requestTime); if (begin.retryWithNewConsenusProtocol) { @@ -806,11 +806,10 @@ public class Paxos continue; } - PaxosPropose.Status propose = propose(proposal, participants, conditionMet, false).awaitUntil(proposeDeadline); + PaxosPropose.Status propose = propose(proposal, participants, conditionMet).awaitUntil(proposeDeadline); switch (propose.outcome) { default: throw new IllegalStateException(); - case MAYBE_FAILURE: throw propose.maybeFailure().markAndThrowAsTimeoutOrFailure(true, consistencyForConsensus, failedAttemptsDueToContention); @@ -843,12 +842,6 @@ public class Paxos .markAndThrowAsTimeoutOrFailure(true, consistencyForConsensus, failedAttemptsDueToContention); case NO: - // Shouldn't retry on this protocol - if (superseded.needsConsensusMigration) - { - casWriteMetrics.acceptMigrationRejects.mark(); - return RETRY_NEW_PROTOCOL; - } // We have been superseded without our proposal being accepted by anyone, so we can safely retry Tracing.trace("Paxos proposal not accepted (pre-empted by a higher ballot)"); if (!waitForContention(proposeDeadline, ++failedAttemptsDueToContention, metadata, partitionKey, consistencyForConsensus, WRITE)) @@ -892,7 +885,7 @@ public class Paxos return read.rowIterator(false); } - public static ConsensusAttemptResult read(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyForConsensus, Dispatcher.RequestTime requestTime) + public static ConsensusAttemptResult read(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyForConsensus, Dispatcher.RequestTime requestTime, long minHLC) throws InvalidRequestException, UnavailableException, ReadFailureException, ReadTimeoutException { long start = nanoTime(); @@ -901,8 +894,10 @@ public class Paxos long deadline = requestTime.computeDeadline(DatabaseDescriptor.getReadRpcTimeout(NANOSECONDS)); int failedAttemptsDueToContention = 0; - Ballot minimumBallot = null; + Ballot minimumBallot = Ballot.atUnixMicrosWithLsb(minHLC, 0, GLOBAL); SinglePartitionReadCommand read = group.queries.get(0); + // Allow potential txn conflicts since Paxos will manage them + read = read.withTransactionalSettings(false, read.nowInSec()); try (PaxosOperationLock lock = PaxosState.lock(read.partitionKey(), read.metadata(), deadline, consistencyForConsensus, false)) { while (true) @@ -934,11 +929,10 @@ public class Paxos } Proposal proposal = Proposal.empty(begin.ballot, read.partitionKey(), read.metadata()); - PaxosPropose.Status propose = propose(proposal, begin.participants, true, false).awaitUntil(deadline); + PaxosPropose.Status propose = propose(proposal, begin.participants, true).awaitUntil(deadline); switch (propose.outcome) { default: throw new IllegalStateException(); - case MAYBE_FAILURE: throw propose.maybeFailure().markAndThrowAsTimeoutOrFailure(false, consistencyForConsensus, failedAttemptsDueToContention); @@ -947,15 +941,6 @@ public class Paxos case SUPERSEDED: Superseded superseded = propose.superseded(); - // For consensus migration we are going to bail out earlier if migration is needed - // otherwise it it will fail every single query that races with migration being started - // during the propose step. Necessary because of CASSANDRA-18276 - // Shouldn't retry again on this protocol - if (superseded.needsConsensusMigration) - { - casReadMetrics.acceptMigrationRejects.mark(); - return RETRY_NEW_PROTOCOL; - } // TODO https://issues.apache.org/jira/browse/CASSANDRA-18276 side effects shouldn't matter for reads switch (superseded.hadSideEffects) { @@ -1071,15 +1056,14 @@ public class Paxos // After performing the prepare phase we may discover that we can't propose // our own transaction on this protocol by discovering a new CM Epoch if (ConsensusRequestRouter.instance.isKeyInMigratingOrMigratedRangeDuringPaxosBegin(query.metadata().id, query.partitionKey()) && prepare.outcome != MAYBE_FAILURE) - { return BeginResult.retryOnNewProtocol(); - } boolean isPromised = false; retry: switch (prepare.outcome) { default: throw new IllegalStateException(); - + case RETRY_DIFFERENT_SYSTEM: + return BeginResult.retryOnNewProtocol(); case FOUND_INCOMPLETE_COMMITTED: { FoundIncompleteCommitted incomplete = prepare.incompleteCommitted(); @@ -1103,11 +1087,10 @@ public class Paxos // and in fact it's possible for a CAS to sometimes determine if side effects occurred by reading // the underlying data and not witnessing the timestamp of its ballot (or any newer for the relevant data). Proposal repropose = new Proposal(inProgress.ballot, inProgress.accepted.update); - PaxosPropose.Status proposeResult = propose(repropose, inProgress.participants, false, true).awaitUntil(deadline); + PaxosPropose.Status proposeResult = propose(repropose, inProgress.participants, false).awaitUntil(deadline); switch (proposeResult.outcome) { default: throw new IllegalStateException(); - case MAYBE_FAILURE: throw proposeResult.maybeFailure().markAndThrowAsTimeoutOrFailure(isWrite, consistencyForConsensus, failedAttemptsDueToContention); @@ -1116,7 +1099,6 @@ public class Paxos break retry; case SUPERSEDED: - checkState(!proposeResult.superseded().needsConsensusMigration, "Should not receive needsConsensusMigration rejects from begin"); // since we are proposing a previous value that was maybe superseded by us before completion // we don't need to test the side effects, as we just want to start again, and fall through // to the superseded section below diff --git a/src/java/org/apache/cassandra/service/paxos/PaxosCommitAndPrepare.java b/src/java/org/apache/cassandra/service/paxos/PaxosCommitAndPrepare.java index b81f6b720b..41546d795a 100644 --- a/src/java/org/apache/cassandra/service/paxos/PaxosCommitAndPrepare.java +++ b/src/java/org/apache/cassandra/service/paxos/PaxosCommitAndPrepare.java @@ -29,11 +29,16 @@ import org.apache.cassandra.net.IVerbHandler; import org.apache.cassandra.net.Message; import org.apache.cassandra.net.MessagingService; import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.KeyMigrationState; import org.apache.cassandra.service.paxos.Commit.Agreed; +import org.apache.cassandra.service.paxos.PaxosPrepare.Rejected; +import org.apache.cassandra.tcm.ClusterMetadata; +import org.apache.cassandra.tcm.ClusterMetadataService; import org.apache.cassandra.tracing.Tracing; import static org.apache.cassandra.exceptions.RequestFailureReason.UNKNOWN; import static org.apache.cassandra.net.Verb.PAXOS2_COMMIT_AND_PREPARE_REQ; +import static org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.getKeyMigrationState; import static org.apache.cassandra.service.paxos.Paxos.newBallot; import static org.apache.cassandra.service.paxos.PaxosPrepare.start; @@ -121,6 +126,8 @@ public class PaxosCommitAndPrepare @Override public void doVerb(Message message) { + ClusterMetadataService.instance().fetchLogFromPeerOrCMS(ClusterMetadata.current(), message.from(), message.epoch()); + PaxosPrepare.Response response = execute(message.payload, message.from()); if (response == null) MessagingService.instance().respondWithFailure(UNKNOWN, message); @@ -134,10 +141,19 @@ public class PaxosCommitAndPrepare if (!Paxos.isInRangeAndShouldProcess(from, commit.update.partitionKey(), commit.update.metadata(), request.read != null)) return null; + // This can be done outside the lock + ClusterMetadata cm = ClusterMetadata.current(); + KeyMigrationState keyMigrationState = getKeyMigrationState(cm, commit.update.metadata().id, commit.update.partitionKey()); + // Make sure the operation is safe and there is no Accord state that needs application + // Also need to know max HLC in order to accept this ballot + long maxHLC = keyMigrationState.maybePerformAccordToPaxosKeyMigration(true); + if (maxHLC >= commit.ballot.unixMicros()) + return new Rejected(Ballot.atUnixMicrosWithLsb(maxHLC + 1, 0, commit.ballot.flag())); + try (PaxosState state = PaxosState.get(commit)) { state.commit(commit); - return PaxosPrepare.RequestHandler.execute(request, state); + return PaxosPrepare.RequestHandler.execute(request, state, cm); } } } diff --git a/src/java/org/apache/cassandra/service/paxos/PaxosPrepare.java b/src/java/org/apache/cassandra/service/paxos/PaxosPrepare.java index b1fb0e5cd6..1c3df86925 100644 --- a/src/java/org/apache/cassandra/service/paxos/PaxosPrepare.java +++ b/src/java/org/apache/cassandra/service/paxos/PaxosPrepare.java @@ -44,6 +44,8 @@ import org.apache.cassandra.db.SinglePartitionReadCommand; import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; import org.apache.cassandra.exceptions.RequestFailure; +import org.apache.cassandra.exceptions.RequestFailureReason; +import org.apache.cassandra.exceptions.RetryOnDifferentSystemException; import org.apache.cassandra.exceptions.UnavailableException; import org.apache.cassandra.gms.EndpointState; import org.apache.cassandra.gms.Gossiper; @@ -59,6 +61,7 @@ import org.apache.cassandra.schema.Schema; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.schema.TableMetadata; import org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.KeyMigrationState; +import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter; import org.apache.cassandra.service.paxos.PaxosPrepare.Status.Outcome; import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tcm.ClusterMetadataService; @@ -67,13 +70,12 @@ import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.utils.vint.VIntCoding; import static java.util.Collections.emptyMap; +import static org.apache.cassandra.exceptions.RequestFailureReason.RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM; import static org.apache.cassandra.exceptions.RequestFailureReason.UNKNOWN; import static org.apache.cassandra.locator.InetAddressAndPort.Serializer.inetAddressAndPortSerializer; import static org.apache.cassandra.net.Verb.PAXOS2_PREPARE_REQ; import static org.apache.cassandra.net.Verb.PAXOS2_PREPARE_RSP; import static org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.getKeyMigrationState; - -import org.apache.cassandra.service.consensus.migration.ConsensusMigratedAt; import static org.apache.cassandra.service.paxos.Ballot.Flag.NONE; import static org.apache.cassandra.service.paxos.Commit.Accepted; import static org.apache.cassandra.service.paxos.Commit.Committed; @@ -148,7 +150,7 @@ public class PaxosPrepare extends PaxosRequestCallback im */ static class Status { - enum Outcome { READ_PERMITTED, PROMISED, SUPERSEDED, FOUND_INCOMPLETE_ACCEPTED, FOUND_INCOMPLETE_COMMITTED, MAYBE_FAILURE, ELECTORATE_MISMATCH } + enum Outcome { READ_PERMITTED, PROMISED, SUPERSEDED, FOUND_INCOMPLETE_ACCEPTED, FOUND_INCOMPLETE_COMMITTED, MAYBE_FAILURE, ELECTORATE_MISMATCH, RETRY_DIFFERENT_SYSTEM } final Outcome outcome; final Participants participants; @@ -306,6 +308,16 @@ public class PaxosPrepare extends PaxosRequestCallback im } } + static class RetryDifferentSystem extends Status + { + private RetryDifferentSystem(Participants participants) + { + super(Outcome.RETRY_DIFFERENT_SYSTEM, participants); + } + + public String toString() { return "RETRY_DIFFERENT_SYSTEM"; } + } + private final boolean acceptEarlyReadPermission; private final AbstractRequest request; private Ballot supersededBy; // cannot be promised, as a newer promise has been made @@ -320,6 +332,7 @@ public class PaxosPrepare extends PaxosRequestCallback im private @Nonnull List withLatest; // promised and have latest commit private @Nullable List needLatest; // promised without having witnessed latest commit, nor yet been refreshed by us private int failures; // failed either on initial request or on refresh + private int retryDifferentSystemFailures; private boolean hasProposalStability = true; // no successful modifying proposal could have raced with us and not been seen private boolean hasOnlyPromises = true; private long maxLowBound; @@ -374,7 +387,7 @@ public class PaxosPrepare extends PaxosRequestCallback im static PaxosPrepare prepareWithBallot(Ballot ballot, Participants participants, SinglePartitionReadCommand readCommand, boolean isWrite, boolean acceptEarlyReadPermission) { Tracing.trace("Preparing {} with read", ballot); - Request request = new Request(ballot, participants.electorate, readCommand, isWrite, true); + Request request = new Request(ballot, participants.electorate, readCommand, isWrite, false); return prepareWithBallotInternal(participants, request, acceptEarlyReadPermission, null); } @@ -845,9 +858,17 @@ public class PaxosPrepare extends PaxosRequestCallback im super.onFailureWithMutex(from, reason.reason); ++failures; + if (reason.reason == RequestFailureReason.RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM) + retryDifferentSystemFailures++; if (failures + participants.sizeOfConsensusQuorum == 1 + participants.sizeOfPoll()) - signalDone(MAYBE_FAILURE); + { + // It failed, but retrying on a different system might succeed + if (participants.sizeOfConsensusQuorum + failures - retryDifferentSystemFailures < 1 + participants.sizeOfPoll()) + signalDone(Outcome.RETRY_DIFFERENT_SYSTEM); + else + signalDone(MAYBE_FAILURE); + } } private void signalDone(Outcome kindOfOutcome) @@ -886,6 +907,8 @@ public class PaxosPrepare extends PaxosRequestCallback im return Success.read(request.ballot, participants, readResponses, supersededBy); case MAYBE_FAILURE: return new MaybeFailure(new Paxos.MaybeFailure(participants, withLatest(), failureReasonsAsMap()), participants); + case RETRY_DIFFERENT_SYSTEM: + return new RetryDifferentSystem(participants); default: throw new IllegalStateException(); } @@ -1001,13 +1024,9 @@ public class PaxosPrepare extends PaxosRequestCallback im @Nonnull final MaybePromise.Outcome outcome; - @Nullable - final ConsensusMigratedAt maybeConsenusMigratedAt; - - Response(@Nonnull MaybePromise.Outcome outcome, @Nullable ConsensusMigratedAt maybeConsenusMigratedAt) + Response(@Nonnull MaybePromise.Outcome outcome) { this.outcome = outcome; - this.maybeConsenusMigratedAt = maybeConsenusMigratedAt; } Permitted permitted() { return (Permitted) this; } Rejected rejected() { return (Rejected) this; } @@ -1037,9 +1056,9 @@ public class PaxosPrepare extends PaxosRequestCallback im @Nullable final Ballot supersededBy; final Epoch electorateEpoch; - Permitted(MaybePromise.Outcome outcome, @Nullable ConsensusMigratedAt maybeConsensusMigratedAt, long lowBound, @Nullable Accepted latestAcceptedButNotCommitted, Committed latestCommitted, @Nullable ReadResponse readResponse, boolean hadProposalStability, Map gossipInfo, Epoch electorateEpoch, @Nullable Ballot supersededBy) + Permitted(MaybePromise.Outcome outcome, long lowBound, @Nullable Accepted latestAcceptedButNotCommitted, Committed latestCommitted, @Nullable ReadResponse readResponse, boolean hadProposalStability, Map gossipInfo, Epoch electorateEpoch, @Nullable Ballot supersededBy) { - super(outcome, maybeConsensusMigratedAt); + super(outcome); this.lowBound = lowBound; this.latestAcceptedButNotCommitted = latestAcceptedButNotCommitted; this.latestCommitted = latestCommitted; @@ -1061,9 +1080,9 @@ public class PaxosPrepare extends PaxosRequestCallback im { final Ballot supersededBy; - Rejected(Ballot supersededBy, @Nullable ConsensusMigratedAt maybeConsensusMigratedAt) + Rejected(Ballot supersededBy) { - super(REJECT, maybeConsensusMigratedAt); + super(REJECT); this.supersededBy = supersededBy; } @@ -1079,13 +1098,20 @@ public class PaxosPrepare extends PaxosRequestCallback im @Override public void doVerb(Message message) { - ClusterMetadataService.instance().fetchLogFromPeerOrCMSAsync(ClusterMetadata.current(), message.from(), message.epoch()); + ClusterMetadataService.instance().fetchLogFromPeerOrCMS(ClusterMetadata.current(), message.from(), message.epoch()); - Response response = execute(message.payload, message.from()); - if (response == null) - MessagingService.instance().respondWithFailure(UNKNOWN, message); - else - MessagingService.instance().respond(response, message); + try + { + Response response = execute(message.payload, message.from()); + if (response == null) + MessagingService.instance().respondWithFailure(UNKNOWN, message); + else + MessagingService.instance().respond(response, message); + } + catch (RetryOnDifferentSystemException e) + { + MessagingService.instance().respondWithFailure(RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM, message); + } } static Response execute(AbstractRequest request, InetAddressAndPort from) @@ -1094,9 +1120,27 @@ public class PaxosPrepare extends PaxosRequestCallback im return null; long start = nanoTime(); - try (PaxosState state = get(request.partitionKey, request.table)) + + try { - return execute(request, state); + // This can be done outside the lock + ClusterMetadata cm = ClusterMetadata.current(); + if (!request.isForRecovery) + { + if (ConsensusRequestRouter.instance.isKeyInMigratingOrMigratedRangeFromPaxos(request.table.id, request.partitionKey)) + throw new RetryOnDifferentSystemException(); + KeyMigrationState keyMigrationState = getKeyMigrationState(cm, request.table.id, request.partitionKey); + // Make sure the operation is safe and there is no Accord state that needs application + // Also need to know max HLC in order to accept this ballot + long maxHLC = keyMigrationState.maybePerformAccordToPaxosKeyMigration(request.isForWrite); + if (maxHLC >= request.ballot.unixMicros()) + return new Rejected(Ballot.atUnixMicrosWithLsb(maxHLC + 1, 0, request.ballot.flag())); + } + + try (PaxosState state = get(request.partitionKey, request.table)) + { + return execute(request, state, cm); + } } finally { @@ -1104,10 +1148,9 @@ public class PaxosPrepare extends PaxosRequestCallback im } } - static Response execute(AbstractRequest request, PaxosState state) + static Response execute(AbstractRequest request, PaxosState state, ClusterMetadata cm) { MaybePromise result = state.promiseIfNewer(request.ballot, request.isForWrite); - KeyMigrationState keyMigrationState = getKeyMigrationState(request.table.id, request.partitionKey); switch (result.outcome) { case PROMISE: @@ -1143,12 +1186,14 @@ public class PaxosPrepare extends PaxosRequestCallback im if (request.read != null) { - // Make sure the read is safe and there is no Accord state that needs application - keyMigrationState.maybePerformAccordToPaxosKeyMigration(request.isForWrite); - try (ReadExecutionController executionController = request.read.executionController(); - UnfilteredPartitionIterator iterator = request.read.executeLocally(executionController)) + SinglePartitionReadCommand readCommand = request.read; + // Allows txn recovery to read even if it would be blocked by migration away from Paxos + if (request.isForRecovery) + readCommand = readCommand.withTransactionalSettings(false, readCommand.nowInSec()); + try (ReadExecutionController executionController = readCommand.executionController(); + UnfilteredPartitionIterator iterator = readCommand.executeLocally(executionController, cm)) { - readResponse = request.read.createResponse(iterator, executionController.getRepairedDataInfo()); + readResponse = readCommand.createResponse(iterator, executionController.getRepairedDataInfo()); } if (hasProposalStability) @@ -1166,10 +1211,10 @@ public class PaxosPrepare extends PaxosRequestCallback im ColumnFamilyStore cfs = Schema.instance.getColumnFamilyStoreInstance(request.table.id); long lowBound = cfs.getPaxosRepairLowBound(request.partitionKey).uuidTimestamp(); - return new Permitted(result.outcome, keyMigrationState.consensusMigratedAt, lowBound, acceptedButNotCommitted, committed, readResponse, hasProposalStability, gossipInfo, electorateEpoch, supersededBy); + return new Permitted(result.outcome, lowBound, acceptedButNotCommitted, committed, readResponse, hasProposalStability, gossipInfo, electorateEpoch, supersededBy); case REJECT: - return new Rejected(result.supersededBy(), keyMigrationState.consensusMigratedAt); + return new Rejected(result.supersededBy()); default: throw new IllegalStateException(); @@ -1290,8 +1335,6 @@ public class PaxosPrepare extends PaxosRequestCallback im if (promised.outcome == PERMIT_READ) promised.supersededBy.serialize(out); } - if (version >= MessagingService.VERSION_51) - ConsensusMigratedAt.serializer.serialize(response.maybeConsenusMigratedAt, out); } public Response deserialize(DataInputPlus in, int version) throws IOException @@ -1300,10 +1343,7 @@ public class PaxosPrepare extends PaxosRequestCallback im if (flags == 0) { Ballot supersededBy = Ballot.deserialize(in); - ConsensusMigratedAt consensusMigratedAt = null; - if (version >= MessagingService.VERSION_51) - consensusMigratedAt = ConsensusMigratedAt.serializer.deserialize(in); - return new Rejected(supersededBy, consensusMigratedAt); + return new Rejected(supersededBy); } else { @@ -1318,10 +1358,7 @@ public class PaxosPrepare extends PaxosRequestCallback im Ballot supersededBy = null; if (outcome == PERMIT_READ) supersededBy = Ballot.deserialize(in); - ConsensusMigratedAt consensusMigratedAt = null; - if (version >= MessagingService.VERSION_51) - consensusMigratedAt = ConsensusMigratedAt.serializer.deserialize(in); - return new Permitted(outcome, consensusMigratedAt, lowBound, acceptedNotCommitted, committed, readResponse, hasProposalStability, gossipInfo, electorateEpoch, supersededBy); + return new Permitted(outcome, lowBound, acceptedNotCommitted, committed, readResponse, hasProposalStability, gossipInfo, electorateEpoch, supersededBy); } } @@ -1343,8 +1380,6 @@ public class PaxosPrepare extends PaxosRequestCallback im + (version >= MessagingService.VERSION_51 ? Epoch.messageSerializer.serializedSize(permitted.electorateEpoch, version) : 0) + (permitted.outcome == PERMIT_READ ? Ballot.sizeInBytes() : 0); } - if (version >= MessagingService.VERSION_51) - size += ConsensusMigratedAt.serializer.serializedSize(response.maybeConsenusMigratedAt); return size; } diff --git a/src/java/org/apache/cassandra/service/paxos/PaxosPrepareRefresh.java b/src/java/org/apache/cassandra/service/paxos/PaxosPrepareRefresh.java index ad909c6a00..ffc9aa610f 100644 --- a/src/java/org/apache/cassandra/service/paxos/PaxosPrepareRefresh.java +++ b/src/java/org/apache/cassandra/service/paxos/PaxosPrepareRefresh.java @@ -26,6 +26,7 @@ import org.slf4j.LoggerFactory; import org.apache.cassandra.exceptions.RequestFailure; import org.apache.cassandra.exceptions.RequestFailureReason; +import org.apache.cassandra.exceptions.RetryOnDifferentSystemException; import org.apache.cassandra.exceptions.WriteTimeoutException; import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; @@ -37,8 +38,11 @@ import org.apache.cassandra.net.MessagingService; import org.apache.cassandra.net.RequestCallbackWithFailure; import org.apache.cassandra.service.paxos.Commit.Agreed; import org.apache.cassandra.service.paxos.Commit.Committed; +import org.apache.cassandra.tcm.ClusterMetadata; +import org.apache.cassandra.tcm.ClusterMetadataService; import org.apache.cassandra.tracing.Tracing; +import static org.apache.cassandra.exceptions.RequestFailureReason.RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM; import static org.apache.cassandra.net.Verb.PAXOS2_PREPARE_REFRESH_REQ; import static org.apache.cassandra.service.paxos.Commit.isAfter; import static org.apache.cassandra.service.paxos.PaxosRequestCallback.shouldExecuteOnSelf; @@ -109,6 +113,7 @@ public class PaxosPrepareRefresh implements RequestCallbackWithFailure message) { + ClusterMetadataService.instance().fetchLogFromPeerOrCMS(ClusterMetadata.current(), message.from(), message.epoch()); onResponse(message.payload, message.from()); } @@ -121,6 +126,11 @@ public class PaxosPrepareRefresh implements RequestCallbackWithFailure message) { - Response response = execute(message.payload, message.from()); - if (response == null) - MessagingService.instance().respondWithFailure(RequestFailureReason.UNKNOWN, message); - else - MessagingService.instance().respond(response, message); + ClusterMetadataService.instance().fetchLogFromPeerOrCMS(ClusterMetadata.current(), message.from(), message.epoch()); + try + { + Response response = execute(message.payload, message.from()); + if (response == null) + MessagingService.instance().respondWithFailure(RequestFailureReason.UNKNOWN, message); + else + MessagingService.instance().respond(response, message); + } + catch (RetryOnDifferentSystemException e) + { + MessagingService.instance().respondWithFailure(RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM, message); + } } public static Response execute(Request request, InetAddressAndPort from) diff --git a/src/java/org/apache/cassandra/service/paxos/PaxosPropose.java b/src/java/org/apache/cassandra/service/paxos/PaxosPropose.java index 650bd8818c..2245f38141 100644 --- a/src/java/org/apache/cassandra/service/paxos/PaxosPropose.java +++ b/src/java/org/apache/cassandra/service/paxos/PaxosPropose.java @@ -31,6 +31,7 @@ import org.slf4j.LoggerFactory; import org.apache.cassandra.db.Keyspace; import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.exceptions.RequestFailure; +import org.apache.cassandra.exceptions.RetryOnDifferentSystemException; import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; @@ -39,11 +40,14 @@ import org.apache.cassandra.net.IVerbHandler; import org.apache.cassandra.net.Message; import org.apache.cassandra.net.MessagingService; import org.apache.cassandra.service.paxos.Commit.Proposal; +import org.apache.cassandra.service.paxos.PaxosPropose.Status.Outcome; +import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tcm.ClusterMetadataService; import org.apache.cassandra.utils.concurrent.ConditionAsConsumer; import static com.google.common.base.Preconditions.checkArgument; import static java.util.Collections.emptyMap; +import static org.apache.cassandra.exceptions.RequestFailureReason.RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM; import static org.apache.cassandra.exceptions.RequestFailureReason.UNKNOWN; import static org.apache.cassandra.net.Verb.PAXOS2_PROPOSE_REQ; import static org.apache.cassandra.service.paxos.PaxosPropose.Status.SideEffects.MAYBE; @@ -91,20 +95,14 @@ public class PaxosPropose> @Nullable final Ballot by; final SideEffects hadSideEffects; - // Consensus migration can occur at the same time that we are superseded - // and it's important to preserve returning the uncertainty of the superseded - // at the same time as enforcing the need for consensus migration - final boolean needsConsensusMigration; - Superseded(@Nullable Ballot by, SideEffects hadSideEffects, boolean needsConsensusMigration) + Superseded(@Nullable Ballot by, SideEffects hadSideEffects) { super(Outcome.SUPERSEDED); - checkArgument(needsConsensusMigration == true || by != null, "Must be superseded by ballot if not due to consensus migration"); this.by = by; this.hadSideEffects = hadSideEffects; - this.needsConsensusMigration = needsConsensusMigration; } - public String toString() { return "Superseded(" + by + ',' + hadSideEffects + ',' + needsConsensusMigration + ')'; } + public String toString() { return "Superseded(" + by + ',' + hadSideEffects + ')'; } } private static class MaybeFailure extends Status @@ -139,9 +137,6 @@ public class PaxosPropose> /** Number of accepts required */ final int required; - /** Repairing an in flight txn not proposing a new one **/ - final boolean isForRecovery; - /** Invoke on reaching a terminal status */ final OnDone onDone; @@ -161,9 +156,7 @@ public class PaxosPropose> /** The newest superseding ballot from a refusal; only returned to the caller if we fail to reach a quorum */ private volatile Ballot supersededBy; - private volatile boolean needsConsensusMigration = false; - - private PaxosPropose(Proposal proposal, int participants, int required, boolean waitForNoSideEffect, boolean isForRecovery, OnDone onDone) + private PaxosPropose(Proposal proposal, int participants, int required, boolean waitForNoSideEffect, OnDone onDone) { this.proposal = proposal; assert required > 0; @@ -171,7 +164,6 @@ public class PaxosPropose> this.participants = participants; this.required = required; this.onDone = onDone; - this.isForRecovery = isForRecovery; } /** @@ -182,7 +174,7 @@ public class PaxosPropose> * @param isForRecovery if true the value being proposed is not a new value it is a value from an existing in flight proposal * and will be allowed to proceed even if the key is migrating to a different consensus protocol */ - static Paxos.Async propose(Proposal proposal, Paxos.Participants participants, boolean waitForNoSideEffect, boolean isForRecovery) + static Paxos.Async propose(Proposal proposal, Paxos.Participants participants, boolean waitForNoSideEffect) { if (waitForNoSideEffect && proposal.update.isEmpty()) waitForNoSideEffect = false; // by definition this has no "side effects" (besides linearizing the operation) @@ -190,9 +182,9 @@ public class PaxosPropose> // to avoid unnecessary object allocations we extend PaxosPropose to implements Paxos.Async class Async extends PaxosPropose> implements Paxos.Async { - private Async(Proposal proposal, int participants, int required, boolean waitForNoSideEffect, boolean isForRecovery) + private Async(Proposal proposal, int participants, int required, boolean waitForNoSideEffect) { - super(proposal, participants, required, waitForNoSideEffect, isForRecovery, newConditionAsConsumer()); + super(proposal, participants, required, waitForNoSideEffect, newConditionAsConsumer()); } public Status awaitUntil(long deadline) @@ -211,24 +203,24 @@ public class PaxosPropose> } } - Async propose = new Async(proposal, participants.sizeOfPoll(), participants.sizeOfConsensusQuorum, waitForNoSideEffect, isForRecovery); + Async propose = new Async(proposal, participants.sizeOfPoll(), participants.sizeOfConsensusQuorum, waitForNoSideEffect); propose.start(participants); return propose; } - static > T propose(Proposal proposal, Paxos.Participants participants, boolean waitForNoSideEffect, boolean isForRecovery, T onDone) + static > T propose(Proposal proposal, Paxos.Participants participants, boolean waitForNoSideEffect, T onDone) { if (waitForNoSideEffect && proposal.update.isEmpty()) waitForNoSideEffect = false; // by definition this has no "side effects" (besides linearizing the operation) - PaxosPropose propose = new PaxosPropose<>(proposal, participants.sizeOfPoll(), participants.sizeOfConsensusQuorum, waitForNoSideEffect, isForRecovery, onDone); + PaxosPropose propose = new PaxosPropose<>(proposal, participants.sizeOfPoll(), participants.sizeOfConsensusQuorum, waitForNoSideEffect, onDone); propose.start(participants); return onDone; } void start(Paxos.Participants participants) { - Message message = Message.out(PAXOS2_PROPOSE_REQ, new Request(proposal, isForRecovery), participants.isUrgent()); + Message message = Message.out(PAXOS2_PROPOSE_REQ, new Request(proposal), participants.isUrgent()); boolean executeOnSelf = false; for (int i = 0, size = participants.sizeOfPoll(); i < size ; ++i) @@ -240,7 +232,7 @@ public class PaxosPropose> } if (executeOnSelf) - PAXOS2_PROPOSE_REQ.stage.execute(() -> executeOnSelf(proposal, isForRecovery)); + PAXOS2_PROPOSE_REQ.stage.execute(() -> executeOnSelf(proposal)); } /** @@ -253,23 +245,22 @@ public class PaxosPropose> if (isSuccessful(responses)) return STATUS_SUCCESS; - if (!canSucceed(responses) && (supersededBy != null || needsConsensusMigration)) + if (!canSucceed(responses) && (supersededBy != null)) { Superseded.SideEffects sideEffects = hasNoSideEffects(responses) ? NO : MAYBE; - return new Superseded(supersededBy, sideEffects, needsConsensusMigration); + return new Superseded(supersededBy, sideEffects); } return new MaybeFailure(new Paxos.MaybeFailure(participants, required, accepts(responses), failureReasonsAsMap())); } - private void executeOnSelf(Proposal proposal, boolean isForRecovery) + private void executeOnSelf(Proposal proposal) { - executeOnSelf(proposal, isForRecovery, RequestHandler::execute); + executeOnSelf(proposal, RequestHandler::execute); } public void onResponse(AcceptResult acceptResult, InetAddressAndPort from) { - checkArgument(!isForRecovery || acceptResult.rejectedDueToConsensusMigration == false, "Repair should never be rejected due to consensus migration"); if (logger.isTraceEnabled()) logger.trace("{} for {} from {}", acceptResult, proposal, from); @@ -277,19 +268,17 @@ public class PaxosPropose> if (supersededBy != null) supersededByUpdater.accumulateAndGet(this, supersededBy, (a, b) -> a == null ? b : b.uuidTimestamp() > a.uuidTimestamp() ? b : a); - long increment = supersededBy == null && !acceptResult.rejectedDueToConsensusMigration + long increment = supersededBy == null ? ACCEPT_INCREMENT : REFUSAL_INCREMENT; - if (acceptResult.rejectedDueToConsensusMigration) - needsConsensusMigration = true; - update(increment); } @Override public void onFailure(InetAddressAndPort from, RequestFailure reason) { + checkArgument(reason.reason != RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM, "Repair should never be rejected due to consensus migration"); if (logger.isTraceEnabled()) logger.trace("{} {} failure from {}", proposal, reason, from); @@ -400,11 +389,9 @@ public class PaxosPropose> static class Request { final Proposal proposal; - final boolean isForRecovery; - Request(Proposal proposal, boolean isForRecovery) + Request(Proposal proposal) { this.proposal = proposal; - this.isForRecovery = isForRecovery; } @Override @@ -412,7 +399,6 @@ public class PaxosPropose> { return "Request{" + "proposal=" + proposal.toString("Propose") + - ", isForRecovery=" + isForRecovery + '}'; } } @@ -425,15 +411,24 @@ public class PaxosPropose> @Override public void doVerb(Message message) { - ClusterMetadataService.instance().fetchLogFromCMS(message.epoch()); - AcceptResult acceptResult = execute(message.payload.proposal, message.payload.isForRecovery, message.from()); - if (acceptResult == null) - MessagingService.instance().respondWithFailure(UNKNOWN, message); - else - MessagingService.instance().respond(acceptResult, message); + ClusterMetadataService.instance().fetchLogFromPeerOrCMS(ClusterMetadata.current(), message.from(), message.epoch()); + + try + { + AcceptResult acceptResult = execute(message.payload.proposal, message.from()); + if (acceptResult == null) + MessagingService.instance().respondWithFailure(UNKNOWN, message); + else + MessagingService.instance().respond(acceptResult, message); + } + catch (RetryOnDifferentSystemException e) + { + // Should not actually be thrown here, but continue to catch and response with the error just in case + MessagingService.instance().respondWithFailure(RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM, message); + } } - public static AcceptResult execute(Proposal proposal, boolean isForRecovery, InetAddressAndPort from) + public static AcceptResult execute(Proposal proposal, InetAddressAndPort from) { if (!Paxos.isInRangeAndShouldProcess(from, proposal.update.partitionKey(), proposal.update.metadata(), false)) return null; @@ -441,7 +436,7 @@ public class PaxosPropose> long start = nanoTime(); try (PaxosState state = PaxosState.get(proposal)) { - return state.acceptIfLatest(proposal, isForRecovery); + return state.acceptIfLatest(proposal); } finally { @@ -456,26 +451,19 @@ public class PaxosPropose> public void serialize(Request request, DataOutputPlus out, int version) throws IOException { Proposal.serializer.serialize(request.proposal, out, version); - if (version >= MessagingService.VERSION_51) - out.writeBoolean(request.isForRecovery); } @Override public Request deserialize(DataInputPlus in, int version) throws IOException { Proposal propose = Proposal.serializer.deserialize(in, version); - boolean isForRecovery = false; - if (version >= MessagingService.VERSION_51) - isForRecovery = in.readBoolean(); - return new Request(propose, isForRecovery); + return new Request(propose); } @Override public long serializedSize(Request request, int version) { long size = Proposal.serializer.serializedSize(request.proposal, version); - if (version >= MessagingService.VERSION_51) - size += TypeSizes.sizeof(request.isForRecovery); return size; } } @@ -487,8 +475,6 @@ public class PaxosPropose> out.writeBoolean(acceptResult.supersededBy != null); if (acceptResult.supersededBy != null) acceptResult.supersededBy.serialize(out); - if (version >= MessagingService.VERSION_51) - out.writeBoolean(acceptResult.rejectedDueToConsensusMigration); } public AcceptResult deserialize(DataInputPlus in, int version) throws IOException @@ -497,10 +483,7 @@ public class PaxosPropose> Ballot supersededBy = null; if (isSuperseded) supersededBy = Ballot.deserialize(in); - boolean rejectedDueToConsensusMigration = false; - if (version >= MessagingService.VERSION_51) - rejectedDueToConsensusMigration = in.readBoolean(); - return new AcceptResult(supersededBy, rejectedDueToConsensusMigration); + return new AcceptResult(supersededBy); } public long serializedSize(AcceptResult acceptResult, int version) @@ -508,8 +491,6 @@ public class PaxosPropose> long size = acceptResult.supersededBy != null ? TypeSizes.sizeof(true) + Ballot.sizeInBytes() : TypeSizes.sizeof(false); - if (version >= MessagingService.VERSION_51) - size += TypeSizes.sizeof(acceptResult.rejectedDueToConsensusMigration); return size; } } diff --git a/src/java/org/apache/cassandra/service/paxos/PaxosRepair.java b/src/java/org/apache/cassandra/service/paxos/PaxosRepair.java index 8c9f219a3b..224b404898 100644 --- a/src/java/org/apache/cassandra/service/paxos/PaxosRepair.java +++ b/src/java/org/apache/cassandra/service/paxos/PaxosRepair.java @@ -63,6 +63,7 @@ import org.apache.cassandra.schema.TableId; import org.apache.cassandra.schema.TableMetadata; import org.apache.cassandra.service.paxos.PaxosPropose.Superseded; import org.apache.cassandra.tcm.ClusterMetadata; +import org.apache.cassandra.tcm.ClusterMetadataService; import org.apache.cassandra.tcm.membership.NodeId; import org.apache.cassandra.utils.CassandraVersion; import org.apache.cassandra.utils.ExecutorUtils; @@ -281,7 +282,7 @@ public class PaxosRepair extends AbstractPaxosRepair // If ballots with same timestamp have been both accepted and rejected by different nodes, // to avoid a livelock we simply try to poison, knowing we will fail but use a new ballot // (note there are alternative approaches but this is conservative) - return PaxosPropose.propose(latestAccepted, participants, false, true, + return PaxosPropose.propose(latestAccepted, participants, false, new ProposingRepair(latestAccepted)); } else if (isAcceptedButNotCommitted || isPromisedButNotAccepted || latestWitnessed.compareTo(latestPreviouslyWitnessed) < 0) @@ -342,7 +343,7 @@ public class PaxosRepair extends AbstractPaxosRepair Proposal propose = new Proposal(incomplete.ballot, incomplete.accepted.update); logger.trace("PaxosRepair of {} found incomplete {}", partitionKey(), incomplete.accepted); - return PaxosPropose.propose(propose, participants, false, true, + return PaxosPropose.propose(propose, participants, false, new ProposingRepair(propose)); // we don't know if we're done, so we must restart } @@ -360,7 +361,7 @@ public class PaxosRepair extends AbstractPaxosRepair // propose the empty ballot logger.trace("PaxosRepair of {} submitting empty proposal", partitionKey()); Proposal proposal = Proposal.empty(input.success().ballot, partitionKey(), table); - return PaxosPropose.propose(proposal, participants, false, true, + return PaxosPropose.propose(proposal, participants, false, new ProposingRepair(proposal)); } @@ -385,10 +386,8 @@ public class PaxosRepair extends AbstractPaxosRepair { case MAYBE_FAILURE: return retry(this); - case SUPERSEDED: Superseded superseded = input.superseded(); - checkState(!superseded.needsConsensusMigration, "Repair should not encounter consensus migration rejection"); if (isAfter(superseded.by, prevSupersededBy)) prevSupersededBy = input.superseded().by; return retry(this); @@ -492,6 +491,12 @@ public class PaxosRepair extends AbstractPaxosRepair return paxosConsistency.isDatacenterLocal() ? ConsistencyLevel.LOCAL_QUORUM : ConsistencyLevel.QUORUM; } + public long maxHlc() + { + checkState(successCriteria != null); + return successCriteria.unixMicros(); + } + static class Request { final DecoratedKey partitionKey; @@ -602,6 +607,11 @@ public class PaxosRepair extends AbstractPaxosRepair return; } + // The epoch has to propagate before checking the state to ensure that acquiring the lock on the paxos state + // creates a happens before relationship with any future paxos requests that should be rejected because of migration + // away from Paxos that is known to the process running this prepare + ClusterMetadataService.instance().fetchLogFromPeerOrCMS(ClusterMetadata.current(), message.from(), message.epoch()); + Ballot latestWitnessed; Accepted acceptedButNotCommited; Committed committed; diff --git a/src/java/org/apache/cassandra/service/paxos/PaxosRequestCallback.java b/src/java/org/apache/cassandra/service/paxos/PaxosRequestCallback.java index 06509edfc8..790f5dbc2d 100644 --- a/src/java/org/apache/cassandra/service/paxos/PaxosRequestCallback.java +++ b/src/java/org/apache/cassandra/service/paxos/PaxosRequestCallback.java @@ -26,6 +26,7 @@ import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CassandraRelevantProperties; import org.apache.cassandra.exceptions.RequestFailure; +import org.apache.cassandra.exceptions.RetryOnDifferentSystemException; import org.apache.cassandra.exceptions.WriteTimeoutException; import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.net.Message; @@ -61,6 +62,11 @@ public abstract class PaxosRequestCallback extends FailureRecordingCallback extends FailureRecordingCallback implements RequestCallbackWithFailure +{ + private final Set waitingResponse; + private final long lowBound; + private final SharedContext ctx; + + public PaxosUpdateLowBallot(SharedContext ctx, Collection endpoints, long maxHlc) + { + checkArgument(maxHlc != IAccordService.NO_HLC); + this.ctx = ctx; + this.waitingResponse = new HashSet<>(endpoints); + this.lowBound = maxHlc + 1; + } + + public synchronized void start() + { + Request request = new Request(lowBound); + + Message message = Message.out(PAXOS2_UPDATE_LOW_BALLOT_REQ, request); + + for (InetAddressAndPort endpoint : waitingResponse) + ctx.messaging().sendWithCallback(message, endpoint, this); + } + + @Override + public void onFailure(InetAddressAndPort from, RequestFailure reason) + { + tryFailure(new PaxosCleanupException("Timed out waiting on response from " + from)); + } + + @Override + public synchronized void onResponse(Message msg) + { + if (isDone()) + return; + + if (!waitingResponse.remove(msg.from())) + throw new IllegalArgumentException("Received unexpected response from " + msg.from()); + + if (waitingResponse.isEmpty()) + trySuccess(null); + } + + public static class Request + { + final long lowBound; + + Request(long lowBound) + { + this.lowBound = lowBound; + } + } + + public static final IVersionedSerializer serializer = new IVersionedSerializer() + { + public void serialize(Request request, DataOutputPlus out, int version) throws IOException + { + out.writeLong(request.lowBound); + } + + public Request deserialize(DataInputPlus in, int version) throws IOException + { + return new Request(in.readLong()); + } + + public long serializedSize(Request request, int version) + { + return TypeSizes.sizeof(request.lowBound); + } + }; + + public static IVerbHandler createVerbHandler(SharedContext ctx) + { + return (in) -> { + PaxosState.ballotTracker().updateLowBound(Ballot.atUnixMicrosWithLsb(in.payload.lowBound, 0, Flag.GLOBAL)); + ctx.messaging().respond(noPayload, in); + }; + } + + public static final IVerbHandler verbHandler = createVerbHandler(SharedContext.Global.instance); +} diff --git a/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java b/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java index 06e4881fb4..c1bf62b6a6 100644 --- a/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java +++ b/src/java/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTracker.java @@ -25,8 +25,6 @@ import java.util.zip.CRC32; import com.google.common.annotations.VisibleForTesting; import com.google.common.base.Preconditions; - -import org.apache.cassandra.service.ClientState; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -34,6 +32,8 @@ import org.apache.cassandra.db.rows.Row; import org.apache.cassandra.io.util.File; import org.apache.cassandra.io.util.RandomAccessReader; import org.apache.cassandra.io.util.SequentialWriter; +import org.apache.cassandra.service.ClientState; +import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.paxos.Ballot; import org.apache.cassandra.service.paxos.Commit; @@ -177,18 +177,27 @@ public class PaxosBallotTracker return directory; } - public synchronized void updateLowBound(Ballot update) throws IOException + public void updateLowBound(Ballot update) throws IOException { - if (!Commit.isAfter(update, lowBound)) + synchronized (this) { - logger.debug("Not updating lower bound with earlier or equal ballot from {} to {}", lowBound, update); - return; + if (!Commit.isAfter(update, lowBound)) + { + logger.debug("Not updating lower bound with earlier or equal ballot from {} to {}", lowBound, update); + return; + } + + logger.debug("Updating lower bound from {} to {}", lowBound, update); + ClientState.getTimestampForPaxos(lowBound.unixMicros()); + lowBound = update; + flush(); } - logger.debug("Updating lower bound from {} to {}", lowBound, update); - ClientState.getTimestampForPaxos(lowBound.unixMicros()); - lowBound = update; - flush(); + // During migration Accord needs to generate HLCs > than any generated by Paxos + // Accord uses repair (which updates this low bound) to discover the minimum HLC that was used by Paxos + // after Paxos stops. This bound will generally be in the past so it's fine to update it in Accord + // all the time + AccordService.instance().ensureMinHlc(lowBound.unixMicros() + 1); } public Ballot getHighBound() diff --git a/src/java/org/apache/cassandra/tcm/EpochAwareDebounce.java b/src/java/org/apache/cassandra/tcm/EpochAwareDebounce.java index d4d0e111b3..967a358371 100644 --- a/src/java/org/apache/cassandra/tcm/EpochAwareDebounce.java +++ b/src/java/org/apache/cassandra/tcm/EpochAwareDebounce.java @@ -23,6 +23,7 @@ import java.util.function.Supplier; import com.github.benmanes.caffeine.cache.Cache; import com.github.benmanes.caffeine.cache.Caffeine; +import org.apache.cassandra.concurrent.ScheduledExecutors; import org.apache.cassandra.utils.Closeable; import org.apache.cassandra.utils.concurrent.Future; import org.apache.cassandra.utils.concurrent.ImmediateFuture; @@ -43,6 +44,7 @@ public class EpochAwareDebounce implements Closeable private final AtomicReference currentFuture = new AtomicReference<>(); private final Cache> inflight = Caffeine.newBuilder() + .executor(ScheduledExecutors.scheduledFastTasks) .maximumSize(getEpochAwareDebounceInFlightTrackerMaxSize()) .expireAfterWrite(getCmsAwaitTimeout().to(MILLISECONDS), MILLISECONDS) diff --git a/src/java/org/apache/cassandra/tools/RepairRunner.java b/src/java/org/apache/cassandra/tools/RepairRunner.java index 8451de2918..97dbe39c37 100644 --- a/src/java/org/apache/cassandra/tools/RepairRunner.java +++ b/src/java/org/apache/cassandra/tools/RepairRunner.java @@ -21,12 +21,12 @@ import java.io.IOException; import java.io.PrintStream; import java.text.SimpleDateFormat; import java.util.List; - import javax.management.ListenerNotFoundException; import javax.management.remote.JMXConnector; import org.apache.cassandra.service.ActiveRepairService.ParentRepairStatus; import org.apache.cassandra.service.StorageServiceMBean; +import org.apache.cassandra.utils.Closeable; import org.apache.cassandra.utils.concurrent.Condition; import org.apache.cassandra.utils.progress.ProgressEvent; import org.apache.cassandra.utils.progress.ProgressEventType; @@ -43,7 +43,7 @@ import static org.apache.cassandra.utils.progress.ProgressEventType.COMPLETE; import static org.apache.cassandra.utils.progress.ProgressEventType.ERROR; import static org.apache.cassandra.utils.progress.ProgressEventType.PROGRESS; -public class RepairRunner extends JMXNotificationProgressListener +public class RepairRunner extends JMXNotificationProgressListener implements Closeable { public static abstract class RepairCmd { @@ -54,8 +54,6 @@ public class RepairRunner extends JMXNotificationProgressListener this.keyspace = keyspace; } - - public abstract Integer start(); } private final SimpleDateFormat format = new SimpleDateFormat("yyyy-MM-dd HH:mm:ss,SSS"); @@ -85,6 +83,7 @@ public class RepairRunner extends JMXNotificationProgressListener this.cmd = repairCmd.start(); } + @Override public void close() { try @@ -109,6 +108,11 @@ public class RepairRunner extends JMXNotificationProgressListener } public void run() throws Exception + { + run(false); + } + + public void run(boolean block) throws Exception { if (cmd == null) return; @@ -121,10 +125,17 @@ public class RepairRunner extends JMXNotificationProgressListener } else { - while (!condition.await(JMX_NOTIFICATION_POLL_INTERVAL_SECONDS, SECONDS)) + if (block) { - queryForCompletedRepair(String.format("After waiting for poll interval of %s seconds", - JMX_NOTIFICATION_POLL_INTERVAL_SECONDS)); + condition.await(); + } + else + { + while (!condition.await(JMX_NOTIFICATION_POLL_INTERVAL_SECONDS, SECONDS)) + { + queryForCompletedRepair(String.format("After waiting for poll interval of %s seconds", + JMX_NOTIFICATION_POLL_INTERVAL_SECONDS)); + } } Exception error = this.error; if (error == null) diff --git a/test/distributed/org/apache/cassandra/distributed/api/IMessage.java b/test/distributed/org/apache/cassandra/distributed/api/IMessage.java index cf7e920041..2d081d626e 100644 --- a/test/distributed/org/apache/cassandra/distributed/api/IMessage.java +++ b/test/distributed/org/apache/cassandra/distributed/api/IMessage.java @@ -32,9 +32,13 @@ public interface IMessage extends Serializable byte[] bytes(); - // TODO: need to make this a long int id(); + default long idAsLong() + { + return id(); + } + int version(); InetSocketAddress from(); diff --git a/test/distributed/org/apache/cassandra/distributed/impl/MessageImpl.java b/test/distributed/org/apache/cassandra/distributed/impl/MessageImpl.java index fc1dbcf0de..2c1b73e0ab 100644 --- a/test/distributed/org/apache/cassandra/distributed/impl/MessageImpl.java +++ b/test/distributed/org/apache/cassandra/distributed/impl/MessageImpl.java @@ -20,6 +20,8 @@ package org.apache.cassandra.distributed.impl; import java.net.InetSocketAddress; +import com.google.common.primitives.Ints; + import org.apache.cassandra.distributed.api.IMessage; import org.apache.cassandra.net.Verb; import org.apache.cassandra.utils.ByteArrayUtil; @@ -45,21 +47,31 @@ public class MessageImpl implements IMessage this.from = from; } + @Override public int verb() { return verb; } + @Override public byte[] bytes() { return bytes; } + @Override public int id() { - return (int) id; + return Ints.checkedCast(id); } + @Override + public long idAsLong() + { + return id; + } + + @Override public int version() { return version; @@ -71,6 +83,7 @@ public class MessageImpl implements IMessage return expiresAtNanos; } + @Override public InetSocketAddress from() { return from; diff --git a/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java b/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java index 4a0b54c386..651e5a35e9 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java +++ b/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java @@ -18,6 +18,7 @@ package org.apache.cassandra.distributed.test; +import java.io.IOException; import java.math.BigDecimal; import java.math.BigInteger; import java.net.InetAddress; @@ -43,6 +44,8 @@ import org.junit.BeforeClass; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import accord.messages.AbstractRequest; +import net.openhft.chronicle.core.util.SerializablePredicate; import org.apache.cassandra.config.CassandraRelevantProperties; import org.apache.cassandra.cql3.Duration; import org.apache.cassandra.db.marshal.AbstractType; @@ -73,6 +76,8 @@ import org.apache.cassandra.distributed.api.IMessage; import org.apache.cassandra.distributed.api.IMessageSink; import org.apache.cassandra.distributed.api.NodeToolResult; import org.apache.cassandra.distributed.shared.DistributedTestBase; +import org.apache.cassandra.io.util.DataInputBuffer; +import org.apache.cassandra.net.Message; import org.apache.cassandra.net.Verb; import org.apache.cassandra.service.accord.AccordCache; @@ -82,6 +87,7 @@ import static org.apache.cassandra.config.CassandraRelevantProperties.JOIN_RING; import static org.apache.cassandra.config.CassandraRelevantProperties.RESET_BOOTSTRAP_PROGRESS; import static org.apache.cassandra.config.CassandraRelevantProperties.SKIP_GC_INSPECTOR; import static org.apache.cassandra.distributed.action.GossipHelper.withProperty; +import static org.apache.cassandra.distributed.impl.TestEndpointCache.toCassandraInetAddressAndPort; import static org.assertj.core.api.Assertions.fail; // checkstyle: suppress below 'blockSystemPropertyUsage' @@ -98,19 +104,61 @@ public class TestBaseImpl extends DistributedTestBase protected static class MessageCountingSink implements IMessageSink { private final Cluster cluster; + private final SerializablePredicate> filter; + + // This isn't perfect at excluding messages so make sure it excludes the ones you care about in your test + public static final SerializablePredicate> EXCLUDE_SYNC_POINT_MESSAGES = message -> { + if (message.payload instanceof AbstractRequest) + return !((AbstractRequest)message.payload).txnId.isSyncPoint(); + return true; + }; public MessageCountingSink(Cluster cluster) + { + this(cluster, message -> true); + } + + public MessageCountingSink(Cluster cluster, SerializablePredicate> filter) { this.cluster = cluster; + this.filter = filter; } @Override - public void accept(InetSocketAddress to, IMessage message) + public void accept(InetSocketAddress to, IMessage iMessage) { - Verb verb = Verb.fromId(message.verb()); - logger.debug("verb {} to {} message {}", verb, to, message); - messageCounts.computeIfAbsent(verb, ignored -> new AtomicLong()).incrementAndGet(); - cluster.get(to).receiveMessage(message); + Verb verb = Verb.fromId(iMessage.verb()); + logger.debug("verb {} to {}", verb, to); + SerializablePredicate> filter = this.filter; + // Do some gymnastics to evaluate the predicate because deserialization fails if done here + // This races with shutdown so when attempting to count/deliver to shut down nodes ignore the exception + // and count anyways since we can't do much better + boolean accepted; + try + { + accepted = cluster.get(to).unsafeCallOnThisThread(() -> { + try (DataInputBuffer in = new DataInputBuffer(iMessage.bytes())) + { + Message message = Message.serializer.deserialize(in, toCassandraInetAddressAndPort(iMessage.from()), iMessage.version()); + return filter.test(message); + } + catch (IOException e) + { + throw new RuntimeException(e); + } + }); + } + catch (IllegalStateException e) + { + if (e.getMessage().startsWith("Can't use shutdown ")) + return; + throw e; + } + if (accepted) + { + messageCounts.computeIfAbsent(verb, ignored -> new AtomicLong()).incrementAndGet(); + cluster.get(to).receiveMessage(iMessage); + } } } diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java index a4931c6ab5..5cb835c205 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationReadRaceTestBase.java @@ -75,6 +75,7 @@ import org.apache.cassandra.service.StorageService; import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.service.accord.api.TokenKey; +import org.apache.cassandra.service.accord.repair.AccordRepair.AccordRepairResult; import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.ConsensusMigrationRepairResult; import org.apache.cassandra.service.consensus.migration.ConsensusTableMigration; @@ -101,6 +102,7 @@ import static org.apache.cassandra.distributed.test.accord.InteropTokenRangeTest import static org.apache.cassandra.distributed.test.accord.InteropTokenRangeTest.TokenOperator.lte; import static org.apache.cassandra.distributed.util.QueryResultUtil.assertThat; import static org.apache.cassandra.utils.ByteBufferUtil.bytesToHex; +import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis; import static org.junit.Assert.assertEquals; /* @@ -413,7 +415,7 @@ public abstract class AccordMigrationReadRaceTestBase extends AccordTestBase { if (BATCH_INDEX == null || BATCH_INDEX == batchCount) { - logger.info("Executing batch {}", batchCount); + logger.info("Executing batch {} with query count {}", batchCount, queryBatch.size()); testBoundsBatch(queryBatch, validationBatch, expectRetry, batchCount); } else @@ -428,7 +430,7 @@ public abstract class AccordMigrationReadRaceTestBase extends AccordTestBase if (!queryBatch.isEmpty()) { - logger.info("Executing batch " + batchCount); + logger.info("Executing trailing batch " + batchCount + " with query count " + queryBatch.size()); testBoundsBatch(queryBatch, validationBatch, expectRetry, batchCount); } } @@ -622,7 +624,8 @@ public abstract class AccordMigrationReadRaceTestBase extends AccordTestBase RepairJobDesc desc = new RepairJobDesc(null, null, keyspace, table, ranges); TokenRange range = TokenRange.create(new TokenKey(tableId, new LongToken(migratingTokenStart)), new TokenKey(tableId, new LongToken(migratingTokenEnd))); Ranges accordRanges = Ranges.of(range); - ConsensusMigrationRepairResult repairResult = ConsensusMigrationRepairResult.fromRepair(startEpoch, accordRanges, true, true, true, false); + AccordRepairResult accordRepairResult = new AccordRepairResult(accordRanges, TimeUnit.MILLISECONDS.toMicros(currentTimeMillis())); + ConsensusMigrationRepairResult repairResult = ConsensusMigrationRepairResult.fromRepair(startEpoch, accordRepairResult, true, true, true, false, false); ConsensusTableMigration.completedRepairJobHandler.onSuccess(new RepairResult(desc, null, repairResult)); }).call(); result.get(); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationTest.java index 6dbfdca2aa..75f01e909a 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationTest.java @@ -112,9 +112,7 @@ import static org.apache.cassandra.distributed.api.ConsistencyLevel.ANY; import static org.apache.cassandra.distributed.api.ConsistencyLevel.SERIAL; import static org.apache.cassandra.schema.SchemaConstants.SYSTEM_KEYSPACE_NAME; import static org.apache.cassandra.schema.SchemaConstants.VIRTUAL_ACCORD_DEBUG; -import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision.paxosV2; import static org.apache.cassandra.service.paxos.PaxosState.MaybePromise.Outcome.PROMISE; -import static org.assertj.core.api.Fail.fail; import static org.junit.Assert.assertEquals; import static org.junit.Assert.assertNotNull; import static org.junit.Assert.assertTrue; @@ -249,7 +247,7 @@ public class AccordMigrationTest extends AccordTestBase if (routed) return super.routeAndMaybeMigrate(cm, tmd, key, consistencyLevel, requestTime, timeoutNanos, isForWrite); routed = true; - return paxosV2; + return ConsensusRoutingDecision.PAXOSV2; } } @@ -266,15 +264,6 @@ public class AccordMigrationTest extends AccordTestBase } } - public static class PaxosToAccordMigrationNotHappeningUpToAccept extends PaxosToAccordMigrationNotHappeningUpToBegin - { - @Override - public boolean isKeyInMigratingOrMigratedRangeDuringPaxosAccept(TableId tableId, DecoratedKey key) - { - return false; - } - } - public static class RoutesToAccordOnce extends ConsensusRequestRouter { boolean routed; @@ -285,7 +274,7 @@ public class AccordMigrationTest extends AccordTestBase if (routed) return super.routeAndMaybeMigrate(cm, tmd, key, consistencyLevel, requestTime, timeoutNanos, isForWrite); routed = true; - return ConsensusRoutingDecision.accord; + return ConsensusRoutingDecision.ACCORD; } } @@ -293,36 +282,32 @@ public class AccordMigrationTest extends AccordTestBase * Helper to invoke a query and assert that the right metrics change indicating the correct * paths were taken to execute the query during migration */ - private static void assertTargetAccordWrite(Consumer query, int coordinatorIndex, int key, List> expectedKeyMigrations, int expectedAccordWriteCount, int expectedCasWriteCount, int expectedKeyMigrationCount, int expectedCasBeginRejects, int expectedCasAcceptRejects) + private static void assertTargetAccordWrite(Consumer query, int coordinatorIndex, int key, List> expectedKeyMigrations, int expectedAccordWriteCount, int expectedCasWriteCount, int expectedKeyMigrationCount, int expectedCasBeginRejects) { int startingWriteCount = getAccordWriteCount(coordinatorIndex); int startingCasWriteCount = getCasWriteCount(coordinatorIndex); int startingKeyMigrationCount = getKeyMigrationCount(coordinatorIndex); int startingCasWriteBeginRejects = getCasWriteBeginRejects(coordinatorIndex); - int startingCasWriteAcceptRejects = getCasWriteAcceptRejects(coordinatorIndex); query.accept(key); validateKeyMigrations(expectedKeyMigrations); assertEquals("Accord writes", expectedAccordWriteCount, getAccordWriteCount(coordinatorIndex) - startingWriteCount); assertEquals("CAS writes", expectedCasWriteCount, getCasWriteCount(coordinatorIndex) - startingCasWriteCount); assertEquals("Key Migrations", expectedKeyMigrationCount, getKeyMigrationCount(coordinatorIndex) - startingKeyMigrationCount); assertEquals("CAS Begin rejects", expectedCasBeginRejects, getCasWriteBeginRejects(coordinatorIndex) - startingCasWriteBeginRejects); - assertEquals("CAS Accept rejects", expectedCasAcceptRejects, getCasWriteAcceptRejects(coordinatorIndex) - startingCasWriteAcceptRejects); } - private static Object[][] assertTargetAccordRead(Function query, int coordinatorIndex, int key, List> expectedKeyMigrations, int expectedAccordReadCount, int expectedCasPrepareCount, int expectedKeyMigrationCount, int expectedCasReadBeginRejects, int expectedCasReadAcceptRejects) + private static Object[][] assertTargetAccordRead(Function query, int coordinatorIndex, int key, List> expectedKeyMigrations, int expectedAccordReadCount, int expectedCasPrepareCount, int expectedKeyMigrationCount, int expectedCasReadBeginRejects) { int startingReadCount = getAccordReadCount(coordinatorIndex); int startingCasPrepareCount = getCasPrepareCount(coordinatorIndex); int startingKeyMigrationCount = getKeyMigrationCount(coordinatorIndex); int startingCasReadBeginRejects = getCasReadBeginRejects(coordinatorIndex); - int startingCasReadAcceptRejects = getCasReadAcceptRejects(coordinatorIndex); Object[][] result = query.apply(key); validateKeyMigrations(expectedKeyMigrations); assertEquals("Accord reads", expectedAccordReadCount, getAccordReadCount(coordinatorIndex) - startingReadCount); assertEquals("CAS prepares", expectedCasPrepareCount, getCasPrepareCount(coordinatorIndex) - startingCasPrepareCount); assertEquals("Key Migrations", expectedKeyMigrationCount, getKeyMigrationCount(coordinatorIndex) - startingKeyMigrationCount); assertEquals("CAS Begin rejects", expectedCasReadBeginRejects, getCasReadBeginRejects(coordinatorIndex) - startingCasReadBeginRejects); - assertEquals("CAS Accept rejects", expectedCasReadAcceptRejects, getCasReadAcceptRejects(coordinatorIndex) - startingCasReadAcceptRejects); return result; } @@ -419,7 +404,7 @@ public class AccordMigrationTest extends AccordTestBase NormalizedRanges migratingRanges = normalizedRanges(ImmutableList.of(migratingRange)); // Not actually migrating yet so should do nothing special - assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 0, 1, 0, 0, 0); + assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 0, 1, 0, 0); // Mark ranges migrating and check migration state is correct nodetool(coordinator, "consensus_admin", "begin-migration", "-st", midToken.toString(), "-et", maxToken.toString(), KEYSPACE, tableName); @@ -431,17 +416,17 @@ public class AccordMigrationTest extends AccordTestBase assertTargetPaxosWrite(runCasNoApply, 1, migratingKey, emptyList(), 0, 1, 0, 0, 0); nodetool(coordinator, "repair", "-st", upperMidToken.toString(), "-et", maxAlignedWithLocalRanges.toString(), "-skip-accord", "-skip-paxos"); // With data repaired the write should now key migrated - assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 0, 1, 0, 0); + assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 0, 1, 0); // Should not repeat key migration, and should still do a migration read in Accord - assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 0, 0, 0, 0); + assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 0, 0, 0); // Should run on Paxos since it is not in the migrating range - assertTargetAccordWrite(runCasNoApply, 1, notMigratingKey, expectedKeyMigrations, 0, 1, 0, 0, 0); + assertTargetAccordWrite(runCasNoApply, 1, notMigratingKey, expectedKeyMigrations, 0, 1, 0, 0); // Check that the coordinator on the other node also has saved that the key migration was performed // and runs the query on Accord immediately without key migration - assertTargetAccordWrite(runCasOnSecondNode, 2, migratingKey, expectedKeyMigrations, 1, 0, 0, 0, 0); + assertTargetAccordWrite(runCasOnSecondNode, 2, migratingKey, expectedKeyMigrations, 1, 0, 0, 0); // Forced repair while a node is down shouldn't work, use repair instead of finish-migration because repair exposes --force // and regular Cassandra repairs are eligible to drive migration so it's important they check --force and down nodes @@ -451,7 +436,7 @@ public class AccordMigrationTest extends AccordTestBase EndpointState endpointState = Gossiper.instance.getEndpointStateForEndpoint(secondNodeBroadcastAddress); Gossiper.runInGossipStageBlocking(() -> Gossiper.instance.markDead(secondNodeBroadcastAddress, endpointState)); }); - nodetool(coordinator, "repair", "--force"); + nodetool(coordinator, "repair", "-full", "--force"); // Data repair was already done for one node's local range NormalizedRanges alreadyDataRepaired = normalizedRanges(ImmutableList.of(new Range<>(upperMidToken, maxAlignedWithLocalRanges))); NormalizedRanges remainingPendingDataRepair = migratingRanges.subtract(alreadyDataRepaired); @@ -463,19 +448,19 @@ public class AccordMigrationTest extends AccordTestBase }); // Full repair should complete the migration and update the metadata, adding --force when nodes are up should be fine - nodetool(coordinator, "repair", "--force" ); + nodetool(coordinator, "repair", "-full", "--force" ); // Some ranges will be migrated because they were already data repaired NormalizedRanges alreadyMigrated = alreadyDataRepaired; assertMigrationState(tableName, ConsensusMigrationTarget.accord, alreadyMigrated, emptyList(), migratingRanges.subtract(alreadyMigrated), 1); // Need to repair a second time to complete the migration to Accord because we are invoking repair directly, finish would do both for us normally - nodetool(coordinator, "repair", "--force"); + nodetool(coordinator, "repair", "-full", "--force"); assertMigrationState(tableName, ConsensusMigrationTarget.accord, migratingRanges, emptyList(), emptyList(), 0); // Should run on Accord, and not perform key migration nor should it need to perform a migration read in Accord now that it is repaired - assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 0, 0, 0, 0); + assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 0, 0, 0); // Should run on Paxos, and not perform key migration - assertTargetAccordWrite(runCasNoApply, 1, notMigratingKey, expectedKeyMigrations, 0, 1, 0, 0, 0); + assertTargetAccordWrite(runCasNoApply, 1, notMigratingKey, expectedKeyMigrations, 0, 1, 0, 0); // Pivot to testing repair with a subrange of the migrating range as well as key migration // Will use the unmigrated range between lowerMidToken and midToken @@ -494,9 +479,9 @@ public class AccordMigrationTest extends AccordTestBase // Need to data repair for key migration to be possible since otherwise it will just run on Paxos nodetool(coordinator, "repair", "-st", lowerMidToken.toString(), "-et", "-3074457345618258603", "-skip-accord", "-skip-paxos"); nodetool(cluster.coordinator(2), "repair", "-st", "-3074457345618258603", "-et", midToken.toString(), "-skip-accord", "-skip-paxos"); - assertTargetAccordWrite(runCasApplies, 1, migratingKey, expectedKeyMigrations, 1, 0, 1, 0, 0); + assertTargetAccordWrite(runCasApplies, 1, migratingKey, expectedKeyMigrations, 1, 0, 1, 0); - // This will force the write to use the normal write patch + // This will force the write to use the normal write path cluster.get(1).runOnInstance(() -> ConsensusRequestRouter.setInstance(new PaxosToAccordMigrationNotHappeningUpToBegin())); // Update inserted row so the condition can apply, if the condition check doesn't apply // then it won't get to propose/accept @@ -516,25 +501,7 @@ public class AccordMigrationTest extends AccordTestBase // and the accept will be rejected at both nodes and we are certain we need to retry the transaction cluster.get(1).runOnInstance(() -> ConsensusRequestRouter.setInstance(new PaxosToAccordMigrationNotHappeningUpToBegin())); addExpectedMigratedKey(expectedKeyMigrations, migratingKey, tableUUID); - assertTargetAccordWrite(runCasApplies, 1, migratingKey, expectedKeyMigrations, 1, 1, 1, 0, 1); - - // One node will now accept the other will reject and we are uncertain if we should retry the transaction - // and should surface that as a timeout exception - migratingKey = testingKeys.next(); - makeCASApply.accept(migratingKey); - cluster.get(1).runOnInstance(() -> ConsensusRequestRouter.setInstance(new PaxosToAccordMigrationNotHappeningUpToAccept())); - try - { - cluster.filters().allVerbs().to(3).from(3).drop(); - runCasNoApply.accept(migratingKey); - cluster.filters().reset(); - fail("Should have thrown timeout exception"); - } - catch (Throwable t) - { - if (!t.getClass().getName().equals("org.apache.cassandra.exceptions.CasWriteTimeoutException")) - throw new RuntimeException(t); - } + assertTargetAccordWrite(runCasApplies, 1, migratingKey, expectedKeyMigrations, 1, 1, 1, 1); // Test that if we find out about a migration from the prepare phase Paxos.begin we // retry it on Accord @@ -542,7 +509,7 @@ public class AccordMigrationTest extends AccordTestBase // Should exit Paxos from begin, key migration should occur because it's a new key, and Accord will need to do a migration read migratingKey = testingKeys.next(); addExpectedMigratedKey(expectedKeyMigrations, migratingKey, tableUUID); - assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 1, 1, 1, 0); + assertTargetAccordWrite(runCasNoApply, 1, migratingKey, expectedKeyMigrations, 1, 1, 1, 1); // Now do two repairs to complete the migration repair, and we are done with black box integration testing // First repair is a range smack dab in the middle @@ -579,7 +546,7 @@ public class AccordMigrationTest extends AccordTestBase List> migratingRanges = ImmutableList.of(migratingRange); int key = 0; - assertTargetAccordRead(runRead, 1, key, expectedKeyMigrations, 0, 1, 0, 0, 0); + assertTargetAccordRead(runRead, 1, key, expectedKeyMigrations, 0, 1, 0, 0); // Mark wrap around range as migrating nodetool(coordinator, "consensus_admin", "begin-migration", "-st", String.valueOf(Long.MIN_VALUE + 1), "-et", String.valueOf(Long.MIN_VALUE), KEYSPACE, accordTableName); assertMigrationState(accordTableName, ConsensusMigrationTarget.accord, emptyList(), migratingRanges, migratingRanges, 1); @@ -588,14 +555,14 @@ public class AccordMigrationTest extends AccordTestBase nodetool(cluster.coordinator(i), "repair", "-skip-paxos", "-skip-accord"); // Should run directly on accord, migrate the key, and perform a quorum read fro Accord, Paxos repair will run prepare once addExpectedMigratedKey(expectedKeyMigrations, key, tableUUID); - assertTargetAccordRead(runRead, 1, key, expectedKeyMigrations, 1, 1, 1, 0, 0); + assertTargetAccordRead(runRead, 1, key, expectedKeyMigrations, 1, 1, 1, 0); key++; // Should run up to accept with both nodes refusing to accept savePromisedAndCommittedPaxosProposal(accordTableName, key); cluster.get(1).runOnInstance(() -> ConsensusRequestRouter.setInstance(new PaxosToAccordMigrationNotHappeningUpToBegin())); addExpectedMigratedKey(expectedKeyMigrations, key, tableUUID); - assertTargetAccordRead(runRead, 1, key, expectedKeyMigrations, 1, 2, 1, 0, 1); + assertTargetAccordRead(runRead, 1, key, expectedKeyMigrations, 1, 2, 1, 1); key++; }); } @@ -660,10 +627,10 @@ public class AccordMigrationTest extends AccordTestBase cluster.get(1).runOnInstance(() -> ConsensusRequestRouter.setInstance(new RoutesToAccordOnce())); nextMigratingKey = paxosMigratingKeys.next(); addExpectedMigratedKey(expectedKeyMigrations, nextMigratingKey, tableUUID); - assertTargetPaxosWrite(runCasNoApply, 1, nextMigratingKey, expectedKeyMigrations, 1, 1, 1, 1, 1); + assertTargetPaxosWrite(runCasNoApply, 1, nextMigratingKey, expectedKeyMigrations, 1, 1, 1, 1, 0); // Repair the currently migrating range from when targets were switched, but it's not an Accord repair, this is to make sure the wrong repair type doesn't trigger progress - nodetool(coordinator, "repair", "-st", upperMidToken.toString(), "-et", maxAlignedWithLocalRanges.toString(), "--skip-accord"); + nodetool(coordinator, "repair", "-full", "-st", upperMidToken.toString(), "-et", maxAlignedWithLocalRanges.toString(), "--skip-accord"); assertMigrationState(tableName, ConsensusMigrationTarget.paxos, ImmutableList.of(new Range(minToken, midToken), new Range(maxToken, minToken)), emptyList(), ImmutableList.of(accordMigratingRange), 1); // Paxos migrating keys should still need key migration after non-Accord repair @@ -832,8 +799,7 @@ public class AccordMigrationTest extends AccordTestBase assertEquals( PROMISE, state.promiseIfNewer(ballot, true).outcome()); PartitionUpdateBuilder updateBuilder = new PartitionUpdateBuilder(metadata, key); updateBuilder.row(CLUSTERING_VALUE).add("v", 42); - // Set isForRepair to true to force accepting the proposal for testing purposes - assertEquals( null, state.acceptIfLatest(new Proposal(ballot, updateBuilder.build()), true).supersededBy); + assertEquals( null, state.acceptIfLatest(new Proposal(ballot, updateBuilder.build())).supersededBy); } }); } diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java index 5e23b58b57..b286e71de1 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordMigrationWriteRaceTestBase.java @@ -92,6 +92,7 @@ import org.apache.cassandra.service.StorageService; import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.TokenRange; import org.apache.cassandra.service.accord.api.TokenKey; +import org.apache.cassandra.service.accord.repair.AccordRepair.AccordRepairResult; import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState; import org.apache.cassandra.service.consensus.migration.ConsensusMigrationRepairResult; @@ -120,6 +121,7 @@ import static org.apache.cassandra.distributed.test.accord.AccordMigrationWriteR import static org.apache.cassandra.distributed.test.accord.AccordMigrationWriteRaceTestBase.Scenario.MUTATION; import static org.apache.cassandra.distributed.util.QueryResultUtil.assertThat; import static org.apache.cassandra.exceptions.RequestFailureReason.RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM; +import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis; import static org.apache.cassandra.utils.Throwables.runUnchecked; import static org.junit.Assert.assertEquals; import static org.junit.Assert.assertFalse; @@ -796,7 +798,8 @@ public abstract class AccordMigrationWriteRaceTestBase extends AccordTestBase RepairJobDesc desc = new RepairJobDesc(null, null, keyspace, table, ranges); TokenRange range = TokenRange.create(new TokenKey(tableId, new LongToken(midTokenLong)), new TokenKey(tableId, new LongToken(maxTokenLong))); Ranges accordRanges = Ranges.of(range); - ConsensusMigrationRepairResult repairResult = ConsensusMigrationRepairResult.fromRepair(startEpoch, accordRanges, true, true, true, false); + AccordRepairResult accordRepairResult = new AccordRepairResult(accordRanges, TimeUnit.MILLISECONDS.toMicros(currentTimeMillis())); + ConsensusMigrationRepairResult repairResult = ConsensusMigrationRepairResult.fromRepair(startEpoch, accordRepairResult, true, true, true, false, false); ConsensusTableMigration.completedRepairJobHandler.onSuccess(new RepairResult(desc, null, repairResult)); return epochAfterRepair.getEpoch(); }); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordReadInteroperabilityTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordReadInteroperabilityTest.java index e8efed6957..f01ff612fd 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordReadInteroperabilityTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordReadInteroperabilityTest.java @@ -141,7 +141,7 @@ public class AccordReadInteroperabilityTest extends AccordTestBase return; test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, PRIMARY KEY(k, c))" + (migrated ? " WITH " + transactionalMode.asCqlParam() : ""), cluster -> { - SHARED_CLUSTER.setMessageSink(new MessageCountingSink(SHARED_CLUSTER)); + SHARED_CLUSTER.setMessageSink(new MessageCountingSink(SHARED_CLUSTER, MessageCountingSink.EXCLUDE_SYNC_POINT_MESSAGES)); if (!migrated) { String alterCQL = "ALTER TABLE " + qualifiedAccordTableName + " WITH " + transactionalMode.asCqlParam(); @@ -155,7 +155,7 @@ public class AccordReadInteroperabilityTest extends AccordTestBase else { nodetool(cluster.coordinator(1), "repair", "-skip-paxos", "-skip-accord", KEYSPACE, accordTableName); - nodetool(cluster.coordinator(1), "repair", "-skip-accord", KEYSPACE, accordTableName); + nodetool(cluster.coordinator(1), "repair", "-full", "-skip-accord", KEYSPACE, accordTableName); } } cluster.coordinator(1).execute(query, cl); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordWriteInteroperabilityTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordWriteInteroperabilityTest.java index 944a02face..f80c22017c 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordWriteInteroperabilityTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordWriteInteroperabilityTest.java @@ -32,41 +32,39 @@ import org.junit.runners.Parameterized; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import org.apache.cassandra.db.ColumnFamilyStore; -import org.apache.cassandra.db.DataRange; -import org.apache.cassandra.db.filter.ColumnFilter; -import org.apache.cassandra.db.memtable.Memtable; -import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; -import org.apache.cassandra.db.rows.Row; -import org.apache.cassandra.db.rows.UnfilteredRowIterator; import org.apache.cassandra.distributed.api.ConsistencyLevel; -import org.apache.cassandra.distributed.shared.InstanceClassLoader; -import org.apache.cassandra.io.sstable.SSTableReadsListener; import org.apache.cassandra.net.Verb; import org.apache.cassandra.service.consensus.TransactionalMode; -import static com.google.common.base.Throwables.getStackTraceAsString; -import static org.apache.cassandra.Util.dk; -import static org.apache.cassandra.Util.spinUntilTrue; +import static org.apache.cassandra.Util.spinAssertEquals; import static org.apache.commons.collections.ListUtils.synchronizedList; import static org.junit.Assert.assertEquals; -import static org.junit.Assert.assertFalse; import static org.junit.Assert.assertTrue; +/** + * Check that the query is sent to Accord and the apply is an interop apply as is required by the transactional + * mode at each step of migration as well as that when the apply response is sent that the memtable actually contains + * the data that the apply should have applied + */ @RunWith(Parameterized.class) public class AccordWriteInteroperabilityTest extends AccordTestBase { private static final Logger logger = LoggerFactory.getLogger(AccordInteroperabilityTest.class); - @Nonnull - private final TransactionalMode mode; - - private final boolean migrated; - - public AccordWriteInteroperabilityTest(@Nonnull TransactionalMode mode, boolean migrated) + enum Migration { - this.mode = mode; - this.migrated = migrated; + notNeeded, + firstPhase, + secondPhase, + finished; + } + + private final Migration migration; + + public AccordWriteInteroperabilityTest(@Nonnull TransactionalMode mode, Migration migration) + { + super(mode); + this.migration = migration; } @Parameterized.Parameters(name = "transactionalMode={0}, migrated={1}") @@ -76,8 +74,8 @@ public class AccordWriteInteroperabilityTest extends AccordTestBase { if (mode.accordIsEnabled) { - tests.add(new Object[]{ mode, true }); - tests.add(new Object[]{ mode, false }); + for (Migration migration : Migration.values()) + tests.add(new Object[]{ mode, migration}); } } return tests; @@ -144,9 +142,9 @@ public class AccordWriteInteroperabilityTest extends AccordTestBase private void testApplyIsInteropApply(String query) throws Throwable { - test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, PRIMARY KEY(k, c))" + (migrated ? " WITH " + transactionalMode.asCqlParam() : ""), + test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, PRIMARY KEY(k, c))" + (migration == Migration.notNeeded ? " WITH " + transactionalMode.asCqlParam() : ""), cluster -> { - MessageCountingSink messageCountingSink = new MessageCountingSink(SHARED_CLUSTER); + MessageCountingSink messageCountingSink = new MessageCountingSink(SHARED_CLUSTER, MessageCountingSink.EXCLUDE_SYNC_POINT_MESSAGES); List failures = synchronizedList(new ArrayList<>()); // Verify that the apply response is only sent after the row has been inserted // TODO (required): Need to delay mutation stage/mutation to ensure this has time to catch it @@ -155,36 +153,46 @@ public class AccordWriteInteroperabilityTest extends AccordTestBase { if (message.verb() == Verb.ACCORD_APPLY_RSP.id) { - // It can be async if it's migrated - if (migrated) - return; - int nodeIndex = ((InstanceClassLoader)ClassLoader.getSystemClassLoader()).getInstanceId(); - try - { - String keyspace = KEYSPACE; - String tableName = accordTableName; - SHARED_CLUSTER.get(nodeIndex).runOnInstance(() -> { - ColumnFamilyStore cfs = ColumnFamilyStore.getIfExists(keyspace, tableName); - Memtable memtable = cfs.getCurrentMemtable(); - int expectedPartitions = query.startsWith("BEGIN BATCH") ? 2 : 1; - assertEquals(expectedPartitions, memtable.partitionCount()); - UnfilteredPartitionIterator partitions = memtable.partitionIterator(ColumnFilter.all(cfs.metadata()), DataRange.allData(cfs.getPartitioner()), SSTableReadsListener.NOOP_LISTENER); - assertTrue(partitions.hasNext()); - for (int i = 0; i < expectedPartitions; i++) - { - UnfilteredRowIterator rows = partitions.next(); - assertTrue(rows.partitionKey().equals(dk(42)) || rows.partitionKey().equals(dk(1))); - assertTrue(rows.hasNext()); - Row row = (Row)rows.next(); - assertFalse(rows.hasNext()); - } - assertFalse(partitions.hasNext()); - }); - } - catch (Throwable t) - { - failures.add(getStackTraceAsString(t)); - } + // TODO (required): https://issues.apache.org/jira/browse/CASSANDRA-20770 +// // It can be async because Paxos no longer reads +// if (migration == Migration.notNeeded || migration == Migration.secondPhase || migration == Migration.finished) +// return; +// // It's a different class loader now? +// int nodeIndex = 0; +// try +// { +// nodeIndex = ((InstanceClassLoader)Thread.currentThread().getContextClassLoader()).getInstanceId(); +// } +// catch (Throwable t) +// { +// t.printStackTrace(); +// } +// try +// { +// String keyspace = KEYSPACE; +// String tableName = accordTableName; +// SHARED_CLUSTER.get(nodeIndex).runOnInstance(() -> { +// ColumnFamilyStore cfs = ColumnFamilyStore.getIfExists(keyspace, tableName); +// Memtable memtable = cfs.getCurrentMemtable(); +// int expectedPartitions = query.startsWith("BEGIN BATCH") ? 2 : 1; +// assertEquals(expectedPartitions, memtable.partitionCount()); +// UnfilteredPartitionIterator partitions = memtable.partitionIterator(ColumnFilter.all(cfs.metadata()), DataRange.allData(cfs.getPartitioner()), SSTableReadsListener.NOOP_LISTENER); +// assertTrue(partitions.hasNext()); +// for (int i = 0; i < expectedPartitions; i++) +// { +// UnfilteredRowIterator rows = partitions.next(); +// assertTrue(rows.partitionKey().equals(dk(42)) || rows.partitionKey().equals(dk(1))); +// assertTrue(rows.hasNext()); +// Row row = (Row)rows.next(); +// assertFalse(rows.hasNext()); +// } +// assertFalse(partitions.hasNext()); +// }); +// } +// catch (Throwable t) +// { +// failures.add(getStackTraceAsString(t)); +// } } } finally @@ -193,28 +201,44 @@ public class AccordWriteInteroperabilityTest extends AccordTestBase } }); - if (!migrated) - { + if (migration != Migration.notNeeded) cluster.coordinator(1).execute("ALTER TABLE " + qualifiedAccordTableName + " WITH " + transactionalMode.asCqlParam(), ConsistencyLevel.ALL); + if (migration == Migration.secondPhase || migration == Migration.finished) nodetool(cluster.coordinator(1), "repair", "-skip-paxos", "-skip-accord", KEYSPACE, accordTableName); - } + if (migration == Migration.finished) + nodetool(cluster.coordinator(1), "repair", "-skip-accord", KEYSPACE, accordTableName); String finalQuery = query; org.apache.cassandra.distributed.api.ConsistencyLevel consistencyLevel = org.apache.cassandra.distributed.api.ConsistencyLevel.QUORUM; - // Need to switch to CAS for it to run through Accord at all - if (!transactionalMode.nonSerialWritesThroughAccord && !query.startsWith("BEGIN TRANSACTION")) + // If the non-serial request wouldn't end up running on Accord then convert it to CAS to make it run on Accord + // and check that it's also an interop apply + if (!transactionalMode.nonSerialWritesThroughAccord && !query.startsWith("BEGIN TRANSACTION") && !query.startsWith("BEGIN BATCH")) { finalQuery = query + " IF NOT EXISTS"; consistencyLevel = org.apache.cassandra.distributed.api.ConsistencyLevel.SERIAL; } - long startingRegularApplyCount = messageCount(Verb.ACCORD_APPLY_REQ); - cluster.coordinator(1).execute(finalQuery, consistencyLevel); - if (transactionalMode.ignoresSuppliedCommitCL() && migrated) + boolean transactionalQuery = finalQuery.startsWith("BEGIN TRANSACTION") || finalQuery.contains("IF NOT EXISTS"); + boolean isCAS = finalQuery.contains("IF NOT EXISTS"); + cluster.coordinator(1).execute(finalQuery, consistencyLevel, ConsistencyLevel.QUORUM); + // If it isn't a transaction query and the mode doesn't write through Accord for non-SERIAL we don't expect to see any apply messages + // If it is a CAS statement, but it's the first phase of migration then we expect it to continue to run on Paxos + if ((!transactionalQuery && !transactionalMode.nonSerialWritesThroughAccord) || (isCAS && migration == migration.firstPhase)) { - // Apply is async and there can be a lot of sources of regular APPLY - spinUntilTrue(() -> messageCount(Verb.ACCORD_APPLY_REQ) > startingRegularApplyCount); + assertEquals(0, messageCount(Verb.ACCORD_APPLY_REQ)); + assertEquals(0, messageCount(Verb.ACCORD_INTEROP_APPLY_REQ)); assertEquals(0, messageCount(Verb.ACCORD_INTEROP_APPLY_REQ)); } + // If it's a transactional mode that ignores supplied commit CL and this isn't the first phase of migration + // then it should be able to do regular Accord apply because nothing besides Accord should be reading the key by the time + // the Accord txn starts + else if (transactionalMode.ignoresSuppliedCommitCL() && migration != migration.firstPhase) + { + spinAssertEquals(3, () -> messageCount(Verb.ACCORD_APPLY_REQ)); + assertEquals(0, messageCount(Verb.ACCORD_INTEROP_APPLY_REQ)); + } + // The apply should be an interop Apply if it's not one of the above exceptions + // Either because the transactional mode respects commit CL or because it's a phase of migration + // that races with non-SERIAL reads or CAS reads and needs to provide them with the expected commit CL else { assertEquals(3, messageCount(Verb.ACCORD_INTEROP_APPLY_REQ)); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/InteropAccordCQLTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/InteropAccordCQLTest.java index 72673b3a42..1c90d3410d 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/InteropAccordCQLTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/InteropAccordCQLTest.java @@ -18,9 +18,6 @@ package org.apache.cassandra.distributed.test.accord; -import org.junit.Ignore; -import org.junit.Test; - import org.apache.cassandra.service.consensus.TransactionalMode; public class InteropAccordCQLTest extends AccordCQLTestBase @@ -29,12 +26,4 @@ public class InteropAccordCQLTest extends AccordCQLTestBase { super(TransactionalMode.test_interop_read); } - - @Ignore - @Override - @Test - public void testCASSimulatorLite() throws Exception - { - super.testCASSimulatorLite(); - } } diff --git a/test/simulator/asm/org/apache/cassandra/simulator/asm/InterceptClasses.java b/test/simulator/asm/org/apache/cassandra/simulator/asm/InterceptClasses.java index 81b4a288e9..82f3b8012a 100644 --- a/test/simulator/asm/org/apache/cassandra/simulator/asm/InterceptClasses.java +++ b/test/simulator/asm/org/apache/cassandra/simulator/asm/InterceptClasses.java @@ -57,6 +57,7 @@ public class InterceptClasses implements BiFunction "|org[/.]apache[/.]cassandra[/.]db[/.]ColumnFamilyStore.*" + "|org[/.]apache[/.]cassandra[/.]db[/.]Keyspace.*" + "|org[/.]apache[/.]cassandra[/.]db[/.]SystemKeyspace.*" + + "|org[/.]apache[/.]cassandra[/.]index[/.].*" + "|org[/.]apache[/.]cassandra[/.]streaming[/.].*" + "|org[/.]apache[/.]cassandra[/.]db.streaming[/.].*" + "|org[/.]apache[/.]cassandra[/.]distributed[/.]impl[/.]DirectStreamingConnectionFactory.*" + diff --git a/test/simulator/main/org/apache/cassandra/simulator/AccordNetworkScheduler.java b/test/simulator/main/org/apache/cassandra/simulator/AccordNetworkScheduler.java new file mode 100644 index 0000000000..4f24346b5a --- /dev/null +++ b/test/simulator/main/org/apache/cassandra/simulator/AccordNetworkScheduler.java @@ -0,0 +1,177 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.simulator; + +import java.util.Arrays; +import java.util.Collections; +import java.util.EnumSet; +import java.util.concurrent.ConcurrentHashMap; +import java.util.stream.Collectors; + +import com.google.common.collect.Multimaps; +import com.google.common.collect.SetMultimap; + +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.distributed.api.IMessage; +import org.apache.cassandra.net.Verb; + +import static org.apache.cassandra.net.Verb.ACCORD_FETCH_TOPOLOGY_REQ; +import static org.apache.cassandra.net.Verb.ACCORD_FETCH_TOPOLOGY_RSP; +import static org.apache.cassandra.net.Verb.ACCORD_FETCH_WATERMARKS_REQ; +import static org.apache.cassandra.net.Verb.ACCORD_FETCH_WATERMARKS_RSP; +import static org.apache.cassandra.net.Verb.ACCORD_GET_DURABLE_BEFORE_REQ; +import static org.apache.cassandra.net.Verb.ACCORD_GET_DURABLE_BEFORE_RSP; +import static org.apache.cassandra.net.Verb.ACCORD_SYNC_NOTIFY_REQ; +import static org.apache.cassandra.net.Verb.ACCORD_SYNC_NOTIFY_RSP; + +/** + * Scheduler for Accord verbs to make certain Accord messages more reliable + */ +public class AccordNetworkScheduler implements FutureActionScheduler +{ + public final FutureActionScheduler defaultScheduler; + public final FutureActionScheduler semiReliableScheduler; + + public static final Verb[] ACCORD_VERBS; + + static + { + ACCORD_VERBS = Arrays.asList(Verb.values()).stream().filter(verb -> verb.name().startsWith("ACCORD_")).collect(Collectors.toList()).toArray(new Verb[0]); + } + + public static final EnumSet UNRELIABLE_ACCORD_VERBS = EnumSet.of( + ACCORD_FETCH_WATERMARKS_REQ, + ACCORD_FETCH_WATERMARKS_RSP, + ACCORD_GET_DURABLE_BEFORE_REQ, + ACCORD_GET_DURABLE_BEFORE_RSP, + ACCORD_FETCH_TOPOLOGY_REQ, + ACCORD_FETCH_TOPOLOGY_RSP, + ACCORD_SYNC_NOTIFY_REQ, + ACCORD_SYNC_NOTIFY_RSP + ); + + public AccordNetworkScheduler(FutureActionScheduler defaultScheduler, FutureActionScheduler semiReliableScheduler) + { + this.defaultScheduler = defaultScheduler; + this.semiReliableScheduler = semiReliableScheduler; + } + + @Override + public DeliverResult shouldDeliver(int from, int to, IInvokableInstance invoker, IMessage message) + { + DeliverResult deliverResult; + boolean protectedMessage = isProtectedMessage(from, to, invoker, message); + if (protectedMessage) + deliverResult = semiReliableScheduler.shouldDeliver(from, to, invoker, message); + else + deliverResult = defaultScheduler.shouldDeliver(from, to, invoker, message); + return new DeliverResult(deliverResult.deliver, protectedMessage | deliverResult.protectedMessage); + } + + @Override + public long messageDeadlineNanos(int from, int to, boolean protectedMessage) + { + return protectedMessage ? semiReliableScheduler.messageDeadlineNanos(from, to, protectedMessage) : defaultScheduler.messageDeadlineNanos(from, to, protectedMessage); + } + + @Override + public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos, boolean protectedMessage) + { + return protectedMessage ? semiReliableScheduler.messageTimeoutNanos(expiresAfterNanos, expirationIntervalNanos, protectedMessage) : defaultScheduler.messageTimeoutNanos(expiresAfterNanos, expirationIntervalNanos, protectedMessage); + } + + @Override + public long messageFailureNanos(int from, int to, boolean protectedMessage) + { + return protectedMessage ? semiReliableScheduler.messageTimeoutNanos(from, to, protectedMessage) : defaultScheduler.messageFailureNanos(from, to, protectedMessage); + } + + @Override + public long schedulerDelayNanos() + { + return 1; + } + + private final SetMultimap inFlightRequestsByNodes = Multimaps.newSetMultimap(new ConcurrentHashMap<>(), () -> Collections.newSetFromMap(new ConcurrentHashMap<>())); + + private boolean isProtectedMessage(int from, int to, IInvokableInstance invoker, IMessage message) + { + return true; +// Verb verb = Verb.fromId(message.verb()); +// // These ones we let be unreliable and don't bother to deserialize to see if it's something where we can infer +// // it is related to an exclusive sync point which we want to be more reliable +// if (UNRELIABLE_ACCORD_VERBS.contains(verb)) +// return false; +// +// // Check if this a protected response to a protected request +// if (verb.name().endsWith("_RSP")) +// { +// long callbackKey = ((long) from << 32) | (to & 0xFFFFFFFFL); +// Set inFlightRequests = inFlightRequestsByNodes.get(callbackKey); +// if (inFlightRequests != null && inFlightRequests.remove(message.id())) +// return true; +// } +// +// Long[] checkResult = invoker.unsafeCallOnThisThread(() -> { +// try +// { +// Message m = Message.serializer.deserialize(new DataInputBuffer(message.bytes()), (InetAddressAndPort) null, message.version()); +// Object payload = m.payload; +// boolean shouldProtect = false; +// boolean noHandler = false; +// if (payload instanceof TxnRequest) +// { +// TxnRequest txnRequest = (TxnRequest)payload; +// shouldProtect = txnRequest.txnId.isSyncPoint(); +// } +// else if (payload instanceof SetGloballyDurable || +// payload instanceof SetShardDurable || +// payload instanceof GetDurableBefore || +// payload instanceof GetMaxConflict) +// { +// shouldProtect = true; +// } +// else +// { +// noHandler = true; +// } +// +// if (shouldProtect) +// return new Long[]{ 1L, m.id()}; +// else +// return new Long[] { 0L, null }; +// } +// catch (IOException e) +// { +// throw new RuntimeException(e); +// } +// }); +// +// boolean shouldBeProtected = checkResult[0] != 0; +// // If this is a protected request expecting a response then store the callback id here so that +// // the response message is also protected +// if (shouldBeProtected && checkResult[1] != null) +// { +// checkState(verb.name().endsWith("_REQ")); +// long callbackKey = ((long) to << 32) | (from & 0xFFFFFFFFL); +// inFlightRequestsByNodes.put(callbackKey, message.idAsLong()); +// } +// return shouldBeProtected; + } +} diff --git a/test/simulator/main/org/apache/cassandra/simulator/AlwaysDeliverNetworkScheduler.java b/test/simulator/main/org/apache/cassandra/simulator/AlwaysDeliverNetworkScheduler.java index ef5bdb1fd3..b45107d76f 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/AlwaysDeliverNetworkScheduler.java +++ b/test/simulator/main/org/apache/cassandra/simulator/AlwaysDeliverNetworkScheduler.java @@ -20,6 +20,8 @@ package org.apache.cassandra.simulator; import java.util.concurrent.TimeUnit; +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.distributed.api.IMessage; import org.apache.cassandra.simulator.systems.SimulatedTime; /** @@ -36,27 +38,27 @@ public class AlwaysDeliverNetworkScheduler implements FutureActionScheduler { this(time, TimeUnit.MILLISECONDS.toNanos(10)); } - public AlwaysDeliverNetworkScheduler(SimulatedTime time, long dealayNanos) + public AlwaysDeliverNetworkScheduler(SimulatedTime time, long delayNanos) { this.time = time; - this.delayNanos = dealayNanos; + this.delayNanos = delayNanos; } - public Deliver shouldDeliver(int from, int to) + public DeliverResult shouldDeliver(int from, int to, IInvokableInstance invoker, IMessage message) { - return Deliver.DELIVER; + return DELIVER_UNPROTECTED_RESULT; } - public long messageDeadlineNanos(int from, int to) + public long messageDeadlineNanos(int from, int to, boolean protectedMessage) { return time.nanoTime() + delayNanos; } - public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos) + public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos, boolean protectedMessage) { return expiresAfterNanos + 1; } - public long messageFailureNanos(int from, int to) + public long messageFailureNanos(int from, int to, boolean protectedMessage) { throw new IllegalStateException(); } diff --git a/test/simulator/main/org/apache/cassandra/simulator/ClusterSimulation.java b/test/simulator/main/org/apache/cassandra/simulator/ClusterSimulation.java index ed21d7079d..bfc2b0576c 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/ClusterSimulation.java +++ b/test/simulator/main/org/apache/cassandra/simulator/ClusterSimulation.java @@ -200,7 +200,7 @@ public class ClusterSimulation implements AutoCloseable protected HeapPool.Logged.Listener memoryListener; protected SimulatedTime.Listener timeListener = (i1, i2) -> {}; protected LongConsumer onThreadLocalRandomCheck; - protected String transactionalMode = "full"; + protected String transactionalMode = "off"; public Builder failures(Failures failures) { @@ -517,7 +517,7 @@ public class ClusterSimulation implements AutoCloseable new SchedulerConfig(schedulerDelayChance, schedulerDelayNanos, schedulerLongDelayNanos)); } - public Map perVerbFutureActionSchedulers(int nodeCount, SimulatedTime time, RandomSource random) + public Map perVerbFutureActionSchedulers(int nodeCount, SimulatedTime time, RandomSource random, FutureActionScheduler defaultScheduler) { return Collections.emptyMap(); } @@ -780,6 +780,7 @@ public class ClusterSimulation implements AutoCloseable .set("file_cache_size", "16MiB") .set("use_deterministic_table_id", true) .set("accord.queue_submission_model", "ASYNC") + .set("accord.range_migration", "explicit") .set("disk_access_mode", "standard") .set("failure_detector", SimulatedFailureDetector.Instance.class.getName()) .set("commitlog_compression", new ParameterizedClass(LZ4Compressor.class.getName(), emptyMap())) @@ -877,7 +878,7 @@ public class ClusterSimulation implements AutoCloseable delivery = new SimulatedMessageDelivery(cluster); failureDetector = new SimulatedFailureDetector(cluster); FutureActionScheduler futureActionScheduler = builder.futureActionScheduler(numOfNodes, time, random); - Map perVerbFutureActionScheduler = builder.perVerbFutureActionSchedulers(numOfNodes, time, random); + Map perVerbFutureActionScheduler = builder.perVerbFutureActionSchedulers(numOfNodes, time, random, futureActionScheduler); simulated = new SimulatedSystems(random, time, delivery, execution, ballots, failureDetector, snitch, futureActionScheduler, perVerbFutureActionScheduler, builder.debug, failures); if (futureActionScheduler instanceof SimulatedFutureActionScheduler) simulated.register((SimulatedFutureActionScheduler) futureActionScheduler); diff --git a/test/simulator/main/org/apache/cassandra/simulator/FixedLossNetworkScheduler.java b/test/simulator/main/org/apache/cassandra/simulator/FixedLossNetworkScheduler.java index 923b691de8..8cfce0b48c 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/FixedLossNetworkScheduler.java +++ b/test/simulator/main/org/apache/cassandra/simulator/FixedLossNetworkScheduler.java @@ -26,6 +26,8 @@ import java.util.concurrent.TimeUnit; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.distributed.api.IMessage; import org.apache.cassandra.simulator.systems.SimulatedTime; import org.apache.cassandra.simulator.utils.ChanceRange; import org.apache.cassandra.simulator.utils.KindOfSequence; @@ -110,13 +112,13 @@ public class FixedLossNetworkScheduler implements FutureActionScheduler private Map pairs = new HashMap<>(); private static final int TIMEOUTS_IN_A_ROW = 1; - public Deliver shouldDeliver(int from, int to) + public DeliverResult shouldDeliver(int from, int to, IInvokableInstance invoker, IMessage message) { DeliveryPair pair = new DeliveryPair(from, to); if (!dropMessage.get(random, from, to)) { pairs.put(pair, 0); - return Deliver.DELIVER; + return DELIVER_UNPROTECTED_RESULT; } int subsequentFailures = pairs.compute(pair, (k, v) -> v == null ? 1 : v+1); @@ -125,28 +127,28 @@ public class FixedLossNetworkScheduler implements FutureActionScheduler { logger.info("Delivering {} after {} failures in a row", pair, TIMEOUTS_IN_A_ROW); pairs.put(pair, 0); - return Deliver.DELIVER; + return DELIVER_UNPROTECTED_RESULT; } logger.info("Timing out {} for the {}th time", pair, subsequentFailures); - return Deliver.TIMEOUT; + return TIMEOUT_RESULT; } - public long messageDeadlineNanos(int from, int to) + public long messageDeadlineNanos(int from, int to, boolean protectedMessage) { return time.nanoTime() + (delayMessage.get(random, from, to) ? normalLatency.get(random, from, to) : delayLatency.get(random, from, to)); } - public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos) + public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos, boolean protectedMessage) { return expiresAfterNanos + random.uniform(0, expirationIntervalNanos / 2); } - public long messageFailureNanos(int from, int to) + public long messageFailureNanos(int from, int to, boolean protectedMessage) { - return messageDeadlineNanos(from, to); + return messageDeadlineNanos(from, to, protectedMessage); } public long schedulerDelayNanos() diff --git a/test/simulator/main/org/apache/cassandra/simulator/FutureActionScheduler.java b/test/simulator/main/org/apache/cassandra/simulator/FutureActionScheduler.java index 67d601d63b..1fa2a5d53b 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/FutureActionScheduler.java +++ b/test/simulator/main/org/apache/cassandra/simulator/FutureActionScheduler.java @@ -18,6 +18,14 @@ package org.apache.cassandra.simulator; +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.distributed.api.IMessage; + +import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.DELIVER; +import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.DELIVER_AND_TIMEOUT; +import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.FAILURE; +import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.TIMEOUT; + /** * Makes decisions about when in the simulated scheduled, in terms of the global simulated nanoTime, * events should occur. @@ -26,28 +34,46 @@ public interface FutureActionScheduler { enum Deliver { DELIVER, TIMEOUT, DELIVER_AND_TIMEOUT, FAILURE } + class DeliverResult + { + public final Deliver deliver; + // A message that should get a better than normal treatment in terms of unreliability + public final boolean protectedMessage; + + public DeliverResult(Deliver deliver, boolean protectedMessage) + { + this.deliver = deliver; + this.protectedMessage = protectedMessage; + } + } + + DeliverResult DELIVER_UNPROTECTED_RESULT = new DeliverResult(DELIVER, false); + DeliverResult TIMEOUT_RESULT = new DeliverResult(TIMEOUT, false); + DeliverResult DELIVER_AND_TIMEOUT_RESULT = new DeliverResult(DELIVER_AND_TIMEOUT, false); + DeliverResult FAILURE_RESULT = new DeliverResult(FAILURE, false); + /** * Make a decision about the result of some attempt to deliver a message. * Note that this includes responses, so for any given message the chance * of a successful reply depends on two of these calls succeeding. */ - Deliver shouldDeliver(int from, int to); + DeliverResult shouldDeliver(int from, int to, IInvokableInstance invoker, IMessage message); /** * The simulated global nanoTime arrival of a message */ - long messageDeadlineNanos(int from, int to); + long messageDeadlineNanos(int from, int to, boolean protectedMessage); /** * The simulated global nanoTime at which a timeout should be reported for a message * with {@code expiresAfterNanos} timeout */ - long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos); + long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos, boolean protectedMessage); /** * The simulated global nanoTime at which a failure should be reported for a message */ - long messageFailureNanos(int from, int to); + long messageFailureNanos(int from, int to, boolean protectedMessage); /** * The additional time in nanos that should elapse for some thread signal event to occur diff --git a/test/simulator/main/org/apache/cassandra/simulator/SimulationRunner.java b/test/simulator/main/org/apache/cassandra/simulator/SimulationRunner.java index b04fb0a4cb..1e75aa01f2 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/SimulationRunner.java +++ b/test/simulator/main/org/apache/cassandra/simulator/SimulationRunner.java @@ -270,7 +270,7 @@ public class SimulationRunner @Option(name = { "--capture" }, title = "wait,wake,now", description = "Capture thread stack traces alongside events, choose from (wait,wake,now)") protected String capture; - @Option(name = { "--transactional-mode" }, title = "off|mixed_reads|full]", description = "What execution strategy to use for CAS and serial read") + @Option(name = { "--transactional-mode" }, title = "off|mixed_reads|full]", description = "Starting transactional mode for the created table") protected String transactionalMode; protected void propagate(B builder) diff --git a/test/simulator/main/org/apache/cassandra/simulator/SimulatorUtils.java b/test/simulator/main/org/apache/cassandra/simulator/SimulatorUtils.java index 622e35ab5d..64747093f0 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/SimulatorUtils.java +++ b/test/simulator/main/org/apache/cassandra/simulator/SimulatorUtils.java @@ -51,36 +51,34 @@ public class SimulatorUtils FastThreadLocal.destroy(); } - public static void verifyAndlogSimulatorArgs(Logger logger, String[] args) + public static void verifyAndlogSimulatorArgs(String[] args) { RuntimeMXBean runtimeMxBean = ManagementFactory.getRuntimeMXBean(); final List jvmArgs = runtimeMxBean.getInputArguments(); - logger.error("JVM Args: {}", jvmArgs.stream().collect(Collectors.joining("\" \"", "\"", "\""))); - logger.error("Command Args: {}", Arrays.stream(args).collect(Collectors.joining("\" \"", "\"", "\""))); + System.err.printf("JVM Args: %s%n", jvmArgs.stream().collect(Collectors.joining("\" \"", "\"", "\""))); + System.err.printf("Command Args: %s%n", Arrays.stream(args).collect(Collectors.joining("\" \"", "\"", "\""))); assert jvmArgs.stream().anyMatch(arg -> arg.startsWith("-Xbootclasspath/a") && arg.endsWith("simulator-bootstrap.jar")) : "must launch JVM with -Xbootclasspath/a:simulator-bootstrap.jar"; assert jvmArgs.stream().anyMatch(arg -> arg.startsWith("-javaagent:") && arg.endsWith("simulator-asm.jar")) : "must launch JVM with -javaagent:simulator-asm.jar"; if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-XX:-BackgroundCompilation"))) - logger.warn("JVM Argument -XX:-BackgroundCompilation not set, non-determinism possible"); + System.err.println("JVM Argument -XX:-BackgroundCompilation not set, non-determinism possible"); if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-XX:-TieredCompilation"))) - logger.warn("JVM Argument -XX:-TieredCompilation not set, non-determinism possible"); + System.err.println("JVM Argument -XX:-TieredCompilation not set, non-determinism possible"); if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-XX:CICompilerCount=1"))) - logger.warn("JVM Argument -XX:CICompilerCount=1 not set, non-determinism possible"); + System.err.println("JVM Argument -XX:CICompilerCount=1 not set, non-determinism possible"); if (!jvmArgs.stream().anyMatch(arg -> arg.startsWith("-XX:Tier4CompileThreshold="))) - logger.warn("JVM Argument -XX:Tier4CompileThreshold not set, non-determinism possible. Typically set -XX:Tier4CompileThreshold=1000"); + System.err.println("JVM Argument -XX:Tier4CompileThreshold not set, non-determinism possible."); if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-Dcassandra.disable_tcactive_openssl=true"))) - logger.warn("JVM Argument -Dcassandra.disable_tcactive_openssl=true not set, non-determinism possible. Typically set -XX:Tier4CompileThreshold=1000"); + System.err.println("JVM Argument -Dcassandra.disable_tcactive_openssl=true not set, non-determinism possible. Typically set -XX:Tier4CompileThreshold=1000"); // log4j support if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-Dlog4j2.disableJmx=true"))) - logger.warn("JVM Argument -Dlog4j2.disableJmx=true not set, non-determinism possible"); - if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-Dlog4j2.disable.jmx=true"))) - logger.warn("JVM Argument -Dlog4j2.disable.jmx=true not set, non-determinism possible"); + System.err.println("JVM Argument -Dlog4j2.disableJmx=true not set, non-determinism possible"); if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-Dlog4j.shutdownHookEnabled=false"))) - logger.warn("JVM Argument -Dlog4j.shutdownHookEnabled=false not set, non-determinism possible"); + System.err.println("JVM Argument -Dlog4j.shutdownHookEnabled=false not set, non-determinism possible"); if (!jvmArgs.stream().anyMatch(arg -> arg.equals("-Dcassandra.simulator.skiplog4jreload=true"))) - logger.warn("JVM Argument -Dcassandra.simulator.skiplog4jreload=true not set, non-determinism possible"); + System.err.println("JVM Argument -Dcassandra.simulator.skiplog4jreload=true not set, non-determinism possible"); } } diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/ClusterActionListener.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/ClusterActionListener.java index c8d873f2a4..c4d37e92be 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/ClusterActionListener.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/ClusterActionListener.java @@ -18,6 +18,8 @@ package org.apache.cassandra.simulator.cluster; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; + public interface ClusterActionListener { interface TopologyChangeValidator @@ -28,7 +30,7 @@ public interface ClusterActionListener interface RepairValidator { - public void before(Topology topology, boolean repairPaxos, boolean repairOnlyPaxos); + public void before(Topology topology, RepairType repairType); public void after(); } @@ -61,7 +63,7 @@ public interface ClusterActionListener return new RepairValidator() { @Override - public void before(Topology topology, boolean repairPaxos, boolean repairOnlyPaxos) + public void before(Topology topology, RepairType repairType) { } diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/KeyspaceActions.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/KeyspaceActions.java index 6ddf61109d..6ae41ab74e 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/KeyspaceActions.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/KeyspaceActions.java @@ -268,7 +268,7 @@ public class KeyspaceActions extends ClusterActions { case ACCORD_MIGRATE: haveConsensusMigrated = true; - return schedule(new OnClusterMigrateConsensus(this), options.topologyChangeInterval); + return schedule(new OnClusterConsensusMigrations(this, options.consensusChangeLimit), options.topologyChangeInterval); } } @@ -280,6 +280,7 @@ public class KeyspaceActions extends ClusterActions if (options.topologyChangeLimit >= 0 && ++topologyChangeCount > options.topologyChangeLimit) return null; + Set dcsWithoutOps = new HashSet<>(); while (!topologyOps.isEmpty() && (!registered.isEmpty() || joined.size() > sum(minRf))) { if (options.changePaxosVariantTo != null && !haveChangedVariant && random.decide(1f / (1 + registered.size()))) @@ -296,8 +297,14 @@ public class KeyspaceActions extends ClusterActions if (registered.size(dc) > 0 && joined.size(dc) > currentRf[dc]) next = options.allChoices.choose(random); else if (registered.size(dc) > 0 && topologyOps.contains(JOIN)) next = options.choicesNoLeave.choose(random); else if (joined.size(dc) > currentRf[dc] && topologyOps.contains(LEAVE)) next = options.choicesNoJoin.choose(random); - else if (joined.size(dc) > minRf[dc]) next = CHANGE_RF; - else continue; + else if (joined.size(dc) > minRf[dc] && topologyOps.contains(CHANGE_RF)) next = CHANGE_RF; + // Don't loop forever if no DC supports an op right now + else if (dcsWithoutOps.size() == currentRf.length) return null; + else + { + dcsWithoutOps.add(dc); + continue; + } // TODO (feature): introduce some time period between cluster actions switch (next) diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterAssertMigrationComplete.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterAssertMigrationComplete.java new file mode 100644 index 0000000000..9ba704c519 --- /dev/null +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterAssertMigrationComplete.java @@ -0,0 +1,67 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.simulator.cluster; + +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.schema.Schema; +import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.service.consensus.migration.ConsensusTableMigration; +import org.apache.cassandra.service.consensus.migration.TableMigrationState; +import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; +import org.apache.cassandra.simulator.Action; +import org.apache.cassandra.simulator.ActionList; +import org.apache.cassandra.simulator.systems.NonInterceptible; + +import static com.google.common.base.Preconditions.checkState; +import static org.apache.cassandra.simulator.Action.Modifiers.NONE; +import static org.apache.cassandra.simulator.systems.NonInterceptible.Permit.REQUIRED; + +public class OnClusterAssertMigrationComplete extends Action +{ + private final KeyspaceActions actions; + + public OnClusterAssertMigrationComplete(KeyspaceActions actions) + { + super("Validate consensus migration completed", NONE, NONE); + this.actions = actions; + } + + @Override + public ActionList performSimple() + { + String keyspace = actions.keyspace; + String table = actions.table; + for (IInvokableInstance instance : actions.cluster) + { + if (instance.isShutdown()) + { + continue; + } + NonInterceptible.apply(REQUIRED, () -> instance.unsafeCallOnThisThread(() -> + { + TableMetadata tm = Schema.instance.getTableMetadata(keyspace, table); + TableMigrationState tms = ConsensusTableMigration.getTableMigrationState(tm.id); + checkState(tms == null, "There should be no table migration state after migration completes, migrating ranges %s migrated ranges %s", tms == null ? null : tms.migratingRanges, tms == null ? null : tms.migratedRanges); + checkState(tm.params.transactionalMigrationFrom == TransactionalMigrationFromMode.none, "transactionalMigrationFrom should be none after migration completes"); + return null; + })); + } + return ActionList.empty(); + } +} diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterChangeRf.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterChangeRf.java index afc8ce1283..80a30038fe 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterChangeRf.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterChangeRf.java @@ -22,6 +22,7 @@ import java.util.Arrays; import org.apache.cassandra.simulator.ActionList; import org.apache.cassandra.simulator.Actions; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; import static org.apache.cassandra.simulator.Action.Modifiers.NONE; import static org.apache.cassandra.simulator.Action.Modifiers.RELIABLE_NO_TIMEOUTS; @@ -58,7 +59,7 @@ class OnClusterChangeRf extends OnClusterChangeTopology return ActionList.of( schemaChange("ALTER KEYSPACE " + description(), actions, on, command.toString()), - new OnClusterFullRepair(actions, after, true, false, false), + new OnClusterFullRepair(actions, after, RepairType.DATA_AND_PAXOS_FULL, false), // TODO: cleanup should clear paxos state tables Actions.of("Flush and Cleanup", !increase ? () -> actions.flushAndCleanup(after.membersOfRing) : ActionList::empty), Quiesce.all(actions).asAction(STRICT, RELIABLE_NO_TIMEOUTS, "Wait for cluster to quiesce") diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterConsensusMigrations.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterConsensusMigrations.java new file mode 100644 index 0000000000..d120683ccf --- /dev/null +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterConsensusMigrations.java @@ -0,0 +1,69 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.simulator.cluster; + +import java.util.ArrayList; +import java.util.List; + +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.schema.Schema; +import org.apache.cassandra.service.consensus.TransactionalMode; +import org.apache.cassandra.simulator.Action; +import org.apache.cassandra.simulator.ActionList; +import org.apache.cassandra.simulator.systems.NonInterceptible; + +import static org.apache.cassandra.simulator.Action.Modifiers.NONE; +import static org.apache.cassandra.simulator.systems.NonInterceptible.Permit.REQUIRED; + +class OnClusterConsensusMigrations extends Action +{ + private final KeyspaceActions actions; + private final int migrations; + + OnClusterConsensusMigrations(KeyspaceActions actions, int migrations) + { + super("Performing " + migrations + " consensus migrations", NONE, NONE); + this.actions = actions; + this.migrations = migrations; + } + + @Override + public ActionList performSimple() + { + IInvokableInstance instance1 = actions.cluster.get(1); + String keyspace = actions.keyspace; + String table = actions.table; + TransactionalMode transactionalMode = TransactionalMode.valueOf( + NonInterceptible.apply(REQUIRED, () -> instance1.unsafeCallOnThisThread(() -> Schema.instance.getTableMetadata(keyspace, table).params.transactionalMode.toString()))); + + // Just migrate back and forth + List result = new ArrayList<>(); + for (int i = 0; i < migrations; i++) + { + + TransactionalMode targetMode = TransactionalMode.full; + if (transactionalMode == TransactionalMode.full) + targetMode = TransactionalMode.off; + result.add(new OnClusterMigrateConsensus(actions, transactionalMode, targetMode)); + transactionalMode = targetMode; + } + + return ActionList.of(result).setStrictlySequential(); + } +} diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterFullRepair.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterFullRepair.java index d415d4e8a4..093ee133a9 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterFullRepair.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterFullRepair.java @@ -24,6 +24,7 @@ import java.util.function.Consumer; import org.apache.cassandra.simulator.Action; import org.apache.cassandra.simulator.ActionList; import org.apache.cassandra.simulator.cluster.ClusterActionListener.RepairValidator; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; import static org.apache.cassandra.simulator.Action.Modifiers.RELIABLE_NO_TIMEOUTS; import static org.apache.cassandra.simulator.Action.Modifiers.STRICT; @@ -36,17 +37,15 @@ class OnClusterFullRepair extends Action implements Consumer final Topology topology; final boolean force; final RepairValidator validator; - final boolean repairPaxos; - final boolean repairOnlyPaxos; + final RepairType repairType; - public OnClusterFullRepair(KeyspaceActions actions, Topology topology, boolean repairPaxos, boolean repairOnlyPaxos, boolean force) + public OnClusterFullRepair(KeyspaceActions actions, Topology topology, RepairType repairType, boolean force) { super(lazy(() -> "Full Repair on " + Arrays.toString(topology.membersOfRing)), STRICT, RELIABLE_NO_TIMEOUTS); this.actions = actions; // STRICT to ensure repairs do not run simultaneously, as seems not to be permitted even for non-overlapping ranges? this.topology = topology; - this.repairPaxos = repairPaxos; - this.repairOnlyPaxos = repairOnlyPaxos; + this.repairType = repairType; this.force = force; this.validator = actions.listener.newRepairValidator(this); register(runAfterTransitiveClosure(this)); @@ -55,8 +54,8 @@ class OnClusterFullRepair extends Action implements Consumer protected ActionList performSimple() { actions.validateReplicasForKeys(actions.cluster.get(topology.membersOfQuorum[0]), actions.keyspace, actions.table, topology); - validator.before(topology, repairPaxos, repairOnlyPaxos); - return actions.on(i -> new OnInstanceRepair(actions, i, repairPaxos, repairOnlyPaxos, force), topology.membersOfRing); + validator.before(topology, repairType); + return actions.on(i -> new OnInstanceRepair(actions, i, repairType, force), topology.membersOfRing); } public void accept(Action ignore) diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensus.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensus.java index d4c4cb87d5..ddb35d31b7 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensus.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensus.java @@ -24,64 +24,128 @@ import java.util.Collections; import java.util.HashMap; import java.util.List; import java.util.Map; -import java.util.Map.Entry; +import java.util.Random; import org.apache.cassandra.dht.IPartitioner; import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; -import org.apache.cassandra.harry.model.TokenPlacementModel; +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.schema.Schema; +import org.apache.cassandra.service.consensus.TransactionalMode; +import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; import org.apache.cassandra.simulator.Action; import org.apache.cassandra.simulator.ActionList; -import org.apache.cassandra.tcm.ClusterMetadata; -import org.apache.cassandra.tcm.membership.NodeId; -import org.apache.cassandra.tcm.ownership.TokenMap; +import org.apache.cassandra.simulator.RandomSource; +import org.apache.cassandra.simulator.systems.NonInterceptible; +import org.apache.cassandra.tcm.compatibility.TokenRingUtils; +import org.apache.cassandra.utils.FBUtilities; import org.apache.cassandra.utils.Pair; import static com.google.common.base.Preconditions.checkState; import static org.apache.cassandra.simulator.Action.Modifiers.NONE; +import static org.apache.cassandra.simulator.systems.NonInterceptible.Permit.REQUIRED; class OnClusterMigrateConsensus extends Action { private final KeyspaceActions actions; + private final TransactionalMode fromMode; + private final TransactionalMode targetMode; - OnClusterMigrateConsensus(KeyspaceActions actions) + OnClusterMigrateConsensus(KeyspaceActions actions, TransactionalMode fromMode, TransactionalMode targetMode) { - super("Performing consensus migration", NONE, NONE); + super("Performing consensus migration from " + fromMode + " to " + targetMode, NONE, NONE); this.actions = actions; + this.fromMode = fromMode; + this.targetMode = targetMode; } + @Override public ActionList performSimple() { - List result = new ArrayList<>(); - List>> ranges = new ArrayList<>(); - ClusterMetadata cm = ClusterMetadata.current(); - TokenMap tm = cm.tokenMap; - IPartitioner partitioner = tm.partitioner(); - TokenPlacementModel.Lookup lookup = actions.factory.lookup(); - Map idToNodeId = new HashMap<>(); - for (int id : actions.all.toArray()) - idToNodeId.put(id, lookup.nodeId(id)); - - for (int ii = 0; ii < actions.all.size(); ii++) + RandomSource rs = actions.random; + IInvokableInstance instance1 = actions.cluster.get(1); + String keyspace = actions.keyspace; + String table = actions.table; + for (IInvokableInstance instance : actions.cluster) { - int nodeIdx = ii + 1; - List tokens = tm.tokens(idToNodeId.get(nodeIdx)); - checkState(tokens.size() == 1, "Expect only 1, not handling vnodes tokenRanges " + tokens); - Token token = tokens.get(0); - Range tokenRange = new Range(tm.getPredecessor(token), token); - Range firstRange = new Range<>(tokenRange.left, partitioner.split(tokenRange.left, tokenRange.right, 0.33)); - Range secondRange = new Range<>(firstRange.right, partitioner.split(tokenRange.left, tokenRange.right, 0.66)); - Range thirdRange = new Range<>(secondRange.right, tokenRange.right); - ranges.add(Pair.create(nodeIdx, new SimpleEntry<>(firstRange.left.toString(), firstRange.right.toString()))); - ranges.add(Pair.create(nodeIdx, new SimpleEntry<>(secondRange.left.toString(), secondRange.right.toString()))); - ranges.add(Pair.create(nodeIdx, new SimpleEntry<>(thirdRange.left.toString(), thirdRange.right.toString()))); + if (instance.isShutdown()) + continue; + TransactionalMigrationFromMode transactionalMigrationFrom = TransactionalMigrationFromMode.valueOf( + NonInterceptible.apply(REQUIRED, () -> instance1.unsafeCallOnThisThread(() -> Schema.instance.getTableMetadata(keyspace, table).params.transactionalMigrationFrom.toString()))); + checkState(transactionalMigrationFrom == TransactionalMigrationFromMode.none); } - Collections.shuffle(ranges); + List result = new ArrayList<>(); + result.add(ClusterReliableQueryAction.schemaChange("ALTER TABLE " + keyspace + "." + table + " WITH TransactionalModel.full", actions, actions.random.uniform(1, actions.cluster.size() + 1), "ALTER TABLE " + keyspace + "." + table + " WITH " + targetMode.asCqlParam())); - System.out.println("Ranges to migrate " + ranges); + Map idToToken = new HashMap<>(); + List tokens = new ArrayList<>(); + IPartitioner partitioner = null; + for (int i = 1; i <= actions.cluster.size(); i++) + { + IInvokableInstance instance = actions.cluster.get(i); + String tokenString = instance.config().getString("initial_token"); + partitioner = FBUtilities.newPartitioner(instance.config().getString("partitioner")); + Token token = partitioner.getTokenFactory().fromString(tokenString); + tokens.add(token); + idToToken.put(i, token); + } - ranges.stream().forEach(p -> result.add(new OnClusterMigrateConsensusOneRange(actions, p.left(), p.right()))); - return ActionList.of(result); + boolean partialMigration = false;//rs.decide(0.25f); + List subRangesForNode = new ArrayList<>(); + int totalSubranges = 0; + for (int ii = 0; ii < actions.all.size; ii++) + { + int subRangesThisNode = rs.uniform(1, 4); + totalSubranges += subRangesThisNode; + subRangesForNode.add(subRangesThisNode); + } + int stopSubrange = partialMigration ? rs.uniform(0, totalSubranges - 1) : Integer.MAX_VALUE; + + List>> ranges = new ArrayList<>(); + for (int ii = 0; ii < actions.all.size; ii++) + { + int nodeIdx = ii + 1; + Token token = idToToken.get(nodeIdx); + Range tokenRange = new Range(TokenRingUtils.getPredecessor(tokens, token), token); + + int numSubRanges = subRangesForNode.get(0); + List> subRanges = new ArrayList<>(); + switch (numSubRanges) + { + default: + throw new IllegalStateException(); + case 1: + subRanges.add(tokenRange); + break; + case 2: + Range firstRange = new Range<>(tokenRange.left, partitioner.split(tokenRange.left, tokenRange.right, 0.5)); + subRanges.add(firstRange); + subRanges.add(new Range<>(firstRange.right, tokenRange.right)); + break; + case 3: + firstRange = new Range<>(tokenRange.left, partitioner.split(tokenRange.left, tokenRange.right, 0.33)); + Range secondRange = new Range<>(firstRange.right, partitioner.split(tokenRange.left, tokenRange.right, 0.66)); + Range thirdRange = new Range<>(secondRange.right, tokenRange.right); + subRanges.add(firstRange); + subRanges.add(secondRange); + subRanges.add(thirdRange); + break; + } + subRanges.stream().map(range -> Pair.create(nodeIdx, new SimpleEntry<>(range.left.toString(), range.right.toString()))).forEach(ranges::add); + if (subRanges.size() >= stopSubrange) + { + ranges = ranges.subList(0, stopSubrange); + break; + } + } + + if (rs.decide(0.5f)) + Collections.shuffle(ranges, new Random(actions.random.uniform(Long.MIN_VALUE, Long.MAX_VALUE))); + + ranges.stream().forEach(p -> result.add(new OnClusterMigrateConsensusOneRange(actions, p.left(), p.right(), targetMode))); + if (!partialMigration) + result.add(new OnClusterAssertMigrationComplete(actions)); + return ActionList.of(result).setStrictlySequential(); } } diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensusOneRange.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensusOneRange.java index 84a1d43c50..3e77994a01 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensusOneRange.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterMigrateConsensusOneRange.java @@ -18,33 +18,62 @@ package org.apache.cassandra.simulator.cluster; +import java.util.ArrayList; +import java.util.List; import java.util.Map; -import com.google.common.collect.ImmutableList; - +import org.apache.cassandra.service.consensus.TransactionalMode; +import org.apache.cassandra.service.consensus.migration.ConsensusMigrationTarget; import org.apache.cassandra.simulator.Action; import org.apache.cassandra.simulator.ActionList; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; import static org.apache.cassandra.simulator.Action.Modifiers.NONE; -import static org.apache.cassandra.simulator.Action.Modifiers.STRICT; class OnClusterMigrateConsensusOneRange extends Action { private final KeyspaceActions actions; private final int repairOn; - Map.Entry startMigrationRange; + private final Map.Entry startMigrationRange; + private final TransactionalMode targetMode; - OnClusterMigrateConsensusOneRange(KeyspaceActions actions, int repairOn, Map.Entry startMigrationRange) + OnClusterMigrateConsensusOneRange(KeyspaceActions actions, int repairOn, Map.Entry startMigrationRange, TransactionalMode targetMode) { - super("Performing consensus migration one range " + startMigrationRange, STRICT, NONE); + super("Performing consensus migration of range " + startMigrationRange + " to " + targetMode, NONE, NONE); this.actions = actions; this.repairOn = repairOn; this.startMigrationRange = startMigrationRange; + this.targetMode = targetMode; } public ActionList performSimple() { - return ActionList.of(new OnInstanceStartConsensusMigration(actions, 1, startMigrationRange ), - new OnClusterRepairRanges(actions, new int[] { repairOn }, true, false, ImmutableList.of(startMigrationRange))); + List result = new ArrayList<>(); + result.add(new OnInstanceStartConsensusMigration(actions, repairOn, startMigrationRange)); + boolean migrateViaRepair = actions.random.decide(0.5f); + String rangeString = startMigrationRange.getKey() + ":" + startMigrationRange.getValue(); + String targetString = targetMode.accordIsEnabled ? ConsensusMigrationTarget.accord.toString() : ConsensusMigrationTarget.paxos.toString(); + if (targetMode == TransactionalMode.off) + { + if (migrateViaRepair) + result.add(new OnInstanceRepair(actions, repairOn, RepairType.ACCORD_ONLY, startMigrationRange, false)); + else + result.add(new OnInstanceFinishConsensusMigration(actions, repairOn, rangeString, targetString)); + } + else + { + if (migrateViaRepair) + { + result.add(new OnInstanceRepair(actions, repairOn, RepairType.DATA_INCREMENTAL, startMigrationRange, false)); + result.add(new OnInstanceRepair(actions, repairOn, RepairType.DATA_AND_PAXOS_FULL, startMigrationRange, false)); + } + else + { + result.add(new OnInstanceFinishConsensusMigration(actions, repairOn, rangeString, targetString)); + result.add(new OnInstanceFinishConsensusMigration(actions, repairOn, rangeString, targetString)); + } + } + + return ActionList.of(result).setStrictlySequential(); } } diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterRepairRanges.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterRepairRanges.java index 94dcea5041..3c61a28e1b 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterRepairRanges.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterRepairRanges.java @@ -23,6 +23,7 @@ import java.util.Map; import org.apache.cassandra.simulator.ActionList; import org.apache.cassandra.simulator.Actions.ReliableAction; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; import static java.util.stream.IntStream.range; import static org.apache.cassandra.simulator.Action.Modifiers.NONE; @@ -30,11 +31,11 @@ import static org.apache.cassandra.simulator.Action.Modifiers.RELIABLE_NO_TIMEOU public class OnClusterRepairRanges extends ReliableAction { - public OnClusterRepairRanges(KeyspaceActions actions, int[] on, boolean repairPaxos, boolean repairOnlyPaxos, List> ranges) + public OnClusterRepairRanges(KeyspaceActions actions, int[] on, RepairType repairType, List> ranges, boolean force) { super("Repair ranges", NONE, RELIABLE_NO_TIMEOUTS, () -> ActionList.of(range(0, on.length) .mapToObj( - i -> new OnInstanceRepair(actions, on[i], repairPaxos, repairOnlyPaxos, ranges.get(i), true)))); + i -> new OnInstanceRepair(actions, on[i], repairType, ranges.get(i), force)))); } } diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterReplace.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterReplace.java index e5642665dc..140da9f18e 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterReplace.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnClusterReplace.java @@ -34,6 +34,7 @@ import org.apache.cassandra.schema.KeyspaceMetadata; import org.apache.cassandra.simulator.Action; import org.apache.cassandra.simulator.ActionList; import org.apache.cassandra.simulator.Actions; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tcm.ClusterMetadataService; import org.apache.cassandra.tcm.MultiStepOperation; @@ -105,7 +106,7 @@ class OnClusterReplace extends OnClusterChangeTopology ).toArray(); local.add(new OnClusterMarkDown(actions, leaving)); - local.add(new OnClusterRepairRanges(actions, others, true, false, repairRanges)); + local.add(new OnClusterRepairRanges(actions, others, RepairType.DATA_AND_PAXOS_FULL, repairRanges, true)); local.add(new ExecuteNextStep(actions, joining, Transformation.Kind.START_REPLACE)); local.addAll(Quiesce.all(actions)); return ActionList.of(local); diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceFinishConsensusMigration.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceFinishConsensusMigration.java new file mode 100644 index 0000000000..5c09b0dacb --- /dev/null +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceFinishConsensusMigration.java @@ -0,0 +1,83 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.simulator.cluster; + +import com.google.common.collect.ImmutableList; + +import org.apache.cassandra.distributed.api.IIsolatedExecutor; +import org.apache.cassandra.service.StorageService; +import org.apache.cassandra.simulator.Action; +import org.apache.cassandra.tcm.ClusterMetadata; +import org.apache.cassandra.tcm.ClusterMetadataService; +import org.apache.cassandra.tcm.Epoch; +import org.apache.cassandra.tools.RepairRunner; +import org.apache.cassandra.tools.RepairRunner.RepairCmd; + +import static com.google.common.base.Preconditions.checkState; +import static org.apache.cassandra.simulator.Action.Modifiers.NONE; +import static org.apache.cassandra.simulator.Action.Modifiers.RELIABLE_NO_TIMEOUTS; + +public class OnInstanceFinishConsensusMigration extends ClusterAction +{ + public OnInstanceFinishConsensusMigration(KeyspaceActions actions, int on, String range, String target) + { + super("Finish consensus migration on " + on + " " + actions.keyspace + "." + actions.table + " " + range, modeForTarget(target), modeForTarget(target), actions, on, invokableBlockingFinishConsensusMigration(actions.keyspace, actions.table, range, target)); + } + + private static Action.Modifiers modeForTarget(String target) + { + if (target.equals("off")) + return RELIABLE_NO_TIMEOUTS; + return NONE; + } + private static IIsolatedExecutor.SerializableRunnable invokableBlockingFinishConsensusMigration(String keyspaceName, String table, String range, String target) + { + return () -> { + try (RepairRunner runner = new RepairRunner(System.out, null, StorageService.instance, new RepairCmd(keyspaceName) + { + @Override + public Integer start() + { + Epoch initialEpoch = ClusterMetadata.current().epoch; + Integer cmd = StorageService.instance.finishConsensusMigration(keyspaceName, ImmutableList.of(table), range, target); + // This isn't going to work well if there are other sources of epoch changes + // but just to start it addresses the issue that TCM hasn't updated on this node yet + if (cmd == null) + { + ClusterMetadataService.instance().fetchLogFromCMS(initialEpoch.nextEpoch()); + cmd = StorageService.instance.finishConsensusMigration(keyspaceName, ImmutableList.of(table), range, target); + checkState(cmd != null); + } + return cmd; + } + })) + { + runner.start(); + try + { + runner.run(); + } + catch (Exception e) + { + throw new RuntimeException(e); + } + } + }; + } +} diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceRepair.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceRepair.java index 20283718a1..55ab4f30a8 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceRepair.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceRepair.java @@ -37,67 +37,99 @@ import org.apache.cassandra.utils.concurrent.Condition; import org.apache.cassandra.utils.progress.ProgressEventType; import static java.util.Collections.singletonList; -import static org.apache.cassandra.simulator.Action.Modifiers.RELIABLE_NO_TIMEOUTS; +import static org.apache.cassandra.simulator.Action.Modifiers.NONE; import static org.apache.cassandra.simulator.cluster.Utils.currentToken; import static org.apache.cassandra.simulator.cluster.Utils.parseTokenRanges; import static org.apache.cassandra.utils.FBUtilities.getBroadcastAddressAndPort; import static org.apache.cassandra.utils.concurrent.Condition.newOneTimeCondition; -class OnInstanceRepair extends ClusterAction +public class OnInstanceRepair extends ClusterAction { - public OnInstanceRepair(KeyspaceActions actions, int on, boolean repairPaxos, boolean repairOnlyPaxos, boolean force) + public enum RepairType { - super("Repair on " + on, RELIABLE_NO_TIMEOUTS, RELIABLE_NO_TIMEOUTS, actions, on, invokableBlockingRepair(actions.keyspace, repairPaxos, repairOnlyPaxos, false, force)); + DATA_INCREMENTAL(true, false, false, true), + PAXOS_FULL(false, true, false, false), + DATA_AND_PAXOS_FULL(true, true, false, false), + ACCORD_ONLY(false, false, true, false); + + public final boolean repairData; + public final boolean repairPaxos; + public final boolean repairAccord; + public final boolean incremental; + + RepairType(boolean repairData, boolean repairPaxos, boolean repairAccord, boolean incremental) + { + this.repairData = repairData; + this.repairPaxos = repairPaxos; + this.repairAccord = repairAccord; + this.incremental = incremental; + } + + @Override + public String toString() + { + return "RepairType{" + + "repairData=" + repairData + + ", repairPaxos=" + repairPaxos + + ", repairAccord=" + repairAccord + + ", incremental=" + incremental + + '}'; + } + + Modifiers modifiers() + { + return NONE; + } } - public OnInstanceRepair(KeyspaceActions actions, int on, boolean repairPaxos, boolean repairOnlyPaxos, Map.Entry repairRange, boolean force) + public OnInstanceRepair(KeyspaceActions actions, int on, RepairType repairType, boolean force) { - this(actions, on, RELIABLE_NO_TIMEOUTS, RELIABLE_NO_TIMEOUTS, repairPaxos, repairOnlyPaxos, repairRange, force); + super("Repair on " + on, repairType.modifiers(), repairType.modifiers(), actions, on, invokableBlockingRepair(actions.keyspace, repairType, false, force)); } - public OnInstanceRepair(KeyspaceActions actions, int on, Modifiers self, Modifiers transitive, String id, boolean repairPaxos, boolean repairOnlyPaxos, boolean primaryRangeOnly, boolean force) + public OnInstanceRepair(KeyspaceActions actions, int on, RepairType repairType, Map.Entry repairRange, boolean force) { - super(id, self, transitive, actions, on, invokableBlockingRepair(actions.keyspace, repairPaxos, repairOnlyPaxos, primaryRangeOnly, force)); + this(actions, on, repairType.modifiers(), repairType.modifiers(), repairType, repairRange, force); } - public OnInstanceRepair(KeyspaceActions actions, int on, Modifiers self, Modifiers transitive, boolean repairPaxos, boolean repairOnlyPaxos, Map.Entry repairRange, boolean force) + public OnInstanceRepair(KeyspaceActions actions, int on, Modifiers self, Modifiers transitive, RepairType repairType, Map.Entry repairRange, boolean force) { - super("Repair on " + on, self, transitive, actions, on, invokableBlockingRepair(actions.keyspace, repairPaxos, repairOnlyPaxos, repairRange, force)); + super("Repair on " + on, self, transitive, actions, on, invokableBlockingRepair(actions.keyspace, repairType, repairRange, force)); } - private static IIsolatedExecutor.SerializableRunnable invokableBlockingRepair(String keyspaceName, boolean repairPaxos, boolean repairOnlyPaxos, boolean primaryRangeOnly, boolean force) + private static IIsolatedExecutor.SerializableRunnable invokableBlockingRepair(String keyspaceName, RepairType repairType, boolean primaryRangeOnly, boolean force) { return () -> { Condition done = newOneTimeCondition(); - invokeRepair(keyspaceName, repairPaxos, repairOnlyPaxos, primaryRangeOnly, force, done::signal); + invokeRepair(keyspaceName, repairType, primaryRangeOnly, force, done::signal); done.awaitThrowUncheckedOnInterrupt(); }; } - private static IIsolatedExecutor.SerializableRunnable invokableBlockingRepair(String keyspaceName, boolean repairPaxos, boolean repairOnlyPaxos, Map.Entry repairRange, boolean force) + private static IIsolatedExecutor.SerializableRunnable invokableBlockingRepair(String keyspaceName, RepairType repairType, Map.Entry repairRange, boolean force) { return () -> { Condition done = newOneTimeCondition(); - invokeRepair(keyspaceName, repairPaxos, repairOnlyPaxos, () -> parseTokenRanges(singletonList(repairRange)), false, force, done::signal); + invokeRepair(keyspaceName, repairType, () -> parseTokenRanges(singletonList(repairRange)), false, force, done::signal); done.awaitThrowUncheckedOnInterrupt(); }; } - private static void invokeRepair(String keyspaceName, boolean repairPaxos, boolean repairOnlyPaxos, boolean primaryRangeOnly, boolean force, Runnable listener) + private static void invokeRepair(String keyspaceName, RepairType repairType, boolean primaryRangeOnly, boolean force, Runnable listener) { Keyspace keyspace = Keyspace.open(keyspaceName); ClusterMetadata metadata = ClusterMetadata.current(); - invokeRepair(keyspaceName, repairPaxos, repairOnlyPaxos, + invokeRepair(keyspaceName, repairType, () -> primaryRangeOnly ? TokenRingUtils.getPrimaryRangesFor(metadata.tokenMap.tokens(), Collections.singleton(currentToken())) : keyspace.getReplicationStrategy().getAddressReplicas(metadata).get(getBroadcastAddressAndPort()).asList(Replica::range), primaryRangeOnly, force, listener); } - private static void invokeRepair(String keyspaceName, boolean repairPaxos, boolean repairOnlyPaxos, IIsolatedExecutor.SerializableCallable>> rangesSupplier, boolean isPrimaryRangeOnly, boolean force, Runnable listener) + private static void invokeRepair(String keyspaceName, RepairType repairType, IIsolatedExecutor.SerializableCallable>> rangesSupplier, boolean isPrimaryRangeOnly, boolean force, Runnable listener) { Collection> ranges = rangesSupplier.call(); // no need to wait for completion, as we track all task submissions and message exchanges, and ensure they finish before continuing to next action - StorageService.instance.repair(keyspaceName, new RepairOption(RepairParallelism.SEQUENTIAL, isPrimaryRangeOnly, false, false, 1, ranges, false, force, PreviewKind.NONE, false, true, !repairOnlyPaxos, repairPaxos, false, false), singletonList((tag, event) -> { + StorageService.instance.repair(keyspaceName, new RepairOption(RepairParallelism.SEQUENTIAL, isPrimaryRangeOnly, repairType.incremental, false, 1, ranges, false, force, PreviewKind.NONE, false, true, repairType.repairData, repairType.repairPaxos, false, repairType.repairAccord), singletonList((tag, event) -> { if (event.getType() == ProgressEventType.COMPLETE) listener.run(); })); diff --git a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceStartConsensusMigration.java b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceStartConsensusMigration.java index e566529a46..de715e7b7d 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceStartConsensusMigration.java +++ b/test/simulator/main/org/apache/cassandra/simulator/cluster/OnInstanceStartConsensusMigration.java @@ -24,15 +24,17 @@ import java.util.Map; import org.apache.cassandra.distributed.api.IIsolatedExecutor; import org.apache.cassandra.service.StorageService; +import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; +import org.apache.cassandra.tcm.ClusterMetadata; +import org.apache.cassandra.tcm.ClusterMetadataService; -import static org.apache.cassandra.simulator.Action.Modifiers.RELIABLE_NO_TIMEOUTS; +import static org.apache.cassandra.simulator.Action.Modifiers.NONE; class OnInstanceStartConsensusMigration extends ClusterAction { - public OnInstanceStartConsensusMigration(KeyspaceActions actions, int on, Map.Entry startMigrationRange) { - this(actions, on, RELIABLE_NO_TIMEOUTS, RELIABLE_NO_TIMEOUTS, startMigrationRange); + this(actions, on, NONE, NONE, startMigrationRange); } public OnInstanceStartConsensusMigration(KeyspaceActions actions, int on, Modifiers self, Modifiers transitive, Map.Entry startMigrationRange) @@ -47,6 +49,12 @@ class OnInstanceStartConsensusMigration extends ClusterAction keyspaces.add(keyspaceName); List tables = new ArrayList<>(); tables.add(cfName); + ClusterMetadata clusterMetadata = ClusterMetadata.current(); + TransactionalMigrationFromMode migrationFrom = clusterMetadata.schema.getTableMetadata(keyspaceName, cfName).params.transactionalMigrationFrom; + // The alter enabling migration could be on another node, need to wait for it to be visible + // This assumes there are no other sources of epoch changes which won't work with topology changes + if (!migrationFrom.isMigrating()) + ClusterMetadataService.instance().fetchLogFromCMS(clusterMetadata.epoch.nextEpoch()); StorageService.instance.migrateConsensusProtocol(keyspaces, tables, range.getKey() + ":" + range.getValue()); }; } diff --git a/test/simulator/main/org/apache/cassandra/simulator/debug/SelfReconcile.java b/test/simulator/main/org/apache/cassandra/simulator/debug/SelfReconcile.java index e00924cd27..ce5b6d7214 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/debug/SelfReconcile.java +++ b/test/simulator/main/org/apache/cassandra/simulator/debug/SelfReconcile.java @@ -21,7 +21,6 @@ package org.apache.cassandra.simulator.debug; import java.util.ArrayList; import java.util.List; import java.util.concurrent.ExecutorService; -import java.util.concurrent.Future; import java.util.function.Supplier; import java.util.regex.Pattern; @@ -35,6 +34,7 @@ import org.apache.cassandra.simulator.ClusterSimulation; import org.apache.cassandra.simulator.OrderOn; import org.apache.cassandra.simulator.RandomSource; import org.apache.cassandra.simulator.Simulation; +import org.apache.cassandra.simulator.SimulationException; import org.apache.cassandra.simulator.SimulationRunner.RecordOption; import org.apache.cassandra.simulator.systems.InterceptedExecution; import org.apache.cassandra.simulator.systems.InterceptedWait; @@ -45,6 +45,7 @@ import org.apache.cassandra.simulator.systems.SimulatedTime; import org.apache.cassandra.utils.CloseableIterator; import org.apache.cassandra.utils.FBUtilities; import org.apache.cassandra.utils.Pair; +import org.apache.cassandra.utils.concurrent.SyncPromise; import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException; import org.apache.cassandra.utils.memory.HeapPool; @@ -267,7 +268,8 @@ public class SelfReconcile InterceptibleThread.setDebugInterceptor(reconciler); reconciler.verifyUninterceptedRng = true; - Future f1 = executor.submit(() -> { + SyncPromise p1 = new SyncPromise<>(); + executor.execute(() -> { try (Simulation simulation = cluster1.simulation(); CloseableIterator iter = simulation.iterator()) { @@ -279,11 +281,20 @@ public class SelfReconcile } catch (Exception e) { - throw new RuntimeException(e); + p1.setFailure(new RuntimeException(e)); + } + try + { + reconciler.verify("done"); + p1.setSuccess(null); + } + catch (Throwable t) + { + p1.setFailure(t); } - reconciler.verify("done"); }); - Future f2 = executor.submit(() -> { + SyncPromise p2 = new SyncPromise<>(); + executor.execute(() -> { try (Simulation simulation = cluster2.simulation(); CloseableIterator iter = simulation.iterator()) { @@ -295,12 +306,25 @@ public class SelfReconcile } catch (Exception e) { - throw new RuntimeException(e); + p2.setFailure(new RuntimeException(e)); + } + try + { + reconciler.verify("done"); + p2.setSuccess(null); + } + catch (Throwable t) + { + p2.setFailure(t); } - reconciler.verify("done"); }); - f1.get(); - f2.get(); + p1.get(); + p2.get(); + } + catch (Throwable t) + { + logger.error("Failed on seed 0x{}", Long.toHexString(seed), t); + throw new SimulationException(seed, t); } finally { @@ -309,8 +333,8 @@ public class SelfReconcile } catch (Throwable t) { - t.printStackTrace(); - throw new RuntimeException("Failed on seed " + Long.toHexString(seed), t); + if (t instanceof SimulationException) throw (SimulationException)t; + throw new SimulationException(seed, "Failure creating the simulation", t); } } diff --git a/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordClusterSimulation.java b/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordClusterSimulation.java index 24e038b3b3..4506c2de40 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordClusterSimulation.java +++ b/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordClusterSimulation.java @@ -46,6 +46,7 @@ class AccordClusterSimulation extends ClusterSimulation impleme config -> config.set("storage_compatibility_mode", "NONE"), (simulated, schedulers, cluster, options) -> { int[] primaryKeys = primaryKeys(seed, builder.primaryKeyCount()); + builder.transactionalMode("full"); KindOfSequence.Period jitter = RandomSource.Choices.uniform(KindOfSequence.values()).choose(random) .period(builder.schedulerJitterNanos(), random); return new PairOfSequencesAccordSimulation(simulated, cluster, options, builder.transactionalMode(), diff --git a/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordSimulationRunner.java b/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordSimulationRunner.java index 147cd3ce59..2f11720f57 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordSimulationRunner.java +++ b/test/simulator/main/org/apache/cassandra/simulator/paxos/AccordSimulationRunner.java @@ -23,9 +23,6 @@ import java.util.concurrent.atomic.AtomicInteger; import org.junit.BeforeClass; -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; - import io.airlift.airline.Cli; import io.airlift.airline.Command; import org.apache.cassandra.config.CassandraRelevantProperties; @@ -36,8 +33,6 @@ import org.apache.cassandra.utils.StorageCompatibilityMode; public class AccordSimulationRunner extends SimulationRunner { - private static Logger logger = LoggerFactory.getLogger(AccordSimulationRunner.class); - @BeforeClass public static void beforeAll() { @@ -94,7 +89,7 @@ public class AccordSimulationRunner extends SimulationRunner */ public static void main(String[] args) throws IOException { - SimulatorUtils.verifyAndlogSimulatorArgs(logger, args); + SimulatorUtils.verifyAndlogSimulatorArgs(args); AccordClusterSimulation.Builder builder = new AccordClusterSimulation.Builder(); builder.unique(uniqueNum.getAndIncrement()); diff --git a/test/simulator/main/org/apache/cassandra/simulator/paxos/PairOfSequencesAccordSimulation.java b/test/simulator/main/org/apache/cassandra/simulator/paxos/PairOfSequencesAccordSimulation.java index 3bb911b41d..4abc65aaf7 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/paxos/PairOfSequencesAccordSimulation.java +++ b/test/simulator/main/org/apache/cassandra/simulator/paxos/PairOfSequencesAccordSimulation.java @@ -307,7 +307,7 @@ public class PairOfSequencesAccordSimulation extends AbstractPairOfSequencesPaxo } sb.append("COMMIT TRANSACTION"); - return new Query(sb.toString(), 0, ConsistencyLevel.ANY, ConsistencyLevel.ANY, binds.toArray(new Object[0])); + return new Query(sb.toString(), 0, ConsistencyLevel.QUORUM, ConsistencyLevel.QUORUM, binds.toArray(new Object[0])); } @Override diff --git a/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosClusterSimulation.java b/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosClusterSimulation.java index dc8a538b6c..cf34b074ca 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosClusterSimulation.java +++ b/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosClusterSimulation.java @@ -19,15 +19,41 @@ package org.apache.cassandra.simulator.paxos; import java.io.IOException; +import java.util.Map; + +import com.google.common.collect.ImmutableMap; import org.apache.cassandra.config.Config.PaxosVariant; import org.apache.cassandra.distributed.api.ConsistencyLevel; +import org.apache.cassandra.net.Verb; +import org.apache.cassandra.simulator.AccordNetworkScheduler; +import org.apache.cassandra.simulator.AlwaysDeliverNetworkScheduler; import org.apache.cassandra.simulator.ClusterSimulation; +import org.apache.cassandra.simulator.FutureActionScheduler; import org.apache.cassandra.simulator.RandomSource; +import org.apache.cassandra.simulator.systems.SimulatedTime; import org.apache.cassandra.simulator.utils.KindOfSequence; import static java.util.concurrent.TimeUnit.SECONDS; import static org.apache.cassandra.distributed.api.ConsistencyLevel.SERIAL; +import static org.apache.cassandra.net.Verb.CLEANUP_MSG; +import static org.apache.cassandra.net.Verb.FAILED_SESSION_MSG; +import static org.apache.cassandra.net.Verb.FINALIZE_COMMIT_MSG; +import static org.apache.cassandra.net.Verb.FINALIZE_PROMISE_MSG; +import static org.apache.cassandra.net.Verb.FINALIZE_PROPOSE_MSG; +import static org.apache.cassandra.net.Verb.PAXOS2_UPDATE_LOW_BALLOT_REQ; +import static org.apache.cassandra.net.Verb.PAXOS2_UPDATE_LOW_BALLOT_RSP; +import static org.apache.cassandra.net.Verb.PREPARE_CONSISTENT_REQ; +import static org.apache.cassandra.net.Verb.PREPARE_CONSISTENT_RSP; +import static org.apache.cassandra.net.Verb.PREPARE_MSG; +import static org.apache.cassandra.net.Verb.REPAIR_RSP; +import static org.apache.cassandra.net.Verb.SNAPSHOT_MSG; +import static org.apache.cassandra.net.Verb.STATUS_REQ; +import static org.apache.cassandra.net.Verb.STATUS_RSP; +import static org.apache.cassandra.net.Verb.SYNC_REQ; +import static org.apache.cassandra.net.Verb.SYNC_RSP; +import static org.apache.cassandra.net.Verb.VALIDATION_REQ; +import static org.apache.cassandra.net.Verb.VALIDATION_RSP; class PaxosClusterSimulation extends ClusterSimulation implements AutoCloseable { @@ -69,6 +95,40 @@ class PaxosClusterSimulation extends ClusterSimulation implemen random.reset(seed); return new PaxosClusterSimulation(random, seed, uniqueNum, this); } + + @Override + public Map perVerbFutureActionSchedulers(int nodeCount, SimulatedTime time, RandomSource random, FutureActionScheduler defaultScheduler) + { + // Mark just the verbs for repair reliable so that other things can continue to be unreliable while repair runs + AlwaysDeliverNetworkScheduler scheduler = new AlwaysDeliverNetworkScheduler(time); + AccordNetworkScheduler accordScheduler = new AccordNetworkScheduler(defaultScheduler, scheduler); + ImmutableMap.Builder builder = ImmutableMap.builder(); + Verb[] repairVerbs = new Verb[] { + REPAIR_RSP, + VALIDATION_RSP, + VALIDATION_REQ, + SYNC_RSP, + SYNC_REQ, + PREPARE_MSG, + SNAPSHOT_MSG, + CLEANUP_MSG, + PREPARE_CONSISTENT_RSP, + PREPARE_CONSISTENT_REQ, + FINALIZE_PROPOSE_MSG, + FINALIZE_PROMISE_MSG, + FINALIZE_COMMIT_MSG, + FAILED_SESSION_MSG, + STATUS_RSP, + STATUS_REQ, + PAXOS2_UPDATE_LOW_BALLOT_REQ, + PAXOS2_UPDATE_LOW_BALLOT_RSP + }; + for (Verb verb : repairVerbs) + builder.put(verb, scheduler); + for (Verb verb : AccordNetworkScheduler.ACCORD_VERBS) + builder.put(verb, accordScheduler); + return builder.build(); + } } PaxosClusterSimulation(RandomSource random, long seed, int uniqueNum, Builder builder) throws IOException diff --git a/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosRepairValidator.java b/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosRepairValidator.java index 572b8ff7cd..5b8482db9e 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosRepairValidator.java +++ b/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosRepairValidator.java @@ -20,6 +20,7 @@ package org.apache.cassandra.simulator.paxos; import org.apache.cassandra.distributed.Cluster; import org.apache.cassandra.simulator.cluster.ClusterActionListener.RepairValidator; +import org.apache.cassandra.simulator.cluster.OnInstanceRepair.RepairType; import org.apache.cassandra.simulator.cluster.Topology; import static java.util.Arrays.stream; @@ -45,12 +46,12 @@ public class PaxosRepairValidator implements RepairValidator } @Override - public void before(Topology topology, boolean repairPaxos, boolean repairOnlyPaxos) + public void before(Topology topology, RepairType repairType) { - if (repairOnlyPaxos) + if (repairType == RepairType.PAXOS_FULL) return; - this.isPaxos = isPaxos; + this.isPaxos = repairType.repairPaxos; this.topology = topology; this.ballotsBefore = Ballots.read(REQUIRED, cluster, keyspace, table, topology.primaryKeys, topology.replicasForKeys, false); } diff --git a/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosSimulationRunner.java b/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosSimulationRunner.java index 7c9062d3d5..4a9389ebca 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosSimulationRunner.java +++ b/test/simulator/main/org/apache/cassandra/simulator/paxos/PaxosSimulationRunner.java @@ -19,13 +19,9 @@ package org.apache.cassandra.simulator.paxos; import java.io.IOException; -import java.util.Objects; import java.util.Optional; import java.util.concurrent.atomic.AtomicInteger; -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; - import io.airlift.airline.Cli; import io.airlift.airline.Command; import io.airlift.airline.Option; @@ -33,12 +29,9 @@ import org.apache.cassandra.config.Config; import org.apache.cassandra.distributed.api.ConsistencyLevel; import org.apache.cassandra.simulator.SimulationRunner; import org.apache.cassandra.simulator.SimulatorUtils; -import org.apache.cassandra.simulator.utils.IntRange; public class PaxosSimulationRunner extends SimulationRunner { - private static Logger logger = LoggerFactory.getLogger(PaxosSimulationRunner.class); - @Command(name = "run") public static class Run extends SimulationRunner.Run { @@ -69,12 +62,6 @@ public class PaxosSimulationRunner extends SimulationRunner @Override protected void run( long seed, PaxosClusterSimulation.Builder builder) throws IOException { - if (!Objects.equals(builder.transactionalMode(), "off")) - { - // Apply handicaps - builder.dcs(new IntRange(1, 1)); - builder.nodes(new IntRange(3, 3)); - } super.run(seed, builder); } } @@ -156,7 +143,7 @@ public class PaxosSimulationRunner extends SimulationRunner */ public static void main(String[] args) throws IOException { - SimulatorUtils.verifyAndlogSimulatorArgs(logger, args); + SimulatorUtils.verifyAndlogSimulatorArgs(args); PaxosClusterSimulation.Builder builder = new PaxosClusterSimulation.Builder(); builder.unique(uniqueNum.getAndIncrement()); diff --git a/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedAction.java b/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedAction.java index b30626136b..9943110de1 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedAction.java +++ b/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedAction.java @@ -42,6 +42,7 @@ import org.apache.cassandra.simulator.ActionList; import org.apache.cassandra.simulator.Actions; import org.apache.cassandra.simulator.FutureActionScheduler; import org.apache.cassandra.simulator.FutureActionScheduler.Deliver; +import org.apache.cassandra.simulator.FutureActionScheduler.DeliverResult; import org.apache.cassandra.simulator.OrderOn; import org.apache.cassandra.simulator.systems.InterceptedExecution.InterceptedRunnableExecution; import org.apache.cassandra.simulator.systems.InterceptedWait.Trigger; @@ -61,6 +62,7 @@ import static org.apache.cassandra.simulator.Action.Modifiers.START_THREAD; import static org.apache.cassandra.simulator.Action.Modifiers.START_TIMEOUT_TASK; import static org.apache.cassandra.simulator.Action.Modifiers.WAKE_UP_THREAD; import static org.apache.cassandra.simulator.Debug.Info.LOG; +import static org.apache.cassandra.simulator.FutureActionScheduler.DELIVER_UNPROTECTED_RESULT; import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.DELIVER; import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.DELIVER_AND_TIMEOUT; import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.FAILURE; @@ -343,7 +345,11 @@ public abstract class SimulatedAction extends Action implements InterceptorOfCon boolean isReliable = is(Modifier.RELIABLE) || self.is(Modifier.RELIABLE); FutureActionScheduler childScheduler = simulated.perVerbFutureSchedulers.getOrDefault(Verb.fromId(message.verb()), simulated.futureScheduler); - Deliver deliver = isReliable ? DELIVER : childScheduler.shouldDeliver(fromNum, toNum); + + DeliverResult deliverResult = isReliable ? DELIVER_UNPROTECTED_RESULT : childScheduler.shouldDeliver(fromNum, toNum, from, message); + Deliver deliver = deliverResult.deliver; + // A message that should get a better than normal treatment in terms of unreliability + boolean protectedMessage = deliverResult.protectedMessage; List actions = new ArrayList<>(deliver == DELIVER_AND_TIMEOUT ? 2 : 1); switch (deliver) @@ -358,7 +364,7 @@ public abstract class SimulatedAction extends Action implements InterceptorOfCon Object description = lazy(() -> String.format("%s(%d) from %s to %s", verb, message.id(), message.from(), to.broadcastAddress())); OrderOn orderOn = task.executor.orderAppliesAfterScheduling(); Action action = applyTo(description, MESSAGE, orderOn, self, verb, task); - long deadlineNanos = childScheduler.messageDeadlineNanos(fromNum, toNum); + long deadlineNanos = childScheduler.messageDeadlineNanos(fromNum, toNum, protectedMessage); if (deliver == DELIVER && deadlineNanos >= expiresAtNanos) { if (isReliable) deadlineNanos = verb.isResponse() ? expiresAtNanos : expiresAtNanos / 2; @@ -412,7 +418,7 @@ public abstract class SimulatedAction extends Action implements InterceptorOfCon { default: throw new AssertionError(); case FAILURE: - long deadlineNanos = childScheduler.messageFailureNanos(toNum, fromNum); + long deadlineNanos = childScheduler.messageFailureNanos(toNum, fromNum, protectedMessage); if (deadlineNanos < expiresAtNanos) { action.setDeadline(simulated.time, deadlineNanos); @@ -421,7 +427,7 @@ public abstract class SimulatedAction extends Action implements InterceptorOfCon case DELIVER_AND_TIMEOUT: case TIMEOUT: long expirationIntervalNanos = from.unsafeCallOnThisThread(RequestCallbacks::defaultExpirationInterval); - action.setDeadline(simulated.time, childScheduler.messageTimeoutNanos(expiresAtNanos, expirationIntervalNanos)); + action.setDeadline(simulated.time, childScheduler.messageTimeoutNanos(expiresAtNanos, expirationIntervalNanos, protectedMessage)); break; } actions.add(action); diff --git a/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedFutureActionScheduler.java b/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedFutureActionScheduler.java index f66999bb5a..ccf234225a 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedFutureActionScheduler.java +++ b/test/simulator/main/org/apache/cassandra/simulator/systems/SimulatedFutureActionScheduler.java @@ -20,6 +20,8 @@ package org.apache.cassandra.simulator.systems; import java.util.BitSet; +import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.distributed.api.IMessage; import org.apache.cassandra.simulator.FutureActionScheduler; import org.apache.cassandra.simulator.RandomSource; import org.apache.cassandra.simulator.cluster.Topology; @@ -31,11 +33,6 @@ import org.apache.cassandra.simulator.utils.KindOfSequence.LinkLatency; import org.apache.cassandra.simulator.utils.KindOfSequence.NetworkDecision; import org.apache.cassandra.simulator.utils.KindOfSequence.Period; -import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.DELIVER; -import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.DELIVER_AND_TIMEOUT; -import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.FAILURE; -import static org.apache.cassandra.simulator.FutureActionScheduler.Deliver.TIMEOUT; - public class SimulatedFutureActionScheduler implements FutureActionScheduler, TopologyListener { static class Network @@ -142,27 +139,27 @@ public class SimulatedFutureActionScheduler implements FutureActionScheduler, To } @Override - public FutureActionScheduler.Deliver shouldDeliver(int from, int to) + public DeliverResult shouldDeliver(int from, int to, IInvokableInstance invoker, IMessage message) { Network config = config(from, to); if (isInDropPartition.get(from) != isInDropPartition.get(to)) - return TIMEOUT; + return TIMEOUT_RESULT; if (!config.dropMessage.get(random, from, to)) - return DELIVER; + return DELIVER_UNPROTECTED_RESULT; if (random.decide(0.5f)) - return DELIVER_AND_TIMEOUT; + return DELIVER_AND_TIMEOUT_RESULT; if (random.decide(0.5f)) - return TIMEOUT; + return TIMEOUT_RESULT; - return FAILURE; + return FAILURE_RESULT; } @Override - public long messageDeadlineNanos(int from, int to) + public long messageDeadlineNanos(int from, int to, boolean protectedMessage) { Network config = config(from, to); return time.nanoTime() + (config.delayMessage.get(random, from, to) @@ -171,15 +168,15 @@ public class SimulatedFutureActionScheduler implements FutureActionScheduler, To } @Override - public long messageTimeoutNanos(long expiresAtNanos, long expirationIntervalNanos) + public long messageTimeoutNanos(long expiresAtNanos, long expirationIntervalNanos, boolean protectedMessage) { return expiresAtNanos + random.uniform(0, expirationIntervalNanos / 2); } @Override - public long messageFailureNanos(int from, int to) + public long messageFailureNanos(int from, int to, boolean protectedMessage) { - return messageDeadlineNanos(from, to); + return messageDeadlineNanos(from, to, protectedMessage); } @Override diff --git a/test/simulator/main/org/apache/cassandra/simulator/utils/KindOfSequence.java b/test/simulator/main/org/apache/cassandra/simulator/utils/KindOfSequence.java index 367dea04b9..dfcd621a2b 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/utils/KindOfSequence.java +++ b/test/simulator/main/org/apache/cassandra/simulator/utils/KindOfSequence.java @@ -104,6 +104,17 @@ public enum KindOfSequence else if (step > 0 && cur >= target) next(random); return result; } + + @Override + public String toString() + { + return "UniformStepPeriod{" + + "range=" + range + + ", target=" + target + + ", cur=" + cur + + ", step=" + step + + '}'; + } } static class RandomWalkPeriod implements Period diff --git a/test/simulator/main/org/apache/cassandra/simulator/utils/LongRange.java b/test/simulator/main/org/apache/cassandra/simulator/utils/LongRange.java index 92c4051239..8c06420dae 100644 --- a/test/simulator/main/org/apache/cassandra/simulator/utils/LongRange.java +++ b/test/simulator/main/org/apache/cassandra/simulator/utils/LongRange.java @@ -91,4 +91,12 @@ public class LongRange return parse.substring(0, parse.length() - (units == SECONDS ? 1 : 2)); } + @Override + public String toString() + { + return "LongRange{" + + "min=" + min + + ", max=" + max + + '}'; + } } diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/HarrySimulatorTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/HarrySimulatorTest.java index 9bc9b6e759..3682d28cba 100644 --- a/test/simulator/test/org/apache/cassandra/simulator/test/HarrySimulatorTest.java +++ b/test/simulator/test/org/apache/cassandra/simulator/test/HarrySimulatorTest.java @@ -523,7 +523,7 @@ public class HarrySimulatorTest } @Override - public Map perVerbFutureActionSchedulers(int nodeCount, SimulatedTime time, RandomSource random) + public Map perVerbFutureActionSchedulers(int nodeCount, SimulatedTime time, RandomSource random, FutureActionScheduler defaultScheduler) { return networkSchedulers(nodeCount, time, random); } diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/ShortAccordSimulationTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/ShortAccordSimulationTest.java index cb83160be8..fdf53fd3e2 100644 --- a/test/simulator/test/org/apache/cassandra/simulator/test/ShortAccordSimulationTest.java +++ b/test/simulator/test/org/apache/cassandra/simulator/test/ShortAccordSimulationTest.java @@ -71,7 +71,6 @@ import org.apache.cassandra.simulator.paxos.AccordSimulationRunner; -XX:Tier4CompileThreshold=1000 -XX:ReservedCodeCacheSize=256M -Xmx16G - -Xmx4G --add-exports java.base/jdk.internal.misc=ALL-UNNAMED --add-exports java.base/jdk.internal.ref=ALL-UNNAMED --add-exports java.base/sun.nio.ch=ALL-UNNAMED diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationFromAccordSimulationTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationFromAccordSimulationTest.java new file mode 100644 index 0000000000..97fb59dc61 --- /dev/null +++ b/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationFromAccordSimulationTest.java @@ -0,0 +1,113 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.simulator.test; + +import java.io.IOException; + +import org.junit.Ignore; +import org.junit.Test; + +import org.apache.cassandra.simulator.paxos.PaxosSimulationRunner; + +/** + * In order to run these tests in your IDE, you need to first build a simulator jara + * + * ant simulator-jars + * + * And then run your test using the following settings (omit add-* if you are running on jdk8): + * + -Dstorage-config=$MODULE_DIR$/test/conf + -Djava.awt.headless=true + -javaagent:$MODULE_DIR$/lib/jamm-0.4.0.jar + -ea + -Dcassandra.debugrefcount=true + -Xss384k + -XX:SoftRefLRUPolicyMSPerMB=0 + -XX:ActiveProcessorCount=2 + -XX:HeapDumpPath=build/test + -Dcassandra.test.driver.connection_timeout_ms=10000 + -Dcassandra.test.driver.read_timeout_ms=24000 + -Dcassandra.memtable_row_overhead_computation_step=100 + -Dcassandra.test.use_prepared=true + -Dcassandra.test.sstableformatdevelopment=true + -Djava.security.egd=file:/dev/urandom + -Dcassandra.testtag=.jdk11 + -Dcassandra.keepBriefBrief=true + -Dcassandra.allow_simplestrategy=true + -Dcassandra.strict.runtime.checks=true + -Dcassandra.reads.thresholds.coordinator.defensive_checks_enabled=true + -Dcassandra.test.flush_local_schema_changes=false + -Dcassandra.test.messagingService.nonGracefulShutdown=true + -Dcassandra.use_nix_recursive_delete=true + -Dcie-cassandra.disable_schema_drop_log=true + -Dlogback.configurationFile=file://$MODULE_DIR$/test/conf/logback-simulator.xml + -Dcassandra.ring_delay_ms=10000 + -Dcassandra.tolerate_sstable_size=true + -Dcassandra.skip_sync=true + -Dcassandra.debugrefcount=false + -Dcassandra.test.simulator.determinismcheck=strict + -Dcassandra.test.simulator.print_asm=none + -javaagent:$MODULE_DIR$/build/test/lib/jars/simulator-asm.jar + -Xbootclasspath/a:$MODULE_DIR$/build/test/lib/jars/simulator-bootstrap.jar + -XX:ActiveProcessorCount=4 + -XX:-TieredCompilation + -XX:-BackgroundCompilation + -XX:CICompilerCount=1 + -XX:Tier4CompileThreshold=1000 + -XX:ReservedCodeCacheSize=256M + -Xmx16G + --add-exports java.base/jdk.internal.misc=ALL-UNNAMED + --add-exports java.base/jdk.internal.ref=ALL-UNNAMED + --add-exports java.base/sun.nio.ch=ALL-UNNAMED + --add-exports java.management.rmi/com.sun.jmx.remote.internal.rmi=ALL-UNNAMED + --add-exports java.rmi/sun.rmi.registry=ALL-UNNAMED + --add-exports java.rmi/sun.rmi.server=ALL-UNNAMED + --add-exports java.sql/java.sql=ALL-UNNAMED + --add-exports java.rmi/sun.rmi.registry=ALL-UNNAMED + --add-opens java.base/java.lang.module=ALL-UNNAMED + --add-opens java.base/java.net=ALL-UNNAMED + --add-opens java.base/jdk.internal.loader=ALL-UNNAMED + --add-opens java.base/jdk.internal.ref=ALL-UNNAMED + --add-opens java.base/jdk.internal.reflect=ALL-UNNAMED + --add-opens java.base/jdk.internal.math=ALL-UNNAMED + --add-opens java.base/jdk.internal.module=ALL-UNNAMED + --add-opens java.base/jdk.internal.util.jar=ALL-UNNAMED + --add-opens jdk.management/com.sun.management.internal=ALL-UNNAMED + --add-opens jdk.management.jfr/jdk.management.jfr=ALL-UNNAMED + --add-opens java.desktop/com.sun.beans.introspect=ALL-UNNAMED + */ +@Ignore +public class ShortPaxosMigrationFromAccordSimulationTest +{ + @Test + @Ignore + public void casOnAccordSimulationTestAccordStart() throws IOException + { + PaxosSimulationRunner.main(new String[] { "run", + "--transactional-mode", "full", + "--seed", "0x2b091cc62b96a2eb", + "-n", "3..6", + "-t", "1000", + "--cluster-action-limit", "-1", + "--consensus-action-limit", "1", + "--consensus-actions", "ACCORD_MIGRATE", + "-c", "2"}); + } +} + diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationToAccordSimulationTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationToAccordSimulationTest.java new file mode 100644 index 0000000000..dab2482f36 --- /dev/null +++ b/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosMigrationToAccordSimulationTest.java @@ -0,0 +1,111 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.simulator.test; + +import java.io.IOException; + +import org.junit.Ignore; +import org.junit.Test; + +import org.apache.cassandra.simulator.paxos.PaxosSimulationRunner; + +/** + * In order to run these tests in your IDE, you need to first build a simulator jara + * + * ant simulator-jars + * + * And then run your test using the following settings (omit add-* if you are running on jdk8): + * + -Dstorage-config=$MODULE_DIR$/test/conf + -Djava.awt.headless=true + -javaagent:$MODULE_DIR$/lib/jamm-0.4.0.jar + -ea + -Dcassandra.debugrefcount=true + -Xss384k + -XX:SoftRefLRUPolicyMSPerMB=0 + -XX:ActiveProcessorCount=2 + -XX:HeapDumpPath=build/test + -Dcassandra.test.driver.connection_timeout_ms=10000 + -Dcassandra.test.driver.read_timeout_ms=24000 + -Dcassandra.memtable_row_overhead_computation_step=100 + -Dcassandra.test.use_prepared=true + -Dcassandra.test.sstableformatdevelopment=true + -Djava.security.egd=file:/dev/urandom + -Dcassandra.testtag=.jdk11 + -Dcassandra.keepBriefBrief=true + -Dcassandra.allow_simplestrategy=true + -Dcassandra.strict.runtime.checks=true + -Dcassandra.reads.thresholds.coordinator.defensive_checks_enabled=true + -Dcassandra.test.flush_local_schema_changes=false + -Dcassandra.test.messagingService.nonGracefulShutdown=true + -Dcassandra.use_nix_recursive_delete=true + -Dcie-cassandra.disable_schema_drop_log=true + -Dlogback.configurationFile=file://$MODULE_DIR$/test/conf/logback-simulator.xml + -Dcassandra.ring_delay_ms=10000 + -Dcassandra.tolerate_sstable_size=true + -Dcassandra.skip_sync=true + -Dcassandra.debugrefcount=false + -Dcassandra.test.simulator.determinismcheck=strict + -Dcassandra.test.simulator.print_asm=none + -javaagent:$MODULE_DIR$/build/test/lib/jars/simulator-asm.jar + -Xbootclasspath/a:$MODULE_DIR$/build/test/lib/jars/simulator-bootstrap.jar + -XX:ActiveProcessorCount=4 + -XX:-TieredCompilation + -XX:-BackgroundCompilation + -XX:CICompilerCount=1 + -XX:Tier4CompileThreshold=1000 + -XX:ReservedCodeCacheSize=256M + -Xmx16G + --add-exports java.base/jdk.internal.misc=ALL-UNNAMED + --add-exports java.base/jdk.internal.ref=ALL-UNNAMED + --add-exports java.base/sun.nio.ch=ALL-UNNAMED + --add-exports java.management.rmi/com.sun.jmx.remote.internal.rmi=ALL-UNNAMED + --add-exports java.rmi/sun.rmi.registry=ALL-UNNAMED + --add-exports java.rmi/sun.rmi.server=ALL-UNNAMED + --add-exports java.sql/java.sql=ALL-UNNAMED + --add-exports java.rmi/sun.rmi.registry=ALL-UNNAMED + --add-opens java.base/java.lang.module=ALL-UNNAMED + --add-opens java.base/java.net=ALL-UNNAMED + --add-opens java.base/jdk.internal.loader=ALL-UNNAMED + --add-opens java.base/jdk.internal.ref=ALL-UNNAMED + --add-opens java.base/jdk.internal.reflect=ALL-UNNAMED + --add-opens java.base/jdk.internal.math=ALL-UNNAMED + --add-opens java.base/jdk.internal.module=ALL-UNNAMED + --add-opens java.base/jdk.internal.util.jar=ALL-UNNAMED + --add-opens jdk.management/com.sun.management.internal=ALL-UNNAMED + --add-opens jdk.management.jfr/jdk.management.jfr=ALL-UNNAMED + --add-opens java.desktop/com.sun.beans.introspect=ALL-UNNAMED + */ +public class ShortPaxosMigrationToAccordSimulationTest +{ + @Test + @Ignore + public void casOnAccordSimulationTestPaxosStart() throws IOException + { + PaxosSimulationRunner.main(new String[] { "run", + "--transactional-mode", "off", + "-n", "3..6", + "-t", "1000", + "--cluster-action-limit", "-1", + "--consensus-action-limit", "1", + "--consensus-actions", "ACCORD_MIGRATE", + "-c", "2"}); + } +} + diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosSimulationTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosSimulationTest.java index e83645bfae..d726c25e94 100644 --- a/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosSimulationTest.java +++ b/test/simulator/test/org/apache/cassandra/simulator/test/ShortPaxosSimulationTest.java @@ -72,7 +72,6 @@ import org.apache.cassandra.simulator.paxos.PaxosSimulationRunner; -XX:Tier4CompileThreshold=1000 -XX:ReservedCodeCacheSize=256M -Xmx16G - -Xmx4G --add-exports java.base/jdk.internal.misc=ALL-UNNAMED --add-exports java.base/jdk.internal.ref=ALL-UNNAMED --add-exports java.base/sun.nio.ch=ALL-UNNAMED @@ -101,20 +100,6 @@ public class ShortPaxosSimulationTest PaxosSimulationRunner.main(new String[] { "run", "--variant", "v2", "-n", "3..6", "-t", "1000", "-c", "2", "--cluster-action-limit", "2", "-s", "30" }); } - @Test - public void casOnAccordSimulationTest() throws IOException - { - PaxosSimulationRunner.main(new String[] { "run", - "--transactional-mode", "full", - "-n", "3...6", - "-t", "1000", - "--cluster-action-limit", "0", - "--consensus-action-limit", "0", - "--consensus-actions", "ACCORD_MIGRATE", - "-c", "10", - "-s", "30"}); - } - @Test @Ignore("fails due to OOM DirectMemory - unclear why") public void selfReconcileTest() throws IOException diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/SimulationTestBase.java b/test/simulator/test/org/apache/cassandra/simulator/test/SimulationTestBase.java index 406725bb1c..0c1088b6eb 100644 --- a/test/simulator/test/org/apache/cassandra/simulator/test/SimulationTestBase.java +++ b/test/simulator/test/org/apache/cassandra/simulator/test/SimulationTestBase.java @@ -29,19 +29,29 @@ import java.util.function.Predicate; import com.google.common.collect.Iterators; import org.junit.BeforeClass; -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; import org.apache.cassandra.concurrent.ExecutorFactory; import org.apache.cassandra.config.CassandraRelevantProperties; import org.apache.cassandra.distributed.Cluster; import org.apache.cassandra.distributed.api.ConsistencyLevel; +import org.apache.cassandra.distributed.api.IInvokableInstance; import org.apache.cassandra.distributed.api.IIsolatedExecutor; +import org.apache.cassandra.distributed.api.IMessage; import org.apache.cassandra.distributed.impl.AbstractCluster; import org.apache.cassandra.distributed.impl.IsolatedExecutor; import org.apache.cassandra.distributed.shared.InstanceClassLoader; -import org.apache.cassandra.simulator.*; +import org.apache.cassandra.simulator.Action; +import org.apache.cassandra.simulator.ActionList; +import org.apache.cassandra.simulator.ActionPlan; +import org.apache.cassandra.simulator.ActionSchedule; import org.apache.cassandra.simulator.ActionSchedule.Work; +import org.apache.cassandra.simulator.ClusterSimulation; +import org.apache.cassandra.simulator.Debug; +import org.apache.cassandra.simulator.FutureActionScheduler; +import org.apache.cassandra.simulator.RandomSource; +import org.apache.cassandra.simulator.RunnableActionScheduler; +import org.apache.cassandra.simulator.Simulation; +import org.apache.cassandra.simulator.SimulationRunner; import org.apache.cassandra.simulator.asm.InterceptClasses; import org.apache.cassandra.simulator.asm.NemesisFieldSelectors; import org.apache.cassandra.simulator.systems.Failures; @@ -85,7 +95,8 @@ public class SimulationTestBase try { Clock.Global.nanoTime(); } catch (IllegalStateException e) {} // make sure static initializer gets called } - private static final Logger logger = LoggerFactory.getLogger(Logger.class); + // Don't use loggers before invoking simulator it messes up initialization order +// private static final Logger logger = LoggerFactory.getLogger(Logger.class); static abstract class DTestClusterSimulation implements Simulation { @@ -206,7 +217,7 @@ public class SimulationTestBase long seed = System.currentTimeMillis(); // Development seed: //long seed = 1687184561194L; - logger.info("Simulation seed: {}L", seed); + System.out.printf("Simulation seed: %dL%n", seed); configure.accept(factory); try (ClusterSimulation cluster = factory.create(seed)) { @@ -277,25 +288,25 @@ public class SimulationTestBase SimulatedSystems simulated = new SimulatedSystems(random, time, null, execution, null, null, null, new FutureActionScheduler() { @Override - public Deliver shouldDeliver(int from, int to) + public DeliverResult shouldDeliver(int from, int to, IInvokableInstance invoker, IMessage message) { - return Deliver.DELIVER; + return DELIVER_UNPROTECTED_RESULT; } @Override - public long messageDeadlineNanos(int from, int to) + public long messageDeadlineNanos(int from, int to, boolean protectedMessage) { return 0; } @Override - public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos) + public long messageTimeoutNanos(long expiresAfterNanos, long expirationIntervalNanos, boolean protectedMessage) { return 0; } @Override - public long messageFailureNanos(int from, int to) + public long messageFailureNanos(int from, int to, boolean protectedMessage) { return 0; } diff --git a/test/unit/org/apache/cassandra/audit/AuditLoggerTest.java b/test/unit/org/apache/cassandra/audit/AuditLoggerTest.java index ee0ad12bcf..311c08b93f 100644 --- a/test/unit/org/apache/cassandra/audit/AuditLoggerTest.java +++ b/test/unit/org/apache/cassandra/audit/AuditLoggerTest.java @@ -34,7 +34,6 @@ import javax.management.remote.JMXConnector; import javax.management.remote.JMXConnectorFactory; import javax.management.remote.JMXServiceURL; -import org.assertj.core.api.Assertions; import org.junit.After; import org.junit.Assert; import org.junit.Before; @@ -45,6 +44,8 @@ import com.datastax.driver.core.BatchStatement; import com.datastax.driver.core.PreparedStatement; import com.datastax.driver.core.ResultSet; import com.datastax.driver.core.Session; +import com.datastax.driver.core.SimpleStatement; +import com.datastax.driver.core.Statement; import com.datastax.driver.core.exceptions.NoHostAvailableException; import com.datastax.driver.core.exceptions.SyntaxError; import net.openhft.chronicle.queue.RollCycles; @@ -65,7 +66,9 @@ import org.apache.cassandra.db.ColumnFamilyStoreMBean; import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.service.StorageService; import org.apache.cassandra.utils.JMXServerUtils; +import org.assertj.core.api.Assertions; +import static com.datastax.driver.core.ConsistencyLevel.QUORUM; import static org.apache.cassandra.config.CassandraRelevantProperties.CASSANDRA_JMX_AUTHORIZER; import static org.apache.cassandra.config.CassandraRelevantProperties.CASSANDRA_JMX_LOCAL_PORT; import static org.apache.cassandra.config.CassandraRelevantProperties.CASSANDRA_JMX_REMOTE_LOGIN_CONFIG; @@ -453,8 +456,9 @@ public class AuditLoggerTest extends CQLTester " INSERT INTO " + fqTableName + " (key, val) VALUES (0, 0);\n" + " END IF\n" + "COMMIT TRANSACTION"; - - session.execute(query); + Statement statement = new SimpleStatement(query); + statement.setConsistencyLevel(QUORUM); + session.execute(statement); AuditLogEntry logEntry = ((InMemoryAuditLogger) AuditLogManager.instance.getLogger()).inMemQueue.poll(); assertLogEntry(query, AuditLogEntryType.TRANSACTION, logEntry, true, null); } diff --git a/test/unit/org/apache/cassandra/auth/TxnAuthTest.java b/test/unit/org/apache/cassandra/auth/TxnAuthTest.java index e26e025951..2b291ab148 100644 --- a/test/unit/org/apache/cassandra/auth/TxnAuthTest.java +++ b/test/unit/org/apache/cassandra/auth/TxnAuthTest.java @@ -21,8 +21,6 @@ package org.apache.cassandra.auth; import java.net.InetSocketAddress; import java.util.Collections; -import org.apache.cassandra.transport.Dispatcher; -import org.assertj.core.api.Assertions; import org.junit.Before; import org.junit.BeforeClass; import org.junit.Test; @@ -36,12 +34,14 @@ import org.apache.cassandra.cql3.statements.TransactionStatement; import org.apache.cassandra.exceptions.UnauthorizedException; import org.apache.cassandra.service.ClientState; import org.apache.cassandra.service.QueryState; +import org.apache.cassandra.transport.Dispatcher; import org.apache.cassandra.transport.messages.ResultMessage; - -import static org.junit.Assert.assertEquals; +import org.assertj.core.api.Assertions; import static org.apache.cassandra.auth.AuthTestUtils.auth; import static org.apache.cassandra.db.ConsistencyLevel.NODE_LOCAL; +import static org.apache.cassandra.db.ConsistencyLevel.QUORUM; +import static org.junit.Assert.assertEquals; public class TxnAuthTest extends CQLTester { @@ -164,7 +164,7 @@ public class TxnAuthTest extends CQLTester { TransactionStatement.Parsed parsed = (TransactionStatement.Parsed) QueryProcessor.parseStatement(query); TransactionStatement statement = (TransactionStatement) parsed.prepare(clientState); - QueryOptions options = QueryOptions.forInternalCalls(NODE_LOCAL, Collections.emptyList()); + QueryOptions options = QueryOptions.forInternalCalls(QUORUM, Collections.emptyList()); QueryState queryState = new QueryState(clientState); return QueryProcessor.instance.process(statement, queryState, options, Dispatcher.RequestTime.forImmediateExecution()); } diff --git a/test/unit/org/apache/cassandra/cql3/PreparedStatementsTest.java b/test/unit/org/apache/cassandra/cql3/PreparedStatementsTest.java index 47cae295d8..a4430ff78f 100644 --- a/test/unit/org/apache/cassandra/cql3/PreparedStatementsTest.java +++ b/test/unit/org/apache/cassandra/cql3/PreparedStatementsTest.java @@ -816,15 +816,18 @@ public class PreparedStatementsTest extends CQLTester PreparedStatement writeOnly = session.prepare(txn( "INSERT INTO " + currentTable() + " (pk, v1, v2) VALUES (?, ?, ?)" )); + writeOnly.setConsistencyLevel(com.datastax.driver.core.ConsistencyLevel.QUORUM); PreparedStatement returnSelect = session.prepare(txn( "SELECT * FROM " + currentTable() + " WHERE pk=?", "UPDATE " + currentTable() + " SET v1 += 1, v2 += 2 WHERE pk = ?" )); + returnSelect.setConsistencyLevel(com.datastax.driver.core.ConsistencyLevel.QUORUM); PreparedStatement returnRef = session.prepare(txn( "LET a = (SELECT * FROM " + currentTable() + " WHERE pk=?)", "SELECT a.pk, a.v1, a.v2", "UPDATE " + currentTable() + " SET v1 += 1, v2 += 2 WHERE pk = ?" )); + returnRef.setConsistencyLevel(com.datastax.driver.core.ConsistencyLevel.QUORUM); // populate every row int numPartitions = 5; int[][] model = new int[numPartitions][]; diff --git a/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java b/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java index f102b169d8..05ca3adc00 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordCommandStoreTest.java @@ -61,10 +61,11 @@ import org.apache.cassandra.schema.TableId; import org.apache.cassandra.schema.TableMetadata; import org.apache.cassandra.service.StorageService; import org.apache.cassandra.service.accord.AccordKeyspace.CommandsForKeyAccessor; -import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.service.accord.api.PartitionKey; +import org.apache.cassandra.service.accord.api.TokenKey; import org.apache.cassandra.service.accord.serializers.CommandsForKeySerializerTest.TestSafeCommandStore; import org.apache.cassandra.service.accord.serializers.ResultSerializers; +import org.apache.cassandra.service.accord.txn.TxnUpdate; import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.utils.Pair; @@ -138,7 +139,9 @@ public class AccordCommandStoreTest waitingOn, result.left, ResultSerializers.APPLIED); AccordSafeCommand safeCommand = new AccordSafeCommand(loaded(txnId, null)); safeCommand.set(expected); - + // In practice we should never need to save it with the condition boolean set + // Not sure why this test does that + ((TxnUpdate)txn.update()).unsafeResetCondition(); AccordTestUtils.appendCommandsBlocking(commandStore, null, expected); logger.info("E: {}", expected); diff --git a/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java b/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java index e050d43a37..210f9c1e7e 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java @@ -57,7 +57,7 @@ public class AccordReadRepairTest extends AccordTestBase public static void setupClass() throws IOException { AccordTestBase.setupCluster(builder -> builder, 2); - SHARED_CLUSTER.setMessageSink(new MessageCountingSink(SHARED_CLUSTER)); + SHARED_CLUSTER.setMessageSink(new MessageCountingSink(SHARED_CLUSTER, MessageCountingSink.EXCLUDE_SYNC_POINT_MESSAGES)); } /* diff --git a/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTrackerTest.java b/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTrackerTest.java index c9ec812f10..556ae24b8e 100644 --- a/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTrackerTest.java +++ b/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosBallotTrackerTest.java @@ -67,6 +67,7 @@ public class PaxosBallotTrackerTest public static void setUpClass() throws Exception { SchemaLoader.prepareServer(); + DatabaseDescriptor.setAccordTransactionsEnabled(false); ks = "coordinatorsessiontest"; cfm = CreateTableStatement.parse("CREATE TABLE tbl (k INT PRIMARY KEY, v INT)", ks).build(); @@ -129,7 +130,7 @@ public class PaxosBallotTrackerTest case PROPOSE: try (PaxosState state = PaxosState.get(commit)) { - state.acceptIfLatest(commit, false); + state.acceptIfLatest(commit); } break; case COMMIT: @@ -223,7 +224,7 @@ public class PaxosBallotTrackerTest DecoratedKey key = dk(1); try (PaxosState state = PaxosState.get(key, cfm)) { - Ballot result = state.acceptIfLatest(new Commit.Proposal(ballot2, PartitionUpdate.emptyUpdate(cfm, key)), false).supersededBy; + Ballot result = state.acceptIfLatest(new Commit.Proposal(ballot2, PartitionUpdate.emptyUpdate(cfm, key))).supersededBy; Assert.assertNull(result); } @@ -231,7 +232,7 @@ public class PaxosBallotTrackerTest ballotTracker.updateLowBound(ballot4); try (PaxosState state = PaxosState.get(key, cfm)) { - Ballot result = state.acceptIfLatest(new Commit.Proposal(ballot3, PartitionUpdate.emptyUpdate(cfm, key)), false).supersededBy; + Ballot result = state.acceptIfLatest(new Commit.Proposal(ballot3, PartitionUpdate.emptyUpdate(cfm, key))).supersededBy; Assert.assertEquals(ballot4, result); } } diff --git a/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosStateTrackerTest.java b/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosStateTrackerTest.java index 2e8668db87..71d4877e57 100644 --- a/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosStateTrackerTest.java +++ b/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosStateTrackerTest.java @@ -48,6 +48,7 @@ import org.apache.cassandra.io.util.FileUtils; import org.apache.cassandra.schema.ColumnMetadata; import org.apache.cassandra.schema.KeyspaceParams; import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.paxos.Ballot; import org.apache.cassandra.service.paxos.Commit; import org.apache.cassandra.service.paxos.PaxosRepairHistory; @@ -73,6 +74,8 @@ public class PaxosStateTrackerTest public static void setUpClass() throws Exception { SchemaLoader.prepareServer(); + DatabaseDescriptor.setAccordTransactionsEnabled(false); + AccordService.startup(null); ks = "coordinatorsessiontest"; cfm1 = TableMetadata.builder(ks, "tbl1").addPartitionKeyColumn("k", Int32Type.instance).addRegularColumn("v", Int32Type.instance).build(); diff --git a/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosUncommittedTrackerIntegrationTest.java b/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosUncommittedTrackerIntegrationTest.java index dc8fafa966..135607d7e5 100644 --- a/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosUncommittedTrackerIntegrationTest.java +++ b/test/unit/org/apache/cassandra/service/paxos/uncommitted/PaxosUncommittedTrackerIntegrationTest.java @@ -100,7 +100,7 @@ public class PaxosUncommittedTrackerIntegrationTest try (PaxosState state = PaxosState.get(key, cfm)) { - state.acceptIfLatest(proposal, false); + state.acceptIfLatest(proposal); } try (CloseableIterator iterator = tracker.uncommittedKeyIterator(cfm.id, ALL_RANGES)) @@ -127,7 +127,7 @@ public class PaxosUncommittedTrackerIntegrationTest try (PaxosState state = PaxosState.get(key, cfm)) { state.promiseIfNewer(proposal.ballot, true); - state.acceptIfLatest(proposal, false); + state.acceptIfLatest(proposal); } try (CloseableIterator iterator = tracker.uncommittedKeyIterator(cfm.id, ALL_RANGES)) {