mirror of https://github.com/apache/cassandra
Compare commits
3 Commits
154b6a38c9
...
119b71981b
| Author | SHA1 | Date |
|---|---|---|
|
|
119b71981b | |
|
|
ab8fb8d9c2 | |
|
|
550dd312fb |
|
|
@ -48,7 +48,6 @@ import org.apache.cassandra.config.DatabaseDescriptor;
|
|||
import org.apache.cassandra.cql3.UntypedResultSet;
|
||||
import org.apache.cassandra.cql3.UntypedResultSet.Row;
|
||||
import org.apache.cassandra.db.ColumnFamilyStore;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.SystemKeyspace;
|
||||
|
|
@ -60,20 +59,17 @@ import org.apache.cassandra.exceptions.InvalidRequestException;
|
|||
import org.apache.cassandra.exceptions.RetryOnDifferentSystemException;
|
||||
import org.apache.cassandra.exceptions.WriteFailureException;
|
||||
import org.apache.cassandra.exceptions.WriteTimeoutException;
|
||||
import org.apache.cassandra.gms.FailureDetector;
|
||||
import org.apache.cassandra.hints.Hint;
|
||||
import org.apache.cassandra.hints.HintsService;
|
||||
import org.apache.cassandra.io.util.DataInputBuffer;
|
||||
import org.apache.cassandra.io.util.DataOutputBuffer;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaLayout;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.Replicas;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessageFlag;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.schema.SchemaConstants;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
import org.apache.cassandra.service.PreserveTimestamp;
|
||||
|
|
@ -609,22 +605,21 @@ public class BatchlogManager implements BatchlogManagerMBean
|
|||
String ks = mutation.getKeyspaceName();
|
||||
Token tk = mutation.key().getToken();
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
KeyspaceMetadata keyspaceMetadata = metadata.schema.getKeyspaceMetadata(ks);
|
||||
Keyspace keyspace = Keyspace.open(ks);
|
||||
|
||||
// TODO: this logic could do with revisiting at some point, as it is unclear what its rationale is
|
||||
// we perform a local write, ignoring errors and inline in this thread (potentially slowing replay down)
|
||||
// effectively bumping CL for locally owned writes and also potentially stalling log replay if an error occurs
|
||||
// once we decide how it should work, it can also probably be simplified, and avoid constructing a ReplicaPlan directly
|
||||
ReplicaLayout.ForTokenWrite allReplias = ReplicaLayout.forTokenWriteLiveAndDown(metadata, keyspaceMetadata, tk);
|
||||
ReplicaPlan.ForWrite replicaPlan = forReplayMutation(metadata, Keyspace.open(ks), tk);
|
||||
ReplicaLayout.ForTokenWrite allReplicas = ReplicaLayout.forTokenWriteLiveAndDown(metadata, keyspace.getMetadata(), tk);
|
||||
CoordinationPlan.ForWrite replayPlan = CoordinationPlan.forReplayMutation(metadata, keyspace, tk);
|
||||
|
||||
Replica selfReplica = allReplias.all().selfIfPresent();
|
||||
Replica selfReplica = allReplicas.all().selfIfPresent();
|
||||
if (selfReplica != null)
|
||||
mutation.apply();
|
||||
|
||||
for (Replica replica : allReplias.all())
|
||||
for (Replica replica : allReplicas.all())
|
||||
{
|
||||
if (replica == selfReplica || replicaPlan.liveAndDown().contains(replica))
|
||||
if (replica == selfReplica || replayPlan.replicas().liveAndDown().contains(replica))
|
||||
continue;
|
||||
|
||||
UUID hostId = metadata.directory.peerId(replica.endpoint()).toUUID();
|
||||
|
|
@ -635,26 +630,13 @@ public class BatchlogManager implements BatchlogManagerMBean
|
|||
}
|
||||
}
|
||||
|
||||
ReplayWriteResponseHandler<Mutation> handler = new ReplayWriteResponseHandler<>(replicaPlan, mutation, Dispatcher.RequestTime.forImmediateExecution());
|
||||
ReplayWriteResponseHandler<Mutation> handler = new ReplayWriteResponseHandler<>(replayPlan, mutation, Dispatcher.RequestTime.forImmediateExecution());
|
||||
Message<Mutation> message = Message.outWithFlag(MUTATION_REQ, mutation, MessageFlag.CALL_BACK_ON_FAILURE);
|
||||
for (Replica replica : replicaPlan.liveAndDown())
|
||||
for (Replica replica : replayPlan.replicas().liveAndDown())
|
||||
MessagingService.instance().sendWriteWithCallback(message, replica, handler);
|
||||
return handler;
|
||||
}
|
||||
|
||||
public static ReplicaPlan.ForWrite forReplayMutation(ClusterMetadata metadata, Keyspace keyspace, Token token)
|
||||
{
|
||||
ReplicaLayout.ForTokenWrite liveAndDown = ReplicaLayout.forTokenWriteLiveAndDown(metadata, keyspace.getMetadata(), token);
|
||||
Replicas.temporaryAssertFull(liveAndDown.all()); // TODO in CASSANDRA-14549
|
||||
|
||||
Replica selfReplica = liveAndDown.all().selfIfPresent();
|
||||
ReplicaLayout.ForTokenWrite liveRemoteOnly = liveAndDown.filter(r -> FailureDetector.isReplicaAlive.test(r) && r != selfReplica);
|
||||
|
||||
return new ReplicaPlan.ForWrite(keyspace, liveAndDown.replicationStrategy(),
|
||||
ConsistencyLevel.ONE, liveRemoteOnly.pending(), liveRemoteOnly.all(), liveRemoteOnly.all(), liveRemoteOnly.all(),
|
||||
(cm) -> forReplayMutation(cm, keyspace, token),
|
||||
metadata.epoch);
|
||||
}
|
||||
private static int gcgs(Collection<Mutation> mutations)
|
||||
{
|
||||
int gcgs = Integer.MAX_VALUE;
|
||||
|
|
@ -672,16 +654,10 @@ public class BatchlogManager implements BatchlogManagerMBean
|
|||
private final Set<InetAddressAndPort> undelivered = Collections.newSetFromMap(new ConcurrentHashMap<>());
|
||||
|
||||
// TODO: should we be hinting here, since presumably batch log will retry? Maintaining historical behaviour for the moment.
|
||||
ReplayWriteResponseHandler(ReplicaPlan.ForWrite replicaPlan, Supplier<Mutation> hintOnFailure, Dispatcher.RequestTime requestTime)
|
||||
ReplayWriteResponseHandler(CoordinationPlan.ForWrite coordinationPlan, Supplier<Mutation> hintOnFailure, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(replicaPlan, null, WriteType.UNLOGGED_BATCH, hintOnFailure, requestTime);
|
||||
Iterables.addAll(undelivered, replicaPlan.contacts().endpoints());
|
||||
}
|
||||
|
||||
@Override
|
||||
protected int blockFor()
|
||||
{
|
||||
return this.replicaPlan.contacts().size();
|
||||
super(coordinationPlan, null, WriteType.UNLOGGED_BATCH, hintOnFailure, requestTime);
|
||||
Iterables.addAll(undelivered, replicaPlan().contacts().endpoints());
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
|
|||
|
|
@ -525,7 +525,13 @@ public enum CassandraRelevantProperties
|
|||
/** Controls the maximum top-k limit for vector search */
|
||||
SAI_VECTOR_SEARCH_MAX_TOP_K("cassandra.sai.vector_search.max_top_k", "1000"),
|
||||
|
||||
/**
|
||||
* enables additional correctness checks in the satellite datacenter replication strategy
|
||||
*/
|
||||
SATELLITE_REPLICATION_ADDITIONAL_CHECKS("cassandra.satellite_replication_addl_checks", "true"),
|
||||
|
||||
SCHEMA_PULL_INTERVAL_MS("cassandra.schema_pull_interval_ms", "60000"),
|
||||
|
||||
SCHEMA_UPDATE_HANDLER_FACTORY_CLASS("cassandra.schema.update_handler_factory.class"),
|
||||
SEARCH_CONCURRENCY_FACTOR("cassandra.search_concurrency_factor", "1"),
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,68 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.db;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||
import org.apache.cassandra.locator.AbstractReplicationStrategy;
|
||||
import org.apache.cassandra.locator.SatelliteFailoverState;
|
||||
import org.apache.cassandra.locator.SatelliteReplicationStrategy;
|
||||
import org.apache.cassandra.net.IVerbHandler;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
|
||||
/**
|
||||
* Verb handler for PAXOS2_COMMIT_REMOTE_REQ that wraps {@link MutationVerbHandler} with a failover state check
|
||||
* for SatelliteReplicationStrategy keyspaces.
|
||||
*
|
||||
* PAXOS2_COMMIT_REMOTE_REQ sends a normal {@link Mutation} that doesn't indicate it came from a paxos commit
|
||||
* so the standard MutationVerbHandler has no awareness that it originated from a paxos commit. This wrapper
|
||||
* rejects mutations during {@link SatelliteFailoverState.State#TRANSITION_ACK} to prevent stale paxos commits
|
||||
* from being applied to satellite/secondary DCs during failover.
|
||||
*/
|
||||
public class PaxosCommitRemoteMutationVerbHandler implements IVerbHandler<Mutation>
|
||||
{
|
||||
public static final PaxosCommitRemoteMutationVerbHandler instance = new PaxosCommitRemoteMutationVerbHandler();
|
||||
private static final Logger logger = LoggerFactory.getLogger(PaxosCommitRemoteMutationVerbHandler.class);
|
||||
|
||||
@Override
|
||||
public void doVerb(Message<Mutation> message)
|
||||
{
|
||||
Mutation mutation = message.payload;
|
||||
Keyspace keyspace = Keyspace.open(mutation.getKeyspaceName());
|
||||
AbstractReplicationStrategy strategy = keyspace.getReplicationStrategy();
|
||||
|
||||
if (strategy instanceof SatelliteReplicationStrategy)
|
||||
{
|
||||
SatelliteReplicationStrategy srs = (SatelliteReplicationStrategy) strategy;
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
SatelliteFailoverState.FailoverInfo failoverInfo = srs.getFailoverInfo(mutation.key().getToken(), metadata);
|
||||
if (failoverInfo.getState() == SatelliteFailoverState.State.TRANSITION_ACK)
|
||||
{
|
||||
logger.debug("Rejecting PAXOS2_COMMIT_REMOTE_REQ for {} during TRANSITION_ACK", mutation.getKeyspaceName());
|
||||
MessagingService.instance().respondWithFailure(RequestFailureReason.UNKNOWN, message);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
MutationVerbHandler.instance.doVerb(message);
|
||||
}
|
||||
}
|
||||
|
|
@ -85,23 +85,21 @@ public class IndexStatusManager
|
|||
/**
|
||||
* Remove endpoints whose indexes are not queryable for the specified {@link Index.QueryPlan}.
|
||||
*
|
||||
* @param liveEndpoints current live endpoints where non-queryable endpoints will be removed
|
||||
* @param keyspace to be queried
|
||||
* @param liveEndpoints current live endpoints where non-queryable endpoints will be removed
|
||||
* @param keyspace to be queried
|
||||
* @param indexQueryPlan index query plan used in the read command
|
||||
* @param level consistency level of read command
|
||||
*/
|
||||
public <E extends Endpoints<E>> E filterForQuery(E liveEndpoints, Keyspace keyspace, Index.QueryPlan indexQueryPlan, ConsistencyLevel level)
|
||||
public <E extends Endpoints<E>> E filterForQuery(E liveEndpoints, String keyspace, Index.QueryPlan indexQueryPlan, Map<InetAddressAndPort, Index.Status> indexStatusMap)
|
||||
{
|
||||
// UNKNOWN states are transient/rare; only a few replicas should have this state at any time. See CASSANDRA-19400
|
||||
Set<Replica> queryableNonSucceeded = new HashSet<>(4);
|
||||
Map<InetAddressAndPort, Index.Status> indexStatusMap = new HashMap<>();
|
||||
|
||||
E queryableEndpoints = liveEndpoints.filter(replica -> {
|
||||
|
||||
boolean allBuilt = true;
|
||||
for (Index index : indexQueryPlan.getIndexes())
|
||||
{
|
||||
Index.Status status = getIndexStatus(replica.endpoint(), keyspace.getName(), index.getIndexMetadata().name);
|
||||
Index.Status status = getIndexStatus(replica.endpoint(), keyspace, index.getIndexMetadata().name);
|
||||
if (!index.isQueryable(status))
|
||||
{
|
||||
indexStatusMap.put(replica.endpoint(), status);
|
||||
|
|
@ -122,6 +120,28 @@ public class IndexStatusManager
|
|||
if (!queryableNonSucceeded.isEmpty() && queryableNonSucceeded.size() != queryableEndpoints.size())
|
||||
queryableEndpoints = queryableEndpoints.sorted(Comparator.comparingInt(e -> queryableNonSucceeded.contains(e) ? 1 : -1));
|
||||
|
||||
return queryableEndpoints;
|
||||
}
|
||||
|
||||
|
||||
public void readFailureException(int filtered, Iterable<Replica> removed, Map<InetAddressAndPort, Index.Status> indexStatusMap, ConsistencyLevel level, int required)
|
||||
{
|
||||
Map<InetAddressAndPort, RequestFailureReason> failureReasons = new HashMap<>();
|
||||
removed.forEach(replica -> {
|
||||
Index.Status status = indexStatusMap.get(replica.endpoint());
|
||||
if (status == Index.Status.FULL_REBUILD_STARTED)
|
||||
failureReasons.put(replica.endpoint(), RequestFailureReason.INDEX_BUILD_IN_PROGRESS);
|
||||
else
|
||||
failureReasons.put(replica.endpoint(), RequestFailureReason.INDEX_NOT_AVAILABLE);
|
||||
});
|
||||
|
||||
throw new ReadFailureException(level, filtered, required, false, failureReasons);
|
||||
}
|
||||
|
||||
public <E extends Endpoints<E>> E filterForQueryOrThrow(E liveEndpoints, Keyspace keyspace, Index.QueryPlan indexQueryPlan, ConsistencyLevel level)
|
||||
{
|
||||
Map<InetAddressAndPort, Index.Status> indexStatusMap = new HashMap<>();
|
||||
E queryableEndpoints = filterForQuery(liveEndpoints, keyspace.getName(), indexQueryPlan, indexStatusMap);
|
||||
int initial = liveEndpoints.size();
|
||||
int filtered = queryableEndpoints.size();
|
||||
|
||||
|
|
@ -132,17 +152,7 @@ public class IndexStatusManager
|
|||
int required = level.blockFor(keyspace.getReplicationStrategy());
|
||||
if (required <= initial && required > filtered)
|
||||
{
|
||||
Map<InetAddressAndPort, RequestFailureReason> failureReasons = new HashMap<>();
|
||||
liveEndpoints.without(queryableEndpoints.endpoints())
|
||||
.forEach(replica -> {
|
||||
Index.Status status = indexStatusMap.get(replica.endpoint());
|
||||
if (status == Index.Status.FULL_REBUILD_STARTED)
|
||||
failureReasons.put(replica.endpoint(), RequestFailureReason.INDEX_BUILD_IN_PROGRESS);
|
||||
else
|
||||
failureReasons.put(replica.endpoint(), RequestFailureReason.INDEX_NOT_AVAILABLE);
|
||||
});
|
||||
|
||||
throw new ReadFailureException(level, filtered, required, false, failureReasons);
|
||||
readFailureException(filtered, liveEndpoints.without(queryableEndpoints.endpoints()), indexStatusMap, level, level.blockFor(keyspace.getReplicationStrategy()));
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -20,30 +20,49 @@ package org.apache.cassandra.locator;
|
|||
import java.lang.reflect.Constructor;
|
||||
import java.lang.reflect.InvocationTargetException;
|
||||
import java.lang.reflect.Method;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReferenceFieldUpdater;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import javax.annotation.Nullable;
|
||||
|
||||
import com.google.common.annotations.VisibleForTesting;
|
||||
import com.google.common.base.Preconditions;
|
||||
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.PartitionPosition;
|
||||
import org.apache.cassandra.db.WriteType;
|
||||
import org.apache.cassandra.dht.AbstractBounds;
|
||||
import org.apache.cassandra.dht.Range;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||
import org.apache.cassandra.index.Index;
|
||||
import org.apache.cassandra.locator.ReplicaCollection.Builder.Conflict;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.schema.ReplicationParams;
|
||||
import org.apache.cassandra.schema.ReplicationType;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.AbstractWriteResponseHandler;
|
||||
import org.apache.cassandra.service.ClientState;
|
||||
import org.apache.cassandra.service.DatacenterSyncWriteResponseHandler;
|
||||
import org.apache.cassandra.service.DatacenterWriteResponseHandler;
|
||||
import org.apache.cassandra.service.WriteResponseHandler;
|
||||
import org.apache.cassandra.service.paxos.Commit.Agreed;
|
||||
import org.apache.cassandra.service.paxos.Paxos;
|
||||
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||
import org.apache.cassandra.service.reads.SpeculativeRetryPolicy;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
import org.apache.cassandra.tcm.compatibility.TokenRingUtils;
|
||||
|
|
@ -51,6 +70,9 @@ import org.apache.cassandra.tcm.ownership.DataPlacement;
|
|||
import org.apache.cassandra.transport.Dispatcher;
|
||||
import org.apache.cassandra.utils.FBUtilities;
|
||||
import org.apache.cassandra.utils.Pair;
|
||||
import org.apache.cassandra.utils.concurrent.Future;
|
||||
|
||||
import static org.apache.cassandra.locator.ReplicaLayout.forTokenWriteLiveAndDown;
|
||||
|
||||
/**
|
||||
* A abstract parent for all replication strategies.
|
||||
|
|
@ -92,57 +114,49 @@ public abstract class AbstractReplicationStrategy
|
|||
|
||||
public abstract DataPlacement calculateDataPlacement(Epoch epoch, List<Range<Token>> ranges, ClusterMetadata metadata);
|
||||
|
||||
public <T> AbstractWriteResponseHandler<T> getWriteResponseHandler(ReplicaPlan.ForWrite replicaPlan,
|
||||
public <T> AbstractWriteResponseHandler<T> getWriteResponseHandler(CoordinationPlan.ForWrite coordinationPlan,
|
||||
CoordinationPlan.ForWrite idealPlan,
|
||||
Runnable callback,
|
||||
WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return getWriteResponseHandler(replicaPlan, callback, writeType, hintOnFailure,
|
||||
requestTime, DatabaseDescriptor.getIdealConsistencyLevel());
|
||||
}
|
||||
|
||||
public <T> AbstractWriteResponseHandler<T> getWriteResponseHandler(ReplicaPlan.ForWrite replicaPlan,
|
||||
Runnable callback,
|
||||
WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure,
|
||||
Dispatcher.RequestTime requestTime,
|
||||
ConsistencyLevel idealConsistencyLevel)
|
||||
{
|
||||
AbstractWriteResponseHandler<T> resultResponseHandler;
|
||||
if (replicaPlan.consistencyLevel().isDatacenterLocal())
|
||||
if (coordinationPlan.consistencyLevel().isDatacenterLocal())
|
||||
{
|
||||
// block for in this context will be localnodes block.
|
||||
resultResponseHandler = new DatacenterWriteResponseHandler<T>(replicaPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
resultResponseHandler = new DatacenterWriteResponseHandler<T>(coordinationPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
}
|
||||
else if (replicaPlan.consistencyLevel() == ConsistencyLevel.EACH_QUORUM && (this instanceof NetworkTopologyStrategy))
|
||||
else if (coordinationPlan.consistencyLevel() == ConsistencyLevel.EACH_QUORUM && (this instanceof NetworkTopologyStrategy))
|
||||
{
|
||||
resultResponseHandler = new DatacenterSyncWriteResponseHandler<T>(replicaPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
resultResponseHandler = new DatacenterSyncWriteResponseHandler<T>(coordinationPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
}
|
||||
else
|
||||
{
|
||||
resultResponseHandler = new WriteResponseHandler<T>(replicaPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
resultResponseHandler = new WriteResponseHandler<T>(coordinationPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
}
|
||||
|
||||
//Check if tracking the ideal consistency level is configured
|
||||
if (idealConsistencyLevel != null)
|
||||
if (idealPlan != null)
|
||||
{
|
||||
//If ideal and requested are the same just use this handler to track the ideal consistency level
|
||||
//This is also used so that the ideal consistency level handler when constructed knows it is the ideal
|
||||
//one for tracking purposes
|
||||
if (idealConsistencyLevel == replicaPlan.consistencyLevel())
|
||||
if (coordinationPlan.consistencyLevel() == idealPlan.consistencyLevel())
|
||||
{
|
||||
resultResponseHandler.setIdealCLResponseHandler(resultResponseHandler);
|
||||
}
|
||||
else
|
||||
{
|
||||
//Construct a delegate response handler to use to track the ideal consistency level
|
||||
AbstractWriteResponseHandler<T> idealHandler = getWriteResponseHandler(replicaPlan.withConsistencyLevel(idealConsistencyLevel),
|
||||
// Construct a delegate response handler to track the ideal consistency level.
|
||||
// We pass idealPlan twice so that the recursive call sees coordinationPlan == idealPlan,
|
||||
// causing the ideal handler to set itself as its own idealCLDelegate. This is required
|
||||
// for the idealCLWriteLatency metric to be recorded (only fires when idealCLDelegate == this).
|
||||
AbstractWriteResponseHandler<T> idealHandler = getWriteResponseHandler(idealPlan, idealPlan,
|
||||
callback,
|
||||
writeType,
|
||||
hintOnFailure,
|
||||
requestTime,
|
||||
idealConsistencyLevel);
|
||||
requestTime);
|
||||
resultResponseHandler.setIdealCLResponseHandler(idealHandler);
|
||||
}
|
||||
}
|
||||
|
|
@ -150,6 +164,16 @@ public abstract class AbstractReplicationStrategy
|
|||
return resultResponseHandler;
|
||||
}
|
||||
|
||||
|
||||
public <T> AbstractWriteResponseHandler<T> getWriteResponseHandler(CoordinationPlan.ForWriteWithIdeal forWritePlan,
|
||||
Runnable callback,
|
||||
WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return getWriteResponseHandler(forWritePlan, forWritePlan.ideal, callback, writeType, hintOnFailure, requestTime);
|
||||
}
|
||||
|
||||
/**
|
||||
* calculate the RF based on strategy_options. When overwriting, ensure that this get()
|
||||
* is FAST, as this is called often.
|
||||
|
|
@ -437,4 +461,364 @@ public abstract class AbstractReplicationStrategy
|
|||
return newRanges;
|
||||
}
|
||||
}
|
||||
|
||||
protected CoordinationPlan.ForWrite planForWriteInternal(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
Function<ClusterMetadata, ReplicaLayout.ForTokenWrite> liveAndDown,
|
||||
ReplicaPlans.Selector selector)
|
||||
{
|
||||
ReplicaPlan.ForWrite plan = ReplicaPlans.forWrite(metadata, keyspace, consistencyLevel, liveAndDown, selector);
|
||||
ResponseTracker tracker = createTrackerForWrite(consistencyLevel, plan, plan.pending, metadata);
|
||||
return new CoordinationPlan.ForWrite(plan, tracker);
|
||||
}
|
||||
|
||||
public CoordinationPlan.ForWriteWithIdeal planForWrite(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
Function<ClusterMetadata, ReplicaLayout.ForTokenWrite> liveAndDown,
|
||||
ReplicaPlans.Selector selector)
|
||||
{
|
||||
CoordinationPlan.ForWrite actual = planForWriteInternal(metadata, keyspace, consistencyLevel, liveAndDown, selector);
|
||||
|
||||
CoordinationPlan.ForWrite ideal = null;
|
||||
ConsistencyLevel idealCL = DatabaseDescriptor.getIdealConsistencyLevel();
|
||||
if (idealCL != null)
|
||||
{
|
||||
if (idealCL == consistencyLevel)
|
||||
{
|
||||
ideal = actual;
|
||||
}
|
||||
else
|
||||
{
|
||||
ideal = planForWriteInternal(metadata, keyspace, idealCL, liveAndDown, selector);
|
||||
}
|
||||
}
|
||||
|
||||
return new CoordinationPlan.ForWriteWithIdeal(actual.replicas(), actual.responses(), ideal);
|
||||
}
|
||||
|
||||
public CoordinationPlan.ForWriteWithIdeal planForWrite(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
Token token,
|
||||
ReplicaPlans.Selector selector)
|
||||
{
|
||||
return planForWrite(metadata, keyspace, consistencyLevel,
|
||||
(newClusterMetadata) -> ReplicaLayout.forTokenWriteLiveAndDown(newClusterMetadata, keyspace, token), selector);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for forwarding a counter write to the leader replica.
|
||||
*
|
||||
* In cases where the original coordinator is not a replica of the counter key, the counter
|
||||
* mutation is forwarded to a leader replica that will coordinate the actual counter update.
|
||||
*/
|
||||
public CoordinationPlan.ForWrite planForForwardingCounterWrite(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
Token token,
|
||||
Function<ClusterMetadata, Replica> replicaSupplier)
|
||||
{
|
||||
ReplicaPlan.ForWrite plan = ReplicaPlans.forSingleReplicaWrite(metadata, keyspace, token, replicaSupplier);
|
||||
ResponseTracker tracker = createTrackerForWrite(plan.consistencyLevel(), plan, plan.pending, metadata);
|
||||
|
||||
return new CoordinationPlan.ForWriteWithIdeal(plan, tracker, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for replaying a mutation from the batchlog.
|
||||
*
|
||||
* When recovering failed batches, mutations are replayed to remote replicas only
|
||||
* (local replica is handled separately). This method creates a replica plan
|
||||
* targeting live remote replicas with CL.ONE, and a response tracker that waits on
|
||||
* all contacts
|
||||
*/
|
||||
public CoordinationPlan.ForWriteWithIdeal planForReplayMutation(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
Token token)
|
||||
{
|
||||
Preconditions.checkState(!replicationType.isTracked(), "Batch replay not supported with tracked keyspaces");
|
||||
|
||||
ReplicaPlan.ForWrite plan = ReplicaPlans.forReplayMutation(metadata, keyspace, token);
|
||||
|
||||
// wait until all contacts respond
|
||||
int blockFor = plan.contacts().size();
|
||||
ResponseTracker tracker = new SimpleResponseTracker(blockFor, blockFor);
|
||||
|
||||
return new CoordinationPlan.ForWriteWithIdeal(plan, tracker, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for a single-partition token read.
|
||||
*/
|
||||
public CoordinationPlan.ForTokenRead planForTokenRead(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
Token token,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
SpeculativeRetryPolicy retry,
|
||||
ReadCoordinator coordinator)
|
||||
{
|
||||
ReplicaPlan.ForTokenRead plan = ReplicaPlans.forRead(metadata, keyspace, tableId, token, indexQueryPlan, consistencyLevel, retry, coordinator);
|
||||
ReplicaPlan.SharedForTokenRead shared = ReplicaPlan.shared(plan);
|
||||
ResponseTracker tracker = createTrackerForRead(plan);
|
||||
return new CoordinationPlan.ForTokenRead(shared, tracker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for a range read.
|
||||
*/
|
||||
public CoordinationPlan.ForRangeRead planForRangeRead(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
int vnodeCount)
|
||||
{
|
||||
ReplicaPlan.ForRangeRead plan = ReplicaPlans.forRangeRead(metadata, keyspace, tableId, indexQueryPlan, consistencyLevel, range, vnodeCount, true);
|
||||
ReplicaPlan.SharedForRangeRead shared = ReplicaPlan.shared(plan);
|
||||
ResponseTracker tracker = createTrackerForRead(plan);
|
||||
return new CoordinationPlan.ForRangeRead(shared, tracker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Attempt to merge two adjacent range read coordination plans into one.
|
||||
*
|
||||
* If the two plans share enough live endpoints to satisfy the consistency level
|
||||
* and the merge is worthwhile returns a merged plan otherwise returns null.
|
||||
*/
|
||||
public CoordinationPlan.ForRangeRead maybeMergeRangeReads(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
ReplicaPlan.ForRangeRead left,
|
||||
ReplicaPlan.ForRangeRead right)
|
||||
{
|
||||
ReplicaPlan.ForRangeRead merged = ReplicaPlans.maybeMerge(metadata, keyspace, tableId, consistencyLevel, left, right);
|
||||
if (merged == null)
|
||||
return null;
|
||||
|
||||
ReplicaPlan.SharedForRangeRead shared = ReplicaPlan.shared(merged);
|
||||
ResponseTracker tracker = createTrackerForRead(merged);
|
||||
return new CoordinationPlan.ForRangeRead(shared, tracker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for a full range read
|
||||
*/
|
||||
public CoordinationPlan.ForRangeRead planForFullRangeRead(Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
Set<InetAddressAndPort> endpointsToContact,
|
||||
int vnodeCount)
|
||||
{
|
||||
ReplicaPlan.ForRangeRead plan = ReplicaPlans.forFullRangeRead(keyspace, consistencyLevel, range, endpointsToContact, vnodeCount);
|
||||
ReplicaPlan.SharedForRangeRead shared = ReplicaPlan.shared(plan);
|
||||
ResponseTracker tracker = createTrackerForRead(plan);
|
||||
return new CoordinationPlan.ForRangeRead(shared, tracker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for a single-replica token read.
|
||||
*/
|
||||
public CoordinationPlan.ForTokenRead planForSingleReplicaTokenRead(Keyspace keyspace, Token token, Replica replica)
|
||||
{
|
||||
ReplicaPlan.ForTokenRead plan = ReplicaPlans.forSingleReplicaRead(keyspace, token, replica);
|
||||
ReplicaPlan.SharedForTokenRead shared = ReplicaPlan.shared(plan);
|
||||
ResponseTracker tracker = createTrackerForRead(plan);
|
||||
return new CoordinationPlan.ForTokenRead(shared, tracker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for a single-replica range read.
|
||||
*
|
||||
* Used by short read protection to fetch additional partitions from a
|
||||
* specific replica. blockFor=1, totalReplicas=1.
|
||||
*/
|
||||
public CoordinationPlan.ForRangeRead planForSingleReplicaRangeRead(Keyspace keyspace,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
Replica replica,
|
||||
int vnodeCount)
|
||||
{
|
||||
ReplicaPlan.ForRangeRead plan = ReplicaPlans.forSingleReplicaRead(keyspace, range, replica, vnodeCount);
|
||||
ReplicaPlan.SharedForRangeRead shared = ReplicaPlan.shared(plan);
|
||||
ResponseTracker tracker = createTrackerForRead(plan);
|
||||
return new CoordinationPlan.ForRangeRead(shared, tracker);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create ResponseTracker for read operation.
|
||||
*/
|
||||
private <E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>> ResponseTracker createTrackerForRead(P plan)
|
||||
{
|
||||
int blockFor = plan.readQuorum();
|
||||
|
||||
// Use candidates.size() for totalReplicas to allow for speculation
|
||||
// (speculation can contact additional candidates beyond initial contacts)
|
||||
int totalReplicas = plan.readCandidates().size();
|
||||
|
||||
return new SimpleResponseTracker(blockFor, totalReplicas);
|
||||
}
|
||||
|
||||
public Paxos.Participants paxosParticipants(ClusterMetadata metadata,
|
||||
TableMetadata table,
|
||||
Token token,
|
||||
ConsistencyLevel consistencyForConsensus,
|
||||
Predicate<Replica> isReplicaAlive)
|
||||
{
|
||||
|
||||
KeyspaceMetadata keyspaceMetadata = metadata.schema.getKeyspaceMetadata(table.keyspace);
|
||||
// MetaStrategy distributes the entire keyspace to all replicas. In addition, its tables (currently only
|
||||
// the dist log table) don't use the globally configured partitioner. For these reasons we don't lookup the
|
||||
// replicas using the supplied token as this can actually be of the incorrect type (for example when
|
||||
// performing Paxos repair).
|
||||
final Token actualToken = table.partitioner == MetaStrategy.partitioner ? MetaStrategy.entireRange.right : token;
|
||||
ReplicaLayout.ForTokenWrite all = forTokenWriteLiveAndDown(metadata, keyspaceMetadata, actualToken);
|
||||
ReplicaLayout.ForTokenWrite electorate = consistencyForConsensus.isDatacenterLocal()
|
||||
? all.filter(InOurDc.replicas()) : all;
|
||||
|
||||
EndpointsForToken live = all.all().filter(isReplicaAlive);
|
||||
return new Paxos.Participants(metadata.epoch, Keyspace.open(table.keyspace), consistencyForConsensus, all, electorate, live,
|
||||
(cm) -> Paxos.Participants.get(cm, table, actualToken, consistencyForConsensus));
|
||||
}
|
||||
|
||||
/**
|
||||
* Hook for replication strategies to send additional mutations alongside a paxos commit.
|
||||
* Called from PaxosCommit.start() after local synchronous execution for tracked keyspaces.
|
||||
*
|
||||
* If the method doesn't return null, the returned future is composed with the paxos consensus:
|
||||
* onDone fires only after both the paxos quorum decision AND this future complete, or after
|
||||
* one of them fails.
|
||||
*/
|
||||
public Future<Void> sendPaxosCommitMutations(Agreed commit, boolean isUrgent)
|
||||
{
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check whether paxos operations should be rejected for the given token.
|
||||
*/
|
||||
public boolean shouldRejectPaxos(Token token)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create ResponseTracker for write operation based on consistency level.
|
||||
*/
|
||||
@VisibleForTesting
|
||||
public ResponseTracker createTrackerForWrite(ConsistencyLevel cl, ReplicaPlan.ForWrite plan, Endpoints<?> pending, ClusterMetadata metadata)
|
||||
{
|
||||
switch (cl)
|
||||
{
|
||||
case ANY:
|
||||
case ONE:
|
||||
case TWO:
|
||||
case THREE:
|
||||
case QUORUM:
|
||||
case ALL:
|
||||
{
|
||||
int totalContacts = plan.contacts().size();
|
||||
int baseBlockFor = cl.blockFor(this);
|
||||
int totalBlockFor = cl.blockForWrite(this, pending);
|
||||
|
||||
// Check if double count model applies (some CLs like ANY don't add pending)
|
||||
// If totalBlockFor == baseBlockFor, no double-count needed (e.g., ANY)
|
||||
if (totalBlockFor == baseBlockFor)
|
||||
return new SimpleResponseTracker(baseBlockFor, totalContacts);
|
||||
|
||||
// Double count model: natural must satisfy base CL, total must include pending
|
||||
int pendingReplicas = pending.size();
|
||||
// contacts() includes both natural and pending replicas
|
||||
int naturalReplicas = totalContacts - pendingReplicas;
|
||||
return new WriteResponseTracker(baseBlockFor, totalBlockFor,
|
||||
naturalReplicas, pendingReplicas,
|
||||
endpoint -> pending.endpoints().contains(endpoint));
|
||||
}
|
||||
|
||||
case LOCAL_ONE:
|
||||
case LOCAL_QUORUM:
|
||||
{
|
||||
int localContacts = plan.contacts().filter(InOurDc.replicas()).size();
|
||||
// Check if double count model applies (depends on local pending)
|
||||
int baseBlockFor = cl.blockFor(this);
|
||||
int totalBlockFor = cl.blockForWrite(this, pending);
|
||||
|
||||
// If totalBlockFor == baseBlockFor, no local pending so no double-count needed
|
||||
if (totalBlockFor == baseBlockFor)
|
||||
return new SimpleResponseTracker(baseBlockFor, localContacts, InOurDc.endpoints());
|
||||
|
||||
// Double count model for local DC
|
||||
int localPending = pending.count(InOurDc.replicas());
|
||||
// localContacts includes both natural and pending in local DC
|
||||
int localNatural = localContacts - localPending;
|
||||
return new WriteResponseTracker(baseBlockFor, totalBlockFor,
|
||||
localNatural, localPending,
|
||||
endpoint -> pending.endpoints().contains(endpoint),
|
||||
InOurDc.endpoints());
|
||||
}
|
||||
|
||||
case EACH_QUORUM:
|
||||
return createCompositeTracker(plan, pending, metadata);
|
||||
|
||||
default:
|
||||
throw new UnsupportedOperationException("Unsupported consistency level for writes: " + cl);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Create per-datacenter tracker for EACH_QUORUM.
|
||||
*/
|
||||
private ResponseTracker createCompositeTracker(ReplicaPlan.ForWrite plan, Endpoints<?> pending, ClusterMetadata metadata)
|
||||
{
|
||||
Map<String, ResponseTracker> trackerPerDc = new HashMap<>();
|
||||
Locator locator = metadata.locator;
|
||||
|
||||
// Group replicas by datacenter
|
||||
Map<String, List<Replica>> replicasByDc = new HashMap<>();
|
||||
for (Replica replica : plan.contacts())
|
||||
{
|
||||
String dc = locator.location(replica.endpoint()).datacenter;
|
||||
replicasByDc.computeIfAbsent(dc, k -> new ArrayList<>()).add(replica);
|
||||
}
|
||||
|
||||
// Group pending replicas by datacenter
|
||||
Map<String, List<Replica>> pendingByDc = new HashMap<>();
|
||||
for (Replica replica : pending)
|
||||
{
|
||||
String dc = locator.location(replica.endpoint()).datacenter;
|
||||
pendingByDc.computeIfAbsent(dc, k -> new ArrayList<>()).add(replica);
|
||||
}
|
||||
|
||||
// Create tracker for each DC
|
||||
for (Map.Entry<String, List<Replica>> entry : replicasByDc.entrySet())
|
||||
{
|
||||
String dc = entry.getKey();
|
||||
int dcContacts = entry.getValue().size();
|
||||
List<Replica> dcPending = pendingByDc.getOrDefault(dc, Collections.emptyList());
|
||||
int dcPendingCount = dcPending.size();
|
||||
int dcNatural = dcContacts - dcPendingCount;
|
||||
int dcBlockFor = dcNatural / 2 + 1;
|
||||
|
||||
// Each sub-tracker must filter by DC since CompositeTracker broadcasts to all children
|
||||
Predicate<InetAddressAndPort> dcFilter = endpoint -> dc.equals(locator.location(endpoint).datacenter);
|
||||
|
||||
if (dcPending.isEmpty())
|
||||
{
|
||||
trackerPerDc.put(dc, new SimpleResponseTracker(dcBlockFor, dcContacts, dcFilter));
|
||||
}
|
||||
else
|
||||
{
|
||||
int totalBlockFor = dcBlockFor + dcPendingCount;
|
||||
trackerPerDc.put(dc, new WriteResponseTracker(dcBlockFor, totalBlockFor,
|
||||
dcNatural, dcPendingCount,
|
||||
endpoint -> pending.endpoints().contains(endpoint),
|
||||
dcFilter));
|
||||
}
|
||||
}
|
||||
|
||||
return new CompositeTracker(trackerPerDc.size(), trackerPerDc.values());
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,168 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.Collection;
|
||||
import java.util.function.ToIntFunction;
|
||||
|
||||
/**
|
||||
* Composite response tracker: broadcasts responses to all child trackers and succeeds when at least
|
||||
* `blockFor` children are individually successful, or fails when `failures > count - blockFor`
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code blockFor == children.length} gives AND semantics (all must succeed)</li>
|
||||
* <li>{@code blockFor == quorum(children.length)} gives majority-quorum semantics</li>
|
||||
* </ul>
|
||||
*/
|
||||
public class CompositeTracker implements ResponseTracker
|
||||
{
|
||||
private final ResponseTracker[] children;
|
||||
private final int blockFor;
|
||||
|
||||
public CompositeTracker(int blockFor, ResponseTracker... children)
|
||||
{
|
||||
if (children == null || children.length == 0)
|
||||
throw new IllegalArgumentException("children cannot be null or empty");
|
||||
if (blockFor < 1 || blockFor > children.length)
|
||||
throw new IllegalArgumentException("blockFor must be between 1 and " + children.length);
|
||||
|
||||
this.children = children;
|
||||
this.blockFor = blockFor;
|
||||
}
|
||||
|
||||
public CompositeTracker(int blockFor, Collection<ResponseTracker> children)
|
||||
{
|
||||
this(blockFor, children.toArray(ResponseTracker[]::new));
|
||||
}
|
||||
|
||||
public static int quorum(int count)
|
||||
{
|
||||
return (count / 2) + 1;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isSuccessful()
|
||||
{
|
||||
int successful = 0;
|
||||
for (ResponseTracker child : children)
|
||||
if (child.isSuccessful())
|
||||
successful++;
|
||||
return successful >= blockFor;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onResponse(InetAddressAndPort from)
|
||||
{
|
||||
for (ResponseTracker child : children)
|
||||
child.onResponse(from);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onFailure(InetAddressAndPort from)
|
||||
{
|
||||
for (ResponseTracker child : children)
|
||||
child.onFailure(from);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isComplete()
|
||||
{
|
||||
if (isSuccessful())
|
||||
return true;
|
||||
|
||||
// Count children that have definitively failed
|
||||
int failed = 0;
|
||||
for (ResponseTracker child : children)
|
||||
if (child.isComplete() && !child.isSuccessful())
|
||||
failed++;
|
||||
int maxPossible = children.length - failed;
|
||||
return maxPossible < blockFor;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int required()
|
||||
{
|
||||
return count(ResponseTracker::required);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int received()
|
||||
{
|
||||
return count(ResponseTracker::received);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int failures()
|
||||
{
|
||||
return count(ResponseTracker::failures);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean countsTowardQuorum(InetAddressAndPort from)
|
||||
{
|
||||
for (ResponseTracker child : children)
|
||||
if (child.countsTowardQuorum(from))
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isPending(InetAddressAndPort from)
|
||||
{
|
||||
for (ResponseTracker child : children)
|
||||
if (child.isPending(from))
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int totalContacts()
|
||||
{
|
||||
return count(ResponseTracker::totalContacts);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pendingContacts()
|
||||
{
|
||||
return count(ResponseTracker::pendingContacts);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("CompositeTracker[children=%d, blockFor=%d]", children.length, blockFor);
|
||||
}
|
||||
|
||||
private int count(ToIntFunction<ResponseTracker> function)
|
||||
{
|
||||
int total = 0;
|
||||
for (ResponseTracker child : children)
|
||||
total += function.applyAsInt(child);
|
||||
return total;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ResponseTracker resetCopy()
|
||||
{
|
||||
ResponseTracker[] resetTrackers = new ResponseTracker[children.length];
|
||||
for (int i=0; i<children.length; i++)
|
||||
resetTrackers[i] = children[i].resetCopy();
|
||||
return new CompositeTracker(blockFor, resetTrackers);
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,321 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import javax.annotation.Nullable;
|
||||
|
||||
import com.google.common.base.Preconditions;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.PartitionPosition;
|
||||
import org.apache.cassandra.dht.AbstractBounds;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.exceptions.UnavailableException;
|
||||
import org.apache.cassandra.index.Index;
|
||||
import org.apache.cassandra.schema.SchemaConstants;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||
import org.apache.cassandra.service.reads.SpeculativeRetryPolicy;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
|
||||
/**
|
||||
* Ties together replica selection and response tracking for a single operation.
|
||||
*
|
||||
* This immutable container ensures that the replica plan (who to contact) and
|
||||
* the response tracker (how to determine success) are created atomically by the
|
||||
* replication strategy, consulting the same state. This is particularly important
|
||||
* for strategies that have per-range state (e.g., failover state in SRS) where
|
||||
* the replica selection and quorum requirements must be consistent.
|
||||
*
|
||||
* The separation between ReplicaPlan and ResponseTracker allows:
|
||||
* <ul>
|
||||
* <li>ReplicaPlan to focus on replica topology and selection</li>
|
||||
* <li>ResponseTracker to encapsulate completion logic</li>
|
||||
* <li>Replication strategies to customize both consistently</li>
|
||||
* </ul>
|
||||
*
|
||||
* No polymorphism is needed at this level - the variation is captured in the
|
||||
* ResponseTracker implementations. This is just a typed tuple ensuring the
|
||||
* two pieces travel together.
|
||||
*
|
||||
* @param <P> the type of ReplicaPlan (ForRead, ForWrite, ForPaxosWrite, etc.)
|
||||
*/
|
||||
public abstract class CoordinationPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>>
|
||||
{
|
||||
// TODO (now): consolidate callback Condition instances into this. Replace all condition await calls with calls to this.
|
||||
private final ResponseTracker responses;
|
||||
|
||||
/**
|
||||
* Create a coordination plan.
|
||||
*
|
||||
* @param responses the response tracker for determining completion/success
|
||||
*/
|
||||
public CoordinationPlan(ResponseTracker responses)
|
||||
{
|
||||
Preconditions.checkNotNull(responses);
|
||||
this.responses = responses;
|
||||
}
|
||||
|
||||
public ConsistencyLevel consistencyLevel()
|
||||
{
|
||||
return replicas().consistencyLevel();
|
||||
}
|
||||
|
||||
public AbstractReplicationStrategy replicationStrategy()
|
||||
{
|
||||
return replicas().replicationStrategy();
|
||||
}
|
||||
|
||||
public abstract P replicas();
|
||||
|
||||
/**
|
||||
* The response tracker for determining completion/success.
|
||||
*
|
||||
* The tracker encapsulates the logic for:
|
||||
* - Recording responses and failures
|
||||
* - Determining when the operation is complete
|
||||
* - Checking if the operation succeeded
|
||||
*
|
||||
* @return the response tracker
|
||||
*/
|
||||
public ResponseTracker responses()
|
||||
{
|
||||
return responses;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("CoordinationPlan[replicaPlan=%s, tracker=%s]", replicas(), responses.getClass().getSimpleName());
|
||||
}
|
||||
|
||||
/**
|
||||
* Extended plan including source cluster metadata and ideal coordination plan.
|
||||
*/
|
||||
public static class ForWrite extends CoordinationPlan<EndpointsForToken, ReplicaPlan.ForWrite>
|
||||
{
|
||||
private final ReplicaPlan.ForWrite replicas;
|
||||
|
||||
public ForWrite(ReplicaPlan.ForWrite replicas, ResponseTracker responses)
|
||||
{
|
||||
super(responses);
|
||||
this.replicas = replicas;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ReplicaPlan.ForWrite replicas()
|
||||
{
|
||||
return replicas;
|
||||
}
|
||||
}
|
||||
|
||||
public static class ForWriteWithIdeal extends CoordinationPlan.ForWrite
|
||||
{
|
||||
public final CoordinationPlan.ForWrite ideal;
|
||||
|
||||
public ForWriteWithIdeal(ReplicaPlan.ForWrite replicas, ResponseTracker responses, CoordinationPlan.ForWrite ideal)
|
||||
{
|
||||
super(replicas, responses);
|
||||
this.ideal = ideal;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create coordination plan for batchlog write.
|
||||
*
|
||||
* The batchlog is a system-level durability mechanism independent of keyspace replication:
|
||||
* - Stored in system.batches regardless of which keyspace(s) the mutations target
|
||||
* - Replica selection is DC-local based on rack diversity and liveness
|
||||
* - Uses simple ack counting (ONE or TWO based on available replicas)
|
||||
*
|
||||
* @param metadata the cluster metadata
|
||||
* @param isAny whether to allow any node (for legacy batch compatibility)
|
||||
* @return coordination plan for batchlog write
|
||||
* @throws UnavailableException if insufficient replicas are available
|
||||
*/
|
||||
public static ForWriteWithIdeal forBatchlogWrite(ClusterMetadata metadata, boolean isAny)
|
||||
throws UnavailableException
|
||||
{
|
||||
ReplicaPlan.ForWrite plan = ReplicaPlans.forBatchlogWrite(metadata, isAny);
|
||||
int blockFor = plan.consistencyLevel().blockFor(plan.replicationStrategy());
|
||||
ResponseTracker tracker = new SimpleResponseTracker(blockFor, plan.contacts().size());
|
||||
return new ForWriteWithIdeal(plan, tracker, null);
|
||||
}
|
||||
}
|
||||
|
||||
public abstract static class ForRead<E extends Endpoints<E>, P extends ReplicaPlan<E, P>> extends CoordinationPlan<E, P> implements Supplier<P>
|
||||
{
|
||||
final ReplicaPlan.Shared<E, P> replicas;
|
||||
|
||||
public ForRead(ReplicaPlan.Shared<E, P> replicas, ResponseTracker responses)
|
||||
{
|
||||
super(responses);
|
||||
this.replicas = replicas;
|
||||
}
|
||||
|
||||
public abstract ForRead<E, P> copyWithResetTracker();
|
||||
|
||||
@Override
|
||||
public P get()
|
||||
{
|
||||
return replicas.get();
|
||||
}
|
||||
|
||||
@Override
|
||||
public P replicas()
|
||||
{
|
||||
return replicas.get();
|
||||
}
|
||||
|
||||
public void addToContacts(Replica replica)
|
||||
{
|
||||
replicas.addToContacts(replica);
|
||||
}
|
||||
}
|
||||
|
||||
public static class ForTokenRead extends ForRead<EndpointsForToken, ReplicaPlan.ForTokenRead>
|
||||
{
|
||||
public ForTokenRead(ReplicaPlan.Shared<EndpointsForToken, ReplicaPlan.ForTokenRead> replicas, ResponseTracker responses)
|
||||
{
|
||||
super(replicas, responses);
|
||||
}
|
||||
|
||||
@Override
|
||||
public ForTokenRead copyWithResetTracker()
|
||||
{
|
||||
return new ForTokenRead(replicas, responses().resetCopy());
|
||||
}
|
||||
}
|
||||
|
||||
public static class ForRangeRead extends ForRead<EndpointsForRange, ReplicaPlan.ForRangeRead>
|
||||
{
|
||||
public ForRangeRead(ReplicaPlan.Shared<EndpointsForRange, ReplicaPlan.ForRangeRead> replicas, ResponseTracker responses)
|
||||
{
|
||||
super(replicas, responses);
|
||||
}
|
||||
|
||||
@Override
|
||||
public ForRangeRead copyWithResetTracker()
|
||||
{
|
||||
return new ForRangeRead(replicas, responses().resetCopy());
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Static convenience methods that look up the replication strategy internally ----
|
||||
|
||||
private static AbstractReplicationStrategy getStrategy(ClusterMetadata metadata, Keyspace keyspace)
|
||||
{
|
||||
if (SchemaConstants.isLocalSystemKeyspace(keyspace.getName()))
|
||||
return keyspace.getReplicationStrategy();
|
||||
|
||||
return metadata.schema.getKeyspaceMetadata(keyspace.getName()).replicationStrategy;
|
||||
}
|
||||
|
||||
public static ForWriteWithIdeal forWrite(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
Function<ClusterMetadata, ReplicaLayout.ForTokenWrite> liveAndDown,
|
||||
ReplicaPlans.Selector selector)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).planForWrite(metadata, keyspace, consistencyLevel, liveAndDown, selector);
|
||||
}
|
||||
|
||||
public static ForWriteWithIdeal forWrite(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
Token token,
|
||||
ReplicaPlans.Selector selector)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).planForWrite(metadata, keyspace, consistencyLevel, token, selector);
|
||||
}
|
||||
|
||||
public static ForWrite forForwardingCounterWrite(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
Token token,
|
||||
Function<ClusterMetadata, Replica> replicaSupplier)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).planForForwardingCounterWrite(metadata, keyspace, token, replicaSupplier);
|
||||
}
|
||||
|
||||
public static ForWriteWithIdeal forReplayMutation(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
Token token)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).planForReplayMutation(metadata, keyspace, token);
|
||||
}
|
||||
|
||||
public static ForTokenRead forTokenRead(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
Token token,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
SpeculativeRetryPolicy retry,
|
||||
ReadCoordinator coordinator)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).planForTokenRead(metadata, keyspace, tableId, token, indexQueryPlan, consistencyLevel, retry, coordinator);
|
||||
}
|
||||
|
||||
public static ForRangeRead forRangeRead(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
int vnodeCount)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).planForRangeRead(metadata, keyspace, tableId, indexQueryPlan, consistencyLevel, range, vnodeCount);
|
||||
}
|
||||
|
||||
public static ForRangeRead maybeMergeRangeReads(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
ForRangeRead left,
|
||||
ForRangeRead right)
|
||||
{
|
||||
return getStrategy(metadata, keyspace).maybeMergeRangeReads(metadata, keyspace, tableId, consistencyLevel, left.replicas(), right.replicas());
|
||||
}
|
||||
|
||||
public static ForRangeRead forFullRangeRead(Keyspace keyspace,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
Set<InetAddressAndPort> endpointsToContact,
|
||||
int vnodeCount)
|
||||
{
|
||||
return keyspace.getReplicationStrategy().planForFullRangeRead(keyspace, consistencyLevel, range, endpointsToContact, vnodeCount);
|
||||
}
|
||||
|
||||
public static ForTokenRead forSingleReplicaTokenRead(Keyspace keyspace, Token token, Replica replica)
|
||||
{
|
||||
return keyspace.getReplicationStrategy().planForSingleReplicaTokenRead(keyspace, token, replica);
|
||||
}
|
||||
|
||||
public static ForRangeRead forSingleReplicaRangeRead(Keyspace keyspace,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
Replica replica,
|
||||
int vnodeCount)
|
||||
{
|
||||
return keyspace.getReplicationStrategy().planForSingleReplicaRangeRead(keyspace, range, replica, vnodeCount);
|
||||
}
|
||||
}
|
||||
|
|
@ -134,11 +134,12 @@ public interface ReplicaPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>
|
|||
E contacts,
|
||||
E liveAndDown,
|
||||
Function<ClusterMetadata, P> recompute,
|
||||
Epoch epoch)
|
||||
Epoch epoch,
|
||||
int readQuorum)
|
||||
{
|
||||
super(keyspace, replicationStrategy, consistencyLevel, contacts, liveAndDown, recompute, epoch);
|
||||
this.candidates = candidates;
|
||||
this.readQuorum = consistencyLevel.blockFor(replicationStrategy);
|
||||
this.readQuorum = readQuorum;
|
||||
}
|
||||
|
||||
public int readQuorum() { return readQuorum; }
|
||||
|
|
@ -200,6 +201,22 @@ public interface ReplicaPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>
|
|||
{
|
||||
private final Function<ReplicaPlan<?, ?>, ForWrite> repairPlan;
|
||||
|
||||
public ForTokenRead(Keyspace keyspace,
|
||||
AbstractReplicationStrategy replicationStrategy,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
EndpointsForToken candidates,
|
||||
EndpointsForToken contacts,
|
||||
EndpointsForToken liveAndDown,
|
||||
Function<ClusterMetadata, ReplicaPlan.ForTokenRead> recompute,
|
||||
Function<ReplicaPlan<?, ?>, ReplicaPlan.ForWrite> repairPlan,
|
||||
Epoch epoch,
|
||||
int readQuorum)
|
||||
{
|
||||
super(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, liveAndDown, recompute, epoch, readQuorum);
|
||||
this.repairPlan = repairPlan;
|
||||
}
|
||||
|
||||
|
||||
public ForTokenRead(Keyspace keyspace,
|
||||
AbstractReplicationStrategy replicationStrategy,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
|
|
@ -210,13 +227,12 @@ public interface ReplicaPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>
|
|||
Function<ReplicaPlan<?, ?>, ReplicaPlan.ForWrite> repairPlan,
|
||||
Epoch epoch)
|
||||
{
|
||||
super(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, liveAndDown, recompute, epoch);
|
||||
this.repairPlan = repairPlan;
|
||||
this(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, liveAndDown, recompute, repairPlan, epoch, consistencyLevel.blockFor(replicationStrategy));
|
||||
}
|
||||
|
||||
public ForTokenRead withContacts(EndpointsForToken newContacts)
|
||||
{
|
||||
ForTokenRead res = new ForTokenRead(keyspace, replicationStrategy, consistencyLevel, candidates, newContacts, liveAndDown, recompute, repairPlan, epoch);
|
||||
ForTokenRead res = new ForTokenRead(keyspace, replicationStrategy, consistencyLevel, candidates, newContacts, liveAndDown, recompute, repairPlan, epoch, readQuorum);
|
||||
res.contacted.addAll(contacted);
|
||||
return res;
|
||||
}
|
||||
|
|
@ -246,14 +262,31 @@ public interface ReplicaPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>
|
|||
int vnodeCount,
|
||||
Function<ClusterMetadata, ReplicaPlan.ForRangeRead> recompute,
|
||||
BiFunction<ReplicaPlan<?, ?>, Token, ReplicaPlan.ForWrite> repairPlan,
|
||||
Epoch epoch)
|
||||
Epoch epoch,
|
||||
int readQuorum)
|
||||
{
|
||||
super(keyspace, replicationStrategy, consistencyLevel, candidates, contact, liveAndDown, recompute, epoch);
|
||||
super(keyspace, replicationStrategy, consistencyLevel, candidates, contact, liveAndDown, recompute, epoch, readQuorum);
|
||||
this.range = range;
|
||||
this.vnodeCount = vnodeCount;
|
||||
this.repairPlan = repairPlan;
|
||||
}
|
||||
|
||||
|
||||
public ForRangeRead(Keyspace keyspace,
|
||||
AbstractReplicationStrategy replicationStrategy,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
EndpointsForRange candidates,
|
||||
EndpointsForRange contact,
|
||||
EndpointsForRange liveAndDown,
|
||||
int vnodeCount,
|
||||
Function<ClusterMetadata, ReplicaPlan.ForRangeRead> recompute,
|
||||
BiFunction<ReplicaPlan<?, ?>, Token, ReplicaPlan.ForWrite> repairPlan,
|
||||
Epoch epoch)
|
||||
{
|
||||
this(keyspace, replicationStrategy, consistencyLevel, range, candidates, contact, liveAndDown, vnodeCount, recompute, repairPlan, epoch, consistencyLevel.blockFor(replicationStrategy));
|
||||
}
|
||||
|
||||
public AbstractBounds<PartitionPosition> range() { return range; }
|
||||
|
||||
/**
|
||||
|
|
@ -263,7 +296,7 @@ public interface ReplicaPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>
|
|||
|
||||
public ForRangeRead withContacts(EndpointsForRange newContact)
|
||||
{
|
||||
ForRangeRead res = new ForRangeRead(keyspace, replicationStrategy, consistencyLevel, range, readCandidates(), newContact, liveAndDown, vnodeCount, recompute, repairPlan, epoch);
|
||||
ForRangeRead res = new ForRangeRead(keyspace, replicationStrategy, consistencyLevel, range, readCandidates(), newContact, liveAndDown, vnodeCount, recompute, repairPlan, epoch, readQuorum);
|
||||
res.contacted.addAll(contacted);
|
||||
return res;
|
||||
}
|
||||
|
|
@ -289,13 +322,27 @@ public interface ReplicaPlan<E extends Endpoints<E>, P extends ReplicaPlan<E, P>
|
|||
EndpointsForRange contact,
|
||||
EndpointsForRange liveAndDown,
|
||||
int vnodeCount,
|
||||
Epoch epoch)
|
||||
Epoch epoch,
|
||||
int readQuorum)
|
||||
{
|
||||
// A FullRangeRead plan, as part of a top K query, is not recomputed to check that it still applies should
|
||||
// the epoch change during the course of query execution so no recomputation function is supplied. Likewise,
|
||||
// no read repair is expected to be performed during this type of query so a null is also used in place of a
|
||||
// function for calculating the repair plan.
|
||||
super(keyspace, replicationStrategy, consistencyLevel, range, candidates, contact, liveAndDown, vnodeCount, null, null, epoch);
|
||||
super(keyspace, replicationStrategy, consistencyLevel, range, candidates, contact, liveAndDown, vnodeCount, null, null, epoch, readQuorum);
|
||||
}
|
||||
|
||||
public ForFullRangeRead(Keyspace keyspace,
|
||||
AbstractReplicationStrategy replicationStrategy,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
AbstractBounds<PartitionPosition> range,
|
||||
EndpointsForRange candidates,
|
||||
EndpointsForRange contact,
|
||||
EndpointsForRange liveAndDown,
|
||||
int vnodeCount,
|
||||
Epoch epoch)
|
||||
{
|
||||
this(keyspace, replicationStrategy, consistencyLevel, range, candidates, contact, liveAndDown, vnodeCount, epoch, consistencyLevel.blockFor(replicationStrategy));
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
|
|||
|
|
@ -264,15 +264,6 @@ public class ReplicaPlans
|
|||
return localReplicas.get(ThreadLocalRandom.current().nextInt(localReplicas.size()));
|
||||
}
|
||||
|
||||
/**
|
||||
* A forwarding counter write is always sent to a single owning coordinator for the range, by the original coordinator
|
||||
* (if it is not itself an owner)
|
||||
*/
|
||||
public static ReplicaPlan.ForWrite forForwardingCounterWrite(ClusterMetadata metadata, Keyspace keyspace, Token token, Function<ClusterMetadata, Replica> replica)
|
||||
{
|
||||
return forSingleReplicaWrite(metadata, keyspace, token, replica);
|
||||
}
|
||||
|
||||
public static ReplicaPlan.ForWrite forLocalBatchlogWrite()
|
||||
{
|
||||
Token token = DatabaseDescriptor.getPartitioner().getMinimumToken();
|
||||
|
|
@ -287,6 +278,36 @@ public class ReplicaPlans
|
|||
return forWrite(systemKeyspace, ConsistencyLevel.ONE, (cm) -> liveAndDown, (cm) -> true, writeAll);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a replica plan for replaying a mutation from the batchlog.
|
||||
*
|
||||
* When recovering failed batches, mutations are replayed to live remote replicas only
|
||||
* (local replica is handled separately by the caller).
|
||||
*
|
||||
* @param metadata the cluster metadata
|
||||
* @param keyspace the keyspace
|
||||
* @param token the token for the mutation
|
||||
* @return replica plan targeting live remote replicas with CL.ONE
|
||||
*/
|
||||
public static ReplicaPlan.ForWrite forReplayMutation(ClusterMetadata metadata, Keyspace keyspace, Token token)
|
||||
{
|
||||
ReplicaLayout.ForTokenWrite liveAndDown = ReplicaLayout.forTokenWriteLiveAndDown(metadata, keyspace.getMetadata(), token);
|
||||
Replicas.temporaryAssertFull(liveAndDown.all()); // TODO in CASSANDRA-14549
|
||||
|
||||
Replica selfReplica = liveAndDown.all().selfIfPresent();
|
||||
ReplicaLayout.ForTokenWrite liveRemoteOnly = liveAndDown.filter(r -> FailureDetector.isReplicaAlive.test(r) && r != selfReplica);
|
||||
|
||||
EndpointsForToken allLiveRemoteOnly = liveRemoteOnly.all();
|
||||
return new ReplicaPlan.ForWrite(keyspace, keyspace.getReplicationStrategy(),
|
||||
ConsistencyLevel.ONE,
|
||||
liveRemoteOnly.pending(),
|
||||
allLiveRemoteOnly,
|
||||
allLiveRemoteOnly,
|
||||
allLiveRemoteOnly,
|
||||
(cm) -> forReplayMutation(cm, keyspace, token),
|
||||
metadata.epoch);
|
||||
}
|
||||
|
||||
/**
|
||||
* Requires that the provided endpoints are alive. Converts them to their relevant system replicas.
|
||||
* Note that the liveAndDown collection and live are equal to the provided endpoints.
|
||||
|
|
@ -800,7 +821,7 @@ public class ReplicaPlans
|
|||
{
|
||||
E replicas = consistencyLevel.isDatacenterLocal() ? liveNaturalReplicas.filter(InOurDc.replicas()) : liveNaturalReplicas;
|
||||
|
||||
return indexQueryPlan != null ? IndexStatusManager.instance.filterForQuery(replicas, keyspace, indexQueryPlan, consistencyLevel) : replicas;
|
||||
return indexQueryPlan != null ? IndexStatusManager.instance.filterForQueryOrThrow(replicas, keyspace, indexQueryPlan, consistencyLevel) : replicas;
|
||||
}
|
||||
|
||||
private static <E extends Endpoints<E>> E contactForEachQuorumRead(Locator locator, NetworkTopologyStrategy replicationStrategy, E candidates)
|
||||
|
|
@ -961,6 +982,31 @@ public class ReplicaPlans
|
|||
return forRead(metadata, keyspace, tableId, token, indexQueryPlan, consistencyLevel, retry, coordinator, true);
|
||||
}
|
||||
|
||||
public static ReplicaPlan.ForTokenRead forRead(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
Token token,
|
||||
AbstractReplicationStrategy replicationStrategy,
|
||||
ReplicaLayout.ForTokenRead forTokenReadLiveAndDown,
|
||||
ReplicaLayout.ForTokenRead forTokenReadLive,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
ConsistencyLevel consistencyLevel,
|
||||
SpeculativeRetryPolicy retry,
|
||||
ReadCoordinator coordinator,
|
||||
boolean throwOnInsufficientLiveReplicas)
|
||||
{
|
||||
EndpointsForToken candidates = candidatesForRead(keyspace, indexQueryPlan, consistencyLevel, forTokenReadLive.all());
|
||||
EndpointsForToken contacts = contactForRead(metadata.locator, replicationStrategy, consistencyLevel, retry.equals(AlwaysSpeculativeRetryPolicy.INSTANCE), candidates);
|
||||
|
||||
if (throwOnInsufficientLiveReplicas)
|
||||
assureSufficientLiveReplicasForRead(metadata.locator, replicationStrategy, consistencyLevel, contacts);
|
||||
|
||||
return new ReplicaPlan.ForTokenRead(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, forTokenReadLiveAndDown.all(),
|
||||
(newClusterMetadata) -> forRead(newClusterMetadata, keyspace, tableId, token, indexQueryPlan, consistencyLevel, retry, coordinator, false),
|
||||
(self) -> forReadRepair(self, metadata, keyspace, tableId, consistencyLevel, token, FailureDetector.isReplicaAlive, coordinator),
|
||||
metadata.epoch);
|
||||
}
|
||||
|
||||
private static ReplicaPlan.ForTokenRead forRead(ClusterMetadata metadata,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
|
|
@ -974,16 +1020,7 @@ public class ReplicaPlans
|
|||
AbstractReplicationStrategy replicationStrategy = keyspace.getReplicationStrategy();
|
||||
ReplicaLayout.ForTokenRead forTokenReadLiveAndDown = ReplicaLayout.forTokenReadSorted(metadata, keyspace, replicationStrategy, tableId, token, coordinator);
|
||||
ReplicaLayout.ForTokenRead forTokenReadLive = forTokenReadLiveAndDown.filter(FailureDetector.isReplicaAlive);
|
||||
EndpointsForToken candidates = candidatesForRead(keyspace, indexQueryPlan, consistencyLevel, forTokenReadLive.all());
|
||||
EndpointsForToken contacts = contactForRead(metadata.locator, replicationStrategy, consistencyLevel, retry.equals(AlwaysSpeculativeRetryPolicy.INSTANCE), candidates);
|
||||
|
||||
if (throwOnInsufficientLiveReplicas)
|
||||
assureSufficientLiveReplicasForRead(metadata.locator, replicationStrategy, consistencyLevel, contacts);
|
||||
|
||||
return new ReplicaPlan.ForTokenRead(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, forTokenReadLiveAndDown.all(),
|
||||
(newClusterMetadata) -> forRead(newClusterMetadata, keyspace, tableId, token, indexQueryPlan, consistencyLevel, retry, coordinator, false),
|
||||
(self) -> forReadRepair(self, metadata, keyspace, tableId, consistencyLevel, token, FailureDetector.isReplicaAlive, coordinator),
|
||||
metadata.epoch);
|
||||
return forRead(metadata, keyspace, tableId, token, replicationStrategy, forTokenReadLiveAndDown, forTokenReadLive, indexQueryPlan, consistencyLevel, retry, coordinator, throwOnInsufficientLiveReplicas);
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
|
|||
|
|
@ -0,0 +1,127 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import com.google.common.annotations.VisibleForTesting;
|
||||
|
||||
/**
|
||||
* Black-box response tracker encapsulating coordination completion logic.
|
||||
*
|
||||
* Created by replication strategy for each operation, this interface allows
|
||||
* strategies to customize how quorum requirements are calculated and enforced.
|
||||
* Different implementations can provide different semantics.
|
||||
*
|
||||
* The tracker is responsible for:
|
||||
* 1. Recording responses and failures from replicas
|
||||
* 2. Determining when the operation has completed (success or definite failure)
|
||||
* 3. Providing metrics for error messages and monitoring
|
||||
*
|
||||
* Thread safety: Implementations must be thread-safe as onResponse/onFailure
|
||||
* can be called concurrently from multiple network threads.
|
||||
*/
|
||||
public interface ResponseTracker
|
||||
{
|
||||
// TODO: review replica plan members and move here as appropriate
|
||||
|
||||
/**
|
||||
* Record a successful response from a replica.
|
||||
*
|
||||
* @param from endpoint that responded successfully
|
||||
*/
|
||||
void onResponse(InetAddressAndPort from);
|
||||
|
||||
/**
|
||||
* Record a failed response from a replica.
|
||||
*
|
||||
* @param from endpoint that failed
|
||||
*/
|
||||
void onFailure(InetAddressAndPort from);
|
||||
|
||||
// TODO: consider having an outcome method that returns an enum (PENDING, SUCCESS, FAILURE)
|
||||
/**
|
||||
* Has the operation completed (either success or definite failure)?
|
||||
*
|
||||
* An operation is complete when:
|
||||
* - Success: Required quorum has been achieved
|
||||
* - Definite failure: Not enough replicas remain to achieve quorum
|
||||
*
|
||||
* @return true if no more responses are needed to make a decision
|
||||
*/
|
||||
boolean isComplete();
|
||||
|
||||
/**
|
||||
* Did the operation succeed (quorum achieved)?
|
||||
*
|
||||
* Only meaningful if isComplete() returns true.
|
||||
*
|
||||
* @return true if required quorum was met
|
||||
*/
|
||||
boolean isSuccessful();
|
||||
|
||||
/**
|
||||
* How many responses are required for success?
|
||||
*
|
||||
* Used for error messages, metrics, and UnavailableException construction.
|
||||
* For complex trackers (e.g., EACH_QUORUM), this may be a sum or other
|
||||
* aggregate value rather than the actual completion criteria.
|
||||
*
|
||||
* @return number of responses required
|
||||
*/
|
||||
int required();
|
||||
|
||||
/**
|
||||
* How many successful responses have been received so far?
|
||||
*
|
||||
* @return number of successful responses
|
||||
*/
|
||||
int received();
|
||||
|
||||
/**
|
||||
* How many failures have been recorded so far?
|
||||
*
|
||||
* @return number of failures
|
||||
*/
|
||||
int failures();
|
||||
|
||||
/**
|
||||
* Should responses from this endpoint be counted toward quorum?
|
||||
*
|
||||
* Allows filtering of responses based on datacenter, state, or other criteria.
|
||||
* For example, LOCAL_QUORUM trackers would return false for remote DC replicas.
|
||||
*
|
||||
* @param from endpoint to check
|
||||
* @return true if responses from this endpoint count toward quorum
|
||||
*/
|
||||
@VisibleForTesting
|
||||
boolean countsTowardQuorum(InetAddressAndPort from);
|
||||
|
||||
/**
|
||||
* Indicates that the given address is a pending replica. Accepting writes but not reads
|
||||
*/
|
||||
boolean isPending(InetAddressAndPort from);
|
||||
|
||||
int totalContacts();
|
||||
|
||||
int pendingContacts();
|
||||
|
||||
/**
|
||||
* creates a copy of the tracker will all response counts reset
|
||||
*/
|
||||
ResponseTracker resetCopy();
|
||||
}
|
||||
|
|
@ -0,0 +1,272 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.Collection;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
|
||||
import com.google.common.base.Preconditions;
|
||||
|
||||
import org.apache.cassandra.dht.Range;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
|
||||
/**
|
||||
* Failover state management for satellite datacenter replication
|
||||
*
|
||||
* Tracks per-range failover state to support different replica layouts / consistency requirements
|
||||
* during different failover states to maintain correctness
|
||||
*
|
||||
* NOTE: Initial implementation stores state in memory. Future work will integrate
|
||||
* with TCM for cluster-wide coordination and persistence.
|
||||
*/
|
||||
public class SatelliteFailoverState
|
||||
{
|
||||
/**
|
||||
* Failover state for a range or keyspace.
|
||||
*
|
||||
* Each state corresponds to a specific phase in the failover process as defined
|
||||
* in CEP-58, with different consistency level requirements.
|
||||
*/
|
||||
public enum State
|
||||
{
|
||||
/**
|
||||
* Normal operation: Primary DC + satellite active.
|
||||
* Read/Write CL: QUORUM_OF_QUORUMS (primary + satellite OR secondary)
|
||||
*/
|
||||
NORMAL,
|
||||
|
||||
/**
|
||||
* First stage of failover. Waits for a QoQ to acknowledge new state before moving to
|
||||
* TRANSITION state. During TRANSITION_ACK, coordinators will not start paxos operations, and the
|
||||
* primary transition state machine will not transition to the transition state until it confirms
|
||||
* that a QoQ nodes for a range are also on TRANSITION_ACK. This temporary gap in paxos availability
|
||||
* prevents the different full dcs from performing conflicting paxos operations concurrently.
|
||||
*/
|
||||
TRANSITION_ACK,
|
||||
|
||||
/**
|
||||
*
|
||||
* Currently syncing to new primary, once sync is complete, failover state will transition to NORMAL
|
||||
* for this range
|
||||
*/
|
||||
TRANSITION,
|
||||
}
|
||||
|
||||
/**
|
||||
* Failover state with DC context.
|
||||
*
|
||||
* Tracks which DC we're failing over from. The target DC is always the current
|
||||
* primary DC configured in the replication strategy, enabling failover between
|
||||
* any configured DCs via schema change (ALTER KEYSPACE ... WITH replication = {'primary': 'DC2'}).
|
||||
*/
|
||||
public static class FailoverInfo
|
||||
{
|
||||
private final State state;
|
||||
private final String fromDC; // DC we're failing over from (old primary), null for NORMAL
|
||||
|
||||
/**
|
||||
* Create failover info.
|
||||
*
|
||||
* @param state The failover state
|
||||
* @param fromDC The old primary DC we're failing from (null for NORMAL state)
|
||||
*/
|
||||
private FailoverInfo(State state, String fromDC)
|
||||
{
|
||||
this.state = state;
|
||||
this.fromDC = fromDC;
|
||||
|
||||
switch (state)
|
||||
{
|
||||
case NORMAL:
|
||||
Preconditions.checkArgument(fromDC == null);
|
||||
break;
|
||||
case TRANSITION_ACK:
|
||||
case TRANSITION:
|
||||
Preconditions.checkArgument(fromDC != null);
|
||||
break;
|
||||
default:
|
||||
throw new IllegalArgumentException("Unknown state: " + state);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the failover state.
|
||||
*/
|
||||
public State getState()
|
||||
{
|
||||
return state;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the DC we're failing over FROM (old primary).
|
||||
* Returns null for NORMAL state.
|
||||
*/
|
||||
public String getFromDC()
|
||||
{
|
||||
return fromDC;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create NORMAL state (no failover in progress).
|
||||
*/
|
||||
public static FailoverInfo normal()
|
||||
{
|
||||
return new FailoverInfo(State.NORMAL, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create TRANSITION_ACK state.
|
||||
*
|
||||
* @param fromDC The old primary DC we're failing over from
|
||||
*/
|
||||
public static FailoverInfo transitionAck(String fromDC)
|
||||
{
|
||||
return new FailoverInfo(State.TRANSITION_ACK, fromDC);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create TRANSITION state
|
||||
*
|
||||
* @param fromDC The old primary DC we're transferring from
|
||||
*/
|
||||
public static FailoverInfo transition(String fromDC)
|
||||
{
|
||||
return new FailoverInfo(State.TRANSITION, fromDC);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
if (fromDC == null)
|
||||
return state.toString();
|
||||
return String.format("%s(from=%s)", state, fromDC);
|
||||
}
|
||||
|
||||
public boolean isTransitioning()
|
||||
{
|
||||
switch (state)
|
||||
{
|
||||
case TRANSITION:
|
||||
case TRANSITION_ACK:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Container for per-range failover state.
|
||||
*
|
||||
* Stores failover state indexed by token range, enabling different ranges
|
||||
* to be in different failover states (though initial implementation uses
|
||||
* uniform state across all ranges).
|
||||
*
|
||||
* NOTE: This is in-memory for initial implementation. Future work will
|
||||
* store this in TCM for persistence and cluster-wide coordination.
|
||||
*/
|
||||
public static class FailoverStateMap
|
||||
{
|
||||
private final Map<Range<Token>, FailoverInfo> perRangeState;
|
||||
private final FailoverInfo defaultState; // Used for ranges not explicitly set
|
||||
|
||||
private FailoverStateMap(Map<Range<Token>, FailoverInfo> perRangeState,
|
||||
FailoverInfo defaultState)
|
||||
{
|
||||
this.perRangeState = Collections.unmodifiableMap(new HashMap<>(perRangeState));
|
||||
this.defaultState = defaultState;
|
||||
}
|
||||
|
||||
private FailoverStateMap(FailoverInfo globalState)
|
||||
{
|
||||
this.perRangeState = Collections.emptyMap();
|
||||
this.defaultState = globalState;
|
||||
}
|
||||
|
||||
public FailoverInfo getFailoverInfo(Range<Token> range)
|
||||
{
|
||||
// FIXME: This is just meant for testing at the moment and assumes that we never query ranges that don't
|
||||
// align with the state map. Fix as part of the failover process implementation
|
||||
return getFailoverInfo(range.right);
|
||||
}
|
||||
|
||||
public FailoverInfo getFailoverInfo(Token token)
|
||||
{
|
||||
for (Map.Entry<Range<Token>, FailoverInfo> entry : perRangeState.entrySet())
|
||||
{
|
||||
if (entry.getKey().contains(token))
|
||||
return entry.getValue();
|
||||
}
|
||||
return defaultState;
|
||||
}
|
||||
|
||||
public static class Builder
|
||||
{
|
||||
private final Map<Range<Token>, FailoverInfo> state = new HashMap<>();
|
||||
private FailoverInfo defaultState = FailoverInfo.normal();
|
||||
|
||||
/**
|
||||
* Set the default state for ranges not explicitly set.
|
||||
*/
|
||||
public Builder setDefault(FailoverInfo info)
|
||||
{
|
||||
this.defaultState = info;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set failover state for a specific range.
|
||||
*/
|
||||
public Builder setState(Range<Token> range, FailoverInfo info)
|
||||
{
|
||||
state.put(range, info);
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set failover state for multiple ranges.
|
||||
*/
|
||||
public Builder setState(Collection<Range<Token>> ranges, FailoverInfo info)
|
||||
{
|
||||
ranges.forEach(r -> state.put(r, info));
|
||||
return this;
|
||||
}
|
||||
|
||||
public FailoverStateMap build()
|
||||
{
|
||||
return new FailoverStateMap(state, defaultState);
|
||||
}
|
||||
}
|
||||
|
||||
public static Builder builder()
|
||||
{
|
||||
return new Builder();
|
||||
}
|
||||
|
||||
/**
|
||||
* Convenience factory: Create state map with single state for all ranges.
|
||||
*/
|
||||
public static FailoverStateMap allRanges(FailoverInfo info)
|
||||
{
|
||||
return new FailoverStateMap(info);
|
||||
}
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
|
|
@ -0,0 +1,164 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicIntegerFieldUpdater;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
/**
|
||||
* Simple response tracker that counts responses against a single threshold.
|
||||
* <p>
|
||||
* Supports optional predicate filtering to selectively count responses
|
||||
* (e.g., LOCAL_* consistency levels that only count local DC responses).
|
||||
*/
|
||||
public class SimpleResponseTracker implements ResponseTracker
|
||||
{
|
||||
private static final AtomicIntegerFieldUpdater<SimpleResponseTracker> RESPONSES_UPDATER
|
||||
= AtomicIntegerFieldUpdater.newUpdater(SimpleResponseTracker.class, "responses");
|
||||
private static final AtomicIntegerFieldUpdater<SimpleResponseTracker> FAILURES_UPDATER
|
||||
= AtomicIntegerFieldUpdater.newUpdater(SimpleResponseTracker.class, "failures");
|
||||
private static final Predicate<InetAddressAndPort> NO_FILTER = address -> true;
|
||||
|
||||
private final int blockFor;
|
||||
private final int totalReplicas;
|
||||
private final Predicate<InetAddressAndPort> filter;
|
||||
private volatile int responses = 0;
|
||||
private volatile int failures = 0;
|
||||
|
||||
/**
|
||||
* Create unfiltered tracker
|
||||
*
|
||||
* @param blockFor number of responses required for quorum
|
||||
* @param totalReplicas total replicas available (for early failure detection)
|
||||
*/
|
||||
public SimpleResponseTracker(int blockFor, int totalReplicas)
|
||||
{
|
||||
this(blockFor, totalReplicas, NO_FILTER);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create filtered tracker
|
||||
*
|
||||
* @param blockFor number of responses required for quorum
|
||||
* @param totalReplicas total replicas available (for early failure detection)
|
||||
* @param filter predicate to test if response counts (null = all count)
|
||||
*/
|
||||
public SimpleResponseTracker(int blockFor, int totalReplicas,
|
||||
Predicate<InetAddressAndPort> filter)
|
||||
{
|
||||
if (blockFor < 0)
|
||||
throw new IllegalArgumentException("blockFor must be non-negative: " + blockFor);
|
||||
if (totalReplicas < 0)
|
||||
throw new IllegalArgumentException("totalReplicas must be non-negative: " + totalReplicas);
|
||||
|
||||
this.blockFor = blockFor;
|
||||
this.totalReplicas = totalReplicas;
|
||||
this.filter = filter != null ? filter : NO_FILTER;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onResponse(InetAddressAndPort from)
|
||||
{
|
||||
if (countsTowardQuorum(from))
|
||||
RESPONSES_UPDATER.incrementAndGet(this);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onFailure(InetAddressAndPort from)
|
||||
{
|
||||
if (countsTowardQuorum(from))
|
||||
FAILURES_UPDATER.incrementAndGet(this);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isComplete()
|
||||
{
|
||||
int r = responses;
|
||||
int f = failures;
|
||||
|
||||
if (r >= blockFor)
|
||||
return true;
|
||||
|
||||
// failure: can't reach blockFor
|
||||
int needed = blockFor - r;
|
||||
int remaining = totalReplicas - (r + f);
|
||||
return needed > remaining;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isSuccessful()
|
||||
{
|
||||
return responses >= blockFor;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int required()
|
||||
{
|
||||
return blockFor;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int received()
|
||||
{
|
||||
return responses;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int failures()
|
||||
{
|
||||
return failures;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean countsTowardQuorum(InetAddressAndPort from)
|
||||
{
|
||||
return filter.test(from);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("SimpleResponseTracker[blockFor=%d, totalReplicas=%d, responses=%d, failures=%d, filtered=%s]",
|
||||
blockFor, totalReplicas, responses, failures, filter != NO_FILTER);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isPending(InetAddressAndPort from)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int totalContacts()
|
||||
{
|
||||
return totalReplicas;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pendingContacts()
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public SimpleResponseTracker resetCopy()
|
||||
{
|
||||
return new SimpleResponseTracker(blockFor, totalReplicas, filter);
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,229 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.function.Predicate;
|
||||
|
||||
/**
|
||||
* Response tracker for writes with pending replicas using the double count model.
|
||||
* <p>
|
||||
* Two requirements must be satisfied for success:
|
||||
* <ol>
|
||||
* <li>Committed replicas must satisfy base CL: {@code naturalSuccesses >= baseBlockFor}</li>
|
||||
* <li>Total replicas (natural + pending) must satisfy: {@code totalSuccesses >= totalBlockFor}</li>
|
||||
* </ol>
|
||||
* <p>
|
||||
* This ensures both consistency (natural replicas have the data) and bootstrap safety
|
||||
* (pending replicas also receive the write).
|
||||
* <p>
|
||||
* Implemented by composing two {@link SimpleResponseTracker}s:
|
||||
* <ul>
|
||||
* <li>naturalTracker: tracks responses from natural replicas only</li>
|
||||
* <li>totalTracker: tracks responses from all replicas (natural + pending)</li>
|
||||
* </ul>
|
||||
* <p>
|
||||
* Thread-safe through delegation to thread-safe SimpleResponseTrackers.
|
||||
*/
|
||||
public class WriteResponseTracker implements ResponseTracker
|
||||
{
|
||||
private final SimpleResponseTracker natural;
|
||||
private final SimpleResponseTracker total;
|
||||
|
||||
/**
|
||||
* Create a write response tracker with the double count model.
|
||||
*
|
||||
* @param baseBlockFor number of natural replica responses required for CL
|
||||
* @param totalBlockFor total responses required (baseBlockFor + pending count)
|
||||
* @param naturalReplicas number of natural replicas available
|
||||
* @param pendingReplicas number of pending replicas available
|
||||
* @param isPending predicate to determine if an endpoint is pending
|
||||
*/
|
||||
public WriteResponseTracker(int baseBlockFor,
|
||||
int totalBlockFor,
|
||||
int naturalReplicas,
|
||||
int pendingReplicas,
|
||||
Predicate<InetAddressAndPort> isPending)
|
||||
{
|
||||
this(baseBlockFor, totalBlockFor, naturalReplicas, pendingReplicas, isPending, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a write response tracker with the double count model and a filter.
|
||||
*
|
||||
* @param naturalBlockFor number of natural replica responses required for CL
|
||||
* @param totalBlockFor total responses required (naturalBlockFor + pending count)
|
||||
* @param naturalReplicas number of natural replicas available
|
||||
* @param pendingReplicas number of pending replicas available
|
||||
* @param isPending predicate to determine if an endpoint is pending
|
||||
* @param filter predicate to filter which responses count (e.g., InOurDc for LOCAL_QUORUM)
|
||||
*/
|
||||
public WriteResponseTracker(int naturalBlockFor,
|
||||
int totalBlockFor,
|
||||
int naturalReplicas,
|
||||
int pendingReplicas,
|
||||
Predicate<InetAddressAndPort> isPending,
|
||||
Predicate<InetAddressAndPort> filter)
|
||||
{
|
||||
if (naturalBlockFor < 0)
|
||||
throw new IllegalArgumentException("naturalBlockFor must be non-negative: " + naturalBlockFor);
|
||||
if (totalBlockFor < naturalBlockFor)
|
||||
throw new IllegalArgumentException("totalBlockFor (" + totalBlockFor + ") must be >= naturalBlockFor (" + naturalBlockFor + ")");
|
||||
if (naturalReplicas < 0)
|
||||
throw new IllegalArgumentException("naturalReplicas must be non-negative: " + naturalReplicas);
|
||||
if (pendingReplicas < 0)
|
||||
throw new IllegalArgumentException("pendingReplicas must be non-negative: " + pendingReplicas);
|
||||
if (naturalBlockFor > naturalReplicas)
|
||||
throw new IllegalArgumentException("naturalBlockFor (" + naturalBlockFor + ") cannot exceed naturalReplicas (" + naturalReplicas + ")");
|
||||
if (totalBlockFor > naturalReplicas + pendingReplicas)
|
||||
throw new IllegalArgumentException("totalBlockFor (" + totalBlockFor + ") cannot exceed total replicas (" + (naturalReplicas + pendingReplicas) + ")");
|
||||
if (isPending == null)
|
||||
throw new IllegalArgumentException("isPending predicate cannot be null");
|
||||
|
||||
Predicate<InetAddressAndPort> naturalFilter = isPending.negate();
|
||||
if (filter != null)
|
||||
naturalFilter = naturalFilter.and(filter);
|
||||
|
||||
this.natural = new SimpleResponseTracker(naturalBlockFor, naturalReplicas, naturalFilter);
|
||||
this.total = new SimpleResponseTracker(totalBlockFor, naturalReplicas + pendingReplicas, filter);
|
||||
}
|
||||
|
||||
private WriteResponseTracker(SimpleResponseTracker natural, SimpleResponseTracker total)
|
||||
{
|
||||
this.natural = natural;
|
||||
this.total = total;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onResponse(InetAddressAndPort from)
|
||||
{
|
||||
natural.onResponse(from);
|
||||
total.onResponse(from);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onFailure(InetAddressAndPort from)
|
||||
{
|
||||
natural.onFailure(from);
|
||||
total.onFailure(from);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isComplete()
|
||||
{
|
||||
// Early failure if either tracker fails
|
||||
if (natural.isComplete() && !natural.isSuccessful())
|
||||
return true;
|
||||
if (total.isComplete() && !total.isSuccessful())
|
||||
return true;
|
||||
|
||||
// Success requires both to succeed
|
||||
return natural.isSuccessful() && total.isSuccessful();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isSuccessful()
|
||||
{
|
||||
return natural.isSuccessful() && total.isSuccessful();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int required()
|
||||
{
|
||||
// Return total requirement for error messages
|
||||
return total.required();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int received()
|
||||
{
|
||||
// Return total successes for error messages
|
||||
return total.received();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int failures()
|
||||
{
|
||||
// Return total failures for error messages
|
||||
return total.failures();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean countsTowardQuorum(InetAddressAndPort from)
|
||||
{
|
||||
return total.countsTowardQuorum(from);
|
||||
}
|
||||
|
||||
public int naturalReceived()
|
||||
{
|
||||
return natural.received();
|
||||
}
|
||||
|
||||
public int pendingReceived()
|
||||
{
|
||||
return total.received() - natural.received();
|
||||
}
|
||||
|
||||
public int naturalFailures()
|
||||
{
|
||||
return natural.failures();
|
||||
}
|
||||
|
||||
public int pendingFailures()
|
||||
{
|
||||
return total.failures() - natural.failures();
|
||||
}
|
||||
|
||||
public int baseBlockFor()
|
||||
{
|
||||
return natural.required();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("WriteResponseTracker[baseBlockFor=%d, totalBlockFor=%d, " +
|
||||
"naturalTracker=%s, totalTracker=%s]",
|
||||
baseBlockFor(), required(),
|
||||
natural, total);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isPending(InetAddressAndPort from)
|
||||
{
|
||||
return total.countsTowardQuorum(from) && !natural.countsTowardQuorum(from);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int totalContacts()
|
||||
{
|
||||
return total.totalContacts();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pendingContacts()
|
||||
{
|
||||
return total.totalContacts() - natural.totalContacts();
|
||||
}
|
||||
|
||||
@Override
|
||||
public ResponseTracker resetCopy()
|
||||
{
|
||||
return new WriteResponseTracker(natural.resetCopy(), total.resetCopy());
|
||||
}
|
||||
}
|
||||
|
|
@ -29,6 +29,7 @@ public class PaxosMetrics
|
|||
private static final MetricNameFactory factory = new DefaultNameFactory(TYPE_NAME);
|
||||
public static final Counter linearizabilityViolations = Metrics.counter(factory.createMetricName("LinearizabilityViolations"));
|
||||
public static final Meter repairPaxosTopologyRetries = Metrics.meter(factory.createMetricName("RepairPaxosTopologyRetries"));
|
||||
public static final Meter rejectedOperations = Metrics.meter(factory.createMetricName("RejectedOperations"));
|
||||
|
||||
public static void initialize() {}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -36,6 +36,7 @@ import org.apache.cassandra.db.CounterMutation;
|
|||
import org.apache.cassandra.db.CounterMutationVerbHandler;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.MutationVerbHandler;
|
||||
import org.apache.cassandra.db.PaxosCommitRemoteMutationVerbHandler;
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.db.ReadCommandVerbHandler;
|
||||
import org.apache.cassandra.db.ReadRepairVerbHandler;
|
||||
|
|
@ -304,7 +305,7 @@ public enum Verb
|
|||
SNAPSHOT_RSP (87, P0, rpcTimeout, MISC, () -> NoPayload.serializer, RESPONSE_HANDLER ),
|
||||
SNAPSHOT_REQ (27, P0, rpcTimeout, MISC, () -> SnapshotCommand.serializer, () -> SnapshotVerbHandler.instance, SNAPSHOT_RSP ),
|
||||
|
||||
PAXOS2_COMMIT_REMOTE_REQ (38, P2, writeTimeout, MUTATION, () -> Mutation.serializer, () -> MutationVerbHandler.instance, MUTATION_RSP ),
|
||||
PAXOS2_COMMIT_REMOTE_REQ (38, P2, writeTimeout, MUTATION, () -> Mutation.serializer, () -> PaxosCommitRemoteMutationVerbHandler.instance, MUTATION_RSP ),
|
||||
PAXOS2_COMMIT_REMOTE_RSP (39, P2, writeTimeout, REQUEST_RESPONSE, () -> NoPayload.serializer, RESPONSE_HANDLER ),
|
||||
PAXOS2_PREPARE_RSP (50, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPrepare.responseSerializer, RESPONSE_HANDLER ),
|
||||
PAXOS2_PREPARE_REQ (40, P2, writeTimeout, MUTATION, () -> PaxosPrepare.requestSerializer, () -> PaxosPrepare.requestHandler, PAXOS2_PREPARE_RSP ),
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ import org.apache.cassandra.io.IVersionedSerializer;
|
|||
import org.apache.cassandra.io.util.DataInputPlus;
|
||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||
import org.apache.cassandra.locator.AbstractReplicationStrategy;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.NodeProximity;
|
||||
|
|
@ -296,7 +297,7 @@ public class ForwardedWrite
|
|||
}
|
||||
}
|
||||
|
||||
public static AbstractWriteResponseHandler<Object> forwardMutation(Mutation mutation, ReplicaPlan.ForWrite plan, AbstractReplicationStrategy strategy, Dispatcher.RequestTime requestTime)
|
||||
public static AbstractWriteResponseHandler<Object> forwardMutation(Mutation mutation, CoordinationPlan.ForWriteWithIdeal plan, AbstractReplicationStrategy strategy, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
// find leader
|
||||
NodeProximity proximity = DatabaseDescriptor.getNodeProximity();
|
||||
|
|
@ -310,7 +311,7 @@ public class ForwardedWrite
|
|||
Replica leader = null;
|
||||
for (Replica replica : proximity.sortedByProximity(FBUtilities.getBroadcastAddressAndPort(), endpoints))
|
||||
{
|
||||
if (plan.isAlive(replica))
|
||||
if (plan.replicas().isAlive(replica))
|
||||
leader = replica;
|
||||
}
|
||||
Preconditions.checkState(leader != null, "Could not find leader for %s", mutation);
|
||||
|
|
@ -321,17 +322,17 @@ public class ForwardedWrite
|
|||
AbstractWriteResponseHandler<Object> handler = strategy.getWriteResponseHandler(plan, null, WriteType.SIMPLE, null, requestTime);
|
||||
|
||||
// Add callbacks for replicas to respond directly to coordinator
|
||||
Message<MutationRequest> toLeader = Message.outWithRequestTime(Verb.FORWARD_WRITE_REQ, new MutationRequest(mutation, plan), requestTime);
|
||||
Message<MutationRequest> toLeader = Message.outWithRequestTime(Verb.FORWARD_WRITE_REQ, new MutationRequest(mutation, plan.replicas()), requestTime);
|
||||
for (Replica endpoint : endpoints)
|
||||
{
|
||||
if (plan.isAlive(endpoint))
|
||||
if (plan.replicas().isAlive(endpoint))
|
||||
{
|
||||
logger.trace("Adding forwarding callback for response from {} id {}", endpoint, toLeader.id());
|
||||
MessagingService.instance().callbacks.addWithExpiration(handler, toLeader, endpoint);
|
||||
}
|
||||
else
|
||||
{
|
||||
handler.expired();
|
||||
handler.expired(endpoint.endpoint());
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -345,7 +346,7 @@ public class ForwardedWrite
|
|||
* The leader will apply the counter mutation, assign a mutation ID, and replicate to other replicas.
|
||||
*/
|
||||
public static AbstractWriteResponseHandler<Object> forwardCounterMutation(CounterMutation counterMutation,
|
||||
ReplicaPlan.ForWrite plan,
|
||||
CoordinationPlan.ForWriteWithIdeal plan,
|
||||
AbstractReplicationStrategy strategy,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
|
|
@ -375,18 +376,19 @@ public class ForwardedWrite
|
|||
// Create response handler for all replicas
|
||||
AbstractWriteResponseHandler<Object> handler = strategy.getWriteResponseHandler(plan, null, WriteType.COUNTER, null, requestTime);
|
||||
|
||||
ReplicaPlan.ForWrite replicas = plan.replicas();
|
||||
// Add callbacks for all live replicas to respond directly to coordinator
|
||||
Message<CounterMutation> forwardMessage = Message.outWithRequestTime(Verb.COUNTER_MUTATION_REQ, counterMutation, requestTime);
|
||||
for (Replica replica : plan.contacts())
|
||||
for (Replica replica : replicas.contacts())
|
||||
{
|
||||
if (plan.isAlive(replica))
|
||||
if (replicas.isAlive(replica))
|
||||
{
|
||||
logger.trace("Adding forwarding callback for tracked counter response from {} id {}", replica, forwardMessage.id());
|
||||
MessagingService.instance().callbacks.addWithExpiration(handler, forwardMessage, replica);
|
||||
}
|
||||
else
|
||||
{
|
||||
handler.expired();
|
||||
handler.expired(replica.endpoint());
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -407,7 +409,7 @@ public class ForwardedWrite
|
|||
* @return the write response handler
|
||||
*/
|
||||
public static AbstractWriteResponseHandler<Object> forward(IMutation mutation,
|
||||
ReplicaPlan.ForWrite plan,
|
||||
CoordinationPlan.ForWriteWithIdeal plan,
|
||||
AbstractReplicationStrategy strategy,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
|
|
|
|||
|
|
@ -43,6 +43,7 @@ import org.apache.cassandra.dht.Token;
|
|||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.exceptions.WriteTimeoutException;
|
||||
import org.apache.cassandra.locator.AbstractReplicationStrategy;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.DynamicEndpointSnitch;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
|
|
@ -171,13 +172,14 @@ public class TrackedWriteRequest
|
|||
|
||||
Preconditions.checkArgument(mutation.id().isNone());
|
||||
String keyspaceName = mutation.getKeyspaceName();
|
||||
ClusterMetadata cm = ClusterMetadata.current();
|
||||
Keyspace keyspace = Keyspace.open(keyspaceName);
|
||||
Token token = mutation.key().getToken();
|
||||
|
||||
ReplicaPlan.ForWrite plan = ReplicaPlans.forWrite(keyspace, consistencyLevel, token, ReplicaPlans.writeAll);
|
||||
AbstractReplicationStrategy rs = plan.replicationStrategy();
|
||||
AbstractReplicationStrategy rs = cm.schema.getKeyspaceMetadata(keyspaceName).replicationStrategy;
|
||||
CoordinationPlan.ForWriteWithIdeal plan = CoordinationPlan.forWrite(cm, keyspace, consistencyLevel, token, ReplicaPlans.writeAll);
|
||||
|
||||
if (plan.lookup(FBUtilities.getBroadcastAddressAndPort()) == null)
|
||||
if (plan.replicas().lookup(FBUtilities.getBroadcastAddressAndPort()) == null)
|
||||
{
|
||||
logger.trace("Remote tracked request {} {}", mutation, plan);
|
||||
writeMetrics.remoteRequests.mark();
|
||||
|
|
@ -193,19 +195,19 @@ public class TrackedWriteRequest
|
|||
if (logger.isTraceEnabled())
|
||||
{
|
||||
logger.trace("Write replication plan for mutation {}: live={}, pending={}, all={}",
|
||||
id, plan.live(), plan.pending(), plan.contacts());
|
||||
id, plan.replicas().live(), plan.replicas().pending(), plan.replicas().contacts());
|
||||
}
|
||||
|
||||
final TrackedWriteResponseHandler handler;
|
||||
if (mutation instanceof CounterMutation)
|
||||
{
|
||||
handler = TrackedWriteResponseHandler.wrap(rs.getWriteResponseHandler(plan, null, WriteType.COUNTER, null, requestTime), id);
|
||||
applyCounterMutationLocally((CounterMutation) mutation, plan, handler);
|
||||
applyCounterMutationLocally((CounterMutation) mutation, plan.replicas(), handler);
|
||||
}
|
||||
else
|
||||
{
|
||||
handler = TrackedWriteResponseHandler.wrap(rs.getWriteResponseHandler(plan, null, WriteType.SIMPLE, null, requestTime), id);
|
||||
applyLocallyAndSendToReplicas((Mutation) mutation, plan, handler);
|
||||
applyLocallyAndSendToReplicas((Mutation) mutation, plan.replicas(), handler);
|
||||
}
|
||||
return handler;
|
||||
}
|
||||
|
|
@ -264,7 +266,7 @@ public class TrackedWriteRequest
|
|||
logger.trace("Skipping dead replica {} for mutation {}", destination, mutation.id());
|
||||
// Only call expired() for AbstractWriteResponseHandler (not for LeaderCallback)
|
||||
if (handler instanceof AbstractWriteResponseHandler)
|
||||
((AbstractWriteResponseHandler<?>) handler).expired(); // immediately mark the response as expired since the request will not be sent
|
||||
((AbstractWriteResponseHandler<?>) handler).expired(destination.endpoint()); // immediately mark the response as expired since the request will not be sent
|
||||
continue;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -29,6 +29,8 @@ import java.util.function.Supplier;
|
|||
|
||||
import javax.annotation.Nullable;
|
||||
|
||||
import com.google.common.annotations.VisibleForTesting;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
|
|
@ -43,9 +45,9 @@ import org.apache.cassandra.exceptions.RequestFailureReason;
|
|||
import org.apache.cassandra.exceptions.RetryOnDifferentSystemException;
|
||||
import org.apache.cassandra.exceptions.WriteFailureException;
|
||||
import org.apache.cassandra.exceptions.WriteTimeoutException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlan.ForWrite;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.RequestCallback;
|
||||
|
|
@ -81,17 +83,17 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
//Count down until all responses and expirations have occured before deciding whether the ideal CL was reached.
|
||||
private AtomicInteger responsesAndExpirations;
|
||||
private final Condition condition = newOneTimeCondition();
|
||||
protected final ReplicaPlan.ForWrite replicaPlan;
|
||||
protected final CoordinationPlan.ForWrite plan;
|
||||
|
||||
protected final Runnable callback;
|
||||
protected final WriteType writeType;
|
||||
private static final AtomicIntegerFieldUpdater<AbstractWriteResponseHandler> failuresUpdater =
|
||||
AtomicIntegerFieldUpdater.newUpdater(AbstractWriteResponseHandler.class, "failures");
|
||||
private volatile int failures = 0;
|
||||
private static final AtomicIntegerFieldUpdater<AbstractWriteResponseHandler> alreadyHintedForRetryOnDifferentSystemUpdater =
|
||||
AtomicIntegerFieldUpdater.newUpdater(AbstractWriteResponseHandler.class, "alreadyHintedForRetryOnDifferentSystem");
|
||||
// Only write a hint to be applied as a transaction once
|
||||
private volatile int alreadyHintedForRetryOnDifferentSystem = 0;
|
||||
private static final AtomicIntegerFieldUpdater<AbstractWriteResponseHandler> signaledUpdater =
|
||||
AtomicIntegerFieldUpdater.newUpdater(AbstractWriteResponseHandler.class, "signaled");
|
||||
private volatile int signaled = 0;
|
||||
private volatile Map<InetAddressAndPort, RequestFailureReason> failureReasonByEndpoint;
|
||||
private final Dispatcher.RequestTime requestTime;
|
||||
private @Nullable final Supplier<Mutation> hintOnFailure;
|
||||
|
|
@ -117,16 +119,26 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
* @param hintOnFailure
|
||||
* @param requestTime
|
||||
*/
|
||||
protected AbstractWriteResponseHandler(ForWrite replicaPlan, Runnable callback, WriteType writeType,
|
||||
protected AbstractWriteResponseHandler(CoordinationPlan.ForWrite plan, Runnable callback, WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this.replicaPlan = replicaPlan;
|
||||
this.plan = plan;
|
||||
this.callback = callback;
|
||||
this.writeType = writeType;
|
||||
this.hintOnFailure = hintOnFailure;
|
||||
this.requestTime = requestTime;
|
||||
}
|
||||
|
||||
public CoordinationPlan.ForWrite coordinationPlan()
|
||||
{
|
||||
return plan;
|
||||
}
|
||||
|
||||
public ForWrite replicaPlan()
|
||||
{
|
||||
return plan.replicas();
|
||||
}
|
||||
|
||||
public void get() throws WriteTimeoutException, WriteFailureException, RetryOnDifferentSystemException
|
||||
{
|
||||
long timeoutNanos = currentTimeoutNanos();
|
||||
|
|
@ -144,9 +156,9 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
if (!signaled)
|
||||
throwTimeout();
|
||||
|
||||
int candidateReplicaCount = candidateReplicaCount();
|
||||
if (blockFor() + failures > candidateReplicaCount)
|
||||
if (!plan.responses().isSuccessful())
|
||||
{
|
||||
int candidateReplicaCount = candidateReplicaCount();
|
||||
// failures keeps incrementing, and this.failureReasonByEndpoint keeps getting new entries after signaling.
|
||||
// Simpler to reason about what happened by copying this.failureReasonByEndpoint and then inferring
|
||||
// failures from it
|
||||
|
|
@ -179,10 +191,10 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
throw new CoordinatorBehindException("Write request failed due to coordinator behind");
|
||||
}
|
||||
|
||||
throw new WriteFailureException(replicaPlan.consistencyLevel(), ackCount(), blockFor(), writeType, getFailureReasonByEndpointMap());
|
||||
throw new WriteFailureException(replicaPlan().consistencyLevel(), ackCount(), blockFor(), writeType, getFailureReasonByEndpointMap());
|
||||
}
|
||||
|
||||
if (replicaPlan.stillAppliesTo(ClusterMetadata.current()))
|
||||
if (replicaPlan().stillAppliesTo(ClusterMetadata.current()))
|
||||
{
|
||||
if (warningContext != null)
|
||||
{
|
||||
|
|
@ -217,7 +229,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
// avoid sending confusing info to the user (see CASSANDRA-6491).
|
||||
if (acks >= blockedFor)
|
||||
acks = blockedFor - 1;
|
||||
throw new WriteTimeoutException(writeType, replicaPlan.consistencyLevel(), acks, blockedFor);
|
||||
throw new WriteTimeoutException(writeType, replicaPlan().consistencyLevel(), acks, blockedFor);
|
||||
}
|
||||
|
||||
public final long currentTimeoutNanos()
|
||||
|
|
@ -236,7 +248,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
public void setIdealCLResponseHandler(AbstractWriteResponseHandler handler)
|
||||
{
|
||||
this.idealCLDelegate = handler;
|
||||
idealCLDelegate.responsesAndExpirations = new AtomicInteger(replicaPlan.contacts().size());
|
||||
idealCLDelegate.responsesAndExpirations = new AtomicInteger(replicaPlan().contacts().size());
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -287,9 +299,12 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
}
|
||||
}
|
||||
|
||||
public final void expired()
|
||||
public final void expired(InetAddressAndPort from)
|
||||
{
|
||||
plan.responses().onFailure(from);
|
||||
logFailureOrTimeoutToIdealCLDelegate();
|
||||
if (plan.responses().isComplete() && !plan.responses().isSuccessful())
|
||||
signal();
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -299,7 +314,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
{
|
||||
// During bootstrap, we have to include the pending endpoints or we may fail the consistency level
|
||||
// guarantees (see #833)
|
||||
return replicaPlan.writeQuorum();
|
||||
return replicaPlan().writeQuorum();
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -309,15 +324,15 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
*/
|
||||
protected int candidateReplicaCount()
|
||||
{
|
||||
if (replicaPlan.consistencyLevel().isDatacenterLocal())
|
||||
return countInOurDc(replicaPlan.liveAndDown()).allReplicas();
|
||||
if (replicaPlan().consistencyLevel().isDatacenterLocal())
|
||||
return countInOurDc(replicaPlan().liveAndDown()).allReplicas();
|
||||
|
||||
return replicaPlan.liveAndDown().size();
|
||||
return replicaPlan().liveAndDown().size();
|
||||
}
|
||||
|
||||
public ConsistencyLevel consistencyLevel()
|
||||
{
|
||||
return replicaPlan.consistencyLevel();
|
||||
return replicaPlan().consistencyLevel();
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -345,14 +360,17 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
|
||||
protected void signal()
|
||||
{
|
||||
if (!signaledUpdater.compareAndSet(this, 0, 1))
|
||||
return;
|
||||
|
||||
//The ideal CL should only count as a strike if the requested CL was achieved.
|
||||
//If the requested CL is not achieved it's fine for the ideal CL to also not be achieved.
|
||||
if (idealCLDelegate != null && blockFor() + failures <= candidateReplicaCount())
|
||||
if (idealCLDelegate != null && plan.responses().isSuccessful())
|
||||
{
|
||||
idealCLDelegate.requestedCLAchieved = true;
|
||||
if (idealCLDelegate == this)
|
||||
{
|
||||
replicaPlan.keyspace().metric.idealCLWriteLatency.addNano(nanoTime() - requestTime.startedAtNanos());
|
||||
replicaPlan().keyspace().metric.idealCLWriteLatency.addNano(nanoTime() - requestTime.startedAtNanos());
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -361,15 +379,17 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
callback.run();
|
||||
}
|
||||
|
||||
@VisibleForTesting
|
||||
public boolean isComplete()
|
||||
{
|
||||
return condition.isSignalled();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onFailure(InetAddressAndPort from, RequestFailure failure)
|
||||
{
|
||||
logger.trace("Got failure {} from {}", failure, from);
|
||||
|
||||
int n = waitingFor(from)
|
||||
? failuresUpdater.incrementAndGet(this)
|
||||
: failures;
|
||||
|
||||
if (failureReasonByEndpoint == null)
|
||||
synchronized (this)
|
||||
{
|
||||
|
|
@ -378,14 +398,16 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
}
|
||||
failureReasonByEndpoint.put(from, failure.reason);
|
||||
|
||||
plan.responses().onFailure(from);
|
||||
|
||||
logFailureOrTimeoutToIdealCLDelegate();
|
||||
|
||||
if (blockFor() + n > candidateReplicaCount())
|
||||
if (plan.responses().isComplete() && !plan.responses().isSuccessful())
|
||||
signal();
|
||||
|
||||
// If the failure was RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM then we only want to hint once
|
||||
// and not for each instance since odds are it will be applied as a transaction at all replicas
|
||||
if (hintOnFailure != null && StorageProxy.shouldHint(replicaPlan.lookup(from)) )
|
||||
if (hintOnFailure != null && StorageProxy.shouldHint(replicaPlan().lookup(from)))
|
||||
{
|
||||
if (failure.reason == RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM)
|
||||
{
|
||||
|
|
@ -394,7 +416,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
}
|
||||
else
|
||||
{
|
||||
StorageProxy.submitHint(hintOnFailure.get(), replicaPlan.lookup(from), null);
|
||||
StorageProxy.submitHint(hintOnFailure.get(), replicaPlan().lookup(from), null);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -419,7 +441,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
// Only mark it as failed if the requested CL was achieved.
|
||||
if (!condition.isSignalled() && requestedCLAchieved)
|
||||
{
|
||||
replicaPlan.keyspace().metric.writeFailedIdealCL.inc();
|
||||
replicaPlan().keyspace().metric.writeFailedIdealCL.inc();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -429,7 +451,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
*/
|
||||
public void maybeTryAdditionalReplicas(IMutation mutation, WritePerformer writePerformer, String localDC)
|
||||
{
|
||||
EndpointsForToken uncontacted = replicaPlan.liveUncontacted();
|
||||
EndpointsForToken uncontacted = replicaPlan().liveUncontacted();
|
||||
if (uncontacted.isEmpty())
|
||||
return;
|
||||
|
||||
|
|
@ -451,7 +473,7 @@ public abstract class AbstractWriteResponseHandler<T> implements RequestCallback
|
|||
for (ColumnFamilyStore cf : cfs)
|
||||
cf.metric.additionalWrites.inc();
|
||||
|
||||
writePerformer.apply(mutation, replicaPlan.withContacts(uncontacted),
|
||||
writePerformer.apply(mutation, replicaPlan().withContacts(uncontacted),
|
||||
(AbstractWriteResponseHandler<IMutation>) this,
|
||||
localDC,
|
||||
requestTime);
|
||||
|
|
|
|||
|
|
@ -38,7 +38,7 @@ public class BatchlogResponseHandler<T> extends AbstractWriteResponseHandler<T>
|
|||
|
||||
public BatchlogResponseHandler(AbstractWriteResponseHandler<T> wrapped, int requiredBeforeFinish, BatchlogCleanup cleanup, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(wrapped.replicaPlan, wrapped.callback, wrapped.writeType, null, requestTime);
|
||||
super(wrapped.plan, wrapped.callback, wrapped.writeType, null, requestTime);
|
||||
this.wrapped = wrapped;
|
||||
this.requiredBeforeFinish = requiredBeforeFinish;
|
||||
this.cleanup = cleanup;
|
||||
|
|
|
|||
|
|
@ -17,109 +17,65 @@
|
|||
*/
|
||||
package org.apache.cassandra.service;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.MessageParams;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.WriteType;
|
||||
import org.apache.cassandra.locator.Locator;
|
||||
import org.apache.cassandra.locator.NetworkTopologyStrategy;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.ParamType;
|
||||
import org.apache.cassandra.service.writes.thresholds.WriteWarningContext;
|
||||
import org.apache.cassandra.transport.Dispatcher;
|
||||
import org.apache.cassandra.utils.FBUtilities;
|
||||
|
||||
/**
|
||||
* This class blocks for a quorum of responses _in all datacenters_ (CL.EACH_QUORUM).
|
||||
*
|
||||
* The CompositeTracker handles per-datacenter counting.
|
||||
*/
|
||||
public class DatacenterSyncWriteResponseHandler<T> extends AbstractWriteResponseHandler<T>
|
||||
{
|
||||
private static final Locator locator = DatabaseDescriptor.getLocator();
|
||||
|
||||
private final Map<String, AtomicInteger> responses = new HashMap<String, AtomicInteger>();
|
||||
private final AtomicInteger acks = new AtomicInteger(0);
|
||||
|
||||
public DatacenterSyncWriteResponseHandler(ReplicaPlan.ForWrite replicaPlan,
|
||||
public DatacenterSyncWriteResponseHandler(CoordinationPlan.ForWrite coordinationPlan,
|
||||
Runnable callback,
|
||||
WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
// Response is been managed by the map so make it 1 for the superclass.
|
||||
super(replicaPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
assert replicaPlan.consistencyLevel() == ConsistencyLevel.EACH_QUORUM;
|
||||
|
||||
if (replicaPlan.replicationStrategy() instanceof NetworkTopologyStrategy)
|
||||
{
|
||||
NetworkTopologyStrategy strategy = (NetworkTopologyStrategy) replicaPlan.replicationStrategy();
|
||||
for (String dc : strategy.getDatacenters())
|
||||
{
|
||||
int rf = strategy.getReplicationFactor(dc).allReplicas;
|
||||
responses.put(dc, new AtomicInteger((rf / 2) + 1));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
responses.put(locator.local().datacenter, new AtomicInteger(ConsistencyLevel.quorumFor(replicaPlan.replicationStrategy())));
|
||||
}
|
||||
|
||||
// During bootstrap, we have to include the pending endpoints or we may fail the consistency level
|
||||
// guarantees (see #833)
|
||||
for (Replica pending : replicaPlan.pending())
|
||||
{
|
||||
responses.get(locator.location(pending.endpoint()).datacenter).incrementAndGet();
|
||||
}
|
||||
super(coordinationPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
assert replicaPlan().consistencyLevel() == ConsistencyLevel.EACH_QUORUM;
|
||||
}
|
||||
|
||||
public void onResponse(Message<T> message)
|
||||
{
|
||||
try
|
||||
{
|
||||
Map<ParamType, Object> params;
|
||||
String dataCenter;
|
||||
|
||||
if (message != null)
|
||||
{
|
||||
params = message.header.params();
|
||||
dataCenter = locator.location(message.from()).datacenter;
|
||||
}
|
||||
else
|
||||
{
|
||||
params = MessageParams.capture();
|
||||
dataCenter = locator.local().datacenter;
|
||||
}
|
||||
Map<ParamType, Object> params = message != null
|
||||
? message.header.params()
|
||||
: MessageParams.capture();
|
||||
|
||||
if (WriteWarningContext.isSupported(params.keySet()))
|
||||
getWarningContext().updateCounters(params);
|
||||
|
||||
responses.get(dataCenter).getAndDecrement();
|
||||
acks.incrementAndGet();
|
||||
InetAddressAndPort from = message == null ? FBUtilities.getBroadcastAddressAndPort() : message.from();
|
||||
|
||||
for (AtomicInteger i : responses.values())
|
||||
{
|
||||
if (i.get() > 0)
|
||||
return;
|
||||
}
|
||||
plan.responses().onResponse(from);
|
||||
|
||||
// all the quorum conditions are met
|
||||
signal();
|
||||
if (plan.responses().isComplete())
|
||||
signal();
|
||||
}
|
||||
finally
|
||||
{
|
||||
//Must be last after all subclass processing
|
||||
// Must be last - forward to ideal CL delegate
|
||||
logResponseToIdealCLDelegate(message);
|
||||
}
|
||||
}
|
||||
|
||||
protected int ackCount()
|
||||
{
|
||||
return acks.get();
|
||||
return plan.responses().received();
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -17,47 +17,63 @@
|
|||
*/
|
||||
package org.apache.cassandra.service;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import org.apache.cassandra.db.MessageParams;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.WriteType;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.InOurDc;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.ParamType;
|
||||
import org.apache.cassandra.service.writes.thresholds.WriteWarningContext;
|
||||
import org.apache.cassandra.transport.Dispatcher;
|
||||
import org.apache.cassandra.utils.FBUtilities;
|
||||
|
||||
/**
|
||||
* This class blocks for a quorum of responses _in the local datacenter only_ (CL.LOCAL_QUORUM).
|
||||
*
|
||||
* The response tracker handles DC filtering via countsTowardQuorum().
|
||||
* This handler still uses waitingFor to filter collectSuccess() calls.
|
||||
*/
|
||||
public class DatacenterWriteResponseHandler<T> extends WriteResponseHandler<T>
|
||||
{
|
||||
private final Predicate<InetAddressAndPort> waitingFor = InOurDc.endpoints();
|
||||
|
||||
public DatacenterWriteResponseHandler(ReplicaPlan.ForWrite replicaPlan,
|
||||
public DatacenterWriteResponseHandler(CoordinationPlan.ForWrite coordinationPlan,
|
||||
Runnable callback,
|
||||
WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(replicaPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
assert replicaPlan.consistencyLevel().isDatacenterLocal();
|
||||
super(coordinationPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
assert coordinationPlan.consistencyLevel().isDatacenterLocal();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onResponse(Message<T> message)
|
||||
{
|
||||
if (message == null || waitingFor(message.from()))
|
||||
InetAddressAndPort from = message == null ? FBUtilities.getBroadcastAddressAndPort() : message.from();
|
||||
Map<ParamType, Object> params = message != null ? message.header.params() : MessageParams.capture();
|
||||
|
||||
if (WriteWarningContext.isSupported(params.keySet()))
|
||||
getWarningContext().updateCounters(params);
|
||||
|
||||
plan.responses().onResponse(from);
|
||||
|
||||
if (message == null || waitingFor(from))
|
||||
{
|
||||
super.onResponse(message);
|
||||
}
|
||||
else
|
||||
{
|
||||
//WriteResponseHandler.response will call logResonseToIdealCLDelegate so only do it if not calling WriteResponseHandler.response.
|
||||
//Must be last after all subclass processing
|
||||
logResponseToIdealCLDelegate(message);
|
||||
replicaPlan().collectSuccess(from);
|
||||
}
|
||||
|
||||
if (plan.responses().isComplete())
|
||||
signal();
|
||||
|
||||
// Must be last (see comment on AbstractWriteResponseHandler.logResponseToIdealCLDelegate for why) - forward to ideal CL delegate
|
||||
logResponseToIdealCLDelegate(message);
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
|
|||
|
|
@ -120,6 +120,7 @@ import org.apache.cassandra.gms.Gossiper;
|
|||
import org.apache.cassandra.hints.Hint;
|
||||
import org.apache.cassandra.hints.HintsService;
|
||||
import org.apache.cassandra.locator.AbstractReplicationStrategy;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.DynamicEndpointSnitch;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
|
|
@ -1088,11 +1089,11 @@ public class StorageProxy implements StorageProxyMBean
|
|||
}
|
||||
|
||||
// NOTE: this ReplicaPlan is a lie, this usage of ReplicaPlan could do with being clarified - the selected() collection is essentially (I think) never used
|
||||
ReplicaPlan.ForWrite replicaPlan = ReplicaPlans.forWrite(keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
AbstractReplicationStrategy rs = replicaPlan.replicationStrategy();
|
||||
CoordinationPlan.ForWriteWithIdeal plan = CoordinationPlan.forWrite(ClusterMetadata.current(), keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
AbstractReplicationStrategy rs = plan.replicationStrategy();
|
||||
|
||||
// If we are coordinating a new mutation id for the first time then create a TrackedWriteResponseHandler
|
||||
AbstractWriteResponseHandler<?> responseHandler = rs.getWriteResponseHandler(replicaPlan, null, WriteType.SIMPLE, null, requestTime);
|
||||
AbstractWriteResponseHandler<?> responseHandler = rs.getWriteResponseHandler(plan, null, WriteType.SIMPLE, null, requestTime);
|
||||
responseHandler = TrackedWriteResponseHandler.wrap(responseHandler, mutationId);
|
||||
|
||||
// For tracked keyspaces, the local commit MUST execute synchronously BEFORE sending to remote replicas.
|
||||
|
|
@ -1100,7 +1101,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
// reconciliation via ActiveLogReconciler.
|
||||
Message<Commit> message = Message.outWithFlag(PAXOS_COMMIT_REQ, proposal, MessageFlag.CALL_BACK_ON_FAILURE);
|
||||
Replica localReplica = null;
|
||||
for (Replica replica : replicaPlan.liveAndDown())
|
||||
for (Replica replica : plan.replicas().liveAndDown())
|
||||
{
|
||||
if (replica.isSelf())
|
||||
{
|
||||
|
|
@ -1129,7 +1130,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
|
||||
// Now send to remote replicas
|
||||
IntHashSet remoteReplicas = new IntHashSet();
|
||||
for (Replica replica : replicaPlan.liveAndDown())
|
||||
for (Replica replica : plan.replicas().liveAndDown())
|
||||
{
|
||||
if (replica.isSelf())
|
||||
continue; // Already executed locally above
|
||||
|
|
@ -1158,22 +1159,22 @@ public class StorageProxy implements StorageProxyMBean
|
|||
|
||||
Token tk = update.partitionKey().getToken();
|
||||
|
||||
ClusterMetadata cm = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForWriteWithIdeal plan = CoordinationPlan.forWrite(cm, keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
|
||||
AbstractWriteResponseHandler<Commit> responseHandler = null;
|
||||
// NOTE: this ReplicaPlan is a lie, this usage of ReplicaPlan could do with being clarified - the selected() collection is essentially (I think) never used
|
||||
ReplicaPlan.ForWrite replicaPlan = ReplicaPlans.forWrite(keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
if (shouldBlock)
|
||||
{
|
||||
AbstractReplicationStrategy rs = replicaPlan.replicationStrategy();
|
||||
responseHandler = rs.getWriteResponseHandler(replicaPlan, null, WriteType.SIMPLE, proposal::makeMutation, requestTime);
|
||||
}
|
||||
responseHandler = plan.replicationStrategy().getWriteResponseHandler(plan, null, WriteType.SIMPLE, proposal::makeMutation, requestTime);
|
||||
|
||||
Message<Commit> message = Message.outWithFlag(PAXOS_COMMIT_REQ, proposal, MessageFlag.CALL_BACK_ON_FAILURE);
|
||||
for (Replica replica : replicaPlan.liveAndDown())
|
||||
for (Replica replica : plan.replicas().liveAndDown())
|
||||
{
|
||||
InetAddressAndPort destination = replica.endpoint();
|
||||
checkHintOverload(replica);
|
||||
|
||||
if (replicaPlan.isAlive(replica))
|
||||
if (plan.replicas().isAlive(replica))
|
||||
{
|
||||
if (shouldBlock)
|
||||
{
|
||||
|
|
@ -1191,7 +1192,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
{
|
||||
if (responseHandler != null)
|
||||
{
|
||||
responseHandler.expired();
|
||||
responseHandler.expired(destination);
|
||||
}
|
||||
if (allowHints && shouldHint(replica))
|
||||
{
|
||||
|
|
@ -1616,11 +1617,11 @@ public class StorageProxy implements StorageProxyMBean
|
|||
pending.get());
|
||||
};
|
||||
|
||||
ReplicaPlan.ForWrite replicaPlan = ReplicaPlans.forWrite(metadata, Keyspace.open(keyspaceName), consistencyLevel, computeReplicas, ReplicaPlans.writeAll);
|
||||
CoordinationPlan.ForWriteWithIdeal plan = CoordinationPlan.forWrite(metadata, Keyspace.open(keyspaceName), consistencyLevel, computeReplicas, ReplicaPlans.writeAll);
|
||||
|
||||
wrappers.add(wrapViewBatchResponseHandler(mutation,
|
||||
consistencyLevel,
|
||||
replicaPlan,
|
||||
plan,
|
||||
baseComplete,
|
||||
WriteType.BATCH,
|
||||
cleanup,
|
||||
|
|
@ -1911,7 +1912,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
{
|
||||
ConsistencyLevel batchConsistencyLevel = consistencyLevelForBatchLog(consistencyLevel, requireQuorumForRemove);
|
||||
// This can't be updated for each iteration because cleanup has to go to the correct replicas which is where the batchlog is originally written
|
||||
ReplicaPlan.ForWrite batchlogReplicaPlan = ReplicaPlans.forBatchlogWrite(ClusterMetadata.current(), batchConsistencyLevel == ConsistencyLevel.ANY);
|
||||
CoordinationPlan.ForWrite coordinationPlan = CoordinationPlan.ForWriteWithIdeal.forBatchlogWrite(ClusterMetadata.current(), batchConsistencyLevel == ConsistencyLevel.ANY);
|
||||
final TimeUUID batchUUID = nextTimeUUID();
|
||||
boolean wroteToBatchLog = false;
|
||||
while (true)
|
||||
|
|
@ -1921,7 +1922,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
attributeNonAccordLatency = true;
|
||||
List<WriteResponseHandlerWrapper> wrappers = new ArrayList<>(mutations.size());
|
||||
List<Mutation> accordMutations = new ArrayList<>(mutations.size());
|
||||
BatchlogCleanup cleanup = new BatchlogCleanup(() -> asyncRemoveFromBatchlog(batchlogReplicaPlan, batchUUID, requestTime));
|
||||
BatchlogCleanup cleanup = new BatchlogCleanup(() -> asyncRemoveFromBatchlog(coordinationPlan.replicas(), batchUUID, requestTime));
|
||||
|
||||
// add a handler for each mutation that will not be written on Accord - includes checking availability, but doesn't initiate any writes, yet
|
||||
SplitConsumer<Mutation> splitConsumer = (accordMutation, untrackedMutation, trackedMutation, originalMutations, mutationIndex) -> {
|
||||
|
|
@ -1940,15 +1941,15 @@ public class StorageProxy implements StorageProxyMBean
|
|||
|
||||
|
||||
// Always construct the replica plan to check availability
|
||||
ReplicaPlan.ForWrite dataReplicaPlan = ReplicaPlans.forWrite(cm, keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
CoordinationPlan.ForWriteWithIdeal plan = CoordinationPlan.forWrite(cm, keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
|
||||
if (dataReplicaPlan.lookup(FBUtilities.getBroadcastAddressAndPort()) != null)
|
||||
if (plan.replicas().lookup(FBUtilities.getBroadcastAddressAndPort()) != null)
|
||||
writeMetrics.localRequests.mark();
|
||||
else
|
||||
writeMetrics.remoteRequests.mark();
|
||||
|
||||
WriteResponseHandlerWrapper wrapper = wrapBatchResponseHandler(untrackedMutation,
|
||||
dataReplicaPlan,
|
||||
plan,
|
||||
batchConsistencyLevel,
|
||||
WriteType.BATCH,
|
||||
cleanup,
|
||||
|
|
@ -1971,7 +1972,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
// with the mutations delivered by the batch log since an unacknowledged Accord txn won't be retried
|
||||
// unless those mutations are also written to the batch log
|
||||
// Only write to the log once and reuse the batchUUID for every attempt to route the mutations correctly
|
||||
doFallibleWriteWithMetricTracking(() -> syncWriteToBatchlog(mutations, batchlogReplicaPlan, batchUUID, requestTime), consistencyLevel);
|
||||
doFallibleWriteWithMetricTracking(() -> syncWriteToBatchlog(mutations, coordinationPlan, batchUUID, requestTime), consistencyLevel);
|
||||
Tracing.trace("Successfully wrote to batchlog");
|
||||
wroteToBatchLog = true;
|
||||
}
|
||||
|
|
@ -2104,17 +2105,17 @@ public class StorageProxy implements StorageProxyMBean
|
|||
}
|
||||
}
|
||||
|
||||
private static void syncWriteToBatchlog(Collection<Mutation> mutations, ReplicaPlan.ForWrite replicaPlan, TimeUUID uuid, Dispatcher.RequestTime requestTime)
|
||||
private static void syncWriteToBatchlog(Collection<Mutation> mutations, CoordinationPlan.ForWrite coordinationPlan, TimeUUID uuid, Dispatcher.RequestTime requestTime)
|
||||
throws WriteTimeoutException, WriteFailureException
|
||||
{
|
||||
WriteResponseHandler<?> handler = new WriteResponseHandler<>(replicaPlan,
|
||||
WriteResponseHandler<?> handler = new WriteResponseHandler<>(coordinationPlan,
|
||||
WriteType.BATCH_LOG,
|
||||
null,
|
||||
requestTime);
|
||||
|
||||
Batch batch = Batch.createLocal(uuid, FBUtilities.timestampMicros(), mutations);
|
||||
Message<Batch> message = Message.out(BATCH_STORE_REQ, batch);
|
||||
for (Replica replica : replicaPlan.liveAndDown())
|
||||
for (Replica replica : coordinationPlan.replicas().liveAndDown())
|
||||
{
|
||||
if (logger.isTraceEnabled())
|
||||
logger.trace("Sending batchlog store request {} to {} for {} mutations", batch.id, replica, batch.size());
|
||||
|
|
@ -2146,8 +2147,8 @@ public class StorageProxy implements StorageProxyMBean
|
|||
{
|
||||
for (WriteResponseHandlerWrapper wrapper : wrappers)
|
||||
{
|
||||
Replicas.temporaryAssertFull(wrapper.handler.replicaPlan.liveAndDown()); // TODO: CASSANDRA-14549
|
||||
ReplicaPlan.ForWrite replicas = wrapper.handler.replicaPlan.withContacts(wrapper.handler.replicaPlan.liveAndDown());
|
||||
Replicas.temporaryAssertFull(wrapper.handler.replicaPlan().liveAndDown()); // TODO: CASSANDRA-14549
|
||||
ReplicaPlan.ForWrite replicas = wrapper.handler.replicaPlan().withContacts(wrapper.handler.replicaPlan().liveAndDown());
|
||||
|
||||
try
|
||||
{
|
||||
|
|
@ -2167,9 +2168,9 @@ public class StorageProxy implements StorageProxyMBean
|
|||
|
||||
for (WriteResponseHandlerWrapper wrapper : wrappers)
|
||||
{
|
||||
EndpointsForToken sendTo = wrapper.handler.replicaPlan.liveAndDown();
|
||||
EndpointsForToken sendTo = wrapper.handler.replicaPlan().liveAndDown();
|
||||
Replicas.temporaryAssertFull(sendTo); // TODO: CASSANDRA-14549
|
||||
sendToHintedReplicas(wrapper.mutation, wrapper.handler.replicaPlan.withContacts(sendTo), wrapper.handler, localDataCenter, stage, requestTime);
|
||||
sendToHintedReplicas(wrapper.mutation, wrapper.handler.replicaPlan().withContacts(sendTo), wrapper.handler, localDataCenter, stage, requestTime);
|
||||
}
|
||||
|
||||
for (WriteResponseHandlerWrapper wrapper : wrappers)
|
||||
|
|
@ -2202,30 +2203,32 @@ public class StorageProxy implements StorageProxyMBean
|
|||
Keyspace keyspace = Keyspace.open(keyspaceName);
|
||||
Token tk = mutation.key().getToken();
|
||||
|
||||
ReplicaPlan.ForWrite replicaPlan = ReplicaPlans.forWrite(keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
|
||||
if (replicaPlan.lookup(FBUtilities.getBroadcastAddressAndPort()) != null)
|
||||
ClusterMetadata cm = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForWriteWithIdeal plan = CoordinationPlan.forWrite(cm, keyspace, consistencyLevel, tk, ReplicaPlans.writeAll);
|
||||
|
||||
if (plan.replicas().lookup(FBUtilities.getBroadcastAddressAndPort()) != null)
|
||||
writeMetrics.localRequests.mark();
|
||||
else
|
||||
writeMetrics.remoteRequests.mark();
|
||||
|
||||
AbstractReplicationStrategy rs = replicaPlan.replicationStrategy();
|
||||
AbstractWriteResponseHandler<IMutation> responseHandler = rs.getWriteResponseHandler(replicaPlan, callback, writeType, mutation.hintOnFailure(), requestTime);
|
||||
AbstractWriteResponseHandler<IMutation> responseHandler = plan.replicationStrategy().getWriteResponseHandler(plan, callback, writeType, mutation.hintOnFailure(), requestTime);
|
||||
|
||||
performer.apply(mutation, replicaPlan, responseHandler, localDataCenter, requestTime);
|
||||
performer.apply(mutation, plan.replicas(), responseHandler, localDataCenter, requestTime);
|
||||
return responseHandler;
|
||||
}
|
||||
|
||||
// same as performWrites except does not initiate writes (but does perform availability checks).
|
||||
private static WriteResponseHandlerWrapper wrapBatchResponseHandler(Mutation mutation,
|
||||
ReplicaPlan.ForWrite replicaPlan,
|
||||
CoordinationPlan.ForWriteWithIdeal plan,
|
||||
ConsistencyLevel batchConsistencyLevel,
|
||||
WriteType writeType,
|
||||
BatchlogResponseHandler.BatchlogCleanup cleanup,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
AbstractReplicationStrategy rs = replicaPlan.replicationStrategy();
|
||||
AbstractWriteResponseHandler<IMutation> writeHandler = rs.getWriteResponseHandler(replicaPlan, null, writeType, mutation, requestTime);
|
||||
AbstractReplicationStrategy rs = plan.replicationStrategy();
|
||||
AbstractWriteResponseHandler<IMutation> writeHandler = rs.getWriteResponseHandler(plan, null, writeType, mutation, requestTime);
|
||||
BatchlogResponseHandler<IMutation> batchHandler = new BatchlogResponseHandler<>(writeHandler, batchConsistencyLevel.blockFor(rs), cleanup, requestTime);
|
||||
return new WriteResponseHandlerWrapper(batchHandler, mutation);
|
||||
}
|
||||
|
|
@ -2236,14 +2239,14 @@ public class StorageProxy implements StorageProxyMBean
|
|||
*/
|
||||
private static WriteResponseHandlerWrapper wrapViewBatchResponseHandler(Mutation mutation,
|
||||
ConsistencyLevel batchConsistencyLevel,
|
||||
ReplicaPlan.ForWrite replicaPlan,
|
||||
CoordinationPlan.ForWriteWithIdeal plan,
|
||||
AtomicLong baseComplete,
|
||||
WriteType writeType,
|
||||
BatchlogResponseHandler.BatchlogCleanup cleanup,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
AbstractReplicationStrategy replicationStrategy = replicaPlan.replicationStrategy();
|
||||
AbstractWriteResponseHandler<IMutation> writeHandler = replicationStrategy.getWriteResponseHandler(replicaPlan, () -> {
|
||||
AbstractReplicationStrategy replicationStrategy = plan.replicationStrategy();
|
||||
AbstractWriteResponseHandler<IMutation> writeHandler = replicationStrategy.getWriteResponseHandler(plan, () -> {
|
||||
long delay = Math.max(0, currentTimeMillis() - baseComplete.get());
|
||||
viewWriteMetrics.viewWriteLatency.update(delay, MILLISECONDS);
|
||||
}, writeType, mutation, requestTime);
|
||||
|
|
@ -2362,7 +2365,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
else
|
||||
{
|
||||
//Immediately mark the response as expired since the request will not be sent
|
||||
responseHandler.expired();
|
||||
responseHandler.expired(destination.endpoint());
|
||||
if (shouldHint(destination))
|
||||
{
|
||||
if (endpointsToHint == null)
|
||||
|
|
@ -2567,10 +2570,10 @@ public class StorageProxy implements StorageProxyMBean
|
|||
// there we'll mark a local request against the metrics.
|
||||
writeMetrics.remoteRequests.mark();
|
||||
|
||||
ReplicaPlan.ForWrite forWrite = ReplicaPlans.forForwardingCounterWrite(metadata, keyspace, tk,
|
||||
clm -> ReplicaPlans.findCounterLeaderReplica(clm, cm.getKeyspaceName(), cm.key(), localDataCenter, cm.consistency()));
|
||||
CoordinationPlan.ForWrite plan = CoordinationPlan.forForwardingCounterWrite(metadata, keyspace, tk,
|
||||
clm -> ReplicaPlans.findCounterLeaderReplica(clm, cm.getKeyspaceName(), cm.key(), localDataCenter, cm.consistency()));
|
||||
// Forward the actual update to the chosen leader replica
|
||||
AbstractWriteResponseHandler<IMutation> responseHandler = new WriteResponseHandler<>(forWrite,
|
||||
AbstractWriteResponseHandler<IMutation> responseHandler = new WriteResponseHandler<>(plan,
|
||||
WriteType.COUNTER, null, requestTime);
|
||||
|
||||
Tracing.trace("Enqueuing counter update to {}", replica);
|
||||
|
|
@ -3831,7 +3834,7 @@ public class StorageProxy implements StorageProxyMBean
|
|||
HintsService.instance.write(hostIds, Hint.create(mutation, creationTime));
|
||||
validTargets.forEach(HintsService.instance.metrics::incrCreatedHints);
|
||||
// Notify the handler only for CL == ANY
|
||||
if (responseHandler != null && responseHandler.replicaPlan.consistencyLevel() == ConsistencyLevel.ANY)
|
||||
if (responseHandler != null && responseHandler.replicaPlan().consistencyLevel() == ConsistencyLevel.ANY)
|
||||
responseHandler.onResponse(null);
|
||||
}
|
||||
};
|
||||
|
|
|
|||
|
|
@ -38,7 +38,7 @@ public class TrackedWriteResponseHandler<T> extends AbstractWriteResponseHandler
|
|||
|
||||
private TrackedWriteResponseHandler(AbstractWriteResponseHandler<T> wrapped, MutationId mutationId)
|
||||
{
|
||||
super(wrapped.replicaPlan, wrapped.callback, wrapped.writeType, null, wrapped.getRequestTime());
|
||||
super(wrapped.plan, wrapped.callback, wrapped.writeType, null, wrapped.getRequestTime());
|
||||
this.wrapped = wrapped;
|
||||
this.mutationId = mutationId;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -18,7 +18,6 @@
|
|||
package org.apache.cassandra.service;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.atomic.AtomicIntegerFieldUpdater;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
|
|
@ -27,8 +26,8 @@ import org.slf4j.LoggerFactory;
|
|||
import org.apache.cassandra.db.MessageParams;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.WriteType;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.ParamType;
|
||||
import org.apache.cassandra.service.writes.thresholds.WriteWarningContext;
|
||||
|
|
@ -37,28 +36,25 @@ import org.apache.cassandra.utils.FBUtilities;
|
|||
|
||||
/**
|
||||
* Handles blocking writes for ONE, ANY, TWO, THREE, QUORUM, and ALL consistency levels.
|
||||
*
|
||||
* Response tracking is delegated to coordinationPlan.tracker().
|
||||
*/
|
||||
public class WriteResponseHandler<T> extends AbstractWriteResponseHandler<T>
|
||||
{
|
||||
protected static final Logger logger = LoggerFactory.getLogger(WriteResponseHandler.class);
|
||||
|
||||
protected volatile int responses;
|
||||
private static final AtomicIntegerFieldUpdater<WriteResponseHandler> responsesUpdater
|
||||
= AtomicIntegerFieldUpdater.newUpdater(WriteResponseHandler.class, "responses");
|
||||
|
||||
public WriteResponseHandler(ReplicaPlan.ForWrite replicaPlan,
|
||||
public WriteResponseHandler(CoordinationPlan.ForWrite coordinationPlan,
|
||||
Runnable callback,
|
||||
WriteType writeType,
|
||||
Supplier<Mutation> hintOnFailure,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(replicaPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
responses = blockFor();
|
||||
super(coordinationPlan, callback, writeType, hintOnFailure, requestTime);
|
||||
}
|
||||
|
||||
public WriteResponseHandler(ReplicaPlan.ForWrite replicaPlan, WriteType writeType, Supplier<Mutation> hintOnFailure, Dispatcher.RequestTime requestTime)
|
||||
public WriteResponseHandler(CoordinationPlan.ForWrite coordinationPlan, WriteType writeType, Supplier<Mutation> hintOnFailure, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this(replicaPlan, null, writeType, hintOnFailure, requestTime);
|
||||
this(coordinationPlan, null, writeType, hintOnFailure, requestTime);
|
||||
}
|
||||
|
||||
public void onResponse(Message<T> m)
|
||||
|
|
@ -69,17 +65,19 @@ public class WriteResponseHandler<T> extends AbstractWriteResponseHandler<T>
|
|||
if (WriteWarningContext.isSupported(params.keySet()))
|
||||
getWarningContext().updateCounters(params);
|
||||
|
||||
replicaPlan.collectSuccess(from);
|
||||
if (responsesUpdater.decrementAndGet(this) == 0)
|
||||
replicaPlan().collectSuccess(from);
|
||||
|
||||
plan.responses().onResponse(from);
|
||||
|
||||
if (plan.responses().isComplete())
|
||||
signal();
|
||||
//Must be last after all subclass processing
|
||||
//The two current subclasses both assume logResponseToIdealCLDelegate is called
|
||||
//here.
|
||||
|
||||
// Must be last (see comment on AbstractWriteResponseHandler.logResponseToIdealCLDelegate for why) - forward to ideal CL delegate
|
||||
logResponseToIdealCLDelegate(m);
|
||||
}
|
||||
|
||||
protected int ackCount()
|
||||
{
|
||||
return blockFor() - responses;
|
||||
return plan.responses().received();
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -28,7 +28,6 @@ import java.util.Set;
|
|||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import javax.annotation.Nullable;
|
||||
|
||||
|
|
@ -47,6 +46,7 @@ import org.apache.cassandra.db.ConsistencyLevel;
|
|||
import org.apache.cassandra.db.DecoratedKey;
|
||||
import org.apache.cassandra.db.IReadResponse;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.db.ReadKind;
|
||||
import org.apache.cassandra.db.ReadResponse;
|
||||
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||
|
|
@ -87,8 +87,8 @@ import org.apache.cassandra.locator.ReplicaLayout.ForTokenWrite;
|
|||
import org.apache.cassandra.locator.ReplicaPlan.ForRead;
|
||||
import org.apache.cassandra.metrics.ClientRequestMetrics;
|
||||
import org.apache.cassandra.metrics.ClientRequestSizeMetrics;
|
||||
import org.apache.cassandra.metrics.PaxosMetrics;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.schema.SchemaConstants;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.ClientState;
|
||||
|
|
@ -103,6 +103,7 @@ import org.apache.cassandra.service.reads.DataResolver;
|
|||
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||
import org.apache.cassandra.service.reads.repair.NoopReadRepair;
|
||||
import org.apache.cassandra.service.reads.tracked.TrackedDataResponse;
|
||||
import org.apache.cassandra.service.reads.tracked.TrackedRead;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
import org.apache.cassandra.tcm.membership.NodeId;
|
||||
|
|
@ -404,7 +405,7 @@ public class Paxos
|
|||
final Epoch epoch;
|
||||
final Function<ClusterMetadata, Participants> recompute;
|
||||
|
||||
Participants(Epoch epoch, Keyspace keyspace, ConsistencyLevel consistencyForConsensus, ReplicaLayout.ForTokenWrite all, ReplicaLayout.ForTokenWrite electorate, EndpointsForToken live,
|
||||
public Participants(Epoch epoch, Keyspace keyspace, ConsistencyLevel consistencyForConsensus, ReplicaLayout.ForTokenWrite all, ReplicaLayout.ForTokenWrite electorate, EndpointsForToken live,
|
||||
Function<ClusterMetadata, Participants> recompute)
|
||||
{
|
||||
this.epoch = epoch;
|
||||
|
|
@ -464,26 +465,15 @@ public class Paxos
|
|||
|
||||
}
|
||||
|
||||
static Participants get(ClusterMetadata metadata, TableMetadata table, Token token, ConsistencyLevel consistencyForConsensus)
|
||||
public static Participants get(ClusterMetadata metadata, TableMetadata table, Token token, ConsistencyLevel consistencyForConsensus)
|
||||
{
|
||||
return get(metadata, table, token, consistencyForConsensus, FailureDetector.isReplicaAlive);
|
||||
}
|
||||
|
||||
static Participants get(ClusterMetadata metadata, TableMetadata table, Token token, ConsistencyLevel consistencyForConsensus, Predicate<Replica> isReplicaAlive)
|
||||
{
|
||||
KeyspaceMetadata keyspaceMetadata = metadata.schema.getKeyspaceMetadata(table.keyspace);
|
||||
// MetaStrategy distributes the entire keyspace to all replicas. In addition, its tables (currently only
|
||||
// the dist log table) don't use the globally configured partitioner. For these reasons we don't lookup the
|
||||
// replicas using the supplied token as this can actually be of the incorrect type (for example when
|
||||
// performing Paxos repair).
|
||||
final Token actualToken = table.partitioner == MetaStrategy.partitioner ? MetaStrategy.entireRange.right : token;
|
||||
ReplicaLayout.ForTokenWrite all = forTokenWriteLiveAndDown(keyspaceMetadata, actualToken);
|
||||
ReplicaLayout.ForTokenWrite electorate = consistencyForConsensus.isDatacenterLocal()
|
||||
? all.filter(InOurDc.replicas()) : all;
|
||||
|
||||
EndpointsForToken live = all.all().filter(isReplicaAlive);
|
||||
return new Participants(metadata.epoch, Keyspace.open(table.keyspace), consistencyForConsensus, all, electorate, live,
|
||||
(cm) -> get(cm, table, actualToken, consistencyForConsensus));
|
||||
AbstractReplicationStrategy strategy = metadata.schema.getKeyspaceMetadata(table.keyspace).replicationStrategy;
|
||||
return strategy.paxosParticipants(metadata, table, token, consistencyForConsensus, isReplicaAlive);
|
||||
}
|
||||
|
||||
static Participants get(TableMetadata table, Token token, ConsistencyLevel consistencyForConsensus)
|
||||
|
|
@ -586,6 +576,28 @@ public class Paxos
|
|||
{
|
||||
return keyspace.getMetadata().params.replication.isMeta();
|
||||
}
|
||||
|
||||
/**
|
||||
* Hook called after electorate messages are sent during a tracked prepare phase.
|
||||
* Base implementation is a no-op. SRS overrides this to fire satellite summary
|
||||
* read requests in parallel with electorate prepare messages.
|
||||
*/
|
||||
public void onPrepareStarted(TrackedRead.Id readId, int dataNodeId, int[] summaryHostIds, ReadCommand readCommand)
|
||||
{
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns additional summary host IDs to include in tracked read reconciliation.
|
||||
* These are nodes that should participate in the ReadReconciliations protocol
|
||||
* but are not part of the paxos electorate (e.g., satellite DC endpoints for SRS).
|
||||
* Base implementation returns an empty array.
|
||||
*/
|
||||
public int[] additionalSummaryHostIds(ClusterMetadata metadata)
|
||||
{
|
||||
return EMPTY_HOST_IDS;
|
||||
}
|
||||
|
||||
private static final int[] EMPTY_HOST_IDS = new int[0];
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -1139,7 +1151,6 @@ public class Paxos
|
|||
// round's proposal (if any).
|
||||
PaxosPrepare.Success success = prepare.success();
|
||||
|
||||
Supplier<Participants> plan = () -> success.participants;
|
||||
List<Message<IReadResponse>> responses = success.responses;
|
||||
|
||||
// There should be only a single response from the coordinator that was selected to do the tracked read
|
||||
|
|
@ -1161,7 +1172,7 @@ public class Paxos
|
|||
}
|
||||
else
|
||||
{
|
||||
DataResolver<EndpointsForToken, Participants> resolver = new DataResolver<>(ReadCoordinator.DEFAULT, query, plan, NoopReadRepair.instance, requestTime);
|
||||
DataResolver<EndpointsForToken, Participants> resolver = new DataResolver<>(ReadCoordinator.DEFAULT, query, success::participants, NoopReadRepair.instance, requestTime);
|
||||
|
||||
for (int i = 0 ; i < responses.size() ; ++i)
|
||||
{
|
||||
|
|
@ -1221,6 +1232,13 @@ public class Paxos
|
|||
// replicas using the supplied token as this can actually be of the incorrect type (for example when
|
||||
// performing Paxos repair).
|
||||
Token token = table.partitioner == MetaStrategy.partitioner ? MetaStrategy.entireRange.right : key.getToken();
|
||||
|
||||
if (keyspace.getReplicationStrategy().shouldRejectPaxos(token))
|
||||
{
|
||||
PaxosMetrics.rejectedOperations.mark();
|
||||
return false;
|
||||
}
|
||||
|
||||
return (includesRead ? EndpointsForToken.natural(keyspace, token).get()
|
||||
: ReplicaLayout.forTokenWriteLiveAndDown(keyspace, token).all()
|
||||
).contains(getBroadcastAddressAndPort());
|
||||
|
|
|
|||
|
|
@ -19,10 +19,15 @@
|
|||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicLongFieldUpdater;
|
||||
import java.util.concurrent.atomic.AtomicReferenceFieldUpdater;
|
||||
import java.util.function.BiFunction;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
import javax.annotation.Nullable;
|
||||
|
||||
import com.google.common.annotations.VisibleForTesting;
|
||||
import com.google.common.base.Preconditions;
|
||||
|
||||
import org.agrona.collections.IntHashSet;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
|
@ -31,9 +36,11 @@ import org.apache.cassandra.concurrent.ExecutorPlus;
|
|||
import org.apache.cassandra.config.CassandraRelevantProperties;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.locator.AbstractReplicationStrategy;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InOurDc;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
|
|
@ -52,6 +59,7 @@ import org.apache.cassandra.tcm.ClusterMetadata;
|
|||
import org.apache.cassandra.tracing.Tracing;
|
||||
import org.apache.cassandra.utils.FBUtilities;
|
||||
import org.apache.cassandra.utils.concurrent.ConditionAsConsumer;
|
||||
import org.apache.cassandra.utils.concurrent.Future;
|
||||
|
||||
import static com.google.common.base.Preconditions.checkState;
|
||||
import static java.util.Collections.emptyMap;
|
||||
|
|
@ -116,6 +124,150 @@ public class PaxosCommit<OnDone extends Consumer<? super PaxosCommit.Status>> ex
|
|||
@Nullable
|
||||
final IntHashSet remoteReplicas;
|
||||
|
||||
/**
|
||||
* Handles composition of paxos and additional commit mutations for replication strategies that
|
||||
* need to send additional mutations (SRS).
|
||||
*
|
||||
* @param <OnDone>
|
||||
*/
|
||||
static class AugmentedCommit<OnDone extends Consumer<? super PaxosCommit.Status>>
|
||||
{
|
||||
private static final AtomicReferenceFieldUpdater<AugmentedCommit, State> stateUpdater = AtomicReferenceFieldUpdater.newUpdater(AugmentedCommit.class, AugmentedCommit.State.class, "state");
|
||||
|
||||
static abstract class State
|
||||
{
|
||||
abstract State onPaxosComplete(Status status);
|
||||
abstract State onMutationComplete(Status status);
|
||||
boolean isComplete()
|
||||
{
|
||||
return false;
|
||||
}
|
||||
Complete asComplete()
|
||||
{
|
||||
throw new IllegalStateException();
|
||||
}
|
||||
}
|
||||
|
||||
static class Pending extends State
|
||||
{
|
||||
final Status commit;
|
||||
final Status addl;
|
||||
|
||||
public Pending(Status commit, Status addl)
|
||||
{
|
||||
this.commit = commit;
|
||||
this.addl = addl;
|
||||
}
|
||||
|
||||
public Pending()
|
||||
{
|
||||
this(null, null);
|
||||
}
|
||||
|
||||
private static State next(Status commit, Status addl)
|
||||
{
|
||||
if (commit != null && !commit.isSuccess())
|
||||
return new Complete(commit);
|
||||
|
||||
if (addl != null && !addl.isSuccess())
|
||||
return new Complete(addl);
|
||||
|
||||
if (commit == null || addl == null)
|
||||
return new Pending(commit, addl);
|
||||
|
||||
return new Complete(commit);
|
||||
}
|
||||
|
||||
@Override
|
||||
State onPaxosComplete(Status status)
|
||||
{
|
||||
Preconditions.checkState(commit == null);
|
||||
return next(status, addl);
|
||||
}
|
||||
|
||||
@Override
|
||||
State onMutationComplete(Status status)
|
||||
{
|
||||
Preconditions.checkState(addl == null);
|
||||
return next(commit, status);
|
||||
}
|
||||
}
|
||||
|
||||
static class Complete extends State
|
||||
{
|
||||
final Status result;
|
||||
|
||||
public Complete(Status result)
|
||||
{
|
||||
this.result = result;
|
||||
}
|
||||
|
||||
@Override
|
||||
State onPaxosComplete(Status status)
|
||||
{
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
State onMutationComplete(Status status)
|
||||
{
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
boolean isComplete()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
Complete asComplete()
|
||||
{
|
||||
return this;
|
||||
}
|
||||
}
|
||||
|
||||
volatile State state = new Pending();
|
||||
|
||||
final OnDone onDone;
|
||||
|
||||
public AugmentedCommit(OnDone onDone)
|
||||
{
|
||||
this.onDone = onDone;
|
||||
}
|
||||
|
||||
private void onCompletion(BiFunction<State, Status, State> update, Status status)
|
||||
{
|
||||
for (;;)
|
||||
{
|
||||
State current = state;
|
||||
if (current.isComplete())
|
||||
return;
|
||||
|
||||
State next = update.apply(current, status);
|
||||
if (stateUpdater.compareAndSet(this, current, next))
|
||||
{
|
||||
if (next.isComplete())
|
||||
onDone.accept(next.asComplete().result);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void onPaxosComplete(Status status)
|
||||
{
|
||||
onCompletion(State::onPaxosComplete, status);
|
||||
}
|
||||
|
||||
void onMutationComplete(Status status)
|
||||
{
|
||||
onCompletion(State::onMutationComplete, status);
|
||||
}
|
||||
}
|
||||
|
||||
@Nullable
|
||||
volatile AugmentedCommit<OnDone> augmentedCommit = null;
|
||||
|
||||
/**
|
||||
* packs two 32-bit integers;
|
||||
* bit 00-31: accepts
|
||||
|
|
@ -272,6 +424,24 @@ public class PaxosCommit<OnDone extends Consumer<? super PaxosCommit.Status>> ex
|
|||
consistencyForConsensus, consistencyForCommit, allowHints, onDone);
|
||||
}
|
||||
|
||||
void setAugmentedCommitFuture(Future<Void> future)
|
||||
{
|
||||
if (future != null)
|
||||
{
|
||||
augmentedCommit = new AugmentedCommit<>(onDone);
|
||||
future.addCallback((result, failure) -> {
|
||||
if (failure != null)
|
||||
{
|
||||
augmentedCommit.onMutationComplete(new Status(new Paxos.MaybeFailure(true, replicas.size(), required, accepts(responses), emptyMap())));
|
||||
}
|
||||
else
|
||||
{
|
||||
augmentedCommit.onMutationComplete(success);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Send commit messages to peers (or self)
|
||||
*/
|
||||
|
|
@ -291,6 +461,19 @@ public class PaxosCommit<OnDone extends Consumer<? super PaxosCommit.Status>> ex
|
|||
boolean localExecutedSynchronously = false;
|
||||
InetAddressAndPort localEndpoint = FBUtilities.getBroadcastAddressAndPort();
|
||||
|
||||
// Set up additional commit work from the replication strategy (e.g., satellite writes for SRS).
|
||||
// This needs to happen before executeOnSelf() can trigger onPaxosDecision(), so that
|
||||
// additionalCommitFuture is set before it's read. The base strategy returns an
|
||||
// already-completed future, so this is a no-op for non-SRS keyspaces.
|
||||
// For SRS, satellite messages are sent here (in parallel with local execution below).
|
||||
// MutationTrackingService.retryFailedWrite for down satellite endpoints schedules async retries,
|
||||
// which will find the mutation in the journal after executeOnSelf() completes below.
|
||||
if (isTrackedKeyspace)
|
||||
{
|
||||
AbstractReplicationStrategy strategy = Keyspace.open(commit.metadata().keyspace).getReplicationStrategy();
|
||||
setAugmentedCommitFuture(strategy.sendPaxosCommitMutations(commit, isUrgent));
|
||||
}
|
||||
|
||||
if (isTrackedKeyspace)
|
||||
{
|
||||
// For tracked keyspaces, we MUST execute locally synchronously, regardless of USE_SELF_EXECUTION setting.
|
||||
|
|
@ -321,14 +504,25 @@ public class PaxosCommit<OnDone extends Consumer<? super PaxosCommit.Status>> ex
|
|||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (localIsReplica)
|
||||
{
|
||||
executeOnSelf();
|
||||
localExecutedSynchronously = true;
|
||||
}
|
||||
|
||||
// Set up additional commit work from the replication strategy (e.g., satellite writes for SRS).
|
||||
// This needs to happen before executeOnSelf() can trigger onPaxosDecision(), so that
|
||||
// additionalCommitFuture is set before it's read. The base strategy returns an
|
||||
// already-completed future, so this is a no-op for non-SRS keyspaces.
|
||||
// For SRS, satellite messages are sent here (in parallel with local execution below).
|
||||
// MutationTrackingService.retryFailedWrite for down satellite endpoints schedules async retries,
|
||||
// which will find the mutation in the journal after executeOnSelf() completes below.
|
||||
AbstractReplicationStrategy strategy = Keyspace.open(commit.metadata().keyspace).getReplicationStrategy();
|
||||
setAugmentedCommitFuture(strategy.sendPaxosCommitMutations(commit, isUrgent));
|
||||
}
|
||||
|
||||
// Now send to remote replicas (and record local execution for non-tracked keyspaces)
|
||||
// Now send to remote replicas in the electorate (and record local execution for non-tracked keyspaces)
|
||||
boolean executeOnSelf = false;
|
||||
for (int i = 0, mi = allLive.size(); i < mi ; ++i)
|
||||
{
|
||||
|
|
@ -470,21 +664,39 @@ public class PaxosCommit<OnDone extends Consumer<? super PaxosCommit.Status>> ex
|
|||
response(response != null, from);
|
||||
}
|
||||
|
||||
@VisibleForTesting
|
||||
protected boolean isFromLocalDc(InetAddressAndPort endpoint)
|
||||
{
|
||||
return InOurDc.endpoints().test(endpoint);
|
||||
}
|
||||
|
||||
/**
|
||||
* Record a failure or success response if {@code from} contributes to our consistency.
|
||||
* If we have reached a final outcome of the commit, run {@code onDone}.
|
||||
*/
|
||||
private void response(boolean success, InetAddressAndPort from)
|
||||
{
|
||||
if (consistencyForCommit.isDatacenterLocal() && !InOurDc.endpoints().test(from))
|
||||
if (consistencyForCommit.isDatacenterLocal() && !isFromLocalDc(from))
|
||||
return;
|
||||
|
||||
long responses = responsesUpdater.addAndGet(this, success ? 0x1L : 0x100000000L);
|
||||
// next two clauses mutually exclusive to ensure we only invoke onDone once, when either failed or succeeded
|
||||
if (accepts(responses) == required) // if we have received _precisely_ the required accepts, we have succeeded
|
||||
onDone.accept(status());
|
||||
onPaxosDecision();
|
||||
else if (replicas.size() - failures(responses) == required - 1) // if we are _unable_ to receive the required accepts, we have failed
|
||||
onPaxosDecision();
|
||||
}
|
||||
|
||||
/**
|
||||
* Called exactly once when the paxos consensus decision (success or failure) is made.
|
||||
* If an additional commit future exists, onDone is deferred until it also completes.
|
||||
*/
|
||||
private void onPaxosDecision()
|
||||
{
|
||||
if (augmentedCommit == null)
|
||||
onDone.accept(status());
|
||||
else
|
||||
augmentedCommit.onPaxosComplete(status());
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
|
|||
|
|
@ -31,6 +31,8 @@ import java.util.stream.Collectors;
|
|||
import javax.annotation.Nonnull;
|
||||
import javax.annotation.Nullable;
|
||||
|
||||
import com.google.common.annotations.VisibleForTesting;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
|
|
@ -41,6 +43,7 @@ import org.apache.cassandra.db.DecoratedKey;
|
|||
import org.apache.cassandra.db.EmbeddableSinglePartitionReadCommand;
|
||||
import org.apache.cassandra.db.IReadResponse;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.db.ReadExecutionController;
|
||||
import org.apache.cassandra.db.ReadResponse;
|
||||
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||
|
|
@ -191,6 +194,7 @@ public class PaxosPrepare extends PaxosRequestCallback<PaxosPrepare.Response> im
|
|||
FoundIncompleteAccepted incompleteAccepted() { return (FoundIncompleteAccepted) this; }
|
||||
FoundIncompleteCommitted incompleteCommitted() { return (FoundIncompleteCommitted) this; }
|
||||
Paxos.MaybeFailure maybeFailure() { return ((MaybeFailure) this).info; }
|
||||
Participants participants() { return participants; }
|
||||
}
|
||||
|
||||
static class Success extends WithRequestedBallot
|
||||
|
|
@ -371,6 +375,7 @@ public class PaxosPrepare extends PaxosRequestCallback<PaxosPrepare.Response> im
|
|||
this.withLatest = new ArrayList<>(participants.sizeOfConsensusQuorum);
|
||||
this.latestAccepted = this.latestCommitted = Committed.none(request.partitionKey, request.table);
|
||||
this.onDone = onDone;
|
||||
this.haveTrackedDataResponseIfNeeded = request.read == null || !isTracked();
|
||||
}
|
||||
|
||||
private boolean hasInProgressProposal()
|
||||
|
|
@ -438,8 +443,6 @@ public class PaxosPrepare extends PaxosRequestCallback<PaxosPrepare.Response> im
|
|||
|
||||
private static <R extends AbstractRequest<R>> void startTracked(PaxosPrepare prepare, Participants participants, Message<R> send, BiFunction<R, RequestTime, Future<Response>> selfHandler)
|
||||
{
|
||||
if (prepare.request.read == null)
|
||||
prepare.haveTrackedDataResponseIfNeeded = true;
|
||||
Message<R> selfMessage = null;
|
||||
Message<R> summaryMessage = null;
|
||||
Id readId = Id.nextId();
|
||||
|
|
@ -471,6 +474,16 @@ public class PaxosPrepare extends PaxosRequestCallback<PaxosPrepare.Response> im
|
|||
summaryHostIds[summaryIndex++] = metadata.directory.peerId(replica.endpoint()).id();
|
||||
}
|
||||
|
||||
// Merge additional summary host IDs (for SRS)
|
||||
int[] additionalIds = participants.additionalSummaryHostIds(metadata);
|
||||
if (additionalIds.length > 0 || summaryIndex < summaryHostIds.length)
|
||||
{
|
||||
int[] merged = new int[summaryIndex + additionalIds.length];
|
||||
System.arraycopy(summaryHostIds, 0, merged, 0, summaryIndex);
|
||||
System.arraycopy(additionalIds, 0, merged, summaryIndex, additionalIds.length);
|
||||
summaryHostIds = merged;
|
||||
}
|
||||
|
||||
for (int i = 0, size = participants.sizeOfPoll() ; i < size ; ++i)
|
||||
{
|
||||
Replica replica = participants.voterReplica(i);
|
||||
|
|
@ -500,11 +513,12 @@ public class PaxosPrepare extends PaxosRequestCallback<PaxosPrepare.Response> im
|
|||
Message<R> selfMessageFinal = selfMessage;
|
||||
send.verb().stage.execute(() -> prepare.executeOnSelfAsync(selfMessageFinal.payload, new RequestTime(selfMessageFinal.createdAtNanos()), selfHandler));
|
||||
}
|
||||
|
||||
participants.onPrepareStarted(readId, dataNodeId, summaryHostIds, (ReadCommand) prepare.request.read);
|
||||
}
|
||||
|
||||
private static <R extends AbstractRequest<R>> void startUntracked(PaxosPrepare prepare, Participants participants, Message<R> send, BiFunction<R, RequestTime, Future<Response>> selfHandler)
|
||||
{
|
||||
prepare.haveTrackedDataResponseIfNeeded = true;
|
||||
Message<R> selfMessage = null;
|
||||
|
||||
for (int i = 0, size = participants.sizeOfPoll() ; i < size ; ++i)
|
||||
|
|
@ -1119,7 +1133,8 @@ public class PaxosPrepare extends PaxosRequestCallback<PaxosPrepare.Response> im
|
|||
super(ballot, electorate, read, isWrite, isForRecovery);
|
||||
}
|
||||
|
||||
private Request(Ballot ballot, Electorate electorate, DecoratedKey partitionKey, TableMetadata table, boolean isWrite, boolean isForRecovery)
|
||||
@VisibleForTesting
|
||||
Request(Ballot ballot, Electorate electorate, DecoratedKey partitionKey, TableMetadata table, boolean isWrite, boolean isForRecovery)
|
||||
{
|
||||
super(ballot, electorate, partitionKey, table, isWrite, isForRecovery);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -167,6 +167,12 @@ public class PaxosPropose<OnDone extends Consumer<? super PaxosPropose.Status>>
|
|||
this.onDone = onDone;
|
||||
}
|
||||
|
||||
@VisibleForTesting
|
||||
PaxosPropose(Proposal proposal, int participants, int required, OnDone onDone)
|
||||
{
|
||||
this(proposal, participants, required, false, onDone);
|
||||
}
|
||||
|
||||
/**
|
||||
* Submit the proposal for commit with all replicas, and return an object that can be waited on synchronously for the result,
|
||||
* or for the present status if the time elapses without a final result being reached.
|
||||
|
|
|
|||
|
|
@ -0,0 +1,98 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.function.Function;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaLayout;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.net.Verb;
|
||||
import org.apache.cassandra.service.reads.tracked.TrackedRead;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
|
||||
/**
|
||||
* Paxos participants for SatelliteReplicationStrategy.
|
||||
*
|
||||
* Paxos consensus operates entirely within the primary DC. Satellites and secondary DCs do not participate in propose
|
||||
* / accept. However paxos reads and writes do need QoQ consistency to support quick failover. This class adds the
|
||||
* additional summary nodes to the prepare/read stage and instructs them to send summaries to the read coordinator.
|
||||
* Additional commit mutations are handled by the replication strategy itself.
|
||||
*/
|
||||
public class SatellitePaxosParticipants extends Paxos.Participants
|
||||
{
|
||||
private static final Logger logger = LoggerFactory.getLogger(SatellitePaxosParticipants.class);
|
||||
|
||||
/** Endpoints in satellite/secondary DCs that receive reads during prepare and writes during commit */
|
||||
private final EndpointsForToken additionalSummaryEndpoints;
|
||||
|
||||
public SatellitePaxosParticipants(Epoch epoch,
|
||||
Keyspace keyspace,
|
||||
ConsistencyLevel consistencyForConsensus,
|
||||
ReplicaLayout.ForTokenWrite all,
|
||||
ReplicaLayout.ForTokenWrite electorate,
|
||||
EndpointsForToken live,
|
||||
Function<ClusterMetadata, Paxos.Participants> recompute,
|
||||
EndpointsForToken additionalSummaryEndpoints)
|
||||
{
|
||||
super(epoch, keyspace, consistencyForConsensus, all, electorate, live, recompute);
|
||||
this.additionalSummaryEndpoints = additionalSummaryEndpoints;
|
||||
}
|
||||
|
||||
public EndpointsForToken getAdditionalSummaryEndpoints()
|
||||
{
|
||||
return additionalSummaryEndpoints;
|
||||
}
|
||||
|
||||
@Override
|
||||
public int[] additionalSummaryHostIds(ClusterMetadata metadata)
|
||||
{
|
||||
if (additionalSummaryEndpoints.isEmpty())
|
||||
return super.additionalSummaryHostIds(metadata);
|
||||
|
||||
int[] ids = new int[additionalSummaryEndpoints.size()];
|
||||
for (int i = 0; i < additionalSummaryEndpoints.size(); i++)
|
||||
ids[i] = metadata.directory.peerId(additionalSummaryEndpoints.endpoint(i)).id();
|
||||
return ids;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onPrepareStarted(TrackedRead.Id readId, int dataNodeId, int[] summaryHostIds, ReadCommand readCommand)
|
||||
{
|
||||
if (additionalSummaryEndpoints.isEmpty() || readCommand == null)
|
||||
return;
|
||||
|
||||
// Send standalone TRACKED_SUMMARY_REQ to each additional satellite endpoint.
|
||||
TrackedRead.SummaryRequest summaryRequest = new TrackedRead.SummaryRequest(readId, readCommand, dataNodeId, summaryHostIds);
|
||||
Message<TrackedRead.SummaryRequest> summaryMessage = Message.out(Verb.TRACKED_SUMMARY_REQ, summaryRequest);
|
||||
for (Replica replica : additionalSummaryEndpoints)
|
||||
{
|
||||
logger.trace("Sending satellite summary request for {} to {}", readId, replica.endpoint());
|
||||
MessagingService.instance().send(summaryMessage, replica.endpoint());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -34,12 +34,12 @@ import org.apache.cassandra.db.transform.DuplicateRowChecker;
|
|||
import org.apache.cassandra.exceptions.ReadFailureException;
|
||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.exceptions.UnavailableException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaCollection;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.service.StorageProxy.LocalReadRunnable;
|
||||
|
|
@ -68,7 +68,7 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
|
||||
protected final ReadCoordinator coordinator;
|
||||
protected final ReadCommand command;
|
||||
private final ReplicaPlan.SharedForTokenRead replicaPlan;
|
||||
private final CoordinationPlan.ForTokenRead plan;
|
||||
protected final ReadRepair<EndpointsForToken, ReplicaPlan.ForTokenRead> readRepair;
|
||||
protected final DigestResolver<EndpointsForToken, ReplicaPlan.ForTokenRead> digestResolver;
|
||||
protected final ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler;
|
||||
|
|
@ -79,16 +79,16 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
private final int initialDataRequestCount;
|
||||
protected volatile PartitionIterator result = null;
|
||||
|
||||
AbstractReadExecutor(ReadCoordinator coordinator, ColumnFamilyStore cfs, ReadCommand command, ReplicaPlan.ForTokenRead replicaPlan, int initialDataRequestCount, Dispatcher.RequestTime requestTime)
|
||||
AbstractReadExecutor(ReadCoordinator coordinator, ColumnFamilyStore cfs, ReadCommand command, CoordinationPlan.ForTokenRead plan, int initialDataRequestCount, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this.coordinator = coordinator;
|
||||
this.command = command;
|
||||
this.replicaPlan = ReplicaPlan.shared(replicaPlan);
|
||||
this.plan = plan;
|
||||
this.initialDataRequestCount = initialDataRequestCount;
|
||||
// the ReadRepair and DigestResolver both need to see our updated
|
||||
this.readRepair = ReadRepair.create(coordinator, command, this.replicaPlan, requestTime);
|
||||
this.digestResolver = new DigestResolver<>(coordinator, command, this.replicaPlan, requestTime);
|
||||
this.handler = new ReadCallback<>(digestResolver, command, this.replicaPlan, requestTime);
|
||||
this.readRepair = ReadRepair.create(coordinator, command, plan, requestTime);
|
||||
this.digestResolver = new DigestResolver<>(coordinator, command, plan, requestTime);
|
||||
this.handler = new ReadCallback<>(digestResolver, command, plan, requestTime);
|
||||
this.cfs = cfs;
|
||||
this.traceState = Tracing.instance.get();
|
||||
this.requestTime = requestTime;
|
||||
|
|
@ -99,7 +99,7 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
// TODO: we need this when talking with pre-3.0 nodes. So if we preserve the digest format moving forward, we can get rid of this once
|
||||
// we stop being compatible with pre-3.0 nodes.
|
||||
int digestVersion = MessagingService.current_version;
|
||||
for (Replica replica : replicaPlan.contacts())
|
||||
for (Replica replica : plan.replicas().contacts())
|
||||
digestVersion = Math.min(digestVersion, MessagingService.instance().versions.get(replica.endpoint()));
|
||||
command.setDigestVersion(digestVersion);
|
||||
}
|
||||
|
|
@ -195,32 +195,32 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
ColumnFamilyStore cfs = keyspace.getColumnFamilyStore(command.metadata().id);
|
||||
SpeculativeRetryPolicy retry = cfs.metadata().params.speculativeRetry;
|
||||
|
||||
ReplicaPlan.ForTokenRead replicaPlan = ReplicaPlans.forRead(metadata,
|
||||
keyspace,
|
||||
command.metadata().id,
|
||||
command.partitionKey().getToken(),
|
||||
command.indexQueryPlan(),
|
||||
consistencyLevel,
|
||||
retry,
|
||||
coordinator);
|
||||
CoordinationPlan.ForTokenRead plan = CoordinationPlan.forTokenRead(metadata,
|
||||
keyspace,
|
||||
command.metadata().id,
|
||||
command.partitionKey().getToken(),
|
||||
command.indexQueryPlan(),
|
||||
consistencyLevel,
|
||||
retry,
|
||||
coordinator);
|
||||
|
||||
// Speculative retry is disabled *OR*
|
||||
// 11980: Disable speculative retry if using EACH_QUORUM in order to prevent miscounting DC responses
|
||||
if (retry.equals(NeverSpeculativeRetryPolicy.INSTANCE) || consistencyLevel == ConsistencyLevel.EACH_QUORUM)
|
||||
return new NeverSpeculatingReadExecutor(coordinator, cfs, command, replicaPlan, requestTime, false);
|
||||
return new NeverSpeculatingReadExecutor(coordinator, cfs, command, plan, requestTime, false);
|
||||
|
||||
if (retry.equals(AlwaysSpeculativeRetryPolicy.INSTANCE))
|
||||
return new AlwaysSpeculatingReadExecutor(coordinator, cfs, command, replicaPlan, requestTime);
|
||||
return new AlwaysSpeculatingReadExecutor(coordinator, cfs, command, plan, requestTime);
|
||||
|
||||
// There are simply no extra replicas to speculate.
|
||||
// Handle this separately so it can record failed attempts to speculate due to lack of replicas
|
||||
if (replicaPlan.contacts().size() == replicaPlan.readCandidates().size())
|
||||
if (plan.replicas().contacts().size() == plan.replicas().readCandidates().size())
|
||||
{
|
||||
boolean recordFailedSpeculation = consistencyLevel != ConsistencyLevel.ALL;
|
||||
return new NeverSpeculatingReadExecutor(coordinator, cfs, command, replicaPlan, requestTime, recordFailedSpeculation);
|
||||
return new NeverSpeculatingReadExecutor(coordinator, cfs, command, plan, requestTime, recordFailedSpeculation);
|
||||
}
|
||||
else // PERCENTILE or CUSTOM.
|
||||
return new SpeculatingReadExecutor(coordinator, cfs, command, replicaPlan, requestTime);
|
||||
return new SpeculatingReadExecutor(coordinator, cfs, command, plan, requestTime);
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -256,7 +256,7 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
@Override
|
||||
public ReplicaPlan.ForTokenRead replicaPlan()
|
||||
{
|
||||
return replicaPlan.get();
|
||||
return plan.replicas();
|
||||
}
|
||||
|
||||
void onReadTimeout() {}
|
||||
|
|
@ -273,11 +273,11 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
public NeverSpeculatingReadExecutor(ReadCoordinator coordinator,
|
||||
ColumnFamilyStore cfs,
|
||||
ReadCommand command,
|
||||
ReplicaPlan.ForTokenRead replicaPlan,
|
||||
CoordinationPlan.ForTokenRead plan,
|
||||
Dispatcher.RequestTime requestTime,
|
||||
boolean logFailedSpeculation)
|
||||
{
|
||||
super(coordinator, cfs, command, replicaPlan, 1, requestTime);
|
||||
super(coordinator, cfs, command, plan, 1, requestTime);
|
||||
this.logFailedSpeculation = logFailedSpeculation;
|
||||
}
|
||||
|
||||
|
|
@ -297,13 +297,13 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
public SpeculatingReadExecutor(ReadCoordinator coordinator,
|
||||
ColumnFamilyStore cfs,
|
||||
ReadCommand command,
|
||||
ReplicaPlan.ForTokenRead replicaPlan,
|
||||
CoordinationPlan.ForTokenRead plan,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
// We're hitting additional targets for read repair (??). Since our "extra" replica is the least-
|
||||
// preferred by the snitch, we do an extra data read to start with against a replica more
|
||||
// likely to respond; better to let RR fail than the entire query.
|
||||
super(coordinator, cfs, command, replicaPlan, replicaPlan.readQuorum() < replicaPlan.contacts().size() ? 2 : 1, requestTime);
|
||||
super(coordinator, cfs, command, plan, plan.replicas().readQuorum() < plan.replicas().contacts().size() ? 2 : 1, requestTime);
|
||||
}
|
||||
|
||||
public void maybeTryAdditionalReplicas()
|
||||
|
|
@ -342,7 +342,7 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
// we must update the plan to include this new node, else when we come to read-repair, we may not include this
|
||||
// speculated response in the data requests we make again, and we will not be able to 'speculate' an extra repair read,
|
||||
// nor would we be able to speculate a new 'write' if the repair writes are insufficient
|
||||
super.replicaPlan.addToContacts(extraReplica);
|
||||
super.plan.addToContacts(extraReplica);
|
||||
|
||||
if (traceState != null)
|
||||
traceState.trace("speculating read retry on {}", extraReplica);
|
||||
|
|
@ -367,12 +367,12 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
public AlwaysSpeculatingReadExecutor(ReadCoordinator coordinator,
|
||||
ColumnFamilyStore cfs,
|
||||
ReadCommand command,
|
||||
ReplicaPlan.ForTokenRead replicaPlan,
|
||||
CoordinationPlan.ForTokenRead plan,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
// presumably, we speculate an extra data request here in case it is our data request that fails to respond,
|
||||
// and there are no more nodes to consult
|
||||
super(coordinator, cfs, command, replicaPlan, replicaPlan.contacts().size() > 1 ? 2 : 1, requestTime);
|
||||
super(coordinator, cfs, command, plan, plan.replicas().contacts().size() > 1 ? 2 : 1, requestTime);
|
||||
}
|
||||
|
||||
public void maybeTryAdditionalReplicas()
|
||||
|
|
@ -397,7 +397,7 @@ public abstract class AbstractReadExecutor implements ReadExecutor
|
|||
public void setResult(PartitionIterator result)
|
||||
{
|
||||
Preconditions.checkState(this.result == null, "Result can only be set once");
|
||||
this.result = DuplicateRowChecker.duringRead(result, this.replicaPlan.get().readCandidates().endpointList());
|
||||
this.result = DuplicateRowChecker.duringRead(result, this.plan.replicas().readCandidates().endpointList());
|
||||
}
|
||||
|
||||
public void awaitResponses() throws ReadTimeoutException
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ package org.apache.cassandra.service.reads;
|
|||
import java.nio.ByteBuffer;
|
||||
import java.util.Collection;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import com.google.common.annotations.VisibleForTesting;
|
||||
import com.google.common.base.Preconditions;
|
||||
|
|
@ -45,7 +46,7 @@ public class DigestResolver<E extends Endpoints<E>, P extends ReplicaPlan.ForRea
|
|||
{
|
||||
private volatile Message<ReadResponse> dataResponse;
|
||||
|
||||
public DigestResolver(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public DigestResolver(ReadCoordinator coordinator, ReadCommand command, Supplier<P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(coordinator, command, replicaPlan, requestTime);
|
||||
Preconditions.checkArgument(command instanceof SinglePartitionReadCommand,
|
||||
|
|
|
|||
|
|
@ -21,7 +21,6 @@ import java.util.HashMap;
|
|||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicIntegerFieldUpdater;
|
||||
import java.util.concurrent.atomic.AtomicReferenceFieldUpdater;
|
||||
|
||||
import com.google.common.collect.ImmutableMap;
|
||||
|
|
@ -40,6 +39,7 @@ import org.apache.cassandra.exceptions.ReadTimeoutException;
|
|||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||
import org.apache.cassandra.exceptions.RetryOnDifferentSystemException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -57,7 +57,6 @@ import org.apache.cassandra.utils.concurrent.Condition;
|
|||
import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException;
|
||||
|
||||
import static java.util.concurrent.TimeUnit.MILLISECONDS;
|
||||
import static java.util.concurrent.atomic.AtomicIntegerFieldUpdater.newUpdater;
|
||||
import static org.apache.cassandra.exceptions.RequestFailureReason.COORDINATOR_BEHIND;
|
||||
import static org.apache.cassandra.exceptions.RequestFailureReason.RETRY_ON_DIFFERENT_TRANSACTION_SYSTEM;
|
||||
import static org.apache.cassandra.tracing.Tracing.isTracing;
|
||||
|
|
@ -72,33 +71,30 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
private final Dispatcher.RequestTime requestTime;
|
||||
// this uses a plain reference, but is initialised before handoff to any other threads; the later updates
|
||||
// may not be visible to the threads immediately, but ReplicaPlan only contains final fields, so they will never see an uninitialised object
|
||||
final ReplicaPlan.Shared<E, P> replicaPlan;
|
||||
final CoordinationPlan.ForRead<E, P> plan;
|
||||
private final ReadCommand command;
|
||||
private static final AtomicIntegerFieldUpdater<ReadCallback> failuresUpdater
|
||||
= newUpdater(ReadCallback.class, "failures");
|
||||
private volatile int failures = 0;
|
||||
private final Map<InetAddressAndPort, RequestFailureReason> failureReasonByEndpoint;
|
||||
private volatile WarningContext warningContext;
|
||||
private static final AtomicReferenceFieldUpdater<ReadCallback, WarningContext> warningsUpdater
|
||||
= AtomicReferenceFieldUpdater.newUpdater(ReadCallback.class, WarningContext.class, "warningContext");
|
||||
|
||||
public ReadCallback(ResponseResolver<E, P> resolver, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public ReadCallback(ResponseResolver<E, P> resolver, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this.command = command;
|
||||
this.resolver = resolver;
|
||||
this.requestTime = requestTime;
|
||||
this.replicaPlan = replicaPlan;
|
||||
this.plan = plan;
|
||||
this.failureReasonByEndpoint = new ConcurrentHashMap<>();
|
||||
// we don't support read repair (or rapid read protection) for range scans yet (CASSANDRA-6897)
|
||||
assert !(command instanceof PartitionRangeReadCommand) || replicaPlan().readQuorum() >= replicaPlan().contacts().size();
|
||||
|
||||
if (logger.isTraceEnabled())
|
||||
logger.trace("Blockfor is {}; setting up requests to {}", replicaPlan().readQuorum(), this.replicaPlan);
|
||||
logger.trace("Blockfor is {}; setting up requests to {}", replicaPlan().readQuorum(), this.plan.replicas());
|
||||
}
|
||||
|
||||
protected P replicaPlan()
|
||||
{
|
||||
return replicaPlan.get();
|
||||
return plan.replicas();
|
||||
}
|
||||
|
||||
public boolean await(long commandTimeout, TimeUnit unit)
|
||||
|
|
@ -141,8 +137,8 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
* See {@link DigestResolver#preprocess(Message)}
|
||||
* CASSANDRA-16097
|
||||
*/
|
||||
int received = resolver.responses.size();
|
||||
boolean failed = failures > 0 && (replicaPlan().readQuorum() > received || !resolver.isDataPresent());
|
||||
int received = plan.responses().received();
|
||||
boolean failed = plan.responses().failures() > 0 && (!plan.responses().isSuccessful() || !resolver.isDataPresent());
|
||||
// If all messages came back as a TIMEOUT then signaled=true and failed=true.
|
||||
// Need to distinguish between a timeout and a failure (network, bad data, etc.), so store an extra field.
|
||||
// see CASSANDRA-17828
|
||||
|
|
@ -168,16 +164,16 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
if (isTracing())
|
||||
{
|
||||
String gotData = received > 0 ? (resolver.isDataPresent() ? " (including data)" : " (only digests)") : "";
|
||||
Tracing.trace("{}; received {} of {} responses{}", !timedout ? "Failed" : "Timed out", received, replicaPlan().readQuorum(), gotData);
|
||||
Tracing.trace("{}; received {} of {} responses{}", !timedout ? "Failed" : "Timed out", received, plan.responses().required(), gotData);
|
||||
}
|
||||
else if (logger.isDebugEnabled())
|
||||
{
|
||||
String gotData = received > 0 ? (resolver.isDataPresent() ? " (including data)" : " (only digests)") : "";
|
||||
logger.debug("{}; received {} of {} responses{}", !timedout ? "Failed" : "Timed out", received, replicaPlan().readQuorum(), gotData);
|
||||
logger.debug("{}; received {} of {} responses{}", !timedout ? "Failed" : "Timed out", received, plan.responses().required(), gotData);
|
||||
}
|
||||
|
||||
if (snapshot != null)
|
||||
snapshot.maybeAbort(command, replicaPlan().consistencyLevel(), received, replicaPlan().readQuorum(), resolver.isDataPresent(), failureReasonByEndpoint);
|
||||
snapshot.maybeAbort(command, replicaPlan().consistencyLevel(), received, plan.responses().required(), resolver.isDataPresent(), failureReasonByEndpoint);
|
||||
|
||||
// failures keeps incrementing, and this.failureReasonByEndpoint keeps getting new entries after signaling.
|
||||
// Simpler to reason about what happened by copying this.failureReasonByEndpoint and then inferring
|
||||
|
|
@ -209,8 +205,8 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
|
||||
// Same as for writes, see AbstractWriteResponseHandler
|
||||
throw !timedout
|
||||
? new ReadFailureException(replicaPlan().consistencyLevel(), received, replicaPlan().readQuorum(), resolver.isDataPresent(), failureReasonByEndpoint)
|
||||
: new ReadTimeoutException(replicaPlan().consistencyLevel(), received, replicaPlan().readQuorum(), resolver.isDataPresent());
|
||||
? new ReadFailureException(replicaPlan().consistencyLevel(), received, plan.responses().required(), resolver.isDataPresent(), failureReasonByEndpoint)
|
||||
: new ReadTimeoutException(replicaPlan().consistencyLevel(), received, plan.responses().required(), resolver.isDataPresent());
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
@ -231,6 +227,7 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
}
|
||||
resolver.preprocess(message);
|
||||
replicaPlan().collectSuccess(message.from());
|
||||
plan.responses().onResponse(from);
|
||||
|
||||
/*
|
||||
* Ensure that data is present and the response accumulator has properly published the
|
||||
|
|
@ -238,7 +235,7 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
* the minimum number of required results, but it guarantees at least the minimum will
|
||||
* be accessible when we do signal. (see CASSANDRA-16807)
|
||||
*/
|
||||
if (resolver.isDataPresent() && resolver.responses.size() >= replicaPlan().readQuorum())
|
||||
if (resolver.isDataPresent() && plan.responses().isSuccessful())
|
||||
condition.signalAll();
|
||||
}
|
||||
|
||||
|
|
@ -274,10 +271,11 @@ public class ReadCallback<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<
|
|||
public void onFailure(InetAddressAndPort from, RequestFailure failure)
|
||||
{
|
||||
assertWaitingFor(from);
|
||||
|
||||
failureReasonByEndpoint.put(from, failure.reason);
|
||||
|
||||
if (replicaPlan().readQuorum() + failuresUpdater.incrementAndGet(this) > replicaPlan().contacts().size())
|
||||
failureReasonByEndpoint.put(from, failure.reason);
|
||||
plan.responses().onFailure(from);
|
||||
|
||||
if (plan.responses().isComplete() && !plan.responses().isSuccessful())
|
||||
condition.signalAll();
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -60,11 +60,11 @@ import org.apache.cassandra.db.rows.UnfilteredRowIterators;
|
|||
import org.apache.cassandra.exceptions.OverloadedException;
|
||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.exceptions.UnavailableException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.metrics.TableMetrics;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.schema.ColumnMetadata;
|
||||
|
|
@ -164,13 +164,13 @@ public class ReplicaFilteringProtection<E extends Endpoints<E>>
|
|||
mergeListener = new QueryMergeListener();
|
||||
}
|
||||
|
||||
private UnfilteredPartitionIterator executeReadCommand(ReadCommand cmd, Replica source, ReplicaPlan.Shared<EndpointsForToken, ReplicaPlan.ForTokenRead> replicaPlan)
|
||||
private UnfilteredPartitionIterator executeReadCommand(ReadCommand cmd, Replica source, CoordinationPlan.ForTokenRead plan)
|
||||
{
|
||||
@SuppressWarnings("unchecked")
|
||||
DataResolver<EndpointsForToken, ReplicaPlan.ForTokenRead> resolver =
|
||||
new DataResolver<>(coordinator, cmd, replicaPlan, (NoopReadRepair<EndpointsForToken, ReplicaPlan.ForTokenRead>) NoopReadRepair.instance, requestTime);
|
||||
new DataResolver<>(coordinator, cmd, plan, (NoopReadRepair<EndpointsForToken, ReplicaPlan.ForTokenRead>) NoopReadRepair.instance, requestTime);
|
||||
|
||||
ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler = new ReadCallback<>(resolver, cmd, replicaPlan, requestTime);
|
||||
ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler = new ReadCallback<>(resolver, cmd, plan, requestTime);
|
||||
// TODO No tracked path here yet so assert it doesn't handle transient replication correctly
|
||||
checkState(!source.isTransient());
|
||||
if (source.isSelf() && coordinator.localReadSupported())
|
||||
|
|
@ -636,20 +636,20 @@ public class ReplicaFilteringProtection<E extends Endpoints<E>>
|
|||
key,
|
||||
filter);
|
||||
|
||||
ReplicaPlan.ForTokenRead replicaPlan = ReplicaPlans.forSingleReplicaRead(keyspace, key.getToken(), source);
|
||||
CoordinationPlan.ForTokenRead plan = CoordinationPlan.forSingleReplicaTokenRead(keyspace, key.getToken(), source);
|
||||
|
||||
try
|
||||
{
|
||||
return executeReadCommand(cmd, source, ReplicaPlan.shared(replicaPlan));
|
||||
return executeReadCommand(cmd, source, plan);
|
||||
}
|
||||
catch (ReadTimeoutException e)
|
||||
{
|
||||
int blockFor = consistency.blockFor(replicaPlan.replicationStrategy());
|
||||
int blockFor = consistency.blockFor(plan.replicationStrategy());
|
||||
throw new ReadTimeoutException(consistency, blockFor - 1, blockFor, true);
|
||||
}
|
||||
catch (UnavailableException e)
|
||||
{
|
||||
int blockFor = consistency.blockFor(replicaPlan.replicationStrategy());
|
||||
int blockFor = consistency.blockFor(plan.replicationStrategy());
|
||||
throw UnavailableException.create(consistency, blockFor, blockFor - 1);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -38,10 +38,10 @@ import org.apache.cassandra.db.transform.Transformation;
|
|||
import org.apache.cassandra.dht.AbstractBounds;
|
||||
import org.apache.cassandra.dht.ExcludingBounds;
|
||||
import org.apache.cassandra.dht.Range;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.service.StorageProxy;
|
||||
import org.apache.cassandra.service.reads.repair.NoopReadRepair;
|
||||
import org.apache.cassandra.tracing.Tracing;
|
||||
|
|
@ -93,17 +93,17 @@ public class ShortReadPartitionsProtection extends Transformation<UnfilteredRowI
|
|||
|
||||
lastPartitionKey = partition.partitionKey();
|
||||
|
||||
Keyspace keyspace = Keyspace.open(command.metadata().keyspace);
|
||||
/*
|
||||
* Extend for moreContents() then apply protection to track lastClustering by applyToRow().
|
||||
*
|
||||
* If we don't apply the transformation *after* extending the partition with MoreRows,
|
||||
* applyToRow() method of protection will not be called on the first row of the new extension iterator.
|
||||
*/
|
||||
ReplicaPlan.ForTokenRead replicaPlan = ReplicaPlans.forSingleReplicaRead(Keyspace.open(command.metadata().keyspace), partition.partitionKey().getToken(), source);
|
||||
ReplicaPlan.SharedForTokenRead sharedReplicaPlan = ReplicaPlan.shared(replicaPlan);
|
||||
CoordinationPlan.ForTokenRead plan = CoordinationPlan.forSingleReplicaTokenRead(keyspace, partition.partitionKey().getToken(), source);
|
||||
ShortReadRowsProtection protection = new ShortReadRowsProtection(partition.partitionKey(),
|
||||
command, source,
|
||||
(cmd) -> executeReadCommand(cmd, sharedReplicaPlan),
|
||||
(cmd) -> executeReadCommand(cmd, plan),
|
||||
singleResultCounter,
|
||||
mergedResultCounter);
|
||||
return Transformation.apply(MoreRows.extend(partition, protection), protection);
|
||||
|
|
@ -177,16 +177,17 @@ public class ShortReadPartitionsProtection extends Transformation<UnfilteredRowI
|
|||
: new ExcludingBounds<>(lastPartitionKey, bounds.right);
|
||||
DataRange newDataRange = cmd.dataRange().forSubRange(newBounds);
|
||||
|
||||
ReplicaPlan.ForRangeRead replicaPlan = ReplicaPlans.forSingleReplicaRead(Keyspace.open(command.metadata().keyspace), cmd.dataRange().keyRange(), source, 1);
|
||||
return executeReadCommand(cmd.withUpdatedLimitsAndDataRange(newLimits, newDataRange), ReplicaPlan.shared(replicaPlan));
|
||||
Keyspace keyspace = Keyspace.open(command.metadata().keyspace);
|
||||
CoordinationPlan.ForRangeRead plan = CoordinationPlan.forSingleReplicaRangeRead(keyspace, cmd.dataRange().keyRange(), source, 1);
|
||||
return executeReadCommand(cmd.withUpdatedLimitsAndDataRange(newLimits, newDataRange), plan);
|
||||
}
|
||||
|
||||
private <E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
UnfilteredPartitionIterator executeReadCommand(ReadCommand cmd, ReplicaPlan.Shared<E, P> replicaPlan)
|
||||
UnfilteredPartitionIterator executeReadCommand(ReadCommand cmd, CoordinationPlan.ForRead<E, P> plan)
|
||||
{
|
||||
cmd = coordinator.maybeAllowOutOfRangeReads(cmd, replicaPlan.get().consistencyLevel());
|
||||
DataResolver<E, P> resolver = new DataResolver<>(coordinator, cmd, replicaPlan, (NoopReadRepair<E, P>)NoopReadRepair.instance, requestTime);
|
||||
ReadCallback<E, P> handler = new ReadCallback<>(resolver, cmd, replicaPlan, requestTime);
|
||||
cmd = coordinator.maybeAllowOutOfRangeReads(cmd, plan.consistencyLevel());
|
||||
DataResolver<E, P> resolver = new DataResolver<>(coordinator, cmd, plan, (NoopReadRepair<E, P>)NoopReadRepair.instance, requestTime);
|
||||
ReadCallback<E, P> handler = new ReadCallback<>(resolver, cmd, plan, requestTime);
|
||||
|
||||
if (source.isSelf() && coordinator.localReadSupported())
|
||||
{
|
||||
|
|
|
|||
|
|
@ -34,8 +34,8 @@ import org.apache.cassandra.dht.AbstractBounds;
|
|||
import org.apache.cassandra.dht.Bounds;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.index.Index;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.schema.ReplicationParams;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
|
|
@ -43,7 +43,7 @@ import org.apache.cassandra.tcm.compatibility.TokenRingUtils;
|
|||
import org.apache.cassandra.utils.AbstractIterator;
|
||||
import org.apache.cassandra.utils.Pair;
|
||||
|
||||
class ReplicaPlanIterator extends AbstractIterator<ReplicaPlan.ForRangeRead>
|
||||
class CoordinationPlanIterator extends AbstractIterator<CoordinationPlan.ForRangeRead>
|
||||
{
|
||||
private final Keyspace keyspace;
|
||||
private final ConsistencyLevel consistency;
|
||||
|
|
@ -53,11 +53,11 @@ class ReplicaPlanIterator extends AbstractIterator<ReplicaPlan.ForRangeRead>
|
|||
final Iterator<? extends AbstractBounds<PartitionPosition>> ranges;
|
||||
private final int rangeCount;
|
||||
|
||||
ReplicaPlanIterator(AbstractBounds<PartitionPosition> keyRange,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
ConsistencyLevel consistency)
|
||||
CoordinationPlanIterator(AbstractBounds<PartitionPosition> keyRange,
|
||||
@Nullable Index.QueryPlan indexQueryPlan,
|
||||
Keyspace keyspace,
|
||||
TableId tableId,
|
||||
ConsistencyLevel consistency)
|
||||
{
|
||||
this.indexQueryPlan = indexQueryPlan;
|
||||
this.keyspace = keyspace;
|
||||
|
|
@ -81,12 +81,12 @@ class ReplicaPlanIterator extends AbstractIterator<ReplicaPlan.ForRangeRead>
|
|||
}
|
||||
|
||||
@Override
|
||||
protected ReplicaPlan.ForRangeRead computeNext()
|
||||
protected CoordinationPlan.ForRangeRead computeNext()
|
||||
{
|
||||
if (!ranges.hasNext())
|
||||
return endOfData();
|
||||
|
||||
return ReplicaPlans.forRangeRead(keyspace, tableId, indexQueryPlan, consistency, ranges.next(), 1);
|
||||
return CoordinationPlan.forRangeRead(ClusterMetadata.current(), keyspace, tableId, indexQueryPlan, consistency, ranges.next(), 1);
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -25,20 +25,19 @@ import com.google.common.collect.PeekingIterator;
|
|||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.utils.AbstractIterator;
|
||||
|
||||
class ReplicaPlanMerger extends AbstractIterator<ReplicaPlan.ForRangeRead>
|
||||
class CoordinationPlanMerger extends AbstractIterator<CoordinationPlan.ForRangeRead>
|
||||
{
|
||||
private final Keyspace keyspace;
|
||||
private final ConsistencyLevel consistency;
|
||||
private final TableId tableId;
|
||||
private final PeekingIterator<ReplicaPlan.ForRangeRead> ranges;
|
||||
private final PeekingIterator<CoordinationPlan.ForRangeRead> ranges;
|
||||
|
||||
ReplicaPlanMerger(Iterator<ReplicaPlan.ForRangeRead> iterator, Keyspace keyspace, TableId tableId, ConsistencyLevel consistency)
|
||||
CoordinationPlanMerger(Iterator<CoordinationPlan.ForRangeRead> iterator, Keyspace keyspace, TableId tableId, ConsistencyLevel consistency)
|
||||
{
|
||||
this.keyspace = keyspace;
|
||||
this.tableId = tableId;
|
||||
|
|
@ -47,12 +46,12 @@ class ReplicaPlanMerger extends AbstractIterator<ReplicaPlan.ForRangeRead>
|
|||
}
|
||||
|
||||
@Override
|
||||
protected ReplicaPlan.ForRangeRead computeNext()
|
||||
protected CoordinationPlan.ForRangeRead computeNext()
|
||||
{
|
||||
if (!ranges.hasNext())
|
||||
return endOfData();
|
||||
|
||||
ReplicaPlan.ForRangeRead current = ranges.next();
|
||||
CoordinationPlan.ForRangeRead current = ranges.next();
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
// getRestrictedRange has broken the queried range into per-[vnode] token ranges, but this doesn't take
|
||||
|
|
@ -65,11 +64,11 @@ class ReplicaPlanMerger extends AbstractIterator<ReplicaPlan.ForRangeRead>
|
|||
// Note: it would be slightly more efficient to have CFS.getRangeSlice on the destination nodes unwraps
|
||||
// the range if necessary and deal with it. However, we can't start sending wrapped range without breaking
|
||||
// wire compatibility, so it's likely easier not to bother;
|
||||
if (current.range().right.isMinimum())
|
||||
if (current.replicas().range().right.isMinimum())
|
||||
break;
|
||||
|
||||
ReplicaPlan.ForRangeRead next = ranges.peek();
|
||||
ReplicaPlan.ForRangeRead merged = ReplicaPlans.maybeMerge(metadata, keyspace, tableId, consistency, current, next);
|
||||
CoordinationPlan.ForRangeRead next = ranges.peek();
|
||||
CoordinationPlan.ForRangeRead merged = CoordinationPlan.maybeMergeRangeReads(metadata, keyspace, tableId, consistency, current, next);
|
||||
if (merged == null)
|
||||
break;
|
||||
|
||||
|
|
@ -45,6 +45,7 @@ import org.apache.cassandra.exceptions.ReadFailureException;
|
|||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.exceptions.RetryOnDifferentSystemException;
|
||||
import org.apache.cassandra.exceptions.UnavailableException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -82,7 +83,7 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
|
||||
public static final ClientRangeRequestMetrics rangeMetrics = new ClientRangeRequestMetrics("RangeSlice");
|
||||
|
||||
final CloseableIterator<ReplicaPlan.ForRangeRead> replicaPlans;
|
||||
final CloseableIterator<CoordinationPlan.ForRangeRead> coordinatorPlans;
|
||||
final int totalRangeCount;
|
||||
final PartitionRangeReadCommand command;
|
||||
final boolean enforceStrictLiveness;
|
||||
|
|
@ -101,7 +102,7 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
// when it was not good enough initially.
|
||||
private int liveReturned;
|
||||
|
||||
RangeCommandIterator(CloseableIterator<ReplicaPlan.ForRangeRead> replicaPlans,
|
||||
RangeCommandIterator(CloseableIterator<CoordinationPlan.ForRangeRead> coordinatorPlans,
|
||||
PartitionRangeReadCommand command,
|
||||
ReadCoordinator readCoordinator,
|
||||
int concurrencyFactor,
|
||||
|
|
@ -109,7 +110,7 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
int totalRangeCount,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this.replicaPlans = replicaPlans;
|
||||
this.coordinatorPlans = coordinatorPlans;
|
||||
this.command = command;
|
||||
this.readCoordinator = readCoordinator;
|
||||
this.concurrencyFactor = concurrencyFactor;
|
||||
|
|
@ -127,7 +128,7 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
while (sentQueryIterator == null || !sentQueryIterator.hasNext())
|
||||
{
|
||||
// If we don't have more range to handle, we're done
|
||||
if (!replicaPlans.hasNext())
|
||||
if (!coordinatorPlans.hasNext())
|
||||
return endOfData();
|
||||
|
||||
// else, sends the next batch of concurrent queries (after having close the previous iterator)
|
||||
|
|
@ -205,31 +206,30 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
return new AccordRangeResponse(result, rangeCommand.isReversed());
|
||||
}
|
||||
|
||||
private SingleRangeResponse executeNormal(ReplicaPlan.ForRangeRead replicaPlan, PartitionRangeReadCommand rangeCommand, ReadCoordinator readCoordinator)
|
||||
private SingleRangeResponse executeNormal(CoordinationPlan.ForRangeRead plan, PartitionRangeReadCommand rangeCommand, ReadCoordinator readCoordinator)
|
||||
{
|
||||
rangeCommand = (PartitionRangeReadCommand) readCoordinator.maybeAllowOutOfRangeReads(rangeCommand, replicaPlan.consistencyLevel());
|
||||
rangeCommand = (PartitionRangeReadCommand) readCoordinator.maybeAllowOutOfRangeReads(rangeCommand, plan.consistencyLevel());
|
||||
// If enabled, request repaired data tracking info from full replicas, but
|
||||
// only if there are multiple full replicas to compare results from.
|
||||
boolean trackRepairedStatus = DatabaseDescriptor.getRepairedDataTrackingForRangeReadsEnabled()
|
||||
&& replicaPlan.contacts().filter(Replica::isFull).size() > 1
|
||||
&& plan.replicas().contacts().filter(Replica::isFull).size() > 1
|
||||
&& !command.metadata().replicationType().isTracked();
|
||||
|
||||
ReplicaPlan.SharedForRangeRead sharedReplicaPlan = ReplicaPlan.shared(replicaPlan);
|
||||
ReadRepair<EndpointsForRange, ReplicaPlan.ForRangeRead> readRepair =
|
||||
ReadRepair.create(readCoordinator, command, sharedReplicaPlan, requestTime);
|
||||
ReadRepair.create(readCoordinator, command, plan, requestTime);
|
||||
DataResolver<EndpointsForRange, ReplicaPlan.ForRangeRead> resolver =
|
||||
new DataResolver<>(readCoordinator, rangeCommand, sharedReplicaPlan, readRepair, requestTime, trackRepairedStatus);
|
||||
new DataResolver<>(readCoordinator, rangeCommand, plan, readRepair, requestTime, trackRepairedStatus);
|
||||
ReadCallback<EndpointsForRange, ReplicaPlan.ForRangeRead> handler =
|
||||
new ReadCallback<>(resolver, rangeCommand, sharedReplicaPlan, requestTime);
|
||||
checkState(!replicaPlan.contacts().anyMatch(Replica::isTransient), "Transient replication requires mutation tracking");
|
||||
new ReadCallback<>(resolver, rangeCommand, plan, requestTime);
|
||||
checkState(!plan.replicas().contacts().anyMatch(Replica::isTransient), "Transient replication requires mutation tracking");
|
||||
|
||||
if (replicaPlan.contacts().size() == 1 && replicaPlan.contacts().get(0).isSelf() && readCoordinator.localReadSupported())
|
||||
if (plan.replicas().contacts().size() == 1 && plan.replicas().contacts().get(0).isSelf() && readCoordinator.localReadSupported())
|
||||
{
|
||||
Stage.READ.execute(new StorageProxy.LocalReadRunnable(rangeCommand, handler, requestTime, trackRepairedStatus));
|
||||
}
|
||||
else
|
||||
{
|
||||
for (Replica replica : replicaPlan.contacts())
|
||||
for (Replica replica : plan.replicas().contacts())
|
||||
{
|
||||
Tracing.trace("Enqueuing request to {}", replica);
|
||||
Message<ReadCommand> message = rangeCommand.createMessage(trackRepairedStatus && replica.isFull(), requestTime);
|
||||
|
|
@ -243,20 +243,20 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
/**
|
||||
* Queries the provided sub-range.
|
||||
*
|
||||
* @param replicaPlan the subRange to query.
|
||||
* @param plan the subRange to query.
|
||||
* @param isFirst in the case where multiple queries are sent in parallel, whether that's the first query on
|
||||
* that batch or not. The reason it matters is that whe paging queries, the command (more specifically the
|
||||
* {@code DataLimits}) may have "state" information and that state may only be valid for the first query (in
|
||||
* that it's the query that "continues" whatever we're previously queried).
|
||||
*/
|
||||
private PartitionIterator query(ClusterMetadata cm, ReplicaPlan.ForRangeRead replicaPlan, ReadCoordinator readCoordinator, List<ReadRepair<?, ?>> readRepairs, boolean isFirst)
|
||||
private PartitionIterator query(ClusterMetadata cm, CoordinationPlan.ForRangeRead plan, ReadCoordinator readCoordinator, List<ReadRepair<?, ?>> readRepairs, boolean isFirst)
|
||||
{
|
||||
PartitionRangeReadCommand rangeCommand = command.forSubRange(replicaPlan.range(), isFirst);
|
||||
PartitionRangeReadCommand rangeCommand = command.forSubRange(plan.replicas().range(), isFirst);
|
||||
|
||||
// Accord interop execution should always be coordinated through the C* plumbing
|
||||
if (!readCoordinator.isEventuallyConsistent())
|
||||
{
|
||||
SingleRangeResponse response = executeNormal(replicaPlan, rangeCommand, readCoordinator);
|
||||
SingleRangeResponse response = executeNormal(plan, rangeCommand, readCoordinator);
|
||||
readRepairs.add(response.getReadRepair());
|
||||
return response;
|
||||
}
|
||||
|
|
@ -271,11 +271,11 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
|
||||
if (accordSplit.target == RangeReadTarget.accord && readCoordinator.isEventuallyConsistent())
|
||||
{
|
||||
return executeAccord(cm, accordSplit.read, replicaPlan.consistencyLevel());
|
||||
return executeAccord(cm, accordSplit.read, plan.consistencyLevel());
|
||||
}
|
||||
else
|
||||
{
|
||||
return executeNormalWithMigrationSplit(cm, replicaPlan, readCoordinator, readRepairs, accordSplit.read);
|
||||
return executeNormalWithMigrationSplit(cm, plan, readCoordinator, readRepairs, accordSplit.read);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -310,11 +310,11 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
{
|
||||
if (accordSplit.target == RangeReadTarget.accord && readCoordinator.isEventuallyConsistent())
|
||||
{
|
||||
responses.add(executeAccord(cm, accordSplit.read, replicaPlan.consistencyLevel()));
|
||||
responses.add(executeAccord(cm, accordSplit.read, plan.consistencyLevel()));
|
||||
}
|
||||
else
|
||||
{
|
||||
responses.add(executeNormalWithMigrationSplit(cm, replicaPlan, readCoordinator, readRepairs, accordSplit.read));
|
||||
responses.add(executeNormalWithMigrationSplit(cm, plan, readCoordinator, readRepairs, accordSplit.read));
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -325,7 +325,7 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
command.metadata().enforceStrictLiveness());
|
||||
}
|
||||
|
||||
private PartitionIterator executeSplit(RangeReadWithReplication split, ReplicaPlan.ForRangeRead replicaPlan, List<ReadRepair<?, ?>> readRepairs)
|
||||
private PartitionIterator executeSplit(RangeReadWithReplication split, CoordinationPlan.ForRangeRead replicaPlan, List<ReadRepair<?, ?>> readRepairs)
|
||||
{
|
||||
if (split.useTracked)
|
||||
{
|
||||
|
|
@ -345,20 +345,20 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
* Execute a normal C* range, splitting for migration if needed.
|
||||
*/
|
||||
private PartitionIterator executeNormalWithMigrationSplit(ClusterMetadata cm,
|
||||
ReplicaPlan.ForRangeRead replicaPlan,
|
||||
ReadCoordinator readCoordinator,
|
||||
List<ReadRepair<?, ?>> readRepairs,
|
||||
PartitionRangeReadCommand rangeCommand)
|
||||
CoordinationPlan.ForRangeRead plan,
|
||||
ReadCoordinator readCoordinator,
|
||||
List<ReadRepair<?, ?>> readRepairs,
|
||||
PartitionRangeReadCommand rangeCommand)
|
||||
{
|
||||
List<RangeReadWithReplication> migrationSplits = MigrationRouter.splitRangeRead(cm, rangeCommand);
|
||||
|
||||
if (migrationSplits.size() == 1)
|
||||
return executeSplit(migrationSplits.get(0), replicaPlan, readRepairs);
|
||||
return executeSplit(migrationSplits.get(0), plan, readRepairs);
|
||||
|
||||
List<PartitionIterator> responses = new ArrayList<>(migrationSplits.size());
|
||||
|
||||
for (RangeReadWithReplication split : migrationSplits)
|
||||
responses.add(executeSplit(split, replicaPlan, readRepairs));
|
||||
responses.add(executeSplit(split, plan, readRepairs));
|
||||
|
||||
// Apply limits since migration splits may have gaps in results
|
||||
return rangeCommand.limits().filter(PartitionIterators.concat(responses),
|
||||
|
|
@ -375,26 +375,26 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
ClusterMetadata cm = ClusterMetadata.current();
|
||||
try
|
||||
{
|
||||
for (int i = 0; i < concurrencyFactor && replicaPlans.hasNext(); )
|
||||
for (int i = 0; i < concurrencyFactor && coordinatorPlans.hasNext(); )
|
||||
{
|
||||
ReplicaPlan.ForRangeRead replicaPlan = replicaPlans.next();
|
||||
CoordinationPlan.ForRangeRead plan = coordinatorPlans.next();
|
||||
boolean isFirst = i == 0;
|
||||
PartitionIterator response;
|
||||
// Only add the retry wrapper to reroute for the top level coordinator execution
|
||||
// not Accord's interop execution
|
||||
if (readCoordinator.isEventuallyConsistent())
|
||||
{
|
||||
Function<ClusterMetadata, PartitionIterator> querySupplier = clusterMetadata -> query(clusterMetadata, replicaPlan, readCoordinator, readRepairs, isFirst);
|
||||
response = retryingPartitionIterator(querySupplier, replicaPlan.consistencyLevel());
|
||||
Function<ClusterMetadata, PartitionIterator> querySupplier = clusterMetadata -> query(clusterMetadata, plan, readCoordinator, readRepairs, isFirst);
|
||||
response = retryingPartitionIterator(querySupplier, plan.consistencyLevel());
|
||||
}
|
||||
else
|
||||
{
|
||||
response = query(cm, replicaPlan, readCoordinator, readRepairs, isFirst);
|
||||
response = query(cm, plan, readCoordinator, readRepairs, isFirst);
|
||||
}
|
||||
concurrentQueries.add(response);
|
||||
// due to RangeMerger, coordinator may fetch more ranges than required by concurrency factor.
|
||||
rangesQueried += replicaPlan.vnodeCount();
|
||||
i += replicaPlan.vnodeCount();
|
||||
rangesQueried += plan.replicas().vnodeCount();
|
||||
i += plan.replicas().vnodeCount();
|
||||
}
|
||||
batchesRequested++;
|
||||
}
|
||||
|
|
@ -480,7 +480,7 @@ public class RangeCommandIterator extends AbstractIterator<RowIterator> implemen
|
|||
if (sentQueryIterator != null)
|
||||
sentQueryIterator.close();
|
||||
|
||||
replicaPlans.close();
|
||||
coordinatorPlans.close();
|
||||
}
|
||||
finally
|
||||
{
|
||||
|
|
|
|||
|
|
@ -77,11 +77,11 @@ public class RangeCommands
|
|||
Tracing.trace("Computing ranges to query");
|
||||
|
||||
Keyspace keyspace = Keyspace.open(command.metadata().keyspace);
|
||||
ReplicaPlanIterator replicaPlans = new ReplicaPlanIterator(command.dataRange().keyRange(),
|
||||
command.indexQueryPlan(),
|
||||
keyspace,
|
||||
command.metadata().id(),
|
||||
consistencyLevel);
|
||||
CoordinationPlanIterator replicaPlans = new CoordinationPlanIterator(command.dataRange().keyRange(),
|
||||
command.indexQueryPlan(),
|
||||
keyspace,
|
||||
command.metadata().id(),
|
||||
consistencyLevel);
|
||||
|
||||
if (command.isTopK())
|
||||
return new ScanAllRangesCommandIterator(keyspace, replicaPlans, command, readCoordinator, replicaPlans.size(), requestTime);
|
||||
|
|
@ -112,7 +112,7 @@ public class RangeCommands
|
|||
Tracing.trace("Submitting range requests on {} ranges with a concurrency of {}", replicaPlans.size(), concurrencyFactor);
|
||||
}
|
||||
|
||||
ReplicaPlanMerger mergedReplicaPlans = new ReplicaPlanMerger(replicaPlans, keyspace, command.metadata().id(), consistencyLevel);
|
||||
CoordinationPlanMerger mergedReplicaPlans = new CoordinationPlanMerger(replicaPlans, keyspace, command.metadata().id(), consistencyLevel);
|
||||
return new RangeCommandIterator(mergedReplicaPlans,
|
||||
command,
|
||||
readCoordinator,
|
||||
|
|
@ -150,15 +150,15 @@ public class RangeCommands
|
|||
try
|
||||
{
|
||||
Keyspace keyspace = Keyspace.open(metadata.keyspace);
|
||||
ReplicaPlanIterator rangeIterator = new ReplicaPlanIterator(DataRange.allData(metadata.partitioner).keyRange(),
|
||||
null,
|
||||
keyspace,
|
||||
metadata.id,
|
||||
consistency);
|
||||
CoordinationPlanIterator rangeIterator = new CoordinationPlanIterator(DataRange.allData(metadata.partitioner).keyRange(),
|
||||
null,
|
||||
keyspace,
|
||||
metadata.id,
|
||||
consistency);
|
||||
|
||||
// Called for the side effect of running assureSufficientLiveReplicasForRead.
|
||||
// Deliberately called with an invalid vnode count in case it is used elsewhere in the future..
|
||||
rangeIterator.forEachRemaining(r -> ReplicaPlans.forRangeRead(keyspace, metadata.id, null, consistency, r.range(), -1));
|
||||
rangeIterator.forEachRemaining(r -> ReplicaPlans.forRangeRead(keyspace, metadata.id, null, consistency, r.replicas().range(), -1));
|
||||
return true;
|
||||
}
|
||||
catch (UnavailableException e)
|
||||
|
|
|
|||
|
|
@ -30,10 +30,10 @@ import org.apache.cassandra.db.PartitionRangeReadCommand;
|
|||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.db.partitions.PartitionIterator;
|
||||
import org.apache.cassandra.index.Index;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.service.reads.DataResolver;
|
||||
|
|
@ -61,13 +61,13 @@ public class ScanAllRangesCommandIterator extends RangeCommandIterator
|
|||
{
|
||||
private final Keyspace keyspace;
|
||||
|
||||
ScanAllRangesCommandIterator(Keyspace keyspace, CloseableIterator<ReplicaPlan.ForRangeRead> replicaPlans,
|
||||
ScanAllRangesCommandIterator(Keyspace keyspace, CloseableIterator<CoordinationPlan.ForRangeRead> coordinationPlans,
|
||||
PartitionRangeReadCommand command,
|
||||
ReadCoordinator readCoordinator,
|
||||
int totalRangeCount,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(replicaPlans, command, readCoordinator, totalRangeCount, totalRangeCount, totalRangeCount, requestTime);
|
||||
super(coordinationPlans, command, readCoordinator, totalRangeCount, totalRangeCount, totalRangeCount, requestTime);
|
||||
Preconditions.checkState(command.isTopK());
|
||||
|
||||
this.keyspace = keyspace;
|
||||
|
|
@ -79,23 +79,22 @@ public class ScanAllRangesCommandIterator extends RangeCommandIterator
|
|||
// get all replicas to contact
|
||||
Set<InetAddressAndPort> replicasToQuery = null;
|
||||
ConsistencyLevel consistencyLevel = null;
|
||||
while (replicaPlans.hasNext())
|
||||
while (coordinatorPlans.hasNext())
|
||||
{
|
||||
if (replicasToQuery == null)
|
||||
replicasToQuery = new HashSet<>();
|
||||
|
||||
ReplicaPlan.ForRangeRead replicaPlan = replicaPlans.next();
|
||||
replicasToQuery.addAll(replicaPlan.contacts().endpoints());
|
||||
CoordinationPlan.ForRangeRead replicaPlan = coordinatorPlans.next();
|
||||
replicasToQuery.addAll(replicaPlan.replicas().contacts().endpoints());
|
||||
consistencyLevel = replicaPlan.consistencyLevel();
|
||||
}
|
||||
|
||||
if (replicasToQuery == null || replicasToQuery.isEmpty())
|
||||
return EmptyIterators.partition();
|
||||
|
||||
ReplicaPlan.ForRangeRead plan = ReplicaPlans.forFullRangeRead(keyspace, consistencyLevel, command.dataRange().keyRange(), replicasToQuery, totalRangeCount);
|
||||
ReplicaPlan.SharedForRangeRead sharedReplicaPlan = ReplicaPlan.shared(plan);
|
||||
DataResolver<EndpointsForRange, ReplicaPlan.ForRangeRead> resolver = new DataResolver<>(ReadCoordinator.DEFAULT, command, sharedReplicaPlan, NoopReadRepair.instance, requestTime, false);
|
||||
ReadCallback<EndpointsForRange, ReplicaPlan.ForRangeRead> handler = new ReadCallback<>(resolver, command, sharedReplicaPlan, requestTime);
|
||||
CoordinationPlan.ForRangeRead plan = CoordinationPlan.forFullRangeRead(keyspace, consistencyLevel, command.dataRange().keyRange(), replicasToQuery, totalRangeCount);
|
||||
DataResolver<EndpointsForRange, ReplicaPlan.ForRangeRead> resolver = new DataResolver<>(ReadCoordinator.DEFAULT, command, plan, NoopReadRepair.instance, requestTime, false);
|
||||
ReadCallback<EndpointsForRange, ReplicaPlan.ForRangeRead> handler = new ReadCallback<>(resolver, command, plan, requestTime);
|
||||
|
||||
int nodes = 0;
|
||||
for (InetAddressAndPort endpoint : replicasToQuery)
|
||||
|
|
@ -106,7 +105,7 @@ public class ScanAllRangesCommandIterator extends RangeCommandIterator
|
|||
nodes++;
|
||||
}
|
||||
|
||||
rangesQueried += plan.vnodeCount();
|
||||
rangesQueried += plan.replicas().vnodeCount();
|
||||
batchesRequested++;
|
||||
|
||||
Tracing.trace("Submitted scanning all ranges requests to {} nodes", nodes);
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ import org.apache.cassandra.db.ReadCommand;
|
|||
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||
import org.apache.cassandra.db.partitions.PartitionIterator;
|
||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -58,7 +59,7 @@ public abstract class AbstractReadRepair<E extends Endpoints<E>, P extends Repli
|
|||
protected final ReadCoordinator coordinator;
|
||||
protected final ReadCommand command;
|
||||
protected final Dispatcher.RequestTime requestTime;
|
||||
protected final ReplicaPlan.Shared<E, P> replicaPlan;
|
||||
protected final CoordinationPlan.ForRead<E, P> plan;
|
||||
protected final ColumnFamilyStore cfs;
|
||||
|
||||
private volatile DigestRepair<E, P> digestRepair = null;
|
||||
|
|
@ -78,19 +79,19 @@ public abstract class AbstractReadRepair<E extends Endpoints<E>, P extends Repli
|
|||
}
|
||||
|
||||
public AbstractReadRepair(ReadCoordinator coordinator, ReadCommand command,
|
||||
ReplicaPlan.Shared<E, P> replicaPlan,
|
||||
CoordinationPlan.ForRead<E, P> plan,
|
||||
Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this.coordinator = coordinator;
|
||||
this.command = command;
|
||||
this.requestTime = requestTime;
|
||||
this.replicaPlan = replicaPlan;
|
||||
this.plan = plan;
|
||||
this.cfs = Keyspace.openAndGetStore(command.metadata());
|
||||
}
|
||||
|
||||
protected P replicaPlan()
|
||||
{
|
||||
return replicaPlan.get();
|
||||
return plan.replicas();
|
||||
}
|
||||
|
||||
void sendReadCommand(Replica to, ReadCallback<E, P> readCallback, boolean speculative, boolean trackRepairedStatus)
|
||||
|
|
@ -135,9 +136,10 @@ public abstract class AbstractReadRepair<E extends Endpoints<E>, P extends Repli
|
|||
*/
|
||||
boolean trackRepairedStatus = DatabaseDescriptor.getRepairedDataTrackingForPartitionReadsEnabled();
|
||||
|
||||
CoordinationPlan.ForRead<E, P> repairPlan = plan.copyWithResetTracker();
|
||||
// Do a full data read to resolve the correct response (and repair node that need be)
|
||||
DataResolver<E, P> resolver = new DataResolver<>(coordinator, command, replicaPlan, this, requestTime, trackRepairedStatus);
|
||||
ReadCallback<E, P> readCallback = new ReadCallback<>(resolver, command, replicaPlan, requestTime);
|
||||
DataResolver<E, P> resolver = new DataResolver<>(coordinator, command, repairPlan, this, requestTime, trackRepairedStatus);
|
||||
ReadCallback<E, P> readCallback = new ReadCallback<>(resolver, command, repairPlan, requestTime);
|
||||
|
||||
digestRepair = new DigestRepair<>(resolver, readCallback, resultConsumer);
|
||||
|
||||
|
|
@ -175,7 +177,7 @@ public abstract class AbstractReadRepair<E extends Endpoints<E>, P extends Repli
|
|||
ConsistencyLevel consistency = replicaPlan().consistencyLevel();
|
||||
ConsistencyLevel speculativeCL = consistency.isDatacenterLocal() ? ConsistencyLevel.LOCAL_QUORUM : ConsistencyLevel.QUORUM;
|
||||
return consistency != ConsistencyLevel.EACH_QUORUM
|
||||
&& consistency.satisfies(speculativeCL, replicaPlan.get().replicationStrategy())
|
||||
&& consistency.satisfies(speculativeCL, plan.replicationStrategy())
|
||||
&& cfs.sampleReadLatencyMicros <= command.getTimeout(MICROSECONDS);
|
||||
}
|
||||
|
||||
|
|
@ -193,7 +195,7 @@ public abstract class AbstractReadRepair<E extends Endpoints<E>, P extends Repli
|
|||
if (uncontacted == null)
|
||||
return;
|
||||
|
||||
replicaPlan.addToContacts(uncontacted);
|
||||
plan.addToContacts(uncontacted);
|
||||
sendReadCommand(uncontacted, repair.readCallback, true, false);
|
||||
ReadRepairMetrics.speculatedRead.mark();
|
||||
ReadRepairDiagnostics.speculatedRead(this, uncontacted.endpoint(), replicaPlan());
|
||||
|
|
|
|||
|
|
@ -43,6 +43,7 @@ import org.apache.cassandra.db.ReadCommand;
|
|||
import org.apache.cassandra.db.ReadCommand.PotentialTxnConflicts;
|
||||
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators;
|
||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -135,9 +136,9 @@ public class BlockingReadRepair<E extends Endpoints<E>, P extends ReplicaPlan.Fo
|
|||
ForWrite repairPlan();
|
||||
}
|
||||
|
||||
BlockingReadRepair(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
BlockingReadRepair(ReadCoordinator coordinator, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(coordinator, command, replicaPlan, requestTime);
|
||||
super(coordinator, command, plan, requestTime);
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
@ -293,7 +294,7 @@ public class BlockingReadRepair<E extends Endpoints<E>, P extends ReplicaPlan.Fo
|
|||
|
||||
public void repairPartitionDirectly(ReadCoordinator readCoordinator, DecoratedKey dk, Map<Replica, Mutation> mutations, ForWrite writePlan)
|
||||
{
|
||||
ReadRepair delegateRR = ReadRepairStrategy.BLOCKING.create(readCoordinator, command, replicaPlan, requestTime);
|
||||
ReadRepair delegateRR = ReadRepairStrategy.BLOCKING.create(readCoordinator, command, plan, requestTime);
|
||||
delegateRR.repairPartition(dk, mutations, writePlan, ReadRepairSource.REPAIR_VIA_ACCORD);
|
||||
delegateRR.maybeSendAdditionalWrites();
|
||||
delegateRR.awaitWrites();
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ import org.apache.cassandra.db.DecoratedKey;
|
|||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -41,9 +42,9 @@ import org.apache.cassandra.transport.Dispatcher;
|
|||
public class ReadOnlyReadRepair<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
extends AbstractReadRepair<E, P>
|
||||
{
|
||||
ReadOnlyReadRepair(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
ReadOnlyReadRepair(ReadCoordinator coordinator, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(coordinator, command, replicaPlan, requestTime);
|
||||
super(coordinator, command, plan, requestTime);
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ import org.apache.cassandra.db.ReadCommand.PotentialTxnConflicts;
|
|||
import org.apache.cassandra.db.partitions.PartitionIterator;
|
||||
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators;
|
||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -41,13 +42,13 @@ public interface ReadRepair<E extends Endpoints<E>, P extends ReplicaPlan.ForRea
|
|||
public interface Factory
|
||||
{
|
||||
<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime);
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime);
|
||||
}
|
||||
|
||||
static <E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return command.metadata().params.readRepair.create(coordinator, command, replicaPlan, requestTime);
|
||||
return command.metadata().params.readRepair.create(coordinator, command, plan, requestTime);
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@
|
|||
package org.apache.cassandra.service.reads.repair;
|
||||
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||
|
|
@ -31,18 +32,18 @@ public enum ReadRepairStrategy implements ReadRepair.Factory
|
|||
NONE
|
||||
{
|
||||
public <E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return new ReadOnlyReadRepair<>(coordinator, command, replicaPlan, requestTime);
|
||||
return new ReadOnlyReadRepair<>(coordinator, command, plan, requestTime);
|
||||
}
|
||||
},
|
||||
|
||||
BLOCKING
|
||||
{
|
||||
public <E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
ReadRepair<E, P> create(ReadCoordinator coordinator, ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return new BlockingReadRepair<>(coordinator, command, replicaPlan, requestTime);
|
||||
return new BlockingReadRepair<>(coordinator, command, plan, requestTime);
|
||||
}
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -53,9 +53,9 @@ import org.apache.cassandra.dht.ExcludingBounds;
|
|||
import org.apache.cassandra.dht.Range;
|
||||
import org.apache.cassandra.index.Index;
|
||||
import org.apache.cassandra.index.transactions.UpdateTransaction;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.transport.Dispatcher;
|
||||
import org.apache.cassandra.utils.concurrent.Future;
|
||||
|
||||
|
|
@ -333,14 +333,15 @@ public abstract class PartialTrackedRangeRead extends PartialTrackedRead
|
|||
|
||||
Keyspace keyspace = Keyspace.open(command.metadata().keyspace);
|
||||
PartitionRangeReadCommand followUpCmd = command.withUpdatedLimitsAndDataRange(newLimits, newDataRange);
|
||||
ReplicaPlan.ForRangeRead replicaPlan = ReplicaPlans.forRangeRead(keyspace,
|
||||
command.metadata().id,
|
||||
followUpCmd.indexQueryPlan(),
|
||||
consistencyLevel,
|
||||
followUpCmd.dataRange().keyRange(),
|
||||
1);
|
||||
CoordinationPlan.ForRangeRead plan = CoordinationPlan.forRangeRead(ClusterMetadata.current(),
|
||||
keyspace,
|
||||
command.metadata().id,
|
||||
followUpCmd.indexQueryPlan(),
|
||||
consistencyLevel,
|
||||
followUpCmd.dataRange().keyRange(),
|
||||
1);
|
||||
|
||||
TrackedRead.Range read = TrackedRead.Range.create(followUpCmd, replicaPlan, requestTime);
|
||||
TrackedRead.Range read = TrackedRead.Range.create(followUpCmd, plan, requestTime);
|
||||
logger.trace("Short read detected, starting followup read {}", read);
|
||||
return read;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -157,10 +157,12 @@ public class ReadReconciliations implements ExpiredStatePurger.Expireable
|
|||
return n;
|
||||
}
|
||||
|
||||
private static final class Coordinator
|
||||
static class Coordinator
|
||||
{
|
||||
private static final Logger logger = LoggerFactory.getLogger(Coordinator.class);
|
||||
|
||||
// FIXME: this will probably break per-DC consistency semantics of SatelliteReplicationStrategy
|
||||
// once read speculation is implemented
|
||||
private static final AtomicLongFieldUpdater<Coordinator> remainingUpdater =
|
||||
AtomicLongFieldUpdater.newUpdater(Coordinator.class, "remaining");
|
||||
private volatile long remaining; // three values packed into one atomic long
|
||||
|
|
@ -190,6 +192,24 @@ public class ReadReconciliations implements ExpiredStatePurger.Expireable
|
|||
summaries = new Accumulator<>(remainingSummaries);
|
||||
}
|
||||
|
||||
/**
|
||||
* Confirm that we're only counting responses from nodes initially chosen by the read coordinator
|
||||
* This is to prevent the implementation of tracked read speculation (doesn't exist yet) from breaking
|
||||
* the per-dc consistency semantics of SatelliteReplicationStrategy because of the simple count completion
|
||||
* mechanics this class uses for tracking received summarys / syncAcks
|
||||
*/
|
||||
private void checkNodeIsExpected(int check)
|
||||
{
|
||||
if (check == dataNode)
|
||||
return;
|
||||
|
||||
for (int node : summaryNodes)
|
||||
if (check == node)
|
||||
return;
|
||||
|
||||
throw new IllegalStateException("Not expecting response from node " + check);
|
||||
}
|
||||
|
||||
/**
|
||||
* For all the logs in the summary that are owned by us, preemptively prioritise delivery of
|
||||
* any mutations that are absent from other participating nodes according to our primary coordinator
|
||||
|
|
@ -197,6 +217,7 @@ public class ReadReconciliations implements ExpiredStatePurger.Expireable
|
|||
*/
|
||||
boolean acceptLocalSummary(MutationSummary summary)
|
||||
{
|
||||
checkNodeIsExpected(LOCAL_NODE);
|
||||
IntArrayList remoteNodes = new IntArrayList(summaryNodes.length, Integer.MIN_VALUE);
|
||||
if (dataNode != LOCAL_NODE)
|
||||
remoteNodes.addInt(dataNode);
|
||||
|
|
@ -238,6 +259,7 @@ public class ReadReconciliations implements ExpiredStatePurger.Expireable
|
|||
*/
|
||||
boolean acceptRemoteSummary(MutationSummary summary, int remoteNode)
|
||||
{
|
||||
checkNodeIsExpected(remoteNode);
|
||||
Log2OffsetsMap.Mutable missingMutations = new Log2OffsetsMap.Mutable();
|
||||
MutationTrackingService.instance().collectLocallyMissingMutations(summary, missingMutations);
|
||||
|
||||
|
|
@ -257,8 +279,9 @@ public class ReadReconciliations implements ExpiredStatePurger.Expireable
|
|||
return updateRemainingAndMaybeComplete(missingCount, -1, 0);
|
||||
}
|
||||
|
||||
boolean acceptSyncAck(int ignoredSyncId)
|
||||
boolean acceptSyncAck(int node)
|
||||
{
|
||||
checkNodeIsExpected(node);
|
||||
return updateRemainingAndMaybeComplete(0, 0, -1);
|
||||
}
|
||||
|
||||
|
|
@ -355,7 +378,7 @@ public class ReadReconciliations implements ExpiredStatePurger.Expireable
|
|||
return next;
|
||||
}
|
||||
|
||||
private boolean updateRemainingAndMaybeComplete(int mutationsDelta, int summariesDelta, int syncAcksDelta)
|
||||
boolean updateRemainingAndMaybeComplete(int mutationsDelta, int summariesDelta, int syncAcksDelta)
|
||||
{
|
||||
return updateRemaining(mutationsDelta, summariesDelta, syncAcksDelta) == 0 && complete();
|
||||
}
|
||||
|
|
|
|||
|
|
@ -57,13 +57,13 @@ import org.apache.cassandra.gms.FailureDetector;
|
|||
import org.apache.cassandra.io.IVersionedSerializer;
|
||||
import org.apache.cassandra.io.util.DataInputPlus;
|
||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.net.IVerbHandler;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessageFlag;
|
||||
|
|
@ -167,7 +167,7 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
|
||||
private final Id readId = Id.nextId();
|
||||
private final ReadCommand command;
|
||||
private final ReplicaPlan.AbstractForRead<E, P> replicaPlan;
|
||||
private final CoordinationPlan.ForRead<E, P> plan;
|
||||
private final ConsistencyLevel consistencyLevel;
|
||||
private final Dispatcher.RequestTime requestTime;
|
||||
|
||||
|
|
@ -188,17 +188,17 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
}
|
||||
}
|
||||
|
||||
public TrackedRead(ReadCommand command, ReplicaPlan.AbstractForRead<E, P> replicaPlan, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
public TrackedRead(ReadCommand command, CoordinationPlan.ForRead<E, P> plan, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
this.command = command;
|
||||
this.replicaPlan = replicaPlan;
|
||||
this.plan = plan;
|
||||
this.consistencyLevel = consistencyLevel;
|
||||
this.requestTime = requestTime;
|
||||
}
|
||||
|
||||
public ReplicaPlan.AbstractForRead<E, P> replicaPlan()
|
||||
public ReplicaPlan.ForRead<E, P> replicaPlan()
|
||||
{
|
||||
return replicaPlan;
|
||||
return plan.replicas();
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
@ -216,9 +216,9 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
|
||||
public static class Partition extends TrackedRead<EndpointsForToken, ReplicaPlan.ForTokenRead>
|
||||
{
|
||||
private Partition(SinglePartitionReadCommand command, ReplicaPlan.AbstractForRead<EndpointsForToken, ReplicaPlan.ForTokenRead> replicaPlan, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
private Partition(SinglePartitionReadCommand command, CoordinationPlan.ForTokenRead plan, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(command, replicaPlan, consistencyLevel, requestTime);
|
||||
super(command, plan, consistencyLevel, requestTime);
|
||||
}
|
||||
|
||||
public static Partition create(ClusterMetadata metadata, SinglePartitionReadCommand command, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
|
|
@ -227,15 +227,15 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
Keyspace keyspace = Keyspace.open(command.metadata().keyspace);
|
||||
ColumnFamilyStore cfs = keyspace.getColumnFamilyStore(command.metadata().id);
|
||||
SpeculativeRetryPolicy retry = cfs.metadata().params.speculativeRetry;
|
||||
ReplicaPlan.ForTokenRead replicaPlan = ReplicaPlans.forRead(metadata,
|
||||
keyspace,
|
||||
cfs.getTableId(),
|
||||
command.partitionKey().getToken(),
|
||||
command.indexQueryPlan(),
|
||||
consistencyLevel,
|
||||
retry,
|
||||
ReadCoordinator.DEFAULT);
|
||||
return new Partition(command, replicaPlan, consistencyLevel, requestTime);
|
||||
CoordinationPlan.ForTokenRead plan = CoordinationPlan.forTokenRead(metadata,
|
||||
keyspace,
|
||||
cfs.getTableId(),
|
||||
command.partitionKey().getToken(),
|
||||
command.indexQueryPlan(),
|
||||
consistencyLevel,
|
||||
retry,
|
||||
ReadCoordinator.DEFAULT);
|
||||
return new Partition(command, plan, consistencyLevel, requestTime);
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
@ -247,15 +247,15 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
|
||||
public static class Range extends TrackedRead<EndpointsForRange, ReplicaPlan.ForRangeRead>
|
||||
{
|
||||
private Range(PartitionRangeReadCommand command, ReplicaPlan.AbstractForRead<EndpointsForRange, ReplicaPlan.ForRangeRead> replicaPlan, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
private Range(PartitionRangeReadCommand command, CoordinationPlan.ForRangeRead plan, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(command, replicaPlan, consistencyLevel, requestTime);
|
||||
super(command, plan, consistencyLevel, requestTime);
|
||||
}
|
||||
|
||||
public static TrackedRead.Range create(PartitionRangeReadCommand command, ReplicaPlan.ForRangeRead replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public static TrackedRead.Range create(PartitionRangeReadCommand command, CoordinationPlan.ForRangeRead plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
Preconditions.checkArgument(command.metadata().replicationType().isTracked());
|
||||
return new Range(command, replicaPlan, replicaPlan.consistencyLevel(), requestTime);
|
||||
return new Range(command, plan, plan.consistencyLevel(), requestTime);
|
||||
}
|
||||
|
||||
@Override
|
||||
|
|
@ -285,7 +285,7 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
{
|
||||
// TODO: skip local coordination if this node knows its recovering from an outage
|
||||
// TODO: read speculation
|
||||
Replica localReplica = replicaPlan.lookup(FBUtilities.getBroadcastAddressAndPort());
|
||||
Replica localReplica = plan.replicas().lookup(FBUtilities.getBroadcastAddressAndPort());
|
||||
if (localReplica != null)
|
||||
readMetrics.localRequests.mark();
|
||||
else
|
||||
|
|
@ -294,10 +294,10 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
// create an id
|
||||
// select data node
|
||||
// select summary nodes
|
||||
E selected = replicaPlan.contacts().filter(r -> FailureDetector.instance.isAlive(r.endpoint()));
|
||||
if (selected.size() < replicaPlan.readQuorum())
|
||||
throw new UnavailableException(String.format("Insufficient replicas available for read (%d < %d)", selected.size(), replicaPlan.readQuorum()),
|
||||
replicaPlan.consistencyLevel(), selected.size(), replicaPlan.readQuorum());
|
||||
E selected = plan.replicas().contacts().filter(r -> FailureDetector.instance.isAlive(r.endpoint()));
|
||||
if (selected.size() < plan.replicas().readQuorum())
|
||||
throw new UnavailableException(String.format("Insufficient replicas available for read (%d < %d)", selected.size(), plan.replicas().readQuorum()),
|
||||
plan.consistencyLevel(), selected.size(), plan.replicas().readQuorum());
|
||||
Replica dataReplica = localReplica != null && localReplica.isFull()
|
||||
? localReplica
|
||||
: Iterables.getOnlyElement(selected.filter(Replica::isFull, 1));
|
||||
|
|
@ -413,17 +413,17 @@ public abstract class TrackedRead<E extends Endpoints<E>, P extends ReplicaPlan.
|
|||
RequestFailure failure = (RequestFailure) ex;
|
||||
if (failure.reason == RequestFailureReason.TIMEOUT)
|
||||
{
|
||||
throw new ReadTimeoutException(replicaPlan.consistencyLevel(), 0, replicaPlan.readQuorum(), false);
|
||||
throw new ReadTimeoutException(plan.consistencyLevel(), 0, plan.replicas().readQuorum(), false);
|
||||
}
|
||||
|
||||
reasons = failure.reasonByEndpoint();
|
||||
}
|
||||
|
||||
throw new ReadFailureException(replicaPlan.consistencyLevel(), 0, replicaPlan.readQuorum(), false, reasons);
|
||||
throw new ReadFailureException(plan.consistencyLevel(), 0, plan.replicas().readQuorum(), false, reasons);
|
||||
}
|
||||
catch (TimeoutException e)
|
||||
{
|
||||
throw new ReadTimeoutException(replicaPlan.consistencyLevel(), 0, replicaPlan.readQuorum(), false);
|
||||
throw new ReadTimeoutException(plan.consistencyLevel(), 0, plan.replicas().readQuorum(), false);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -155,9 +155,6 @@ public final class ServerTestUtils
|
|||
{
|
||||
daemonInitialization();
|
||||
|
||||
// Need to happen after daemonInitialization for config to be set, but before CFS initialization
|
||||
MutationJournal.start();
|
||||
|
||||
if (isServerPrepared)
|
||||
return;
|
||||
|
||||
|
|
@ -174,6 +171,10 @@ public final class ServerTestUtils
|
|||
throw new RuntimeException(e);
|
||||
}
|
||||
|
||||
// Need to happen after daemonInitialization for config to be set and after
|
||||
// cleanupAndLeaveDirs for directories to exist, but before CFS initialization
|
||||
MutationJournal.start();
|
||||
|
||||
try
|
||||
{
|
||||
remoteAddrs.add(InetAddressAndPort.getByName("127.0.0.4"));
|
||||
|
|
|
|||
|
|
@ -50,7 +50,8 @@ public class MutationJournalTableTest extends CQLTester
|
|||
@Before
|
||||
public void setUp()
|
||||
{
|
||||
schemaChange("CREATE TABLE " + KEYSPACE + ".tbl(pk int PRIMARY KEY, v int)");
|
||||
schemaChange("CREATE KEYSPACE ks WITH replication={'class':'SimpleStrategy', 'replication_factor':1} AND replication_type='tracked'");
|
||||
schemaChange("CREATE TABLE ks.tbl(pk int PRIMARY KEY, v int)");
|
||||
}
|
||||
|
||||
@Test
|
||||
|
|
@ -58,11 +59,12 @@ public class MutationJournalTableTest extends CQLTester
|
|||
{
|
||||
// Start the mutation journal
|
||||
MutationJournal.start();
|
||||
enableCoordinatorExecution();
|
||||
|
||||
// Write data to trigger journal writes
|
||||
for (int i = 0; i < 100; i++)
|
||||
{
|
||||
execute("INSERT INTO " + KEYSPACE + ".tbl(pk, v) VALUES (?, ?)", i, i);
|
||||
execute("INSERT INTO ks.tbl(pk, v) VALUES (?, ?)", i, i);
|
||||
}
|
||||
|
||||
// Query the virtual table
|
||||
|
|
|
|||
|
|
@ -393,7 +393,7 @@ public class IndexStatusManagerTest
|
|||
Index.QueryPlan qp = mockedQueryPlan(indexes);
|
||||
ConsistencyLevel cl = mockedConsistencyLevel(testcase.numRequired);
|
||||
|
||||
EndpointsForRange actual = IndexStatusManager.instance.filterForQuery(endpoints, ks, qp, cl);
|
||||
EndpointsForRange actual = IndexStatusManager.instance.filterForQueryOrThrow(endpoints, ks, qp, cl);
|
||||
|
||||
assertArrayEquals(
|
||||
testcase.expected.stream().toArray(),
|
||||
|
|
|
|||
|
|
@ -0,0 +1,427 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.UnknownHostException;
|
||||
import java.util.Arrays;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import org.junit.Test;
|
||||
|
||||
import static org.junit.Assert.*;
|
||||
import static org.mockito.Mockito.*;
|
||||
|
||||
public class CompositeTrackerTest
|
||||
{
|
||||
private InetAddressAndPort endpoint(String ip) throws UnknownHostException
|
||||
{
|
||||
return InetAddressAndPort.getByName(ip);
|
||||
}
|
||||
|
||||
private ResponseTracker createMockTracker(boolean successful, boolean complete)
|
||||
{
|
||||
ResponseTracker tracker = mock(ResponseTracker.class);
|
||||
when(tracker.isSuccessful()).thenReturn(successful);
|
||||
when(tracker.isComplete()).thenReturn(complete);
|
||||
when(tracker.required()).thenReturn(2);
|
||||
when(tracker.received()).thenReturn(successful ? 2 : 0);
|
||||
when(tracker.failures()).thenReturn(successful ? 0 : 2);
|
||||
return tracker;
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testQuorumCalculation()
|
||||
{
|
||||
assertEquals(1, CompositeTracker.quorum(1));
|
||||
assertEquals(2, CompositeTracker.quorum(2));
|
||||
assertEquals(2, CompositeTracker.quorum(3));
|
||||
assertEquals(3, CompositeTracker.quorum(4));
|
||||
assertEquals(3, CompositeTracker.quorum(5));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testQuorumSuccessAndFailure()
|
||||
{
|
||||
// N=1: quorum=1
|
||||
assertTrue(new CompositeTracker(CompositeTracker.quorum(1), Arrays.asList(
|
||||
createMockTracker(true, true)
|
||||
)).isSuccessful());
|
||||
|
||||
// N=2: quorum=2 (both required)
|
||||
assertFalse(new CompositeTracker(CompositeTracker.quorum(2), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(false, false)
|
||||
)).isSuccessful());
|
||||
|
||||
assertTrue(new CompositeTracker(CompositeTracker.quorum(2), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(true, true)
|
||||
)).isSuccessful());
|
||||
|
||||
// N=3: quorum=2, last child fails
|
||||
assertTrue(new CompositeTracker(CompositeTracker.quorum(3), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(true, true), createMockTracker(false, false)
|
||||
)).isSuccessful());
|
||||
|
||||
// N=3: quorum=2, first child fails (any child can be the failure)
|
||||
assertTrue(new CompositeTracker(CompositeTracker.quorum(3), Arrays.asList(
|
||||
createMockTracker(false, true), createMockTracker(true, true), createMockTracker(true, true)
|
||||
)).isSuccessful());
|
||||
|
||||
// N=3: only 1 succeeds → not successful
|
||||
assertFalse(new CompositeTracker(CompositeTracker.quorum(3), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(false, true), createMockTracker(false, false)
|
||||
)).isSuccessful());
|
||||
|
||||
// N=4: quorum=3
|
||||
assertFalse(new CompositeTracker(CompositeTracker.quorum(4), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(true, true), createMockTracker(false, false), createMockTracker(false, false)
|
||||
)).isSuccessful());
|
||||
|
||||
assertTrue(new CompositeTracker(CompositeTracker.quorum(4), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(true, true), createMockTracker(true, true), createMockTracker(false, false)
|
||||
)).isSuccessful());
|
||||
|
||||
// N=5: quorum=3
|
||||
assertTrue(new CompositeTracker(CompositeTracker.quorum(5), Arrays.asList(
|
||||
createMockTracker(true, true), createMockTracker(true, true), createMockTracker(true, true), createMockTracker(false, false), createMockTracker(false, false)
|
||||
)).isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllSuccessAndFailure()
|
||||
{
|
||||
// All succeed
|
||||
CompositeTracker tracker = new CompositeTracker(2,
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(true, true)
|
||||
);
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertTrue(tracker.isComplete());
|
||||
|
||||
// First fails
|
||||
assertFalse(new CompositeTracker(2,
|
||||
createMockTracker(false, true),
|
||||
createMockTracker(true, true)
|
||||
).isSuccessful());
|
||||
|
||||
// Second fails
|
||||
assertFalse(new CompositeTracker(2,
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, true)
|
||||
).isSuccessful());
|
||||
|
||||
// Both fail
|
||||
tracker = new CompositeTracker(2,
|
||||
createMockTracker(false, true),
|
||||
createMockTracker(false, true)
|
||||
);
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertTrue(tracker.isComplete());
|
||||
|
||||
// Any single failure in N children → overall failure
|
||||
assertFalse(new CompositeTracker(3,
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, true)
|
||||
).isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testQuorumEarlyCompletionWhenSuccessful()
|
||||
{
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(3), Arrays.asList(
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, false)
|
||||
));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testQuorumEarlyCompletionWhenImpossible()
|
||||
{
|
||||
// 4 children, 2 failed → max possible = 2 < quorum(4) → complete
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(4), Arrays.asList(
|
||||
createMockTracker(true, false),
|
||||
createMockTracker(false, true),
|
||||
createMockTracker(false, true),
|
||||
createMockTracker(false, false)
|
||||
));
|
||||
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertTrue(tracker.isComplete());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testQuorumNotCompleteWhenStillPossible()
|
||||
{
|
||||
// 4 children: 1 succeeded, 1 failed, 2 pending → max possible = 3 >= quorum(4)
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(4), Arrays.asList(
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, true),
|
||||
createMockTracker(false, false),
|
||||
createMockTracker(false, false)
|
||||
));
|
||||
|
||||
assertFalse(tracker.isComplete());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllEarlyCompletionOnAnyFailure()
|
||||
{
|
||||
CompositeTracker tracker = new CompositeTracker(3,
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, true), // Failed
|
||||
createMockTracker(false, false) // Still pending
|
||||
);
|
||||
|
||||
// One child has definitively failed → can't all succeed
|
||||
assertTrue(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllNotCompleteWhenPending()
|
||||
{
|
||||
CompositeTracker tracker = new CompositeTracker(2,
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, false) // Pending, not failed
|
||||
);
|
||||
|
||||
assertFalse(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testOnResponseDelegatesToAll() throws Exception
|
||||
{
|
||||
ResponseTracker c0 = mock(ResponseTracker.class);
|
||||
ResponseTracker c1 = mock(ResponseTracker.class);
|
||||
ResponseTracker c2 = mock(ResponseTracker.class);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(3), c0, c1, c2);
|
||||
|
||||
InetAddressAndPort ep = endpoint("127.0.0.1");
|
||||
tracker.onResponse(ep);
|
||||
|
||||
verify(c0).onResponse(ep);
|
||||
verify(c1).onResponse(ep);
|
||||
verify(c2).onResponse(ep);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testOnFailureDelegatesToAll() throws Exception
|
||||
{
|
||||
ResponseTracker c0 = mock(ResponseTracker.class);
|
||||
ResponseTracker c1 = mock(ResponseTracker.class);
|
||||
ResponseTracker c2 = mock(ResponseTracker.class);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(3, c0, c1, c2);
|
||||
|
||||
InetAddressAndPort ep = endpoint("127.0.0.1");
|
||||
tracker.onFailure(ep);
|
||||
|
||||
verify(c0).onFailure(ep);
|
||||
verify(c1).onFailure(ep);
|
||||
verify(c2).onFailure(ep);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAggregatesSums()
|
||||
{
|
||||
ResponseTracker c0 = mock(ResponseTracker.class);
|
||||
when(c0.received()).thenReturn(2);
|
||||
ResponseTracker c1 = mock(ResponseTracker.class);
|
||||
when(c1.received()).thenReturn(1);
|
||||
ResponseTracker c2 = mock(ResponseTracker.class);
|
||||
when(c2.received()).thenReturn(3);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(3), c0, c1, c2);
|
||||
|
||||
assertEquals(6, tracker.received());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testFailuresSum()
|
||||
{
|
||||
ResponseTracker c0 = mock(ResponseTracker.class);
|
||||
when(c0.failures()).thenReturn(1);
|
||||
ResponseTracker c1 = mock(ResponseTracker.class);
|
||||
when(c1.failures()).thenReturn(2);
|
||||
ResponseTracker c2 = mock(ResponseTracker.class);
|
||||
when(c2.failures()).thenReturn(0);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(3, c0, c1, c2);
|
||||
|
||||
assertEquals(3, tracker.failures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCountsTowardQuorumFromAny() throws Exception
|
||||
{
|
||||
ResponseTracker c0 = mock(ResponseTracker.class);
|
||||
when(c0.countsTowardQuorum(any())).thenReturn(true);
|
||||
ResponseTracker c1 = mock(ResponseTracker.class);
|
||||
when(c1.countsTowardQuorum(any())).thenReturn(false);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(2, c0, c1);
|
||||
|
||||
assertTrue(tracker.countsTowardQuorum(endpoint("127.0.0.1")));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCountsTowardQuorumFromNone() throws Exception
|
||||
{
|
||||
ResponseTracker c0 = mock(ResponseTracker.class);
|
||||
when(c0.countsTowardQuorum(any())).thenReturn(false);
|
||||
ResponseTracker c1 = mock(ResponseTracker.class);
|
||||
when(c1.countsTowardQuorum(any())).thenReturn(false);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(2), c0, c1);
|
||||
|
||||
assertFalse(tracker.countsTowardQuorum(endpoint("127.0.0.1")));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testNestedComposition()
|
||||
{
|
||||
// CompositeTracker(all) containing CompositeTracker(quorum)s
|
||||
CompositeTracker inner1 = new CompositeTracker(CompositeTracker.quorum(3),
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(false, false)
|
||||
);
|
||||
|
||||
CompositeTracker inner2 = new CompositeTracker(CompositeTracker.quorum(2),
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(true, true)
|
||||
);
|
||||
|
||||
CompositeTracker outer = new CompositeTracker(2, inner1, inner2);
|
||||
|
||||
assertTrue(outer.isSuccessful());
|
||||
assertTrue(outer.isComplete());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testNestedCompositionWithFailure()
|
||||
{
|
||||
CompositeTracker inner1 = new CompositeTracker(CompositeTracker.quorum(2),
|
||||
createMockTracker(true, true),
|
||||
createMockTracker(true, true)
|
||||
);
|
||||
|
||||
CompositeTracker inner2 = new CompositeTracker(CompositeTracker.quorum(2),
|
||||
createMockTracker(false, true),
|
||||
createMockTracker(false, true)
|
||||
);
|
||||
|
||||
CompositeTracker outer = new CompositeTracker(2, inner1, inner2);
|
||||
|
||||
assertFalse(outer.isSuccessful());
|
||||
assertTrue(outer.isComplete());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testConcurrentResponses() throws Exception
|
||||
{
|
||||
SimpleResponseTracker c0 = new SimpleResponseTracker(2, 3);
|
||||
SimpleResponseTracker c1 = new SimpleResponseTracker(2, 3);
|
||||
SimpleResponseTracker c2 = new SimpleResponseTracker(2, 3);
|
||||
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(3), c0, c1, c2);
|
||||
|
||||
ExecutorService executor = Executors.newFixedThreadPool(10);
|
||||
CountDownLatch startLatch = new CountDownLatch(1);
|
||||
CountDownLatch doneLatch = new CountDownLatch(9);
|
||||
|
||||
try
|
||||
{
|
||||
for (int i = 0; i < 9; i++)
|
||||
{
|
||||
final int index = i;
|
||||
executor.submit(() -> {
|
||||
try
|
||||
{
|
||||
startLatch.await();
|
||||
tracker.onResponse(endpoint("127.0.0." + index));
|
||||
doneLatch.countDown();
|
||||
}
|
||||
catch (Exception e)
|
||||
{
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
startLatch.countDown();
|
||||
assertTrue(doneLatch.await(10, TimeUnit.SECONDS));
|
||||
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertTrue(tracker.isComplete());
|
||||
assertEquals(27, tracker.received());
|
||||
}
|
||||
finally
|
||||
{
|
||||
executor.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testNullChildren()
|
||||
{
|
||||
new CompositeTracker(1, (ResponseTracker[]) null);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testEmptyChildren()
|
||||
{
|
||||
new CompositeTracker(1);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testBlockForZero()
|
||||
{
|
||||
new CompositeTracker(0, mock(ResponseTracker.class));
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testBlockForExceedsChildren()
|
||||
{
|
||||
new CompositeTracker(3, mock(ResponseTracker.class), mock(ResponseTracker.class));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testToString()
|
||||
{
|
||||
CompositeTracker tracker = new CompositeTracker(CompositeTracker.quorum(3),
|
||||
mock(ResponseTracker.class),
|
||||
mock(ResponseTracker.class),
|
||||
mock(ResponseTracker.class)
|
||||
);
|
||||
|
||||
String str = tracker.toString();
|
||||
assertTrue(str.contains("CompositeTracker"));
|
||||
assertTrue(str.contains("children=3"));
|
||||
assertTrue(str.contains("blockFor=2"));
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,73 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
|
||||
/**
|
||||
* Creates coordination plans with arbitrary replica plans
|
||||
*/
|
||||
public class CoordinationPlans
|
||||
{
|
||||
private static ResponseTracker createTrackerForWrite(ReplicaPlan.ForWrite plan)
|
||||
{
|
||||
return plan.replicationStrategy().createTrackerForWrite(plan.consistencyLevel(), plan, plan.pending, ClusterMetadata.current());
|
||||
}
|
||||
|
||||
public static CoordinationPlan.ForWriteWithIdeal create(ReplicaPlan.ForWrite plan, ConsistencyLevel idealCL)
|
||||
{
|
||||
ResponseTracker tracker = createTrackerForWrite(plan);
|
||||
|
||||
CoordinationPlan.ForWrite idealPlan = null;
|
||||
if (idealCL != null && idealCL != plan.consistencyLevel())
|
||||
{
|
||||
ReplicaPlan.ForWrite idealReplicaPlan = plan.withConsistencyLevel(idealCL);
|
||||
ResponseTracker idealTracker = createTrackerForWrite(idealReplicaPlan);
|
||||
idealPlan = new CoordinationPlan.ForWrite(idealReplicaPlan, idealTracker);
|
||||
}
|
||||
|
||||
return new CoordinationPlan.ForWriteWithIdeal(plan, tracker, idealPlan);
|
||||
}
|
||||
|
||||
public static CoordinationPlan.ForTokenRead create(ReplicaPlan.ForTokenRead plan)
|
||||
{
|
||||
return new CoordinationPlan.ForTokenRead(ReplicaPlan.shared(plan), trackerForRead(plan));
|
||||
}
|
||||
|
||||
public static CoordinationPlan.ForTokenRead create(ReplicaPlan.SharedForTokenRead shared)
|
||||
{
|
||||
return new CoordinationPlan.ForTokenRead(shared, trackerForRead(shared.get()));
|
||||
}
|
||||
|
||||
public static CoordinationPlan.ForRangeRead create(ReplicaPlan.ForRangeRead plan)
|
||||
{
|
||||
return new CoordinationPlan.ForRangeRead(ReplicaPlan.shared(plan), trackerForRead(plan));
|
||||
}
|
||||
|
||||
public static CoordinationPlan.ForRangeRead create(ReplicaPlan.SharedForRangeRead shared)
|
||||
{
|
||||
return new CoordinationPlan.ForRangeRead(shared, trackerForRead(shared.get()));
|
||||
}
|
||||
|
||||
private static <E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>> ResponseTracker trackerForRead(P plan)
|
||||
{
|
||||
return new SimpleResponseTracker(plan.readQuorum(), plan.readCandidates().size());
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,199 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.UnknownHostException;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import org.junit.After;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.CassandraTestBase;
|
||||
import org.apache.cassandra.CassandraTestBase.UseMurmur3Partitioner;
|
||||
import org.apache.cassandra.ServerTestUtils;
|
||||
import org.apache.cassandra.config.Config;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.marshal.AsciiType;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.distributed.test.log.ClusterMetadataTestHelper;
|
||||
import org.apache.cassandra.locator.AbstractReplicaCollection.ReplicaList;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.paxos.Paxos;
|
||||
import org.apache.cassandra.service.paxos.SatellitePaxosParticipants;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.membership.Location;
|
||||
|
||||
import static org.apache.cassandra.CassandraTestBase.DisableMBeanRegistration;
|
||||
import static org.apache.cassandra.CassandraTestBase.PrepareServerNoRegister;
|
||||
import static org.junit.Assert.assertFalse;
|
||||
import static org.junit.Assert.assertTrue;
|
||||
|
||||
@PrepareServerNoRegister
|
||||
@DisableMBeanRegistration
|
||||
@UseMurmur3Partitioner
|
||||
public class SatelliteCommitPlanTest extends CassandraTestBase
|
||||
{
|
||||
private static final String KEYSPACE = "scp_test";
|
||||
private static final LongToken TOKEN = new LongToken(150);
|
||||
|
||||
@After
|
||||
public void teardown()
|
||||
{
|
||||
ServerTestUtils.resetCMS();
|
||||
}
|
||||
|
||||
private void addToken(long token, String address, Location location) throws UnknownHostException
|
||||
{
|
||||
InetAddressAndPort addr = InetAddressAndPort.getByName(address);
|
||||
ClusterMetadataTestHelper.addEndpoint(addr, new LongToken(token), location);
|
||||
}
|
||||
|
||||
private void setupTopology() throws UnknownHostException
|
||||
{
|
||||
DatabaseDescriptor.setPaxosVariant(Config.PaxosVariant.v2);
|
||||
|
||||
Location dc1 = new Location("dc1", "rack1");
|
||||
Location dc2 = new Location("dc2", "rack1");
|
||||
Location sat1 = new Location("sat1", "rack1");
|
||||
Location sat2 = new Location("sat2", "rack1");
|
||||
|
||||
addToken(100, "10.0.0.10", dc1);
|
||||
addToken(200, "10.0.0.11", dc1);
|
||||
addToken(300, "10.0.0.12", dc1);
|
||||
|
||||
addToken(400, "10.1.0.10", dc2);
|
||||
addToken(500, "10.1.0.11", dc2);
|
||||
addToken(600, "10.1.0.12", dc2);
|
||||
|
||||
addToken(700, "10.2.0.10", sat1);
|
||||
addToken(800, "10.2.0.11", sat1);
|
||||
addToken(1100, "10.2.0.12", sat1);
|
||||
|
||||
addToken(900, "10.3.0.10", sat2);
|
||||
addToken(1000, "10.3.0.11", sat2);
|
||||
addToken(1200, "10.3.0.12", sat2);
|
||||
}
|
||||
|
||||
private void createDualDCKeyspace() throws Exception
|
||||
{
|
||||
String cql = "CREATE KEYSPACE " + KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
"'dc1.satellite.sat1': '3/3', " +
|
||||
"'dc2': '3', " +
|
||||
"'dc2.satellite.sat2': '3/3', " +
|
||||
"'primary': 'dc1'" +
|
||||
"} AND replication_type = 'tracked'";
|
||||
ClusterMetadataTestHelper.createKeyspace(cql);
|
||||
}
|
||||
|
||||
private SatelliteReplicationStrategy getSRS()
|
||||
{
|
||||
KeyspaceMetadata ksm = ClusterMetadata.current().schema.getKeyspaces().getNullable(KEYSPACE);
|
||||
return (SatelliteReplicationStrategy) ksm.replicationStrategy;
|
||||
}
|
||||
|
||||
private SatelliteReplicationStrategy.SatelliteCommitPlan createPlan() throws Exception
|
||||
{
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
SatelliteReplicationStrategy srs = getSRS();
|
||||
Keyspace keyspace = Keyspace.mockKS(metadata.schema.getKeyspaces().getNullable(KEYSPACE));
|
||||
return srs.createSatelliteCommitPlan(metadata, keyspace, TOKEN);
|
||||
}
|
||||
|
||||
private Set<String> collectDCs(AbstractReplicaCollection<?> endpoints)
|
||||
{
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
Set<String> dcs = new HashSet<>();
|
||||
ReplicaList epList = endpoints.list;
|
||||
for (int i = 0; i < endpoints.size(); i++)
|
||||
dcs.add(metadata.locator.location(epList.get(i).endpoint()).datacenter);
|
||||
return dcs;
|
||||
}
|
||||
|
||||
private void assertExpectedDCs(Set<String> dcs)
|
||||
{
|
||||
assertTrue("Should include sat1 (primary's satellite)", dcs.contains("sat1"));
|
||||
assertTrue("Should include dc2 (other full DC)", dcs.contains("dc2"));
|
||||
assertFalse("Should NOT include sat2 (dc2's satellite, not primary's)", dcs.contains("sat2"));
|
||||
assertFalse("Should NOT include dc1 (primary, handled by paxos)", dcs.contains("dc1"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Both createSatelliteCommitPlan and paxosParticipants should include only
|
||||
* the primary DC's satellite (sat1) and other full DCs (dc2), excluding
|
||||
* the primary DC itself (dc1) and non-primary satellites (sat2).
|
||||
*/
|
||||
@Test
|
||||
public void testEndpointDCSelection() throws Exception
|
||||
{
|
||||
setupTopology();
|
||||
createDualDCKeyspace();
|
||||
|
||||
// check commit plan endpoint selection
|
||||
SatelliteReplicationStrategy.SatelliteCommitPlan plan = createPlan();
|
||||
assertExpectedDCs(collectDCs(plan.liveEndpoints));
|
||||
|
||||
// check paxos participant endpoint selection
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
SatelliteReplicationStrategy srs = getSRS();
|
||||
TableMetadata table = TableMetadata.builder(KEYSPACE, "test_table")
|
||||
.addPartitionKeyColumn("key", AsciiType.instance)
|
||||
.build();
|
||||
|
||||
Paxos.Participants participants = srs.paxosParticipants(metadata, table,
|
||||
TOKEN,
|
||||
ConsistencyLevel.SERIAL,
|
||||
r -> true);
|
||||
|
||||
assertTrue(participants instanceof SatellitePaxosParticipants);
|
||||
SatellitePaxosParticipants spp = (SatellitePaxosParticipants) participants;
|
||||
assertExpectedDCs(collectDCs(spp.getAdditionalSummaryEndpoints()));
|
||||
}
|
||||
|
||||
/**
|
||||
* The tracker should not be complete with only the pre-completed primary DC,
|
||||
* but should complete once a quorum of groups has responded (dc1 pre-completed + sat1).
|
||||
*/
|
||||
@Test
|
||||
public void testTrackerCompletesWithQuorumOfGroups() throws Exception
|
||||
{
|
||||
setupTopology();
|
||||
createDualDCKeyspace();
|
||||
|
||||
SatelliteReplicationStrategy.SatelliteCommitPlan plan = createPlan();
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
assertFalse("Tracker should not be complete with only primary DC pre-completed", plan.tracker.isComplete());
|
||||
|
||||
// meet quorum in sat1
|
||||
for (int i = 0; i < plan.liveEndpoints.size(); i++)
|
||||
{
|
||||
InetAddressAndPort ep = plan.liveEndpoints.endpoint(i);
|
||||
if (metadata.locator.location(ep).datacenter.equals("sat1"))
|
||||
plan.tracker.onResponse(ep);
|
||||
}
|
||||
|
||||
assertTrue("Should be complete with quorum of groups", plan.tracker.isComplete());
|
||||
assertTrue("Should be successful", plan.tracker.isSuccessful());
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,394 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.InetAddress;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import org.junit.After;
|
||||
import org.junit.Before;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.DecoratedKey;
|
||||
import org.apache.cassandra.db.BufferDecoratedKey;
|
||||
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||
import org.apache.cassandra.db.marshal.AsciiType;
|
||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.distributed.test.log.ClusterMetadataTestHelper;
|
||||
import org.apache.cassandra.exceptions.UnavailableException;
|
||||
import org.apache.cassandra.gms.Gossiper;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.MessagingService;
|
||||
import org.apache.cassandra.net.NoPayload;
|
||||
import org.apache.cassandra.net.Verb;
|
||||
import org.apache.cassandra.replication.MutationId;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.paxos.Ballot;
|
||||
import org.apache.cassandra.service.paxos.Commit;
|
||||
import org.apache.cassandra.service.paxos.Paxos;
|
||||
import org.apache.cassandra.service.paxos.SatellitePaxosParticipants;
|
||||
import org.apache.cassandra.service.reads.tracked.TrackedRead;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||
import org.apache.cassandra.utils.concurrent.Future;
|
||||
|
||||
import static org.apache.cassandra.utils.ByteBufferUtil.bytes;
|
||||
import static org.junit.Assert.*;
|
||||
|
||||
public class SatellitePaxosFailoverTest extends SatelliteReplicationStrategyTestBase
|
||||
{
|
||||
private static final LongToken TOKEN = new LongToken(150);
|
||||
|
||||
@Before
|
||||
public void registerLocalNode() throws Exception
|
||||
{
|
||||
// Register the local broadcast address in dc1 so shouldRejectPaxos can resolve the local DC
|
||||
InetAddress localAddr = InetAddress.getByName("127.0.0.1");
|
||||
DatabaseDescriptor.setBroadcastAddress(localAddr);
|
||||
InetAddressAndPort localEndpoint = InetAddressAndPort.getByAddress(localAddr);
|
||||
ClusterMetadataTestHelper.register(localEndpoint, "dc1", "rack1");
|
||||
}
|
||||
|
||||
@After
|
||||
public void clearSinks()
|
||||
{
|
||||
MessagingService.instance().outboundSink.clear();
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testShouldRejectPaxosReturnsTrueDuringTransitionAck() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(
|
||||
SatelliteFailoverState.FailoverInfo.transitionAck("dc1")));
|
||||
|
||||
assertTrue("Should reject paxos during TRANSITION_ACK", strategy.shouldRejectPaxos(TOKEN));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testShouldRejectPaxosReturnsTrueWhenNotInPrimaryDC() throws Exception
|
||||
{
|
||||
// Local node is in dc1, but primary is dc2 — should reject
|
||||
createDualDCKeyspace("dc2");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
assertTrue("Should reject paxos when local node is not in primary DC", strategy.shouldRejectPaxos(TOKEN));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testShouldRejectPaxosReturnsFalseInNormalState() throws Exception
|
||||
{
|
||||
// Local node is in dc1 and dc1 is primary — should allow
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
assertFalse("Should not reject paxos in NORMAL state when in primary DC", strategy.shouldRejectPaxos(TOKEN));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testShouldRejectPaxosReturnsFalseDuringTransition() throws Exception
|
||||
{
|
||||
// Local node is in dc1 and dc1 is the new primary during TRANSITION — should allow
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(
|
||||
SatelliteFailoverState.FailoverInfo.transition("dc2")));
|
||||
|
||||
assertFalse("Should not reject paxos during TRANSITION when in primary DC", strategy.shouldRejectPaxos(TOKEN));
|
||||
}
|
||||
|
||||
private TableMetadata tableMetadata(String keyspace)
|
||||
{
|
||||
return TableMetadata.builder(keyspace, "test_table")
|
||||
.addPartitionKeyColumn("key", AsciiType.instance)
|
||||
.build();
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosParticipantsRejectedDuringTransitionAck() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc2");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(
|
||||
SatelliteFailoverState.FailoverInfo.transitionAck("dc1")));
|
||||
|
||||
try
|
||||
{
|
||||
strategy.paxosParticipants(ClusterMetadata.current(), tableMetadata(DUAL_DC_KEYSPACE),
|
||||
TOKEN, ConsistencyLevel.SERIAL, r -> true);
|
||||
fail("paxosParticipants should throw UnavailableException during TRANSITION_ACK");
|
||||
}
|
||||
catch (UnavailableException e)
|
||||
{
|
||||
// expected
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosParticipantsAllowedInNormalState() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
Paxos.Participants participants = strategy.paxosParticipants(
|
||||
ClusterMetadata.current(), tableMetadata(DUAL_DC_KEYSPACE),
|
||||
TOKEN, ConsistencyLevel.SERIAL, r -> true);
|
||||
assertNotNull(participants);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSendPaxosCommitMutationsRejectedDuringTransitionAck() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc2");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(
|
||||
SatelliteFailoverState.FailoverInfo.transitionAck("dc1")));
|
||||
|
||||
TableMetadata table = tableMetadata(DUAL_DC_KEYSPACE);
|
||||
DecoratedKey key = table.partitioner.decorateKey(bytes("test_key"));
|
||||
PartitionUpdate update = PartitionUpdate.emptyUpdate(table, key);
|
||||
Commit.Agreed commit = new Commit.Agreed(Ballot.none(), update);
|
||||
|
||||
try
|
||||
{
|
||||
strategy.sendPaxosCommitMutations(commit, false);
|
||||
fail("sendPaxosCommitMutations should throw UnavailableException during TRANSITION_ACK");
|
||||
}
|
||||
catch (UnavailableException e)
|
||||
{
|
||||
// expected
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSendPaxosCommitMutationsFailsWhenSatelliteRequestsFail() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
// Mark all non-primary (satellite/secondary) endpoints alive so they are contacted with a callback,
|
||||
// rather than being pre-marked as failed via the downEndpoints path.
|
||||
markAllEndpointsAlive();
|
||||
|
||||
// Capture and swallow the outbound satellite commit requests so no real network I/O happens; we drive
|
||||
// the failure ourselves via callback expiration below.
|
||||
List<MessageCapture> captured = new CopyOnWriteArrayList<>();
|
||||
MessagingService.instance().outboundSink.add((message, to) -> {
|
||||
if (message.verb() == Verb.PAXOS2_COMMIT_REMOTE_REQ)
|
||||
captured.add(new MessageCapture(message, to));
|
||||
return false;
|
||||
});
|
||||
|
||||
Commit.Agreed commit = agreedCommit();
|
||||
|
||||
Future<Void> future = strategy.sendPaxosCommitMutations(commit, false);
|
||||
|
||||
// A quorum of satellite endpoints must have been contacted with a callback.
|
||||
assertFalse("Expected satellite commit requests to be sent", captured.isEmpty());
|
||||
assertFalse("Future should not complete before any responses/failures", future.isDone());
|
||||
|
||||
// Simulate a request failure (timeout) for every satellite endpoint. This exercises the callback's
|
||||
// onFailure path; it only runs because invokeOnFailure() returns true.
|
||||
for (MessageCapture cap : captured)
|
||||
MessagingService.instance().callbacks.onExpired(cap.message, cap.to);
|
||||
|
||||
assertTrue("Future should fail once satellite quorum is unreachable", future.awaitUninterruptibly(30, TimeUnit.SECONDS));
|
||||
assertTrue("Future should have failed", future.isDone() && !future.isSuccess());
|
||||
assertNotNull("Failure cause should be set", future.cause());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSendPaxosCommitMutationsSucceedsWhenSatelliteRequestsRespond() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
markAllEndpointsAlive();
|
||||
|
||||
List<MessageCapture> captured = new CopyOnWriteArrayList<>();
|
||||
MessagingService.instance().outboundSink.add((message, to) -> {
|
||||
if (message.verb() == Verb.PAXOS2_COMMIT_REMOTE_REQ)
|
||||
captured.add(new MessageCapture(message, to));
|
||||
return false;
|
||||
});
|
||||
|
||||
Commit.Agreed commit = agreedCommit();
|
||||
|
||||
Future<Void> future = strategy.sendPaxosCommitMutations(commit, false);
|
||||
|
||||
assertFalse("Expected satellite commit requests to be sent", captured.isEmpty());
|
||||
|
||||
// Deliver a successful response for every satellite endpoint via the registered callback.
|
||||
for (MessageCapture cap : captured)
|
||||
{
|
||||
Message<NoPayload> response = Message.internalResponse(Verb.PAXOS2_COMMIT_REMOTE_RSP, NoPayload.noPayload)
|
||||
.withFrom(cap.to);
|
||||
MessagingService.instance().callbacks.removeAndRespond(cap.message.id(), cap.to, response);
|
||||
}
|
||||
|
||||
assertTrue("Future should complete once satellite quorum responds",
|
||||
future.awaitUninterruptibly(30, TimeUnit.SECONDS));
|
||||
assertTrue("Future should have succeeded", future.isSuccess());
|
||||
}
|
||||
|
||||
private Commit.Agreed agreedCommit()
|
||||
{
|
||||
TableMetadata table = tableMetadata(DUAL_DC_KEYSPACE);
|
||||
// Pin the key to TOKEN (150) so it maps to the configured ring ranges set up by the test base.
|
||||
DecoratedKey key = new BufferDecoratedKey(TOKEN, bytes("test_key"));
|
||||
PartitionUpdate update = PartitionUpdate.emptyUpdate(table, key);
|
||||
// A non-none mutation id is required by sendPaxosCommitMutations (it registers with mutation tracking).
|
||||
return new Commit.Agreed(Ballot.none(), update).withMutationId(new MutationId(1L, 1L));
|
||||
}
|
||||
|
||||
private void markAllEndpointsAlive()
|
||||
{
|
||||
for (InetAddressAndPort endpoint : ClusterMetadata.current().directory.allAddresses())
|
||||
Gossiper.instance.initializeNodeUnsafe(endpoint, UUID.randomUUID(), 1);
|
||||
}
|
||||
|
||||
private SatellitePaxosParticipants getParticipants(String keyspace) throws Exception
|
||||
{
|
||||
SatelliteReplicationStrategy strategy = getSRS(keyspace);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
Paxos.Participants participants = strategy.paxosParticipants(
|
||||
metadata, tableMetadata(keyspace), TOKEN, ConsistencyLevel.SERIAL, r -> true);
|
||||
assertTrue("SRS should return SatellitePaxosParticipants",
|
||||
participants instanceof SatellitePaxosParticipants);
|
||||
return (SatellitePaxosParticipants) participants;
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosParticipantsReturnsSatelliteEndpoints() throws Exception
|
||||
{
|
||||
// Dual DC with dc1 primary: satellite endpoints should include sat1 (dc1's satellite) and dc2 (other full DC)
|
||||
createDualDCKeyspace("dc1");
|
||||
SatellitePaxosParticipants spp = getParticipants(DUAL_DC_KEYSPACE);
|
||||
|
||||
EndpointsForToken satelliteEndpoints = spp.getAdditionalSummaryEndpoints();
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
Set<String> dcs = replicaDCs(satelliteEndpoints, metadata);
|
||||
|
||||
assertTrue("Should include sat1 (primary's satellite)", dcs.contains("sat1"));
|
||||
assertTrue("Should include dc2 (other full DC)", dcs.contains("dc2"));
|
||||
assertFalse("Should not include dc1 (primary DC)", dcs.contains("dc1"));
|
||||
assertFalse("Should not include sat2 (other DC's satellite)", dcs.contains("sat2"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosParticipantsSingleDCHasSatelliteOnly() throws Exception
|
||||
{
|
||||
createSingleDCKeyspace();
|
||||
SatellitePaxosParticipants spp = getParticipants(SINGLE_DC_KEYSPACE);
|
||||
|
||||
EndpointsForToken satelliteEndpoints = spp.getAdditionalSummaryEndpoints();
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
Set<String> dcs = replicaDCs(satelliteEndpoints, metadata);
|
||||
|
||||
assertTrue("Should include sat1", dcs.contains("sat1"));
|
||||
assertEquals("Should only have sat1", 1, dcs.size());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAdditionalSummaryHostIdsMatchesSatelliteEndpoints() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatellitePaxosParticipants spp = getParticipants(DUAL_DC_KEYSPACE);
|
||||
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
EndpointsForToken satelliteEndpoints = spp.getAdditionalSummaryEndpoints();
|
||||
int[] additionalIds = spp.additionalSummaryHostIds(metadata);
|
||||
|
||||
assertEquals(satelliteEndpoints.size(), additionalIds.length);
|
||||
for (int i = 0; i < satelliteEndpoints.size(); i++)
|
||||
{
|
||||
int expectedId = metadata.directory.peerId(satelliteEndpoints.endpoint(i)).id();
|
||||
assertEquals(expectedId, additionalIds[i]);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testOnPrepareStartedSendsSummaryRequestToSatellites() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatellitePaxosParticipants spp = getParticipants(DUAL_DC_KEYSPACE);
|
||||
|
||||
List<MessageCapture> captured = new CopyOnWriteArrayList<>();
|
||||
MessagingService.instance().outboundSink.add((message, to) -> {
|
||||
captured.add(new MessageCapture(message, to));
|
||||
return false;
|
||||
});
|
||||
|
||||
TrackedRead.Id readId = new TrackedRead.Id(1, 100L);
|
||||
TableMetadata table = tableMetadata(DUAL_DC_KEYSPACE);
|
||||
SinglePartitionReadCommand readCommand = SinglePartitionReadCommand.fullPartitionRead(table, 0, ByteBufferUtil.bytes(0));
|
||||
|
||||
spp.onPrepareStarted(readId, 42, new int[] { 1, 2, 3 }, readCommand);
|
||||
|
||||
EndpointsForToken satelliteEndpoints = spp.getAdditionalSummaryEndpoints();
|
||||
assertEquals(satelliteEndpoints.size(), captured.size());
|
||||
|
||||
Set<InetAddressAndPort> sentTo = captured.stream().map(c -> c.to).collect(Collectors.toSet());
|
||||
for (MessageCapture cap : captured)
|
||||
assertEquals(Verb.TRACKED_SUMMARY_REQ, cap.message.verb());
|
||||
for (int i = 0; i < satelliteEndpoints.size(); i++)
|
||||
assertTrue("Should send to satellite endpoint", sentTo.contains(satelliteEndpoints.endpoint(i)));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testOnPrepareStartedNoOpWhenReadCommandNull() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatellitePaxosParticipants spp = getParticipants(DUAL_DC_KEYSPACE);
|
||||
|
||||
List<MessageCapture> captured = new CopyOnWriteArrayList<>();
|
||||
MessagingService.instance().outboundSink.add((message, to) -> {
|
||||
captured.add(new MessageCapture(message, to));
|
||||
return false;
|
||||
});
|
||||
|
||||
spp.onPrepareStarted(new TrackedRead.Id(1, 100L), 42, new int[] { 1, 2, 3 }, null);
|
||||
|
||||
assertEquals(0, captured.size());
|
||||
}
|
||||
|
||||
private static class MessageCapture
|
||||
{
|
||||
final Message<?> message;
|
||||
final InetAddressAndPort to;
|
||||
|
||||
MessageCapture(Message<?> message, InetAddressAndPort to)
|
||||
{
|
||||
this.message = message;
|
||||
this.to = to;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,291 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.Arrays;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import org.junit.Assume;
|
||||
import org.junit.Test;
|
||||
import org.junit.runner.RunWith;
|
||||
import org.junit.runners.Parameterized;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.dht.Range;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.service.reads.AlwaysSpeculativeRetryPolicy;
|
||||
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
|
||||
import static org.junit.Assert.assertEquals;
|
||||
import static org.junit.Assert.assertFalse;
|
||||
import static org.junit.Assert.assertTrue;
|
||||
import static org.junit.Assert.fail;
|
||||
|
||||
@RunWith(Parameterized.class)
|
||||
public class SatelliteReadPlanTest extends SatelliteReplicationStrategyTestBase
|
||||
{
|
||||
enum ReadType
|
||||
{
|
||||
TOKEN, RANGE
|
||||
}
|
||||
|
||||
@Parameterized.Parameters(name = "{0}/{1}")
|
||||
public static Collection<Object[]> params()
|
||||
{
|
||||
return Arrays.asList(new Object[][] {
|
||||
{ ReadType.TOKEN, SatelliteFailoverState.State.NORMAL },
|
||||
{ ReadType.TOKEN, SatelliteFailoverState.State.TRANSITION_ACK },
|
||||
{ ReadType.TOKEN, SatelliteFailoverState.State.TRANSITION },
|
||||
{ ReadType.RANGE, SatelliteFailoverState.State.NORMAL },
|
||||
{ ReadType.RANGE, SatelliteFailoverState.State.TRANSITION_ACK },
|
||||
{ ReadType.RANGE, SatelliteFailoverState.State.TRANSITION },
|
||||
});
|
||||
}
|
||||
|
||||
private final ReadType readType;
|
||||
private final SatelliteFailoverState.State failoverState;
|
||||
|
||||
public SatelliteReadPlanTest(ReadType readType, SatelliteFailoverState.State failoverState)
|
||||
{
|
||||
this.readType = readType;
|
||||
this.failoverState = failoverState;
|
||||
}
|
||||
|
||||
private boolean isTransition()
|
||||
{
|
||||
return failoverState != SatelliteFailoverState.State.NORMAL;
|
||||
}
|
||||
|
||||
private SatelliteFailoverState.FailoverInfo failoverInfo()
|
||||
{
|
||||
switch (failoverState)
|
||||
{
|
||||
case TRANSITION_ACK: return SatelliteFailoverState.FailoverInfo.transitionAck("dc1");
|
||||
case TRANSITION: return SatelliteFailoverState.FailoverInfo.transition("dc1");
|
||||
default: throw new IllegalStateException("No failover info for NORMAL");
|
||||
}
|
||||
}
|
||||
|
||||
private CoordinationPlan.ForRead<?, ?> createPlan(SatelliteReplicationStrategy strategy, String keyspaceName) throws Exception
|
||||
{
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
KeyspaceMetadata ksm = metadata.schema.getKeyspaces().getNullable(keyspaceName);
|
||||
Keyspace keyspace = Keyspace.mockKS(ksm);
|
||||
|
||||
switch (readType)
|
||||
{
|
||||
case TOKEN:
|
||||
return strategy.planForTokenRead(metadata, keyspace, TABLE_ID,
|
||||
new LongToken(150), null,
|
||||
ConsistencyLevel.QUORUM,
|
||||
AlwaysSpeculativeRetryPolicy.INSTANCE,
|
||||
ReadCoordinator.DEFAULT);
|
||||
case RANGE:
|
||||
return strategy.planForRangeRead(metadata, keyspace, TABLE_ID, null,
|
||||
ConsistencyLevel.QUORUM,
|
||||
Range.makeRowRange(new LongToken(100),
|
||||
new LongToken(200)),
|
||||
1);
|
||||
default:
|
||||
throw new IllegalStateException();
|
||||
}
|
||||
}
|
||||
|
||||
private ReplicaPlan.ForRead<?, ?> replicas(CoordinationPlan.ForRead<?, ?> plan)
|
||||
{
|
||||
return (ReplicaPlan.ForRead<?, ?>) plan.replicas();
|
||||
}
|
||||
|
||||
private void assertNoDuplicateEndpoints(String label, Iterable<Replica> replicas)
|
||||
{
|
||||
Set<InetAddressAndPort> seen = new HashSet<>();
|
||||
for (Replica r : replicas)
|
||||
assertTrue(label + " contains duplicate endpoint: " + r.endpoint(),
|
||||
seen.add(r.endpoint()));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testReadPlanDualDC() throws Exception
|
||||
{
|
||||
createDualDCKeyspace(isTransition() ? "dc2" : "dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
if (isTransition())
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(failoverInfo()));
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
|
||||
Set<String> contactDCs = replicaDCs(plan.replicas().contacts(), metadata);
|
||||
if (isTransition())
|
||||
{
|
||||
assertTrue("Should include dc2 (new primary)", contactDCs.contains("dc2"));
|
||||
assertTrue("Should include dc1 (old primary)", contactDCs.contains("dc1"));
|
||||
}
|
||||
else
|
||||
{
|
||||
assertTrue("Should include dc1", contactDCs.contains("dc1"));
|
||||
assertTrue("Should include sat1 (dc1's satellite)", contactDCs.contains("sat1"));
|
||||
assertFalse("Should NOT include sat2 (dc2's satellite)", contactDCs.contains("sat2"));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testReadPlanSingleDC() throws Exception
|
||||
{
|
||||
createSingleDCKeyspace();
|
||||
SatelliteReplicationStrategy strategy = getSRS(SINGLE_DC_KEYSPACE);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, SINGLE_DC_KEYSPACE);
|
||||
|
||||
Set<String> contactDCs = replicaDCs(plan.replicas().contacts(), metadata);
|
||||
assertTrue("Should include dc1", contactDCs.contains("dc1"));
|
||||
assertTrue("Should include sat1", contactDCs.contains("sat1"));
|
||||
assertFalse("Should NOT include dc2", contactDCs.contains("dc2"));
|
||||
assertFalse("Should NOT include sat2", contactDCs.contains("sat2"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testReadPlanExcludesDisabledDC() throws Exception
|
||||
{
|
||||
createDisabledDCKeyspace();
|
||||
SatelliteReplicationStrategy strategy = getSRS(DISABLED_DC_KEYSPACE);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, DISABLED_DC_KEYSPACE);
|
||||
|
||||
Set<String> contactDCs = replicaDCs(plan.replicas().contacts(), metadata);
|
||||
assertTrue("Should include dc1", contactDCs.contains("dc1"));
|
||||
assertTrue("Should include sat1 (dc1's satellite)", contactDCs.contains("sat1"));
|
||||
assertFalse("Should NOT include dc2 (disabled)", contactDCs.contains("dc2"));
|
||||
assertFalse("Should NOT include sat2 (disabled dc2's satellite)", contactDCs.contains("sat2"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testReadPlanPrimaryDCFirst() throws Exception
|
||||
{
|
||||
createDualDCKeyspace(isTransition() ? "dc2" : "dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
if (isTransition())
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(failoverInfo()));
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
|
||||
String expectedPrimary = isTransition() ? "dc2" : "dc1";
|
||||
Replica first = plan.replicas().contacts().iterator().next();
|
||||
assertEquals("First contact should be in primary DC",
|
||||
expectedPrimary, metadata.locator.location(first.endpoint()).datacenter);
|
||||
|
||||
if (isTransition())
|
||||
{
|
||||
// dc2 contacts should appear before dc1 contacts
|
||||
boolean seenDc1 = false;
|
||||
for (Replica r : plan.replicas().contacts())
|
||||
{
|
||||
String dc = metadata.locator.location(r.endpoint()).datacenter;
|
||||
if (dc.equals("dc1"))
|
||||
seenDc1 = true;
|
||||
if (dc.equals("dc2") && seenDc1)
|
||||
fail("dc2 contact appeared after dc1 contact — primary should come first");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testReadPlanNoMerge() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
|
||||
Set<String> candidateDCs = replicaDCs(replicas(plan).readCandidates(), metadata);
|
||||
assertTrue("Should have dc1 candidates", candidateDCs.contains("dc1"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testTransitionReadPlanMergesDCs() throws Exception
|
||||
{
|
||||
Assume.assumeTrue(isTransition());
|
||||
|
||||
createDualDCKeyspace("dc2");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(failoverInfo()));
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
|
||||
Set<String> candidateDCs = replicaDCs(replicas(plan).readCandidates(), metadata);
|
||||
assertTrue("Merged candidates should include dc2", candidateDCs.contains("dc2"));
|
||||
assertTrue("Merged candidates should include dc1", candidateDCs.contains("dc1"));
|
||||
|
||||
Set<String> liveAndDownDCs = replicaDCs(plan.replicas().liveAndDown(), metadata);
|
||||
assertTrue("Merged liveAndDown should include dc2", liveAndDownDCs.contains("dc2"));
|
||||
assertTrue("Merged liveAndDown should include dc1", liveAndDownDCs.contains("dc1"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testTransitionReadPlanNoDuplicates() throws Exception
|
||||
{
|
||||
Assume.assumeTrue(isTransition());
|
||||
|
||||
createDualDCKeyspace("dc2");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(failoverInfo()));
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> plan = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
ReplicaPlan.ForRead<?, ?> replicas = replicas(plan);
|
||||
|
||||
assertNoDuplicateEndpoints("contacts", replicas.contacts());
|
||||
assertNoDuplicateEndpoints("candidates", replicas.readCandidates());
|
||||
assertNoDuplicateEndpoints("liveAndDown", replicas.liveAndDown());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testTransitionReadPlanQuorum() throws Exception
|
||||
{
|
||||
Assume.assumeTrue(isTransition());
|
||||
|
||||
createDualDCKeyspace("dc2");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
|
||||
// Get the individual plan quorum before merging
|
||||
CoordinationPlan.ForRead<?, ?> primaryOnly = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
int primaryQuorum = replicas(primaryOnly).readQuorum();
|
||||
|
||||
// Now set transition state and get merged plan
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(failoverInfo()));
|
||||
|
||||
CoordinationPlan.ForRead<?, ?> merged = createPlan(strategy, DUAL_DC_KEYSPACE);
|
||||
int mergedQuorum = replicas(merged).readQuorum();
|
||||
|
||||
assertTrue("Merged quorum (" + mergedQuorum + ") should be >= primary quorum (" + primaryQuorum + ")",
|
||||
mergedQuorum >= primaryQuorum);
|
||||
}
|
||||
}
|
||||
|
|
@ -17,87 +17,29 @@
|
|||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.UnknownHostException;
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
|
||||
import org.junit.After;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.CassandraTestBase;
|
||||
import org.apache.cassandra.CassandraTestBase.UseMurmur3Partitioner;
|
||||
import org.apache.cassandra.ServerTestUtils;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.config.Config;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.distributed.test.log.ClusterMetadataTestHelper;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.schema.ReplicationType;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.membership.Location;
|
||||
|
||||
import static org.apache.cassandra.CassandraTestBase.DisableMBeanRegistration;
|
||||
import static org.apache.cassandra.CassandraTestBase.PrepareServerNoRegister;
|
||||
import static org.junit.Assert.assertEquals;
|
||||
import static org.junit.Assert.assertFalse;
|
||||
import static org.junit.Assert.assertTrue;
|
||||
import static org.junit.Assert.fail;
|
||||
|
||||
@PrepareServerNoRegister
|
||||
@DisableMBeanRegistration
|
||||
@UseMurmur3Partitioner
|
||||
public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
||||
public class SatelliteReplicationStrategyTest extends SatelliteReplicationStrategyTestBase
|
||||
{
|
||||
private static final String KEYSPACE = "test";
|
||||
|
||||
@After
|
||||
public void teardown()
|
||||
{
|
||||
ServerTestUtils.resetCMS();
|
||||
}
|
||||
|
||||
private void addToken(long token, String address, Location location) throws UnknownHostException
|
||||
{
|
||||
InetAddressAndPort addr = InetAddressAndPort.getByName(address);
|
||||
ClusterMetadataTestHelper.addEndpoint(addr, new LongToken(token), location);
|
||||
}
|
||||
|
||||
private void setupDCs() throws UnknownHostException
|
||||
{
|
||||
Location dc1 = new Location("dc1", "rack1");
|
||||
Location dc2 = new Location("dc2", "rack1");
|
||||
Location sat1 = new Location("sat1", "rack1");
|
||||
Location sat2 = new Location("sat2", "rack1");
|
||||
|
||||
// DC1
|
||||
addToken(100, "10.0.0.10", dc1);
|
||||
addToken(200, "10.0.0.11", dc1);
|
||||
addToken(300, "10.0.0.12", dc1);
|
||||
|
||||
// DC2
|
||||
addToken(400, "10.1.0.10", dc2);
|
||||
addToken(500, "10.1.0.11", dc2);
|
||||
addToken(600, "10.1.0.12", dc2);
|
||||
|
||||
// SAT1
|
||||
addToken(700, "10.2.0.10", sat1);
|
||||
addToken(800, "10.2.0.11", sat1);
|
||||
|
||||
// SAT2
|
||||
addToken(900, "10.3.0.10", sat2);
|
||||
addToken(1000, "10.3.0.11", sat2);
|
||||
}
|
||||
|
||||
private static SatelliteReplicationStrategy getSRS(String keyspace)
|
||||
{
|
||||
KeyspaceMetadata ksm = ClusterMetadata.current().schema.getKeyspaces().getNullable(keyspace);
|
||||
return (SatelliteReplicationStrategy) ksm.replicationStrategy;
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testValidSingleDCWithSatellite() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
String cql = "CREATE KEYSPACE " + KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
|
|
@ -119,8 +61,6 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
@Test
|
||||
public void testValidMultipleDCsWithSatellites() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
String cql = "CREATE KEYSPACE " + KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
|
|
@ -139,10 +79,8 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
assertEquals(2, strategy.getSatellites().size());
|
||||
}
|
||||
|
||||
private void testConfigurationException(Map<String, String> options, String messageContains) throws UnknownHostException
|
||||
private void testConfigurationException(Map<String, String> options, String messageContains)
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
try
|
||||
{
|
||||
new SatelliteReplicationStrategy(KEYSPACE, options, ReplicationType.tracked);
|
||||
|
|
@ -176,8 +114,6 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
@Test
|
||||
public void testUntrackedReplicationFails() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
Map<String, String> options = new HashMap<>();
|
||||
options.put("dc1", "3");
|
||||
options.put("primary", "dc1");
|
||||
|
|
@ -196,6 +132,33 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosV1Fails() throws Exception
|
||||
{
|
||||
Map<String, String> options = new HashMap<>();
|
||||
options.put("dc1", "3");
|
||||
options.put("primary", "dc1");
|
||||
|
||||
SatelliteReplicationStrategy strategy = new SatelliteReplicationStrategy(
|
||||
KEYSPACE, options, ReplicationType.tracked);
|
||||
|
||||
Config.PaxosVariant prev = DatabaseDescriptor.getPaxosVariant();
|
||||
try
|
||||
{
|
||||
DatabaseDescriptor.setPaxosVariant(Config.PaxosVariant.v1);
|
||||
strategy.validateExpectedOptions(ClusterMetadata.current());
|
||||
fail("ConfigurationException expected");
|
||||
}
|
||||
catch (ConfigurationException e)
|
||||
{
|
||||
assertTrue(e.getMessage().contains("requires paxos_variant=v2"));
|
||||
}
|
||||
finally
|
||||
{
|
||||
DatabaseDescriptor.setPaxosVariant(prev);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testDotsInDCNamesFails() throws Exception
|
||||
{
|
||||
|
|
@ -242,12 +205,10 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
@Test
|
||||
public void testReplicaCalculationWithSatellites() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
String cql = "CREATE KEYSPACE " + KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
"'dc1.satellite.sat1': '2/2', " +
|
||||
"'dc1.satellite.sat1': '3/3', " +
|
||||
"'primary': 'dc1'" +
|
||||
"} AND replication_type = 'tracked'";
|
||||
|
||||
|
|
@ -258,8 +219,8 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
EndpointsForRange replicas = strategy.calculateNaturalReplicas(
|
||||
new LongToken(150), ClusterMetadata.current());
|
||||
|
||||
// Should have 3 full replicas from dc1 + 2 satellite replicas from sat1
|
||||
assertEquals(5, replicas.size());
|
||||
// Should have 3 full replicas from dc1 + 3 satellite replicas from sat1
|
||||
assertEquals(6, replicas.size());
|
||||
|
||||
int fullCount = 0;
|
||||
int witnessCount = 0;
|
||||
|
|
@ -272,14 +233,12 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
}
|
||||
|
||||
assertEquals(3, fullCount);
|
||||
assertEquals(2, witnessCount);
|
||||
assertEquals(3, witnessCount);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testDisableNonPrimaryDC() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
Map<String, String> options = new HashMap<>();
|
||||
options.put("dc1", "3");
|
||||
options.put("dc2", "3");
|
||||
|
|
@ -331,14 +290,12 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
@Test
|
||||
public void testDisabledDCSatelliteStillGetsReplicas() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
String cql = "CREATE KEYSPACE " + KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
"'dc1.satellite.sat1': '2/2', " +
|
||||
"'dc1.satellite.sat1': '3/3', " +
|
||||
"'dc2': '3', " +
|
||||
"'dc2.satellite.sat2': '2/2', " +
|
||||
"'dc2.satellite.sat2': '3/3', " +
|
||||
"'dc2.disabled': 'true', " +
|
||||
"'primary': 'dc1'" +
|
||||
"} AND replication_type = 'tracked'";
|
||||
|
|
@ -351,8 +308,8 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
new LongToken(150), ClusterMetadata.current());
|
||||
|
||||
// Disabled does not affect placement — all DCs and satellites still get replicas
|
||||
// 3 full from dc1 + 3 full from dc2 + 2 witness from sat1 + 2 witness from sat2
|
||||
assertEquals(10, replicas.size());
|
||||
// 3 full from dc1 + 3 full from dc2 + 3 witness from sat1 + 3 witness from sat2
|
||||
assertEquals(12, replicas.size());
|
||||
}
|
||||
|
||||
@Test
|
||||
|
|
@ -370,8 +327,6 @@ public class SatelliteReplicationStrategyTest extends CassandraTestBase
|
|||
@Test
|
||||
public void testHasSameSettingsWithDisabled() throws Exception
|
||||
{
|
||||
setupDCs();
|
||||
|
||||
Map<String, String> optionsA = new HashMap<>();
|
||||
optionsA.put("dc1", "3");
|
||||
optionsA.put("dc2", "3");
|
||||
|
|
|
|||
|
|
@ -0,0 +1,166 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.UnknownHostException;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import org.junit.After;
|
||||
import org.junit.Before;
|
||||
import org.junit.BeforeClass;
|
||||
|
||||
import org.apache.cassandra.ServerTestUtils;
|
||||
import org.apache.cassandra.config.Config;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner;
|
||||
import org.apache.cassandra.distributed.test.log.ClusterMetadataTestHelper;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
import org.apache.cassandra.service.StorageService;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.membership.Location;
|
||||
|
||||
import static org.apache.cassandra.config.CassandraRelevantProperties.ORG_APACHE_CASSANDRA_DISABLE_MBEAN_REGISTRATION;
|
||||
|
||||
public abstract class SatelliteReplicationStrategyTestBase
|
||||
{
|
||||
protected static final String KEYSPACE = "test";
|
||||
protected static final TableId TABLE_ID = TableId.generate();
|
||||
protected static final String DUAL_DC_KEYSPACE = "dual_dc_test";
|
||||
protected static final String SINGLE_DC_KEYSPACE = "single_dc_test";
|
||||
protected static final String DISABLED_DC_KEYSPACE = "disabled_dc_test";
|
||||
|
||||
@BeforeClass
|
||||
public static void setUpClass()
|
||||
{
|
||||
ORG_APACHE_CASSANDRA_DISABLE_MBEAN_REGISTRATION.setBoolean(true);
|
||||
ServerTestUtils.daemonInitialization();
|
||||
StorageService.instance.setPartitionerUnsafe(Murmur3Partitioner.instance);
|
||||
DatabaseDescriptor.setPaxosVariant(Config.PaxosVariant.v2);
|
||||
ServerTestUtils.prepareServerNoRegister();
|
||||
}
|
||||
|
||||
@Before
|
||||
public void setup() throws UnknownHostException
|
||||
{
|
||||
setupDCs();
|
||||
}
|
||||
|
||||
@After
|
||||
public void teardown()
|
||||
{
|
||||
ServerTestUtils.resetCMS();
|
||||
}
|
||||
|
||||
private void addToken(long token, String address, Location location) throws UnknownHostException
|
||||
{
|
||||
InetAddressAndPort addr = InetAddressAndPort.getByName(address);
|
||||
ClusterMetadataTestHelper.addEndpoint(addr, new LongToken(token), location);
|
||||
}
|
||||
|
||||
private void setupDCs() throws UnknownHostException
|
||||
{
|
||||
Location dc1 = new Location("dc1", "rack1");
|
||||
Location dc2 = new Location("dc2", "rack1");
|
||||
Location sat1 = new Location("sat1", "rack1");
|
||||
Location sat2 = new Location("sat2", "rack1");
|
||||
|
||||
// DC1
|
||||
addToken(100, "10.0.0.10", dc1);
|
||||
addToken(200, "10.0.0.11", dc1);
|
||||
addToken(300, "10.0.0.12", dc1);
|
||||
|
||||
// DC2
|
||||
addToken(400, "10.1.0.10", dc2);
|
||||
addToken(500, "10.1.0.11", dc2);
|
||||
addToken(600, "10.1.0.12", dc2);
|
||||
|
||||
// SAT1
|
||||
addToken(700, "10.2.0.10", sat1);
|
||||
addToken(800, "10.2.0.11", sat1);
|
||||
addToken(900, "10.2.0.12", sat1);
|
||||
|
||||
// SAT2
|
||||
addToken(1000, "10.3.0.10", sat2);
|
||||
addToken(1100, "10.3.0.11", sat2);
|
||||
addToken(1200, "10.3.0.12", sat2);
|
||||
}
|
||||
|
||||
protected static SatelliteReplicationStrategy getSRS(String keyspace)
|
||||
{
|
||||
KeyspaceMetadata ksm = ClusterMetadata.current().schema.getKeyspaces().getNullable(keyspace);
|
||||
return (SatelliteReplicationStrategy) ksm.replicationStrategy;
|
||||
}
|
||||
|
||||
protected void createDualDCKeyspace(String primary) throws Exception
|
||||
{
|
||||
String cql = "CREATE KEYSPACE " + DUAL_DC_KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
"'dc1.satellite.sat1': '3/3', " +
|
||||
"'dc2': '3', " +
|
||||
"'dc2.satellite.sat2': '3/3', " +
|
||||
"'primary': '" + primary + "'" +
|
||||
"} AND replication_type = 'tracked'";
|
||||
ClusterMetadataTestHelper.createKeyspace(cql);
|
||||
}
|
||||
|
||||
protected void createSingleDCKeyspace() throws Exception
|
||||
{
|
||||
String cql = "CREATE KEYSPACE " + SINGLE_DC_KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
"'dc1.satellite.sat1': '3/3', " +
|
||||
"'primary': 'dc1'" +
|
||||
"} AND replication_type = 'tracked'";
|
||||
ClusterMetadataTestHelper.createKeyspace(cql);
|
||||
}
|
||||
|
||||
protected void createDisabledDCKeyspace() throws Exception
|
||||
{
|
||||
String cql = "CREATE KEYSPACE " + DISABLED_DC_KEYSPACE + " WITH replication = {" +
|
||||
"'class': 'SatelliteReplicationStrategy', " +
|
||||
"'dc1': '3', " +
|
||||
"'dc1.satellite.sat1': '3/3', " +
|
||||
"'dc2': '3', " +
|
||||
"'dc2.satellite.sat2': '3/3', " +
|
||||
"'dc2.disabled': 'true', " +
|
||||
"'primary': 'dc1'" +
|
||||
"} AND replication_type = 'tracked'";
|
||||
ClusterMetadataTestHelper.createKeyspace(cql);
|
||||
}
|
||||
|
||||
protected Set<String> replicaDCs(Iterable<Replica> replicas, ClusterMetadata metadata)
|
||||
{
|
||||
Set<String> dcs = new HashSet<>();
|
||||
for (Replica r : replicas)
|
||||
dcs.add(metadata.locator.location(r.endpoint()).datacenter);
|
||||
return dcs;
|
||||
}
|
||||
|
||||
protected Set<InetAddressAndPort> replicasInDC(Iterable<Replica> replicas, String dc, ClusterMetadata metadata)
|
||||
{
|
||||
Set<InetAddressAndPort> eps = new HashSet<>();
|
||||
for (Replica r : replicas)
|
||||
if (metadata.locator.location(r.endpoint()).datacenter.equals(dc))
|
||||
eps.add(r.endpoint());
|
||||
return eps;
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,153 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.util.Arrays;
|
||||
import java.util.Collection;
|
||||
import java.util.Set;
|
||||
|
||||
import org.junit.Test;
|
||||
import org.junit.runner.RunWith;
|
||||
import org.junit.runners.Parameterized;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner.LongToken;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
|
||||
import static org.junit.Assert.assertFalse;
|
||||
import static org.junit.Assert.assertTrue;
|
||||
|
||||
@RunWith(Parameterized.class)
|
||||
public class SatelliteWritePlanTest extends SatelliteReplicationStrategyTestBase
|
||||
{
|
||||
@Parameterized.Parameters(name = "{0}")
|
||||
public static Collection<Object[]> params()
|
||||
{
|
||||
return Arrays.asList(new Object[][] {
|
||||
{ SatelliteFailoverState.State.NORMAL },
|
||||
{ SatelliteFailoverState.State.TRANSITION_ACK },
|
||||
{ SatelliteFailoverState.State.TRANSITION },
|
||||
});
|
||||
}
|
||||
|
||||
private final SatelliteFailoverState.State failoverState;
|
||||
|
||||
public SatelliteWritePlanTest(SatelliteFailoverState.State failoverState)
|
||||
{
|
||||
this.failoverState = failoverState;
|
||||
}
|
||||
|
||||
private boolean isTransition()
|
||||
{
|
||||
return failoverState != SatelliteFailoverState.State.NORMAL;
|
||||
}
|
||||
|
||||
private void applyFailoverState(SatelliteReplicationStrategy strategy)
|
||||
{
|
||||
if (!isTransition())
|
||||
return;
|
||||
|
||||
SatelliteFailoverState.FailoverInfo info;
|
||||
switch (failoverState)
|
||||
{
|
||||
case TRANSITION_ACK: info = SatelliteFailoverState.FailoverInfo.transitionAck("dc1"); break;
|
||||
case TRANSITION: info = SatelliteFailoverState.FailoverInfo.transition("dc1"); break;
|
||||
default: throw new IllegalStateException();
|
||||
}
|
||||
strategy.setFailoverState(SatelliteFailoverState.FailoverStateMap.allRanges(info));
|
||||
}
|
||||
|
||||
private CoordinationPlan.ForWrite callPlanForWrite(SatelliteReplicationStrategy strategy,
|
||||
String keyspaceName, Token token)
|
||||
{
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
KeyspaceMetadata ksm = metadata.schema.getKeyspaces().getNullable(keyspaceName);
|
||||
Keyspace keyspace = Keyspace.mockKS(ksm);
|
||||
return strategy.planForWriteInternal(metadata, keyspace, ConsistencyLevel.QUORUM,
|
||||
(cm) -> ReplicaLayout.forTokenWriteLiveAndDown(cm, keyspace, token),
|
||||
ReplicaPlans.writeAll);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testWriteContactsExcludeOtherSatellite() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
applyFailoverState(strategy);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForWrite plan = callPlanForWrite(strategy, DUAL_DC_KEYSPACE, new LongToken(150));
|
||||
|
||||
Set<String> dcs = replicaDCs(plan.replicas().contacts(), metadata);
|
||||
assertTrue("Should include dc1", dcs.contains("dc1"));
|
||||
assertTrue("Should include dc2", dcs.contains("dc2"));
|
||||
assertTrue("Should include sat1 (dc1's satellite)", dcs.contains("sat1"));
|
||||
assertFalse("Should NOT include sat2 (dc2's satellite)", dcs.contains("sat2"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testWriteContactsExcludeDisabledDC() throws Exception
|
||||
{
|
||||
createDisabledDCKeyspace();
|
||||
SatelliteReplicationStrategy strategy = getSRS(DISABLED_DC_KEYSPACE);
|
||||
applyFailoverState(strategy);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForWrite plan = callPlanForWrite(strategy, DISABLED_DC_KEYSPACE, new LongToken(150));
|
||||
|
||||
Set<String> dcs = replicaDCs(plan.replicas().contacts(), metadata);
|
||||
assertTrue("Should include dc1", dcs.contains("dc1"));
|
||||
assertTrue("Should include sat1 (dc1's satellite)", dcs.contains("sat1"));
|
||||
assertFalse("Should NOT include dc2 (disabled)", dcs.contains("dc2"));
|
||||
assertFalse("Should NOT include sat2 (disabled dc2's satellite)", dcs.contains("sat2"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testWriteTrackerComposition() throws Exception
|
||||
{
|
||||
createDualDCKeyspace("dc1");
|
||||
SatelliteReplicationStrategy strategy = getSRS(DUAL_DC_KEYSPACE);
|
||||
applyFailoverState(strategy);
|
||||
ClusterMetadata metadata = ClusterMetadata.current();
|
||||
|
||||
CoordinationPlan.ForWrite plan = callPlanForWrite(strategy, DUAL_DC_KEYSPACE, new LongToken(150));
|
||||
ResponseTracker tracker = plan.responses();
|
||||
|
||||
assertTrue("Write should use CompositeTracker",
|
||||
tracker instanceof CompositeTracker);
|
||||
|
||||
Set<InetAddressAndPort> dc1Contacts = replicasInDC(plan.replicas().contacts(), "dc1", metadata);
|
||||
Set<InetAddressAndPort> sat1Contacts = replicasInDC(plan.replicas().contacts(), "sat1", metadata);
|
||||
|
||||
int count = 0;
|
||||
for (InetAddressAndPort ep : dc1Contacts)
|
||||
{
|
||||
tracker.onResponse(ep);
|
||||
if (++count >= 2) break;
|
||||
}
|
||||
assertFalse("dc1 quorum alone should not suffice (1 of 3 groups)", tracker.isSuccessful());
|
||||
|
||||
for (InetAddressAndPort ep : sat1Contacts)
|
||||
tracker.onResponse(ep);
|
||||
|
||||
assertTrue("Should succeed with primary + satellite quorums (2 of 3 groups)", tracker.isSuccessful());
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,361 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.UnknownHostException;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
import org.junit.Test;
|
||||
|
||||
import static org.junit.Assert.*;
|
||||
|
||||
public class SimpleResponseTrackerTest
|
||||
{
|
||||
private InetAddressAndPort endpoint(String ip) throws UnknownHostException
|
||||
{
|
||||
return InetAddressAndPort.getByName(ip);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testQuorumReached() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 3);
|
||||
|
||||
assertFalse(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertEquals(0, tracker.received());
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
assertFalse(tracker.isComplete());
|
||||
assertEquals(1, tracker.received());
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.received());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPartialProgress() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(3, 5);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
|
||||
assertFalse(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.received());
|
||||
assertEquals(3, tracker.required());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testEarlyFailure() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(3, 5);
|
||||
|
||||
// Need 3, have 5 total
|
||||
tracker.onResponse(endpoint("127.0.0.1")); // 1 success
|
||||
tracker.onFailure(endpoint("127.0.0.2")); // 1 failure
|
||||
tracker.onFailure(endpoint("127.0.0.3")); // 2 failures
|
||||
tracker.onFailure(endpoint("127.0.0.4")); // 3 failures
|
||||
|
||||
// Have 1 success, 3 failures, 1 remaining
|
||||
// Need 2 more but only 1 remaining -> impossible
|
||||
assertTrue(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertEquals(1, tracker.received());
|
||||
assertEquals(3, tracker.failures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllSucceed() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(3, 3);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(3, tracker.received());
|
||||
assertEquals(0, tracker.failures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllFail() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 3);
|
||||
|
||||
tracker.onFailure(endpoint("127.0.0.1"));
|
||||
tracker.onFailure(endpoint("127.0.0.2"));
|
||||
tracker.onFailure(endpoint("127.0.0.3"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertEquals(0, tracker.received());
|
||||
assertEquals(3, tracker.failures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testBlockForOne() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(1, 3);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(1, tracker.received());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testBlockForAll() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(3, 3);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertFalse(tracker.isComplete());
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testZeroResponses() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 3);
|
||||
|
||||
assertFalse(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertEquals(0, tracker.received());
|
||||
assertEquals(0, tracker.failures());
|
||||
assertEquals(2, tracker.required());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testConcurrentResponses() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(50, 100);
|
||||
ExecutorService executor = Executors.newFixedThreadPool(10);
|
||||
CountDownLatch startLatch = new CountDownLatch(1);
|
||||
CountDownLatch doneLatch = new CountDownLatch(50);
|
||||
|
||||
try
|
||||
{
|
||||
// Launch 50 threads to call onResponse concurrently
|
||||
for (int i = 0; i < 50; i++)
|
||||
{
|
||||
final int index = i;
|
||||
executor.submit(() -> {
|
||||
try
|
||||
{
|
||||
startLatch.await();
|
||||
tracker.onResponse(endpoint("127.0.0." + index));
|
||||
doneLatch.countDown();
|
||||
}
|
||||
catch (Exception e)
|
||||
{
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Start all threads at once
|
||||
startLatch.countDown();
|
||||
|
||||
// Wait for completion
|
||||
assertTrue(doneLatch.await(10, TimeUnit.SECONDS));
|
||||
|
||||
// Verify no lost updates
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(50, tracker.received());
|
||||
assertEquals(0, tracker.failures());
|
||||
}
|
||||
finally
|
||||
{
|
||||
executor.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
// Filtering tests
|
||||
|
||||
@Test
|
||||
public void testUnfilteredTracker() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 4);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("192.168.1.1"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.received());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testFilteredTracker() throws Exception
|
||||
{
|
||||
// Filter that only accepts local endpoints (127.0.0.*)
|
||||
Predicate<InetAddressAndPort> localFilter = endpoint ->
|
||||
endpoint.getHostAddress(false).startsWith("127.0.0.");
|
||||
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 3, localFilter);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.received());
|
||||
assertTrue(tracker.countsTowardQuorum(endpoint("127.0.0.1")));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testFilteredIgnoresNonMatching() throws Exception
|
||||
{
|
||||
// Filter that only accepts local endpoints
|
||||
Predicate<InetAddressAndPort> localFilter = endpoint ->
|
||||
endpoint.getHostAddress(false).startsWith("127.0.0.");
|
||||
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 2, localFilter);
|
||||
|
||||
// Remote endpoint response is ignored
|
||||
tracker.onResponse(endpoint("192.168.1.1"));
|
||||
assertFalse(tracker.isComplete());
|
||||
assertEquals(0, tracker.received());
|
||||
assertFalse(tracker.countsTowardQuorum(endpoint("192.168.1.1")));
|
||||
|
||||
// Local endpoints count
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.received());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCountsTowardQuorum() throws Exception
|
||||
{
|
||||
Predicate<InetAddressAndPort> filter = endpoint ->
|
||||
endpoint.getHostAddress(false).startsWith("127.0.0.");
|
||||
|
||||
SimpleResponseTracker unfilteredTracker = new SimpleResponseTracker(2, 3);
|
||||
assertTrue(unfilteredTracker.countsTowardQuorum(endpoint("127.0.0.1")));
|
||||
assertTrue(unfilteredTracker.countsTowardQuorum(endpoint("192.168.1.1")));
|
||||
|
||||
SimpleResponseTracker filteredTracker = new SimpleResponseTracker(2, 3, filter);
|
||||
assertTrue(filteredTracker.countsTowardQuorum(endpoint("127.0.0.1")));
|
||||
assertFalse(filteredTracker.countsTowardQuorum(endpoint("192.168.1.1")));
|
||||
}
|
||||
|
||||
// Usage pattern tests
|
||||
|
||||
@Test
|
||||
public void testQuorumUsage() throws Exception
|
||||
{
|
||||
// Simulates QUORUM with RF=5
|
||||
int rf = 5;
|
||||
int blockFor = rf / 2 + 1; // 3
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(blockFor, rf);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(3, tracker.required());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testLocalQuorumUsage() throws Exception
|
||||
{
|
||||
// Simulates LOCAL_QUORUM with localRF=3
|
||||
int localRf = 3;
|
||||
int blockFor = localRf / 2 + 1; // 2
|
||||
Predicate<InetAddressAndPort> localFilter = endpoint ->
|
||||
endpoint.getHostAddress(false).startsWith("127.0.0.");
|
||||
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(blockFor, localRf, localFilter);
|
||||
|
||||
// Remote response ignored
|
||||
tracker.onResponse(endpoint("192.168.1.1"));
|
||||
assertFalse(tracker.isComplete());
|
||||
|
||||
// Local responses count
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.required());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testSerialUsage() throws Exception
|
||||
{
|
||||
// Simulates SERIAL paxos with participants=5 (RF=4 + 1 pending)
|
||||
int participants = 5;
|
||||
int blockFor = participants / 2 + 1; // 3
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(blockFor, participants);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(3, tracker.required());
|
||||
}
|
||||
|
||||
// Validation tests
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testNegativeBlockFor()
|
||||
{
|
||||
new SimpleResponseTracker(-1, 3);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testNegativeTotalReplicas()
|
||||
{
|
||||
new SimpleResponseTracker(2, -1);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testToString() throws Exception
|
||||
{
|
||||
SimpleResponseTracker tracker = new SimpleResponseTracker(2, 3);
|
||||
String str = tracker.toString();
|
||||
|
||||
assertTrue(str.contains("SimpleResponseTracker"));
|
||||
assertTrue(str.contains("blockFor=2"));
|
||||
assertTrue(str.contains("totalReplicas=3"));
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,331 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.locator;
|
||||
|
||||
import java.net.UnknownHostException;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
import org.junit.Test;
|
||||
|
||||
import static org.junit.Assert.*;
|
||||
|
||||
/**
|
||||
* Tests for WriteResponseTracker implementing the double count model.
|
||||
*/
|
||||
public class WriteResponseTrackerTest
|
||||
{
|
||||
private InetAddressAndPort endpoint(String ip) throws UnknownHostException
|
||||
{
|
||||
return InetAddressAndPort.getByName(ip);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testBothRequirementsMet() throws Exception
|
||||
{
|
||||
// RF=3, pending=1: baseBlockFor=2, totalBlockFor=3
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
// 2 committed successes
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertFalse("Should not be complete - need 3 total", tracker.isComplete());
|
||||
|
||||
// 1 pending success -> 3 total
|
||||
tracker.onResponse(endpoint("127.0.0.4"));
|
||||
assertTrue("Should be complete", tracker.isComplete());
|
||||
assertTrue("Should be successful", tracker.isSuccessful());
|
||||
assertEquals(2, tracker.naturalReceived());
|
||||
assertEquals(1, tracker.pendingReceived());
|
||||
assertEquals(3, tracker.received());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCommittedRequirementNotMet() throws Exception
|
||||
{
|
||||
// RF=3, pending=1: baseBlockFor=2, totalBlockFor=3
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
// 1 committed success, 1 pending success
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.4"));
|
||||
assertFalse("Should not be complete - need 2 committed", tracker.isComplete());
|
||||
assertEquals(1, tracker.naturalReceived());
|
||||
assertEquals(1, tracker.pendingReceived());
|
||||
|
||||
// 2 committed failures -> can't reach 2 committed
|
||||
tracker.onFailure(endpoint("127.0.0.2"));
|
||||
tracker.onFailure(endpoint("127.0.0.3"));
|
||||
assertTrue("Should be complete - impossible to reach committed requirement", tracker.isComplete());
|
||||
assertFalse("Should not be successful", tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testTotalRequirementNotMet() throws Exception
|
||||
{
|
||||
// RF=3, pending=2: baseBlockFor=2, totalBlockFor=4
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
pending.add(endpoint("127.0.0.5"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 4, 3, 2, isPending);
|
||||
|
||||
// 2 committed successes (meets base requirement)
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertFalse("Should not be complete - need 4 total", tracker.isComplete());
|
||||
assertEquals(2, tracker.naturalReceived());
|
||||
|
||||
// 1 committed failure, 2 pending failures -> only 3 total possible, need 4
|
||||
tracker.onFailure(endpoint("127.0.0.3"));
|
||||
tracker.onFailure(endpoint("127.0.0.4"));
|
||||
tracker.onFailure(endpoint("127.0.0.5"));
|
||||
assertTrue("Should be complete - impossible to reach total requirement", tracker.isComplete());
|
||||
assertFalse("Should not be successful", tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testNoPendingReplicas() throws Exception
|
||||
{
|
||||
// RF=3, pending=0: baseBlockFor=2, totalBlockFor=2 (degenerates to simple case)
|
||||
Predicate<InetAddressAndPort> isPending = addr -> false;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 2, 3, 0, isPending);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
assertFalse(tracker.isComplete());
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(2, tracker.naturalReceived());
|
||||
assertEquals(0, tracker.pendingReceived());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllFail() throws Exception
|
||||
{
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
tracker.onFailure(endpoint("127.0.0.1"));
|
||||
tracker.onFailure(endpoint("127.0.0.2"));
|
||||
// After 2 committed failures, can't reach baseBlockFor=2 with only 1 remaining
|
||||
assertTrue(tracker.isComplete());
|
||||
assertFalse(tracker.isSuccessful());
|
||||
assertEquals(0, tracker.received());
|
||||
assertEquals(2, tracker.naturalFailures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAllSucceed() throws Exception
|
||||
{
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
tracker.onResponse(endpoint("127.0.0.4"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(3, tracker.naturalReceived());
|
||||
assertEquals(1, tracker.pendingReceived());
|
||||
assertEquals(4, tracker.received());
|
||||
assertEquals(0, tracker.failures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPendingSuccessBeforeCommitted() throws Exception
|
||||
{
|
||||
// Pending responses arrive first
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
// Pending arrives first
|
||||
tracker.onResponse(endpoint("127.0.0.4"));
|
||||
assertFalse(tracker.isComplete());
|
||||
assertEquals(0, tracker.naturalReceived());
|
||||
assertEquals(1, tracker.pendingReceived());
|
||||
|
||||
// Then committed
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testExactlyMeetsRequirements() throws Exception
|
||||
{
|
||||
// RF=2, pending=1: baseBlockFor=2, totalBlockFor=3
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.3"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 2, 1, isPending);
|
||||
|
||||
// Exactly 2 committed (all of them)
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
assertFalse("Need 3 total", tracker.isComplete());
|
||||
|
||||
// Exactly 1 pending (all of them)
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testMixedSuccessesAndFailures() throws Exception
|
||||
{
|
||||
// RF=5, pending=2: baseBlockFor=3, totalBlockFor=5
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.6"));
|
||||
pending.add(endpoint("127.0.0.7"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(3, 5, 5, 2, isPending);
|
||||
|
||||
// 3 committed successes, 2 committed failures
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onResponse(endpoint("127.0.0.2"));
|
||||
tracker.onResponse(endpoint("127.0.0.3"));
|
||||
tracker.onFailure(endpoint("127.0.0.4"));
|
||||
tracker.onFailure(endpoint("127.0.0.5"));
|
||||
|
||||
assertFalse("Need 5 total, only have 3", tracker.isComplete());
|
||||
assertEquals(3, tracker.naturalReceived());
|
||||
assertEquals(2, tracker.naturalFailures());
|
||||
|
||||
// 2 pending successes
|
||||
tracker.onResponse(endpoint("127.0.0.6"));
|
||||
tracker.onResponse(endpoint("127.0.0.7"));
|
||||
|
||||
assertTrue(tracker.isComplete());
|
||||
assertTrue(tracker.isSuccessful());
|
||||
assertEquals(3, tracker.naturalReceived());
|
||||
assertEquals(2, tracker.pendingReceived());
|
||||
assertEquals(5, tracker.received());
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testNegativeBaseBlockFor()
|
||||
{
|
||||
new WriteResponseTracker(-1, 2, 3, 1, addr -> false);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testTotalRequiredLessThanBase()
|
||||
{
|
||||
new WriteResponseTracker(3, 2, 3, 1, addr -> false);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testBaseBlockForExceedsCommitted()
|
||||
{
|
||||
new WriteResponseTracker(4, 5, 3, 2, addr -> false);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testTotalBlockForExceedsTotalReplicas()
|
||||
{
|
||||
new WriteResponseTracker(2, 6, 3, 2, addr -> false);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalArgumentException.class)
|
||||
public void testNullPredicate()
|
||||
{
|
||||
new WriteResponseTracker(2, 3, 3, 1, null);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAccessors() throws Exception
|
||||
{
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
assertEquals(2, tracker.baseBlockFor());
|
||||
assertEquals(3, tracker.required()); // Returns totalBlockFor for error messages
|
||||
|
||||
tracker.onResponse(endpoint("127.0.0.1"));
|
||||
tracker.onFailure(endpoint("127.0.0.2"));
|
||||
tracker.onResponse(endpoint("127.0.0.4"));
|
||||
|
||||
assertEquals(1, tracker.naturalReceived());
|
||||
assertEquals(1, tracker.pendingReceived());
|
||||
assertEquals(1, tracker.naturalFailures());
|
||||
assertEquals(0, tracker.pendingFailures());
|
||||
assertEquals(2, tracker.received());
|
||||
assertEquals(1, tracker.failures());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCountsTowardQuorum() throws Exception
|
||||
{
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
|
||||
// All endpoints count toward quorum in writes
|
||||
assertTrue(tracker.countsTowardQuorum(endpoint("127.0.0.1")));
|
||||
assertTrue(tracker.countsTowardQuorum(endpoint("127.0.0.4")));
|
||||
assertTrue(tracker.countsTowardQuorum(endpoint("192.168.1.1")));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testToString() throws Exception
|
||||
{
|
||||
Set<InetAddressAndPort> pending = new HashSet<>();
|
||||
pending.add(endpoint("127.0.0.4"));
|
||||
Predicate<InetAddressAndPort> isPending = pending::contains;
|
||||
|
||||
WriteResponseTracker tracker = new WriteResponseTracker(2, 3, 3, 1, isPending);
|
||||
String str = tracker.toString();
|
||||
|
||||
assertTrue(str.contains("WriteResponseTracker"));
|
||||
assertTrue(str.contains("baseBlockFor=2"));
|
||||
assertTrue(str.contains("totalBlockFor=3"));
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,563 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import com.google.common.collect.ImmutableList;
|
||||
import org.junit.Assert;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.ReadResponse;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.service.reads.ReadCallback;
|
||||
import org.apache.cassandra.service.reads.ResponseResolver;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
import org.apache.cassandra.transport.Dispatcher;
|
||||
|
||||
import static org.quicktheories.QuickTheory.qt;
|
||||
|
||||
/**
|
||||
* Property-based tests for ReadCallback with dynamic topology generation.
|
||||
* Tests incremental response behavior: applies responses one at a time and validates
|
||||
* completion state after each response.
|
||||
*/
|
||||
public class ReadCallbackPropertyTest extends ResponseHandlerPropertyTestBase
|
||||
{
|
||||
private static final ImmutableList<ConsistencyLevel> consistencyLevels = ImmutableList.of(
|
||||
ConsistencyLevel.ONE,
|
||||
ConsistencyLevel.TWO,
|
||||
ConsistencyLevel.THREE,
|
||||
ConsistencyLevel.QUORUM,
|
||||
ConsistencyLevel.ALL,
|
||||
ConsistencyLevel.LOCAL_ONE,
|
||||
ConsistencyLevel.LOCAL_QUORUM,
|
||||
ConsistencyLevel.EACH_QUORUM
|
||||
);
|
||||
|
||||
@Override
|
||||
protected List<ConsistencyLevel> consistencyLevels()
|
||||
{
|
||||
return consistencyLevels;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate expected outcome based on responses so far.
|
||||
* Returns null if more responses needed (incomplete).
|
||||
*
|
||||
* Single-requirement model for reads (no pending replica complexity):
|
||||
* - Reads only count responses from full (non-pending) replicas
|
||||
* - blockFor does NOT include pending replicas
|
||||
* - Success when: full_replica_successes >= CL@RF
|
||||
* - For LOCAL_* CLs, only local DC replicas are contacted
|
||||
*/
|
||||
private static ExpectedOutcome calculateExpectedOutcome(TopologyConfig topology,
|
||||
ConsistencyLevel cl,
|
||||
List<ResponseMessage> responses,
|
||||
int blockFor,
|
||||
int contactedReplicas)
|
||||
{
|
||||
// Count successes from full replicas only (pending can't serve reads)
|
||||
int fullReplicaSuccesses = 0;
|
||||
int fullReplicaResponses = 0;
|
||||
|
||||
Map<String, Integer> successesPerDc = new HashMap<>();
|
||||
Map<String, Integer> responsesPerDc = new HashMap<>();
|
||||
|
||||
for (ResponseMessage resp : responses)
|
||||
{
|
||||
// Skip pending replicas - they can't serve reads
|
||||
if (resp.replicaIdx >= topology.totalReplicas)
|
||||
continue;
|
||||
|
||||
String dc = getReplicaDatacenter(resp.replicaIdx, topology);
|
||||
fullReplicaResponses++;
|
||||
responsesPerDc.merge(dc, 1, Integer::sum);
|
||||
|
||||
if (resp.isSuccess())
|
||||
{
|
||||
fullReplicaSuccesses++;
|
||||
successesPerDc.merge(dc, 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
|
||||
// Check completion conditions based on CL type
|
||||
switch (cl)
|
||||
{
|
||||
case ONE:
|
||||
case TWO:
|
||||
case THREE:
|
||||
case QUORUM:
|
||||
case ALL:
|
||||
{
|
||||
// Global CLs: need blockFor successes from full replicas
|
||||
if (fullReplicaSuccesses >= blockFor)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: can't satisfy requirement
|
||||
int remaining = contactedReplicas - fullReplicaResponses;
|
||||
if (fullReplicaSuccesses + remaining < blockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
case LOCAL_ONE:
|
||||
case LOCAL_QUORUM:
|
||||
{
|
||||
// Local CLs: only count responses from local DC (datacenter1)
|
||||
String localDc = "datacenter1";
|
||||
int localSuccesses = successesPerDc.getOrDefault(localDc, 0);
|
||||
int localResponses = responsesPerDc.getOrDefault(localDc, 0);
|
||||
|
||||
if (localSuccesses >= blockFor)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: can't satisfy requirement from local DC
|
||||
int localRemaining = contactedReplicas - localResponses;
|
||||
if (localSuccesses + localRemaining < blockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
case EACH_QUORUM:
|
||||
{
|
||||
// Note: ReadCallback doesn't have special per-DC tracking for EACH_QUORUM.
|
||||
// It just counts total successes against blockFor (sum of DC quorums).
|
||||
// This is different from DatacenterSyncWriteResponseHandler which tracks per-DC.
|
||||
if (fullReplicaSuccesses >= blockFor)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: can't satisfy requirement
|
||||
int remaining = contactedReplicas - fullReplicaResponses;
|
||||
if (fullReplicaSuccesses + remaining < blockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
default:
|
||||
throw new IllegalArgumentException("Unsupported CL: " + cl);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A minimal ResponseResolver for testing that only tracks response counts and data presence.
|
||||
* This allows testing ReadCallback's completion logic without full read infrastructure.
|
||||
*/
|
||||
private static class TestResponseResolver extends ResponseResolver<EndpointsForToken, ReplicaPlan.ForTokenRead>
|
||||
{
|
||||
private volatile boolean dataPresent = false;
|
||||
|
||||
public TestResponseResolver(CoordinationPlan.ForTokenRead plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(null, null, plan, requestTime);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean isDataPresent()
|
||||
{
|
||||
return dataPresent;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void preprocess(Message<ReadResponse> message)
|
||||
{
|
||||
// Add to accumulator (parent class behavior)
|
||||
responses.add(message);
|
||||
// First full replica response makes data present
|
||||
Replica replica = replicaPlan().lookup(message.from());
|
||||
if (replica != null && replica.isFull())
|
||||
dataPresent = true;
|
||||
}
|
||||
|
||||
public int responseCount()
|
||||
{
|
||||
return responses.size();
|
||||
}
|
||||
|
||||
/** Expose replicaPlan for testing */
|
||||
public ReplicaPlan.ForTokenRead getReplicaPlan()
|
||||
{
|
||||
return replicaPlan();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a ReplicaPlan for testing with the given parameters.
|
||||
*/
|
||||
private static CoordinationPlan.ForTokenRead createReplicaPlan(Keyspace ks,
|
||||
ConsistencyLevel cl,
|
||||
EndpointsForToken contacts)
|
||||
{
|
||||
ReplicaPlan.ForTokenRead plan = new ReplicaPlan.ForTokenRead(
|
||||
ks,
|
||||
ks.getReplicationStrategy(),
|
||||
cl,
|
||||
contacts, // candidates
|
||||
contacts, // contacts
|
||||
contacts, // liveAndDown
|
||||
(cm) -> null, // recompute function
|
||||
(self) -> null, // repair plan function
|
||||
Epoch.EMPTY
|
||||
);
|
||||
return CoordinationPlans.create(plan);
|
||||
}
|
||||
|
||||
/**
|
||||
* Gets the set of replica indices that belong to the local datacenter (datacenter1).
|
||||
*/
|
||||
private static Set<Integer> getLocalDcReplicaIndices(TopologyConfig topology)
|
||||
{
|
||||
Set<Integer> localIndices = new HashSet<>();
|
||||
int idx = 0;
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
String dc = entry.getKey();
|
||||
int rf = entry.getValue();
|
||||
if ("datacenter1".equals(dc))
|
||||
{
|
||||
for (int i = 0; i < rf; i++)
|
||||
localIndices.add(idx + i);
|
||||
}
|
||||
idx += rf;
|
||||
}
|
||||
return localIndices;
|
||||
}
|
||||
|
||||
/**
|
||||
* Filters replicas to only include those from the local datacenter.
|
||||
*/
|
||||
private static EndpointsForToken filterToLocalDc(EndpointsForToken replicas, TopologyConfig topology)
|
||||
{
|
||||
Set<Integer> localIndices = getLocalDcReplicaIndices(topology);
|
||||
List<Replica> localReplicas = new ArrayList<>();
|
||||
for (int i = 0; i < replicas.size(); i++)
|
||||
{
|
||||
if (localIndices.contains(i))
|
||||
localReplicas.add(replicas.get(i));
|
||||
}
|
||||
return EndpointsForToken.of(replicas.token(), localReplicas.toArray(new Replica[0]));
|
||||
}
|
||||
|
||||
/**
|
||||
* Bundles a handler with the set of contacted replica indices.
|
||||
*/
|
||||
private static class HandlerWithContacts
|
||||
{
|
||||
final ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler;
|
||||
final Set<Integer> contactedIndices;
|
||||
|
||||
HandlerWithContacts(ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler, Set<Integer> contactedIndices)
|
||||
{
|
||||
this.handler = handler;
|
||||
this.contactedIndices = contactedIndices;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a fresh ReadCallback for testing.
|
||||
* For LOCAL_* CLs, only local DC replicas are contacted.
|
||||
*/
|
||||
private static HandlerWithContacts createHandler(
|
||||
Keyspace ks,
|
||||
ConsistencyLevel cl,
|
||||
EndpointsForToken fullReplicas,
|
||||
TopologyConfig topology)
|
||||
{
|
||||
EndpointsForToken contacts;
|
||||
Set<Integer> contactedIndices;
|
||||
|
||||
if (cl == ConsistencyLevel.LOCAL_ONE || cl == ConsistencyLevel.LOCAL_QUORUM)
|
||||
{
|
||||
// For LOCAL_* CLs, only contact local DC replicas
|
||||
contacts = filterToLocalDc(fullReplicas, topology);
|
||||
contactedIndices = getLocalDcReplicaIndices(topology);
|
||||
}
|
||||
else
|
||||
{
|
||||
// For global CLs, contact all full replicas
|
||||
contacts = fullReplicas;
|
||||
contactedIndices = new HashSet<>();
|
||||
for (int i = 0; i < fullReplicas.size(); i++)
|
||||
contactedIndices.add(i);
|
||||
}
|
||||
|
||||
CoordinationPlan.ForTokenRead plan = createReplicaPlan(ks, cl, contacts);
|
||||
Dispatcher.RequestTime requestTime = new Dispatcher.RequestTime(System.nanoTime(), System.nanoTime());
|
||||
|
||||
TestResponseResolver resolver = new TestResponseResolver(plan, requestTime);
|
||||
ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler = new ReadCallback<>(resolver, null, plan, requestTime);
|
||||
|
||||
return new HandlerWithContacts(handler, contactedIndices);
|
||||
}
|
||||
|
||||
/**
|
||||
* Gets the endpoint for a replica index from the replica sets.
|
||||
*/
|
||||
private static InetAddressAndPort getEndpoint(int replicaIdx, ReplicaSets replicaSets)
|
||||
{
|
||||
if (replicaIdx < replicaSets.fullReplicas.size())
|
||||
return replicaSets.fullReplicas.get(replicaIdx).endpoint();
|
||||
|
||||
int pendingIdx = replicaIdx - replicaSets.fullReplicas.size();
|
||||
if (pendingIdx < replicaSets.pendingReplicas.size())
|
||||
return replicaSets.pendingReplicas.get(pendingIdx).endpoint();
|
||||
|
||||
throw new IllegalArgumentException("Invalid replica index: " + replicaIdx);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a minimal ReadResponse message for testing.
|
||||
* Uses Message.synthetic() which is designed for testing and allows creating messages from arbitrary nodes.
|
||||
* The TestResponseResolver doesn't access the payload, so we pass null.
|
||||
*/
|
||||
private static Message<ReadResponse> createResponseMessage(InetAddressAndPort from)
|
||||
{
|
||||
return Message.synthetic(from, org.apache.cassandra.net.Verb.READ_RSP, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a single response to the handler.
|
||||
* Returns true if the response was applied (i.e., from a replica in contacts).
|
||||
*/
|
||||
private static boolean applyResponse(ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler,
|
||||
ResponseMessage response,
|
||||
ReplicaSets replicaSets,
|
||||
Set<Integer> contactedIndices)
|
||||
{
|
||||
// Skip pending replicas - they can't serve reads
|
||||
if (response.replicaIdx >= replicaSets.fullReplicas.size())
|
||||
return false;
|
||||
|
||||
// Skip replicas not in the contact set (important for LOCAL_* CLs)
|
||||
if (!contactedIndices.contains(response.replicaIdx))
|
||||
return false;
|
||||
|
||||
InetAddressAndPort endpoint = getEndpoint(response.replicaIdx, replicaSets);
|
||||
|
||||
if (response.isSuccess())
|
||||
{
|
||||
Message<ReadResponse> msg = createResponseMessage(endpoint);
|
||||
handler.onResponse(msg);
|
||||
}
|
||||
else
|
||||
{
|
||||
handler.onFailure(endpoint, new RequestFailure(response.failureReason, null));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks if the handler has signaled completion.
|
||||
*/
|
||||
private static boolean isComplete(ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler)
|
||||
{
|
||||
// ReadCallback signals completion via its condition
|
||||
// We check by attempting a zero-timeout await
|
||||
return handler.await(0, java.util.concurrent.TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a pre-generated response sequence to a handler, stopping when complete.
|
||||
* Returns the subset of responses that were actually applied.
|
||||
*/
|
||||
private static List<ResponseMessage> applyResponseSequence(
|
||||
ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler,
|
||||
List<ResponseMessage> responses,
|
||||
ReplicaSets replicaSets,
|
||||
Set<Integer> contactedIndices)
|
||||
{
|
||||
List<ResponseMessage> applied = new ArrayList<>();
|
||||
|
||||
for (ResponseMessage response : responses)
|
||||
{
|
||||
// Check if already complete before applying
|
||||
if (isComplete(handler))
|
||||
break;
|
||||
|
||||
if (applyResponse(handler, response, replicaSets, contactedIndices))
|
||||
applied.add(response);
|
||||
}
|
||||
|
||||
return applied;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validates handler completed with expected outcome.
|
||||
*/
|
||||
private static void validateOutcome(ReadCallback<EndpointsForToken, ReplicaPlan.ForTokenRead> handler,
|
||||
List<ResponseMessage> appliedResponses,
|
||||
TopologyConfig topology,
|
||||
ConsistencyLevel cl,
|
||||
int contactedReplicas)
|
||||
{
|
||||
TestResponseResolver resolver = (TestResponseResolver) handler.resolver;
|
||||
int blockFor = resolver.getReplicaPlan().readQuorum();
|
||||
ExpectedOutcome expected = calculateExpectedOutcome(topology, cl, appliedResponses, blockFor, contactedReplicas);
|
||||
|
||||
boolean complete = isComplete(handler);
|
||||
int successes = resolver.responseCount();
|
||||
int failures = contactedReplicas - successes;
|
||||
|
||||
if (expected == null)
|
||||
{
|
||||
// Should not be complete yet
|
||||
Assert.assertFalse(
|
||||
String.format("Handler completed prematurely with %d successes, %d failures (blockFor=%d, CL=%s)",
|
||||
successes, failures, blockFor, cl),
|
||||
complete
|
||||
);
|
||||
}
|
||||
else if (expected == ExpectedOutcome.SUCCESS)
|
||||
{
|
||||
Assert.assertTrue(
|
||||
String.format("Handler should have completed successfully with %d successes (blockFor=%d, CL=%s)",
|
||||
successes, blockFor, cl),
|
||||
complete
|
||||
);
|
||||
Assert.assertTrue(
|
||||
String.format("Data should be present with %d successes", successes),
|
||||
resolver.isDataPresent()
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Expected failure - handler should be complete due to too many failures
|
||||
Assert.assertTrue(
|
||||
String.format("Handler should have completed (failed) with %d failures (blockFor=%d, contacts=%d, CL=%s)",
|
||||
failures, blockFor, contactedReplicas, cl),
|
||||
complete
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Property Tests
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void readCallbackBehavior()
|
||||
{
|
||||
qt()
|
||||
.withExamples(250)
|
||||
.forAll(testCaseGen())
|
||||
.assuming(testCase -> {
|
||||
// Filter out topologies where any scenario has insufficient replicas for its CL
|
||||
for (CLScenario scenario : testCase.scenarios)
|
||||
{
|
||||
if (testCase.topology.totalReplicas < minReplicasForCL(scenario.cl))
|
||||
return false;
|
||||
if (!hasEnoughLocalReplicas(testCase.topology, scenario.cl))
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
})
|
||||
.checkAssert(testCase -> {
|
||||
try
|
||||
{
|
||||
Keyspace ks = getOrCreateKeyspace(testCase.topology);
|
||||
ReplicaSets replicaSets = createReplicaSets(testCase.topology);
|
||||
|
||||
// Test all scenarios for this topology
|
||||
for (CLScenario scenario : testCase.scenarios)
|
||||
{
|
||||
HandlerWithContacts hwc = createHandler(ks, scenario.cl, replicaSets.fullReplicas, testCase.topology);
|
||||
List<ResponseMessage> appliedResponses = applyResponseSequence(
|
||||
hwc.handler, scenario.responses, replicaSets, hwc.contactedIndices);
|
||||
validateOutcome(hwc.handler, appliedResponses, testCase.topology, scenario.cl, hwc.contactedIndices.size());
|
||||
}
|
||||
}
|
||||
catch (Throwable e)
|
||||
{
|
||||
if (e instanceof AssertionError)
|
||||
throw (AssertionError) e;
|
||||
throw new AssertionError("Test setup failed: " + e.getMessage(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimum replicas needed for a consistency level to be valid.
|
||||
* For LOCAL_* CLs, this returns the minimum needed in the local DC.
|
||||
*/
|
||||
private static int minReplicasForCL(ConsistencyLevel cl)
|
||||
{
|
||||
switch (cl)
|
||||
{
|
||||
case ONE:
|
||||
case LOCAL_ONE:
|
||||
return 1;
|
||||
case TWO:
|
||||
return 2;
|
||||
case THREE:
|
||||
return 3;
|
||||
case QUORUM:
|
||||
case LOCAL_QUORUM:
|
||||
return 3; // Need at least 3 for quorum to be meaningful (3/2 + 1 = 2)
|
||||
case EACH_QUORUM:
|
||||
return 1; // Just need at least 1 replica per DC (checked separately)
|
||||
case ALL:
|
||||
return 1;
|
||||
default:
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if the local DC has enough replicas for the given CL.
|
||||
*/
|
||||
private static boolean hasEnoughLocalReplicas(TopologyConfig topology, ConsistencyLevel cl)
|
||||
{
|
||||
if (cl == ConsistencyLevel.LOCAL_ONE || cl == ConsistencyLevel.LOCAL_QUORUM)
|
||||
{
|
||||
Integer localRf = topology.replicationFactors.get("datacenter1");
|
||||
if (localRf == null)
|
||||
return false;
|
||||
return localRf >= minReplicasForCL(cl);
|
||||
}
|
||||
|
||||
if (cl == ConsistencyLevel.EACH_QUORUM)
|
||||
{
|
||||
// EACH_QUORUM requires at least 1 replica in every DC
|
||||
for (int rf : topology.replicationFactors.values())
|
||||
{
|
||||
if (rf < 1)
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,597 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service;
|
||||
|
||||
import java.net.InetAddress;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.function.BiFunction;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.junit.BeforeClass;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import org.apache.cassandra.SchemaLoader;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner;
|
||||
import org.apache.cassandra.distributed.test.log.ClusterMetadataTestHelper;
|
||||
import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||
import org.apache.cassandra.locator.BaseProximity;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.NodeProximity;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaCollection;
|
||||
import org.apache.cassandra.locator.ReplicaUtils;
|
||||
import org.apache.cassandra.schema.KeyspaceParams;
|
||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||
import org.quicktheories.core.Gen;
|
||||
|
||||
import static org.quicktheories.generators.SourceDSL.arbitrary;
|
||||
import static org.quicktheories.generators.SourceDSL.booleans;
|
||||
import static org.quicktheories.generators.SourceDSL.integers;
|
||||
import static org.quicktheories.generators.SourceDSL.lists;
|
||||
|
||||
public abstract class ResponseHandlerPropertyTestBase
|
||||
{
|
||||
protected static final Logger logger = LoggerFactory.getLogger(ResponseHandlerPropertyTestBase.class);
|
||||
protected static final AtomicInteger keyspaceCounter = new AtomicInteger(0);
|
||||
protected static final Map<String, Keyspace> topologyCache = new HashMap<>();
|
||||
protected static final Set<InetAddressAndPort> registeredNodes = new HashSet<>();
|
||||
|
||||
@BeforeClass
|
||||
public static void setUpClass() throws Throwable
|
||||
{
|
||||
// Set partitioner system property BEFORE DatabaseDescriptor initialization
|
||||
System.setProperty("cassandra.partitioner", Murmur3Partitioner.class.getName());
|
||||
|
||||
SchemaLoader.loadSchema();
|
||||
|
||||
// Configure node proximity
|
||||
NodeProximity sorter = new BaseProximity()
|
||||
{
|
||||
public <C extends ReplicaCollection<? extends C>> C sortedByProximity(InetAddressAndPort address, C replicas)
|
||||
{
|
||||
return replicas;
|
||||
}
|
||||
|
||||
public int compareEndpoints(InetAddressAndPort target, Replica a1, Replica a2)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
public boolean isWorthMergingForRangeQuery(ReplicaCollection<?> merged, ReplicaCollection<?> l1, ReplicaCollection<?> l2)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
};
|
||||
DatabaseDescriptor.setNodeProximity(sorter);
|
||||
// Set broadcast address to match first replica of datacenter1 (replicaIdx=0 → 127.1.0.255)
|
||||
// This ensures InOurDc.endpoints() correctly identifies local DC replicas
|
||||
InetAddress broadcastAddr = InetAddress.getByName("127.1.0.255");
|
||||
DatabaseDescriptor.setBroadcastAddress(broadcastAddr);
|
||||
// Register broadcast address with datacenter1 so locator knows our DC
|
||||
InetAddressAndPort broadcastEndpoint = InetAddressAndPort.getByAddress(broadcastAddr);
|
||||
ClusterMetadataTestHelper.register(broadcastEndpoint, "datacenter1", "rack1");
|
||||
registeredNodes.add(broadcastEndpoint);
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Data Structures
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Describes a cluster topology configuration.
|
||||
*/
|
||||
public static class TopologyConfig
|
||||
{
|
||||
public final int numDatacenters;
|
||||
public final Map<String, Integer> replicationFactors; // DC name -> RF
|
||||
public final Map<String, Integer> pendingReplicas; // DC name -> pending count
|
||||
public final int totalReplicas;
|
||||
public final int totalPending;
|
||||
|
||||
TopologyConfig(int numDatacenters, Map<String, Integer> replicationFactors, Map<String, Integer> pendingReplicas)
|
||||
{
|
||||
this.numDatacenters = numDatacenters;
|
||||
this.replicationFactors = replicationFactors;
|
||||
this.pendingReplicas = pendingReplicas;
|
||||
this.totalReplicas = replicationFactors.values().stream().mapToInt(Integer::intValue).sum();
|
||||
this.totalPending = pendingReplicas.values().stream().mapToInt(Integer::intValue).sum();
|
||||
}
|
||||
|
||||
String signature()
|
||||
{
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("dcs=").append(numDatacenters);
|
||||
replicationFactors.forEach((dc, rf) -> sb.append(",").append(dc).append(":").append(rf));
|
||||
if (totalPending > 0)
|
||||
{
|
||||
sb.append(",pending:");
|
||||
pendingReplicas.forEach((dc, count) -> {
|
||||
if (count > 0)
|
||||
sb.append(dc).append("=").append(count).append(";");
|
||||
});
|
||||
}
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return signature();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A single response message (success or failure).
|
||||
*/
|
||||
public static class ResponseMessage
|
||||
{
|
||||
public final int replicaIdx;
|
||||
public final RequestFailureReason failureReason; // null = success
|
||||
|
||||
public ResponseMessage(int replicaIdx, RequestFailureReason failureReason)
|
||||
{
|
||||
this.replicaIdx = replicaIdx;
|
||||
this.failureReason = failureReason;
|
||||
}
|
||||
|
||||
public boolean isSuccess()
|
||||
{
|
||||
return failureReason == null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("replica=%d, %s", replicaIdx, isSuccess() ? "SUCCESS" : "FAILURE(" + failureReason + ")");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A test scenario for a specific consistency level with pre-generated responses.
|
||||
*/
|
||||
public static class CLScenario
|
||||
{
|
||||
public final ConsistencyLevel cl;
|
||||
public final List<ResponseMessage> responses;
|
||||
|
||||
public CLScenario(ConsistencyLevel cl, List<ResponseMessage> responses)
|
||||
{
|
||||
this.cl = cl;
|
||||
this.responses = responses;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("CL=%s, responses=%d", cl, responses.size());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A complete test case with topology and multiple CL scenarios.
|
||||
*/
|
||||
public static class TestCase
|
||||
{
|
||||
public final TopologyConfig topology;
|
||||
public final List<CLScenario> scenarios;
|
||||
|
||||
public TestCase(TopologyConfig topology, List<CLScenario> scenarios)
|
||||
{
|
||||
this.topology = topology;
|
||||
this.scenarios = scenarios;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("topology=%s, scenarios=%d", topology, scenarios.size());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Holds both full and pending replica sets for a topology.
|
||||
*/
|
||||
public static class ReplicaSets
|
||||
{
|
||||
public final EndpointsForToken fullReplicas;
|
||||
public final EndpointsForToken pendingReplicas;
|
||||
|
||||
public ReplicaSets(EndpointsForToken fullReplicas, EndpointsForToken pendingReplicas)
|
||||
{
|
||||
this.fullReplicas = fullReplicas;
|
||||
this.pendingReplicas = pendingReplicas;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Expected outcome of a response sequence
|
||||
*/
|
||||
public enum ExpectedOutcome
|
||||
{
|
||||
SUCCESS, // Handler should complete successfully
|
||||
FAILURE // Handler should fail
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Generators
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Generates varied datacenter topologies (1-7 DCs with varying RF and pending replicas).
|
||||
*/
|
||||
protected Gen<TopologyConfig> topologyGen()
|
||||
{
|
||||
return integers().between(1, 7).flatMap(numDcs -> {
|
||||
return lists().of(integers().between(1, 5)).ofSize(numDcs).flatMap(rfList -> {
|
||||
// Generate 0-2 pending replicas per DC
|
||||
return lists().of(integers().between(0, 2)).ofSize(numDcs).map(pendingList -> {
|
||||
Map<String, Integer> replicationFactors = new LinkedHashMap<>();
|
||||
Map<String, Integer> pendingReplicas = new LinkedHashMap<>();
|
||||
for (int i = 0; i < numDcs; i++)
|
||||
{
|
||||
String dcName = "datacenter" + (i + 1);
|
||||
replicationFactors.put(dcName, rfList.get(i));
|
||||
pendingReplicas.put(dcName, pendingList.get(i));
|
||||
}
|
||||
return new TopologyConfig(numDcs, replicationFactors, pendingReplicas);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// ========================================
|
||||
// Topology Setup
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Gets or creates a keyspace for the given topology.
|
||||
*/
|
||||
public static Keyspace getOrCreateKeyspace(TopologyConfig topology) throws Exception
|
||||
{
|
||||
String signature = topology.signature();
|
||||
if (topologyCache.containsKey(signature))
|
||||
return topologyCache.get(signature);
|
||||
|
||||
String keyspaceName = "PropTest" + keyspaceCounter.incrementAndGet();
|
||||
|
||||
// Register full replica nodes
|
||||
int replicaIdx = 0;
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
String dcName = entry.getKey();
|
||||
int rf = entry.getValue();
|
||||
int dcNum = Integer.parseInt(dcName.substring("datacenter".length()));
|
||||
|
||||
for (int i = 0; i < rf; i++)
|
||||
{
|
||||
String ip = String.format("127.%d.0.%d", dcNum, 255 - replicaIdx);
|
||||
InetAddressAndPort endpoint = InetAddressAndPort.getByName(ip);
|
||||
|
||||
if (!registeredNodes.contains(endpoint))
|
||||
{
|
||||
ClusterMetadataTestHelper.register(endpoint, dcName, "rack1");
|
||||
registeredNodes.add(endpoint);
|
||||
}
|
||||
replicaIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
// Pending replica nodes don't need ClusterMetadata registration
|
||||
// We pass them directly to ReplicaPlan in createHandler()
|
||||
// Just register them with basic info so InOurDc can identify their datacenter
|
||||
for (Map.Entry<String, Integer> entry : topology.pendingReplicas.entrySet())
|
||||
{
|
||||
String dcName = entry.getKey();
|
||||
int pending = entry.getValue();
|
||||
int dcNum = Integer.parseInt(dcName.substring("datacenter".length()));
|
||||
|
||||
for (int i = 0; i < pending; i++)
|
||||
{
|
||||
String ip = String.format("127.%d.0.%d", dcNum, 255 - replicaIdx);
|
||||
InetAddressAndPort endpoint = InetAddressAndPort.getByName(ip);
|
||||
|
||||
if (!registeredNodes.contains(endpoint))
|
||||
{
|
||||
// Just register for DC/rack identification - no join process needed
|
||||
ClusterMetadataTestHelper.register(endpoint, dcName, "rack1");
|
||||
registeredNodes.add(endpoint);
|
||||
}
|
||||
replicaIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
// Create keyspace
|
||||
Object[] dcRfPairs = topology.replicationFactors.entrySet().stream()
|
||||
.flatMap(e -> Stream.of(e.getKey(), e.getValue()))
|
||||
.toArray();
|
||||
SchemaLoader.createKeyspace(keyspaceName, KeyspaceParams.nts(dcRfPairs),
|
||||
SchemaLoader.standardCFMD(keyspaceName, "Standard"));
|
||||
|
||||
Keyspace ks = Keyspace.open(keyspaceName);
|
||||
topologyCache.put(signature, ks);
|
||||
return ks;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates full and pending replica sets for the given topology.
|
||||
*/
|
||||
public static ReplicaSets createReplicaSets(TopologyConfig topology) throws Exception
|
||||
{
|
||||
List<Replica> fullReplicas = new ArrayList<>();
|
||||
List<Replica> pendingReplicas = new ArrayList<>();
|
||||
int replicaIdx = 0;
|
||||
|
||||
// Create full replicas for all DCs first (must match getOrCreateKeyspace ordering)
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
int rf = entry.getValue();
|
||||
int dcNum = Integer.parseInt(entry.getKey().substring("datacenter".length()));
|
||||
|
||||
for (int i = 0; i < rf; i++)
|
||||
{
|
||||
String ip = String.format("127.%d.0.%d", dcNum, 255 - replicaIdx);
|
||||
fullReplicas.add(ReplicaUtils.full(InetAddressAndPort.getByName(ip)));
|
||||
replicaIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
// Then create pending replicas for all DCs
|
||||
for (Map.Entry<String, Integer> entry : topology.pendingReplicas.entrySet())
|
||||
{
|
||||
int pending = entry.getValue();
|
||||
int dcNum = Integer.parseInt(entry.getKey().substring("datacenter".length()));
|
||||
|
||||
for (int i = 0; i < pending; i++)
|
||||
{
|
||||
String ip = String.format("127.%d.0.%d", dcNum, 255 - replicaIdx);
|
||||
pendingReplicas.add(ReplicaUtils.full(InetAddressAndPort.getByName(ip)));
|
||||
replicaIdx++;
|
||||
}
|
||||
}
|
||||
|
||||
var token = Murmur3Partitioner.instance.getToken(ByteBufferUtil.bytes(0));
|
||||
return new ReplicaSets(
|
||||
EndpointsForToken.of(token, fullReplicas.toArray(new Replica[0])),
|
||||
EndpointsForToken.of(token, pendingReplicas.toArray(new Replica[0]))
|
||||
);
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Response Generation
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Maps replica index to its datacenter.
|
||||
* Indices 0..(totalReplicas-1) are full replicas, totalReplicas..(totalReplicas+totalPending-1) are pending.
|
||||
*/
|
||||
public static String getReplicaDatacenter(int replicaIdx, TopologyConfig topology)
|
||||
{
|
||||
int idx = 0;
|
||||
|
||||
// First check full replicas
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
int rf = entry.getValue();
|
||||
if (replicaIdx < idx + rf)
|
||||
return entry.getKey();
|
||||
idx += rf;
|
||||
}
|
||||
|
||||
// Then check pending replicas (indices continue from where full replicas left off)
|
||||
int pendingStartIdx = topology.totalReplicas;
|
||||
idx = pendingStartIdx;
|
||||
for (Map.Entry<String, Integer> entry : topology.pendingReplicas.entrySet())
|
||||
{
|
||||
int pending = entry.getValue();
|
||||
if (replicaIdx < idx + pending)
|
||||
return entry.getKey();
|
||||
idx += pending;
|
||||
}
|
||||
|
||||
throw new IllegalArgumentException("Invalid replica index: " + replicaIdx);
|
||||
}
|
||||
|
||||
protected abstract List<ConsistencyLevel> consistencyLevels();
|
||||
|
||||
/**
|
||||
* Generates write-applicable consistency levels.
|
||||
*/
|
||||
protected Gen<ConsistencyLevel> consistencyLevelGen()
|
||||
{
|
||||
return arbitrary().pick(consistencyLevels());
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates failure reasons for failed responses.
|
||||
*/
|
||||
protected Gen<RequestFailureReason> failureReasonGen()
|
||||
{
|
||||
return arbitrary().pick(
|
||||
RequestFailureReason.TIMEOUT,
|
||||
RequestFailureReason.UNKNOWN,
|
||||
RequestFailureReason.INCOMPATIBLE_SCHEMA,
|
||||
RequestFailureReason.COORDINATOR_BEHIND
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a random permutation of integers [0, n-1] using Fisher-Yates shuffle.
|
||||
* This ensures deterministic reproducibility from QuickTheories' seed.
|
||||
*/
|
||||
protected static Gen<List<Integer>> permutationGen(int n)
|
||||
{
|
||||
if (n == 0) return lists().of(integers().all()).ofSize(0);
|
||||
if (n == 1) return lists().of(integers().all()).ofSize(1).map(list -> {
|
||||
List<Integer> result = new ArrayList<>();
|
||||
result.add(0);
|
||||
return result;
|
||||
});
|
||||
|
||||
// Generate n-1 random swap indices for Fisher-Yates shuffle
|
||||
return lists().of(integers().between(0, n - 1)).ofSize(n - 1).map(swaps -> {
|
||||
List<Integer> result = new ArrayList<>();
|
||||
for (int i = 0; i < n; i++)
|
||||
result.add(i);
|
||||
|
||||
// Fisher-Yates shuffle using generated swap indices
|
||||
for (int i = 0; i < n - 1; i++)
|
||||
{
|
||||
int maxSwap = n - i - 1;
|
||||
int j = i + Math.min(swaps.get(i), maxSwap);
|
||||
Collections.swap(result, i, j);
|
||||
}
|
||||
|
||||
return result;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a response sequence of the given size where each element gets a random type
|
||||
* from the provided enum values, in a random order.
|
||||
*
|
||||
* @param size number of responses to generate
|
||||
* @param values possible response types
|
||||
* @param factory creates a response message from (index, type)
|
||||
*/
|
||||
protected static <T extends Enum<T>, R> Gen<List<R>> typedResponseSequenceGen(
|
||||
int size, T[] values, BiFunction<Integer, T, R> factory)
|
||||
{
|
||||
return lists().of(arbitrary().pick(values)).ofSize(size).flatMap(types ->
|
||||
permutationGen(size).map(ordering -> {
|
||||
List<R> responses = new ArrayList<>();
|
||||
for (int idx : ordering)
|
||||
responses.add(factory.apply(idx, types.get(idx)));
|
||||
return responses;
|
||||
})
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a hardcoded sequence where all responses have the same type.
|
||||
*/
|
||||
protected static <T extends Enum<T>, R> List<R> allSameTypeSequence(
|
||||
int size, T type, BiFunction<Integer, T, R> factory)
|
||||
{
|
||||
List<R> responses = new ArrayList<>();
|
||||
for (int i = 0; i < size; i++)
|
||||
responses.add(factory.apply(i, type));
|
||||
return responses;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a response sequence for a given topology.
|
||||
* Each replica (full + pending) gets a success/failure outcome, and responses are in random order.
|
||||
*/
|
||||
private Gen<List<ResponseMessage>> responseSequenceGen(TopologyConfig topology)
|
||||
{
|
||||
int totalEndpoints = topology.totalReplicas + topology.totalPending;
|
||||
|
||||
// Generate success/failure for each endpoint (full + pending)
|
||||
return lists().of(booleans().all()).ofSize(totalEndpoints).flatMap(successFlags -> {
|
||||
// Generate failure reasons for each endpoint (used only if failure)
|
||||
return lists().of(failureReasonGen()).ofSize(totalEndpoints).flatMap(failureReasons -> {
|
||||
// Generate random ordering of responses
|
||||
return permutationGen(totalEndpoints).map(ordering -> {
|
||||
List<ResponseMessage> responses = new ArrayList<>();
|
||||
for (int idx : ordering)
|
||||
{
|
||||
RequestFailureReason reason = successFlags.get(idx) ? null : failureReasons.get(idx);
|
||||
responses.add(new ResponseMessage(idx, reason));
|
||||
}
|
||||
return responses;
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates hardcoded all-success response sequence (full + pending replicas).
|
||||
*/
|
||||
private static List<ResponseMessage> allSuccessSequence(TopologyConfig topology)
|
||||
{
|
||||
List<ResponseMessage> responses = new ArrayList<>();
|
||||
int totalEndpoints = topology.totalReplicas + topology.totalPending;
|
||||
for (int i = 0; i < totalEndpoints; i++)
|
||||
responses.add(new ResponseMessage(i, null));
|
||||
return responses;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates hardcoded all-failure response sequence (full + pending replicas).
|
||||
*/
|
||||
private static List<ResponseMessage> allFailureSequence(TopologyConfig topology)
|
||||
{
|
||||
List<ResponseMessage> responses = new ArrayList<>();
|
||||
int totalEndpoints = topology.totalReplicas + topology.totalPending;
|
||||
for (int i = 0; i < totalEndpoints; i++)
|
||||
responses.add(new ResponseMessage(i, RequestFailureReason.TIMEOUT));
|
||||
return responses;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a CL scenario (consistency level + response sequence).
|
||||
*/
|
||||
private Gen<CLScenario> scenarioGen(TopologyConfig topology)
|
||||
{
|
||||
return consistencyLevelGen().flatMap(cl ->
|
||||
responseSequenceGen(topology).map(responses ->
|
||||
new CLScenario(cl, responses)
|
||||
)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a complete test case with topology and multiple CL scenarios.
|
||||
* Includes 3 random scenarios plus 2 hardcoded scenarios (all-success, all-failure).
|
||||
*/
|
||||
protected Gen<TestCase> testCaseGen()
|
||||
{
|
||||
return random -> {
|
||||
TopologyConfig topology = topologyGen().generate(random);
|
||||
|
||||
List<CLScenario> scenarios = new ArrayList<>();
|
||||
for (int i=0; i<50; i++)
|
||||
scenarios.add(scenarioGen(topology).generate(random));
|
||||
|
||||
// Add hardcoded all-success and failure scenarios for each CL
|
||||
for (ConsistencyLevel cl: consistencyLevels())
|
||||
{
|
||||
scenarios.add(new CLScenario(cl, allSuccessSequence(topology)));
|
||||
scenarios.add(new CLScenario(cl, allFailureSequence(topology)));
|
||||
}
|
||||
|
||||
return new TestCase(topology, scenarios);
|
||||
};
|
||||
}
|
||||
|
||||
}
|
||||
|
|
@ -0,0 +1,641 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service;
|
||||
|
||||
import com.google.common.base.Predicates;
|
||||
import com.google.common.collect.ImmutableList;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.WriteType;
|
||||
import org.apache.cassandra.exceptions.*;
|
||||
import org.apache.cassandra.locator.*;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.Verb;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.transport.Dispatcher;
|
||||
import org.junit.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.apache.cassandra.net.NoPayload.noPayload;
|
||||
import static org.quicktheories.QuickTheory.qt;
|
||||
|
||||
/**
|
||||
* Property-based tests for WriteResponseHandler with dynamic topology generation.
|
||||
* Tests incremental response behavior: applies responses one at a time and validates
|
||||
* completion state after each response.
|
||||
*/
|
||||
public class WriteResponseHandlerPropertyTest extends ResponseHandlerPropertyTestBase
|
||||
{
|
||||
|
||||
private static final ImmutableList<ConsistencyLevel> consistencyLevels = ImmutableList.of(
|
||||
ConsistencyLevel.ANY,
|
||||
ConsistencyLevel.ONE,
|
||||
ConsistencyLevel.TWO,
|
||||
ConsistencyLevel.THREE,
|
||||
ConsistencyLevel.QUORUM,
|
||||
ConsistencyLevel.ALL,
|
||||
ConsistencyLevel.LOCAL_ONE,
|
||||
ConsistencyLevel.LOCAL_QUORUM,
|
||||
ConsistencyLevel.EACH_QUORUM
|
||||
);
|
||||
|
||||
@Override
|
||||
protected List<ConsistencyLevel> consistencyLevels()
|
||||
{
|
||||
return consistencyLevels;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate expected outcome based on responses so far.
|
||||
* Returns null if more responses needed (incomplete).
|
||||
*
|
||||
* Two-requirement model for all CLs with pending replicas:
|
||||
* 1. Consistency: committed replicas must satisfy CL@RF
|
||||
* 2. Bootstrap safety: total replicas (committed + pending) must satisfy CL@RF + pending
|
||||
*/
|
||||
public static ExpectedOutcome calculateExpectedOutcome(TopologyConfig topology,
|
||||
ConsistencyLevel cl,
|
||||
List<ResponseMessage> responses,
|
||||
int blockFor)
|
||||
{
|
||||
// Count successes/failures by committed vs pending status
|
||||
int committedSuccesses = 0;
|
||||
int committedTotal = 0;
|
||||
int totalSuccesses = 0;
|
||||
int totalResponses = responses.size();
|
||||
|
||||
Map<String, Integer> committedSuccessesPerDc = new HashMap<>();
|
||||
Map<String, Integer> committedTotalPerDc = new HashMap<>();
|
||||
Map<String, Integer> totalSuccessesPerDc = new HashMap<>();
|
||||
Map<String, Integer> totalResponsesPerDc = new HashMap<>();
|
||||
|
||||
for (ResponseMessage resp : responses)
|
||||
{
|
||||
String dc = getReplicaDatacenter(resp.replicaIdx, topology);
|
||||
boolean isPending = resp.replicaIdx >= topology.totalReplicas;
|
||||
|
||||
totalResponsesPerDc.merge(dc, 1, Integer::sum);
|
||||
if (!isPending) committedTotalPerDc.merge(dc, 1, Integer::sum);
|
||||
|
||||
if (resp.isSuccess())
|
||||
{
|
||||
totalSuccesses++;
|
||||
totalSuccessesPerDc.merge(dc, 1, Integer::sum);
|
||||
|
||||
if (!isPending)
|
||||
{
|
||||
committedSuccesses++;
|
||||
committedSuccessesPerDc.merge(dc, 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
|
||||
if (!isPending) committedTotal++;
|
||||
}
|
||||
|
||||
int totalEndpoints = topology.totalReplicas + topology.totalPending;
|
||||
int committedEndpoints = topology.totalReplicas;
|
||||
|
||||
// Check completion conditions based on CL type
|
||||
switch (cl)
|
||||
{
|
||||
case ANY:
|
||||
{
|
||||
// ANY only needs 1 success from anywhere (can hint otherwise)
|
||||
// ANY doesn't add pending to blockFor - it just needs any 1 ack
|
||||
if (totalSuccesses >= 1)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: can't get even 1 success
|
||||
int totalRemaining = totalEndpoints - totalResponses;
|
||||
if (totalSuccesses + totalRemaining < 1)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
case ONE:
|
||||
case TWO:
|
||||
case THREE:
|
||||
case QUORUM:
|
||||
case ALL:
|
||||
{
|
||||
// Global CLs: base requirement from RF, total includes pending
|
||||
int totalBlockFor = blockFor;
|
||||
int baseBlockFor = totalBlockFor - topology.totalPending;
|
||||
|
||||
if (committedSuccesses >= baseBlockFor && totalSuccesses >= totalBlockFor)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: can't satisfy either requirement
|
||||
int committedRemaining = committedEndpoints - committedTotal;
|
||||
int totalRemaining = totalEndpoints - totalResponses;
|
||||
|
||||
if (committedSuccesses + committedRemaining < baseBlockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
if (totalSuccesses + totalRemaining < totalBlockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
case LOCAL_ONE:
|
||||
case LOCAL_QUORUM:
|
||||
{
|
||||
// Local DC requirements
|
||||
String localDc = "datacenter1";
|
||||
int localRf = topology.replicationFactors.get(localDc);
|
||||
int localPending = topology.pendingReplicas.get(localDc);
|
||||
|
||||
int baseBlockFor = (cl == ConsistencyLevel.LOCAL_ONE) ? 1 : (localRf / 2 + 1);
|
||||
int totalBlockFor = baseBlockFor + localPending;
|
||||
|
||||
int localCommittedSuccesses = committedSuccessesPerDc.getOrDefault(localDc, 0);
|
||||
int localTotalSuccesses = totalSuccessesPerDc.getOrDefault(localDc, 0);
|
||||
int localCommittedResponses = committedTotalPerDc.getOrDefault(localDc, 0);
|
||||
int localTotalResponses = totalResponsesPerDc.getOrDefault(localDc, 0);
|
||||
|
||||
if (localCommittedSuccesses >= baseBlockFor && localTotalSuccesses >= totalBlockFor)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: can't satisfy either requirement
|
||||
int committedRemaining = localRf - localCommittedResponses;
|
||||
int totalRemaining = (localRf + localPending) - localTotalResponses;
|
||||
|
||||
if (localCommittedSuccesses + committedRemaining < baseBlockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
if (localTotalSuccesses + totalRemaining < totalBlockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
case EACH_QUORUM:
|
||||
{
|
||||
// Check if all DCs satisfy both requirements
|
||||
boolean allDcsSatisfied = true;
|
||||
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
String dc = entry.getKey();
|
||||
int dcRf = entry.getValue();
|
||||
int dcPending = topology.pendingReplicas.get(dc);
|
||||
|
||||
int baseBlockFor = (dcRf / 2 + 1);
|
||||
int totalBlockFor = baseBlockFor + dcPending;
|
||||
|
||||
int dcCommittedSuccesses = committedSuccessesPerDc.getOrDefault(dc, 0);
|
||||
int dcTotalSuccesses = totalSuccessesPerDc.getOrDefault(dc, 0);
|
||||
|
||||
if (dcCommittedSuccesses < baseBlockFor || dcTotalSuccesses < totalBlockFor)
|
||||
allDcsSatisfied = false;
|
||||
}
|
||||
|
||||
if (allDcsSatisfied) return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Early failure: check if any DC can't satisfy either requirement
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
String dc = entry.getKey();
|
||||
int dcRf = entry.getValue();
|
||||
int dcPending = topology.pendingReplicas.get(dc);
|
||||
|
||||
int baseBlockFor = (dcRf / 2 + 1);
|
||||
int totalBlockFor = baseBlockFor + dcPending;
|
||||
|
||||
int dcCommittedSuccesses = committedSuccessesPerDc.getOrDefault(dc, 0);
|
||||
int dcTotalSuccesses = totalSuccessesPerDc.getOrDefault(dc, 0);
|
||||
int dcCommittedResponses = committedTotalPerDc.getOrDefault(dc, 0);
|
||||
int dcTotalResponses = totalResponsesPerDc.getOrDefault(dc, 0);
|
||||
|
||||
int committedRemaining = dcRf - dcCommittedResponses;
|
||||
int totalRemaining = (dcRf + dcPending) - dcTotalResponses;
|
||||
|
||||
if (dcCommittedSuccesses + committedRemaining < baseBlockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
if (dcTotalSuccesses + totalRemaining < totalBlockFor)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
default:
|
||||
throw new IllegalArgumentException("Unsupported CL: " + cl);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a pre-generated response sequence to a handler, stopping when complete.
|
||||
* Returns the subset of responses that were actually applied.
|
||||
*/
|
||||
private static List<ResponseMessage> applyResponseSequence(AbstractWriteResponseHandler handler,
|
||||
List<ResponseMessage> responses,
|
||||
ReplicaSets replicaSets)
|
||||
{
|
||||
List<ResponseMessage> appliedResponses = new ArrayList<>();
|
||||
for (ResponseMessage response : responses)
|
||||
{
|
||||
if (handler.isComplete())
|
||||
break;
|
||||
|
||||
applyResponse(handler, response, replicaSets);
|
||||
appliedResponses.add(response);
|
||||
}
|
||||
return appliedResponses;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Testing Helpers
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Creates a fresh handler for testing.
|
||||
*/
|
||||
private static AbstractWriteResponseHandler createHandler(Keyspace ks,
|
||||
ConsistencyLevel cl,
|
||||
EndpointsForToken targets,
|
||||
EndpointsForToken pending) throws Exception
|
||||
{
|
||||
ReplicaPlan.ForWrite replicaPlan = ReplicaPlans.forWrite(ks, cl,
|
||||
(cm) -> targets,
|
||||
(cm) -> pending,
|
||||
ClusterMetadata.current().epoch,
|
||||
Predicates.alwaysTrue(),
|
||||
ReplicaPlans.writeAll);
|
||||
CoordinationPlan.ForWriteWithIdeal coordinationPlan = CoordinationPlans.create(replicaPlan, null);
|
||||
return ks.getReplicationStrategy().getWriteResponseHandler(coordinationPlan, null, WriteType.SIMPLE, null,
|
||||
Dispatcher.RequestTime.forImmediateExecution());
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a single response to the handler.
|
||||
*/
|
||||
private static void applyResponse(AbstractWriteResponseHandler handler,
|
||||
ResponseMessage response,
|
||||
ReplicaSets replicaSets)
|
||||
{
|
||||
// Determine which replica set to use based on index
|
||||
EndpointsForToken targets;
|
||||
int adjustedIdx;
|
||||
|
||||
if (response.replicaIdx < replicaSets.fullReplicas.size())
|
||||
{
|
||||
// Full replica
|
||||
targets = replicaSets.fullReplicas;
|
||||
adjustedIdx = response.replicaIdx;
|
||||
}
|
||||
else
|
||||
{
|
||||
// Pending replica
|
||||
targets = replicaSets.pendingReplicas;
|
||||
adjustedIdx = response.replicaIdx - replicaSets.fullReplicas.size();
|
||||
}
|
||||
|
||||
InetAddressAndPort endpoint = targets.get(adjustedIdx).endpoint();
|
||||
|
||||
if (response.isSuccess())
|
||||
{
|
||||
Message msg = Message.builder(Verb.ECHO_REQ, noPayload)
|
||||
.from(endpoint)
|
||||
.build();
|
||||
handler.onResponse(msg);
|
||||
}
|
||||
else
|
||||
{
|
||||
handler.onFailure(endpoint,
|
||||
RequestFailure.forReason(response.failureReason));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Known bug: candidateReplicaCount() doesn't include pending replicas for LOCAL CLs.
|
||||
*
|
||||
* When local DC has >1 pending replicas:
|
||||
* - blockFor = baseQuorum + localPending (e.g., RF=3: blockFor = 2 + 2 = 4)
|
||||
* - candidateReplicaCount = localRF only (e.g., 3 full replicas)
|
||||
* - Early failure detection: blockFor + failures > candidates
|
||||
* - Result: 4 + 0 > 3, handler fails immediately even with zero failures
|
||||
*
|
||||
* With pending=1, blockFor=candidates, so bug only triggers with failures.
|
||||
* With pending>1, blockFor>candidates, so bug triggers deterministically.
|
||||
*
|
||||
* TODO: Remove this workaround after coordinator plan refactor fixes candidateReplicaCount
|
||||
*/
|
||||
public static boolean isKnownBugScenario(TopologyConfig topology, ConsistencyLevel cl, List<ResponseMessage> responses)
|
||||
{
|
||||
if (cl == ConsistencyLevel.LOCAL_ONE || cl == ConsistencyLevel.LOCAL_QUORUM)
|
||||
{
|
||||
String localDc = "datacenter1";
|
||||
int localPending = topology.pendingReplicas.get(localDc);
|
||||
|
||||
// Bug deterministically triggers when pending > 1
|
||||
return localPending > 1;
|
||||
}
|
||||
|
||||
if (cl == ConsistencyLevel.EACH_QUORUM)
|
||||
{
|
||||
// Known bug: EACH_QUORUM early failure detection doesn't work with pending replicas.
|
||||
// When a DC has pending replicas AND receives failures, the handler can get stuck
|
||||
// incomplete instead of failing early.
|
||||
//
|
||||
// Example: DC1 with RF=2, 1 pending needs 3 acks total. If it gets 2 successes and
|
||||
// 1 failure (all 3 nodes responded), it can't possibly succeed but handler doesn't fail.
|
||||
//
|
||||
// TODO: Remove after EACH_QUORUM early failure detection is fixed
|
||||
for (Map.Entry<String, Integer> entry : topology.replicationFactors.entrySet())
|
||||
{
|
||||
String dc = entry.getKey();
|
||||
int dcPending = topology.pendingReplicas.get(dc);
|
||||
|
||||
if (dcPending > 0)
|
||||
{
|
||||
// Check if this DC has any failures
|
||||
boolean hasFailures = responses.stream()
|
||||
.anyMatch(r -> {
|
||||
String respDc = getReplicaDatacenter(r.replicaIdx, topology);
|
||||
return respDc.equals(dc) && !r.isSuccess();
|
||||
});
|
||||
|
||||
if (hasFailures)
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validates handler completed with expected outcome.
|
||||
*/
|
||||
private static void validateOutcome(AbstractWriteResponseHandler handler,
|
||||
List<ResponseMessage> responses,
|
||||
TopologyConfig topology,
|
||||
ConsistencyLevel cl) throws Exception
|
||||
{
|
||||
// TODO: Remove after coordinator plan refactor fixes candidateReplicaCount
|
||||
if (isKnownBugScenario(topology, cl, responses))
|
||||
{
|
||||
if (cl == ConsistencyLevel.LOCAL_ONE || cl == ConsistencyLevel.LOCAL_QUORUM)
|
||||
{
|
||||
logger.info("[KNOWN BUG] Skipping validation: {} with {} pending replicas in datacenter1 triggers candidateReplicaCount bug (blockFor > candidates)",
|
||||
cl, topology.pendingReplicas.get("datacenter1"));
|
||||
}
|
||||
else if (cl == ConsistencyLevel.EACH_QUORUM)
|
||||
{
|
||||
logger.info("[KNOWN BUG] Skipping validation: {} with pending replicas and failures triggers early failure detection bug",
|
||||
cl);
|
||||
}
|
||||
|
||||
// Drain handler to avoid hanging futures
|
||||
try { handler.get(); } catch (Exception ignored) {}
|
||||
return;
|
||||
}
|
||||
|
||||
if (!handler.isComplete())
|
||||
{
|
||||
// Detailed diagnostics for incomplete handler
|
||||
Map<String, Integer> successesPerDc = new HashMap<>();
|
||||
Map<String, Integer> totalPerDc = new HashMap<>();
|
||||
int successCount = 0;
|
||||
|
||||
for (ResponseMessage resp : responses)
|
||||
{
|
||||
String dc = getReplicaDatacenter(resp.replicaIdx, topology);
|
||||
totalPerDc.merge(dc, 1, Integer::sum);
|
||||
if (resp.isSuccess())
|
||||
{
|
||||
successCount++;
|
||||
successesPerDc.merge(dc, 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
|
||||
StringBuilder diagnostic = new StringBuilder();
|
||||
diagnostic.append(String.format("Handler not complete after %d responses\n", responses.size()));
|
||||
diagnostic.append(String.format("Topology: %s, CL: %s, blockFor: %d\n", topology, cl, handler.blockFor()));
|
||||
diagnostic.append(String.format("Total successes: %d\n", successCount));
|
||||
diagnostic.append("Per-DC breakdown:\n");
|
||||
for (String dc : topology.replicationFactors.keySet())
|
||||
{
|
||||
int dcRf = topology.replicationFactors.get(dc);
|
||||
int dcQuorum = dcRf / 2 + 1;
|
||||
int dcSuccesses = successesPerDc.getOrDefault(dc, 0);
|
||||
int dcTotal = totalPerDc.getOrDefault(dc, 0);
|
||||
diagnostic.append(String.format(" %s: RF=%d, quorum=%d, responses=%d, successes=%d\n",
|
||||
dc, dcRf, dcQuorum, dcTotal, dcSuccesses));
|
||||
}
|
||||
throw new AssertionError(diagnostic.toString());
|
||||
}
|
||||
|
||||
// Calculate what the outcome should be based on responses
|
||||
ExpectedOutcome expected = calculateExpectedOutcome(topology, cl, responses, handler.blockFor());
|
||||
if (expected == null)
|
||||
{
|
||||
// Detailed diagnostics for model error
|
||||
int successCount = 0;
|
||||
Map<String, Integer> successesPerDc = new HashMap<>();
|
||||
for (ResponseMessage resp : responses)
|
||||
{
|
||||
if (resp.isSuccess())
|
||||
{
|
||||
successCount++;
|
||||
String dc = getReplicaDatacenter(resp.replicaIdx, topology);
|
||||
successesPerDc.merge(dc, 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
|
||||
StringBuilder diagnostic = new StringBuilder();
|
||||
diagnostic.append(String.format("Handler completed but model says incomplete - model error\n"));
|
||||
diagnostic.append(String.format("Topology: %s, CL: %s, blockFor: %d\n", topology, cl, handler.blockFor()));
|
||||
diagnostic.append(String.format("Responses applied: %d, total successes: %d\n", responses.size(), successCount));
|
||||
if (cl == ConsistencyLevel.LOCAL_ONE || cl == ConsistencyLevel.LOCAL_QUORUM)
|
||||
{
|
||||
int localSuccesses = successesPerDc.getOrDefault("datacenter1", 0);
|
||||
diagnostic.append(String.format("LOCAL DC (datacenter1) successes: %d\n", localSuccesses));
|
||||
|
||||
// Check handler type
|
||||
diagnostic.append(String.format("Handler type: %s\n", handler.getClass().getSimpleName()));
|
||||
diagnostic.append(String.format("Handler ackCount: %d\n", handler.ackCount()));
|
||||
|
||||
// Check what the locator thinks
|
||||
diagnostic.append(String.format("Locator's local DC: %s\n",
|
||||
DatabaseDescriptor.getLocalDataCenter()));
|
||||
diagnostic.append(String.format("Broadcast address: %s\n",
|
||||
DatabaseDescriptor.getBroadcastAddress()));
|
||||
|
||||
// Show which responses were applied and their DCs
|
||||
diagnostic.append("Applied responses:\n");
|
||||
|
||||
// Build endpoint lookup for replica indices
|
||||
List<InetAddressAndPort> allEndpoints = new ArrayList<>();
|
||||
handler.replicaPlan().contacts().forEach(r -> allEndpoints.add(r.endpoint()));
|
||||
handler.replicaPlan().pending().forEach(r -> allEndpoints.add(r.endpoint()));
|
||||
|
||||
for (int i = 0; i < Math.min(responses.size(), 10); i++)
|
||||
{
|
||||
ResponseMessage r = responses.get(i);
|
||||
String dc = getReplicaDatacenter(r.replicaIdx, topology);
|
||||
String type = r.replicaIdx < topology.totalReplicas ? "full" : "pending";
|
||||
|
||||
// Check what InOurDc thinks about this endpoint
|
||||
InetAddressAndPort endpoint = allEndpoints.get(r.replicaIdx);
|
||||
boolean inOurDc = InOurDc.isInOurDc(endpoint);
|
||||
|
||||
diagnostic.append(String.format(" [%d] %s, DC=%s, success=%s, InOurDc=%s\n",
|
||||
r.replicaIdx, type, dc, r.isSuccess(), inOurDc));
|
||||
}
|
||||
if (responses.size() > 10)
|
||||
diagnostic.append(String.format(" ... and %d more responses\n", responses.size() - 10));
|
||||
}
|
||||
diagnostic.append(String.format("Total endpoints: %d (full=%d, pending=%d)\n",
|
||||
topology.totalReplicas + topology.totalPending,
|
||||
topology.totalReplicas, topology.totalPending));
|
||||
diagnostic.append("Per-DC successes:\n");
|
||||
for (String dc : topology.replicationFactors.keySet())
|
||||
{
|
||||
int dcSuccesses = successesPerDc.getOrDefault(dc, 0);
|
||||
diagnostic.append(String.format(" %s: %d successes (RF=%d, pending=%d)\n",
|
||||
dc, dcSuccesses,
|
||||
topology.replicationFactors.get(dc),
|
||||
topology.pendingReplicas.get(dc)));
|
||||
}
|
||||
|
||||
// Debug: show what the handler thinks about pending replicas
|
||||
diagnostic.append(String.format("Handler's replicaPlan.pending() size: %d\n", handler.replicaPlan().pending().size()));
|
||||
for (Replica replica : handler.replicaPlan().pending())
|
||||
{
|
||||
diagnostic.append(String.format(" Pending replica: %s\n", replica.endpoint()));
|
||||
}
|
||||
|
||||
throw new AssertionError(diagnostic.toString());
|
||||
}
|
||||
|
||||
try
|
||||
{
|
||||
handler.get();
|
||||
if (expected == ExpectedOutcome.FAILURE)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Expected failure but succeeded: topology=%s, CL=%s, responses=%d",
|
||||
topology, cl, responses.size()));
|
||||
}
|
||||
}
|
||||
catch (WriteTimeoutException | WriteFailureException |
|
||||
CoordinatorBehindException | RetryOnDifferentSystemException e)
|
||||
{
|
||||
if (expected == ExpectedOutcome.SUCCESS)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Expected success but failed: topology=%s, CL=%s, responses=%d, error=%s",
|
||||
topology, cl, responses.size(), e.getMessage()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Property Tests
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void writeResponseHandlerBehavior()
|
||||
{
|
||||
qt()
|
||||
.withExamples(250)
|
||||
.forAll(testCaseGen())
|
||||
.assuming(testCase -> {
|
||||
// Filter out topologies where any scenario has insufficient replicas for its CL
|
||||
for (CLScenario scenario : testCase.scenarios)
|
||||
{
|
||||
if (testCase.topology.totalReplicas < minReplicasForCL(scenario.cl))
|
||||
return false;
|
||||
|
||||
// For EACH_QUORUM with pending, verify each DC has enough replicas
|
||||
// to satisfy: dcQuorum = (dcRf/2+1) + dcPending
|
||||
if (scenario.cl == ConsistencyLevel.EACH_QUORUM)
|
||||
{
|
||||
for (Map.Entry<String, Integer> entry : testCase.topology.replicationFactors.entrySet())
|
||||
{
|
||||
String dc = entry.getKey();
|
||||
int dcRf = entry.getValue();
|
||||
int dcPending = testCase.topology.pendingReplicas.get(dc);
|
||||
int dcQuorum = (dcRf / 2 + 1) + dcPending;
|
||||
int dcTotal = dcRf + dcPending;
|
||||
|
||||
// Skip if impossible to achieve quorum
|
||||
if (dcTotal < dcQuorum)
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
})
|
||||
.checkAssert(testCase -> {
|
||||
try
|
||||
{
|
||||
Keyspace ks = getOrCreateKeyspace(testCase.topology);
|
||||
ReplicaSets replicaSets = createReplicaSets(testCase.topology);
|
||||
|
||||
// Test all scenarios for this topology (3 random + 2 hardcoded)
|
||||
for (CLScenario scenario : testCase.scenarios)
|
||||
{
|
||||
AbstractWriteResponseHandler handler = createHandler(ks, scenario.cl, replicaSets.fullReplicas, replicaSets.pendingReplicas);
|
||||
List<ResponseMessage> appliedResponses = applyResponseSequence(handler, scenario.responses, replicaSets);
|
||||
validateOutcome(handler, appliedResponses, testCase.topology, scenario.cl);
|
||||
}
|
||||
}
|
||||
catch (Throwable e)
|
||||
{
|
||||
if (e instanceof AssertionError)
|
||||
throw (AssertionError) e;
|
||||
throw new AssertionError("Test setup failed: " + e.getMessage(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimum replicas needed for a consistency level to be valid.
|
||||
*/
|
||||
private static int minReplicasForCL(ConsistencyLevel cl)
|
||||
{
|
||||
switch (cl)
|
||||
{
|
||||
case ANY:
|
||||
case ONE:
|
||||
case LOCAL_ONE:
|
||||
return 1;
|
||||
case TWO:
|
||||
return 2;
|
||||
case THREE:
|
||||
return 3;
|
||||
case QUORUM:
|
||||
case LOCAL_QUORUM:
|
||||
case EACH_QUORUM:
|
||||
return 3;
|
||||
case ALL:
|
||||
return 1;
|
||||
default:
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -25,6 +25,8 @@ import java.util.concurrent.TimeUnit;
|
|||
|
||||
import com.google.common.base.Predicates;
|
||||
|
||||
import org.apache.cassandra.locator.*;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.junit.Before;
|
||||
import org.junit.BeforeClass;
|
||||
import org.junit.Test;
|
||||
|
|
@ -57,6 +59,7 @@ import static java.util.concurrent.TimeUnit.DAYS;
|
|||
import static org.apache.cassandra.net.NoPayload.noPayload;
|
||||
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
|
||||
import static org.junit.Assert.assertEquals;
|
||||
import static org.junit.Assert.assertFalse;
|
||||
import static org.junit.Assert.assertTrue;
|
||||
|
||||
public class WriteResponseHandlerTest
|
||||
|
|
@ -148,8 +151,8 @@ public class WriteResponseHandlerTest
|
|||
assertEquals(startingCount + 1, ks.metric.idealCLWriteLatency.latency.getCount());
|
||||
|
||||
//Don't need the others
|
||||
awr.expired();
|
||||
awr.expired();
|
||||
awr.expired(targets.get(2).endpoint());
|
||||
awr.expired(targets.get(3).endpoint());
|
||||
|
||||
assertEquals(0, ks.metric.writeFailedIdealCL.getCount());
|
||||
}
|
||||
|
|
@ -216,10 +219,9 @@ public class WriteResponseHandlerTest
|
|||
awr.onResponse(createDummyMessage(2));
|
||||
|
||||
//Fail in remote DC
|
||||
awr.expired();
|
||||
awr.expired();
|
||||
awr.expired();
|
||||
assertEquals(1, ks.metric.writeFailedIdealCL.getCount());
|
||||
awr.expired(targets.get(3).endpoint());
|
||||
awr.expired(targets.get(4).endpoint());
|
||||
awr.expired(targets.get(5).endpoint());
|
||||
assertEquals(0, ks.metric.idealCLWriteLatency.totalLatency.getCount());
|
||||
}
|
||||
|
||||
|
|
@ -260,14 +262,14 @@ public class WriteResponseHandlerTest
|
|||
|
||||
// Failure in local DC
|
||||
awr.onResponse(createDummyMessage(0));
|
||||
|
||||
awr.expired();
|
||||
awr.expired();
|
||||
|
||||
awr.expired(targets.get(1).endpoint());
|
||||
awr.expired(targets.get(2).endpoint());
|
||||
|
||||
//Fail in remote DC
|
||||
awr.expired();
|
||||
awr.expired();
|
||||
awr.expired();
|
||||
awr.expired(targets.get(3).endpoint());
|
||||
awr.expired(targets.get(4).endpoint());
|
||||
awr.expired(targets.get(5).endpoint());
|
||||
|
||||
assertEquals(startingCount, ks.metric.writeFailedIdealCL.getCount());
|
||||
}
|
||||
|
|
@ -297,6 +299,28 @@ public class WriteResponseHandlerTest
|
|||
}
|
||||
|
||||
|
||||
/**
|
||||
* expired(from) must notify the ResponseTracker so that when enough down-node expirations
|
||||
* accumulate to make quorum mathematically impossible, the handler signals failure immediately
|
||||
* rather than blocking until the full RPC timeout.
|
||||
*/
|
||||
@Test
|
||||
public void expiredUpdatesResponseTrackerAndSignalsFailureWhenQuorumImpossible()
|
||||
{
|
||||
// QUORUM on 6 replicas (3 DC1 + 3 DC2) requires 4 acks.
|
||||
// If 3 replicas are expired (down at dispatch time), only 3 remain — quorum is impossible.
|
||||
AbstractWriteResponseHandler awr = createWriteResponseHandler(ConsistencyLevel.QUORUM, null);
|
||||
|
||||
awr.expired(targets.get(0).endpoint());
|
||||
awr.expired(targets.get(1).endpoint());
|
||||
awr.expired(targets.get(2).endpoint()); // 3 remaining, blockFor=4: impossible to succeed
|
||||
|
||||
assertTrue("handler must be complete (failed) after quorum becomes impossible via expired()",
|
||||
awr.isComplete());
|
||||
assertFalse("handler must not report success",
|
||||
awr.coordinationPlan().responses().isSuccessful());
|
||||
}
|
||||
|
||||
private static AbstractWriteResponseHandler createWriteResponseHandler(ConsistencyLevel cl, ConsistencyLevel ideal)
|
||||
{
|
||||
return createWriteResponseHandler(cl, ideal, Dispatcher.RequestTime.forImmediateExecution());
|
||||
|
|
@ -304,8 +328,9 @@ public class WriteResponseHandlerTest
|
|||
|
||||
private static AbstractWriteResponseHandler createWriteResponseHandler(ConsistencyLevel cl, ConsistencyLevel ideal, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return ks.getReplicationStrategy().getWriteResponseHandler(ReplicaPlans.forWrite(ks, cl, (cm) -> targets, (cm) -> pending, Epoch.FIRST, Predicates.alwaysTrue(), ReplicaPlans.writeAll),
|
||||
null, WriteType.SIMPLE, null, requestTime, ideal);
|
||||
ReplicaPlan.ForWrite replicaPlan = ReplicaPlans.forWrite(ks, cl, (cm) -> targets, (cm) -> pending, Epoch.FIRST, Predicates.alwaysTrue(), ReplicaPlans.writeAll);
|
||||
CoordinationPlan.ForWriteWithIdeal coordinationPlan = CoordinationPlans.create(replicaPlan, ideal);
|
||||
return ks.getReplicationStrategy().getWriteResponseHandler(coordinationPlan, null, WriteType.SIMPLE, null, requestTime);
|
||||
}
|
||||
|
||||
private static Message createDummyMessage(int target)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,217 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
import org.junit.Test;
|
||||
|
||||
import static java.util.Collections.emptyMap;
|
||||
import static org.junit.Assert.*;
|
||||
|
||||
public class AugmentedCommitTest
|
||||
{
|
||||
private static final PaxosCommit.Status SUCCESS = new PaxosCommit.Status(null);
|
||||
private static final PaxosCommit.Status FAILURE = new PaxosCommit.Status(
|
||||
new Paxos.MaybeFailure(true, 3, 2, 0, emptyMap()));
|
||||
|
||||
private static PaxosCommit.AugmentedCommit<Consumer<PaxosCommit.Status>> create(AtomicReference<PaxosCommit.Status> capture)
|
||||
{
|
||||
return new PaxosCommit.AugmentedCommit<>(capture::set);
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Both succeed
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testBothSucceed_paxosFirst()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
assertNull("Should not complete with only paxos", result.get());
|
||||
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
assertNotNull("Should complete when both done", result.get());
|
||||
assertTrue("Should be success", result.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testBothSucceed_mutationFirst()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
assertNull("Should not complete with only mutation", result.get());
|
||||
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
assertNotNull("Should complete when both done", result.get());
|
||||
assertTrue("Should be success", result.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Paxos fails
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testPaxosFails_immediateCompletion()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onPaxosComplete(FAILURE);
|
||||
assertNotNull("Should complete immediately on paxos failure", result.get());
|
||||
assertFalse("Should report failure", result.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosFails_afterMutationSucceeds()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
assertNull(result.get());
|
||||
|
||||
ac.onPaxosComplete(FAILURE);
|
||||
assertNotNull("Should complete on paxos failure", result.get());
|
||||
assertFalse("Should report failure", result.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Mutation fails
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testMutationFails_immediateCompletion()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onMutationComplete(FAILURE);
|
||||
assertNotNull("Should complete immediately on mutation failure", result.get());
|
||||
assertFalse("Should report failure", result.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testMutationFails_afterPaxosSucceeds()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
assertNull(result.get());
|
||||
|
||||
ac.onMutationComplete(FAILURE);
|
||||
assertNotNull("Should complete on mutation failure", result.get());
|
||||
assertFalse("Should report failure", result.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Both fail
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testBothFail_paxosFirst()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onPaxosComplete(FAILURE);
|
||||
assertNotNull("Should complete immediately", result.get());
|
||||
assertFalse(result.get().isSuccess());
|
||||
|
||||
// Second failure is a no-op
|
||||
ac.onMutationComplete(FAILURE);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testBothFail_mutationFirst()
|
||||
{
|
||||
AtomicReference<PaxosCommit.Status> result = new AtomicReference<>();
|
||||
var ac = create(result);
|
||||
|
||||
ac.onMutationComplete(FAILURE);
|
||||
assertNotNull("Should complete immediately", result.get());
|
||||
assertFalse(result.get().isSuccess());
|
||||
|
||||
// Second failure is a no-op
|
||||
ac.onPaxosComplete(FAILURE);
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Terminal state is idempotent
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testCompleteState_ignoresFurtherUpdates()
|
||||
{
|
||||
AtomicInteger callCount = new AtomicInteger();
|
||||
var ac = new PaxosCommit.AugmentedCommit<Consumer<PaxosCommit.Status>>(s -> callCount.incrementAndGet());
|
||||
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
assertEquals("onDone should be called exactly once", 1, callCount.get());
|
||||
|
||||
// Further calls should be no-ops
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
ac.onMutationComplete(FAILURE);
|
||||
ac.onPaxosComplete(FAILURE);
|
||||
assertEquals("onDone should still be called exactly once", 1, callCount.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCompleteViaFailure_ignoresFurtherUpdates()
|
||||
{
|
||||
AtomicInteger callCount = new AtomicInteger();
|
||||
var ac = new PaxosCommit.AugmentedCommit<Consumer<PaxosCommit.Status>>(s -> callCount.incrementAndGet());
|
||||
|
||||
ac.onPaxosComplete(FAILURE);
|
||||
assertEquals(1, callCount.get());
|
||||
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
ac.onMutationComplete(FAILURE);
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
assertEquals("onDone should still be called exactly once", 1, callCount.get());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Duplicate calls to same side
|
||||
// ========================================
|
||||
|
||||
@Test(expected = IllegalStateException.class)
|
||||
public void testDuplicatePaxosComplete_throws()
|
||||
{
|
||||
var ac = create(new AtomicReference<>());
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
ac.onPaxosComplete(SUCCESS);
|
||||
}
|
||||
|
||||
@Test(expected = IllegalStateException.class)
|
||||
public void testDuplicateMutationComplete_throws()
|
||||
{
|
||||
var ac = create(new AtomicReference<>());
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
ac.onMutationComplete(SUCCESS);
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,418 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
import com.google.common.collect.ImmutableList;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.DecoratedKey;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.NoPayload;
|
||||
import org.apache.cassandra.net.Verb;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.ResponseHandlerPropertyTestBase;
|
||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||
|
||||
import static org.apache.cassandra.net.NoPayload.noPayload;
|
||||
import static org.apache.cassandra.service.paxos.Commit.Agreed;
|
||||
import static org.quicktheories.QuickTheory.qt;
|
||||
|
||||
/**
|
||||
* Property-based tests for PaxosCommit response tracking logic.
|
||||
*
|
||||
* PaxosCommit uses a single-counter model: it counts accepts and failures in a
|
||||
* flat manner (no committed vs pending distinction). DC-local filtering applies
|
||||
* for datacenter-local consistency levels.
|
||||
*
|
||||
* Tests all write consistency levels: ANY, ONE, TWO, THREE, QUORUM, ALL,
|
||||
* LOCAL_ONE, LOCAL_QUORUM, EACH_QUORUM.
|
||||
*/
|
||||
public class PaxosCommitPropertyTest extends ResponseHandlerPropertyTestBase
|
||||
{
|
||||
private static final ImmutableList<ConsistencyLevel> commitConsistencyLevels = ImmutableList.of(
|
||||
ConsistencyLevel.ANY,
|
||||
ConsistencyLevel.ONE,
|
||||
ConsistencyLevel.TWO,
|
||||
ConsistencyLevel.THREE,
|
||||
ConsistencyLevel.QUORUM,
|
||||
ConsistencyLevel.ALL,
|
||||
ConsistencyLevel.LOCAL_ONE,
|
||||
ConsistencyLevel.LOCAL_QUORUM,
|
||||
ConsistencyLevel.EACH_QUORUM
|
||||
);
|
||||
|
||||
@Override
|
||||
protected List<ConsistencyLevel> consistencyLevels()
|
||||
{
|
||||
return commitConsistencyLevels;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Independent Model
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Independent model for PaxosCommit completion.
|
||||
*
|
||||
* PaxosCommit uses a single counter:
|
||||
* - DC-local CLs filter out non-local DC responses entirely
|
||||
* - SUCCESS: accepts == required (exact match for once-only signaling)
|
||||
* - FAILURE: replicas.size() - failures == required - 1 (impossible to reach required)
|
||||
*
|
||||
* Note: replicas.size() is ALL replicas across ALL DCs, even for local CLs.
|
||||
* This asymmetry (count only local, but use global size for failure) is
|
||||
* production behavior.
|
||||
*/
|
||||
static ExpectedOutcome calculateCommitOutcome(TopologyConfig topology,
|
||||
ConsistencyLevel cl,
|
||||
List<ResponseMessage> responses,
|
||||
int required,
|
||||
int totalReplicaCount)
|
||||
{
|
||||
boolean dcLocalFilter = cl.isDatacenterLocal();
|
||||
int accepts = 0;
|
||||
int failures = 0;
|
||||
|
||||
for (ResponseMessage resp : responses)
|
||||
{
|
||||
if (dcLocalFilter)
|
||||
{
|
||||
String dc = getReplicaDatacenter(resp.replicaIdx, topology);
|
||||
if (!"datacenter1".equals(dc))
|
||||
continue;
|
||||
}
|
||||
|
||||
if (resp.isSuccess())
|
||||
accepts++;
|
||||
else
|
||||
failures++;
|
||||
}
|
||||
|
||||
// Success: reached required accepts
|
||||
if (accepts >= required)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
// Failure: impossible to reach required
|
||||
// (uses totalReplicaCount which includes all DCs, even for local CLs)
|
||||
if (totalReplicaCount - failures < required)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null; // still in progress
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Handler creation and response application
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Testable subclass that overrides the DC membership check using topology knowledge,
|
||||
* bypassing the InOurDc/Locator infrastructure which isn't fully initialized in unit tests.
|
||||
*/
|
||||
static class TestableCommit<T extends Consumer<? super PaxosCommit.Status>>
|
||||
extends PaxosCommit<T>
|
||||
{
|
||||
private final Set<InetAddressAndPort> localEndpoints;
|
||||
|
||||
TestableCommit(Agreed commit, EndpointsForToken replicas, int required,
|
||||
ConsistencyLevel consistencyForCommit, T onDone,
|
||||
Set<InetAddressAndPort> localEndpoints)
|
||||
{
|
||||
super(commit, false, consistencyForCommit, consistencyForCommit, replicas, required, onDone);
|
||||
this.localEndpoints = localEndpoints;
|
||||
}
|
||||
|
||||
@Override
|
||||
protected boolean isFromLocalDc(InetAddressAndPort endpoint)
|
||||
{
|
||||
return localEndpoints.contains(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the set of datacenter1 endpoints from the replica sets, used to override
|
||||
* DC filtering in TestableCommit.
|
||||
*/
|
||||
private static Set<InetAddressAndPort> localEndpointSet(TopologyConfig topology,
|
||||
ReplicaSets replicaSets)
|
||||
{
|
||||
Set<InetAddressAndPort> local = new HashSet<>();
|
||||
int dc1Full = topology.replicationFactors.getOrDefault("datacenter1", 0);
|
||||
|
||||
for (int i = 0; i < dc1Full; i++)
|
||||
local.add(replicaSets.fullReplicas.get(i).endpoint());
|
||||
|
||||
// pending replicas for DC1 are at indices [dc1Full, dc1Full + dc1Pending)
|
||||
// within replicaSets.pendingReplicas, but pendingReplicas is ordered per-DC too
|
||||
int pendingOffset = 0;
|
||||
for (Map.Entry<String, Integer> entry : topology.pendingReplicas.entrySet())
|
||||
{
|
||||
int count = entry.getValue();
|
||||
if ("datacenter1".equals(entry.getKey()))
|
||||
{
|
||||
for (int i = 0; i < count; i++)
|
||||
local.add(replicaSets.pendingReplicas.get(pendingOffset + i).endpoint());
|
||||
break;
|
||||
}
|
||||
pendingOffset += count;
|
||||
}
|
||||
return local;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute blockFor for PaxosCommit, matching Participants.requiredFor() behavior.
|
||||
*/
|
||||
private static int computeBlockFor(Keyspace ks, ConsistencyLevel cl, EndpointsForToken pending)
|
||||
{
|
||||
return cl.blockForWrite(ks.getReplicationStrategy(), pending);
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a PaxosCommit handler for testing.
|
||||
* Uses TestableCommit subclass to override DC filtering for LOCAL_* consistency levels.
|
||||
* Builds a synthetic Agreed with an untracked keyspace so mutation tracking is bypassed.
|
||||
*/
|
||||
private static PaxosCommit<Consumer<PaxosCommit.Status>> createCommitHandler(
|
||||
EndpointsForToken allReplicas, int required, ConsistencyLevel commitCl,
|
||||
AtomicReference<PaxosCommit.Status> statusCapture,
|
||||
TopologyConfig topology, ReplicaSets replicaSets, Keyspace ks)
|
||||
{
|
||||
Set<InetAddressAndPort> localEndpoints = localEndpointSet(topology, replicaSets);
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
DecoratedKey key = table.partitioner.decorateKey(ByteBufferUtil.bytes(0));
|
||||
Agreed commit = new Agreed(Ballot.none(), PartitionUpdate.emptyUpdate(table, key));
|
||||
return new TestableCommit<>(commit, allReplicas, required, commitCl, statusCapture::set, localEndpoints);
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies a single response to the PaxosCommit handler.
|
||||
*/
|
||||
private static void applyCommitResponse(PaxosCommit<?> handler,
|
||||
ResponseMessage response,
|
||||
ReplicaSets replicaSets)
|
||||
{
|
||||
EndpointsForToken targets;
|
||||
int adjustedIdx;
|
||||
|
||||
if (response.replicaIdx < replicaSets.fullReplicas.size())
|
||||
{
|
||||
targets = replicaSets.fullReplicas;
|
||||
adjustedIdx = response.replicaIdx;
|
||||
}
|
||||
else
|
||||
{
|
||||
targets = replicaSets.pendingReplicas;
|
||||
adjustedIdx = response.replicaIdx - replicaSets.fullReplicas.size();
|
||||
}
|
||||
|
||||
InetAddressAndPort endpoint = targets.get(adjustedIdx).endpoint();
|
||||
|
||||
if (response.isSuccess())
|
||||
{
|
||||
Message<NoPayload> msg = Message.builder(Verb.ECHO_REQ, noPayload)
|
||||
.from(endpoint)
|
||||
.build();
|
||||
handler.onResponse(msg);
|
||||
}
|
||||
else
|
||||
{
|
||||
handler.onFailure(endpoint, RequestFailure.forReason(response.failureReason));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies responses to the handler, stopping when onDone fires.
|
||||
*/
|
||||
private static List<ResponseMessage> applyCommitResponseSequence(
|
||||
PaxosCommit<?> handler,
|
||||
List<ResponseMessage> responses,
|
||||
ReplicaSets replicaSets,
|
||||
AtomicReference<PaxosCommit.Status> statusCapture)
|
||||
{
|
||||
List<ResponseMessage> appliedResponses = new ArrayList<>();
|
||||
for (ResponseMessage response : responses)
|
||||
{
|
||||
if (statusCapture.get() != null)
|
||||
break;
|
||||
|
||||
applyCommitResponse(handler, response, replicaSets);
|
||||
appliedResponses.add(response);
|
||||
}
|
||||
return appliedResponses;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Property Test
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void paxosCommitBehavior()
|
||||
{
|
||||
qt()
|
||||
.withExamples(250)
|
||||
.forAll(testCaseGen())
|
||||
.assuming(testCase -> {
|
||||
for (CLScenario scenario : testCase.scenarios)
|
||||
{
|
||||
if (testCase.topology.totalReplicas < minReplicasForCL(scenario.cl))
|
||||
return false;
|
||||
|
||||
if (scenario.cl == ConsistencyLevel.EACH_QUORUM)
|
||||
{
|
||||
for (Map.Entry<String, Integer> entry : testCase.topology.replicationFactors.entrySet())
|
||||
{
|
||||
int dcRf = entry.getValue();
|
||||
int dcPending = testCase.topology.pendingReplicas.get(entry.getKey());
|
||||
int dcQuorum = (dcRf / 2 + 1) + dcPending;
|
||||
int dcTotal = dcRf + dcPending;
|
||||
|
||||
if (dcTotal < dcQuorum)
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
})
|
||||
.checkAssert(testCase -> {
|
||||
try
|
||||
{
|
||||
Keyspace ks = getOrCreateKeyspace(testCase.topology);
|
||||
ReplicaSets replicaSets = createReplicaSets(testCase.topology);
|
||||
|
||||
// Build combined replica list (full + pending) as PaxosCommit sees it
|
||||
List<Replica> allReplicaList = new ArrayList<>();
|
||||
replicaSets.fullReplicas.forEach(allReplicaList::add);
|
||||
replicaSets.pendingReplicas.forEach(allReplicaList::add);
|
||||
EndpointsForToken allReplicas = EndpointsForToken.of(
|
||||
replicaSets.fullReplicas.token(),
|
||||
allReplicaList.toArray(new Replica[0]));
|
||||
|
||||
for (CLScenario scenario : testCase.scenarios)
|
||||
{
|
||||
int blockFor = computeBlockFor(ks, scenario.cl, replicaSets.pendingReplicas);
|
||||
|
||||
AtomicReference<PaxosCommit.Status> statusCapture = new AtomicReference<>();
|
||||
PaxosCommit<Consumer<PaxosCommit.Status>> handler =
|
||||
createCommitHandler(allReplicas, blockFor, scenario.cl, statusCapture,
|
||||
testCase.topology, replicaSets, ks);
|
||||
|
||||
List<ResponseMessage> appliedResponses =
|
||||
applyCommitResponseSequence(handler, scenario.responses, replicaSets, statusCapture);
|
||||
|
||||
validateCommitOutcome(statusCapture.get(), appliedResponses,
|
||||
testCase.topology, scenario.cl, blockFor,
|
||||
allReplicas.size());
|
||||
}
|
||||
}
|
||||
catch (Throwable e)
|
||||
{
|
||||
if (e instanceof AssertionError)
|
||||
throw (AssertionError) e;
|
||||
throw new AssertionError("Test setup failed: " + e.getMessage(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Validates the PaxosCommit outcome against the independent model.
|
||||
*/
|
||||
private static void validateCommitOutcome(PaxosCommit.Status status,
|
||||
List<ResponseMessage> responses,
|
||||
TopologyConfig topology,
|
||||
ConsistencyLevel cl,
|
||||
int blockFor,
|
||||
int totalReplicaCount)
|
||||
{
|
||||
ExpectedOutcome expected = calculateCommitOutcome(topology, cl, responses, blockFor, totalReplicaCount);
|
||||
|
||||
if (status == null)
|
||||
{
|
||||
if (expected != null)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Model says %s but PaxosCommit handler didn't complete: " +
|
||||
"topology=%s, CL=%s, blockFor=%d, responses=%d, totalReplicas=%d",
|
||||
expected, topology, cl, blockFor, responses.size(), totalReplicaCount));
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (expected == null)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"PaxosCommit handler completed but model says incomplete: " +
|
||||
"topology=%s, CL=%s, blockFor=%d, responses=%d, totalReplicas=%d, status=%s",
|
||||
topology, cl, blockFor, responses.size(), totalReplicaCount, status));
|
||||
}
|
||||
|
||||
if (expected == ExpectedOutcome.SUCCESS && !status.isSuccess())
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Expected success but got failure: topology=%s, CL=%s, blockFor=%d, responses=%d",
|
||||
topology, cl, blockFor, responses.size()));
|
||||
}
|
||||
|
||||
if (expected == ExpectedOutcome.FAILURE && status.isSuccess())
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Expected failure but got success: topology=%s, CL=%s, blockFor=%d, responses=%d",
|
||||
topology, cl, blockFor, responses.size()));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimum replicas needed for a consistency level.
|
||||
*/
|
||||
private static int minReplicasForCL(ConsistencyLevel cl)
|
||||
{
|
||||
switch (cl)
|
||||
{
|
||||
case ANY:
|
||||
case ONE:
|
||||
case LOCAL_ONE:
|
||||
return 1;
|
||||
case TWO:
|
||||
return 2;
|
||||
case THREE:
|
||||
return 3;
|
||||
case QUORUM:
|
||||
case LOCAL_QUORUM:
|
||||
case EACH_QUORUM:
|
||||
return 3;
|
||||
case ALL:
|
||||
return 1;
|
||||
default:
|
||||
throw new IllegalArgumentException("Unsupported CL: " + cl);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,352 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
import com.google.common.collect.ImmutableList;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.DecoratedKey;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.ResponseHandlerPropertyTestBase;
|
||||
import org.apache.cassandra.service.paxos.Commit.Committed;
|
||||
import org.apache.cassandra.service.paxos.PaxosPrepare.Permitted;
|
||||
import org.apache.cassandra.service.paxos.PaxosPrepare.Rejected;
|
||||
import org.apache.cassandra.service.paxos.PaxosPrepare.Response;
|
||||
import org.apache.cassandra.service.paxos.PaxosPrepare.Status;
|
||||
import org.apache.cassandra.service.paxos.PaxosState.MaybePromise;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||
import org.quicktheories.core.Gen;
|
||||
|
||||
import static java.util.Collections.emptyMap;
|
||||
import static org.quicktheories.QuickTheory.qt;
|
||||
|
||||
/**
|
||||
* Property-based tests for PaxosPrepare response tracking logic.
|
||||
* Tests the quorum counting aspects: when does a prepare achieve a quorum of
|
||||
* permissions, and when is failure due to too many failures.
|
||||
*
|
||||
* Simplified model: all responses agree on the same "latest commit" (no divergence).
|
||||
* Tests SERIAL and LOCAL_SERIAL only.
|
||||
*/
|
||||
public class PaxosPreparePropertyTest extends ResponseHandlerPropertyTestBase
|
||||
{
|
||||
private static final ImmutableList<ConsistencyLevel> CONSISTENCY_LEVELS = ImmutableList.of(
|
||||
ConsistencyLevel.SERIAL,
|
||||
ConsistencyLevel.LOCAL_SERIAL
|
||||
);
|
||||
|
||||
@Override
|
||||
protected List<ConsistencyLevel> consistencyLevels()
|
||||
{
|
||||
return CONSISTENCY_LEVELS;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Data Structures
|
||||
// ========================================
|
||||
|
||||
enum PrepareResponseType { PERMIT_WITH_LATEST, REJECT, FAIL }
|
||||
|
||||
static class PrepareResponseMessage
|
||||
{
|
||||
final int replicaIdx;
|
||||
final PrepareResponseType type;
|
||||
|
||||
PrepareResponseMessage(int replicaIdx, PrepareResponseType type)
|
||||
{
|
||||
this.replicaIdx = replicaIdx;
|
||||
this.type = type;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("replica=%d, %s", replicaIdx, type);
|
||||
}
|
||||
}
|
||||
|
||||
static class PrepareScenario
|
||||
{
|
||||
final ConsistencyLevel cl;
|
||||
final Paxos.Participants participants;
|
||||
final List<PrepareResponseMessage> responses;
|
||||
|
||||
PrepareScenario(ConsistencyLevel cl, Paxos.Participants participants,
|
||||
List<PrepareResponseMessage> responses)
|
||||
{
|
||||
this.cl = cl;
|
||||
this.participants = participants;
|
||||
this.responses = responses;
|
||||
}
|
||||
}
|
||||
|
||||
static class PrepareTestCase
|
||||
{
|
||||
final TopologyConfig topology;
|
||||
final List<PrepareScenario> scenarios;
|
||||
|
||||
PrepareTestCase(TopologyConfig topology, List<PrepareScenario> scenarios)
|
||||
{
|
||||
this.topology = topology;
|
||||
this.scenarios = scenarios;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("topology=%s, scenarios=%d", topology, scenarios.size());
|
||||
}
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Independent Model
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Simplified independent model for PaxosPrepare completion.
|
||||
*
|
||||
* All permitted responses are treated as having the latest commit.
|
||||
*
|
||||
* SUPERSEDED: any rejection received (immediate termination)
|
||||
* SUCCESS: withLatest >= sizeOfConsensusQuorum (and no rejection)
|
||||
* FAILURE: failures + sizeOfConsensusQuorum > sizeOfPoll
|
||||
* INCOMPLETE: otherwise
|
||||
*/
|
||||
static ExpectedOutcome calculatePrepareOutcome(int sizeOfPoll, int sizeOfConsensusQuorum,
|
||||
List<PrepareResponseMessage> appliedResponses)
|
||||
{
|
||||
int withLatest = 0;
|
||||
int failures = 0;
|
||||
|
||||
for (PrepareResponseMessage r : appliedResponses)
|
||||
{
|
||||
switch (r.type)
|
||||
{
|
||||
case PERMIT_WITH_LATEST: withLatest++; break;
|
||||
case REJECT: return ExpectedOutcome.FAILURE;
|
||||
case FAIL: failures++; break;
|
||||
}
|
||||
}
|
||||
|
||||
if (withLatest >= sizeOfConsensusQuorum)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
if (failures + sizeOfConsensusQuorum > sizeOfPoll)
|
||||
return ExpectedOutcome.FAILURE;
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// PaxosPrepare construction helpers
|
||||
// ========================================
|
||||
|
||||
private static Paxos.Participants buildParticipants(ConsistencyLevel cl, Keyspace ks) throws Exception
|
||||
{
|
||||
var token = Murmur3Partitioner.instance.getToken(ByteBufferUtil.bytes(0));
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
Predicate<Replica> allAlive = r -> true;
|
||||
return Paxos.Participants.get(ClusterMetadata.current(), table, token, cl, allAlive);
|
||||
}
|
||||
|
||||
private static Committed buildCommittedNone(Keyspace ks)
|
||||
{
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
DecoratedKey key = Murmur3Partitioner.instance.decorateKey(ByteBufferUtil.bytes(0));
|
||||
return Committed.none(key, table);
|
||||
}
|
||||
|
||||
private static Response buildPermitWithLatest(Committed latestCommitted)
|
||||
{
|
||||
return new Permitted(
|
||||
MaybePromise.Outcome.PROMISE, 0L, null, latestCommitted,
|
||||
null, true, emptyMap(), Epoch.EMPTY, null
|
||||
);
|
||||
}
|
||||
|
||||
private static Response buildRejected()
|
||||
{
|
||||
return new Rejected(Ballot.none());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Test Case Generator
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Generates test cases. Builds Participants from the actual replication strategy
|
||||
* to ensure quorum parameters match reality, then generates responses sized
|
||||
* to the actual electorate.
|
||||
*/
|
||||
Gen<PrepareTestCase> prepareTestCaseGen()
|
||||
{
|
||||
return random -> {
|
||||
TopologyConfig topology = topologyGen().generate(random);
|
||||
|
||||
Keyspace ks;
|
||||
try { ks = getOrCreateKeyspace(topology); }
|
||||
catch (Exception e) { throw new RuntimeException(e); }
|
||||
|
||||
List<PrepareScenario> scenarios = new ArrayList<>();
|
||||
|
||||
for (ConsistencyLevel cl : CONSISTENCY_LEVELS)
|
||||
{
|
||||
Paxos.Participants participants;
|
||||
try { participants = buildParticipants(cl, ks); }
|
||||
catch (Exception e) { continue; }
|
||||
|
||||
int pollSize = participants.sizeOfPoll();
|
||||
if (pollSize <= 0 || participants.sizeOfConsensusQuorum > pollSize)
|
||||
continue;
|
||||
|
||||
Gen<List<PrepareResponseMessage>> responseGen =
|
||||
typedResponseSequenceGen(pollSize, PrepareResponseType.values(), PrepareResponseMessage::new);
|
||||
|
||||
for (int i = 0; i < 25; i++)
|
||||
scenarios.add(new PrepareScenario(cl, participants, responseGen.generate(random)));
|
||||
|
||||
for (PrepareResponseType type : PrepareResponseType.values())
|
||||
scenarios.add(new PrepareScenario(cl, participants,
|
||||
allSameTypeSequence(pollSize, type, PrepareResponseMessage::new)));
|
||||
}
|
||||
|
||||
return new PrepareTestCase(topology, scenarios);
|
||||
};
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Property Test
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void paxosPrepareQuorumCounting()
|
||||
{
|
||||
qt()
|
||||
.withExamples(250)
|
||||
.forAll(prepareTestCaseGen())
|
||||
.assuming(testCase -> !testCase.scenarios.isEmpty())
|
||||
.checkAssert(testCase -> {
|
||||
try
|
||||
{
|
||||
Keyspace ks = getOrCreateKeyspace(testCase.topology);
|
||||
Committed committedNone = buildCommittedNone(ks);
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
DecoratedKey key = Murmur3Partitioner.instance.decorateKey(ByteBufferUtil.bytes(0));
|
||||
|
||||
for (PrepareScenario scenario : testCase.scenarios)
|
||||
{
|
||||
Paxos.Participants participants = scenario.participants;
|
||||
int consensusQuorum = participants.sizeOfConsensusQuorum;
|
||||
int pollSize = participants.sizeOfPoll();
|
||||
|
||||
AtomicReference<Status> statusCapture = new AtomicReference<>();
|
||||
PaxosPrepare prepare = new PaxosPrepare(participants,
|
||||
new PaxosPrepare.Request(Ballot.none(), participants.electorate, key, table, true, true),
|
||||
false, statusCapture::set);
|
||||
|
||||
List<PrepareResponseMessage> applied = new ArrayList<>();
|
||||
boolean completed = false;
|
||||
|
||||
for (PrepareResponseMessage msg : scenario.responses)
|
||||
{
|
||||
if (statusCapture.get() != null)
|
||||
{
|
||||
completed = true;
|
||||
break;
|
||||
}
|
||||
|
||||
InetAddressAndPort from = participants.voter(msg.replicaIdx);
|
||||
|
||||
switch (msg.type)
|
||||
{
|
||||
case PERMIT_WITH_LATEST:
|
||||
prepare.onResponse(buildPermitWithLatest(committedNone), from);
|
||||
break;
|
||||
case REJECT:
|
||||
prepare.onResponse(buildRejected(), from);
|
||||
break;
|
||||
case FAIL:
|
||||
prepare.onFailure(from, RequestFailure.forReason(RequestFailureReason.TIMEOUT));
|
||||
break;
|
||||
}
|
||||
|
||||
applied.add(msg);
|
||||
|
||||
if (statusCapture.get() != null)
|
||||
completed = true;
|
||||
|
||||
ExpectedOutcome expected = calculatePrepareOutcome(pollSize, consensusQuorum, applied);
|
||||
|
||||
if (expected != null && !completed)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Model says %s but PaxosPrepare not complete: topology=%s, CL=%s, " +
|
||||
"applied=%d, quorum=%d/%d, responses=%s",
|
||||
expected, testCase.topology, scenario.cl,
|
||||
applied.size(), consensusQuorum, pollSize, applied));
|
||||
}
|
||||
|
||||
if (expected == null && completed)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"PaxosPrepare completed but model says incomplete: topology=%s, CL=%s, " +
|
||||
"applied=%d, quorum=%d/%d, status=%s, responses=%s",
|
||||
testCase.topology, scenario.cl,
|
||||
applied.size(), consensusQuorum, pollSize,
|
||||
statusCapture.get(), applied));
|
||||
}
|
||||
}
|
||||
|
||||
if (!completed)
|
||||
{
|
||||
ExpectedOutcome expected = calculatePrepareOutcome(pollSize, consensusQuorum, applied);
|
||||
if (expected != null)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"After all %d responses, model says %s but PaxosPrepare not complete: " +
|
||||
"topology=%s, CL=%s, quorum=%d/%d",
|
||||
applied.size(), expected, testCase.topology, scenario.cl,
|
||||
consensusQuorum, pollSize));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
catch (Throwable e)
|
||||
{
|
||||
if (e instanceof AssertionError)
|
||||
throw (AssertionError) e;
|
||||
throw new AssertionError("Test setup failed: " + e.getMessage(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,341 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
import com.google.common.collect.ImmutableList;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.DecoratedKey;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.ResponseHandlerPropertyTestBase;
|
||||
import org.apache.cassandra.service.paxos.Commit.Proposal;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||
import org.quicktheories.core.Gen;
|
||||
|
||||
import static org.quicktheories.QuickTheory.qt;
|
||||
|
||||
/**
|
||||
* Property-based tests for PaxosPropose response tracking logic.
|
||||
* Tests through actual PaxosPropose instances, calling onResponse/onFailure
|
||||
* and checking that the onDone callback fires at the correct point.
|
||||
*
|
||||
* Tests SERIAL and LOCAL_SERIAL only. Uses an independent model for the quorum math.
|
||||
*/
|
||||
public class PaxosProposePropertyTest extends ResponseHandlerPropertyTestBase
|
||||
{
|
||||
private static final ImmutableList<ConsistencyLevel> CONSISTENCY_LEVELS = ImmutableList.of(
|
||||
ConsistencyLevel.SERIAL,
|
||||
ConsistencyLevel.LOCAL_SERIAL
|
||||
);
|
||||
|
||||
@Override
|
||||
protected List<ConsistencyLevel> consistencyLevels()
|
||||
{
|
||||
return CONSISTENCY_LEVELS;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Data Structures
|
||||
// ========================================
|
||||
|
||||
enum ProposeResponseType { ACCEPT, REFUSE, FAIL }
|
||||
|
||||
static class ProposeResponseMessage
|
||||
{
|
||||
final int replicaIdx;
|
||||
final ProposeResponseType type;
|
||||
|
||||
ProposeResponseMessage(int replicaIdx, ProposeResponseType type)
|
||||
{
|
||||
this.replicaIdx = replicaIdx;
|
||||
this.type = type;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("replica=%d, %s", replicaIdx, type);
|
||||
}
|
||||
}
|
||||
|
||||
static class ProposeScenario
|
||||
{
|
||||
final ConsistencyLevel cl;
|
||||
final Paxos.Participants participants;
|
||||
final List<ProposeResponseMessage> responses;
|
||||
|
||||
ProposeScenario(ConsistencyLevel cl, Paxos.Participants participants,
|
||||
List<ProposeResponseMessage> responses)
|
||||
{
|
||||
this.cl = cl;
|
||||
this.participants = participants;
|
||||
this.responses = responses;
|
||||
}
|
||||
}
|
||||
|
||||
static class ProposeTestCase
|
||||
{
|
||||
final TopologyConfig topology;
|
||||
final List<ProposeScenario> scenarios;
|
||||
|
||||
ProposeTestCase(TopologyConfig topology, List<ProposeScenario> scenarios)
|
||||
{
|
||||
this.topology = topology;
|
||||
this.scenarios = scenarios;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("topology=%s, scenarios=%d", topology, scenarios.size());
|
||||
}
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Independent Model
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Independent model for PaxosPropose completion.
|
||||
*
|
||||
* Rules:
|
||||
* - SUCCESS: accepts >= required
|
||||
* - Can still succeed: refusals == 0 AND required <= participants - failures
|
||||
* - FAILURE: cannot succeed (any refusal, or too many failures)
|
||||
*/
|
||||
static ExpectedOutcome calculateProposeOutcome(int participants, int required,
|
||||
List<ProposeResponseMessage> appliedResponses)
|
||||
{
|
||||
int accepts = 0, refusals = 0, failures = 0;
|
||||
for (ProposeResponseMessage r : appliedResponses)
|
||||
{
|
||||
switch (r.type)
|
||||
{
|
||||
case ACCEPT: accepts++; break;
|
||||
case REFUSE: refusals++; break;
|
||||
case FAIL: failures++; break;
|
||||
}
|
||||
}
|
||||
|
||||
if (accepts >= required)
|
||||
return ExpectedOutcome.SUCCESS;
|
||||
|
||||
boolean canSucceed = refusals == 0 && required <= participants - failures;
|
||||
if (canSucceed)
|
||||
return null; // still in progress
|
||||
|
||||
return ExpectedOutcome.FAILURE;
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// PaxosPropose construction helpers
|
||||
// ========================================
|
||||
|
||||
private static Paxos.Participants buildParticipants(ConsistencyLevel cl, Keyspace ks) throws Exception
|
||||
{
|
||||
var token = Murmur3Partitioner.instance.getToken(ByteBufferUtil.bytes(0));
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
Predicate<Replica> allAlive = r -> true;
|
||||
return Paxos.Participants.get(ClusterMetadata.current(), table, token, cl, allAlive);
|
||||
}
|
||||
|
||||
private static Proposal buildEmptyProposal(Keyspace ks)
|
||||
{
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
DecoratedKey key = Murmur3Partitioner.instance.decorateKey(ByteBufferUtil.bytes(0));
|
||||
return Proposal.empty(Ballot.none(), key, table);
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Test Case Generator
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Generates test cases. Builds Participants from the actual replication strategy
|
||||
* to ensure quorum parameters match reality, then generates responses sized
|
||||
* to the actual electorate.
|
||||
*/
|
||||
Gen<ProposeTestCase> proposeTestCaseGen()
|
||||
{
|
||||
return random -> {
|
||||
TopologyConfig topology = topologyGen().generate(random);
|
||||
|
||||
Keyspace ks;
|
||||
try
|
||||
{
|
||||
ks = getOrCreateKeyspace(topology);
|
||||
}
|
||||
catch (Exception e)
|
||||
{
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
|
||||
List<ProposeScenario> scenarios = new ArrayList<>();
|
||||
|
||||
for (ConsistencyLevel cl : CONSISTENCY_LEVELS)
|
||||
{
|
||||
Paxos.Participants participants;
|
||||
try
|
||||
{
|
||||
participants = buildParticipants(cl, ks);
|
||||
}
|
||||
catch (Exception e)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
int pollSize = participants.sizeOfPoll();
|
||||
if (pollSize <= 0 || participants.sizeOfConsensusQuorum > pollSize)
|
||||
continue;
|
||||
|
||||
Gen<List<ProposeResponseMessage>> responseGen =
|
||||
typedResponseSequenceGen(pollSize, ProposeResponseType.values(), ProposeResponseMessage::new);
|
||||
|
||||
for (int i = 0; i < 25; i++)
|
||||
scenarios.add(new ProposeScenario(cl, participants, responseGen.generate(random)));
|
||||
|
||||
for (ProposeResponseType type : ProposeResponseType.values())
|
||||
scenarios.add(new ProposeScenario(cl, participants,
|
||||
allSameTypeSequence(pollSize, type, ProposeResponseMessage::new)));
|
||||
}
|
||||
|
||||
return new ProposeTestCase(topology, scenarios);
|
||||
};
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Property Test
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void paxosProposeSignaling()
|
||||
{
|
||||
qt()
|
||||
.withExamples(250)
|
||||
.forAll(proposeTestCaseGen())
|
||||
.assuming(testCase -> !testCase.scenarios.isEmpty())
|
||||
.checkAssert(testCase -> {
|
||||
try
|
||||
{
|
||||
Keyspace ks = getOrCreateKeyspace(testCase.topology);
|
||||
Proposal proposal = buildEmptyProposal(ks);
|
||||
|
||||
for (ProposeScenario scenario : testCase.scenarios)
|
||||
verifyScenario(testCase.topology, scenario, proposal);
|
||||
}
|
||||
catch (Throwable e)
|
||||
{
|
||||
if (e instanceof AssertionError)
|
||||
throw (AssertionError) e;
|
||||
throw new AssertionError("Test setup failed: " + e.getMessage(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
private void verifyScenario(TopologyConfig topology, ProposeScenario scenario, Proposal proposal)
|
||||
{
|
||||
Paxos.Participants participants = scenario.participants;
|
||||
int consensusQuorum = participants.sizeOfConsensusQuorum;
|
||||
int pollSize = participants.sizeOfPoll();
|
||||
|
||||
AtomicReference<PaxosPropose.Status> statusCapture = new AtomicReference<>();
|
||||
PaxosPropose<Consumer<PaxosPropose.Status>> propose = new PaxosPropose<>(
|
||||
proposal, pollSize, consensusQuorum, statusCapture::set);
|
||||
|
||||
List<ProposeResponseMessage> applied = new ArrayList<>();
|
||||
boolean completed = false;
|
||||
|
||||
for (ProposeResponseMessage msg : scenario.responses)
|
||||
{
|
||||
if (statusCapture.get() != null)
|
||||
{
|
||||
completed = true;
|
||||
break;
|
||||
}
|
||||
|
||||
InetAddressAndPort from = participants.voter(msg.replicaIdx);
|
||||
|
||||
switch (msg.type)
|
||||
{
|
||||
case ACCEPT:
|
||||
propose.onResponse(PaxosState.AcceptResult.SUCCESS, from);
|
||||
break;
|
||||
case REFUSE:
|
||||
propose.onResponse(new PaxosState.AcceptResult(Ballot.none()), from);
|
||||
break;
|
||||
case FAIL:
|
||||
propose.onFailure(from, RequestFailure.forReason(RequestFailureReason.TIMEOUT));
|
||||
break;
|
||||
}
|
||||
|
||||
applied.add(msg);
|
||||
|
||||
if (statusCapture.get() != null)
|
||||
completed = true;
|
||||
|
||||
ExpectedOutcome expected = calculateProposeOutcome(pollSize, consensusQuorum, applied);
|
||||
|
||||
if (expected != null && !completed)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"Model says %s but PaxosPropose not complete: topology=%s, CL=%s, " +
|
||||
"applied=%d, quorum=%d/%d, responses=%s",
|
||||
expected, topology, scenario.cl,
|
||||
applied.size(), consensusQuorum, pollSize, applied));
|
||||
}
|
||||
|
||||
if (expected == null && completed)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"PaxosPropose completed but model says incomplete: topology=%s, CL=%s, " +
|
||||
"applied=%d, quorum=%d/%d, status=%s, responses=%s",
|
||||
topology, scenario.cl,
|
||||
applied.size(), consensusQuorum, pollSize,
|
||||
statusCapture.get(), applied));
|
||||
}
|
||||
}
|
||||
|
||||
if (!completed)
|
||||
{
|
||||
ExpectedOutcome expected = calculateProposeOutcome(pollSize, consensusQuorum, applied);
|
||||
if (expected != null)
|
||||
{
|
||||
throw new AssertionError(String.format(
|
||||
"After all %d responses, model says %s but PaxosPropose not complete: " +
|
||||
"topology=%s, CL=%s, quorum=%d/%d",
|
||||
applied.size(), expected, topology, scenario.cl,
|
||||
consensusQuorum, pollSize));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,286 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.apache.cassandra.service.paxos;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
import org.junit.BeforeClass;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.SchemaLoader;
|
||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||
import org.apache.cassandra.db.ConsistencyLevel;
|
||||
import org.apache.cassandra.db.DecoratedKey;
|
||||
import org.apache.cassandra.db.Keyspace;
|
||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.net.NoPayload;
|
||||
import org.apache.cassandra.net.Verb;
|
||||
import org.apache.cassandra.schema.KeyspaceParams;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.service.paxos.PaxosCommitPropertyTest.TestableCommit;
|
||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||
import org.apache.cassandra.utils.concurrent.AsyncPromise;
|
||||
import org.apache.cassandra.utils.concurrent.ImmediateFuture;
|
||||
|
||||
import static org.apache.cassandra.net.NoPayload.noPayload;
|
||||
import static org.apache.cassandra.service.paxos.Commit.Agreed;
|
||||
import static org.junit.Assert.assertFalse;
|
||||
import static org.junit.Assert.assertNotNull;
|
||||
import static org.junit.Assert.assertNull;
|
||||
import static org.junit.Assert.assertTrue;
|
||||
|
||||
/**
|
||||
* Tests for PaxosCommit's augmented commit composition logic.
|
||||
*
|
||||
* Verifies that onDone fires correctly when paxos consensus is combined with an additional
|
||||
* commit future (from the replication strategy, e.g. satellite DC writes for SRS).
|
||||
* Either side failing should cause immediate completion with failure.
|
||||
*/
|
||||
public class SatellitePaxosCommitTest
|
||||
{
|
||||
private static final String KEYSPACE = "spc_test";
|
||||
|
||||
@BeforeClass
|
||||
public static void setup() throws Exception
|
||||
{
|
||||
SchemaLoader.loadSchema();
|
||||
SchemaLoader.createKeyspace(KEYSPACE, KeyspaceParams.simple(3),
|
||||
SchemaLoader.standardCFMD(KEYSPACE, "Standard"));
|
||||
}
|
||||
|
||||
private static TestableCommit<Consumer<PaxosCommit.Status>> createHandler(EndpointsForToken replicas,
|
||||
int required,
|
||||
AtomicReference<PaxosCommit.Status> statusCapture)
|
||||
{
|
||||
Keyspace ks = Keyspace.open(KEYSPACE);
|
||||
TableMetadata table = ks.getColumnFamilyStores().iterator().next().metadata();
|
||||
DecoratedKey key = table.partitioner.decorateKey(ByteBufferUtil.bytes(0));
|
||||
Agreed commit = new Agreed(Ballot.none(), PartitionUpdate.emptyUpdate(table, key));
|
||||
return new TestableCommit<>(commit, replicas, required,
|
||||
ConsistencyLevel.QUORUM, statusCapture::set,
|
||||
Collections.singleton(replicas.get(0).endpoint()));
|
||||
}
|
||||
|
||||
private static EndpointsForToken threeReplicas() throws Exception
|
||||
{
|
||||
InetAddressAndPort ep1 = InetAddressAndPort.getByName("127.0.0.1");
|
||||
InetAddressAndPort ep2 = InetAddressAndPort.getByName("127.0.0.2");
|
||||
InetAddressAndPort ep3 = InetAddressAndPort.getByName("127.0.0.3");
|
||||
Token minToken = DatabaseDescriptor.getPartitioner().getMinimumToken();
|
||||
Token maxToken = DatabaseDescriptor.getPartitioner().getRandomToken();
|
||||
return EndpointsForToken.of(maxToken,
|
||||
Replica.fullReplica(ep1, minToken, maxToken),
|
||||
Replica.fullReplica(ep2, minToken, maxToken),
|
||||
Replica.fullReplica(ep3, minToken, maxToken));
|
||||
}
|
||||
|
||||
private static void sendSuccess(PaxosCommit<?> handler, InetAddressAndPort from)
|
||||
{
|
||||
Message<NoPayload> msg = Message.builder(Verb.ECHO_REQ, noPayload)
|
||||
.from(from)
|
||||
.build();
|
||||
handler.onResponse(msg);
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// No augmented commit (default behavior)
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testNoAugmentedCommit_paxosQuorumFiresOnDone() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
sendSuccess(handler, replicas.get(0).endpoint());
|
||||
assertNull("Should not fire after 1 response", status.get());
|
||||
|
||||
sendSuccess(handler, replicas.get(1).endpoint());
|
||||
assertNotNull("Should fire after quorum", status.get());
|
||||
assertTrue("Should be success", status.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Already-completed futures
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testCompletedSuccessFuture_paxosQuorumFiresOnDone() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
handler.setAugmentedCommitFuture(ImmediateFuture.success(null));
|
||||
|
||||
sendSuccess(handler, replicas.get(0).endpoint());
|
||||
assertNull(status.get());
|
||||
|
||||
sendSuccess(handler, replicas.get(1).endpoint());
|
||||
assertNotNull("Should fire after quorum (future already done)", status.get());
|
||||
assertTrue("Should be success", status.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testCompletedFailureFuture_failsImmediately() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
AsyncPromise<Void> failed = new AsyncPromise<>();
|
||||
failed.tryFailure(new RuntimeException("satellite quorum not met"));
|
||||
handler.setAugmentedCommitFuture(failed);
|
||||
|
||||
// Future already failed — onDone should fire immediately
|
||||
assertNotNull("Should fire immediately on failed future", status.get());
|
||||
assertFalse("Should report failure", status.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Paxos completes first
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testPaxosSucceedsFirst_defersUntilFutureSucceeds() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
AsyncPromise<Void> promise = new AsyncPromise<>();
|
||||
handler.setAugmentedCommitFuture(promise);
|
||||
|
||||
sendSuccess(handler, replicas.get(0).endpoint());
|
||||
sendSuccess(handler, replicas.get(1).endpoint());
|
||||
assertNull("onDone should NOT fire yet (future pending)", status.get());
|
||||
|
||||
promise.trySuccess(null);
|
||||
assertNotNull("onDone should fire after future resolves", status.get());
|
||||
assertTrue("Should be success", status.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosSucceedsFirst_futureFailsCausesFailure() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
AsyncPromise<Void> promise = new AsyncPromise<>();
|
||||
handler.setAugmentedCommitFuture(promise);
|
||||
|
||||
sendSuccess(handler, replicas.get(0).endpoint());
|
||||
sendSuccess(handler, replicas.get(1).endpoint());
|
||||
assertNull("onDone deferred", status.get());
|
||||
|
||||
promise.tryFailure(new RuntimeException("satellite quorum not met"));
|
||||
assertNotNull("onDone should fire", status.get());
|
||||
assertFalse("Should report failure", status.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testPaxosFailsFirst_failsImmediately() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
AsyncPromise<Void> promise = new AsyncPromise<>();
|
||||
handler.setAugmentedCommitFuture(promise);
|
||||
|
||||
// Paxos fails (enough failures to make quorum impossible)
|
||||
handler.onFailure(replicas.get(0).endpoint(), RequestFailure.UNKNOWN);
|
||||
handler.onFailure(replicas.get(1).endpoint(), RequestFailure.UNKNOWN);
|
||||
|
||||
// Paxos failure should fire onDone immediately without waiting for the future
|
||||
assertNotNull("onDone should fire immediately on paxos failure", status.get());
|
||||
assertFalse("Should report paxos failure", status.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Future completes first
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testFutureSucceedsFirst_defersUntilPaxosSucceeds() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
AsyncPromise<Void> promise = new AsyncPromise<>();
|
||||
handler.setAugmentedCommitFuture(promise);
|
||||
|
||||
promise.trySuccess(null);
|
||||
assertNull("onDone should NOT fire yet (paxos not done)", status.get());
|
||||
|
||||
sendSuccess(handler, replicas.get(0).endpoint());
|
||||
assertNull(status.get());
|
||||
sendSuccess(handler, replicas.get(1).endpoint());
|
||||
assertNotNull("onDone should fire after paxos quorum", status.get());
|
||||
assertTrue("Should be success", status.get().isSuccess());
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testFutureFailsFirst_failsImmediately() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
AsyncPromise<Void> promise = new AsyncPromise<>();
|
||||
handler.setAugmentedCommitFuture(promise);
|
||||
|
||||
promise.tryFailure(new RuntimeException("satellite quorum not met"));
|
||||
|
||||
// Future failure should fire onDone immediately without waiting for paxos
|
||||
assertNotNull("onDone should fire immediately on future failure", status.get());
|
||||
assertFalse("Should report failure", status.get().isSuccess());
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Null future (no-op)
|
||||
// ========================================
|
||||
|
||||
@Test
|
||||
public void testNullFuture_behavesLikeNoAugmentedCommit() throws Exception
|
||||
{
|
||||
EndpointsForToken replicas = threeReplicas();
|
||||
AtomicReference<PaxosCommit.Status> status = new AtomicReference<>();
|
||||
PaxosCommit<?> handler = createHandler(replicas, 2, status);
|
||||
|
||||
handler.setAugmentedCommitFuture(null);
|
||||
|
||||
sendSuccess(handler, replicas.get(0).endpoint());
|
||||
assertNull(status.get());
|
||||
|
||||
sendSuccess(handler, replicas.get(1).endpoint());
|
||||
assertNotNull("Should fire after quorum", status.get());
|
||||
assertTrue("Should be success", status.get().isSuccess());
|
||||
}
|
||||
}
|
||||
|
|
@ -58,6 +58,8 @@ import org.apache.cassandra.db.rows.Row;
|
|||
import org.apache.cassandra.db.rows.RowIterator;
|
||||
import org.apache.cassandra.dht.Murmur3Partitioner;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
|
|
@ -1250,7 +1252,7 @@ public class DataResolverTest extends AbstractReadResponseTest
|
|||
}
|
||||
|
||||
private DataResolver resolverWithVerifier(final ReadCommand command,
|
||||
final ReplicaPlan.SharedForRangeRead plan,
|
||||
final CoordinationPlan.ForRangeRead plan,
|
||||
final ReadRepair readRepair,
|
||||
final Dispatcher.RequestTime requestTime,
|
||||
final RepairedDataVerifier verifier)
|
||||
|
|
@ -1258,7 +1260,7 @@ public class DataResolverTest extends AbstractReadResponseTest
|
|||
class TestableDataResolver extends DataResolver
|
||||
{
|
||||
|
||||
public TestableDataResolver(ReadCommand command, ReplicaPlan.SharedForRangeRead plan, ReadRepair readRepair, Dispatcher.RequestTime requestTime)
|
||||
public TestableDataResolver(ReadCommand command, CoordinationPlan.ForRangeRead plan, ReadRepair readRepair, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(ReadCoordinator.DEFAULT, command, plan, readRepair, requestTime, true);
|
||||
}
|
||||
|
|
@ -1324,17 +1326,17 @@ public class DataResolverTest extends AbstractReadResponseTest
|
|||
assertEquals(update.metadata().name, cfm.name);
|
||||
}
|
||||
|
||||
private ReplicaPlan.SharedForRangeRead plan(EndpointsForRange replicas, ConsistencyLevel consistencyLevel)
|
||||
private CoordinationPlan.ForRangeRead plan(EndpointsForRange replicas, ConsistencyLevel consistencyLevel)
|
||||
{
|
||||
BiFunction<ReplicaPlan<?, ?>, Token, ReplicaPlan.ForWrite> repairPlan = (self, t) -> ReplicaPlans.forReadRepair(self, ClusterMetadata.current(), ks, null, consistencyLevel, t, (i) -> true, ReadCoordinator.DEFAULT);
|
||||
return ReplicaPlan.shared(new ReplicaPlan.ForRangeRead(ks,
|
||||
ks.getReplicationStrategy(),
|
||||
consistencyLevel,
|
||||
ReplicaUtils.FULL_BOUNDS,
|
||||
replicas, replicas, replicas,
|
||||
1, null,
|
||||
repairPlan,
|
||||
Epoch.EMPTY));
|
||||
return CoordinationPlans.create(ReplicaPlan.shared(new ReplicaPlan.ForRangeRead(ks,
|
||||
ks.getReplicationStrategy(),
|
||||
consistencyLevel,
|
||||
ReplicaUtils.FULL_BOUNDS,
|
||||
replicas, replicas, replicas,
|
||||
1, null,
|
||||
repairPlan,
|
||||
Epoch.EMPTY)));
|
||||
}
|
||||
|
||||
private static void resolveAndConsume(DataResolver resolver)
|
||||
|
|
|
|||
|
|
@ -23,6 +23,9 @@ import java.util.concurrent.ExecutorService;
|
|||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.junit.Assert;
|
||||
import org.junit.Test;
|
||||
|
||||
|
|
@ -32,7 +35,6 @@ import org.apache.cassandra.db.SinglePartitionReadCommand;
|
|||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||
import org.apache.cassandra.db.rows.Row;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.schema.TableMetadata;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
import org.apache.cassandra.transport.Dispatcher;
|
||||
|
|
@ -91,7 +93,7 @@ public class DigestResolverTest extends AbstractReadResponseTest
|
|||
SinglePartitionReadCommand command = SinglePartitionReadCommand.fullPartitionRead(cfm, nowInSec, dk);
|
||||
EndpointsForToken targetReplicas = EndpointsForToken.of(dk.getToken(), full(EP1), full(EP2));
|
||||
PartitionUpdate response = update(row(1000, 4, 4), row(1000, 5, 5)).build();
|
||||
ReplicaPlan.SharedForTokenRead plan = plan(ConsistencyLevel.ONE, targetReplicas);
|
||||
CoordinationPlan.ForTokenRead plan = plan(ConsistencyLevel.ONE, targetReplicas);
|
||||
|
||||
ExecutorService pool = Executors.newFixedThreadPool(2);
|
||||
long endTime = System.nanoTime() + TimeUnit.MINUTES.toNanos(2);
|
||||
|
|
@ -213,9 +215,9 @@ public class DigestResolverTest extends AbstractReadResponseTest
|
|||
resolver.getData());
|
||||
}
|
||||
|
||||
private ReplicaPlan.SharedForTokenRead plan(ConsistencyLevel consistencyLevel, EndpointsForToken replicas)
|
||||
private CoordinationPlan.ForTokenRead plan(ConsistencyLevel consistencyLevel, EndpointsForToken replicas)
|
||||
{
|
||||
return ReplicaPlan.shared(new ReplicaPlan.ForTokenRead(ks, ks.getReplicationStrategy(), consistencyLevel, replicas, replicas, replicas, null, (self) -> null, Epoch.EMPTY));
|
||||
return CoordinationPlans.create(new ReplicaPlan.ForTokenRead(ks, ks.getReplicationStrategy(), consistencyLevel, replicas, replicas, replicas, null, (self) -> null, Epoch.EMPTY));
|
||||
}
|
||||
|
||||
private void waitForLatch(CountDownLatch startlatch)
|
||||
|
|
|
|||
|
|
@ -39,6 +39,8 @@ import org.apache.cassandra.exceptions.ReadFailureException;
|
|||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||
import org.apache.cassandra.exceptions.RequestFailure;
|
||||
import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
|
|
@ -207,7 +209,7 @@ public class ReadExecutorTest
|
|||
public void testRaceWithNonSpeculativeFailure()
|
||||
{
|
||||
MockSinglePartitionReadCommand command = new MockSinglePartitionReadCommand(TimeUnit.DAYS.toMillis(365));
|
||||
ReplicaPlan.ForTokenRead plan = plan(ConsistencyLevel.LOCAL_ONE, targets, targets.subList(0, 1));
|
||||
CoordinationPlan.ForTokenRead plan = plan(ConsistencyLevel.LOCAL_ONE, targets, targets.subList(0, 1));
|
||||
AbstractReadExecutor executor = new AbstractReadExecutor.SpeculatingReadExecutor(ReadCoordinator.DEFAULT, cfs, command, plan, Dispatcher.RequestTime.forImmediateExecution());
|
||||
|
||||
// Issue an initial request against the first endpoint...
|
||||
|
|
@ -271,13 +273,13 @@ public class ReadExecutorTest
|
|||
}
|
||||
}
|
||||
|
||||
private ReplicaPlan.ForTokenRead plan(EndpointsForToken targets, ConsistencyLevel consistencyLevel)
|
||||
private CoordinationPlan.ForTokenRead plan(EndpointsForToken targets, ConsistencyLevel consistencyLevel)
|
||||
{
|
||||
return plan(consistencyLevel, targets, targets);
|
||||
}
|
||||
|
||||
private ReplicaPlan.ForTokenRead plan(ConsistencyLevel consistencyLevel, EndpointsForToken natural, EndpointsForToken selected)
|
||||
private CoordinationPlan.ForTokenRead plan(ConsistencyLevel consistencyLevel, EndpointsForToken natural, EndpointsForToken selected)
|
||||
{
|
||||
return new ReplicaPlan.ForTokenRead(ks, ks.getReplicationStrategy(), consistencyLevel, natural, selected, natural, (cm) -> null, (self) -> null, Epoch.EMPTY);
|
||||
return CoordinationPlans.create(new ReplicaPlan.ForTokenRead(ks, ks.getReplicationStrategy(), consistencyLevel, natural, selected, natural, (cm) -> null, (self) -> null, Epoch.EMPTY));
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ import static org.apache.cassandra.Util.rp;
|
|||
import static org.apache.cassandra.Util.token;
|
||||
import static org.junit.Assert.assertEquals;
|
||||
|
||||
public class ReplicaPlanIteratorTest
|
||||
public class CoordinationPlanIteratorTest
|
||||
{
|
||||
private static final String KEYSPACE = "ReplicaPlanIteratorTest";
|
||||
private static final TableId TABLE_ID = TableId.generate();
|
||||
|
|
@ -165,11 +165,11 @@ public class ReplicaPlanIteratorTest
|
|||
@SafeVarargs
|
||||
private final void testRanges(Keyspace keyspace, AbstractBounds<PartitionPosition> queryRange, AbstractBounds<PartitionPosition>... expected)
|
||||
{
|
||||
try (ReplicaPlanIterator iterator = new ReplicaPlanIterator(queryRange, null, keyspace, TABLE_ID, ConsistencyLevel.ANY))
|
||||
try (CoordinationPlanIterator iterator = new CoordinationPlanIterator(queryRange, null, keyspace, TABLE_ID, ConsistencyLevel.ANY))
|
||||
{
|
||||
List<AbstractBounds<PartitionPosition>> restrictedRanges = new ArrayList<>(expected.length);
|
||||
while (iterator.hasNext())
|
||||
restrictedRanges.add(iterator.next().range());
|
||||
restrictedRanges.add(iterator.next().replicas().range());
|
||||
|
||||
// verify range counts
|
||||
assertEquals(expected.length, restrictedRanges.size());
|
||||
|
|
@ -63,9 +63,9 @@ import static org.junit.Assert.assertEquals;
|
|||
import static org.junit.Assert.assertFalse;
|
||||
|
||||
/**
|
||||
* Tests for {@link ReplicaPlanMerger}.
|
||||
* Tests for {@link CoordinationPlanMerger}.
|
||||
*/
|
||||
public class ReplicaPlanMergerTest
|
||||
public class CoordinationPlanMergerTest
|
||||
{
|
||||
private static final String KEYSPACE = "ReplicaPlanMergerTest";
|
||||
private static Keyspace keyspace;
|
||||
|
|
@ -416,13 +416,13 @@ public class ReplicaPlanMergerTest
|
|||
AbstractBounds<PartitionPosition> queryRange,
|
||||
AbstractBounds<PartitionPosition>... expected)
|
||||
{
|
||||
try (ReplicaPlanIterator originals = new ReplicaPlanIterator(queryRange, null, keyspace, null, ANY); // ANY avoids endpoint erros
|
||||
ReplicaPlanMerger merger = new ReplicaPlanMerger(originals, keyspace, null, consistencyLevel))
|
||||
try (CoordinationPlanIterator originals = new CoordinationPlanIterator(queryRange, null, keyspace, null, ANY); // ANY avoids endpoint erros
|
||||
CoordinationPlanMerger merger = new CoordinationPlanMerger(originals, keyspace, null, consistencyLevel))
|
||||
{
|
||||
// collect the merged ranges
|
||||
List<AbstractBounds<PartitionPosition>> mergedRanges = new ArrayList<>(expected.length);
|
||||
while (merger.hasNext())
|
||||
mergedRanges.add(merger.next().range());
|
||||
mergedRanges.add(merger.next().replicas().range());
|
||||
|
||||
assertFalse("The number of merged ranges should never be greater than the number of original ranges",
|
||||
mergedRanges.size() > originals.size());
|
||||
|
|
@ -38,7 +38,8 @@ import org.apache.cassandra.dht.AbstractBounds;
|
|||
import org.apache.cassandra.dht.Range;
|
||||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.ReplicaPlans;
|
||||
import org.apache.cassandra.schema.KeyspaceParams;
|
||||
import org.apache.cassandra.schema.TableId;
|
||||
|
|
@ -70,18 +71,18 @@ public class RangeCommandIteratorTest
|
|||
int vnodeCount = 0;
|
||||
|
||||
Keyspace keyspace = Keyspace.open(KEYSPACE1);
|
||||
List<ReplicaPlan.ForRangeRead> ranges = new ArrayList<>();
|
||||
List<CoordinationPlan.ForRangeRead> ranges = new ArrayList<>();
|
||||
for (int i = 0; i + 1 < tokens.size(); i++)
|
||||
{
|
||||
Range<PartitionPosition> range = Range.makeRowRange(tokens.get(i), tokens.get(i + 1));
|
||||
ranges.add(ReplicaPlans.forRangeRead(keyspace, TABLE_ID, null, ConsistencyLevel.ONE, range, 1));
|
||||
ranges.add(CoordinationPlans.create(ReplicaPlans.forRangeRead(keyspace, TABLE_ID, null, ConsistencyLevel.ONE, range, 1)));
|
||||
vnodeCount++;
|
||||
}
|
||||
|
||||
ReplicaPlanMerger merge = new ReplicaPlanMerger(ranges.iterator(), keyspace, TABLE_ID, ConsistencyLevel.ONE);
|
||||
ReplicaPlan.ForRangeRead mergedRange = Iterators.getOnlyElement(merge);
|
||||
CoordinationPlanMerger merge = new CoordinationPlanMerger(ranges.iterator(), keyspace, TABLE_ID, ConsistencyLevel.ONE);
|
||||
CoordinationPlan.ForRangeRead mergedRange = Iterators.getOnlyElement(merge);
|
||||
// all ranges are merged as test has only one node.
|
||||
assertEquals(vnodeCount, mergedRange.vnodeCount());
|
||||
assertEquals(vnodeCount, mergedRange.replicas().vnodeCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
|
|
@ -108,27 +109,27 @@ public class RangeCommandIteratorTest
|
|||
AbstractBounds<PartitionPosition> keyRange = command.dataRange().keyRange();
|
||||
|
||||
// without range merger, there will be 2 batches requested: 1st batch with 1 range and 2nd batch with remaining ranges
|
||||
CloseableIterator<ReplicaPlan.ForRangeRead> replicaPlans = replicaPlanIterator(keyRange, keyspace, false);
|
||||
CloseableIterator<CoordinationPlan.ForRangeRead> replicaPlans = coordinatorPlanIterator(keyRange, keyspace, false);
|
||||
RangeCommandIterator data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution());
|
||||
verifyRangeCommandIterator(data, rows, 2, vnodeCount);
|
||||
|
||||
// without range merger and initial cf=5, there will be 1 batches requested: 5 vnode ranges for 1st batch
|
||||
replicaPlans = replicaPlanIterator(keyRange, keyspace, false);
|
||||
replicaPlans = coordinatorPlanIterator(keyRange, keyspace, false);
|
||||
data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, vnodeCount, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution());
|
||||
verifyRangeCommandIterator(data, rows, 1, vnodeCount);
|
||||
|
||||
// without range merger and max cf=1, there will be 5 batches requested: 1 vnode range per batch
|
||||
replicaPlans = replicaPlanIterator(keyRange, keyspace, false);
|
||||
replicaPlans = coordinatorPlanIterator(keyRange, keyspace, false);
|
||||
data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1, vnodeCount, Dispatcher.RequestTime.forImmediateExecution());
|
||||
verifyRangeCommandIterator(data, rows, vnodeCount, vnodeCount);
|
||||
|
||||
// with range merger, there will be only 1 batch requested, as all ranges share the same replica - localhost
|
||||
replicaPlans = replicaPlanIterator(keyRange, keyspace, true);
|
||||
replicaPlans = coordinatorPlanIterator(keyRange, keyspace, true);
|
||||
data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution());
|
||||
verifyRangeCommandIterator(data, rows, 1, vnodeCount);
|
||||
|
||||
// with range merger and max cf=1, there will be only 1 batch requested, as all ranges share the same replica - localhost
|
||||
replicaPlans = replicaPlanIterator(keyRange, keyspace, true);
|
||||
replicaPlans = coordinatorPlanIterator(keyRange, keyspace, true);
|
||||
data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1, vnodeCount, Dispatcher.RequestTime.forImmediateExecution());
|
||||
verifyRangeCommandIterator(data, rows, 1, vnodeCount);
|
||||
}
|
||||
|
|
@ -164,13 +165,13 @@ public class RangeCommandIteratorTest
|
|||
return new TokenUpdater().withKeys(values).update().getTokens();
|
||||
}
|
||||
|
||||
private static CloseableIterator<ReplicaPlan.ForRangeRead> replicaPlanIterator(AbstractBounds<PartitionPosition> keyRange,
|
||||
Keyspace keyspace,
|
||||
boolean withRangeMerger)
|
||||
private static CloseableIterator<CoordinationPlan.ForRangeRead> coordinatorPlanIterator(AbstractBounds<PartitionPosition> keyRange,
|
||||
Keyspace keyspace,
|
||||
boolean withRangeMerger)
|
||||
{
|
||||
CloseableIterator<ReplicaPlan.ForRangeRead> replicaPlans = new ReplicaPlanIterator(keyRange, null, keyspace, null, ConsistencyLevel.ONE);
|
||||
CloseableIterator<CoordinationPlan.ForRangeRead> replicaPlans = new CoordinationPlanIterator(keyRange, null, keyspace, null, ConsistencyLevel.ONE);
|
||||
if (withRangeMerger)
|
||||
replicaPlans = new ReplicaPlanMerger(replicaPlans, keyspace, null, ConsistencyLevel.ONE);
|
||||
replicaPlans = new CoordinationPlanMerger(replicaPlans, keyspace, null, ConsistencyLevel.ONE);
|
||||
|
||||
return replicaPlans;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -80,7 +80,7 @@ public class RangeCommandsTest extends CQLTester
|
|||
// verify that a low concurrency factor is not capped by the max concurrency factor
|
||||
PartitionRangeReadCommand command = command(cfs, 50, 50);
|
||||
try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, ReadCoordinator.DEFAULT, Dispatcher.RequestTime.forImmediateExecution());
|
||||
ReplicaPlanIterator ranges = new ReplicaPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE))
|
||||
CoordinationPlanIterator ranges = new CoordinationPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE))
|
||||
{
|
||||
assertEquals(2, partitions.concurrencyFactor());
|
||||
assertEquals(MAX_CONCURRENCY_FACTOR, partitions.maxConcurrencyFactor());
|
||||
|
|
@ -90,7 +90,7 @@ public class RangeCommandsTest extends CQLTester
|
|||
// verify that a high concurrency factor is capped by the max concurrency factor
|
||||
command = command(cfs, 1000, 50);
|
||||
try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, ReadCoordinator.DEFAULT, Dispatcher.RequestTime.forImmediateExecution());
|
||||
ReplicaPlanIterator ranges = new ReplicaPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE))
|
||||
CoordinationPlanIterator ranges = new CoordinationPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE))
|
||||
{
|
||||
assertEquals(MAX_CONCURRENCY_FACTOR, partitions.concurrencyFactor());
|
||||
assertEquals(MAX_CONCURRENCY_FACTOR, partitions.maxConcurrencyFactor());
|
||||
|
|
@ -100,7 +100,7 @@ public class RangeCommandsTest extends CQLTester
|
|||
// with 0 estimated results per range the concurrency factor should be 1
|
||||
command = command(cfs, 1000, 0);
|
||||
try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, ReadCoordinator.DEFAULT, Dispatcher.RequestTime.forImmediateExecution());
|
||||
ReplicaPlanIterator ranges = new ReplicaPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE))
|
||||
CoordinationPlanIterator ranges = new CoordinationPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE))
|
||||
{
|
||||
assertEquals(1, partitions.concurrencyFactor());
|
||||
assertEquals(MAX_CONCURRENCY_FACTOR, partitions.maxConcurrencyFactor());
|
||||
|
|
|
|||
|
|
@ -61,6 +61,8 @@ import org.apache.cassandra.dht.ByteOrderedPartitioner.BytesToken;
|
|||
import org.apache.cassandra.dht.Token;
|
||||
import org.apache.cassandra.distributed.test.log.ClusterMetadataTestHelper;
|
||||
import org.apache.cassandra.gms.Gossiper;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.EndpointsForToken;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
|
|
@ -366,11 +368,11 @@ public abstract class AbstractReadRepairTest
|
|||
Epoch.EMPTY);
|
||||
}
|
||||
|
||||
public abstract InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, ReplicaPlan.Shared<?, ?> replicaPlan, Dispatcher.RequestTime requestTime);
|
||||
public abstract InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, CoordinationPlan.ForRead<?, ?> plan, Dispatcher.RequestTime requestTime);
|
||||
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReplicaPlan.Shared<?, ?> replicaPlan)
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(CoordinationPlan.ForRead<?, ?> plan)
|
||||
{
|
||||
return createInstrumentedReadRepair(command, replicaPlan, Dispatcher.RequestTime.forImmediateExecution());
|
||||
return createInstrumentedReadRepair(command, plan, Dispatcher.RequestTime.forImmediateExecution());
|
||||
|
||||
}
|
||||
|
||||
|
|
@ -381,7 +383,7 @@ public abstract class AbstractReadRepairTest
|
|||
@Test
|
||||
public void readSpeculationCycle()
|
||||
{
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(ReplicaPlan.shared(replicaPlan(replicas, EndpointsForRange.of(replica1, replica2))));
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(CoordinationPlans.create(replicaPlan(replicas, EndpointsForRange.of(replica1, replica2))));
|
||||
ResultConsumer consumer = new ResultConsumer();
|
||||
|
||||
Assert.assertEquals(epSet(), repair.getReadRecipients());
|
||||
|
|
@ -400,7 +402,7 @@ public abstract class AbstractReadRepairTest
|
|||
@Test
|
||||
public void noSpeculationRequired()
|
||||
{
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(ReplicaPlan.shared(replicaPlan(replicas, EndpointsForRange.of(replica1, replica2))));
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(CoordinationPlans.create(replicaPlan(replicas, EndpointsForRange.of(replica1, replica2))));
|
||||
ResultConsumer consumer = new ResultConsumer();
|
||||
|
||||
Assert.assertEquals(epSet(), repair.getReadRecipients());
|
||||
|
|
|
|||
|
|
@ -35,6 +35,7 @@ import org.apache.cassandra.db.ConsistencyLevel;
|
|||
import org.apache.cassandra.db.Mutation;
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.locator.AbstractReplicationStrategy;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
|
|
@ -86,9 +87,9 @@ public class BlockingReadRepairTest extends AbstractReadRepairTest
|
|||
private static class InstrumentedBlockingReadRepair<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
extends BlockingReadRepair<E, P> implements InstrumentedReadRepair<E, P>
|
||||
{
|
||||
public InstrumentedBlockingReadRepair(ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public InstrumentedBlockingReadRepair(ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(ReadCoordinator.DEFAULT, command, replicaPlan, requestTime);
|
||||
super(ReadCoordinator.DEFAULT, command, plan, requestTime);
|
||||
}
|
||||
|
||||
Set<InetAddressAndPort> readCommandRecipients = new HashSet<>();
|
||||
|
|
@ -116,9 +117,9 @@ public class BlockingReadRepairTest extends AbstractReadRepairTest
|
|||
}
|
||||
|
||||
@Override
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, ReplicaPlan.Shared<?, ?> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, CoordinationPlan.ForRead<?, ?> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return new InstrumentedBlockingReadRepair(command, replicaPlan, requestTime);
|
||||
return new InstrumentedBlockingReadRepair(command, plan, requestTime);
|
||||
}
|
||||
|
||||
@Test
|
||||
|
|
|
|||
|
|
@ -29,6 +29,8 @@ import java.util.function.Predicate;
|
|||
|
||||
import com.google.common.collect.Lists;
|
||||
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.junit.After;
|
||||
import org.junit.Assert;
|
||||
import org.junit.BeforeClass;
|
||||
|
|
@ -44,7 +46,6 @@ import org.apache.cassandra.locator.Endpoints;
|
|||
import org.apache.cassandra.locator.EndpointsForRange;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
import org.apache.cassandra.locator.ReplicaPlan;
|
||||
import org.apache.cassandra.net.Message;
|
||||
import org.apache.cassandra.service.reads.ReadCallback;
|
||||
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||
|
|
@ -119,9 +120,9 @@ public class DiagEventsBlockingReadRepairTest extends AbstractReadRepairTest
|
|||
}
|
||||
|
||||
@Override
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, ReplicaPlan.Shared<?,?> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, CoordinationPlan.ForRead<?,?> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return new DiagnosticBlockingRepairHandler(command, replicaPlan, requestTime);
|
||||
return new DiagnosticBlockingRepairHandler(command, plan, requestTime);
|
||||
}
|
||||
|
||||
private static DiagnosticPartitionReadRepairHandler createRepairHandler(Map<Replica, Mutation> repairs, ReplicaPlan.ForWrite writePlan)
|
||||
|
|
@ -134,9 +135,9 @@ public class DiagEventsBlockingReadRepairTest extends AbstractReadRepairTest
|
|||
private Set<InetAddressAndPort> recipients = Collections.emptySet();
|
||||
private ReadCallback readCallback = null;
|
||||
|
||||
DiagnosticBlockingRepairHandler(ReadCommand command, ReplicaPlan.Shared<?,?> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
DiagnosticBlockingRepairHandler(ReadCommand command, CoordinationPlan.ForRead<?,?> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(ReadCoordinator.DEFAULT, command, replicaPlan, requestTime);
|
||||
super(ReadCoordinator.DEFAULT, command, plan, requestTime);
|
||||
DiagnosticEventService.instance().subscribe(ReadRepairEvent.class, this::onRepairEvent);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -28,6 +28,8 @@ import org.junit.Test;
|
|||
|
||||
import org.apache.cassandra.db.ReadCommand;
|
||||
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators;
|
||||
import org.apache.cassandra.locator.CoordinationPlan;
|
||||
import org.apache.cassandra.locator.CoordinationPlans;
|
||||
import org.apache.cassandra.locator.Endpoints;
|
||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||
import org.apache.cassandra.locator.Replica;
|
||||
|
|
@ -42,9 +44,9 @@ public class ReadOnlyReadRepairTest extends AbstractReadRepairTest
|
|||
private static class InstrumentedReadOnlyReadRepair<E extends Endpoints<E>, P extends ReplicaPlan.ForRead<E, P>>
|
||||
extends ReadOnlyReadRepair implements InstrumentedReadRepair
|
||||
{
|
||||
public InstrumentedReadOnlyReadRepair(ReadCommand command, ReplicaPlan.Shared<E, P> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public InstrumentedReadOnlyReadRepair(ReadCommand command, CoordinationPlan.ForRead<E, P> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
super(ReadCoordinator.DEFAULT, command, replicaPlan, requestTime);
|
||||
super(ReadCoordinator.DEFAULT, command, plan, requestTime);
|
||||
}
|
||||
|
||||
Set<InetAddressAndPort> readCommandRecipients = new HashSet<>();
|
||||
|
|
@ -78,25 +80,25 @@ public class ReadOnlyReadRepairTest extends AbstractReadRepairTest
|
|||
}
|
||||
|
||||
@Override
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, ReplicaPlan.Shared<?, ?> replicaPlan, Dispatcher.RequestTime requestTime)
|
||||
public InstrumentedReadRepair createInstrumentedReadRepair(ReadCommand command, CoordinationPlan.ForRead<?, ?> plan, Dispatcher.RequestTime requestTime)
|
||||
{
|
||||
return new InstrumentedReadOnlyReadRepair(command, replicaPlan, requestTime);
|
||||
return new InstrumentedReadOnlyReadRepair(command, plan, requestTime);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void getMergeListener()
|
||||
{
|
||||
ReplicaPlan.SharedForRangeRead replicaPlan = ReplicaPlan.shared(replicaPlan(replicas, replicas));
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(replicaPlan);
|
||||
Assert.assertSame(UnfilteredPartitionIterators.MergeListener.NOOP, repair.getMergeListener(replicaPlan.get()));
|
||||
CoordinationPlan.ForRangeRead plan = CoordinationPlans.create(replicaPlan(replicas, replicas));
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(plan);
|
||||
Assert.assertSame(UnfilteredPartitionIterators.MergeListener.NOOP, repair.getMergeListener(plan.replicas()));
|
||||
}
|
||||
|
||||
@Test(expected = UnsupportedOperationException.class)
|
||||
public void repairPartitionFailure()
|
||||
{
|
||||
ReplicaPlan.SharedForRangeRead readPlan = ReplicaPlan.shared(replicaPlan(replicas, replicas));
|
||||
CoordinationPlan.ForRangeRead plan = CoordinationPlans.create(replicaPlan(replicas, replicas));
|
||||
ReplicaPlan.ForWrite writePlan = repairPlan(replicas, replicas);
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(readPlan);
|
||||
InstrumentedReadRepair repair = createInstrumentedReadRepair(plan);
|
||||
repair.repairPartition(null, Collections.emptyMap(), writePlan, ReadRepairSource.OTHER);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,537 @@
|
|||
/*
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.service.reads.tracked;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
|
||||
import org.junit.Assert;
|
||||
import org.junit.BeforeClass;
|
||||
import org.junit.Test;
|
||||
|
||||
import org.apache.cassandra.SchemaLoader;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.quicktheories.core.Gen;
|
||||
|
||||
import static org.quicktheories.QuickTheory.qt;
|
||||
import static org.quicktheories.generators.SourceDSL.booleans;
|
||||
import static org.quicktheories.generators.SourceDSL.integers;
|
||||
import static org.quicktheories.generators.SourceDSL.lists;
|
||||
|
||||
/**
|
||||
* Property-based tests for ReadReconciliations.Coordinator completion logic.
|
||||
* Tests that the three-counter system (mutations, summaries, syncAcks) correctly
|
||||
* determines when reconciliation is complete.
|
||||
*/
|
||||
public class ReadReconciliationsCoordinatorPropertyTest
|
||||
{
|
||||
// LOCAL_NODE is initialized from ClusterMetadata to match production code
|
||||
private static int LOCAL_NODE;
|
||||
private static int REMOTE_NODE;
|
||||
|
||||
@BeforeClass
|
||||
public static void setUpClass() throws Throwable
|
||||
{
|
||||
SchemaLoader.loadSchema();
|
||||
// Get the actual local node ID that ReadReconciliations will use
|
||||
LOCAL_NODE = ClusterMetadata.current().myNodeId().id();
|
||||
REMOTE_NODE = LOCAL_NODE + 1; // Use a distinct remote node ID
|
||||
}
|
||||
|
||||
/**
|
||||
* Testable subclass that overrides complete() to skip messaging side effects,
|
||||
* and exposes test methods to directly manipulate counters.
|
||||
*/
|
||||
static class TestableCoordinator extends ReadReconciliations.Coordinator
|
||||
{
|
||||
TestableCoordinator(int dataNode, int[] summaryNodes)
|
||||
{
|
||||
super(new TrackedRead.Id(LOCAL_NODE, 0), dataNode, summaryNodes);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected boolean complete()
|
||||
{
|
||||
// Skip messaging - just return true
|
||||
return true;
|
||||
}
|
||||
|
||||
// Test methods that directly update counters without messaging side effects
|
||||
boolean testAcceptLocalSummary()
|
||||
{
|
||||
return updateRemainingAndMaybeComplete(0, -1, 0);
|
||||
}
|
||||
|
||||
boolean testAcceptRemoteSummary(int missingCount)
|
||||
{
|
||||
return updateRemainingAndMaybeComplete(missingCount, -1, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Configuration for a coordinator test case.
|
||||
*/
|
||||
static class CoordinatorConfig
|
||||
{
|
||||
final int summaryNodeCount; // 0-5 summary nodes
|
||||
final boolean isDataNode; // Whether local node is the data node
|
||||
|
||||
CoordinatorConfig(int summaryNodeCount, boolean isDataNode)
|
||||
{
|
||||
this.summaryNodeCount = summaryNodeCount;
|
||||
this.isDataNode = isDataNode;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("summaryNodes=%d, isDataNode=%s", summaryNodeCount, isDataNode);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Response types for coordinator testing.
|
||||
*/
|
||||
enum ResponseType
|
||||
{
|
||||
LOCAL_SUMMARY,
|
||||
REMOTE_SUMMARY,
|
||||
SYNC_ACK,
|
||||
MUTATION
|
||||
}
|
||||
|
||||
/**
|
||||
* A single response event.
|
||||
*/
|
||||
static class CoordinatorResponse
|
||||
{
|
||||
final ResponseType type;
|
||||
final int missingCount; // Only used for REMOTE_SUMMARY
|
||||
final int nodeId; // Used for SYNC_ACK
|
||||
|
||||
CoordinatorResponse(ResponseType type)
|
||||
{
|
||||
this(type, 0, 0);
|
||||
}
|
||||
|
||||
CoordinatorResponse(ResponseType type, int missingCount)
|
||||
{
|
||||
this(type, missingCount, 0);
|
||||
}
|
||||
|
||||
CoordinatorResponse(ResponseType type, int missingCount, int nodeId)
|
||||
{
|
||||
this.type = type;
|
||||
this.missingCount = missingCount;
|
||||
this.nodeId = nodeId;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
if (type == ResponseType.REMOTE_SUMMARY)
|
||||
return String.format("%s(missing=%d)", type, missingCount);
|
||||
if (type == ResponseType.SYNC_ACK)
|
||||
return String.format("%s(node=%d)", type, nodeId);
|
||||
return type.toString();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A complete test case with configuration and response sequence.
|
||||
*/
|
||||
static class TestCase
|
||||
{
|
||||
final CoordinatorConfig config;
|
||||
final List<CoordinatorResponse> responses;
|
||||
|
||||
TestCase(CoordinatorConfig config, List<CoordinatorResponse> responses)
|
||||
{
|
||||
this.config = config;
|
||||
this.responses = responses;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("config=%s, responses=%d", config, responses.size());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Expected state tracker for validating coordinator behavior.
|
||||
*/
|
||||
static class ExpectedState
|
||||
{
|
||||
int remainingMutations;
|
||||
int remainingSummaries;
|
||||
int remainingSyncAcks;
|
||||
|
||||
ExpectedState(int summaryNodeCount, boolean isDataNode)
|
||||
{
|
||||
this.remainingMutations = 0;
|
||||
this.remainingSummaries = 1 + summaryNodeCount;
|
||||
this.remainingSyncAcks = isDataNode ? summaryNodeCount : 0;
|
||||
}
|
||||
|
||||
boolean apply(CoordinatorResponse response)
|
||||
{
|
||||
switch (response.type)
|
||||
{
|
||||
case LOCAL_SUMMARY:
|
||||
remainingSummaries--;
|
||||
break;
|
||||
case REMOTE_SUMMARY:
|
||||
remainingMutations += response.missingCount;
|
||||
remainingSummaries--;
|
||||
break;
|
||||
case SYNC_ACK:
|
||||
remainingSyncAcks--;
|
||||
break;
|
||||
case MUTATION:
|
||||
remainingMutations--;
|
||||
break;
|
||||
}
|
||||
return isComplete();
|
||||
}
|
||||
|
||||
boolean isComplete()
|
||||
{
|
||||
return remainingMutations == 0
|
||||
&& remainingSummaries == 0
|
||||
&& remainingSyncAcks == 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString()
|
||||
{
|
||||
return String.format("mutations=%d, summaries=%d, syncAcks=%d",
|
||||
remainingMutations, remainingSummaries, remainingSyncAcks);
|
||||
}
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Generators
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Generate coordinator configurations.
|
||||
*/
|
||||
Gen<CoordinatorConfig> configGen()
|
||||
{
|
||||
return integers().between(0, 5).flatMap(summaryNodes ->
|
||||
booleans().all().map(isDataNode ->
|
||||
new CoordinatorConfig(summaryNodes, isDataNode)
|
||||
)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate test cases with valid response sequences.
|
||||
* Responses are generated such that mutations always come after their parent remote summary.
|
||||
*/
|
||||
Gen<TestCase> testCaseGen()
|
||||
{
|
||||
return configGen().flatMap(config -> responseSequenceGen(config).map(responses ->
|
||||
new TestCase(config, responses)
|
||||
));
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a valid response sequence for a given configuration.
|
||||
*/
|
||||
Gen<List<CoordinatorResponse>> responseSequenceGen(CoordinatorConfig config)
|
||||
{
|
||||
// Generate missingCount for each remote summary (0-10)
|
||||
return lists().of(integers().between(0, 10)).ofSize(config.summaryNodeCount).flatMap(missingCounts -> {
|
||||
// Calculate total mutations needed
|
||||
int totalMutations = missingCounts.stream().mapToInt(Integer::intValue).sum();
|
||||
|
||||
// Generate a random permutation using indices
|
||||
int totalEvents = 1 + config.summaryNodeCount + totalMutations + (config.isDataNode ? config.summaryNodeCount : 0);
|
||||
return permutationGen(totalEvents).map(permutation -> {
|
||||
// Build the response sequence respecting ordering constraints
|
||||
return buildResponseSequence(config, missingCounts, permutation);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Build response sequence respecting the constraint that mutations must come after their parent remote summary.
|
||||
*/
|
||||
private List<CoordinatorResponse> buildResponseSequence(CoordinatorConfig config,
|
||||
List<Integer> missingCounts,
|
||||
List<Integer> permutation)
|
||||
{
|
||||
// Create all events
|
||||
List<CoordinatorResponse> events = new ArrayList<>();
|
||||
|
||||
// Add local summary
|
||||
events.add(new CoordinatorResponse(ResponseType.LOCAL_SUMMARY));
|
||||
|
||||
// Add remote summaries with their mutations linked
|
||||
List<List<CoordinatorResponse>> remoteSummaryGroups = new ArrayList<>();
|
||||
for (int i = 0; i < config.summaryNodeCount; i++)
|
||||
{
|
||||
List<CoordinatorResponse> group = new ArrayList<>();
|
||||
group.add(new CoordinatorResponse(ResponseType.REMOTE_SUMMARY, missingCounts.get(i)));
|
||||
for (int j = 0; j < missingCounts.get(i); j++)
|
||||
group.add(new CoordinatorResponse(ResponseType.MUTATION));
|
||||
remoteSummaryGroups.add(group);
|
||||
}
|
||||
|
||||
// Add sync acks (if data node)
|
||||
List<CoordinatorResponse> syncAcks = new ArrayList<>();
|
||||
if (config.isDataNode)
|
||||
{
|
||||
for (int i = 0; i < config.summaryNodeCount; i++)
|
||||
syncAcks.add(new CoordinatorResponse(ResponseType.SYNC_ACK, 0, REMOTE_NODE + i + 1));
|
||||
}
|
||||
|
||||
// Shuffle the independent groups using the permutation
|
||||
// We shuffle: local summary, each remote summary group (as a unit), and sync acks
|
||||
List<Object> shuffleable = new ArrayList<>();
|
||||
shuffleable.add(events.get(0)); // local summary
|
||||
shuffleable.addAll(remoteSummaryGroups);
|
||||
shuffleable.addAll(syncAcks);
|
||||
|
||||
// Use permutation to reorder (just use first N elements of permutation as indices)
|
||||
List<Object> shuffled = new ArrayList<>(shuffleable);
|
||||
Collections.shuffle(shuffled, new java.util.Random(permutation.hashCode()));
|
||||
|
||||
// Flatten back to response list
|
||||
List<CoordinatorResponse> result = new ArrayList<>();
|
||||
for (Object item : shuffled)
|
||||
{
|
||||
if (item instanceof CoordinatorResponse)
|
||||
result.add((CoordinatorResponse) item);
|
||||
else if (item instanceof List)
|
||||
{
|
||||
@SuppressWarnings("unchecked")
|
||||
List<CoordinatorResponse> group = (List<CoordinatorResponse>) item;
|
||||
result.addAll(group);
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a random permutation of integers [0, n-1].
|
||||
*/
|
||||
private static Gen<List<Integer>> permutationGen(int n)
|
||||
{
|
||||
if (n == 0)
|
||||
return lists().of(integers().all()).ofSize(0);
|
||||
if (n == 1)
|
||||
{
|
||||
return lists().of(integers().all()).ofSize(1).map(list -> {
|
||||
List<Integer> result = new ArrayList<>();
|
||||
result.add(0);
|
||||
return result;
|
||||
});
|
||||
}
|
||||
|
||||
// Generate n random integers for shuffling
|
||||
return lists().of(integers().between(0, Integer.MAX_VALUE)).ofSize(n).map(randoms -> {
|
||||
List<Integer> result = new ArrayList<>();
|
||||
for (int i = 0; i < n; i++)
|
||||
result.add(i);
|
||||
|
||||
// Fisher-Yates shuffle using generated random values
|
||||
for (int i = n - 1; i > 0; i--)
|
||||
{
|
||||
int j = Math.abs(randoms.get(i)) % (i + 1);
|
||||
Collections.swap(result, i, j);
|
||||
}
|
||||
|
||||
return result;
|
||||
});
|
||||
}
|
||||
|
||||
// ========================================
|
||||
// Test Methods
|
||||
// ========================================
|
||||
|
||||
/**
|
||||
* Apply a response to the coordinator and return whether it completed.
|
||||
*/
|
||||
private boolean applyResponse(TestableCoordinator coordinator, CoordinatorResponse response)
|
||||
{
|
||||
switch (response.type)
|
||||
{
|
||||
case LOCAL_SUMMARY:
|
||||
return coordinator.testAcceptLocalSummary();
|
||||
case REMOTE_SUMMARY:
|
||||
return coordinator.testAcceptRemoteSummary(response.missingCount);
|
||||
case SYNC_ACK:
|
||||
return coordinator.acceptSyncAck(response.nodeId);
|
||||
case MUTATION:
|
||||
return coordinator.acceptMutation(null); // mutationId is ignored
|
||||
default:
|
||||
throw new IllegalArgumentException("Unknown response type: " + response.type);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a coordinator for testing.
|
||||
*/
|
||||
private TestableCoordinator createCoordinator(CoordinatorConfig config)
|
||||
{
|
||||
int dataNode = config.isDataNode ? LOCAL_NODE : REMOTE_NODE;
|
||||
int[] summaryNodes = new int[config.summaryNodeCount];
|
||||
for (int i = 0; i < config.summaryNodeCount; i++)
|
||||
summaryNodes[i] = REMOTE_NODE + i + 1; // Use distinct node IDs
|
||||
|
||||
return new TestableCoordinator(dataNode, summaryNodes);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void coordinatorCompletionBehavior()
|
||||
{
|
||||
qt()
|
||||
.withExamples(500)
|
||||
.forAll(testCaseGen())
|
||||
.checkAssert(testCase -> {
|
||||
TestableCoordinator coordinator = createCoordinator(testCase.config);
|
||||
ExpectedState expected = new ExpectedState(
|
||||
testCase.config.summaryNodeCount,
|
||||
testCase.config.isDataNode
|
||||
);
|
||||
|
||||
for (CoordinatorResponse response : testCase.responses)
|
||||
{
|
||||
boolean actualComplete = applyResponse(coordinator, response);
|
||||
boolean expectedComplete = expected.apply(response);
|
||||
|
||||
Assert.assertEquals(
|
||||
String.format("Completion mismatch after %s: expected=%s, actual=%s, state=%s, config=%s",
|
||||
response, expectedComplete, actualComplete, expected, testCase.config),
|
||||
expectedComplete, actualComplete
|
||||
);
|
||||
|
||||
if (actualComplete)
|
||||
break;
|
||||
}
|
||||
|
||||
// Final state should be complete
|
||||
Assert.assertTrue(
|
||||
String.format("Should be complete after all responses: state=%s, config=%s",
|
||||
expected, testCase.config),
|
||||
expected.isComplete()
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Test edge case: minimal coordinator with 0 summary nodes as data node.
|
||||
* Should complete immediately after local summary.
|
||||
*/
|
||||
@Test
|
||||
public void minimalDataNodeCompletes()
|
||||
{
|
||||
TestableCoordinator coordinator = new TestableCoordinator(LOCAL_NODE, new int[0]);
|
||||
|
||||
// Should complete after just the local summary
|
||||
Assert.assertTrue("Should complete after local summary with 0 summary nodes",
|
||||
coordinator.testAcceptLocalSummary());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test edge case: minimal coordinator with 0 summary nodes as summary node.
|
||||
* Should complete immediately after local summary.
|
||||
*/
|
||||
@Test
|
||||
public void minimalSummaryNodeCompletes()
|
||||
{
|
||||
TestableCoordinator coordinator = new TestableCoordinator(REMOTE_NODE, new int[0]);
|
||||
|
||||
// Should complete after just the local summary
|
||||
Assert.assertTrue("Should complete after local summary with 0 summary nodes",
|
||||
coordinator.testAcceptLocalSummary());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test that data node requires sync acks from all summary nodes.
|
||||
*/
|
||||
@Test
|
||||
public void dataNodeRequiresSyncAcks()
|
||||
{
|
||||
TestableCoordinator coordinator = new TestableCoordinator(
|
||||
LOCAL_NODE, new int[]{REMOTE_NODE, REMOTE_NODE + 1}
|
||||
);
|
||||
|
||||
// Local summary
|
||||
Assert.assertFalse(coordinator.testAcceptLocalSummary());
|
||||
|
||||
// Remote summaries with no missing mutations
|
||||
Assert.assertFalse(coordinator.testAcceptRemoteSummary(0));
|
||||
Assert.assertFalse(coordinator.testAcceptRemoteSummary(0));
|
||||
|
||||
// First sync ack
|
||||
Assert.assertFalse(coordinator.acceptSyncAck(REMOTE_NODE));
|
||||
|
||||
// Second sync ack should complete
|
||||
Assert.assertTrue(coordinator.acceptSyncAck(REMOTE_NODE + 1));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test that summary node does NOT require sync acks.
|
||||
*/
|
||||
@Test
|
||||
public void summaryNodeDoesNotRequireSyncAcks()
|
||||
{
|
||||
TestableCoordinator coordinator = new TestableCoordinator(
|
||||
REMOTE_NODE, new int[]{REMOTE_NODE + 1, REMOTE_NODE + 2}
|
||||
);
|
||||
|
||||
// Local summary
|
||||
Assert.assertFalse(coordinator.testAcceptLocalSummary());
|
||||
|
||||
// Remote summaries with no missing mutations
|
||||
Assert.assertFalse(coordinator.testAcceptRemoteSummary(0));
|
||||
|
||||
// Last remote summary should complete (no sync acks needed)
|
||||
Assert.assertTrue(coordinator.testAcceptRemoteSummary(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test that mutations must be received before completion.
|
||||
*/
|
||||
@Test
|
||||
public void mutationsMustBeReceived()
|
||||
{
|
||||
TestableCoordinator coordinator = new TestableCoordinator(
|
||||
REMOTE_NODE, new int[]{REMOTE_NODE + 1}
|
||||
);
|
||||
|
||||
// Local summary
|
||||
Assert.assertFalse(coordinator.testAcceptLocalSummary());
|
||||
|
||||
// Remote summary with 2 missing mutations
|
||||
Assert.assertFalse(coordinator.testAcceptRemoteSummary(2));
|
||||
|
||||
// First mutation
|
||||
Assert.assertFalse(coordinator.acceptMutation(null));
|
||||
|
||||
// Second mutation should complete
|
||||
Assert.assertTrue(coordinator.acceptMutation(null));
|
||||
}
|
||||
}
|
||||
Loading…
Reference in New Issue