Merge branch 'cassandra-4.1' into cassandra-5.0

This commit is contained in:
Stefan Miklosovic 2024-02-22 23:57:43 +01:00
commit bfc6e1241d
No known key found for this signature in database
GPG Key ID: 32F35CB2F546D93E
7 changed files with 109 additions and 98 deletions

View File

@ -20,6 +20,7 @@ Merged from 4.1:
* Memoize Cassandra verion and add a backoff interval for failed schema pulls (CASSANDRA-18902)
* Fix StackOverflowError on ALTER after many previous schema changes (CASSANDRA-19166)
Merged from 4.0:
* Filter remote DC replicas out when constructing the initial replica plan for the local read repair (CASSANDRA-19120)
* Remove redundant code in StorageProxy#sendToHintedReplicas (CASSANDRA-19412)
* Remove bashisms for mx4j tool in cassandra-env.sh (CASSANDRA-19416)
* Add new concurrent_merkle_tree_requests config property to prevent OOM during multi-range and/or multi-table repairs (CASSANDRA-19336)

View File

@ -446,6 +446,7 @@ public class ReplicaPlans
{
return new Selector()
{
@Override
public <E extends Endpoints<E>, L extends ReplicaLayout.ForWrite<E>>
E select(ConsistencyLevel consistencyLevel, L liveAndDown, L live)
@ -462,7 +463,8 @@ public class ReplicaPlans
int add = consistencyLevel.blockForWrite(liveAndDown.replicationStrategy(), liveAndDown.pending()) - contacts.size();
if (add > 0)
{
for (Replica replica : filter(live.all(), r -> !contacts.contains(r)))
E all = consistencyLevel.isDatacenterLocal() ? live.all().filter(InOurDc.replicas()) : live.all();
for (Replica replica : filter(all, r -> !contacts.contains(r)))
{
contacts.add(replica);
if (--add == 0)

View File

@ -25,11 +25,9 @@ import java.util.concurrent.ConcurrentHashMap;
import org.apache.cassandra.utils.concurrent.AsyncFuture;
import org.apache.cassandra.utils.concurrent.CountDownLatch;
import java.util.concurrent.TimeUnit;
import java.util.function.Predicate;
import com.google.common.annotations.VisibleForTesting;
import com.google.common.base.Preconditions;
import com.google.common.base.Predicates;
import com.google.common.collect.Iterables;
import com.google.common.collect.Lists;
@ -56,6 +54,7 @@ import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException;
import static org.apache.cassandra.net.Verb.*;
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
import static org.apache.cassandra.utils.concurrent.CountDownLatch.newCountDownLatch;
import static com.google.common.collect.Iterables.all;
public class BlockingPartitionRepair
extends AsyncFuture<Object> implements RequestCallback<Object>
@ -64,31 +63,30 @@ public class BlockingPartitionRepair
private final ReplicaPlan.ForWrite writePlan;
private final Map<Replica, Mutation> pendingRepairs;
private final CountDownLatch latch;
private final Predicate<InetAddressAndPort> shouldBlockOn;
private volatile long mutationsSentTime;
public BlockingPartitionRepair(DecoratedKey key, Map<Replica, Mutation> repairs, ReplicaPlan.ForWrite writePlan)
{
this(key, repairs, writePlan,
writePlan.consistencyLevel().isDatacenterLocal() ? InOurDc.endpoints() : Predicates.alwaysTrue());
}
public BlockingPartitionRepair(DecoratedKey key, Map<Replica, Mutation> repairs, ReplicaPlan.ForWrite writePlan, Predicate<InetAddressAndPort> shouldBlockOn)
{
this.key = key;
this.pendingRepairs = new ConcurrentHashMap<>(repairs);
this.writePlan = writePlan;
this.shouldBlockOn = shouldBlockOn;
// make sure all the read repair targets are contact of the repair write plan
Preconditions.checkState(all(repairs.keySet(), (r) -> writePlan.contacts().contains(r)),
"All repair targets should be part of contacts of read repair write plan.");
int blockFor = writePlan.writeQuorum();
// here we remove empty repair mutations from the block for total, since
// we're not sending them mutations
for (Replica participant : writePlan.contacts())
{
// remote dcs can sometimes get involved in dc-local reads. We want to repair
// them if they do, but they shouldn't interfere with blocking the client read.
if (!repairs.containsKey(participant) && shouldBlockOn.test(participant.endpoint()))
if (!repairs.containsKey(participant))
blockFor--;
// make sure for local consistency, all contacts are local replicas
Preconditions.checkState(!writePlan.consistencyLevel().isDatacenterLocal() || InOurDc.replicas().test(participant),
"Local consistency blocking read repair is trying to contact remote DC node: " + participant.endpoint());
}
// there are some cases where logically identical data can return different digests
@ -113,11 +111,8 @@ public class BlockingPartitionRepair
@VisibleForTesting
void ack(InetAddressAndPort from)
{
if (shouldBlockOn.test(from))
{
pendingRepairs.remove(writePlan.lookup(from));
latch.decrement();
}
pendingRepairs.remove(writePlan.lookup(from));
latch.decrement();
}
@Override
@ -163,9 +158,6 @@ public class BlockingPartitionRepair
// use a separate verb here to avoid writing hints on timeouts
sendRR(Message.out(READ_REPAIR_REQ, mutation), destination.endpoint());
ColumnFamilyStore.metricsFor(tableId).readRepairRequests.mark();
if (!shouldBlockOn.test(destination.endpoint()))
pendingRepairs.remove(destination);
ReadRepairDiagnostics.sendInitialRepair(this, destination.endpoint(), mutation);
}
}
@ -208,7 +200,7 @@ public class BlockingPartitionRepair
if (awaitRepairsUntil(timeout + timeoutUnit.convert(mutationsSentTime, TimeUnit.NANOSECONDS), timeoutUnit))
return;
EndpointsForToken newCandidates = writePlan.liveUncontacted();
EndpointsForToken newCandidates = writePlan.consistencyLevel().isDatacenterLocal() ? writePlan.liveUncontacted().filter(InOurDc.replicas()) : writePlan.liveUncontacted();
if (newCandidates.isEmpty())
return;

View File

@ -33,6 +33,7 @@ import com.google.common.primitives.Ints;
import org.apache.cassandra.dht.ByteOrderedPartitioner;
import org.apache.cassandra.dht.Token;
import org.apache.cassandra.gms.Gossiper;
import org.apache.cassandra.locator.AbstractNetworkTopologySnitch;
import org.apache.cassandra.locator.EndpointsForToken;
import org.apache.cassandra.locator.ReplicaPlan;
import org.junit.Assert;
@ -91,13 +92,22 @@ public abstract class AbstractReadRepairTest
static InetAddressAndPort target1;
static InetAddressAndPort target2;
static InetAddressAndPort target3;
static InetAddressAndPort remote1;
static InetAddressAndPort remote2;
static InetAddressAndPort remote3;
static List<InetAddressAndPort> targets;
static List<InetAddressAndPort> remotes;
static Replica replica1;
static Replica replica2;
static Replica replica3;
static EndpointsForRange replicas;
static ReplicaPlan.ForRead<?, ?> replicaPlan;
static Replica remoteReplica1;
static Replica remoteReplica2;
static Replica remoteReplica3;
static EndpointsForRange remoteReplicas;
static long now = TimeUnit.NANOSECONDS.toMicros(nanoTime());
static DecoratedKey key;
@ -214,6 +224,66 @@ public abstract class AbstractReadRepairTest
static void configureClass(ReadRepairStrategy repairStrategy) throws Throwable
{
SchemaLoader.loadSchema();
DatabaseDescriptor.setEndpointSnitch(new AbstractNetworkTopologySnitch()
{
public String getRack(InetAddressAndPort endpoint)
{
return "rack1";
}
public String getDatacenter(InetAddressAndPort endpoint)
{
byte[] address = endpoint.addressBytes;
if (address[1] == 2) {
return "datacenter2";
}
return "datacenter1";
}
});
target1 = InetAddressAndPort.getByName("127.1.0.255");
target2 = InetAddressAndPort.getByName("127.1.0.254");
target3 = InetAddressAndPort.getByName("127.1.0.253");
remote1 = InetAddressAndPort.getByName("127.2.0.255");
remote2 = InetAddressAndPort.getByName("127.2.0.254");
remote3 = InetAddressAndPort.getByName("127.2.0.253");
targets = ImmutableList.of(target1, target2, target3);
remotes = ImmutableList.of(remote1, remote2, remote3);
replica1 = fullReplica(target1, FULL_RANGE);
replica2 = fullReplica(target2, FULL_RANGE);
replica3 = fullReplica(target3, FULL_RANGE);
replicas = EndpointsForRange.of(replica1, replica2, replica3);
remoteReplica1 = fullReplica(remote1, FULL_RANGE);
remoteReplica2 = fullReplica(remote2, FULL_RANGE);
remoteReplica3 = fullReplica(remote3, FULL_RANGE);
remoteReplicas = EndpointsForRange.of(remoteReplica1, remoteReplica2, remoteReplica3);
StorageService.instance.getTokenMetadata().clearUnsafe();
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 0 })), replica1.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 1 })), replica2.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 2 })), replica3.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 3 })), remoteReplica1.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 4 })), remoteReplica2.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 5 })), remoteReplica3.endpoint());
for (Replica replica : replicas)
{
UUID hostId = UUID.randomUUID();
Gossiper.instance.initializeNodeUnsafe(replica.endpoint(), hostId, 1);
StorageService.instance.getTokenMetadata().updateHostId(hostId, replica.endpoint());
}
for (Replica replica : remoteReplicas)
{
UUID hostId = UUID.randomUUID();
Gossiper.instance.initializeNodeUnsafe(replica.endpoint(), hostId, 1);
StorageService.instance.getTokenMetadata().updateHostId(hostId, replica.endpoint());
}
String ksName = "ks";
String ddl = String.format("CREATE TABLE tbl (k int primary key, v text) WITH read_repair='%s'",
@ -221,7 +291,8 @@ public abstract class AbstractReadRepairTest
cfm = CreateTableStatement.parse(ddl, ksName).build();
assert cfm.params.readRepair == repairStrategy;
KeyspaceMetadata ksm = KeyspaceMetadata.create(ksName, KeyspaceParams.simple(3), Tables.of(cfm));
KeyspaceMetadata ksm = KeyspaceMetadata.create(ksName, KeyspaceParams.nts("datacenter1", 3, "datacenter2", 3), Tables.of(cfm));
SchemaTestUtil.announceNewKeyspace(ksm);
ks = Keyspace.open(ksName);
@ -230,27 +301,6 @@ public abstract class AbstractReadRepairTest
cfs.sampleReadLatencyMicros = 0;
cfs.additionalWriteLatencyMicros = 0;
target1 = InetAddressAndPort.getByName("127.0.0.255");
target2 = InetAddressAndPort.getByName("127.0.0.254");
target3 = InetAddressAndPort.getByName("127.0.0.253");
targets = ImmutableList.of(target1, target2, target3);
replica1 = fullReplica(target1, FULL_RANGE);
replica2 = fullReplica(target2, FULL_RANGE);
replica3 = fullReplica(target3, FULL_RANGE);
replicas = EndpointsForRange.of(replica1, replica2, replica3);
replicaPlan = replicaPlan(ConsistencyLevel.QUORUM, replicas);
StorageService.instance.getTokenMetadata().clearUnsafe();
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 0 })), replica1.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 1 })), replica2.endpoint());
StorageService.instance.getTokenMetadata().updateNormalToken(ByteOrderedPartitioner.instance.getToken(ByteBuffer.wrap(new byte[] { 2 })), replica3.endpoint());
Gossiper.instance.initializeNodeUnsafe(replica1.endpoint(), UUID.randomUUID(), 1);
Gossiper.instance.initializeNodeUnsafe(replica2.endpoint(), UUID.randomUUID(), 1);
Gossiper.instance.initializeNodeUnsafe(replica3.endpoint(), UUID.randomUUID(), 1);
// default test values
key = dk(5);
cell1 = cell("v", "val1", now);
@ -297,7 +347,7 @@ public abstract class AbstractReadRepairTest
Token token = readPlan.range().left.getToken();
EndpointsForToken pending = EndpointsForToken.empty(token);
return ReplicaPlans.forWrite(readPlan.keyspace(),
ConsistencyLevel.TWO,
readPlan.consistencyLevel(),
liveAndDown.forToken(token),
pending,
replica -> true,
@ -305,7 +355,7 @@ public abstract class AbstractReadRepairTest
}
static ReplicaPlan.ForRangeRead replicaPlan(EndpointsForRange replicas, EndpointsForRange targets)
{
return replicaPlan(ks, ConsistencyLevel.QUORUM, replicas, targets);
return replicaPlan(ks, ConsistencyLevel.LOCAL_QUORUM, replicas, targets);
}
static ReplicaPlan.ForRangeRead replicaPlan(Keyspace keyspace, ConsistencyLevel consistencyLevel, EndpointsForRange replicas)
{

View File

@ -39,7 +39,6 @@ import org.apache.cassandra.locator.EndpointsForRange;
import org.apache.cassandra.locator.InetAddressAndPort;
import org.apache.cassandra.locator.Replica;
import org.apache.cassandra.locator.ReplicaPlan;
import org.apache.cassandra.locator.ReplicaUtils;
import org.apache.cassandra.net.Message;
import org.apache.cassandra.service.reads.ReadCallback;
@ -53,7 +52,7 @@ public class BlockingReadRepairTest extends AbstractReadRepairTest
{
public InstrumentedReadRepairHandler(Map<Replica, Mutation> repairs, ReplicaPlan.ForWrite writePlan)
{
super(Util.dk("not a real usable value"), repairs, writePlan, e -> targets.contains(e));
super(Util.dk("not a real usable value"), repairs, writePlan);
}
Map<InetAddressAndPort, Mutation> mutationsSent = new HashMap<>();
@ -124,10 +123,10 @@ public class BlockingReadRepairTest extends AbstractReadRepairTest
{
AbstractReplicationStrategy rs = ks.getReplicationStrategy();
Assert.assertTrue(ConsistencyLevel.QUORUM.satisfies(ConsistencyLevel.QUORUM, rs));
Assert.assertTrue(ConsistencyLevel.THREE.satisfies(ConsistencyLevel.QUORUM, rs));
Assert.assertTrue(ConsistencyLevel.TWO.satisfies(ConsistencyLevel.QUORUM, rs));
Assert.assertFalse(ConsistencyLevel.ONE.satisfies(ConsistencyLevel.QUORUM, rs));
Assert.assertFalse(ConsistencyLevel.ANY.satisfies(ConsistencyLevel.QUORUM, rs));
Assert.assertTrue(ConsistencyLevel.THREE.satisfies(ConsistencyLevel.LOCAL_QUORUM, rs));
Assert.assertTrue(ConsistencyLevel.TWO.satisfies(ConsistencyLevel.LOCAL_QUORUM, rs));
Assert.assertFalse(ConsistencyLevel.ONE.satisfies(ConsistencyLevel.LOCAL_QUORUM, rs));
Assert.assertFalse(ConsistencyLevel.ANY.satisfies(ConsistencyLevel.LOCAL_QUORUM, rs));
}
@ -256,36 +255,18 @@ public class BlockingReadRepairTest extends AbstractReadRepairTest
}
/**
* For dc local consistency levels, noop mutations and responses from remote dcs should not affect effective blockFor
* For dc local consistency levels, we will run into assertion error because no remote DC replicas should be contacted
*/
@Test
public void remoteDCTest() throws Exception
@Test(expected = IllegalStateException.class)
public void remoteDCSpeculativeRetryTest() throws Exception
{
Map<Replica, Mutation> repairs = new HashMap<>();
repairs.put(replica1, mutation(cell1));
repairs.put(remoteReplica1, mutation(cell1));
Replica remote1 = ReplicaUtils.full(InetAddressAndPort.getByName("10.0.0.1"));
Replica remote2 = ReplicaUtils.full(InetAddressAndPort.getByName("10.0.0.2"));
repairs.put(remote1, mutation(cell1));
EndpointsForRange participants = EndpointsForRange.of(replica1, replica2, remote1, remote2);
EndpointsForRange participants = EndpointsForRange.of(replica1, replica2, remoteReplica1, remoteReplica2);
ReplicaPlan.ForWrite writePlan = repairPlan(replicaPlan(ks, ConsistencyLevel.LOCAL_QUORUM, participants));
InstrumentedReadRepairHandler handler = createRepairHandler(repairs, writePlan);
handler.sendInitialRepairs();
Assert.assertEquals(2, handler.mutationsSent.size());
Assert.assertTrue(handler.mutationsSent.containsKey(replica1.endpoint()));
Assert.assertTrue(handler.mutationsSent.containsKey(remote1.endpoint()));
Assert.assertEquals(1, handler.waitingOn());
Assert.assertFalse(getCurrentRepairStatus(handler));
handler.ack(remote1.endpoint());
Assert.assertEquals(1, handler.waitingOn());
Assert.assertFalse(getCurrentRepairStatus(handler));
handler.ack(replica1.endpoint());
Assert.assertEquals(0, handler.waitingOn());
Assert.assertTrue(getCurrentRepairStatus(handler));
createRepairHandler(repairs, writePlan);
}
private boolean getCurrentRepairStatus(BlockingPartitionRepair handler)

View File

@ -181,7 +181,7 @@ public class DiagEventsBlockingReadRepairTest extends AbstractReadRepairTest
DiagnosticPartitionReadRepairHandler(DecoratedKey key, Map<Replica, Mutation> repairs, ReplicaPlan.ForWrite writePlan)
{
super(key, repairs, writePlan, isLocal());
super(key, repairs, writePlan);
DiagnosticEventService.instance().subscribe(PartitionRepairEvent.class, this::onRepairEvent);
}

View File

@ -77,7 +77,7 @@ public class ReadRepairTest
{
public InstrumentedReadRepairHandler(Map<Replica, Mutation> repairs, ReplicaPlan.ForWrite writePlan)
{
super(Util.dk("not a valid key"), repairs, writePlan, e -> targets.endpoints().contains(e));
super(Util.dk("not a valid key"), repairs, writePlan);
}
Map<InetAddressAndPort, Mutation> mutationsSent = new HashMap<>();
@ -315,10 +315,10 @@ public class ReadRepairTest
}
/**
* For dc local consistency levels, noop mutations and responses from remote dcs should not affect effective blockFor
* For dc local consistency levels, if repair map has remote DC node, we will get assertion failure
*/
@Test
public void remoteDCTest() throws Exception
@Test(expected = IllegalStateException.class)
public void remoteDCNodeInvolveInLocalConsistencyTest() throws Exception
{
Map<Replica, Mutation> repairs = new HashMap<>();
repairs.put(target1, mutation(cell1));
@ -330,22 +330,7 @@ public class ReadRepairTest
EndpointsForRange participants = EndpointsForRange.of(target1, target2, remote1, remote2);
EndpointsForRange targets = EndpointsForRange.of(target1, target2);
InstrumentedReadRepairHandler handler = createRepairHandler(repairs, participants, targets);
handler.sendInitialRepairs();
Assert.assertEquals(2, handler.mutationsSent.size());
Assert.assertTrue(handler.mutationsSent.containsKey(target1.endpoint()));
Assert.assertTrue(handler.mutationsSent.containsKey(remote1.endpoint()));
Assert.assertEquals(1, handler.waitingOn());
Assert.assertFalse(getCurrentRepairStatus(handler));
handler.ack(remote1.endpoint());
Assert.assertEquals(1, handler.waitingOn());
Assert.assertFalse(getCurrentRepairStatus(handler));
handler.ack(target1.endpoint());
Assert.assertEquals(0, handler.waitingOn());
Assert.assertTrue(getCurrentRepairStatus(handler));
createRepairHandler(repairs, participants, targets);
}
private boolean getCurrentRepairStatus(BlockingPartitionRepair handler)