mirror of https://github.com/apache/cassandra
Fix operationMode reporting DECOMMISSION_FAILED instead of LEAVING when resuming a failed decommission
Patch by Minal Kyada; reviewed by Caleb Rackliffe and Sam Tunnicliffe for CASSANDRA-21493
This commit is contained in:
parent
aa1447571b
commit
feef3fcf51
|
|
@ -1,4 +1,5 @@
|
|||
6.0-alpha2
|
||||
* Fix operationMode reporting DECOMMISSION_FAILED instead of LEAVING when resuming a failed decommission (CASSANDRA-21493)
|
||||
* Avoid megamorphic calls when serializing and deserializing fixed-length values (CASSANDRA-21536)
|
||||
* Avoid megamorphic calls for Cell.timestamp/ttl/path/localDeletionTimeAsUnsignedInt methods (CASSANDRA-21526)
|
||||
* Allow unreserved keywords as user and identity names in USER and IDENTITY statements (CASSANDRA-21510)
|
||||
|
|
|
|||
|
|
@ -92,6 +92,7 @@ public interface SingleNodeSequences
|
|||
else if (InProgressSequences.isLeave(inProgress))
|
||||
{
|
||||
logger.info("Resuming decommission @ {} (current epoch = {}): {}", inProgress.latestModification, metadata.epoch, inProgress.status());
|
||||
StorageService.instance.clearTransientMode();
|
||||
}
|
||||
else
|
||||
{
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ package org.apache.cassandra.distributed.test;
|
|||
import java.io.IOException;
|
||||
import java.util.concurrent.Callable;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
import net.bytebuddy.ByteBuddy;
|
||||
import net.bytebuddy.dynamic.loading.ClassLoadingStrategy;
|
||||
|
|
@ -141,6 +142,7 @@ public class DecommissionTest extends TestBaseImpl
|
|||
|
||||
public static class BB
|
||||
{
|
||||
static final AtomicBoolean injected = new AtomicBoolean(false);
|
||||
public static void install(ClassLoader classLoader, Integer num)
|
||||
{
|
||||
if (num == 2)
|
||||
|
|
@ -157,7 +159,7 @@ public class DecommissionTest extends TestBaseImpl
|
|||
public static void execute(NodeId leaving, PlacementDeltas startLeave, PlacementDeltas midLeave, PlacementDeltas finishLeave,
|
||||
@SuperCall Callable<?> zuper) throws ExecutionException, InterruptedException
|
||||
{
|
||||
if (!StorageService.instance.isDecommissionFailed())
|
||||
if (injected.compareAndSet(false, true))
|
||||
throw new ExecutionException(new RuntimeException("simulated error in prepareUnbootstrapStreaming"));
|
||||
|
||||
try
|
||||
|
|
|
|||
|
|
@ -53,8 +53,10 @@ import org.apache.cassandra.service.StorageService;
|
|||
import org.apache.cassandra.streaming.StreamSession;
|
||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||
import org.apache.cassandra.tcm.ClusterMetadataService;
|
||||
import org.apache.cassandra.tcm.Epoch;
|
||||
import org.apache.cassandra.tcm.membership.NodeId;
|
||||
import org.apache.cassandra.tcm.membership.NodeVersion;
|
||||
import org.apache.cassandra.tcm.transformations.PrepareLeave;
|
||||
import org.apache.cassandra.tcm.transformations.Startup;
|
||||
import org.apache.cassandra.utils.CassandraVersion;
|
||||
import org.apache.cassandra.utils.FBUtilities;
|
||||
|
|
@ -62,6 +64,8 @@ import org.apache.cassandra.utils.FBUtilities;
|
|||
import static net.bytebuddy.matcher.ElementMatchers.named;
|
||||
import static org.apache.cassandra.distributed.api.Feature.GOSSIP;
|
||||
import static org.apache.cassandra.distributed.api.Feature.NETWORK;
|
||||
import static org.apache.cassandra.distributed.shared.ClusterUtils.pauseBeforeCommit;
|
||||
import static org.apache.cassandra.distributed.shared.ClusterUtils.unpauseCommits;
|
||||
import static org.apache.cassandra.distributed.shared.NetworkTopology.dcAndRack;
|
||||
import static org.apache.cassandra.distributed.shared.NetworkTopology.networkTopology;
|
||||
import static org.apache.cassandra.distributed.test.ring.BootstrapTest.populate;
|
||||
|
|
@ -84,6 +88,43 @@ public class DecommissionTest extends TestBaseImpl
|
|||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testOperationModeOnDecomResume() throws Exception
|
||||
{
|
||||
// Node 2's first decomission attempt fails mid-stream (injected by BB), leaving it in
|
||||
// DECOMISSION_FAILED. On resume, operationMode() must transition back to LEAVING before
|
||||
// MID_LEAVE is committed. We pause the CMS just before that commit to assert the mode
|
||||
// in the window where streaming has finished but the epoch has not yet advanced.
|
||||
try (Cluster cluster = builder().withNodes(3)
|
||||
.withConfig(config -> config.with(NETWORK, GOSSIP))
|
||||
.withInstanceInitializer(BB::install)
|
||||
.start())
|
||||
{
|
||||
populate(cluster, 0, 100, 1, 2, ConsistencyLevel.QUORUM);
|
||||
|
||||
IInvokableInstance cmsNode = cluster.get(1);
|
||||
IInvokableInstance leavingNode = cluster.get(2);
|
||||
|
||||
// --force required: cie_internal keyspace has RF=3, decommission would fail replication check otherwise
|
||||
leavingNode.nodetoolResult("decommission", "--force").asserts().failure();
|
||||
leavingNode.runOnInstance(() -> assertEquals(StorageService.Mode.DECOMMISSION_FAILED, StorageService.instance.operationMode()));
|
||||
|
||||
Callable<Epoch> midLeavePaused = pauseBeforeCommit(cmsNode, e -> e instanceof PrepareLeave.MidLeave);
|
||||
|
||||
Thread resumeThread = new Thread(() -> leavingNode.nodetoolResult("decommission").asserts().success());
|
||||
resumeThread.start();
|
||||
midLeavePaused.call();
|
||||
|
||||
leavingNode.runOnInstance(() ->
|
||||
assertEquals("operationMode during resumed decommission streaming should be LEAVING, not DECOMMISSION_FAILED",
|
||||
StorageService.Mode.LEAVING,
|
||||
StorageService.instance.operationMode()));
|
||||
|
||||
unpauseCommits(cmsNode);
|
||||
resumeThread.join(TimeUnit.MINUTES.toMillis(2));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testAddressReuseAfterDecommission() throws IOException, ExecutionException, InterruptedException
|
||||
{
|
||||
|
|
|
|||
Loading…
Reference in New Issue