diff --git a/CHANGES.txt b/CHANGES.txt
index c4fde4445e..ffe042e3f3 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -1,4 +1,5 @@
5.1
+ * Parallelized UCS compactions (CASSANDRA-18802)
* Avoid prepared statement invalidation race when committing schema changes (CASSANDRA-20116)
* Restore optimization in MultiCBuilder around building one clustering (CASSANDRA-20129)
* Consolidate all snapshot management to SnapshotManager and introduce SnapshotManagerMBean (CASSANDRA-18111)
diff --git a/NEWS.txt b/NEWS.txt
index 740be2eb96..ae1ba39d30 100644
--- a/NEWS.txt
+++ b/NEWS.txt
@@ -92,6 +92,16 @@ New features
- There is new MBean of name org.apache.cassandra.service.snapshot:type=SnapshotManager which exposes user-facing
snapshot operations. Snapshot-related methods on StorageServiceMBean are still present and functional
but marked as deprecated.
+ - The unified compaction strategy will now parallelize individual compactions by splitting operations into separate
+ tasks per output shard. This significantly shortens compaction durations but cannot take advantage of preemptive
+ SSTable opening and thus can be less efficient if the strategy is configured to create large sstables.
+ This functionality is controlled by the parallelize_output_shards compaction option (the default can also be set
+ using -Dunified_compaction.parallelize_output_shards=false/true).
+ - Compaction parallelism for the unified compaction strategy also applies to major compactions. Because the number
+ of shards for a major compaction is usually quite high, this can dramatically improve the duration of major
+ compactions. To avoid the possibility of starving background compactions from resources, the number of threads
+ used for major compactions is limited to half the available compaction threads by default, and can be controlled
+ by a new --jobs / -j option of nodetool compact.
Upgrading
diff --git a/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java b/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java
index 9862f22782..e409e975f6 100644
--- a/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java
+++ b/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java
@@ -610,6 +610,7 @@ public enum CassandraRelevantProperties
UCS_BASE_SHARD_COUNT("unified_compaction.base_shard_count", "4"),
UCS_MIN_SSTABLE_SIZE("unified_compaction.min_sstable_size", "100MiB"),
UCS_OVERLAP_INCLUSION_METHOD("unified_compaction.overlap_inclusion_method"),
+ UCS_PARALLELIZE_OUTPUT_SHARDS("unified_compaction.parallelize_output_shards", "true"),
UCS_SCALING_PARAMETER("unified_compaction.scaling_parameters", "T4"),
UCS_SSTABLE_GROWTH("unified_compaction.sstable_growth", "0.333"),
UCS_SURVIVAL_FACTOR("unified_compaction.survival_factor", "1"),
diff --git a/src/java/org/apache/cassandra/db/ColumnFamilyStore.java b/src/java/org/apache/cassandra/db/ColumnFamilyStore.java
index cbd3d97c6e..91a3a40152 100644
--- a/src/java/org/apache/cassandra/db/ColumnFamilyStore.java
+++ b/src/java/org/apache/cassandra/db/ColumnFamilyStore.java
@@ -2208,12 +2208,22 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean, Memtable.Owner
public void forceMajorCompaction()
{
- forceMajorCompaction(false);
+ forceMajorCompaction(false, -1);
}
public void forceMajorCompaction(boolean splitOutput)
{
- CompactionManager.instance.performMaximal(this, splitOutput);
+ forceMajorCompaction(splitOutput, -1);
+ }
+
+ public void forceMajorCompaction(int permittedParallelism)
+ {
+ forceMajorCompaction(false, permittedParallelism);
+ }
+
+ public void forceMajorCompaction(boolean splitOutput, int permittedParallelism)
+ {
+ CompactionManager.instance.performMaximal(this, splitOutput, permittedParallelism);
}
@Override
diff --git a/src/java/org/apache/cassandra/db/ColumnFamilyStoreMBean.java b/src/java/org/apache/cassandra/db/ColumnFamilyStoreMBean.java
index 7d7b9e58ee..33cd6ebbc2 100644
--- a/src/java/org/apache/cassandra/db/ColumnFamilyStoreMBean.java
+++ b/src/java/org/apache/cassandra/db/ColumnFamilyStoreMBean.java
@@ -52,6 +52,15 @@ public interface ColumnFamilyStoreMBean
*/
public void forceMajorCompaction(boolean splitOutput) throws ExecutionException, InterruptedException;
+ /**
+ * force a major compaction of this column family
+ *
+ * @param permittedParallelism The maximum number of compaction threads that can be used by the operation.
+ * If 0, the operation can use all available threads.
+ * If <0, the default parallelism will be used.
+ */
+ public void forceMajorCompaction(int permittedParallelism) throws ExecutionException, InterruptedException;
+
/**
* Forces a major compaction of specified token ranges in this column family.
*
diff --git a/src/java/org/apache/cassandra/db/compaction/AbstractCompactionStrategy.java b/src/java/org/apache/cassandra/db/compaction/AbstractCompactionStrategy.java
index fd210d6d02..9d189b9975 100644
--- a/src/java/org/apache/cassandra/db/compaction/AbstractCompactionStrategy.java
+++ b/src/java/org/apache/cassandra/db/compaction/AbstractCompactionStrategy.java
@@ -92,7 +92,7 @@ public abstract class AbstractCompactionStrategy
private final Directories directories;
/**
- * pause/resume/getNextBackgroundTask must synchronize. This guarantees that after pause completes,
+ * pause/resume/getNextBackgroundTasks must synchronize. This guarantees that after pause completes,
* no new tasks will be generated; or put another way, pause can't run until in-progress tasks are
* done being created.
*
@@ -176,21 +176,22 @@ public abstract class AbstractCompactionStrategy
/**
* @param gcBefore throw away tombstones older than this
*
- * @return the next background/minor compaction task to run; null if nothing to do.
+ * @return the next background/minor compaction tasks to run; an empty collection if nothing to do.
*
* Is responsible for marking its sstables as compaction-pending.
*/
- public abstract AbstractCompactionTask getNextBackgroundTask(final long gcBefore);
+ public abstract Collection getNextBackgroundTasks(final long gcBefore);
/**
- * @param gcBefore throw away tombstones older than this
- *
- * @return a compaction task that should be run to compact this columnfamilystore
- * as much as possible. Null if nothing to do.
- *
+ * @param gcBefore throw away tombstones older than this
+ * @return A list of compaction tasks that should be run to compact this columnfamilystore
+ * as much as possible. Empty if nothing to do.
+ * Order matters if a parallelism limit is applied, as the tasks are run in way that parallelizes ones that
+ * are close together in the list.
+ *
* Is responsible for marking its sstables as compaction-pending.
*/
- public abstract Collection getMaximalTask(final long gcBefore, boolean splitOutput);
+ public abstract List getMaximalTasks(final long gcBefore, boolean splitOutput);
/**
* @param sstables SSTables to compact. Must be marked as compacting.
@@ -418,7 +419,7 @@ public abstract class AbstractCompactionStrategy
// there is no overlap, tombstones are safely droppable
return true;
}
- else if (CompactionController.getFullyExpiredSSTables(cfs, Collections.singleton(sstable), overlaps, gcBefore).size() > 0)
+ else if (CompactionController.getFullyExpiredSSTables(cfs, Collections.singleton(sstable), s -> overlaps, gcBefore).size() > 0)
{
return true;
}
diff --git a/src/java/org/apache/cassandra/db/compaction/AbstractCompactionTask.java b/src/java/org/apache/cassandra/db/compaction/AbstractCompactionTask.java
index 40c4cb49e1..d0b3dc498f 100644
--- a/src/java/org/apache/cassandra/db/compaction/AbstractCompactionTask.java
+++ b/src/java/org/apache/cassandra/db/compaction/AbstractCompactionTask.java
@@ -23,18 +23,16 @@ import java.util.Set;
import com.google.common.base.Preconditions;
import org.apache.cassandra.db.ColumnFamilyStore;
-import org.apache.cassandra.db.Directories;
-import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.io.FSDiskFullWriteError;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.utils.TimeUUID;
import org.apache.cassandra.utils.WrappedRunnable;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
public abstract class AbstractCompactionTask extends WrappedRunnable
{
protected final ColumnFamilyStore cfs;
- protected LifecycleTransaction transaction;
+ protected ILifecycleTransaction transaction;
protected boolean isUserDefined;
protected OperationType compactionType;
@@ -42,18 +40,30 @@ public abstract class AbstractCompactionTask extends WrappedRunnable
* @param cfs
* @param transaction the modifying managing the status of the sstables we're replacing
*/
- public AbstractCompactionTask(ColumnFamilyStore cfs, LifecycleTransaction transaction)
+ public AbstractCompactionTask(ColumnFamilyStore cfs, ILifecycleTransaction transaction)
{
this.cfs = cfs;
this.transaction = transaction;
this.isUserDefined = false;
this.compactionType = OperationType.COMPACTION;
- // enforce contract that caller should mark sstables compacting
- Set compacting = transaction.tracker.getCompacting();
- for (SSTableReader sstable : transaction.originals())
- assert compacting.contains(sstable) : sstable.getFilename() + " is not correctly marked compacting";
- validateSSTables(transaction.originals());
+ try
+ {
+ if (!transaction.isOffline())
+ {
+ // enforce contract that caller should mark sstables compacting
+ Set compacting = cfs.getTracker().getCompacting();
+ for (SSTableReader sstable : transaction.originals())
+ assert compacting.contains(sstable) : sstable.getFilename() + " is not correctly marked compacting";
+ }
+
+ validateSSTables(transaction.originals());
+ }
+ catch (Throwable err)
+ {
+ rejected();
+ throw err;
+ }
}
/**
@@ -61,7 +71,7 @@ public abstract class AbstractCompactionTask extends WrappedRunnable
*/
private void validateSSTables(Set sstables)
{
- // do not allow to be compacted together
+ // do not allow sstables in different repair states to be compacted together
if (!sstables.isEmpty())
{
Iterator iter = sstables.iterator();
@@ -93,11 +103,11 @@ public abstract class AbstractCompactionTask extends WrappedRunnable
/**
* executes the task and unmarks sstables compacting
*/
- public int execute(ActiveCompactionsTracker activeCompactions)
+ public void execute(ActiveCompactionsTracker activeCompactions)
{
try
{
- return executeInternal(activeCompactions);
+ executeInternal(activeCompactions);
}
catch(FSDiskFullWriteError e)
{
@@ -107,12 +117,21 @@ public abstract class AbstractCompactionTask extends WrappedRunnable
}
finally
{
- transaction.close();
+ cleanup();
}
}
- public abstract CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables);
- protected abstract int executeInternal(ActiveCompactionsTracker activeCompactions);
+ protected void cleanup()
+ {
+ transaction.close();
+ }
+
+ public void rejected()
+ {
+ cleanup();
+ }
+
+ protected abstract void executeInternal(ActiveCompactionsTracker activeCompactions);
public AbstractCompactionTask setUserDefined(boolean isUserDefined)
{
diff --git a/src/java/org/apache/cassandra/db/compaction/AbstractStrategyHolder.java b/src/java/org/apache/cassandra/db/compaction/AbstractStrategyHolder.java
index a1471c7746..2db7e1db60 100644
--- a/src/java/org/apache/cassandra/db/compaction/AbstractStrategyHolder.java
+++ b/src/java/org/apache/cassandra/db/compaction/AbstractStrategyHolder.java
@@ -53,15 +53,15 @@ public abstract class AbstractStrategyHolder
public static class TaskSupplier implements Comparable
{
private final int numRemaining;
- private final Supplier supplier;
+ private final Supplier> supplier;
- TaskSupplier(int numRemaining, Supplier supplier)
+ TaskSupplier(int numRemaining, Supplier> supplier)
{
this.numRemaining = numRemaining;
this.supplier = supplier;
}
- public AbstractCompactionTask getTask()
+ public Collection getTasks()
{
return supplier.get();
}
diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionController.java b/src/java/org/apache/cassandra/db/compaction/CompactionController.java
index b4096c81dd..1753d7e664 100644
--- a/src/java/org/apache/cassandra/db/compaction/CompactionController.java
+++ b/src/java/org/apache/cassandra/db/compaction/CompactionController.java
@@ -26,6 +26,7 @@ import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.function.LongPredicate;
+import java.util.function.UnaryOperator;
import com.google.common.base.Predicates;
import com.google.common.collect.Iterables;
@@ -130,7 +131,12 @@ public class CompactionController extends AbstractCompactionController
public Set getFullyExpiredSSTables()
{
- return getFullyExpiredSSTables(cfs, compacting, overlappingSSTables, gcBefore, ignoreOverlaps());
+ return getFullyExpiredSSTables(cfs,
+ compacting,
+ x -> overlappingSSTables != null ? overlappingSSTables
+ : Collections.emptyList(),
+ gcBefore,
+ ignoreOverlaps());
}
/**
@@ -145,20 +151,20 @@ public class CompactionController extends AbstractCompactionController
*
* @param cfStore
* @param compacting we take the drop-candidates from this set, it is usually the sstables included in the compaction
- * @param overlapping the sstables that overlap the ones in compacting.
+ * @param overlappingSupplier called on the compacting sstables to compute the set of sstables that overlap with them if needed
* @param gcBefore
* @param ignoreOverlaps don't check if data shadows/overlaps any data in other sstables
* @return
*/
public static Set getFullyExpiredSSTables(ColumnFamilyStore cfStore,
Iterable compacting,
- Iterable overlapping,
+ UnaryOperator> overlappingSupplier,
long gcBefore,
boolean ignoreOverlaps)
{
logger.trace("Checking droppable sstables in {}", cfStore);
- if (NEVER_PURGE_TOMBSTONES_PROPERTY_VALUE || compacting == null || cfStore.getNeverPurgeTombstones() || overlapping == null)
+ if (NEVER_PURGE_TOMBSTONES_PROPERTY_VALUE || compacting == null || cfStore.getNeverPurgeTombstones())
return Collections.emptySet();
if (cfStore.getCompactionStrategyManager().onlyPurgeRepairedTombstones() && !Iterables.all(compacting, SSTableReader::isRepaired))
@@ -183,7 +189,7 @@ public class CompactionController extends AbstractCompactionController
List candidates = new ArrayList<>();
long minTimestamp = Long.MAX_VALUE;
- for (SSTableReader sstable : overlapping)
+ for (SSTableReader sstable : overlappingSupplier.apply(compacting))
{
// Overlapping might include fully expired sstables. What we care about here is
// the min timestamp of the overlapping sstables that actually contain live data.
@@ -230,10 +236,10 @@ public class CompactionController extends AbstractCompactionController
public static Set getFullyExpiredSSTables(ColumnFamilyStore cfStore,
Iterable compacting,
- Iterable overlapping,
+ UnaryOperator> overlappingSupplier,
long gcBefore)
{
- return getFullyExpiredSSTables(cfStore, compacting, overlapping, gcBefore, false);
+ return getFullyExpiredSSTables(cfStore, compacting, overlappingSupplier, gcBefore, false);
}
/**
diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionManager.java b/src/java/org/apache/cassandra/db/compaction/CompactionManager.java
index 67b847e0ef..801b1ec06b 100644
--- a/src/java/org/apache/cassandra/db/compaction/CompactionManager.java
+++ b/src/java/org/apache/cassandra/db/compaction/CompactionManager.java
@@ -34,6 +34,7 @@ import java.util.concurrent.ExecutorService;
import java.util.concurrent.RejectedExecutionException;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.atomic.AtomicInteger;
+import java.util.concurrent.atomic.AtomicReference;
import java.util.function.BooleanSupplier;
import java.util.function.Predicate;
import java.util.function.Supplier;
@@ -57,6 +58,7 @@ import com.google.common.util.concurrent.RateLimiter;
import com.google.common.util.concurrent.Uninterruptibles;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
+
import com.codahale.metrics.Meter;
import net.openhft.chronicle.core.util.ThrowingSupplier;
import org.apache.cassandra.cache.AutoSavingCache;
@@ -108,6 +110,8 @@ import org.apache.cassandra.schema.TableMetadata;
import org.apache.cassandra.service.ActiveRepairService;
import org.apache.cassandra.service.StorageService;
import org.apache.cassandra.streaming.PreviewKind;
+import org.apache.cassandra.tcm.ClusterMetadata;
+import org.apache.cassandra.tcm.ownership.DataPlacement;
import org.apache.cassandra.utils.FBUtilities;
import org.apache.cassandra.utils.JVMStabilityInspector;
import org.apache.cassandra.utils.MBeanWrapper;
@@ -115,10 +119,10 @@ import org.apache.cassandra.utils.OutputHandler;
import org.apache.cassandra.utils.Throwables;
import org.apache.cassandra.utils.TimeUUID;
import org.apache.cassandra.utils.WrappedRunnable;
-import org.apache.cassandra.tcm.ClusterMetadata;
-import org.apache.cassandra.tcm.ownership.DataPlacement;
+import org.apache.cassandra.utils.concurrent.AsyncPromise;
import org.apache.cassandra.utils.concurrent.Future;
import org.apache.cassandra.utils.concurrent.ImmediateFuture;
+import org.apache.cassandra.utils.concurrent.Promise;
import org.apache.cassandra.utils.concurrent.Refs;
import static java.util.Collections.singleton;
@@ -268,9 +272,9 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
cfs.getCompactionStrategyManager().getName());
List> futures = new ArrayList<>(1);
- Future> fut = executor.submitIfRunning(new BackgroundCompactionCandidate(cfs), "background task");
- if (!fut.isCancelled())
- futures.add(fut);
+ Promise promise = new AsyncPromise<>();
+ if (!executor.submitIfRunning(new BackgroundCompactionCandidate(cfs, promise), "background task").isCancelled())
+ futures.add(promise);
else
compactingCF.remove(cfs);
return futures;
@@ -356,16 +360,22 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
class BackgroundCompactionCandidate implements Runnable
{
private final ColumnFamilyStore cfs;
+ private final Promise toSignalWhenDone;
+ private final AtomicReference asyncErrors = new AtomicReference<>(null);
- BackgroundCompactionCandidate(ColumnFamilyStore cfs)
+ BackgroundCompactionCandidate(ColumnFamilyStore cfs, Promise toSignalWhenDone)
{
compactingCF.add(cfs);
this.cfs = cfs;
+ this.toSignalWhenDone = toSignalWhenDone;
}
public void run()
{
boolean ranCompaction = false;
+ boolean async = false;
+ Throwable error = null;
+ Collection tasks = null;
try
{
logger.trace("Checking {}.{}", cfs.getKeyspaceName(), cfs.name);
@@ -376,24 +386,131 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
}
CompactionStrategyManager strategy = cfs.getCompactionStrategyManager();
- AbstractCompactionTask task = strategy.getNextBackgroundTask(getDefaultGcBefore(cfs, FBUtilities.nowInSeconds()));
- if (task == null)
+ tasks = strategy.getNextBackgroundTasks(getDefaultGcBefore(cfs, FBUtilities.nowInSeconds()));
+ if (tasks == null || tasks.isEmpty())
{
if (DatabaseDescriptor.automaticSSTableUpgrade())
ranCompaction = maybeRunUpgradeTask(strategy);
}
- else
+ else if (tasks.size() == 1)
{
- task.execute(active);
+ // If just one task, run it directly on this thread
+ for (AbstractCompactionTask task : tasks)
+ task.execute(active);
ranCompaction = true;
}
+ else
+ {
+ // more than 1 task: we need to do this outside the catch and complete block
+ async = true;
+ }
+ }
+ catch (Throwable t)
+ {
+ error = t;
+ }
+
+ if (!async)
+ {
+ complete(ranCompaction, error);
+ Throwables.maybeFail(error);
+ }
+ else // async
+ processTasksAsync(tasks);
+ }
+
+ private void processTasksAsync(Collection tasks)
+ {
+ assert tasks != null;
+
+ // We should only signal overall completion when all tasks are done
+ AtomicInteger toComplete = new AtomicInteger(tasks.size());
+
+ AbstractCompactionTask lastTask = null;
+ // Submit all but the last task for execution,
+ for (AbstractCompactionTask task : tasks)
+ {
+ if (lastTask != null)
+ {
+ AbstractCompactionTask toRun = lastTask;
+ try
+ {
+ executor.submit(() -> runTask(toRun, toComplete));
+ }
+ catch (RejectedExecutionException e)
+ {
+ logger.debug("Failed to submit background compaction task: {}", e.getMessage());
+ rejectTask(toRun, toComplete);
+ }
+ }
+ lastTask = task;
+ }
+
+ // and run the last task directly in this thread.
+ assert lastTask != null;
+ runTask(lastTask, toComplete);
+ }
+
+ private void runTask(AbstractCompactionTask task, AtomicInteger toComplete)
+ {
+ try
+ {
+ task.execute(active);
+ }
+ catch (Throwable t)
+ {
+ addAsyncError(t);
}
finally
{
- compactingCF.remove(cfs);
+ if (toComplete.decrementAndGet() == 0)
+ complete(true, asyncErrors.get());
}
- if (ranCompaction) // only submit background if we actually ran a compaction - otherwise we end up in an infinite loop submitting noop background tasks
- submitBackground(cfs);
+ }
+
+ private void rejectTask(AbstractCompactionTask task, AtomicInteger toComplete)
+ {
+ try
+ {
+ task.rejected();
+ throw new RuntimeException("Failed to submit background compaction task");
+ }
+ catch (Throwable t) // make sure we catch exceptions thrown by task.rejected() as well
+ {
+ addAsyncError(t);
+ }
+ finally
+ {
+ if (toComplete.decrementAndGet() == 0)
+ complete(false, asyncErrors.get());
+ }
+ }
+
+ private void complete(boolean submitNew, Throwable error)
+ {
+ compactingCF.remove(cfs);
+ if (error == null)
+ {
+ toSignalWhenDone.setSuccess(null);
+ if (submitNew)
+ submitBackground(cfs);
+ }
+ else
+ toSignalWhenDone.setFailure(error);
+ }
+
+ private void addAsyncError(Throwable t)
+ {
+ asyncErrors.accumulateAndGet(t, (a, b) ->
+ {
+ if (a == null)
+ return b;
+ else
+ {
+ a.addSuppressed(b);
+ return a;
+ }
+ });
}
boolean maybeRunUpgradeTask(CompactionStrategyManager strategy)
@@ -423,7 +540,7 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
@VisibleForTesting
public BackgroundCompactionCandidate getBackgroundCompactionCandidate(ColumnFamilyStore cfs)
{
- return new BackgroundCompactionCandidate(cfs);
+ return new BackgroundCompactionCandidate(cfs, new AsyncPromise<>());
}
/**
@@ -721,7 +838,7 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
}
catch (Throwable t)
{
- logger.warn(String.format("Unable to cancel %s from transaction %s", sstable, transaction.opId()), t);
+ logger.warn(String.format("Unable to cancel %s from transaction %s", sstable, transaction.opIdString()), t);
}
}
else
@@ -1004,22 +1121,32 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
return fullyContainedSSTables;
}
- public void performMaximal(final ColumnFamilyStore cfStore, boolean splitOutput)
+ public void performMaximal(final ColumnFamilyStore cfStore)
{
- FBUtilities.waitOnFutures(submitMaximal(cfStore, getDefaultGcBefore(cfStore, FBUtilities.nowInSeconds()), splitOutput));
+ performMaximal(cfStore, false, -1);
}
- public List> submitMaximal(final ColumnFamilyStore cfStore, final long gcBefore, boolean splitOutput)
+ public void performMaximal(final ColumnFamilyStore cfStore, boolean splitOutput, int permittedParallelism)
{
- return submitMaximal(cfStore, gcBefore, splitOutput, OperationType.MAJOR_COMPACTION);
+ FBUtilities.waitOnFutures(submitMaximal(cfStore, getDefaultGcBefore(cfStore, FBUtilities.nowInSeconds()), splitOutput, permittedParallelism));
}
- public List> submitMaximal(final ColumnFamilyStore cfStore, final long gcBefore, boolean splitOutput, OperationType operationType)
+ public List> submitMaximal(final ColumnFamilyStore cfStore, final long gcBefore, boolean splitOutput, int permittedParallelism)
{
+ return submitMaximal(cfStore, gcBefore, splitOutput, permittedParallelism, OperationType.MAJOR_COMPACTION);
+ }
+
+ public List> submitMaximal(final ColumnFamilyStore cfStore, final long gcBefore, boolean splitOutput, int permittedParallelism, OperationType operationType)
+ {
+ if (permittedParallelism < 0)
+ permittedParallelism = getCoreCompactorThreads() / 2;
+ else if (permittedParallelism == 0)
+ permittedParallelism = Integer.MAX_VALUE;
+
// here we compute the task off the compaction executor, so having that present doesn't
// confuse runWithCompactionsDisabled -- i.e., we don't want to deadlock ourselves, waiting
// for ourselves to finish/acknowledge cancellation before continuing.
- CompactionTasks tasks = cfStore.getCompactionStrategyManager().getMaximalTasks(gcBefore, splitOutput, operationType);
+ CompactionTasks tasks = cfStore.getCompactionStrategyManager().getMaximalTasks(gcBefore, splitOutput, permittedParallelism, operationType);
if (tasks.isEmpty())
return Collections.emptyList();
@@ -1043,6 +1170,8 @@ public class CompactionManager implements CompactionManagerMBean, ICompactionMan
Future> fut = executor.submitIfRunning(runnable, "maximal task");
if (!fut.isCancelled())
futures.add(fut);
+ else
+ task.rejected();
}
if (nonEmptyTasks > 1)
logger.info("Major compaction will not result in a single sstable - repaired and unrepaired data is kept separate and compaction runs per data_file_directory.");
diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionStrategyHolder.java b/src/java/org/apache/cassandra/db/compaction/CompactionStrategyHolder.java
index becd3b954a..9bc4d6d7cb 100644
--- a/src/java/org/apache/cassandra/db/compaction/CompactionStrategyHolder.java
+++ b/src/java/org/apache/cassandra/db/compaction/CompactionStrategyHolder.java
@@ -106,7 +106,7 @@ public class CompactionStrategyHolder extends AbstractStrategyHolder
{
List suppliers = new ArrayList<>(strategies.size());
for (AbstractCompactionStrategy strategy : strategies)
- suppliers.add(new TaskSupplier(strategy.getEstimatedRemainingTasks(), () -> strategy.getNextBackgroundTask(gcBefore)));
+ suppliers.add(new TaskSupplier(strategy.getEstimatedRemainingTasks(), () -> strategy.getNextBackgroundTasks(gcBefore)));
return suppliers;
}
@@ -117,7 +117,7 @@ public class CompactionStrategyHolder extends AbstractStrategyHolder
List tasks = new ArrayList<>(strategies.size());
for (AbstractCompactionStrategy strategy : strategies)
{
- Collection task = strategy.getMaximalTask(gcBefore, splitOutput);
+ Collection task = strategy.getMaximalTasks(gcBefore, splitOutput);
if (task != null)
tasks.addAll(task);
}
diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionStrategyManager.java b/src/java/org/apache/cassandra/db/compaction/CompactionStrategyManager.java
index d022be45c4..dfb908425a 100644
--- a/src/java/org/apache/cassandra/db/compaction/CompactionStrategyManager.java
+++ b/src/java/org/apache/cassandra/db/compaction/CompactionStrategyManager.java
@@ -195,26 +195,26 @@ public class CompactionStrategyManager implements INotificationConsumer
*
* Returns a task for the compaction strategy that needs it the most (most estimated remaining tasks)
*/
- public AbstractCompactionTask getNextBackgroundTask(long gcBefore)
+ public Collection getNextBackgroundTasks(long gcBefore)
{
maybeReloadDiskBoundaries();
readLock.lock();
try
{
if (!isEnabled())
- return null;
+ return Collections.emptyList();
int numPartitions = getNumTokenPartitions();
// first try to promote/demote sstables from completed repairs
- AbstractCompactionTask repairFinishedTask;
- repairFinishedTask = pendingRepairs.getNextRepairFinishedTask();
- if (repairFinishedTask != null)
- return repairFinishedTask;
+ Collection repairFinishedTasks;
+ repairFinishedTasks = pendingRepairs.getNextRepairFinishedTasks();
+ if (repairFinishedTasks != null && !repairFinishedTasks.isEmpty())
+ return repairFinishedTasks;
- repairFinishedTask = transientRepairs.getNextRepairFinishedTask();
- if (repairFinishedTask != null)
- return repairFinishedTask;
+ repairFinishedTasks = transientRepairs.getNextRepairFinishedTasks();
+ if (repairFinishedTasks != null && !repairFinishedTasks.isEmpty())
+ return repairFinishedTasks;
// sort compaction task suppliers by remaining tasks descending
List suppliers = new ArrayList<>(numPartitions * holders.size());
@@ -226,12 +226,12 @@ public class CompactionStrategyManager implements INotificationConsumer
// return the first non-null task
for (TaskSupplier supplier : suppliers)
{
- AbstractCompactionTask task = supplier.getTask();
- if (task != null)
- return task;
+ Collection tasks = supplier.getTasks();
+ if (tasks != null && !tasks.isEmpty())
+ return tasks;
}
- return null;
+ return Collections.emptyList();
}
finally
{
@@ -1079,7 +1079,7 @@ public class CompactionStrategyManager implements INotificationConsumer
}
}
- public CompactionTasks getMaximalTasks(final long gcBefore, final boolean splitOutput, OperationType operationType)
+ public CompactionTasks getMaximalTasks(final long gcBefore, final boolean splitOutput, int permittedParallelism, OperationType operationType)
{
maybeReloadDiskBoundaries();
// runWithCompactionsDisabled cancels active compactions and disables them, then we are able
@@ -1097,6 +1097,7 @@ public class CompactionStrategyManager implements INotificationConsumer
tasks.add(task.setCompactionType(operationType));
}
}
+ tasks = CompositeCompactionTask.applyParallelismLimit(tasks, permittedParallelism);
}
finally
{
diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionTask.java b/src/java/org/apache/cassandra/db/compaction/CompactionTask.java
index 4f27736f26..0cbd9fe3a8 100644
--- a/src/java/org/apache/cassandra/db/compaction/CompactionTask.java
+++ b/src/java/org/apache/cassandra/db/compaction/CompactionTask.java
@@ -26,10 +26,13 @@ import java.util.Set;
import java.util.concurrent.TimeUnit;
import com.google.common.base.Predicate;
+import com.google.common.collect.ImmutableList;
import com.google.common.collect.ImmutableMap;
import com.google.common.collect.Iterables;
-import com.google.common.collect.Sets;
import com.google.common.util.concurrent.RateLimiter;
+import org.apache.cassandra.db.compaction.unified.UnifiedCompactionTask;
+import org.apache.cassandra.dht.Range;
+import org.apache.cassandra.dht.Token;
import org.apache.commons.lang3.StringUtils;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
@@ -40,7 +43,7 @@ import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.SystemKeyspace;
import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.compaction.writers.DefaultCompactionWriter;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.sstable.metadata.MetadataCollector;
import org.apache.cassandra.io.util.File;
@@ -64,12 +67,12 @@ public class CompactionTask extends AbstractCompactionTask
protected static long totalBytesCompacted = 0;
private ActiveCompactionsTracker activeCompactions;
- public CompactionTask(ColumnFamilyStore cfs, LifecycleTransaction txn, long gcBefore)
+ public CompactionTask(ColumnFamilyStore cfs, ILifecycleTransaction txn, long gcBefore)
{
this(cfs, txn, gcBefore, false);
}
- public CompactionTask(ColumnFamilyStore cfs, LifecycleTransaction txn, long gcBefore, boolean keepOriginals)
+ public CompactionTask(ColumnFamilyStore cfs, ILifecycleTransaction txn, long gcBefore, boolean keepOriginals)
{
super(cfs, txn);
this.gcBefore = gcBefore;
@@ -81,44 +84,73 @@ public class CompactionTask extends AbstractCompactionTask
return totalBytesCompacted += bytesCompacted;
}
- protected int executeInternal(ActiveCompactionsTracker activeCompactions)
+ protected void executeInternal(ActiveCompactionsTracker activeCompactions)
{
this.activeCompactions = activeCompactions == null ? ActiveCompactionsTracker.NOOP : activeCompactions;
run();
- return transaction.originals().size();
}
public boolean reduceScopeForLimitedSpace(Set nonExpiredSSTables, long expectedSize)
{
- if (partialCompactionsAcceptable() && transaction.originals().size() > 1)
+ if (partialCompactionsAcceptable() && nonExpiredSSTables.size() > 1)
{
// Try again w/o the largest one.
SSTableReader removedSSTable = cfs.getMaxSizeFile(nonExpiredSSTables);
logger.warn("insufficient space to compact all requested files. {}MiB required, {} for compaction {} - removing largest SSTable: {}",
(float) expectedSize / 1024 / 1024,
- StringUtils.join(transaction.originals(), ", "),
- transaction.opId(),
+ StringUtils.join(nonExpiredSSTables, ", "),
+ transaction.opIdString(),
removedSSTable);
// Note that we have removed files that are still marked as compacting.
// This suboptimal but ok since the caller will unmark all the sstables at the end.
transaction.cancel(removedSSTable);
+ nonExpiredSSTables.remove(removedSSTable);
return true;
}
return false;
}
+ /**
+ * @return The token range that the operation should compact. This is usually null, but if we have a parallelizable
+ * multi-task operation (see {@link UnifiedCompactionStrategy#createCompactionTasks}), it will specify a subrange.
+ */
+ protected Range tokenRange()
+ {
+ return null;
+ }
+
+ /**
+ * @return The set of input sstables for this compaction. This must be a subset of the transaction originals and
+ * must reflect any removal of sstables from the originals set for correct overlap tracking.
+ * See {@link UnifiedCompactionTask} for an example.
+ */
+ protected Set inputSSTables()
+ {
+ return transaction.originals();
+ }
+
+ /**
+ * @return True if the task should try to limit the operation size to the available space by removing sstables from
+ * the compacting set. This cannot be done if this is part of a multi-task operation with a shared transaction.
+ */
+ protected boolean shouldReduceScopeForSpace()
+ {
+ return true;
+ }
+
/**
* For internal use and testing only. The rest of the system should go through the submit* methods,
* which are properly serialized.
* Caller is in charge of marking/unmarking the sstables as compacting.
*/
+ @Override
protected void runMayThrow() throws Exception
{
// The collection of sstables passed may be empty (but not null); even if
// it is not empty, it may compact down to nothing if all rows are deleted.
assert transaction != null;
- if (transaction.originals().isEmpty())
+ if (inputSSTables().isEmpty())
return;
// Note that the current compaction strategy, is not necessarily the one this task was created under.
@@ -131,22 +163,35 @@ public class CompactionTask extends AbstractCompactionTask
SnapshotManager.instance.takeSnapshot(options);
}
- try (CompactionController controller = getCompactionController(transaction.originals()))
+ try (CompactionController controller = getCompactionController(inputSSTables()))
{
+ // Note: the controller set-up above relies on using the transaction-provided sstable list, from which
+ // fully-expired sstables should not be removed (so that the overlap tracker does not include them), but
+ // sstables excluded for scope reduction should be removed.
+ Set actuallyCompact = new HashSet<>(inputSSTables());
final Set fullyExpiredSSTables = controller.getFullyExpiredSSTables();
+ if (!fullyExpiredSSTables.isEmpty())
+ {
+ logger.debug("Compaction {} dropping expired sstables: {}", transaction.opIdString(), fullyExpiredSSTables);
+ actuallyCompact.removeAll(fullyExpiredSSTables);
+ }
TimeUUID taskId = transaction.opId();
// select SSTables to compact based on available disk space.
- if (!buildCompactionCandidatesForAvailableDiskSpace(fullyExpiredSSTables, taskId))
+ final boolean hasExpirations = !fullyExpiredSSTables.isEmpty();
+ if ((shouldReduceScopeForSpace() && !buildCompactionCandidatesForAvailableDiskSpace(actuallyCompact, hasExpirations, taskId))
+ || hasExpirations)
{
// The set of sstables has changed (one or more were excluded due to limited available disk space).
- // We need to recompute the overlaps between sstables.
+ // We need to recompute the overlaps between sstables. The iterators used in the compaction controller
+ // and tracker will reflect the changed set of sstables made by LifecycleTransaction.cancel(),
+ // so refreshing the overlaps will be based on the updated set of sstables.
controller.refreshOverlaps();
}
// sanity check: all sstables must belong to the same cfs
- assert !Iterables.any(transaction.originals(), new Predicate()
+ assert !Iterables.any(actuallyCompact, new Predicate()
{
@Override
public boolean apply(SSTableReader sstable)
@@ -159,13 +204,15 @@ public class CompactionTask extends AbstractCompactionTask
// so in our single-threaded compaction world this is a valid way of determining if we're compacting
// all the sstables (that existed when we started)
StringBuilder ssTableLoggerMsg = new StringBuilder("[");
- for (SSTableReader sstr : transaction.originals())
+ for (SSTableReader sstr : actuallyCompact)
{
- ssTableLoggerMsg.append(String.format("%s:level=%d, ", sstr.getFilename(), sstr.getSSTableLevel()));
+ ssTableLoggerMsg.append(sstr.getSSTableLevel() != 0 ? String.format("%s:level=%d", sstr.getFilename(), sstr.getSSTableLevel())
+ : sstr.getFilename());
+ ssTableLoggerMsg.append(", ");
}
ssTableLoggerMsg.append("]");
- logger.info("Compacting ({}) {}", taskId, ssTableLoggerMsg);
+ logger.info("Compacting ({}) {}", transaction.opIdString(), ssTableLoggerMsg);
RateLimiter limiter = CompactionManager.instance.getRateLimiter();
long start = nanoTime();
@@ -175,15 +222,17 @@ public class CompactionTask extends AbstractCompactionTask
long inputSizeBytes;
long timeSpentWritingKeys;
- Set actuallyCompact = Sets.difference(transaction.originals(), fullyExpiredSSTables);
Collection newSStables;
long[] mergedRowCounts;
long totalSourceCQLRows;
+ Range tokenRange = tokenRange();
+ List> rangeList = tokenRange != null ? ImmutableList.of(tokenRange) : null;
+
long nowInSec = FBUtilities.nowInSeconds();
try (Refs refs = Refs.ref(actuallyCompact);
- AbstractCompactionStrategy.ScannerList scanners = strategy.getScanners(actuallyCompact);
+ AbstractCompactionStrategy.ScannerList scanners = strategy.getScanners(actuallyCompact, rangeList);
CompactionIterator ci = new CompactionIterator(compactionType, scanners.scanners, controller, nowInSec, taskId))
{
long lastCheckObsoletion = start;
@@ -257,37 +306,36 @@ public class CompactionTask extends AbstractCompactionTask
ImmutableMap.of(COMPACTION_TYPE_PROPERTY, compactionType.type));
logger.info(String.format("Compacted (%s) %d sstables to [%s] to level=%d. %s to %s (~%d%% of original) in %,dms. Read Throughput = %s, Write Throughput = %s, Row Throughput = ~%,d/s. %,d total partitions merged to %,d. Partition merge counts were {%s}. Time spent writing keys = %,dms",
- taskId,
- transaction.originals().size(),
- newSSTableNames.toString(),
- getLevel(),
- FBUtilities.prettyPrintMemory(startsize),
- FBUtilities.prettyPrintMemory(endsize),
- (int) (ratio * 100),
- dTime,
- FBUtilities.prettyPrintMemoryPerSecond(startsize, durationInNano),
- FBUtilities.prettyPrintMemoryPerSecond(endsize, durationInNano),
- (int) totalSourceCQLRows / (TimeUnit.NANOSECONDS.toSeconds(durationInNano) + 1),
- totalSourceRows,
- totalKeysWritten,
- mergeSummary,
- timeSpentWritingKeys));
+ transaction.opIdString(),
+ actuallyCompact.size(),
+ newSSTableNames.toString(),
+ getLevel(),
+ FBUtilities.prettyPrintMemory(startsize),
+ FBUtilities.prettyPrintMemory(endsize),
+ (int) (ratio * 100),
+ dTime,
+ FBUtilities.prettyPrintMemoryPerSecond(startsize, durationInNano),
+ FBUtilities.prettyPrintMemoryPerSecond(endsize, durationInNano),
+ (int) totalSourceCQLRows / (TimeUnit.NANOSECONDS.toSeconds(durationInNano) + 1),
+ totalSourceRows,
+ totalKeysWritten,
+ mergeSummary,
+ timeSpentWritingKeys));
if (logger.isTraceEnabled())
{
logger.trace("CF Total Bytes Compacted: {}", FBUtilities.prettyPrintMemory(CompactionTask.addToTotalBytesCompacted(endsize)));
logger.trace("Actual #keys: {}, Estimated #keys:{}, Err%: {}", totalKeysWritten, estimatedKeys, ((double)(totalKeysWritten - estimatedKeys)/totalKeysWritten));
}
- cfs.getCompactionStrategyManager().compactionLogger.compaction(startTime, transaction.originals(), currentTimeMillis(), newSStables);
+ cfs.getCompactionStrategyManager().compactionLogger.compaction(startTime, inputSSTables(), currentTimeMillis(), newSStables);
// update the metrics
cfs.metric.compactionBytesWritten.inc(endsize);
}
}
- @Override
public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction transaction,
+ ILifecycleTransaction transaction,
Set nonExpiredSSTables)
{
return new DefaultCompactionWriter(cfs, directories, transaction, nonExpiredSSTables, keepOriginals, getLevel());
@@ -366,7 +414,7 @@ public class CompactionTask extends AbstractCompactionTask
*
* @return true if there is enough disk space to execute the complete compaction, false if some sstables are excluded.
*/
- protected boolean buildCompactionCandidatesForAvailableDiskSpace(final Set fullyExpiredSSTables, TimeUUID taskId)
+ protected boolean buildCompactionCandidatesForAvailableDiskSpace(final Set nonExpiredSSTables, boolean containsExpired, TimeUUID taskId)
{
if(!cfs.isCompactionDiskSpaceCheckEnabled() && compactionType == OperationType.COMPACTION)
{
@@ -374,7 +422,6 @@ public class CompactionTask extends AbstractCompactionTask
return true;
}
- final Set nonExpiredSSTables = Sets.difference(transaction.originals(), fullyExpiredSSTables);
CompactionStrategyManager strategy = cfs.getCompactionStrategyManager();
int sstablesRemoved = 0;
@@ -408,11 +455,13 @@ public class CompactionTask extends AbstractCompactionTask
// we end up here if we can't take any more sstables out of the compaction.
// usually means we've run out of disk space
- // but we can still compact expired SSTables
- if(partialCompactionsAcceptable() && fullyExpiredSSTables.size() > 0 )
+ // but we can still remove expired SSTables
+ if (partialCompactionsAcceptable() && containsExpired)
{
- // sanity check to make sure we compact only fully expired SSTables.
- assert transaction.originals().equals(fullyExpiredSSTables);
+ for (SSTableReader rdr : nonExpiredSSTables)
+ transaction.cancel(rdr);
+ nonExpiredSSTables.clear();
+ assert transaction.originals().size() > 0;
break;
}
diff --git a/src/java/org/apache/cassandra/db/compaction/CompositeCompactionTask.java b/src/java/org/apache/cassandra/db/compaction/CompositeCompactionTask.java
new file mode 100644
index 0000000000..7ff39fc749
--- /dev/null
+++ b/src/java/org/apache/cassandra/db/compaction/CompositeCompactionTask.java
@@ -0,0 +1,124 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction;
+
+import java.util.ArrayList;
+import java.util.Collections;
+import java.util.List;
+
+import com.google.common.annotations.VisibleForTesting;
+
+import org.apache.cassandra.utils.Throwables;
+
+/// A composition of several compaction tasks into one. This object executes the given tasks sequentially and
+/// is used to limit the parallelism of some compaction tasks that split into a large number of parallelizable ones
+/// but should not be allowed to take all compaction executor threads.
+public class CompositeCompactionTask extends AbstractCompactionTask
+{
+ @VisibleForTesting
+ final ArrayList tasks;
+
+ public CompositeCompactionTask(AbstractCompactionTask first)
+ {
+ super(first.cfs, first.cfs.getTracker().tryModify(Collections.emptyList(), OperationType.COMPACTION));
+ tasks = new ArrayList<>();
+ addTask(first);
+ }
+
+ /// Add a task to the composition.
+ public CompositeCompactionTask addTask(AbstractCompactionTask task)
+ {
+ tasks.add(task);
+ return this;
+ }
+
+ @Override
+ protected void executeInternal(ActiveCompactionsTracker tracker)
+ {
+ // Run all tasks in sequence, regardless if any of them fail.
+ Throwables.perform(tasks.stream().map(x -> () -> x.execute(tracker)));
+ }
+
+ @Override
+ protected void runMayThrow() throws Exception {
+ throw new IllegalStateException("CompositeCompactionTask should be run through execute");
+ }
+
+ @Override
+ public void rejected()
+ {
+ Throwables.perform(tasks.stream().map(t -> t::rejected), () -> super.rejected());
+ }
+
+ @Override
+ public AbstractCompactionTask setUserDefined(boolean isUserDefined)
+ {
+ for (AbstractCompactionTask task : tasks)
+ task.setUserDefined(isUserDefined);
+ return super.setUserDefined(isUserDefined);
+ }
+
+ @Override
+ public AbstractCompactionTask setCompactionType(OperationType compactionType)
+ {
+ for (AbstractCompactionTask task : tasks)
+ task.setCompactionType(compactionType);
+ return super.setCompactionType(compactionType);
+ }
+
+ @Override
+ public String toString()
+ {
+ return "Composite " + tasks;
+ }
+
+ /// Limit the parallelism of a list of compaction tasks by combining them into a smaller number of composite tasks.
+ /// This method assumes that the caller has preference for the tasks to be executed in order close to the order of
+ /// the input list. See [UnifiedCompactionStrategy#getMaximalTasks] for an example of how to use this method.
+ public static List applyParallelismLimit(List tasks, int parallelismLimit)
+ {
+ if (tasks.size() <= parallelismLimit || parallelismLimit <= 0)
+ return tasks;
+
+ List result = new ArrayList<>(parallelismLimit);
+ int taskIndex = 0;
+ for (AbstractCompactionTask task : tasks) {
+ if (result.size() < parallelismLimit)
+ result.add(task);
+ else {
+ result.set(taskIndex, combineTasks(result.get(taskIndex), task));
+ if (++taskIndex == parallelismLimit)
+ taskIndex = 0;
+ }
+ }
+ return result;
+ }
+
+ /// Make a composite tasks that combines two tasks. If the former is already a composite task, the latter is added
+ /// to it. Otherwise, a new composite task is created.
+ public static CompositeCompactionTask combineTasks(AbstractCompactionTask task1, AbstractCompactionTask task2)
+ {
+ CompositeCompactionTask composite;
+ if (task1 instanceof CompositeCompactionTask)
+ composite = (CompositeCompactionTask) task1;
+ else
+ composite = new CompositeCompactionTask(task1);
+ return composite.addTask(task2);
+ }
+}
diff --git a/src/java/org/apache/cassandra/db/compaction/LeveledCompactionStrategy.java b/src/java/org/apache/cassandra/db/compaction/LeveledCompactionStrategy.java
index 58166630bc..ff90219224 100644
--- a/src/java/org/apache/cassandra/db/compaction/LeveledCompactionStrategy.java
+++ b/src/java/org/apache/cassandra/db/compaction/LeveledCompactionStrategy.java
@@ -126,7 +126,7 @@ public class LeveledCompactionStrategy extends AbstractCompactionStrategy
* the only difference between background and maximal in LCS is that maximal is still allowed
* (by explicit user request) even when compaction is disabled.
*/
- public AbstractCompactionTask getNextBackgroundTask(long gcBefore)
+ public Collection getNextBackgroundTasks(long gcBefore)
{
Collection previousCandidate = null;
while (true)
@@ -140,7 +140,7 @@ public class LeveledCompactionStrategy extends AbstractCompactionStrategy
if (sstable == null)
{
logger.trace("No compaction necessary for {}", this);
- return null;
+ return Collections.emptyList();
}
candidate = new LeveledManifest.CompactionCandidate(Collections.singleton(sstable),
sstable.getSSTableLevel(),
@@ -159,7 +159,7 @@ public class LeveledCompactionStrategy extends AbstractCompactionStrategy
logger.warn("Could not acquire references for compacting SSTables {} which is not a problem per se," +
"unless it happens frequently, in which case it must be reported. Will retry later.",
candidate.sstables);
- return null;
+ return Collections.emptyList();
}
LifecycleTransaction txn = cfs.getTracker().tryModify(candidate.sstables, OperationType.COMPACTION);
@@ -172,13 +172,13 @@ public class LeveledCompactionStrategy extends AbstractCompactionStrategy
newTask = new SingleSSTableLCSTask(cfs, txn, candidate.level);
newTask.setCompactionType(op);
- return newTask;
+ return Collections.singletonList(newTask);
}
previousCandidate = candidate.sstables;
}
}
- public synchronized Collection getMaximalTask(long gcBefore, boolean splitOutput)
+ public synchronized List getMaximalTasks(long gcBefore, boolean splitOutput)
{
Iterable sstables = manifest.getSSTables();
diff --git a/src/java/org/apache/cassandra/db/compaction/LeveledCompactionTask.java b/src/java/org/apache/cassandra/db/compaction/LeveledCompactionTask.java
index 8f5a70a84d..d5867410f1 100644
--- a/src/java/org/apache/cassandra/db/compaction/LeveledCompactionTask.java
+++ b/src/java/org/apache/cassandra/db/compaction/LeveledCompactionTask.java
@@ -26,7 +26,7 @@ import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.compaction.writers.MajorLeveledCompactionWriter;
import org.apache.cassandra.db.compaction.writers.MaxSSTableSizeWriter;
import org.apache.cassandra.io.sstable.format.SSTableReader;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
public class LeveledCompactionTask extends CompactionTask
{
@@ -34,7 +34,7 @@ public class LeveledCompactionTask extends CompactionTask
private final long maxSSTableBytes;
private final boolean majorCompaction;
- public LeveledCompactionTask(ColumnFamilyStore cfs, LifecycleTransaction txn, int level, long gcBefore, long maxSSTableBytes, boolean majorCompaction)
+ public LeveledCompactionTask(ColumnFamilyStore cfs, ILifecycleTransaction txn, int level, long gcBefore, long maxSSTableBytes, boolean majorCompaction)
{
super(cfs, txn, gcBefore);
this.level = level;
@@ -45,7 +45,7 @@ public class LeveledCompactionTask extends CompactionTask
@Override
public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables)
{
if (majorCompaction)
@@ -67,7 +67,7 @@ public class LeveledCompactionTask extends CompactionTask
@Override
public boolean reduceScopeForLimitedSpace(Set nonExpiredSSTables, long expectedSize)
{
- if (transaction.originals().size() > 1 && level <= 1)
+ if (nonExpiredSSTables.size() > 1 && level <= 1)
{
// Try again w/o the largest one.
logger.warn("insufficient space to do L0 -> L{} compaction. {}MiB required, {} for compaction {}",
@@ -100,6 +100,7 @@ public class LeveledCompactionTask extends CompactionTask
largestL0SSTable.onDiskLength(),
transaction.opId());
transaction.cancel(largestL0SSTable);
+ nonExpiredSSTables.remove(largestL0SSTable);
return true;
}
}
diff --git a/src/java/org/apache/cassandra/db/compaction/PendingRepairHolder.java b/src/java/org/apache/cassandra/db/compaction/PendingRepairHolder.java
index 0c5d53c1d8..1f2ff4a1a1 100644
--- a/src/java/org/apache/cassandra/db/compaction/PendingRepairHolder.java
+++ b/src/java/org/apache/cassandra/db/compaction/PendingRepairHolder.java
@@ -116,7 +116,7 @@ public class PendingRepairHolder extends AbstractStrategyHolder
{
List suppliers = new ArrayList<>(managers.size());
for (PendingRepairManager manager : managers)
- suppliers.add(new TaskSupplier(manager.getMaxEstimatedRemainingTasks(), () -> manager.getNextBackgroundTask(gcBefore)));
+ suppliers.add(new TaskSupplier(manager.getMaxEstimatedRemainingTasks(), () -> manager.getNextBackgroundTasks(gcBefore)));
return suppliers;
}
@@ -156,7 +156,7 @@ public class PendingRepairHolder extends AbstractStrategyHolder
managers.get(router.getIndexForSSTable(sstable)).addSSTable(sstable);
}
- AbstractCompactionTask getNextRepairFinishedTask()
+ Collection getNextRepairFinishedTasks()
{
List repairFinishedSuppliers = getRepairFinishedTaskSuppliers();
if (!repairFinishedSuppliers.isEmpty())
@@ -164,9 +164,9 @@ public class PendingRepairHolder extends AbstractStrategyHolder
Collections.sort(repairFinishedSuppliers);
for (TaskSupplier supplier : repairFinishedSuppliers)
{
- AbstractCompactionTask task = supplier.getTask();
- if (task != null)
- return task;
+ Collection tasks = supplier.getTasks();
+ if (tasks != null && !tasks.isEmpty())
+ return tasks;
}
}
return null;
@@ -180,7 +180,7 @@ public class PendingRepairHolder extends AbstractStrategyHolder
int numPending = manager.getNumPendingRepairFinishedTasks();
if (numPending > 0)
{
- suppliers.add(new TaskSupplier(numPending, manager::getNextRepairFinishedTask));
+ suppliers.add(new TaskSupplier(numPending, manager::getNextRepairFinishedTasks));
}
}
diff --git a/src/java/org/apache/cassandra/db/compaction/PendingRepairManager.java b/src/java/org/apache/cassandra/db/compaction/PendingRepairManager.java
index 7251c04dcf..5adb829970 100644
--- a/src/java/org/apache/cassandra/db/compaction/PendingRepairManager.java
+++ b/src/java/org/apache/cassandra/db/compaction/PendingRepairManager.java
@@ -33,13 +33,10 @@ import com.google.common.base.Preconditions;
import com.google.common.collect.ImmutableMap;
import com.google.common.collect.Iterables;
import com.google.common.collect.Maps;
-
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
import org.apache.cassandra.db.ColumnFamilyStore;
-import org.apache.cassandra.db.Directories;
-import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
import org.apache.cassandra.dht.Range;
import org.apache.cassandra.dht.Token;
@@ -352,22 +349,25 @@ class PendingRepairManager
return count;
}
- synchronized AbstractCompactionTask getNextRepairFinishedTask()
+ synchronized Collection getNextRepairFinishedTasks()
{
+ List tasks = new ArrayList<>();
for (TimeUUID sessionID : strategies.keySet())
{
if (canCleanup(sessionID))
{
- return getRepairFinishedCompactionTask(sessionID);
+ RepairFinishedCompactionTask repairFinishedTask = getRepairFinishedCompactionTask(sessionID);
+ if (repairFinishedTask != null)
+ tasks.add(repairFinishedTask);
}
}
- return null;
+ return tasks;
}
- synchronized AbstractCompactionTask getNextBackgroundTask(long gcBefore)
+ synchronized Collection getNextBackgroundTasks(long gcBefore)
{
if (strategies.isEmpty())
- return null;
+ return Collections.emptyList();
Map numTasks = new HashMap<>(strategies.size());
ArrayList sessions = new ArrayList<>(strategies.size());
@@ -382,13 +382,13 @@ class PendingRepairManager
}
if (sessions.isEmpty())
- return null;
+ return Collections.emptyList();
// we want the session with the most compactions at the head of the list
sessions.sort((o1, o2) -> numTasks.get(o2) - numTasks.get(o1));
TimeUUID sessionID = sessions.get(0);
- return get(sessionID).getNextBackgroundTask(gcBefore);
+ return get(sessionID).getNextBackgroundTasks(gcBefore);
}
synchronized Collection getMaximalTasks(long gcBefore, boolean splitOutput)
@@ -405,7 +405,7 @@ class PendingRepairManager
}
else
{
- Collection tasks = entry.getValue().getMaximalTask(gcBefore, splitOutput);
+ Collection tasks = entry.getValue().getMaximalTasks(gcBefore, splitOutput);
if (tasks != null)
maximalTasks.addAll(tasks);
}
@@ -536,7 +536,8 @@ class PendingRepairManager
{
if (obsoleteSSTables)
{
- transaction.finish();
+ transaction.prepareToCommit();
+ transaction.commit();
}
else
{
@@ -552,15 +553,9 @@ class PendingRepairManager
}
}
- public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables)
- {
- throw new UnsupportedOperationException();
- }
-
- protected int executeInternal(ActiveCompactionsTracker activeCompactions)
+ protected void executeInternal(ActiveCompactionsTracker activeCompactions)
{
run();
- return transaction.originals().size();
}
}
diff --git a/src/java/org/apache/cassandra/db/compaction/SSTableSplitter.java b/src/java/org/apache/cassandra/db/compaction/SSTableSplitter.java
index 6f68c340f0..f35e1659be 100644
--- a/src/java/org/apache/cassandra/db/compaction/SSTableSplitter.java
+++ b/src/java/org/apache/cassandra/db/compaction/SSTableSplitter.java
@@ -24,13 +24,13 @@ import org.apache.cassandra.db.*;
import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.compaction.writers.MaxSSTableSizeWriter;
import org.apache.cassandra.io.sstable.format.SSTableReader;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
public class SSTableSplitter
{
private final SplittingCompactionTask task;
- public SSTableSplitter(ColumnFamilyStore cfs, LifecycleTransaction transaction, int sstableSizeInMB)
+ public SSTableSplitter(ColumnFamilyStore cfs, ILifecycleTransaction transaction, int sstableSizeInMB)
{
this.task = new SplittingCompactionTask(cfs, transaction, sstableSizeInMB);
}
@@ -44,7 +44,7 @@ public class SSTableSplitter
{
private final int sstableSizeInMiB;
- public SplittingCompactionTask(ColumnFamilyStore cfs, LifecycleTransaction transaction, int sstableSizeInMB)
+ public SplittingCompactionTask(ColumnFamilyStore cfs, ILifecycleTransaction transaction, int sstableSizeInMB)
{
super(cfs, transaction, CompactionManager.NO_GC, false);
this.sstableSizeInMiB = sstableSizeInMB;
@@ -62,7 +62,7 @@ public class SSTableSplitter
@Override
public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables)
{
return new MaxSSTableSizeWriter(cfs, directories, txn, nonExpiredSSTables, sstableSizeInMiB * 1024L * 1024L, 0, false);
diff --git a/src/java/org/apache/cassandra/db/compaction/ShardManager.java b/src/java/org/apache/cassandra/db/compaction/ShardManager.java
index 6ea2cd72a8..7a5c65b228 100644
--- a/src/java/org/apache/cassandra/db/compaction/ShardManager.java
+++ b/src/java/org/apache/cassandra/db/compaction/ShardManager.java
@@ -18,7 +18,8 @@
package org.apache.cassandra.db.compaction;
-import java.util.Set;
+import java.util.*;
+import java.util.function.BiFunction;
import java.util.stream.Collectors;
import com.google.common.collect.ImmutableList;
@@ -144,7 +145,7 @@ public interface ShardManager
/**
* Estimate the density of the sstable that will be the result of compacting the given sources.
*/
- default double calculateCombinedDensity(Set extends SSTableReader> sstables)
+ default double calculateCombinedDensity(Collection sstables)
{
if (sstables.isEmpty())
return 0;
@@ -163,4 +164,43 @@ public interface ShardManager
else
return onDiskLength;
}
+
+ /**
+ * Seggregate the given sstables into the shard ranges that intersect sstables from the collection, and call
+ * the given function on the combination of each shard range and the intersecting sstable set.
+ */
+ default List splitSSTablesInShards(Collection sstables,
+ int numShardsForDensity,
+ BiFunction, Range, T> maker)
+ {
+ ShardTracker boundaries = boundaries(numShardsForDensity);
+ List tasks = new ArrayList<>();
+ SSTableReader[] items = sstables.toArray(SSTableReader[]::new);
+ Arrays.sort(items, SSTableReader.firstKeyComparator);
+ PriorityQueue active = new PriorityQueue<>(SSTableReader.lastKeyComparator);
+ int i = 0;
+ while (i < items.length || !active.isEmpty())
+ {
+ if (active.isEmpty())
+ {
+ boundaries.advanceTo(items[i].getFirst().getToken());
+ active.add(items[i++]);
+ }
+ Token shardEnd = boundaries.shardEnd();
+
+ while (i < items.length && (shardEnd == null || items[i].getFirst().getToken().compareTo(shardEnd) <= 0))
+ active.add(items[i++]);
+
+ final T result = maker.apply(active, boundaries.shardSpan());
+ if (result != null)
+ tasks.add(result);
+
+ while (!active.isEmpty() && (shardEnd == null || active.peek().getLast().getToken().compareTo(shardEnd) <= 0))
+ active.poll();
+
+ if (!active.isEmpty()) // shardEnd must be non-null (otherwise the line above exhausts all)
+ boundaries.advanceTo(shardEnd.nextValidToken());
+ }
+ return tasks;
+ }
}
diff --git a/src/java/org/apache/cassandra/db/compaction/ShardManagerDiskAware.java b/src/java/org/apache/cassandra/db/compaction/ShardManagerDiskAware.java
index 4f8aba283a..41c5d67002 100644
--- a/src/java/org/apache/cassandra/db/compaction/ShardManagerDiskAware.java
+++ b/src/java/org/apache/cassandra/db/compaction/ShardManagerDiskAware.java
@@ -204,7 +204,7 @@ public class ShardManagerDiskAware extends ShardManagerNoDisks
public int count()
{
- return countPerDisk;
+ return countPerDisk * diskBoundaryPositions.length;
}
/**
@@ -231,7 +231,7 @@ public class ShardManagerDiskAware extends ShardManagerNoDisks
public int shardIndex()
{
- return nextShardIndex - 1;
+ return diskIndex * countPerDisk + nextShardIndex - 1;
}
}
}
diff --git a/src/java/org/apache/cassandra/db/compaction/ShardManagerTrivial.java b/src/java/org/apache/cassandra/db/compaction/ShardManagerTrivial.java
index 407bff4f0d..064b5033b5 100644
--- a/src/java/org/apache/cassandra/db/compaction/ShardManagerTrivial.java
+++ b/src/java/org/apache/cassandra/db/compaction/ShardManagerTrivial.java
@@ -18,7 +18,10 @@
package org.apache.cassandra.db.compaction;
+import java.util.Collection;
+import java.util.List;
import java.util.Set;
+import java.util.function.BiFunction;
import org.apache.cassandra.db.PartitionPosition;
import org.apache.cassandra.dht.IPartitioner;
@@ -54,7 +57,7 @@ public class ShardManagerTrivial implements ShardManager
}
@Override
- public double calculateCombinedDensity(Set extends SSTableReader> sstables)
+ public double calculateCombinedDensity(Collection sstables)
{
double totalSize = 0;
for (SSTableReader sstable : sstables)
@@ -62,6 +65,14 @@ public class ShardManagerTrivial implements ShardManager
return totalSize;
}
+ @Override
+ public List splitSSTablesInShards(Collection sstables,
+ int numShardsForDensity,
+ BiFunction, Range, T> maker)
+ {
+ return List.of(maker.apply(sstables, new Range<>(partitioner.getMinimumToken(), partitioner.getMinimumToken())));
+ }
+
@Override
public double localSpaceCoverage()
{
diff --git a/src/java/org/apache/cassandra/db/compaction/SingleSSTableLCSTask.java b/src/java/org/apache/cassandra/db/compaction/SingleSSTableLCSTask.java
index 1f73c4cd30..2d9768924c 100644
--- a/src/java/org/apache/cassandra/db/compaction/SingleSSTableLCSTask.java
+++ b/src/java/org/apache/cassandra/db/compaction/SingleSSTableLCSTask.java
@@ -18,19 +18,14 @@
package org.apache.cassandra.db.compaction;
-import java.util.Set;
-
-import org.slf4j.Logger;
-import org.slf4j.LoggerFactory;
-
import org.apache.cassandra.db.ColumnFamilyStore;
-import org.apache.cassandra.db.Directories;
-import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
import org.apache.cassandra.io.sstable.CorruptSSTableException;
import org.apache.cassandra.io.sstable.format.SSTableFormat.Components;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.sstable.metadata.StatsMetadata;
+import org.slf4j.Logger;
+import org.slf4j.LoggerFactory;
/**
* Special compaction task that does not do any compaction, instead it
@@ -51,16 +46,9 @@ public class SingleSSTableLCSTask extends AbstractCompactionTask
}
@Override
- public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables)
- {
- throw new UnsupportedOperationException("This method should never be called on SingleSSTableLCSTask");
- }
-
- @Override
- protected int executeInternal(ActiveCompactionsTracker activeCompactions)
+ protected void executeInternal(ActiveCompactionsTracker activeCompactions)
{
run();
- return 1;
}
@Override
diff --git a/src/java/org/apache/cassandra/db/compaction/SizeTieredCompactionStrategy.java b/src/java/org/apache/cassandra/db/compaction/SizeTieredCompactionStrategy.java
index 74a96ca211..1ed39ea562 100644
--- a/src/java/org/apache/cassandra/db/compaction/SizeTieredCompactionStrategy.java
+++ b/src/java/org/apache/cassandra/db/compaction/SizeTieredCompactionStrategy.java
@@ -30,7 +30,7 @@ import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.compaction.writers.SplittingSizeTieredCompactionWriter;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.exceptions.ConfigurationException;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.schema.CompactionParams;
@@ -175,7 +175,7 @@ public class SizeTieredCompactionStrategy extends AbstractCompactionStrategy
return sstr.getReadMeter() == null ? 0.0 : sstr.getReadMeter().twoHourRate() / sstr.estimatedKeys();
}
- public AbstractCompactionTask getNextBackgroundTask(long gcBefore)
+ public Collection getNextBackgroundTasks(long gcBefore)
{
List previousCandidate = null;
while (true)
@@ -183,7 +183,7 @@ public class SizeTieredCompactionStrategy extends AbstractCompactionStrategy
List hottestBucket = getNextBackgroundSSTables(gcBefore);
if (hottestBucket.isEmpty())
- return null;
+ return Collections.emptyList();
// Already tried acquiring references without success. It means there is a race with
// the tracker but candidate SSTables were not yet replaced in the compaction strategy manager
@@ -192,22 +192,22 @@ public class SizeTieredCompactionStrategy extends AbstractCompactionStrategy
logger.warn("Could not acquire references for compacting SSTables {} which is not a problem per se," +
"unless it happens frequently, in which case it must be reported. Will retry later.",
hottestBucket);
- return null;
+ return Collections.emptyList();
}
- LifecycleTransaction transaction = cfs.getTracker().tryModify(hottestBucket, OperationType.COMPACTION);
+ ILifecycleTransaction transaction = cfs.getTracker().tryModify(hottestBucket, OperationType.COMPACTION);
if (transaction != null)
- return new CompactionTask(cfs, transaction, gcBefore);
+ return Collections.singletonList(new CompactionTask(cfs, transaction, gcBefore));
previousCandidate = hottestBucket;
}
}
- public synchronized Collection getMaximalTask(final long gcBefore, boolean splitOutput)
+ public synchronized List getMaximalTasks(final long gcBefore, boolean splitOutput)
{
Iterable filteredSSTables = filterSuspectSSTables(sstables);
if (Iterables.isEmpty(filteredSSTables))
return null;
- LifecycleTransaction txn = cfs.getTracker().tryModify(filteredSSTables, OperationType.COMPACTION);
+ ILifecycleTransaction txn = cfs.getTracker().tryModify(filteredSSTables, OperationType.COMPACTION);
if (txn == null)
return null;
if (splitOutput)
@@ -219,7 +219,7 @@ public class SizeTieredCompactionStrategy extends AbstractCompactionStrategy
{
assert !sstables.isEmpty(); // checked for by CM.submitUserDefined
- LifecycleTransaction transaction = cfs.getTracker().tryModify(sstables, OperationType.COMPACTION);
+ ILifecycleTransaction transaction = cfs.getTracker().tryModify(sstables, OperationType.COMPACTION);
if (transaction == null)
{
logger.trace("Unable to mark {} for compaction; probably a background compaction got to it first. You can disable background compactions temporarily if this is a problem", sstables);
@@ -347,7 +347,7 @@ public class SizeTieredCompactionStrategy extends AbstractCompactionStrategy
private static class SplittingCompactionTask extends CompactionTask
{
- public SplittingCompactionTask(ColumnFamilyStore cfs, LifecycleTransaction txn, long gcBefore)
+ public SplittingCompactionTask(ColumnFamilyStore cfs, ILifecycleTransaction txn, long gcBefore)
{
super(cfs, txn, gcBefore);
}
@@ -355,7 +355,7 @@ public class SizeTieredCompactionStrategy extends AbstractCompactionStrategy
@Override
public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables)
{
return new SplittingSizeTieredCompactionWriter(cfs, directories, txn, nonExpiredSSTables);
diff --git a/src/java/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategy.java b/src/java/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategy.java
index 2709d43ae5..ef4d78a542 100644
--- a/src/java/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategy.java
+++ b/src/java/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategy.java
@@ -80,7 +80,7 @@ public class TimeWindowCompactionStrategy extends AbstractCompactionStrategy
}
@Override
- public AbstractCompactionTask getNextBackgroundTask(long gcBefore)
+ public Collection getNextBackgroundTasks(long gcBefore)
{
List previousCandidate = null;
while (true)
@@ -88,7 +88,7 @@ public class TimeWindowCompactionStrategy extends AbstractCompactionStrategy
List latestBucket = getNextBackgroundSSTables(gcBefore);
if (latestBucket.isEmpty())
- return null;
+ return Collections.emptyList();
// Already tried acquiring references without success. It means there is a race with
// the tracker but candidate SSTables were not yet replaced in the compaction strategy manager
@@ -97,12 +97,12 @@ public class TimeWindowCompactionStrategy extends AbstractCompactionStrategy
logger.warn("Could not acquire references for compacting SSTables {} which is not a problem per se," +
"unless it happens frequently, in which case it must be reported. Will retry later.",
latestBucket);
- return null;
+ return Collections.emptyList();
}
LifecycleTransaction modifier = cfs.getTracker().tryModify(latestBucket, OperationType.COMPACTION);
if (modifier != null)
- return new TimeWindowCompactionTask(cfs, modifier, gcBefore, options.ignoreOverlaps);
+ return Collections.singletonList(new TimeWindowCompactionTask(cfs, modifier, gcBefore, options.ignoreOverlaps));
previousCandidate = latestBucket;
}
}
@@ -125,7 +125,7 @@ public class TimeWindowCompactionStrategy extends AbstractCompactionStrategy
if (currentTimeMillis() - lastExpiredCheck > options.expiredSSTableCheckFrequency)
{
logger.debug("TWCS expired check sufficiently far in the past, checking for fully expired SSTables");
- expired = CompactionController.getFullyExpiredSSTables(cfs, uncompacting, options.ignoreOverlaps ? Collections.emptySet() : cfs.getOverlappingLiveSSTables(uncompacting),
+ expired = CompactionController.getFullyExpiredSSTables(cfs, uncompacting, cfs::getOverlappingLiveSSTables,
gcBefore, options.ignoreOverlaps);
lastExpiredCheck = currentTimeMillis();
}
@@ -379,7 +379,7 @@ public class TimeWindowCompactionStrategy extends AbstractCompactionStrategy
}
@Override
- public synchronized Collection getMaximalTask(long gcBefore, boolean splitOutput)
+ public synchronized List getMaximalTasks(long gcBefore, boolean splitOutput)
{
Iterable filteredSSTables = filterSuspectSSTables(sstables);
if (Iterables.isEmpty(filteredSSTables))
@@ -387,7 +387,7 @@ public class TimeWindowCompactionStrategy extends AbstractCompactionStrategy
LifecycleTransaction txn = cfs.getTracker().tryModify(filteredSSTables, OperationType.COMPACTION);
if (txn == null)
return null;
- return Collections.singleton(new TimeWindowCompactionTask(cfs, txn, gcBefore, options.ignoreOverlaps));
+ return Collections.singletonList(new TimeWindowCompactionTask(cfs, txn, gcBefore, options.ignoreOverlaps));
}
/**
diff --git a/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.java b/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.java
index 089045f28a..c23e600e7b 100644
--- a/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.java
+++ b/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.java
@@ -31,6 +31,7 @@ import java.util.regex.Pattern;
import com.google.common.annotations.VisibleForTesting;
import com.google.common.base.Preconditions;
import com.google.common.base.Predicate;
+import com.google.common.collect.ImmutableList;
import com.google.common.collect.ImmutableSet;
import com.google.common.collect.Iterables;
import com.google.common.collect.Sets;
@@ -44,8 +45,10 @@ import org.apache.cassandra.db.commitlog.IntervalSet;
import org.apache.cassandra.db.compaction.unified.Controller;
import org.apache.cassandra.db.compaction.unified.ShardedMultiWriter;
import org.apache.cassandra.db.compaction.unified.UnifiedCompactionTask;
+import org.apache.cassandra.db.lifecycle.CompositeLifecycleTransaction;
import org.apache.cassandra.db.lifecycle.LifecycleNewTracker;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.PartialLifecycleTransaction;
import org.apache.cassandra.exceptions.ConfigurationException;
import org.apache.cassandra.index.Index;
import org.apache.cassandra.io.sstable.Descriptor;
@@ -139,15 +142,25 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
public static String printScalingParameter(int w)
{
if (w < 0)
- return "L" + Integer.toString(2 - w);
+ return 'L' + Integer.toString(2 - w);
else if (w > 0)
- return "T" + Integer.toString(w + 2);
+ return 'T' + Integer.toString(w + 2);
else
return "N";
}
+ private TimeUUID nextTimeUUID()
+ {
+ // Make a time-UUID with sequence 0. The reason to do this is to accommodate parallelized compactions:
+ // - Sequence 0 (visible as -8000- in the UUID string) denotes single-task (i.e. non-parallelized) compactions.
+ // - Sequence >0 (-800n-) denotes the individual task's index of a parallelized compaction.
+ // - Parallelized compactions use sequence 0 as the transaction id, and sequences from 1 to the number of tasks
+ // for the ids of individual tasks.
+ return TimeUUID.Generator.nextTimeUUID().withSequence(0);
+ }
+
@Override
- public synchronized Collection getMaximalTask(long gcBefore, boolean splitOutput)
+ public synchronized List getMaximalTasks(long gcBefore, boolean splitOutput)
{
maybeUpdateShardManager();
// The tasks are split by repair status and disk, as well as in non-overlapping sections to enable some
@@ -155,31 +168,55 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
// split across shards according to its density. Depending on the parallelism, the operation may require up to
// 100% extra space to complete.
List tasks = new ArrayList<>();
- List> nonOverlapping = splitInNonOverlappingSets(filterSuspectSSTables(getSSTables()));
- for (Set set : nonOverlapping)
- {
- LifecycleTransaction txn = cfs.getTracker().tryModify(set, OperationType.COMPACTION);
- if (txn != null)
- tasks.add(createCompactionTask(txn, gcBefore));
+
+ try {
+ List sstables = getSuitableSSTables();
+ if (sstables.isEmpty())
+ return Collections.emptyList();
+
+ // If possible, we want to issue separate compactions for non-overlapping sets of sstables, to allow
+ // for smaller extra space requirements. However, if the sharding configuration has changed, a major
+ // compaction should combine non-overlapping sets if they are split on a boundary that is no longer
+ // in effect.
+ final ShardManager shardManager = getShardManager();
+ List> groups =
+ shardManager.splitSSTablesInShards(sstables,
+ controller.getNumShards(shardManager.calculateCombinedDensity(sstables)),
+ (sstableShard, shardRange) -> Sets.newHashSet(sstableShard));
+
+ // Now combine all of these groups that share an sstable so that we have valid independent transactions.
+ groups = combineSetsWithCommonElement(groups);
+
+ for (Collection set : groups)
+ {
+ LifecycleTransaction txn = cfs.getTracker().tryModify(set, OperationType.COMPACTION, nextTimeUUID());
+ // The tasks may be further split by output shard to increase the parallelism.
+ if (txn != null)
+ tasks.addAll(createCompactionTasks(gcBefore, txn));
+ // we ignore splitOutput (always split according to the strategy's sharding) and do not need isMaximal
+ }
+
+ return tasks;
+ }
+ catch (Throwable t)
+ {
+ for (AbstractCompactionTask task : tasks)
+ task.rejected();
+ throw t;
}
- return tasks;
}
- private static List> splitInNonOverlappingSets(Collection sstables)
+ /**
+ * Transform a list to transitively combine adjacent sets that have a common element, resulting in disjoint sets.
+ */
+ private static List> combineSetsWithCommonElement(List extends Set> overlapSets)
{
- List> overlapSets = Overlaps.constructOverlapSets(new ArrayList<>(sstables),
- UnifiedCompactionStrategy::startsAfter,
- SSTableReader.firstKeyComparator,
- SSTableReader.lastKeyComparator);
- if (overlapSets.isEmpty())
- return overlapSets;
-
- Set group = overlapSets.get(0);
- List> groups = new ArrayList<>();
+ Set group = overlapSets.get(0);
+ List> groups = new ArrayList<>();
for (int i = 1; i < overlapSets.size(); ++i)
{
- Set current = overlapSets.get(i);
- if (Sets.intersection(current, group).isEmpty())
+ Set current = overlapSets.get(i);
+ if (Collections.disjoint(current, group))
{
groups.add(group);
group = current;
@@ -198,7 +235,7 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
{
assert !sstables.isEmpty(); // checked for by CM.submitUserDefined
- LifecycleTransaction transaction = cfs.getTracker().tryModify(sstables, OperationType.COMPACTION);
+ LifecycleTransaction transaction = cfs.getTracker().tryModify(sstables, OperationType.COMPACTION, nextTimeUUID());
if (transaction == null)
{
logger.trace("Unable to mark {} for compaction; probably a background compaction got to it first. You can disable background compactions temporarily if this is a problem", sstables);
@@ -209,37 +246,37 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
}
/**
- * Returns a compaction task to run next.
- *
- * This method is synchronized because task creation is significantly more expensive in UCS; the strategy is
+ * Returns a collections of compaction tasks.
+ *
+ * This method is synchornized because task creation is significantly more expensive in UCS; the strategy is
* stateless, therefore it has to compute the shard/bucket structure on each call.
*
* @param gcBefore throw away tombstones older than this
+ * @return collection of AbstractCompactionTask, which could be either a CompactionTask or an UnifiedCompactionTask
*/
@Override
- public synchronized UnifiedCompactionTask getNextBackgroundTask(long gcBefore)
+ public synchronized Collection getNextBackgroundTasks(long gcBefore)
{
while (true)
{
CompactionPick pick = getNextCompactionPick(gcBefore);
if (pick == null)
- return null;
- UnifiedCompactionTask task = createCompactionTask(pick, gcBefore);
- if (task != null)
- return task;
+ return Collections.emptyList();
+ Collection tasks = createCompactionTasks(pick, gcBefore);
+ if (tasks != null)
+ return tasks;
}
}
- private UnifiedCompactionTask createCompactionTask(CompactionPick pick, long gcBefore)
+ private Collection createCompactionTasks(CompactionPick pick, long gcBefore)
{
Preconditions.checkNotNull(pick);
Preconditions.checkArgument(!pick.isEmpty());
- LifecycleTransaction transaction = cfs.getTracker().tryModify(pick,
- OperationType.COMPACTION);
+ LifecycleTransaction transaction = cfs.getTracker().tryModify(pick, OperationType.COMPACTION, nextTimeUUID());
if (transaction != null)
{
- return createCompactionTask(transaction, gcBefore);
+ return createCompactionTasks(gcBefore, transaction);
}
else
{
@@ -284,6 +321,15 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
boundaries);
}
+ @VisibleForTesting
+ List createCompactionTasks(long gcBefore, LifecycleTransaction transaction)
+ {
+ if (controller.parallelizeOutputShards())
+ return createParallelCompactionTasks(transaction, gcBefore);
+ else
+ return ImmutableList.of(createCompactionTask(transaction, gcBefore));
+ }
+
/**
* Create the task that in turns creates the sstable writer used for compaction.
*
@@ -294,6 +340,42 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
return new UnifiedCompactionTask(cfs, this, transaction, gcBefore, getShardManager());
}
+ private List createParallelCompactionTasks(LifecycleTransaction transaction, long gcBefore)
+ {
+ Collection sstables = transaction.originals();
+ ShardManager shardManager = getShardManager();
+ CompositeLifecycleTransaction compositeTransaction = new CompositeLifecycleTransaction(transaction);
+
+ double density = shardManager.calculateCombinedDensity(sstables);
+ int numShards = controller.getNumShards(density * shardManager.shardSetCoverage());
+ if (numShards <= 1)
+ return Collections.singletonList(createCompactionTask(transaction, gcBefore));
+ List tasks = shardManager.splitSSTablesInShards(
+ sstables,
+ numShards,
+ (rangeSSTables, range) ->
+ new UnifiedCompactionTask(cfs,
+ this,
+ new PartialLifecycleTransaction(compositeTransaction),
+ gcBefore,
+ shardManager,
+ range,
+ rangeSSTables)
+ );
+ compositeTransaction.completeInitialization();
+
+ if (tasks.isEmpty())
+ transaction.close(); // this should not be reachable normally, close the transaction for safety
+
+ if (tasks.size() == 1) // if there's just one range, make it a non-ranged task (to apply early open etc.)
+ {
+ assert ((CompactionTask) tasks.get(0)).inputSSTables().equals(sstables);
+ return Collections.singletonList(createCompactionTask(transaction, gcBefore));
+ }
+ else
+ return tasks;
+ }
+
private void maybeUpdateShardManager()
{
// TODO - modify ShardManager::isOutOfDate to take an Epoch
@@ -358,7 +440,7 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
lastExpiredCheck = ts;
expired = CompactionController.getFullyExpiredSSTables(cfs,
suitable,
- cfs.getOverlappingLiveSSTables(suitable),
+ cfs::getOverlappingLiveSSTables,
gcBefore,
controller.getIgnoreOverlapsInExpirationCheck());
if (logger.isTraceEnabled() && !expired.isEmpty())
@@ -452,7 +534,7 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
* new compactions, and by external tools to analyze the strategy decisions.
*
* @param sstables a collection of the sstables to be assigned to levels
- * @param compactionFilter a filter to exclude CompactionSSTables,
+ * @param compactionFilter a filter to exclude SSTableReaders,
* e.g., {@link #isSuitableForCompaction}
*
* @return a list of the levels in the compaction hierarchy
@@ -512,6 +594,11 @@ public class UnifiedCompactionStrategy extends AbstractCompactionStrategy
return levels;
}
+ List getSuitableSSTables()
+ {
+ return getCompactableSSTables(getSSTables(), UnifiedCompactionStrategy::isSuitableForCompaction);
+ }
+
private List getCompactableSSTables(Collection sstables,
Predicate compactionFilter)
{
diff --git a/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.md b/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.md
index 5f8d548af9..23d309ea46 100644
--- a/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.md
+++ b/src/java/org/apache/cassandra/db/compaction/UnifiedCompactionStrategy.md
@@ -218,7 +218,7 @@ This sharding mechanism is independent of the compaction specification.
This sharding scheme easily admits extensions. In particular, when the size of the data set is expected to grow very
large, to avoid having to pre-specify a high enough target size to avoid problems with per-sstable overhead, we can
-apply an "SSTtable growth" parameter, which determines what part of the density growth should be assigned to increased
+apply an "SSTable growth" parameter, which determines what part of the density growth should be assigned to increased
SSTable size, reducing the growth of the number of shards (and hence non-overlapping sstables).
Additionally, to allow for a mode of operation with a fixed number of shards, and splitting conditional on reaching
@@ -332,14 +332,38 @@ with legacy strategies (e.g. all resources consumed by L0 and sstables accumulat
steady state where compactions always use more sstables than the assigned threshold and fan factor and maintain a tiered
hierarchy based on the lowest overlap they are able to maintain for the load.
+## Output shard parallelization
+
+Because the sharding of the output of a compaction operation is known in advance, we can parallelize the compaction
+process by starting a separate task for each shard. This can dramatically speed the throughput of compaction and is
+especially helpful for the lower levels of the compaction heirarchy, where the number of input shards is very low
+(often just one). To make sure that we correctly change the state of input and output sstables, such operations will
+share a transaction and will complete only when all individual tasks complete (and, conversely, abort if any of the
+individual tasks abort). Early opening of sstables is not supported in this mode, because we currently do not support
+arbitraty filtering of the requests to an sstable; it is expected that the smaller size and quicker completion time of
+compactions should make up for this.
+
+This is controlled by the `parallelize_output_shards` parameter, which is `true` by default.
+
## Major compaction
-Under the working principles of UCS, a major compaction is an operation which compacts together all sstables that have
-(transitive) overlap, and where the output is split on shard boundaries appropriate for the expected result density.
+Major compaction in UCS always splits the output into a shard number suitable for the expected result density.
+If the input sstables can be split into non-overlapping sets that correspond to current shard boundaries, the compaction
+will construct independent operations that work over these sets, to improve the space overhead of the operation as well
+as the time needed to persistently complete individual steps. Because all levels will usually be split in $b$ shards,
+it will very often be the case that major compactions split into $b$ individual jobs, reducing the space overhead by a
+factor close to $b$. Note that this does not always apply; for example, if a topology change causes the sharding
+boundaries to move, the mismatch between old and new sharding boundaries will cause the compaction to produce a single
+operation and require 100% space overhead.
-In other words, it is expected that a major compaction will result in $b$ concurrent compactions, each containing all
-sstables covered in each of the base shards, and that the result will be split on shard boundaries whose number
-depends on the total size of data contained in the shard.
+Output shard parallelization also applies to major compactions: if the `parallelize_output_shards` option is enabled,
+shards of individual compactions will be compacted concurrently, which can significantly reduce the time needed to
+perform the compaction; if the option is not enabled, major compaction will only be parallelized up to the number of
+individual non-overlapping sets the sstables can be split into. In either case, the number of parallel operations is
+limited to a number specified as a parameter of the operation (e.g. `nodetool compact -j n`), which is set to half the
+compaction thread count by default. Using a jobs of 0 will let the compaction use all available threads and run
+as quickly as possible, but this will prevent other compaction operations from running until it completes and thus
+should be used with caution, only while the database is known to not receive any writes.
## Differences with STCS and LCS
@@ -441,6 +465,11 @@ UCS accepts these compaction strategy parameters:
that are considered too small. If set, the strategy will split the space into fewer than the base count shards, to
make the estimated sstables size at least as large as this value. A value of 0 disables this feature.
The default value is 100MiB.
+* **parallelize_output_shards**. Enables or disables parallelization of compaction tasks for the output shards of a
+ compaction. This can dramatically improve compaction throughput especially on the lowest levels of the hierarchy,
+ but disables early open and thus may be less efficient when compaction is configured to produce very large
+ sstables.
+ The default value is `true`.
* **expired_sstable_check_frequency_seconds**. Determines how often to check for expired SSTables.
The default value is 10 minutes.
diff --git a/src/java/org/apache/cassandra/db/compaction/unified/Controller.java b/src/java/org/apache/cassandra/db/compaction/unified/Controller.java
index 5faab27027..df2a66590c 100644
--- a/src/java/org/apache/cassandra/db/compaction/unified/Controller.java
+++ b/src/java/org/apache/cassandra/db/compaction/unified/Controller.java
@@ -31,6 +31,7 @@ import org.slf4j.LoggerFactory;
import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.config.CassandraRelevantProperties;
import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.compaction.CompactionManager;
import org.apache.cassandra.db.compaction.UnifiedCompactionStrategy;
import org.apache.cassandra.exceptions.ConfigurationException;
import org.apache.cassandra.utils.FBUtilities;
@@ -150,6 +151,14 @@ public class Controller
static final Overlaps.InclusionMethod DEFAULT_OVERLAP_INCLUSION_METHOD =
CassandraRelevantProperties.UCS_OVERLAP_INCLUSION_METHOD.getEnum(Overlaps.InclusionMethod.TRANSITIVE);
+ /**
+ * Whether to create subtask for the output shards of individual compactions and execute them in parallel.
+ * Defaults to true for improved parallelization and efficiency.
+ */
+ static final String PARALLELIZE_OUTPUT_SHARDS_OPTION = "parallelize_output_shards";
+ static final boolean DEFAULT_PARALLELIZE_OUTPUT_SHARDS =
+ CassandraRelevantProperties.UCS_PARALLELIZE_OUTPUT_SHARDS.getBoolean(true);
+
protected final ColumnFamilyStore cfs;
protected final MonotonicClock clock;
private final int[] scalingParameters;
@@ -172,6 +181,7 @@ public class Controller
private static final double INVERSE_LOG_2 = 1.0 / Math.log(2);
protected final Overlaps.InclusionMethod overlapInclusionMethod;
+ protected final boolean parallelizeOutputShards;
Controller(ColumnFamilyStore cfs,
MonotonicClock clock,
@@ -185,7 +195,8 @@ public class Controller
int baseShardCount,
double targetSStableSize,
double sstableGrowthModifier,
- Overlaps.InclusionMethod overlapInclusionMethod)
+ Overlaps.InclusionMethod overlapInclusionMethod,
+ boolean parallelizeOutputShards)
{
this.cfs = cfs;
this.clock = clock;
@@ -199,6 +210,7 @@ public class Controller
this.targetSSTableSize = targetSStableSize;
this.overlapInclusionMethod = overlapInclusionMethod;
this.sstableGrowthModifier = sstableGrowthModifier;
+ this.parallelizeOutputShards = parallelizeOutputShards;
if (maxSSTablesToCompact <= 0)
maxSSTablesToCompact = Integer.MAX_VALUE;
@@ -356,6 +368,11 @@ public class Controller
}
}
+ public boolean parallelizeOutputShards()
+ {
+ return parallelizeOutputShards;
+ }
+
/**
* @return the survival factor o
* @param index
@@ -446,6 +463,10 @@ public class Controller
? Overlaps.InclusionMethod.valueOf(toUpperCaseLocalized(options.get(OVERLAP_INCLUSION_METHOD_OPTION)))
: DEFAULT_OVERLAP_INCLUSION_METHOD;
+ boolean parallelizeOutputShards = options.containsKey(PARALLELIZE_OUTPUT_SHARDS_OPTION)
+ ? Boolean.parseBoolean(options.get(PARALLELIZE_OUTPUT_SHARDS_OPTION))
+ : DEFAULT_PARALLELIZE_OUTPUT_SHARDS;
+
return new Controller(cfs,
MonotonicClock.Global.preciseTime,
Ws,
@@ -458,7 +479,8 @@ public class Controller
baseShardCount,
targetSStableSize,
sstableGrowthModifier,
- inclusionMethod);
+ inclusionMethod,
+ parallelizeOutputShards);
}
public static Map validateOptions(Map options) throws ConfigurationException
@@ -572,12 +594,8 @@ public class Controller
}
}
- s = options.remove(ALLOW_UNSAFE_AGGRESSIVE_SSTABLE_EXPIRATION_OPTION);
- if (s != null && !s.equalsIgnoreCase("true") && !s.equalsIgnoreCase("false"))
- {
- throw new ConfigurationException(String.format("%s should either be 'true' or 'false', not %s",
- ALLOW_UNSAFE_AGGRESSIVE_SSTABLE_EXPIRATION_OPTION, s));
- }
+ validateBoolean(options, ALLOW_UNSAFE_AGGRESSIVE_SSTABLE_EXPIRATION_OPTION);
+ validateBoolean(options, PARALLELIZE_OUTPUT_SHARDS_OPTION);
s = options.remove(OVERLAP_INCLUSION_METHOD_OPTION);
if (s != null)
@@ -645,6 +663,16 @@ public class Controller
return options;
}
+ private static void validateBoolean(Map options, String option)
+ {
+ String s;
+ s = options.remove(option);
+ if (s != null && !s.equalsIgnoreCase("true") && !s.equalsIgnoreCase("false")) {
+ throw new ConfigurationException(String.format("%s should either be 'true' or 'false', not %s",
+ option, s));
+ }
+ }
+
// The methods below are implemented here (rather than directly in UCS) to aid testability.
public double getBaseSstableSize(int F)
@@ -675,7 +703,7 @@ public class Controller
public int maxConcurrentCompactions()
{
- return DatabaseDescriptor.getConcurrentCompactors();
+ return CompactionManager.instance.getMaximumCompactorThreads();
}
public int maxSSTablesToCompact()
diff --git a/src/java/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriter.java b/src/java/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriter.java
index ca5e99749c..c90afaa8d7 100644
--- a/src/java/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriter.java
@@ -28,7 +28,7 @@ import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.compaction.ShardTracker;
import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.sstable.format.SSTableWriter;
import org.apache.cassandra.utils.FBUtilities;
@@ -47,12 +47,13 @@ public class ShardedCompactionWriter extends CompactionAwareWriter
public ShardedCompactionWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables,
boolean keepOriginals,
+ boolean earlyOpenAllowed,
ShardTracker boundaries)
{
- super(cfs, directories, txn, nonExpiredSSTables, keepOriginals);
+ super(cfs, directories, txn, nonExpiredSSTables, keepOriginals, earlyOpenAllowed);
this.boundaries = boundaries;
long totalKeyCount = nonExpiredSSTables.stream()
diff --git a/src/java/org/apache/cassandra/db/compaction/unified/UnifiedCompactionTask.java b/src/java/org/apache/cassandra/db/compaction/unified/UnifiedCompactionTask.java
index 5a729f6432..0b3e9a6f4a 100644
--- a/src/java/org/apache/cassandra/db/compaction/unified/UnifiedCompactionTask.java
+++ b/src/java/org/apache/cassandra/db/compaction/unified/UnifiedCompactionTask.java
@@ -18,15 +18,20 @@
package org.apache.cassandra.db.compaction.unified;
+import java.util.Collection;
import java.util.Set;
+import com.google.common.collect.ImmutableSet;
+import com.google.common.collect.Sets;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.compaction.CompactionTask;
import org.apache.cassandra.db.compaction.ShardManager;
import org.apache.cassandra.db.compaction.UnifiedCompactionStrategy;
import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.dht.Range;
+import org.apache.cassandra.dht.Token;
import org.apache.cassandra.io.sstable.format.SSTableReader;
/**
@@ -36,26 +41,80 @@ public class UnifiedCompactionTask extends CompactionTask
{
private final ShardManager shardManager;
private final Controller controller;
+ private final Range operationRange;
+ private final Set actuallyCompact;
public UnifiedCompactionTask(ColumnFamilyStore cfs,
UnifiedCompactionStrategy strategy,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
long gcBefore,
ShardManager shardManager)
+ {
+ this(cfs, strategy, txn, gcBefore, shardManager, null, null);
+ }
+
+ public UnifiedCompactionTask(ColumnFamilyStore cfs,
+ UnifiedCompactionStrategy strategy,
+ ILifecycleTransaction txn,
+ long gcBefore,
+ ShardManager shardManager,
+ Range operationRange,
+ Collection actuallyCompact)
{
super(cfs, txn, gcBefore);
this.controller = strategy.getController();
this.shardManager = shardManager;
+
+ if (operationRange != null)
+ {
+ assert actuallyCompact != null : "Ranged tasks should use a set of sstables to compact";
+ }
+ this.operationRange = operationRange;
+ // To make sure actuallyCompact tracks any removals from txn.originals(), we intersect the given set with it.
+ // This should not be entirely necessary (as shouldReduceScopeForSpace() is false for ranged tasks), but it
+ // is cleaner to enforce inputSSTables()'s requirements.
+ this.actuallyCompact = actuallyCompact != null ? Sets.intersection(ImmutableSet.copyOf(actuallyCompact),
+ txn.originals())
+ : txn.originals();
}
@Override
public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables)
{
double density = shardManager.calculateCombinedDensity(nonExpiredSSTables);
int numShards = controller.getNumShards(density * shardManager.shardSetCoverage());
- return new ShardedCompactionWriter(cfs, directories, txn, nonExpiredSSTables, keepOriginals, shardManager.boundaries(numShards));
+ // In multi-task operations we need to expire many ranges in a source sstable for early open. Not doable yet.
+ boolean earlyOpenAllowed = tokenRange() == null;
+ return new ShardedCompactionWriter(cfs,
+ directories,
+ txn,
+ nonExpiredSSTables,
+ keepOriginals,
+ earlyOpenAllowed,
+ shardManager.boundaries(numShards));
+ }
+
+ @Override
+ protected Range tokenRange()
+ {
+ return operationRange;
+ }
+
+ @Override
+ protected boolean shouldReduceScopeForSpace()
+ {
+ // Because parallelized tasks share input sstables, we can't reduce the scope of individual tasks
+ // (as doing that will leave some part of an sstable out of the compaction but still drop the whole sstable
+ // when the task set completes).
+ return tokenRange() == null;
+ }
+
+ @Override
+ protected Set inputSSTables()
+ {
+ return actuallyCompact;
}
}
\ No newline at end of file
diff --git a/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java b/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java
index 145163a39c..ea21f7be57 100644
--- a/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java
@@ -33,7 +33,7 @@ import org.apache.cassandra.db.DiskBoundaries;
import org.apache.cassandra.db.PartitionPosition;
import org.apache.cassandra.db.SerializationHeader;
import org.apache.cassandra.db.compaction.CompactionTask;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
import org.apache.cassandra.io.sstable.Descriptor;
import org.apache.cassandra.io.sstable.SSTableRewriter;
@@ -64,7 +64,7 @@ public abstract class CompactionAwareWriter extends Transactional.AbstractTransa
protected final boolean isTransient;
protected final SSTableRewriter sstableWriter;
- protected final LifecycleTransaction txn;
+ protected final ILifecycleTransaction txn;
private final List locations;
private final List diskBoundaries;
private int locationIndex;
@@ -72,9 +72,19 @@ public abstract class CompactionAwareWriter extends Transactional.AbstractTransa
public CompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables,
boolean keepOriginals)
+ {
+ this(cfs, directories, txn, nonExpiredSSTables, keepOriginals, true);
+ }
+
+ public CompactionAwareWriter(ColumnFamilyStore cfs,
+ Directories directories,
+ ILifecycleTransaction txn,
+ Set nonExpiredSSTables,
+ boolean keepOriginals,
+ boolean earlyOpenAllowed)
{
this.cfs = cfs;
this.directories = directories;
@@ -83,7 +93,7 @@ public abstract class CompactionAwareWriter extends Transactional.AbstractTransa
estimatedTotalKeys = SSTableReader.getApproximateKeyCount(nonExpiredSSTables);
maxAge = CompactionTask.getMaxDataAge(nonExpiredSSTables);
- sstableWriter = SSTableRewriter.construct(cfs, txn, keepOriginals, maxAge);
+ sstableWriter = SSTableRewriter.construct(cfs, txn, keepOriginals, maxAge, earlyOpenAllowed);
minRepairedAt = CompactionTask.getMinRepairedAt(nonExpiredSSTables);
pendingRepair = CompactionTask.getPendingRepair(nonExpiredSSTables);
isTransient = CompactionTask.getIsTransient(nonExpiredSSTables);
diff --git a/src/java/org/apache/cassandra/db/compaction/writers/DefaultCompactionWriter.java b/src/java/org/apache/cassandra/db/compaction/writers/DefaultCompactionWriter.java
index fbb0e27a99..2b124f4417 100644
--- a/src/java/org/apache/cassandra/db/compaction/writers/DefaultCompactionWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/writers/DefaultCompactionWriter.java
@@ -26,7 +26,7 @@ import org.slf4j.LoggerFactory;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.Directories;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.io.sstable.format.SSTableReader;
/**
@@ -37,12 +37,12 @@ public class DefaultCompactionWriter extends CompactionAwareWriter
protected static final Logger logger = LoggerFactory.getLogger(DefaultCompactionWriter.class);
private final int sstableLevel;
- public DefaultCompactionWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables)
+ public DefaultCompactionWriter(ColumnFamilyStore cfs, Directories directories, ILifecycleTransaction txn, Set nonExpiredSSTables)
{
this(cfs, directories, txn, nonExpiredSSTables, false, 0);
}
- public DefaultCompactionWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables, boolean keepOriginals, int sstableLevel)
+ public DefaultCompactionWriter(ColumnFamilyStore cfs, Directories directories, ILifecycleTransaction txn, Set nonExpiredSSTables, boolean keepOriginals, int sstableLevel)
{
super(cfs, directories, txn, nonExpiredSSTables, keepOriginals);
this.sstableLevel = sstableLevel;
diff --git a/src/java/org/apache/cassandra/db/compaction/writers/MajorLeveledCompactionWriter.java b/src/java/org/apache/cassandra/db/compaction/writers/MajorLeveledCompactionWriter.java
index 09263df853..d0fb70587c 100644
--- a/src/java/org/apache/cassandra/db/compaction/writers/MajorLeveledCompactionWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/writers/MajorLeveledCompactionWriter.java
@@ -23,7 +23,7 @@ import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.compaction.LeveledManifest;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
import org.apache.cassandra.io.sstable.format.SSTableReader;
@@ -40,7 +40,7 @@ public class MajorLeveledCompactionWriter extends CompactionAwareWriter
public MajorLeveledCompactionWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables,
long maxSSTableSize)
{
@@ -49,7 +49,7 @@ public class MajorLeveledCompactionWriter extends CompactionAwareWriter
public MajorLeveledCompactionWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables,
long maxSSTableSize,
boolean keepOriginals)
diff --git a/src/java/org/apache/cassandra/db/compaction/writers/MaxSSTableSizeWriter.java b/src/java/org/apache/cassandra/db/compaction/writers/MaxSSTableSizeWriter.java
index 1ded2128e7..75f85b1e4d 100644
--- a/src/java/org/apache/cassandra/db/compaction/writers/MaxSSTableSizeWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/writers/MaxSSTableSizeWriter.java
@@ -23,7 +23,7 @@ import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.compaction.OperationType;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.io.sstable.format.SSTableReader;
public class MaxSSTableSizeWriter extends CompactionAwareWriter
@@ -34,7 +34,7 @@ public class MaxSSTableSizeWriter extends CompactionAwareWriter
public MaxSSTableSizeWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables,
long maxSSTableSize,
int level)
@@ -44,7 +44,7 @@ public class MaxSSTableSizeWriter extends CompactionAwareWriter
public MaxSSTableSizeWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables,
long maxSSTableSize,
int level,
diff --git a/src/java/org/apache/cassandra/db/compaction/writers/SplittingSizeTieredCompactionWriter.java b/src/java/org/apache/cassandra/db/compaction/writers/SplittingSizeTieredCompactionWriter.java
index 4cd0858e18..917fbb9cf0 100644
--- a/src/java/org/apache/cassandra/db/compaction/writers/SplittingSizeTieredCompactionWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/writers/SplittingSizeTieredCompactionWriter.java
@@ -26,7 +26,7 @@ import org.slf4j.LoggerFactory;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.Directories;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.io.sstable.format.SSTableReader;
/**
@@ -46,12 +46,12 @@ public class SplittingSizeTieredCompactionWriter extends CompactionAwareWriter
private long currentBytesToWrite;
private int currentRatioIndex = 0;
- public SplittingSizeTieredCompactionWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables)
+ public SplittingSizeTieredCompactionWriter(ColumnFamilyStore cfs, Directories directories, ILifecycleTransaction txn, Set nonExpiredSSTables)
{
this(cfs, directories, txn, nonExpiredSSTables, DEFAULT_SMALLEST_SSTABLE_BYTES);
}
- public SplittingSizeTieredCompactionWriter(ColumnFamilyStore cfs, Directories directories, LifecycleTransaction txn, Set nonExpiredSSTables, long smallestSSTable)
+ public SplittingSizeTieredCompactionWriter(ColumnFamilyStore cfs, Directories directories, ILifecycleTransaction txn, Set nonExpiredSSTables, long smallestSSTable)
{
super(cfs, directories, txn, nonExpiredSSTables, false);
this.allSSTables = txn.originals();
diff --git a/src/java/org/apache/cassandra/db/lifecycle/CompositeLifecycleTransaction.java b/src/java/org/apache/cassandra/db/lifecycle/CompositeLifecycleTransaction.java
new file mode 100644
index 0000000000..1628baf3e9
--- /dev/null
+++ b/src/java/org/apache/cassandra/db/lifecycle/CompositeLifecycleTransaction.java
@@ -0,0 +1,161 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.lifecycle;
+
+import java.util.concurrent.atomic.AtomicInteger;
+
+import org.slf4j.Logger;
+import org.slf4j.LoggerFactory;
+
+import org.apache.cassandra.utils.TimeUUID;
+
+/// Composite lifecycle transaction. This is a wrapper around a lifecycle transaction that allows for multiple partial
+/// operations that comprise the whole transaction. This is used to parallelize compaction operations over individual
+/// output shards where the compaction sources are shared among the operations; in this case we can only release the
+/// shared sources once all operations are complete.
+///
+/// A composite transaction is initialized with a main transaction that will be used to commit the transaction. Each
+/// part of the composite transaction must be registered with the transaction before it is used. The transaction must
+/// be initialized by calling [#completeInitialization()] before any of the processing is allowed to proceed.
+///
+/// The transaction is considered complete when all parts have been committed or aborted. If any part is aborted, the
+/// whole transaction is also aborted ([PartialLifecycleTransaction] will also throw an exception on other parts when
+/// they access it if the composite is already aborted).
+///
+/// When all parts are committed, the full transaction is applied by performing a checkpoint, obsoletion of the
+/// originals if any of the parts requested it, preparation and commit. This may somewhat violate the rules of
+/// transactions as a part that has been committed may actually have no effect if another part is aborted later.
+/// There are also restrictions on the operations that this model can accept, e.g. replacement of sources and partial
+/// checkpointing are not supported (as they are parts of early open which we don't aim to support at this time),
+/// and we consider that all parts will have the same opinion about the obsoletion of the originals.
+public class CompositeLifecycleTransaction
+{
+ protected static final Logger logger = LoggerFactory.getLogger(CompositeLifecycleTransaction.class);
+
+ final LifecycleTransaction mainTransaction;
+ private final AtomicInteger partsToCommitOrAbort;
+ private volatile boolean obsoleteOriginalsRequested;
+ private volatile boolean wasAborted;
+ private volatile boolean initializationComplete;
+ private volatile int partsCount = 0;
+
+ /// Create a composite transaction wrapper over the given transaction. After construction, the individual parts of
+ /// the operation must be registered using [#register] and the composite sealed by calling [#completeInitialization].
+ /// The composite will then track the state of the parts and commit after all of them have committed (respectively
+ /// abort if one aborts but only after waiting for all the other tasks to complete, successfully or not).
+ ///
+ /// To make it easy to recognize the parts of a composite transaction, the given transaction should have an id with
+ /// sequence number 0, and partial transactions should use the id that [#register] returns.
+ public CompositeLifecycleTransaction(LifecycleTransaction mainTransaction)
+ {
+ this.mainTransaction = mainTransaction;
+ this.partsToCommitOrAbort = new AtomicInteger(0);
+ this.wasAborted = false;
+ this.obsoleteOriginalsRequested = false;
+ }
+
+ /// Register one part of the composite transaction. Every part must register itself before the composite transaction
+ /// is initialized and the parts are allowed to proceed.
+ /// @param part the part to register
+ public TimeUUID register(PartialLifecycleTransaction part)
+ {
+ int index = partsToCommitOrAbort.incrementAndGet();
+ return mainTransaction.opId().withSequence(index);
+ }
+
+ /// Complete the initialization of the composite transaction. This must be called before any of the parts are
+ /// executed.
+ public void completeInitialization()
+ {
+ partsCount = partsToCommitOrAbort.get();
+ initializationComplete = true;
+ if (logger.isTraceEnabled())
+ logger.trace("Composite transaction {} initialized with {} parts.", mainTransaction.opIdString(), partsCount);
+ }
+
+ /// Get the number of parts in the composite transaction. 0 if the transaction is not yet initialized.
+ public int partsCount()
+ {
+ return partsCount;
+ }
+
+ /// Request that the original sstables are obsoleted when the transaction is committed. Note that this class has
+ /// an expectation that all parts will have the same opinion about this, and one request will be sufficient to
+ /// trigger obsoletion.
+ public void requestObsoleteOriginals()
+ {
+ obsoleteOriginalsRequested = true;
+ }
+
+ /// Commit a part of the composite transaction. This will trigger the final commit of the whole transaction if it is
+ /// the last part to complete. A part has to commit or abort exactly once.
+ public void commitPart()
+ {
+ partCommittedOrAborted();
+ }
+
+ /// Signal an abort of one part of the transaction. If this is the last part to signal, the whole transaction will
+ /// now abort. Otherwise the composite transaction will wait for the other parts to complete and will abort the
+ /// composite when they all give their commit or abort signal. A part has to commit or abort exactly once.
+ ///
+ /// [PartialLifecycleTransaction] will attempt to abort other parts sooner by throwing an exception when any of its
+ /// methods are called when the composite transaction is already aborted.
+ public void abortPart()
+ {
+ wasAborted = true;
+ partCommittedOrAborted();
+ }
+
+ boolean wasAborted()
+ {
+ return wasAborted;
+ }
+
+ private void partCommittedOrAborted()
+ {
+ if (!initializationComplete)
+ throw new IllegalStateException("Composite transaction used before initialization is complete.");
+ if (partsToCommitOrAbort.decrementAndGet() == 0)
+ {
+ if (wasAborted)
+ {
+ if (logger.isTraceEnabled())
+ logger.trace("Composite transaction {} with {} parts aborted.",
+ mainTransaction.opIdString(),
+ partsCount);
+
+ mainTransaction.abort();
+ }
+ else
+ {
+ if (logger.isTraceEnabled())
+ logger.trace("Composite transaction {} with {} parts completed{}.",
+ mainTransaction.opIdString(),
+ partsCount,
+ obsoleteOriginalsRequested ? " with obsoletion" : "");
+
+ mainTransaction.checkpoint();
+ if (obsoleteOriginalsRequested)
+ mainTransaction.obsoleteOriginals();
+ mainTransaction.prepareToCommit();
+ mainTransaction.commit();
+ }
+ }
+ }
+}
diff --git a/src/java/org/apache/cassandra/db/lifecycle/ILifecycleTransaction.java b/src/java/org/apache/cassandra/db/lifecycle/ILifecycleTransaction.java
index c014e3865f..f3e9c9b7c5 100644
--- a/src/java/org/apache/cassandra/db/lifecycle/ILifecycleTransaction.java
+++ b/src/java/org/apache/cassandra/db/lifecycle/ILifecycleTransaction.java
@@ -21,7 +21,11 @@ package org.apache.cassandra.db.lifecycle;
import java.util.Collection;
import java.util.Set;
+import com.google.common.collect.Iterables;
+
import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.Throwables;
+import org.apache.cassandra.utils.TimeUUID;
import org.apache.cassandra.utils.concurrent.Transactional;
public interface ILifecycleTransaction extends Transactional, LifecycleNewTracker
@@ -29,10 +33,37 @@ public interface ILifecycleTransaction extends Transactional, LifecycleNewTracke
void checkpoint();
void update(SSTableReader reader, boolean original);
void update(Collection readers, boolean original);
- public SSTableReader current(SSTableReader reader);
+ SSTableReader current(SSTableReader reader);
void obsolete(SSTableReader reader);
void obsoleteOriginals();
Set originals();
boolean isObsolete(SSTableReader reader);
boolean isOffline();
+ TimeUUID opId();
+
+ /// Op identifier as a string to use in debug prints. Usually just the opId, with added part information for partial
+ /// transactions.
+ default String opIdString()
+ {
+ return opId().toString();
+ }
+
+ void cancel(SSTableReader removedSSTable);
+
+ default void abort()
+ {
+ Throwables.maybeFail(abort(null));
+ }
+
+ default void commit()
+ {
+ Throwables.maybeFail(commit(null));
+ }
+
+ default SSTableReader onlyOne()
+ {
+ final Set originals = originals();
+ assert originals.size() == 1;
+ return Iterables.getFirst(originals, null);
+ }
}
diff --git a/src/java/org/apache/cassandra/db/lifecycle/LifecycleTransaction.java b/src/java/org/apache/cassandra/db/lifecycle/LifecycleTransaction.java
index f6e3bb8a1f..4c871cb8e4 100644
--- a/src/java/org/apache/cassandra/db/lifecycle/LifecycleTransaction.java
+++ b/src/java/org/apache/cassandra/db/lifecycle/LifecycleTransaction.java
@@ -125,7 +125,7 @@ public class LifecycleTransaction extends Transactional.AbstractTransactional im
}
}
- public final Tracker tracker;
+ private final Tracker tracker;
// The transaction logs keep track of new and old sstable files
private final LogTransaction log;
// the original readers this transaction was opened over, and that it guards
@@ -184,6 +184,11 @@ public class LifecycleTransaction extends Transactional.AbstractTransactional im
this(tracker, new LogTransaction(operationType, tracker), readers);
}
+ LifecycleTransaction(Tracker tracker, OperationType operationType, Iterable extends SSTableReader> readers, TimeUUID id)
+ {
+ this(tracker, new LogTransaction(operationType, tracker, id), readers);
+ }
+
LifecycleTransaction(Tracker tracker, LogTransaction log, Iterable extends SSTableReader> readers)
{
this.tracker = tracker;
@@ -207,11 +212,18 @@ public class LifecycleTransaction extends Transactional.AbstractTransactional im
return log.type();
}
+ @Override
public TimeUUID opId()
{
return log.id();
}
+ @VisibleForTesting
+ public Tracker tracker()
+ {
+ return tracker;
+ }
+
public void doPrepare()
{
// note for future: in anticompaction two different operations use the same Transaction, and both prepareToCommit()
@@ -576,6 +588,7 @@ public class LifecycleTransaction extends Transactional.AbstractTransactional im
}
// convenience method for callers that know only one sstable is involved in the transaction
+ // overridden to avoid defensive copying
public SSTableReader onlyOne()
{
assert originals.size() == 1;
diff --git a/src/java/org/apache/cassandra/db/lifecycle/LogTransaction.java b/src/java/org/apache/cassandra/db/lifecycle/LogTransaction.java
index b1d13abf3d..635ab38ec6 100644
--- a/src/java/org/apache/cassandra/db/lifecycle/LogTransaction.java
+++ b/src/java/org/apache/cassandra/db/lifecycle/LogTransaction.java
@@ -133,9 +133,14 @@ class LogTransaction extends Transactional.AbstractTransactional implements Tran
}
LogTransaction(OperationType opType, Tracker tracker)
+ {
+ this(opType, tracker, nextTimeUUID());
+ }
+
+ LogTransaction(OperationType opType, Tracker tracker, TimeUUID id)
{
this.tracker = tracker;
- this.txnFile = new LogFile(opType, nextTimeUUID());
+ this.txnFile = new LogFile(opType, id);
this.lock = new Object();
this.selfRef = new Ref<>(this, new TransactionTidier(txnFile, lock));
diff --git a/src/java/org/apache/cassandra/db/lifecycle/PartialLifecycleTransaction.java b/src/java/org/apache/cassandra/db/lifecycle/PartialLifecycleTransaction.java
new file mode 100644
index 0000000000..fccf79479c
--- /dev/null
+++ b/src/java/org/apache/cassandra/db/lifecycle/PartialLifecycleTransaction.java
@@ -0,0 +1,225 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.lifecycle;
+
+import java.util.Collection;
+import java.util.Set;
+import java.util.concurrent.atomic.AtomicBoolean;
+
+import org.apache.cassandra.db.compaction.OperationType;
+import org.apache.cassandra.io.sstable.SSTable;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.Throwables;
+import org.apache.cassandra.utils.TimeUUID;
+
+/// Partial lifecycle transaction. This works together with a CompositeLifecycleTransaction to allow for multiple
+/// tasks using a shared transaction to be committed or aborted together. This is used to parallelize compaction
+/// operations over the same sources. See [CompositeLifecycleTransaction] for more details.
+///
+/// This class takes care of synchronizing various operations on the shared transaction, making sure that an abort
+/// or commit signal is given exactly once (provided that this partial transaction is closed), and throwing an exception
+/// when progress is made when the transaction was already aborted by another part.
+public class PartialLifecycleTransaction implements ILifecycleTransaction
+{
+ final CompositeLifecycleTransaction composite;
+ final ILifecycleTransaction mainTransaction;
+ final AtomicBoolean committedOrAborted = new AtomicBoolean(false);
+ final TimeUUID id;
+
+ public PartialLifecycleTransaction(CompositeLifecycleTransaction composite)
+ {
+ this.composite = composite;
+ this.mainTransaction = composite.mainTransaction;
+ this.id = composite.register(this);
+ }
+
+ public void checkpoint()
+ {
+ // don't do anything, composite will checkpoint at end
+ }
+
+ private RuntimeException earlyOpenUnsupported()
+ {
+ throw new UnsupportedOperationException("PartialLifecycleTransaction does not support early opening of SSTables");
+ }
+
+ public void update(SSTableReader reader, boolean original)
+ {
+ throwIfCompositeAborted();
+ if (original)
+ throw earlyOpenUnsupported();
+
+ synchronized (mainTransaction)
+ {
+ mainTransaction.update(reader, original);
+ }
+ }
+
+ public void update(Collection readers, boolean original)
+ {
+ throwIfCompositeAborted();
+ if (original)
+ throw earlyOpenUnsupported();
+
+ synchronized (mainTransaction)
+ {
+ mainTransaction.update(readers, original);
+ }
+ }
+
+ public SSTableReader current(SSTableReader reader)
+ {
+ synchronized (mainTransaction)
+ {
+ return mainTransaction.current(reader);
+ }
+ }
+
+ public void obsolete(SSTableReader reader)
+ {
+ earlyOpenUnsupported();
+ }
+
+ public void obsoleteOriginals()
+ {
+ composite.requestObsoleteOriginals();
+ }
+
+ public Set originals()
+ {
+ return mainTransaction.originals();
+ }
+
+ public boolean isObsolete(SSTableReader reader)
+ {
+ throw earlyOpenUnsupported();
+ }
+
+ private boolean markCommittedOrAborted()
+ {
+ return committedOrAborted.compareAndSet(false, true);
+ }
+
+ /// Commit the transaction part. Because this is a part of a composite transaction, the actual commit will be
+ /// carried out only after all parts have committed.
+ ///
+ ///
+ public Throwable commit(Throwable accumulate)
+ {
+ Throwables.maybeFail(accumulate); // we must be called with a null accumulate
+ if (markCommittedOrAborted())
+ composite.commitPart();
+ else
+ throw new IllegalStateException("Partial transaction already committed or aborted.");
+ return null;
+ }
+
+ public Throwable abort(Throwable accumulate)
+ {
+ Throwables.maybeFail(accumulate); // we must be called with a null accumulate
+ if (markCommittedOrAborted())
+ composite.abortPart();
+ else
+ throw new IllegalStateException("Partial transaction already committed or aborted.");
+ return null;
+ }
+
+ private void throwIfCompositeAborted()
+ {
+ if (composite.wasAborted())
+ throw new AbortedException("Transaction aborted, likely by another partial operation.");
+ }
+
+ public void prepareToCommit()
+ {
+ if (committedOrAborted.get())
+ throw new IllegalStateException("Partial transaction already committed or aborted.");
+
+ throwIfCompositeAborted();
+ // nothing else to do, the composite transaction will perform the preparation when all parts are done
+ }
+
+ public void close()
+ {
+ if (markCommittedOrAborted()) // close should abort if not committed
+ composite.abortPart();
+ }
+
+ public void trackNew(SSTable table)
+ {
+ throwIfCompositeAborted();
+ synchronized (mainTransaction)
+ {
+ mainTransaction.trackNew(table);
+ }
+ }
+
+ public void untrackNew(SSTable table)
+ {
+ synchronized (mainTransaction)
+ {
+ mainTransaction.untrackNew(table);
+ }
+ }
+
+ public OperationType opType()
+ {
+ return mainTransaction.opType();
+ }
+
+ public boolean isOffline()
+ {
+ return mainTransaction.isOffline();
+ }
+
+ @Override
+ public TimeUUID opId()
+ {
+ return id;
+ }
+
+ @Override
+ public String opIdString()
+ {
+ return String.format("%s (%d/%d)", id, id.sequence(), composite.partsCount());
+ }
+
+ @Override
+ public void cancel(SSTableReader removedSSTable)
+ {
+ synchronized (mainTransaction)
+ {
+ mainTransaction.cancel(removedSSTable);
+ }
+ }
+
+ @Override
+ public String toString()
+ {
+ return opIdString();
+ }
+
+ public static class AbortedException extends RuntimeException
+ {
+ public AbortedException(String message)
+ {
+ super(message);
+ }
+ }
+}
diff --git a/src/java/org/apache/cassandra/db/lifecycle/Tracker.java b/src/java/org/apache/cassandra/db/lifecycle/Tracker.java
index adc18cd3b5..ea8c17783f 100644
--- a/src/java/org/apache/cassandra/db/lifecycle/Tracker.java
+++ b/src/java/org/apache/cassandra/db/lifecycle/Tracker.java
@@ -62,6 +62,7 @@ import org.apache.cassandra.notifications.TablePreScrubNotification;
import org.apache.cassandra.notifications.TruncationNotification;
import org.apache.cassandra.utils.Pair;
import org.apache.cassandra.utils.Throwables;
+import org.apache.cassandra.utils.TimeUUID;
import org.apache.cassandra.utils.concurrent.OpOrder;
import static com.google.common.base.Predicates.and;
@@ -122,12 +123,20 @@ public class Tracker
* @return a Transaction over the provided sstables if we are able to mark the given @param sstables as compacted, before anyone else
*/
public LifecycleTransaction tryModify(Iterable extends SSTableReader> sstables, OperationType operationType)
+ {
+ return tryModify(sstables, operationType, TimeUUID.Generator.nextTimeUUID());
+ }
+
+ /**
+ * @return a Transaction over the provided sstables if we are able to mark the given @param sstables as compacted, before anyone else
+ */
+ public LifecycleTransaction tryModify(Iterable extends SSTableReader> sstables, OperationType operationType, TimeUUID operationId)
{
if (Iterables.isEmpty(sstables))
- return new LifecycleTransaction(this, operationType, sstables);
+ return new LifecycleTransaction(this, operationType, sstables, operationId);
if (null == apply(permitCompacting(sstables), updateCompacting(emptySet(), sstables)))
return null;
- return new LifecycleTransaction(this, operationType, sstables);
+ return new LifecycleTransaction(this, operationType, sstables, operationId);
}
diff --git a/src/java/org/apache/cassandra/db/lifecycle/WrappedLifecycleTransaction.java b/src/java/org/apache/cassandra/db/lifecycle/WrappedLifecycleTransaction.java
index 12c46a9573..74d0a75670 100644
--- a/src/java/org/apache/cassandra/db/lifecycle/WrappedLifecycleTransaction.java
+++ b/src/java/org/apache/cassandra/db/lifecycle/WrappedLifecycleTransaction.java
@@ -24,6 +24,7 @@ import java.util.Set;
import org.apache.cassandra.db.compaction.OperationType;
import org.apache.cassandra.io.sstable.SSTable;
import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.TimeUUID;
public class WrappedLifecycleTransaction implements ILifecycleTransaction
{
@@ -113,4 +114,14 @@ public class WrappedLifecycleTransaction implements ILifecycleTransaction
{
return delegate.isOffline();
}
+
+ public TimeUUID opId()
+ {
+ return delegate.opId();
+ }
+
+ public void cancel(SSTableReader removedSSTable)
+ {
+ delegate.cancel(removedSSTable);
+ }
}
diff --git a/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java b/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java
index 31df8b1120..1b2c9c03a0 100644
--- a/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java
@@ -99,7 +99,12 @@ public class SSTableRewriter extends Transactional.AbstractTransactional impleme
public static SSTableRewriter construct(ColumnFamilyStore cfs, ILifecycleTransaction transaction, boolean keepOriginals, long maxAge)
{
- return new SSTableRewriter(transaction, maxAge, calculateOpenInterval(cfs.supportsEarlyOpen()), keepOriginals, true);
+ return construct(cfs, transaction, keepOriginals, maxAge, true);
+ }
+
+ public static SSTableRewriter construct(ColumnFamilyStore cfs, ILifecycleTransaction transaction, boolean keepOriginals, long maxAge, boolean earlyOpenAllowed)
+ {
+ return new SSTableRewriter(transaction, maxAge, calculateOpenInterval(earlyOpenAllowed && cfs.supportsEarlyOpen()), keepOriginals, true);
}
private static long calculateOpenInterval(boolean shouldOpenEarly)
diff --git a/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java b/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java
index fe99b8430d..8e2db41f6c 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java
@@ -1019,7 +1019,8 @@ public abstract class SSTableReader extends SSTable implements UnfilteredSource,
{
if (range == null)
return getScanner();
- return getScanner(Collections.singletonList(range));
+ else
+ return getScanner(Collections.singletonList(range));
}
/**
diff --git a/src/java/org/apache/cassandra/service/StorageService.java b/src/java/org/apache/cassandra/service/StorageService.java
index 47bb83a5ec..28d9d124e9 100644
--- a/src/java/org/apache/cassandra/service/StorageService.java
+++ b/src/java/org/apache/cassandra/service/StorageService.java
@@ -2689,6 +2689,14 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
}
}
+ public void forceKeyspaceCompaction(boolean splitOutput, int parallelism, String keyspaceName, String... tableNames) throws IOException, ExecutionException, InterruptedException
+ {
+ for (ColumnFamilyStore cfStore : getValidColumnFamilies(true, false, keyspaceName, tableNames))
+ {
+ cfStore.forceMajorCompaction(splitOutput, parallelism);
+ }
+ }
+
public int relocateSSTables(String keyspaceName, String... tableNames) throws IOException, ExecutionException, InterruptedException
{
return relocateSSTables(0, keyspaceName, tableNames);
diff --git a/src/java/org/apache/cassandra/service/StorageServiceMBean.java b/src/java/org/apache/cassandra/service/StorageServiceMBean.java
index 1529a59a56..46ecc0e05a 100644
--- a/src/java/org/apache/cassandra/service/StorageServiceMBean.java
+++ b/src/java/org/apache/cassandra/service/StorageServiceMBean.java
@@ -375,6 +375,11 @@ public interface StorageServiceMBean extends NotificationEmitter
*/
public void forceKeyspaceCompaction(boolean splitOutput, String keyspaceName, String... tableNames) throws IOException, ExecutionException, InterruptedException;
+ /**
+ * Forces major compaction of a single keyspace with the given parallelism limit
+ */
+ public void forceKeyspaceCompaction(boolean splitOutput, int parallelism, String keyspaceName, String... tableNames) throws IOException, ExecutionException, InterruptedException;
+
/** @deprecated See CASSANDRA-11179 */
@Deprecated(since = "3.5")
public int relocateSSTables(String keyspace, String ... cfnames) throws IOException, ExecutionException, InterruptedException;
diff --git a/src/java/org/apache/cassandra/tools/NodeProbe.java b/src/java/org/apache/cassandra/tools/NodeProbe.java
index b2951a49b0..8191bb73ec 100644
--- a/src/java/org/apache/cassandra/tools/NodeProbe.java
+++ b/src/java/org/apache/cassandra/tools/NodeProbe.java
@@ -469,6 +469,11 @@ public class NodeProbe implements AutoCloseable
compactionProxy.forceUserDefinedCompaction(datafiles);
}
+ public void forceKeyspaceCompaction(boolean splitOutput, int parallelism, String keyspaceName, String... tableNames) throws IOException, ExecutionException, InterruptedException
+ {
+ ssProxy.forceKeyspaceCompaction(splitOutput, parallelism, keyspaceName, tableNames);
+ }
+
public void forceKeyspaceCompaction(boolean splitOutput, String keyspaceName, String... tableNames) throws IOException, ExecutionException, InterruptedException
{
ssProxy.forceKeyspaceCompaction(splitOutput, keyspaceName, tableNames);
diff --git a/src/java/org/apache/cassandra/tools/nodetool/Compact.java b/src/java/org/apache/cassandra/tools/nodetool/Compact.java
index f5a83ed904..1f8b4c4b46 100644
--- a/src/java/org/apache/cassandra/tools/nodetool/Compact.java
+++ b/src/java/org/apache/cassandra/tools/nodetool/Compact.java
@@ -50,6 +50,12 @@ public class Compact extends NodeToolCmd
@Option(title = "partition_key", name = {"--partition"}, description = "String representation of the partition key")
private String partitionKey = EMPTY;
+ @Option(title = "jobs",
+ name = {"-j", "--jobs"},
+ description = "Use -j to specify the maximum number of threads to use for parallel compaction. " +
+ "If not set, up to half the compaction threads will be used. " +
+ "If set to 0, the major compaction will use all threads and will not permit other compactions to run until it completes (use with caution).")
+ private Integer parallelism = null;
@Override
public void execute(NodeProbe probe)
@@ -95,7 +101,10 @@ public class Compact extends NodeToolCmd
}
else
{
- probe.forceKeyspaceCompaction(splitOutput, keyspace, tableNames);
+ if (parallelism != null)
+ probe.forceKeyspaceCompaction(splitOutput, parallelism, keyspace, tableNames);
+ else // avoid referring to the new method to work with older versions
+ probe.forceKeyspaceCompaction(splitOutput, keyspace, tableNames);
}
} catch (Exception e)
{
diff --git a/src/java/org/apache/cassandra/utils/Throwables.java b/src/java/org/apache/cassandra/utils/Throwables.java
index 3665dfca82..242c768808 100644
--- a/src/java/org/apache/cassandra/utils/Throwables.java
+++ b/src/java/org/apache/cassandra/utils/Throwables.java
@@ -333,4 +333,11 @@ public final class Throwables
if (!anyCauseMatches(err, cause::isInstance))
throw new AssertionError("The exception is not caused by " + cause.getName(), err);
}
+
+ @VisibleForTesting
+ public static void assertAnyCause(Throwable err, Class extends Throwable>... causeClasses)
+ {
+ if (Arrays.stream(causeClasses).noneMatch(c -> anyCauseMatches(err, c::isInstance)))
+ throw new AssertionError("The exception is not caused by any of " + Arrays.toString(causeClasses), err);
+ }
}
diff --git a/src/java/org/apache/cassandra/utils/TimeUUID.java b/src/java/org/apache/cassandra/utils/TimeUUID.java
index d993f171af..0b17504763 100644
--- a/src/java/org/apache/cassandra/utils/TimeUUID.java
+++ b/src/java/org/apache/cassandra/utils/TimeUUID.java
@@ -293,6 +293,25 @@ public class TimeUUID implements Serializable, Comparable
return ballot == null ? "null" : String.format("%s(%d:%s)", kind, ballot.uuidTimestamp(), ballot);
}
+ public int sequence()
+ {
+ return (int) ((lsb >> 48) & 0x0000000000003FFFL);
+ }
+
+ /**
+ * Returns a new TimeUUID with the same data as this one, but with the provided sequence value.
+ *
+ * Warning: the uniqueness of the returned TimeUUID is not guaranteed by this method. Caller must ensure that
+ * the sequence numbers in use are distinct.
+ */
+ public TimeUUID withSequence(long sequence)
+ {
+ long sequenceBits = 0x0000000000003FFFL;
+ long sequenceMask = ~(sequenceBits << 48);
+ final long bits = (sequence & sequenceBits) << 48;
+ return new TimeUUID(uuidTimestamp, lsb() & sequenceMask | bits);
+ }
+
@Override
public int compareTo(TimeUUID that)
{
diff --git a/test/distributed/org/apache/cassandra/distributed/test/PaxosRepair2Test.java b/test/distributed/org/apache/cassandra/distributed/test/PaxosRepair2Test.java
index ed066b2c3b..8b043ab6aa 100644
--- a/test/distributed/org/apache/cassandra/distributed/test/PaxosRepair2Test.java
+++ b/test/distributed/org/apache/cassandra/distributed/test/PaxosRepair2Test.java
@@ -287,7 +287,7 @@ public class PaxosRepair2Test extends TestBaseImpl
{
ColumnFamilyStore paxos = Keyspace.open(SYSTEM_KEYSPACE_NAME).getColumnFamilyStore(SystemKeyspace.PAXOS);
FBUtilities.waitOnFuture(paxos.forceFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS));
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(paxos, 0, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(paxos, 0, false, 0));
}
private static Map getPaxosRows()
diff --git a/test/distributed/org/apache/cassandra/distributed/test/PaxosRepairTest.java b/test/distributed/org/apache/cassandra/distributed/test/PaxosRepairTest.java
index 8e3db2039f..e22a6be3ea 100644
--- a/test/distributed/org/apache/cassandra/distributed/test/PaxosRepairTest.java
+++ b/test/distributed/org/apache/cassandra/distributed/test/PaxosRepairTest.java
@@ -597,7 +597,7 @@ public class PaxosRepairTest extends TestBaseImpl
{
ColumnFamilyStore paxos = Keyspace.open(SYSTEM_KEYSPACE_NAME).getColumnFamilyStore(SystemKeyspace.PAXOS);
FBUtilities.waitOnFuture(paxos.forceFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS));
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(paxos, 0, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(paxos, 0, false, 0));
}
private static Map getPaxosRows()
diff --git a/test/distributed/org/apache/cassandra/distributed/test/UpgradeSSTablesTest.java b/test/distributed/org/apache/cassandra/distributed/test/UpgradeSSTablesTest.java
index c2caac3e0e..3ec8952221 100644
--- a/test/distributed/org/apache/cassandra/distributed/test/UpgradeSSTablesTest.java
+++ b/test/distributed/org/apache/cassandra/distributed/test/UpgradeSSTablesTest.java
@@ -83,7 +83,7 @@ public class UpgradeSSTablesTest extends TestBaseImpl
Future> future = cluster.get(1).asyncAcceptsOnInstance((String ks) -> {
ColumnFamilyStore cfs = Keyspace.open(ks).getColumnFamilyStore("tbl");
- CompactionManager.instance.submitMaximal(cfs, FBUtilities.nowInSeconds(), false, OperationType.COMPACTION);
+ CompactionManager.instance.submitMaximal(cfs, FBUtilities.nowInSeconds(), false, 1, OperationType.COMPACTION);
}).apply(KEYSPACE);
Assert.assertTrue(cluster.get(1).callOnInstance(() -> CompactionLatchByteman.starting.awaitUninterruptibly(1, TimeUnit.MINUTES)));
@@ -129,7 +129,7 @@ public class UpgradeSSTablesTest extends TestBaseImpl
cluster.get(1).acceptsOnInstance((String ks) -> {
ColumnFamilyStore cfs = Keyspace.open(ks).getColumnFamilyStore("tbl");
- FBUtilities.allOf(CompactionManager.instance.submitMaximal(cfs, FBUtilities.nowInSeconds(), false, OperationType.COMPACTION))
+ FBUtilities.allOf(CompactionManager.instance.submitMaximal(cfs, FBUtilities.nowInSeconds(), false, 1, OperationType.COMPACTION))
.awaitUninterruptibly(1, TimeUnit.MINUTES);
}).accept(KEYSPACE);
diff --git a/test/long/org/apache/cassandra/db/compaction/LongCompactionsTest.java b/test/long/org/apache/cassandra/db/compaction/LongCompactionsTest.java
index caf9a359a3..e7043ba5ae 100644
--- a/test/long/org/apache/cassandra/db/compaction/LongCompactionsTest.java
+++ b/test/long/org/apache/cassandra/db/compaction/LongCompactionsTest.java
@@ -200,7 +200,7 @@ public class LongCompactionsTest
if (cfs.getLiveSSTables().size() > 1)
{
- CompactionManager.instance.performMaximal(cfs, false);
+ CompactionManager.instance.performMaximal(cfs);
}
}
}
diff --git a/test/long/org/apache/cassandra/db/compaction/LongLeveledCompactionStrategyTest.java b/test/long/org/apache/cassandra/db/compaction/LongLeveledCompactionStrategyTest.java
index 6c4a8ddb9b..51500d0bab 100644
--- a/test/long/org/apache/cassandra/db/compaction/LongLeveledCompactionStrategyTest.java
+++ b/test/long/org/apache/cassandra/db/compaction/LongLeveledCompactionStrategyTest.java
@@ -99,16 +99,19 @@ public class LongLeveledCompactionStrategyTest
{
while (true)
{
- final AbstractCompactionTask nextTask = lcs.getNextBackgroundTask(Integer.MIN_VALUE);
- if (nextTask == null)
+ final var nextTasks = lcs.getNextBackgroundTasks(Integer.MIN_VALUE);
+ if (nextTasks == null || nextTasks.isEmpty())
break;
- tasks.add(new Runnable()
+ for (var nextTask : nextTasks)
{
- public void run()
+ tasks.add(new Runnable()
{
- nextTask.execute(ActiveCompactionsTracker.NOOP);
- }
- });
+ public void run()
+ {
+ nextTask.execute(ActiveCompactionsTracker.NOOP);
+ }
+ });
+ }
}
if (tasks.isEmpty())
break;
diff --git a/test/microbench/org/apache/cassandra/test/microbench/CachingBenchTest.java b/test/microbench/org/apache/cassandra/test/microbench/CachingBenchTest.java
index 3a0f09fc2a..8ffd8150be 100644
--- a/test/microbench/org/apache/cassandra/test/microbench/CachingBenchTest.java
+++ b/test/microbench/org/apache/cassandra/test/microbench/CachingBenchTest.java
@@ -262,7 +262,7 @@ public class CachingBenchTest extends CQLTester
String hashesBefore = getHashes();
long startTime = currentTimeMillis();
- CompactionManager.instance.performMaximal(cfs, true);
+ CompactionManager.instance.performMaximal(cfs, true, 0);
long endTime = currentTimeMillis();
int endRowCount = countRows(cfs);
diff --git a/test/microbench/org/apache/cassandra/test/microbench/ZeroCopyStreamingBench.java b/test/microbench/org/apache/cassandra/test/microbench/ZeroCopyStreamingBench.java
index 7b2bc3afaa..e10b91d9b2 100644
--- a/test/microbench/org/apache/cassandra/test/microbench/ZeroCopyStreamingBench.java
+++ b/test/microbench/org/apache/cassandra/test/microbench/ZeroCopyStreamingBench.java
@@ -202,7 +202,7 @@ public class ZeroCopyStreamingBench
.applyUnsafe();
}
store.forceBlockingFlush(ColumnFamilyStore.FlushReason.USER_FORCED);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
}
@TearDown
diff --git a/test/unit/org/apache/cassandra/cql3/GcCompactionTest.java b/test/unit/org/apache/cassandra/cql3/GcCompactionTest.java
index 31b303471e..7e118e7928 100644
--- a/test/unit/org/apache/cassandra/cql3/GcCompactionTest.java
+++ b/test/unit/org/apache/cassandra/cql3/GcCompactionTest.java
@@ -232,7 +232,7 @@ public class GcCompactionTest extends CQLTester
assertEquals(0, collected.getSSTableLevel()); // garbagecollect should leave the LCS level where it was
- CompactionManager.instance.performMaximal(cfs, false);
+ CompactionManager.instance.performMaximal(cfs);
assertEquals(1, cfs.getLiveSSTables().size());
SSTableReader compacted = cfs.getLiveSSTables().iterator().next();
diff --git a/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java b/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java
index 74a0a4b17f..b7d142536a 100644
--- a/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java
+++ b/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java
@@ -188,7 +188,7 @@ public class CrcCheckChanceTest extends CQLTester
}
DatabaseDescriptor.setCompactionThroughputMebibytesPerSec(1);
- List extends Future>> futures = CompactionManager.instance.submitMaximal(cfs, CompactionManager.getDefaultGcBefore(cfs, FBUtilities.nowInSeconds()), false);
+ List extends Future>> futures = CompactionManager.instance.submitMaximal(cfs, CompactionManager.getDefaultGcBefore(cfs, FBUtilities.nowInSeconds()), false, 1);
execute("DROP TABLE %s");
try
diff --git a/test/unit/org/apache/cassandra/db/KeyspaceTest.java b/test/unit/org/apache/cassandra/db/KeyspaceTest.java
index f78d081dce..a42359de46 100644
--- a/test/unit/org/apache/cassandra/db/KeyspaceTest.java
+++ b/test/unit/org/apache/cassandra/db/KeyspaceTest.java
@@ -419,7 +419,7 @@ public class KeyspaceTest extends CQLTester
// compact so we have a big row with more than the minimum index count
if (cfs.getLiveSSTables().size() > 1)
- CompactionManager.instance.performMaximal(cfs, false);
+ CompactionManager.instance.performMaximal(cfs);
// verify that we do indeed have multiple index entries
SSTableReader sstable = cfs.getLiveSSTables().iterator().next();
diff --git a/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java b/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java
index 2fee758779..2eb521c688 100644
--- a/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java
+++ b/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java
@@ -421,7 +421,7 @@ public class RangeTombstoneTest
partition.getRow(Clustering.make(bb(i))).hasLiveData(nowInSec, enforceStrictLiveness));
// Compact everything and re-test
- CompactionManager.instance.performMaximal(cfs, false);
+ CompactionManager.instance.performMaximal(cfs);
partition = Util.getOnlyPartitionUnfiltered(Util.cmd(cfs, key).build());
for (int i = 0; i < 5; i++)
@@ -515,7 +515,7 @@ public class RangeTombstoneTest
assertEquals(10, index.rowsInserted.size());
- CompactionManager.instance.performMaximal(cfs, false);
+ CompactionManager.instance.performMaximal(cfs);
// compacted down to single sstable
assertEquals(1, cfs.getLiveSSTables().size());
@@ -547,7 +547,7 @@ public class RangeTombstoneTest
assertEquals(2, cfs.getLiveSSTables().size());
// compact down to single sstable
- CompactionManager.instance.performMaximal(cfs, false);
+ CompactionManager.instance.performMaximal(cfs);
assertEquals(1, cfs.getLiveSSTables().size());
// test the physical structure of the sstable i.e. rt & columns on disk
diff --git a/test/unit/org/apache/cassandra/db/compaction/AbstractCompactionStrategyTest.java b/test/unit/org/apache/cassandra/db/compaction/AbstractCompactionStrategyTest.java
index 1fc43bdcc3..043778092e 100644
--- a/test/unit/org/apache/cassandra/db/compaction/AbstractCompactionStrategyTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/AbstractCompactionStrategyTest.java
@@ -20,6 +20,7 @@ package org.apache.cassandra.db.compaction;
import java.util.Collections;
+import com.google.common.collect.Iterables;
import org.junit.After;
import org.junit.BeforeClass;
import org.junit.Test;
@@ -31,7 +32,7 @@ import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.Keyspace;
import org.apache.cassandra.db.RowUpdateBuilder;
-import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.exceptions.ConfigurationException;
import org.apache.cassandra.schema.CompactionParams;
import org.apache.cassandra.schema.KeyspaceParams;
@@ -100,7 +101,8 @@ public class AbstractCompactionStrategyTest
}
// Check they are returned on the next background task
- try (LifecycleTransaction txn = strategy.getNextBackgroundTask(FBUtilities.nowInSeconds()).transaction)
+ long gcBefore1 = FBUtilities.nowInSeconds();
+ try (ILifecycleTransaction txn = Iterables.getOnlyElement(strategy.getNextBackgroundTasks(gcBefore1), null).transaction)
{
Assert.assertEquals(cfs.getLiveSSTables(), txn.originals());
}
@@ -109,7 +111,8 @@ public class AbstractCompactionStrategyTest
cfs.getTracker().removeUnsafe(cfs.getLiveSSTables());
// verify the compaction strategy will return null
- Assert.assertNull(strategy.getNextBackgroundTask(FBUtilities.nowInSeconds()));
+ long gcBefore = FBUtilities.nowInSeconds();
+ Assert.assertNull(Iterables.getOnlyElement(strategy.getNextBackgroundTasks(gcBefore), null));
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/CancelCompactionsTest.java b/test/unit/org/apache/cassandra/db/compaction/CancelCompactionsTest.java
index b3a182f186..757ad1c086 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CancelCompactionsTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CancelCompactionsTest.java
@@ -31,6 +31,7 @@ import java.util.concurrent.TimeUnit;
import java.util.stream.Collectors;
import com.google.common.collect.ImmutableSet;
+import com.google.common.collect.Iterables;
import com.google.common.util.concurrent.Uninterruptibles;
import org.junit.Assume;
import org.junit.Test;
@@ -485,7 +486,7 @@ public class CancelCompactionsTest extends CQLTester
{
for (AbstractCompactionStrategy cs : css)
{
- ct = cs.getNextBackgroundTask(0);
+ ct = Iterables.getOnlyElement(cs.getNextBackgroundTasks(0), null);
if (ct != null)
break;
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/CompactionControllerTest.java b/test/unit/org/apache/cassandra/db/compaction/CompactionControllerTest.java
index 1f15529aa2..f46150cffc 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CompactionControllerTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CompactionControllerTest.java
@@ -184,19 +184,19 @@ public class CompactionControllerTest extends SchemaLoader
// the first sstable should be expired because the overlapping sstable is newer and the gc period is later
long gcBefore = (System.currentTimeMillis() / 1000) + 5;
- Set expired = CompactionController.getFullyExpiredSSTables(cfs, compacting, overlapping, gcBefore);
+ Set expired = CompactionController.getFullyExpiredSSTables(cfs, compacting, x -> overlapping, gcBefore);
assertNotNull(expired);
assertEquals(1, expired.size());
assertEquals(compacting.iterator().next(), expired.iterator().next());
// however if we add an older mutation to the memtable then the sstable should not be expired
applyMutation(cfs.metadata(), key, timestamp3);
- expired = CompactionController.getFullyExpiredSSTables(cfs, compacting, overlapping, gcBefore);
+ expired = CompactionController.getFullyExpiredSSTables(cfs, compacting, x -> overlapping, gcBefore);
assertNotNull(expired);
assertEquals(0, expired.size());
// Now if we explicitly ask to ignore overlaped sstables, we should get back our expired sstable
- expired = CompactionController.getFullyExpiredSSTables(cfs, compacting, overlapping, gcBefore, true);
+ expired = CompactionController.getFullyExpiredSSTables(cfs, compacting, x -> overlapping, gcBefore, true);
assertNotNull(expired);
assertEquals(1, expired.size());
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/CompactionStrategyManagerPendingRepairTest.java b/test/unit/org/apache/cassandra/db/compaction/CompactionStrategyManagerPendingRepairTest.java
index 483b77a29a..c8eca664bd 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CompactionStrategyManagerPendingRepairTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CompactionStrategyManagerPendingRepairTest.java
@@ -253,7 +253,7 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
Assert.assertFalse(sstable.isRepaired());
cfs.getCompactionStrategyManager().enable(); // enable compaction to fetch next background task
- AbstractCompactionTask compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds());
+ AbstractCompactionTask compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null);
Assert.assertNotNull(compactionTask);
Assert.assertSame(PendingRepairManager.RepairFinishedCompactionTask.class, compactionTask.getClass());
@@ -294,7 +294,7 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
Assert.assertFalse(sstable.isRepaired());
cfs.getCompactionStrategyManager().enable(); // enable compaction to fetch next background task
- AbstractCompactionTask compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds());
+ AbstractCompactionTask compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null);
Assert.assertNotNull(compactionTask);
Assert.assertSame(PendingRepairManager.RepairFinishedCompactionTask.class, compactionTask.getClass());
@@ -348,7 +348,7 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
cfs.getCompactionStrategyManager().enable();
for (SSTableReader sstable : sstables)
pendingContains(sstable);
- AbstractCompactionTask compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds());
+ AbstractCompactionTask compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null);
// Finalize the repair session
LocalSessionAccessor.finalizeUnsafe(repairID);
@@ -374,13 +374,13 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
System.out.println("*********************************************************************************************");
// Run compaction again. It should pick up the pending repair sstable
- compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds());
+ compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null);
if (compactionTask != null)
{
Assert.assertSame(PendingRepairManager.RepairFinishedCompactionTask.class, compactionTask.getClass());
compactionTask.execute(ActiveCompactionsTracker.NOOP);
- while ((compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds())) != null)
+ while ((compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null)) != null)
compactionTask.execute(ActiveCompactionsTracker.NOOP);
}
@@ -388,7 +388,7 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
Util.spinAssertEquals(Boolean.FALSE,
() -> {
AbstractCompactionTask ctask;
- while ((ctask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds())) != null)
+ while ((ctask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null)) != null)
ctask.execute(ActiveCompactionsTracker.NOOP);
return hasPendingStrategiesFor(repairID);
@@ -434,7 +434,7 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
Assert.assertFalse(unrepairedContains(sstable));
cfs.getCompactionStrategyManager().enable(); // enable compaction to fetch next background task
- AbstractCompactionTask compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds());
+ AbstractCompactionTask compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null);
Assert.assertNotNull(compactionTask);
Assert.assertSame(PendingRepairManager.RepairFinishedCompactionTask.class, compactionTask.getClass());
@@ -465,7 +465,7 @@ public class CompactionStrategyManagerPendingRepairTest extends AbstractPendingR
Assert.assertFalse(unrepairedContains(sstable));
cfs.getCompactionStrategyManager().enable(); // enable compaction to fetch next background task
- AbstractCompactionTask compactionTask = csm.getNextBackgroundTask(FBUtilities.nowInSeconds());
+ AbstractCompactionTask compactionTask = Iterables.getOnlyElement(csm.getNextBackgroundTasks(FBUtilities.nowInSeconds()), null);
Assert.assertNotNull(compactionTask);
Assert.assertSame(PendingRepairManager.RepairFinishedCompactionTask.class, compactionTask.getClass());
diff --git a/test/unit/org/apache/cassandra/db/compaction/CompactionTaskTest.java b/test/unit/org/apache/cassandra/db/compaction/CompactionTaskTest.java
index 066aa92067..8109e40dfe 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CompactionTaskTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CompactionTaskTest.java
@@ -213,13 +213,13 @@ public class CompactionTaskTest
try (LifecycleTransaction txn = LifecycleTransaction.offline(OperationType.COMPACTION, sstables))
{
- Assert.assertEquals(4, txn.tracker.getView().liveSSTables().size());
+ Assert.assertEquals(4, txn.tracker().getView().liveSSTables().size());
CompactionTask task = new CompactionTask(cfs, txn, 1000);
task.execute(null);
// Check that new SSTable was not released
- Assert.assertEquals(1, txn.tracker.getView().liveSSTables().size());
- SSTableReader newSSTable = txn.tracker.getView().liveSSTables().iterator().next();
+ Assert.assertEquals(1, txn.tracker().getView().liveSSTables().size());
+ SSTableReader newSSTable = txn.tracker().getView().liveSSTables().iterator().next();
Assert.assertNotNull(newSSTable.tryRef());
}
finally
@@ -233,7 +233,7 @@ public class CompactionTaskTest
public void testMajorCompactTask()
{
//major compact without range/pk specified
- CompactionTasks compactionTasks = cfs.getCompactionStrategyManager().getMaximalTasks(Integer.MAX_VALUE, false, OperationType.MAJOR_COMPACTION);
+ CompactionTasks compactionTasks = cfs.getCompactionStrategyManager().getMaximalTasks(Integer.MAX_VALUE, false, Integer.MAX_VALUE, OperationType.MAJOR_COMPACTION);
Assert.assertTrue(compactionTasks.stream().allMatch(task -> task.compactionType.equals(OperationType.MAJOR_COMPACTION)));
}
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/CompactionsCQLTest.java b/test/unit/org/apache/cassandra/db/compaction/CompactionsCQLTest.java
index ca9f5bbcd1..ce04222172 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CompactionsCQLTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CompactionsCQLTest.java
@@ -17,36 +17,17 @@
*/
package org.apache.cassandra.db.compaction;
-import java.io.IOException;
-import java.nio.ByteBuffer;
-import java.nio.file.FileStore;
-import java.util.Collection;
-import java.util.HashMap;
-import java.util.List;
-import java.util.Map;
-import java.util.Random;
-import java.util.Set;
-
-import org.apache.commons.lang3.StringUtils;
-
-import org.junit.After;
-import org.junit.Assert;
-import org.junit.Before;
-import org.junit.Test;
-
+import com.google.common.collect.Iterables;
import org.apache.cassandra.Util;
import org.apache.cassandra.config.Config;
import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.cql3.CQLTester;
import org.apache.cassandra.cql3.UntypedResultSet;
-import org.apache.cassandra.db.ColumnFamilyStore;
-import org.apache.cassandra.db.DecoratedKey;
-import org.apache.cassandra.db.Directories;
-import org.apache.cassandra.db.Keyspace;
-import org.apache.cassandra.db.RowUpdateBuilder;
+import org.apache.cassandra.db.*;
import org.apache.cassandra.db.commitlog.CommitLog;
import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.compaction.writers.MaxSSTableSizeWriter;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
import org.apache.cassandra.db.rows.Cell;
import org.apache.cassandra.db.rows.Row;
@@ -61,14 +42,20 @@ import org.apache.cassandra.io.util.File;
import org.apache.cassandra.io.util.PathUtils;
import org.apache.cassandra.schema.CompactionParams;
import org.apache.cassandra.serializers.MarshalException;
+import org.apache.commons.lang3.StringUtils;
import org.assertj.core.api.Assertions;
+import org.junit.After;
+import org.junit.Assert;
+import org.junit.Before;
+import org.junit.Test;
+
+import java.io.IOException;
+import java.nio.ByteBuffer;
+import java.nio.file.FileStore;
+import java.util.*;
import static org.apache.cassandra.utils.TimeUUID.Generator.nextTimeUUID;
-import static org.junit.Assert.assertEquals;
-import static org.junit.Assert.assertFalse;
-import static org.junit.Assert.assertNotNull;
-import static org.junit.Assert.assertTrue;
-import static org.junit.Assert.fail;
+import static org.junit.Assert.*;
public class CompactionsCQLTest extends CQLTester
{
@@ -491,7 +478,7 @@ public class CompactionsCQLTest extends CQLTester
}
assertEquals(50, cfs.getLiveSSTables().size());
LeveledCompactionStrategy lcs = (LeveledCompactionStrategy) cfs.getCompactionStrategyManager().getUnrepairedUnsafe().first();
- AbstractCompactionTask act = lcs.getNextBackgroundTask(0);
+ AbstractCompactionTask act = Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null);
// we should be compacting all 50 sstables:
assertEquals(50, act.transaction.originals().size());
act.execute(ActiveCompactionsTracker.NOOP);
@@ -525,7 +512,7 @@ public class CompactionsCQLTest extends CQLTester
// mark the L1 sstable as compacting to make sure we trigger STCS in L0:
LifecycleTransaction txn = cfs.getTracker().tryModify(l1sstable, OperationType.COMPACTION);
LeveledCompactionStrategy lcs = (LeveledCompactionStrategy) cfs.getCompactionStrategyManager().getUnrepairedUnsafe().first();
- AbstractCompactionTask act = lcs.getNextBackgroundTask(0);
+ AbstractCompactionTask act = Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null);
// note that max_threshold is 60 (more than the amount of L0 sstables), but MAX_COMPACTING_L0 is 32, which means we will trigger STCS with at most max_threshold sstables
assertEquals(50, act.transaction.originals().size());
assertEquals(0, ((LeveledCompactionTask)act).getLevel());
@@ -558,7 +545,7 @@ public class CompactionsCQLTest extends CQLTester
LeveledCompactionTask lcsTask;
while (true)
{
- lcsTask = (LeveledCompactionTask) lcs.getNextBackgroundTask(0);
+ lcsTask = (LeveledCompactionTask) Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null);
if (lcsTask != null)
{
lcsTask.execute(CompactionManager.instance.active);
@@ -595,7 +582,7 @@ public class CompactionsCQLTest extends CQLTester
// sstables have been removed.
try
{
- AbstractCompactionTask task = new NotifyingCompactionTask((LeveledCompactionTask) lcs.getNextBackgroundTask(0));
+ AbstractCompactionTask task = new NotifyingCompactionTask((LeveledCompactionTask) Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null));
task.execute(CompactionManager.instance.active);
fail("task should throw exception");
}
@@ -604,7 +591,7 @@ public class CompactionsCQLTest extends CQLTester
// ignored
}
- lcsTask = (LeveledCompactionTask) lcs.getNextBackgroundTask(0);
+ lcsTask = (LeveledCompactionTask) Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null);
try
{
assertNotNull(lcsTask);
@@ -626,7 +613,7 @@ public class CompactionsCQLTest extends CQLTester
@Override
public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
Directories directories,
- LifecycleTransaction txn,
+ ILifecycleTransaction txn,
Set nonExpiredSSTables)
{
return new MaxSSTableSizeWriter(cfs, directories, txn, nonExpiredSSTables, 1 << 20, 1)
diff --git a/test/unit/org/apache/cassandra/db/compaction/CompactionsPurgeTest.java b/test/unit/org/apache/cassandra/db/compaction/CompactionsPurgeTest.java
index 4f1b639d70..9916bffa74 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CompactionsPurgeTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CompactionsPurgeTest.java
@@ -120,7 +120,7 @@ public class CompactionsPurgeTest
Util.flush(cfs);
// major compact and test that all columns but the resurrected one is completely gone
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Integer.MAX_VALUE, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Long.MAX_VALUE, false, 0));
cfs.invalidateCachedPartition(dk(key));
ImmutableBTreePartition partition = Util.getOnlyPartitionUnfiltered(Util.cmd(cfs, key).build());
@@ -156,7 +156,7 @@ public class CompactionsPurgeTest
Util.flush(cfs);
// major compact - tombstones should be purged
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Integer.MAX_VALUE, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Long.MAX_VALUE, false, 0));
// resurrect one column
RowUpdateBuilder builder = new RowUpdateBuilder(cfs.metadata(), 2, key);
@@ -200,7 +200,7 @@ public class CompactionsPurgeTest
Util.flush(cfs);
// major compact - tombstones should be purged
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Integer.MAX_VALUE, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Long.MAX_VALUE, false, 0));
// resurrect one column
RowUpdateBuilder builder = new RowUpdateBuilder(cfs.metadata(), 2, key);
@@ -242,7 +242,7 @@ public class CompactionsPurgeTest
Util.flush(cfs);
// major compact - tombstones should be purged
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Integer.MAX_VALUE, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Long.MAX_VALUE, false, 0));
// resurrect one column
RowUpdateBuilder builder = new RowUpdateBuilder(cfs.metadata(), 2, key);
@@ -519,7 +519,7 @@ public class CompactionsPurgeTest
assertEquals(0, result.size());
// compact the two sstables with a gcBefore that does *not* allow the row tombstone to be purged
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, (int) (System.currentTimeMillis() / 1000) - 10000, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, (int) (System.currentTimeMillis() / 1000) - 10000, false, 0));
// the data should be gone, but the tombstone should still exist
assertEquals(1, cfs.getLiveSSTables().size());
@@ -539,7 +539,7 @@ public class CompactionsPurgeTest
Util.flush(cfs);
// compact the two sstables with a gcBefore that *does* allow the row tombstone to be purged
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, (int) (System.currentTimeMillis() / 1000) + 10000, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, (int) (System.currentTimeMillis() / 1000) + 10000, false, 0));
// both the data and the tombstone should be gone this time
assertEquals(0, cfs.getLiveSSTables().size());
diff --git a/test/unit/org/apache/cassandra/db/compaction/CorruptedSSTablesCompactionsTest.java b/test/unit/org/apache/cassandra/db/compaction/CorruptedSSTablesCompactionsTest.java
index dc78276254..1705256c76 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CorruptedSSTablesCompactionsTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CorruptedSSTablesCompactionsTest.java
@@ -23,34 +23,47 @@ package org.apache.cassandra.db.compaction;
import java.nio.ByteBuffer;
import java.nio.channels.FileChannel;
-import java.util.*;
+import java.util.Collection;
+import java.util.Collections;
+import java.util.HashMap;
+import java.util.HashSet;
+import java.util.Random;
+import java.util.Set;
+import com.google.common.collect.ImmutableMap;
import org.junit.After;
import org.junit.AfterClass;
import org.junit.BeforeClass;
import org.junit.Test;
-
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
-import static org.apache.cassandra.utils.Clock.Global.nanoTime;
-import static org.junit.Assert.assertEquals;
-import static org.junit.Assert.assertNotNull;
-
import org.apache.cassandra.SchemaLoader;
import org.apache.cassandra.Util;
import org.apache.cassandra.cache.ChunkCache;
-import org.apache.cassandra.config.*;
-import org.apache.cassandra.db.*;
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.Keyspace;
+import org.apache.cassandra.db.RowUpdateBuilder;
+import org.apache.cassandra.db.lifecycle.PartialLifecycleTransaction;
import org.apache.cassandra.db.marshal.LongType;
+import org.apache.cassandra.db.rows.UnfilteredRowIterator;
+import org.apache.cassandra.dht.Murmur3Partitioner;
import org.apache.cassandra.exceptions.ConfigurationException;
import org.apache.cassandra.io.sstable.CorruptSSTableException;
+import org.apache.cassandra.io.sstable.ISSTableScanner;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.util.File;
import org.apache.cassandra.io.util.FileUtils;
-import org.apache.cassandra.schema.*;
+import org.apache.cassandra.schema.CompactionParams;
+import org.apache.cassandra.schema.KeyspaceParams;
+import org.apache.cassandra.schema.TableMetadata;
import org.apache.cassandra.utils.Throwables;
+import static org.apache.cassandra.utils.Clock.Global.nanoTime;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
import static org.junit.Assert.assertTrue;
public class CorruptedSSTablesCompactionsTest
@@ -63,6 +76,7 @@ public class CorruptedSSTablesCompactionsTest
private static final String STANDARD_STCS = "Standard_STCS";
private static final String STANDARD_LCS = "Standard_LCS";
private static final String STANDARD_UCS = "Standard_UCS";
+ private static final String STANDARD_UCS_PARALLEL = "Standard_UCS_Parallel";
private static int maxValueSize;
@After
@@ -77,6 +91,9 @@ public class CorruptedSSTablesCompactionsTest
@BeforeClass
public static void defineSchema() throws ConfigurationException
{
+ DatabaseDescriptor.daemonInitialization(); // because of all the static initialization in CFS
+ DatabaseDescriptor.setPartitionerUnsafe(Murmur3Partitioner.instance);
+
long seed = nanoTime();
//long seed = 754271160974509L; // CASSANDRA-9530: use this seed to reproduce compaction failures if reading empty rows
@@ -91,7 +108,8 @@ public class CorruptedSSTablesCompactionsTest
KeyspaceParams.simple(1),
makeTable(STANDARD_STCS).compaction(CompactionParams.stcs(Collections.emptyMap())),
makeTable(STANDARD_LCS).compaction(CompactionParams.lcs(Collections.emptyMap())),
- makeTable(STANDARD_UCS).compaction(CompactionParams.ucs(Collections.emptyMap())));
+ makeTable(STANDARD_UCS).compaction(CompactionParams.ucs(Collections.emptyMap())),
+ makeTable(STANDARD_UCS_PARALLEL).compaction(CompactionParams.ucs(new HashMap<>(ImmutableMap.of("min_sstable_size", "1KiB")))));
maxValueSize = DatabaseDescriptor.getMaxValueSize();
DatabaseDescriptor.setMaxValueSize(1024 * 1024);
@@ -140,6 +158,12 @@ public class CorruptedSSTablesCompactionsTest
testCorruptedSSTables(STANDARD_UCS);
}
+ @Test
+ public void testCorruptedSSTablesWithUnifiedCompactionStrategyParallelized() throws Exception
+ {
+ testCorruptedSSTables(STANDARD_UCS_PARALLEL);
+ }
+
public void testCorruptedSSTables(String tableName) throws Exception
{
@@ -189,30 +213,32 @@ public class CorruptedSSTablesCompactionsTest
if (currentSSTable + 1 > SSTABLES_TO_CORRUPT)
break;
- FileChannel fc = null;
-
- try
+ do
{
- int corruptionSize = 100;
- fc = new File(sstable.getFilename()).newReadWriteChannel();
- assertNotNull(fc);
- assertTrue(fc.size() > corruptionSize);
- long pos = random.nextInt((int)(fc.size() - corruptionSize));
- logger.info("Corrupting sstable {} [{}] at pos {} / {}", currentSSTable, sstable.getFilename(), pos, fc.size());
- fc.position(pos);
- // We want to write something large enough that the corruption cannot get undetected
- // (even without compression)
- byte[] corruption = new byte[corruptionSize];
- random.nextBytes(corruption);
- fc.write(ByteBuffer.wrap(corruption));
+ FileChannel fc = null;
+ try
+ {
+ int corruptionSize = 25;
+ fc = new File(sstable.getFilename()).newReadWriteChannel();
+ assertNotNull(fc);
+ assertTrue(fc.size() > corruptionSize);
+ long pos = random.nextInt((int) (fc.size() - corruptionSize));
+ logger.info("Corrupting sstable {} [{}] at pos {} / {}", currentSSTable, sstable.getFilename(), pos, fc.size());
+ fc.position(pos);
+ // We want to write something large enough that the corruption cannot get undetected
+ // (even without compression)
+ byte[] corruption = new byte[corruptionSize];
+ random.nextBytes(corruption);
+ fc.write(ByteBuffer.wrap(corruption));
+ }
+ finally
+ {
+ FileUtils.closeQuietly(fc);
+ }
if (ChunkCache.instance != null)
ChunkCache.instance.invalidateFile(sstable.getFilename());
-
- }
- finally
- {
- FileUtils.closeQuietly(fc);
}
+ while (readsWithoutError(sstable));
currentSSTable++;
}
@@ -231,12 +257,38 @@ public class CorruptedSSTablesCompactionsTest
{
// This is the expected path. The SSTable should be marked corrupted, and retrying the compaction
// should move on to the next corruption.
- Throwables.assertAnyCause(e, CorruptSSTableException.class);
+ Throwables.assertAnyCause(e, CorruptSSTableException.class, PartialLifecycleTransaction.AbortedException.class);
failures++;
}
}
cfs.truncateBlocking();
- assertEquals(SSTABLES_TO_CORRUPT, failures);
+ if (tableName != STANDARD_UCS_PARALLEL)
+ assertEquals(SSTABLES_TO_CORRUPT, failures);
+ else
+ {
+ // Since we proceed in parallel, we can mark more than one SSTable as corrupted in an iteration.
+ assertTrue(failures > 0 && failures <= SSTABLES_TO_CORRUPT);
+ }
+ }
+
+ private boolean readsWithoutError(SSTableReader sstable)
+ {
+ try
+ {
+ ISSTableScanner scanner = sstable.getScanner();
+ while (scanner.hasNext())
+ {
+ UnfilteredRowIterator iter = scanner.next();
+ while (iter.hasNext())
+ iter.next();
+ }
+ return true;
+ }
+ catch (Throwable t)
+ {
+ sstable.unmarkSuspect();
+ return false;
+ }
}
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/LeveledCompactionStrategyTest.java b/test/unit/org/apache/cassandra/db/compaction/LeveledCompactionStrategyTest.java
index d56003ae2a..a7166bad5f 100644
--- a/test/unit/org/apache/cassandra/db/compaction/LeveledCompactionStrategyTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/LeveledCompactionStrategyTest.java
@@ -593,7 +593,7 @@ public class LeveledCompactionStrategyTest
{
for (AbstractCompactionStrategy strategy : strategies)
{
- AbstractCompactionTask task = strategy.getNextBackgroundTask(0);
+ AbstractCompactionTask task = Iterables.getOnlyElement(strategy.getNextBackgroundTasks(0), null);
if (task != null)
{
try
@@ -922,7 +922,7 @@ public class LeveledCompactionStrategyTest
l0sstables.add(MockSchema.sstable(i, (i + 1) * 1024 * 1024, cfs));
try (LifecycleTransaction txn = LifecycleTransaction.offline(OperationType.COMPACTION, Iterables.concat(l0sstables, l1sstables)))
{
- Set nonExpired = Sets.difference(txn.originals(), Collections.emptySet());
+ Set nonExpired = new HashSet<>(Sets.difference(txn.originals(), Collections.emptySet()));
CompactionTask task = new LeveledCompactionTask(cfs, txn, 1, 0, 1024*1024, false);
SSTableReader lastRemoved = null;
boolean removed = true;
@@ -974,7 +974,8 @@ public class LeveledCompactionStrategyTest
for (int i = 0; i < l0sstables.size(); i++)
{
Set before = new HashSet<>(txn.originals());
- removed = task.reduceScopeForLimitedSpace(before, 0);
+ Set sources = new HashSet<>(before);
+ removed = task.reduceScopeForLimitedSpace(sources, 0);
SSTableReader removedSSTable = Sets.difference(before, txn.originals()).stream().findFirst().orElse(null);
if (removed)
{
diff --git a/test/unit/org/apache/cassandra/db/compaction/PartialCompactionsTest.java b/test/unit/org/apache/cassandra/db/compaction/PartialCompactionsTest.java
index d9fcccbaa7..1d877904a2 100644
--- a/test/unit/org/apache/cassandra/db/compaction/PartialCompactionsTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/PartialCompactionsTest.java
@@ -98,7 +98,7 @@ public class PartialCompactionsTest extends SchemaLoader
LimitableDataDirectory.setAvailableSpace(cfs, enoughSpaceForAllButTheLargestSSTable(cfs));
// when - run a compaction where all tombstones have timed out
- FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Integer.MAX_VALUE, false));
+ FBUtilities.waitOnFutures(CompactionManager.instance.submitMaximal(cfs, Integer.MAX_VALUE, false, 0));
// then - the tombstones should not be removed
assertEquals("live sstables after compaction", 2, cfs.getLiveSSTables().size());
diff --git a/test/unit/org/apache/cassandra/db/compaction/PendingRepairManagerTest.java b/test/unit/org/apache/cassandra/db/compaction/PendingRepairManagerTest.java
index d2b3693032..8c5535c1eb 100644
--- a/test/unit/org/apache/cassandra/db/compaction/PendingRepairManagerTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/PendingRepairManagerTest.java
@@ -21,6 +21,7 @@ package org.apache.cassandra.db.compaction;
import java.util.Collection;
import java.util.Collections;
+import com.google.common.collect.Iterables;
import com.google.common.collect.Lists;
import org.junit.Assert;
import org.junit.Test;
@@ -152,8 +153,8 @@ public class PendingRepairManagerTest extends AbstractPendingRepairTest
LocalSessionAccessor.finalizeUnsafe(repairID);
Assert.assertEquals(2, prm.getSessions().size());
- Assert.assertNull(prm.getNextBackgroundTask(FBUtilities.nowInSeconds()));
- AbstractCompactionTask compactionTask = prm.getNextRepairFinishedTask();
+ Assert.assertEquals(0, prm.getNextBackgroundTasks(FBUtilities.nowInSeconds()).size());
+ AbstractCompactionTask compactionTask = Iterables.getOnlyElement(prm.getNextRepairFinishedTasks());
try
{
Assert.assertNotNull(compactionTask);
@@ -171,7 +172,7 @@ public class PendingRepairManagerTest extends AbstractPendingRepairTest
public void getNextBackgroundTaskNoSessions()
{
PendingRepairManager prm = csm.getPendingRepairManagers().get(0);
- Assert.assertNull(prm.getNextBackgroundTask(FBUtilities.nowInSeconds()));
+ Assert.assertEquals(0, prm.getNextBackgroundTasks(FBUtilities.nowInSeconds()).size());
}
/**
@@ -191,7 +192,7 @@ public class PendingRepairManagerTest extends AbstractPendingRepairTest
Assert.assertNotNull(prm.get(repairID));
LocalSessionAccessor.finalizeUnsafe(repairID);
- Assert.assertNull(prm.getNextBackgroundTask(FBUtilities.nowInSeconds()));
+ Assert.assertEquals(0, prm.getNextBackgroundTasks(FBUtilities.nowInSeconds()).size());
}
@@ -303,7 +304,7 @@ public class PendingRepairManagerTest extends AbstractPendingRepairTest
prm.getOrCreate(sstable);
cfs.truncateBlocking();
Assert.assertFalse(cfs.getSSTables(SSTableSet.LIVE).iterator().hasNext());
- Assert.assertNull(cfs.getCompactionStrategyManager().getNextBackgroundTask(0));
+ Assert.assertEquals(0, cfs.getCompactionStrategyManager().getNextBackgroundTasks(0).size());
}
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/ShardManagerTest.java b/test/unit/org/apache/cassandra/db/compaction/ShardManagerTest.java
index 71e5d8a064..ced5d78821 100644
--- a/test/unit/org/apache/cassandra/db/compaction/ShardManagerTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/ShardManagerTest.java
@@ -21,10 +21,13 @@ package org.apache.cassandra.db.compaction;
import java.nio.ByteBuffer;
import java.util.ArrayList;
import java.util.Arrays;
+import java.util.HashSet;
import java.util.List;
+import java.util.Set;
import java.util.stream.Collectors;
import com.google.common.collect.ImmutableList;
+import org.apache.cassandra.dht.*;
import org.junit.Assert;
import org.junit.Before;
import org.junit.Test;
@@ -37,12 +40,8 @@ import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.DiskBoundaries;
import org.apache.cassandra.db.PartitionPosition;
-import org.apache.cassandra.dht.IPartitioner;
-import org.apache.cassandra.dht.Murmur3Partitioner;
-import org.apache.cassandra.dht.Range;
-import org.apache.cassandra.dht.Splitter;
-import org.apache.cassandra.dht.Token;
import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.Pair;
import org.mockito.Mockito;
import static org.apache.cassandra.db.ColumnFamilyStore.RING_VERSION_IRRELEVANT;
@@ -409,4 +408,149 @@ public class ShardManagerTest
}
}
}
+
+
+ @Test
+ public void testSplitSSTablesInRanges()
+ {
+ testSplitSSTablesInRanges(8, ints(1, 2, 4));
+ testSplitSSTablesInRanges(4, ints(1, 2, 4));
+ testSplitSSTablesInRanges(2, ints(1, 2, 4));
+ testSplitSSTablesInRanges(5, ints(1, 2, 4));
+ testSplitSSTablesInRanges(5, ints(2, 4, 8));
+ testSplitSSTablesInRanges(3, ints(1, 3, 5));
+ testSplitSSTablesInRanges(3, ints(3, 3, 3));
+
+ testSplitSSTablesInRanges(1, ints(1, 2, 3));
+
+ testSplitSSTablesInRanges(3, ints());
+ }
+
+ @Test
+ public void testSplitSSTablesInRangesMissingParts()
+ {
+ // Drop some sstables without losing ranges
+ testSplitSSTablesInRanges(8, ints(2, 4, 8),
+ ints(1));
+
+ testSplitSSTablesInRanges(8, ints(2, 4, 8),
+ ints(1), ints(0), ints(2, 7));
+
+ testSplitSSTablesInRanges(5, ints(2, 4, 8),
+ ints(1), ints(0), ints(2, 7));
+ }
+
+ @Test
+ public void testSplitSSTablesInRangesOneRange()
+ {
+ // Drop second half
+ testSplitSSTablesInRanges(2, ints(2, 4, 8),
+ ints(1), ints(2, 3), ints(4, 5, 6, 7));
+ // Drop all except center, within shard
+ testSplitSSTablesInRanges(3, ints(5, 7, 9),
+ ints(0, 1, 3, 4), ints(0, 1, 2, 4, 5, 6), ints(0, 1, 2, 6, 7, 8));
+ }
+
+ @Test
+ public void testSplitSSTablesInRangesSkippedRange()
+ {
+ // Drop all sstables containing the 4/8-5/8 range.
+ testSplitSSTablesInRanges(8, ints(2, 4, 8),
+ ints(1), ints(2), ints(4));
+ // Drop all sstables containing the 4/8-6/8 range.
+ testSplitSSTablesInRanges(8, ints(2, 4, 8),
+ ints(1), ints(2), ints(4, 5));
+ // Drop all sstables containing the 4/8-8/8 range.
+ testSplitSSTablesInRanges(8, ints(2, 4, 8),
+ ints(1), ints(2, 3), ints(4, 5, 6, 7));
+
+ // Drop all sstables containing the 0/8-2/8 range.
+ testSplitSSTablesInRanges(5, ints(2, 4, 8),
+ ints(0), ints(0), ints(0, 1));
+ // Drop all sstables containing the 6/8-8/8 range.
+ testSplitSSTablesInRanges(5, ints(2, 4, 8),
+ ints(1), ints(3), ints(6, 7));
+ // Drop sstables on both ends.
+ testSplitSSTablesInRanges(5, ints(3, 4, 8),
+ ints(0, 2), ints(0, 3), ints(0, 1, 6, 7));
+ }
+
+ public void testSplitSSTablesInRanges(int numShards, int[] perLevelCounts, int[]... dropsPerLevel)
+ {
+ weightedRanges.clear();
+ weightedRanges.add(new Splitter.WeightedRange(1.0, new Range<>(minimumToken, minimumToken)));
+ ShardManager manager = new ShardManagerNoDisks(weightedRanges);
+
+ Set allSSTables = new HashSet<>();
+ int levelNum = 0;
+ for (int perLevelCount : perLevelCounts)
+ {
+ List ssTables = mockNonOverlappingSSTables(perLevelCount);
+ if (levelNum < dropsPerLevel.length)
+ {
+ for (int i = dropsPerLevel[levelNum].length - 1; i >= 0; i--)
+ ssTables.remove(dropsPerLevel[levelNum][i]);
+ }
+ allSSTables.addAll(ssTables);
+ ++levelNum;
+ }
+
+ var results = new ArrayList, Set>>();
+ manager.splitSSTablesInShards(allSSTables, numShards, (sstables, range) -> results.add(Pair.create(range, Set.copyOf(sstables))));
+ int i = 0;
+ int[] expectedSSTablesInTasks = new int[results.size()];
+ int[] collectedSSTablesPerTask = new int[results.size()];
+ for (var t : results)
+ {
+ collectedSSTablesPerTask[i] = t.right().size();
+ expectedSSTablesInTasks[i] = (int) allSSTables.stream().filter(x -> intersects(x, t.left())).count();
+ ++i;
+ }
+ Assert.assertEquals(Arrays.toString(expectedSSTablesInTasks), Arrays.toString(collectedSSTablesPerTask));
+ System.out.println(Arrays.toString(expectedSSTablesInTasks));
+ }
+
+ private boolean intersects(SSTableReader r, Range range)
+ {
+ if (range == null)
+ return true;
+ return range.intersects(range(r));
+ }
+
+
+ private Bounds range(SSTableReader x)
+ {
+ return new Bounds<>(x.getFirst().getToken(), x.getLast().getToken());
+ }
+
+ List mockNonOverlappingSSTables(int numSSTables)
+ {
+ if (!partitioner.splitter().isPresent())
+ throw new IllegalStateException(String.format("Cannot split ranges with current partitioner %s", partitioner));
+
+ ByteBuffer emptyBuffer = ByteBuffer.allocate(0);
+
+ List sstables = new ArrayList<>(numSSTables);
+ for (int i = 0; i < numSSTables; i++)
+ {
+ DecoratedKey first = new BufferDecoratedKey(boundary(numSSTables, i).nextValidToken(), emptyBuffer);
+ DecoratedKey last = new BufferDecoratedKey(boundary(numSSTables, i+1), emptyBuffer);
+ sstables.add(mockSSTable(first, last));
+ }
+
+ return sstables;
+ }
+
+ private Token boundary(int numSSTables, int i)
+ {
+ return partitioner.split(partitioner.getMinimumToken(), partitioner.getMaximumToken(), i * 1.0 / numSSTables);
+ }
+
+ private SSTableReader mockSSTable(DecoratedKey first, DecoratedKey last)
+ {
+ SSTableReader sstable = Mockito.mock(SSTableReader.class);
+ when(sstable.getFirst()).thenReturn(first);
+ when(sstable.getLast()).thenReturn(last);
+ return sstable;
+ }
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/SingleSSTableLCSTaskTest.java b/test/unit/org/apache/cassandra/db/compaction/SingleSSTableLCSTaskTest.java
index 9e03ec3a9a..e14d48eae0 100644
--- a/test/unit/org/apache/cassandra/db/compaction/SingleSSTableLCSTaskTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/SingleSSTableLCSTaskTest.java
@@ -22,6 +22,7 @@ import java.nio.ByteBuffer;
import java.nio.channels.FileChannel;
import java.util.Random;
+import com.google.common.collect.Iterables;
import org.apache.commons.lang3.StringUtils;
import org.junit.Test;
@@ -101,14 +102,14 @@ public class SingleSSTableLCSTaskTest extends CQLTester
}
// now we have a bunch of data in L0, first compaction will be a normal one, containing all sstables:
LeveledCompactionStrategy lcs = (LeveledCompactionStrategy) cfs.getCompactionStrategyManager().getUnrepairedUnsafe().first();
- AbstractCompactionTask act = lcs.getNextBackgroundTask(0);
+ AbstractCompactionTask act = Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null);
act.execute(ActiveCompactionsTracker.NOOP);
// now all sstables are laid out non-overlapping in L1, this means that the rest of the compactions
// will be single sstable ones, make sure that we use SingleSSTableLCSTask if singleSSTUplevel is true:
while (lcs.getEstimatedRemainingTasks() > 0)
{
- act = lcs.getNextBackgroundTask(0);
+ act = Iterables.getOnlyElement(lcs.getNextBackgroundTasks(0), null);
assertEquals(singleSSTUplevel, act instanceof SingleSSTableLCSTask);
act.execute(ActiveCompactionsTracker.NOOP);
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/TTLExpiryTest.java b/test/unit/org/apache/cassandra/db/compaction/TTLExpiryTest.java
index e7e97bed99..bf3c508af4 100644
--- a/test/unit/org/apache/cassandra/db/compaction/TTLExpiryTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/TTLExpiryTest.java
@@ -141,7 +141,7 @@ public class TTLExpiryTest
Set expired = CompactionController.getFullyExpiredSSTables(
cfs,
sstables,
- Collections.EMPTY_SET,
+ s -> Collections.EMPTY_SET,
gcBefore);
assertEquals(2, expired.size());
diff --git a/test/unit/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategyTest.java b/test/unit/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategyTest.java
index 4576db78f7..5387732a71 100644
--- a/test/unit/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategyTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/TimeWindowCompactionStrategyTest.java
@@ -297,11 +297,13 @@ public class TimeWindowCompactionStrategyTest extends SchemaLoader
twcs.addSSTable(sstable);
twcs.startup();
- assertNull(twcs.getNextBackgroundTask(nowInSeconds()));
+ long gcBefore1 = nowInSeconds();
+ assertNull(Iterables.getOnlyElement(twcs.getNextBackgroundTasks(gcBefore1), null));
// Wait for the expiration of the first sstable
Thread.sleep(TimeUnit.SECONDS.toMillis(TTL_SECONDS + 1));
- AbstractCompactionTask t = twcs.getNextBackgroundTask(nowInSeconds());
+ long gcBefore = nowInSeconds();
+ AbstractCompactionTask t = Iterables.getOnlyElement(twcs.getNextBackgroundTasks(gcBefore), null);
assertNotNull(t);
assertEquals(1, Iterables.size(t.transaction.originals()));
SSTableReader sstable = t.transaction.originals().iterator().next();
@@ -352,11 +354,13 @@ public class TimeWindowCompactionStrategyTest extends SchemaLoader
twcs.addSSTable(sstable);
twcs.startup();
- assertNull(twcs.getNextBackgroundTask(nowInSeconds()));
+ long gcBefore2 = nowInSeconds();
+ assertNull(Iterables.getOnlyElement(twcs.getNextBackgroundTasks(gcBefore2), null));
// Wait for the expiration of the first sstable
Thread.sleep(TimeUnit.SECONDS.toMillis(TTL_SECONDS + 1));
- assertNull(twcs.getNextBackgroundTask(nowInSeconds()));
+ long gcBefore1 = nowInSeconds();
+ assertNull(Iterables.getOnlyElement(twcs.getNextBackgroundTasks(gcBefore1), null));
options.put(TimeWindowCompactionStrategyOptions.UNSAFE_AGGRESSIVE_SSTABLE_EXPIRATION_KEY, "true");
twcs = new TimeWindowCompactionStrategy(cfs, options);
@@ -364,7 +368,8 @@ public class TimeWindowCompactionStrategyTest extends SchemaLoader
twcs.addSSTable(sstable);
twcs.startup();
- AbstractCompactionTask t = twcs.getNextBackgroundTask(nowInSeconds());
+ long gcBefore = nowInSeconds();
+ AbstractCompactionTask t = Iterables.getOnlyElement(twcs.getNextBackgroundTasks(gcBefore), null);
assertNotNull(t);
assertEquals(1, Iterables.size(t.transaction.originals()));
SSTableReader sstable = t.transaction.originals().iterator().next();
diff --git a/test/unit/org/apache/cassandra/db/compaction/UnifiedCompactionStrategyTest.java b/test/unit/org/apache/cassandra/db/compaction/UnifiedCompactionStrategyTest.java
index 7b3d85d0b7..8bca5e8f80 100644
--- a/test/unit/org/apache/cassandra/db/compaction/UnifiedCompactionStrategyTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/UnifiedCompactionStrategyTest.java
@@ -20,7 +20,7 @@ package org.apache.cassandra.db.compaction;
import java.nio.ByteBuffer;
import java.util.ArrayList;
-import java.util.Collection;
+import java.util.Arrays;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
@@ -29,9 +29,11 @@ import java.util.Random;
import java.util.Set;
import java.util.TreeMap;
import java.util.concurrent.TimeUnit;
+import java.util.stream.Collectors;
import com.google.common.collect.Iterables;
import org.apache.commons.math3.random.JDKRandomGenerator;
+import org.junit.Assert;
import org.junit.Before;
import org.junit.BeforeClass;
import org.junit.Test;
@@ -45,11 +47,15 @@ import org.apache.cassandra.db.Directories;
import org.apache.cassandra.db.DiskBoundaries;
import org.apache.cassandra.db.compaction.unified.Controller;
import org.apache.cassandra.db.compaction.unified.UnifiedCompactionTask;
+import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.PartialLifecycleTransaction;
import org.apache.cassandra.db.lifecycle.SSTableSet;
import org.apache.cassandra.db.lifecycle.Tracker;
import org.apache.cassandra.db.marshal.AsciiType;
+import org.apache.cassandra.dht.Bounds;
import org.apache.cassandra.dht.IPartitioner;
import org.apache.cassandra.dht.Murmur3Partitioner;
+import org.apache.cassandra.dht.Range;
import org.apache.cassandra.dht.Splitter;
import org.apache.cassandra.dht.Token;
import org.apache.cassandra.io.sstable.format.SSTableReader;
@@ -58,12 +64,15 @@ import org.apache.cassandra.tcm.ClusterMetadata;
import org.apache.cassandra.utils.FBUtilities;
import org.apache.cassandra.utils.Overlaps;
import org.apache.cassandra.utils.Pair;
+import org.apache.cassandra.utils.concurrent.Transactional;
import org.mockito.Answers;
import org.mockito.Mock;
import org.mockito.Mockito;
import org.mockito.MockitoAnnotations;
+import static org.assertj.core.api.Assertions.assertThat;
import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertFalse;
import static org.junit.Assert.assertNotNull;
import static org.junit.Assert.assertNull;
import static org.junit.Assert.assertSame;
@@ -175,7 +184,8 @@ public class UnifiedCompactionStrategyTest
UnifiedCompactionStrategy strategy = new UnifiedCompactionStrategy(cfs, new HashMap<>(), controller);
- assertNull(strategy.getNextBackgroundTask(FBUtilities.nowInSeconds()));
+ long gcBefore = FBUtilities.nowInSeconds();
+ assertNull(Iterables.getOnlyElement(strategy.getNextBackgroundTasks(gcBefore), null));
assertEquals(0, strategy.getEstimatedRemainingTasks());
}
@@ -474,7 +484,7 @@ public class UnifiedCompactionStrategyTest
strategy.addSSTables(sstables);
dataTracker.addInitialSSTables(sstables);
- AbstractCompactionTask task = strategy.getNextBackgroundTask(0);
+ AbstractCompactionTask task = Iterables.getOnlyElement(strategy.getNextBackgroundTasks(0), null);
assertSame(UnifiedCompactionTask.class, task.getClass());
task.transaction.abort();
}
@@ -656,38 +666,228 @@ public class UnifiedCompactionStrategyTest
}
@Test
- public void testMaximalSelection()
+ public void testCreateParallelTasks()
{
+ testCreateParallelTasks(8, arr(1, 2, 4));
+ testCreateParallelTasks(4, arr(1, 2, 4));
+ testCreateParallelTasks(2, arr(1, 2, 4));
+ testCreateParallelTasks(5, arr(1, 2, 4));
+ testCreateParallelTasks(5, arr(2, 4, 8));
+ testCreateParallelTasks(3, arr(1, 3, 5));
+ testCreateParallelTasks(3, arr(3, 3, 3));
+
+ testCreateParallelTasks(1, arr(1, 2, 3));
+ }
+
+ @Test
+ public void testCreateParallelTasksMissingParts()
+ {
+ // Drop some sstables without losing ranges
+ testCreateParallelTasks(8, arr(2, 4, 8),
+ arr(1));
+
+ testCreateParallelTasks(8, arr(2, 4, 8),
+ arr(1), arr(0), arr(2, 7));
+
+ testCreateParallelTasks(5, arr(2, 4, 8),
+ arr(1), arr(0), arr(2, 7));
+ }
+
+ @Test
+ public void testCreateParallelTasksOneRange()
+ {
+ // Drop second half
+ testCreateParallelTasks(2, arr(2, 4, 8),
+ arr(1), arr(2, 3), arr(4, 5, 6, 7));
+ // Drop all except center, within shard
+ testCreateParallelTasks(3, arr(5, 7, 9),
+ arr(0, 1, 3, 4), arr(0, 1, 2, 4, 5, 6), arr(0, 1, 2, 6, 7, 8));
+ }
+
+ @Test
+ public void testCreateParallelTasksSkippedRange()
+ {
+ // Drop all sstables containing the 4/8-5/8 range.
+ testCreateParallelTasks(8, arr(2, 4, 8),
+ arr(1), arr(2), arr(4));
+ // Drop all sstables containing the 4/8-6/8 range.
+ testCreateParallelTasks(8, arr(2, 4, 8),
+ arr(1), arr(2), arr(4, 5));
+ // Drop all sstables containing the 4/8-8/8 range.
+ testCreateParallelTasks(8, arr(2, 4, 8),
+ arr(1), arr(2, 3), arr(4, 5, 6, 7));
+
+ // Drop all sstables containing the 0/8-2/8 range.
+ testCreateParallelTasks(5, arr(2, 4, 8),
+ arr(0), arr(0), arr(0, 1));
+ // Drop all sstables containing the 6/8-8/8 range.
+ testCreateParallelTasks(5, arr(2, 4, 8),
+ arr(1), arr(3), arr(6, 7));
+ // Drop sstables on both ends.
+ testCreateParallelTasks(5, arr(3, 4, 8),
+ arr(0, 2), arr(0, 3), arr(0, 1, 6, 7));
+ }
+
+ public void testCreateParallelTasks(int numShards, int[] perLevelCounts, int[]... dropsPerLevel)
+ {
+ // Note: This test has a counterpart in ShardManagerTest that exercises splitSSTablesInShards directly and more thoroughly.
+ // This one ensures the data is correctly passed to and presented in compaction tasks.
+ Set allSSTables = new HashSet<>();
+ int levelNum = 0;
+ for (int perLevelCount : perLevelCounts)
+ {
+ List ssTables = mockNonOverlappingSSTables(perLevelCount, levelNum, 100 << (20 + levelNum));
+ if (levelNum < dropsPerLevel.length)
+ {
+ for (int i = dropsPerLevel[levelNum].length - 1; i >= 0; i--)
+ ssTables.remove(dropsPerLevel[levelNum][i]);
+ }
+ allSSTables.addAll(ssTables);
+ ++levelNum;
+ }
+ dataTracker.addInitialSSTables(allSSTables);
+
+ Controller controller = Mockito.mock(Controller.class);
+ when(controller.getNumShards(anyDouble())).thenReturn(numShards);
+ when(controller.parallelizeOutputShards()).thenReturn(true);
+ UnifiedCompactionStrategy strategy = new UnifiedCompactionStrategy(cfs, new HashMap<>(), controller);
+ strategy.startup();
+ LifecycleTransaction txn = dataTracker.tryModify(allSSTables, OperationType.COMPACTION);
+ var tasks = strategy.createCompactionTasks(0, txn);
+ int i = 0;
+ int[] expectedSSTablesInTasks = new int[tasks.size()];
+ int[] collectedSSTablesPerTask = new int[tasks.size()];
+ for (AbstractCompactionTask act : tasks)
+ {
+ CompactionTask t = (CompactionTask) act;
+ assertTrue(t instanceof UnifiedCompactionTask);
+ assertFalse(t.inputSSTables().isEmpty());
+ collectedSSTablesPerTask[i] = t.inputSSTables().size();
+ expectedSSTablesInTasks[i] = (int) allSSTables.stream().filter(x -> intersects(x, t.tokenRange())).count();
+ t.rejected(); // close transaction
+ ++i;
+ }
+ if (tasks.size() == 1)
+ assertNull(((CompactionTask) tasks.get(0)).tokenRange()); // make sure single-task compactions are not ranged
+ Assert.assertEquals(Arrays.toString(expectedSSTablesInTasks), Arrays.toString(collectedSSTablesPerTask));
+ System.out.println(Arrays.toString(expectedSSTablesInTasks));
+ assertThat(tasks.size()).isLessThanOrEqualTo(numShards);
+ assertEquals(allSSTables, tasks.stream().flatMap(t -> ((CompactionTask) t).inputSSTables().stream()).collect(Collectors.toSet()));
+ for (var t : tasks)
+ for (var q : tasks)
+ if (t != q)
+ assertFalse("Subranges " + ((CompactionTask)t).tokenRange() + " and " + ((CompactionTask)q).tokenRange() + "intersect", ((CompactionTask)t).tokenRange().intersects(((CompactionTask)q).tokenRange()));
+
+ // make sure the composite transaction has the correct number of tasks
+ assertEquals(Transactional.AbstractTransactional.State.ABORTED, txn.state());
+ }
+
+ private boolean intersects(SSTableReader r, Range range)
+ {
+ if (range == null)
+ return true;
+ return range.intersects(range(r));
+ }
+
+
+ private Bounds range(SSTableReader x)
+ {
+ return new Bounds<>(x.getFirst().getToken(), x.getLast().getToken());
+ }
+
+ @Test
+ public void testDontCreateParallelTasks()
+ {
+ int numShards = 5;
Set allSSTables = new HashSet<>();
allSSTables.addAll(mockNonOverlappingSSTables(10, 0, 100 << 20));
allSSTables.addAll(mockNonOverlappingSSTables(15, 1, 200 << 20));
allSSTables.addAll(mockNonOverlappingSSTables(25, 2, 400 << 20));
-
+ dataTracker.addInitialSSTables(allSSTables);
Controller controller = Mockito.mock(Controller.class);
+ when(controller.getNumShards(anyDouble())).thenReturn(numShards);
+ when(controller.parallelizeOutputShards()).thenReturn(false);
UnifiedCompactionStrategy strategy = new UnifiedCompactionStrategy(cfs, new HashMap<>(), controller);
- strategy.addSSTables(allSSTables);
+ strategy.startup();
+ LifecycleTransaction txn = dataTracker.tryModify(allSSTables, OperationType.COMPACTION);
+ var tasks = strategy.createCompactionTasks(0, txn);
+ assertEquals(1, tasks.size());
+ assertEquals(allSSTables, ((CompactionTask) tasks.get(0)).inputSSTables());
+ }
+ @Test
+ public void testMaximalSelection()
+ {
+ // shared transaction, all tasks refer to the same input sstables
+ testMaximalSelection(1, 1, 0, false, 12 + 18 + 30, ((12 * 100L + 18 * 200 + 30 * 400) << 20));
+ testMaximalSelection(5, 5, 0, true, 12 + 18 + 30, ((12 * 100L + 18 * 200 + 30 * 400) << 20));
+ // when there's a common split point of existing and new sharding (i.e. gcd(num_shards,12,18,30) > 1), it should be used
+ testMaximalSelection(3, 3, 0, false, 4 + 6 + 10, ((4 * 100L + 6 * 200 + 10 * 400) << 20));
+ testMaximalSelection(9, 3, 0, false, 4 + 6 + 10, ((4 * 100L + 6 * 200 + 10 * 400) << 20));
+ testMaximalSelection(9, 9, 0, true, 4 + 6 + 10, ((4 * 100L + 6 * 200 + 10 * 400) << 20));
+ testMaximalSelection(2, 2, 0, false, 6 + 9 + 15, ((6 * 100L + 9 * 200 + 15 * 400) << 20));
+ testMaximalSelection(4, 2, 0, false, 6 + 9 + 15, ((6 * 100L + 9 * 200 + 15 * 400) << 20));
+ testMaximalSelection(4, 4, 0, true, 6 + 9 + 15, ((6 * 100L + 9 * 200 + 15 * 400) << 20));
+ testMaximalSelection(18, 6, 0, false, 2 + 3 + 5, ((2 * 100L + 3 * 200 + 5 * 400) << 20));
+ testMaximalSelection(18, 18, 0, true, 2 + 3 + 5, ((2 * 100L + 3 * 200 + 5 * 400) << 20));
+ }
+
+ @Test
+ public void testMaximalSelectionWithLimit()
+ {
+ // shared transaction, all tasks refer to the same input sstables
+ testMaximalSelection(5, 5, 2, true, 12 + 18 + 30, ((12 * 100L + 18 * 200 + 30 * 400) << 20));
+ // when there's a common split point of existing and new sharding (i.e. gcd(num_shards,12,18,30) > 1), it should be used
+ testMaximalSelection(3, 3, 2, false, 4 + 6 + 10, ((4 * 100L + 6 * 200 + 10 * 400) << 20));
+ testMaximalSelection(9, 3, 1, false, 4 + 6 + 10, ((4 * 100L + 6 * 200 + 10 * 400) << 20));
+ testMaximalSelection(9, 9, 3, true, 4 + 6 + 10, ((4 * 100L + 6 * 200 + 10 * 400) << 20));
+ testMaximalSelection(18, 6, 4, false, 2 + 3 + 5, ((2 * 100L + 3 * 200 + 5 * 400) << 20));
+ testMaximalSelection(18, 18, 5, true, 2 + 3 + 5, ((2 * 100L + 3 * 200 + 5 * 400) << 20));
+ }
+
+ private void testMaximalSelection(int numShards, int expectedTaskCount, int parallelismLimit, boolean parallelize, int originalsCount, long onDiskLength)
+ {
+ if (parallelismLimit == 0)
+ parallelismLimit = Integer.MAX_VALUE;
+ Set allSSTables = new HashSet<>();
+ allSSTables.addAll(mockNonOverlappingSSTables(12, 0, 100 << 20));
+ allSSTables.addAll(mockNonOverlappingSSTables(18, 1, 200 << 20));
+ allSSTables.addAll(mockNonOverlappingSSTables(30, 2, 400 << 20));
dataTracker.addInitialSSTables(allSSTables);
- Collection tasks = strategy.getMaximalTask(0, false);
- assertEquals(5, tasks.size()); // 5 (gcd of 10,15,25) common boundaries
- for (AbstractCompactionTask task : tasks)
+ Controller controller = Mockito.mock(Controller.class);
+ when(controller.getNumShards(anyDouble())).thenReturn(numShards);
+ when(controller.parallelizeOutputShards()).thenReturn(parallelize);
+ when(controller.maxConcurrentCompactions()).thenReturn(1000);
+ UnifiedCompactionStrategy strategy = new UnifiedCompactionStrategy(cfs, new HashMap<>(), controller);
+ strategy.addSSTables(allSSTables);
+
+ List allTasks = strategy.getMaximalTasks(0, false);
+ List limitedParallelismTasks = CompositeCompactionTask.applyParallelismLimit(allTasks, parallelismLimit);
+ assertEquals(expectedTaskCount, allTasks.size());
+ for (AbstractCompactionTask task : allTasks)
{
Set compacting = task.transaction.originals();
- assertEquals(2 + 3 + 5, compacting.size()); // count / gcd sstables of each level
- assertEquals((2 * 100L + 3 * 200 + 5 * 400) << 20, compacting.stream().mapToLong(SSTableReader::onDiskLength).sum());
+ assertEquals(originalsCount, compacting.size()); // count / gcd sstables of each level
+ assertEquals(onDiskLength, compacting.stream().mapToLong(SSTableReader::onDiskLength).sum());
- // None of the selected sstables may intersect any in any other set.
- for (AbstractCompactionTask task2 : tasks)
+ if (!(task.transaction instanceof PartialLifecycleTransaction))
{
- if (task == task2)
- continue;
+ // None of the selected sstables may intersect any in any other set.
+ for (AbstractCompactionTask task2 : allTasks)
+ {
+ if (task == task2)
+ continue;
- Set compacting2 = task2.transaction.originals();
- for (SSTableReader r1 : compacting)
- for (SSTableReader r2 : compacting2)
- assertTrue(r1 + " intersects " + r2, r1.getFirst().compareTo(r2.getLast()) > 0 || r1.getLast().compareTo(r2.getFirst()) < 0);
+ Set compacting2 = task2.transaction.originals();
+ for (SSTableReader r1 : compacting)
+ for (SSTableReader r2 : compacting2)
+ assertTrue(r1 + " intersects " + r2, r1.getFirst().compareTo(r2.getLast()) > 0 || r1.getLast().compareTo(r2.getFirst()) < 0);
+ }
}
}
+
+ if (parallelismLimit > 0)
+ assertTrue(limitedParallelismTasks.size() <= parallelismLimit);
}
@Test
@@ -899,8 +1099,8 @@ public class UnifiedCompactionStrategyTest
List sstables = new ArrayList<>(numSSTables);
for (int i = 0; i < numSSTables; i++)
{
- DecoratedKey first = new BufferDecoratedKey(boundary(numSSTables, i).nextValidToken(), emptyBuffer);
- DecoratedKey last = new BufferDecoratedKey(boundary(numSSTables, i+1), emptyBuffer);
+ DecoratedKey first = new BufferDecoratedKey(boundary(numSSTables, i + 0.0001), emptyBuffer);
+ DecoratedKey last = new BufferDecoratedKey(boundary(numSSTables, i + 0.9999), emptyBuffer);
sstables.add(mockSSTable(level, bytesOnDisk, timestamp, 0., first, last));
timestamp+=10;
@@ -909,8 +1109,9 @@ public class UnifiedCompactionStrategyTest
return sstables;
}
- private Token boundary(int numSSTables, int i)
+ private Token boundary(int numSSTables, double i)
{
- return partitioner.split(partitioner.getMinimumToken(), partitioner.getMaximumToken(), i * 1.0 / numSSTables);
+ return partitioner.split(partitioner.getMinimumToken(), partitioner.getMaximumToken(), i / numSSTables);
}
+
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/unified/BackgroundCompactionTrackingTest.java b/test/unit/org/apache/cassandra/db/compaction/unified/BackgroundCompactionTrackingTest.java
new file mode 100644
index 0000000000..212182e2b8
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/unified/BackgroundCompactionTrackingTest.java
@@ -0,0 +1,194 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.unified;
+
+import java.nio.ByteBuffer;
+import java.util.ArrayList;
+import java.util.BitSet;
+import java.util.HashSet;
+import java.util.List;
+import java.util.Random;
+import java.util.Set;
+import java.util.stream.Collectors;
+
+import org.junit.After;
+import org.junit.Assert;
+import org.junit.Before;
+import org.junit.Test;
+import org.junit.runner.RunWith;
+
+import org.apache.cassandra.cql3.CQLTester;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.compaction.CompactionInfo;
+import org.apache.cassandra.db.compaction.CompactionManager;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.TimeUUID;
+import org.jboss.byteman.contrib.bmunit.BMRule;
+import org.jboss.byteman.contrib.bmunit.BMUnitConfig;
+import org.jboss.byteman.contrib.bmunit.BMUnitRunner;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+@RunWith(BMUnitRunner.class)
+@BMUnitConfig(debug=true)
+@BMRule(
+name = "Get stats before task completion",
+targetClass = "org.apache.cassandra.db.compaction.ActiveCompactions",
+targetMethod = "finishCompaction",
+targetLocation = "AT ENTRY",
+action = "org.apache.cassandra.db.compaction.unified.BackgroundCompactionTrackingTest.getStats()"
+)
+public class BackgroundCompactionTrackingTest extends CQLTester
+{
+ // Get rid of commitlog noise
+ @Before
+ public void disableCommitlog()
+ {
+ schemaChange("ALTER KEYSPACE " + KEYSPACE + " WITH durable_writes = false");
+ }
+ @After
+ public void enableCommitlog()
+ {
+ schemaChange("ALTER KEYSPACE " + KEYSPACE + " WITH durable_writes = true");
+ }
+
+ @Test
+ public void testBackgroundCompactionTracking()
+ {
+ testBackgroundCompactionTracking(false, 3);
+ }
+
+ @Test
+ public void testBackgroundCompactionTrackingParallelized()
+ {
+ testBackgroundCompactionTracking(true, 3);
+ }
+
+ public void testBackgroundCompactionTracking(boolean parallelize, int shards)
+ {
+ int bytes_per_row = 5000;
+ int partitions = 5000;
+ int rows_per_partition = 10;
+ long targetSize = 1L * partitions * rows_per_partition * bytes_per_row / shards;
+ CompactionManager.instance.setMaximumCompactorThreads(50);
+ CompactionManager.instance.setCoreCompactorThreads(50);
+ String table = createTable(String.format("CREATE TABLE %%s (k int, t int, v blob, PRIMARY KEY (k, t))" +
+ " with compression = {'enabled': false} " +
+ " and compaction = {" +
+ "'class': 'UnifiedCompactionStrategy', " +
+ "'parallelize_output_shards': '%s', " +
+ "'base_shard_count': %d, " +
+ "'min_sstable_size': '%dB', " + // to disable sharding on flush
+ "'scaling_parameters': 'T4, T7'" +
+ "}",
+ parallelize, shards, targetSize));
+ ColumnFamilyStore cfs = getColumnFamilyStore(KEYSPACE, table);
+ cfs.disableAutoCompaction();
+
+ for (int iter = 1; iter <= 5; ++iter)
+ {
+ byte [] payload = new byte[bytes_per_row];
+ new Random(42).nextBytes(payload);
+ ByteBuffer b = ByteBuffer.wrap(payload);
+ Set before = new HashSet<>(cfs.getLiveSSTables());
+
+ for (int i = 0; i < partitions; i++)
+ {
+ for (int j = 0; j < rows_per_partition; j++)
+ execute(String.format("INSERT INTO %s.%s(k, t, v) VALUES (?, ?, ?)", KEYSPACE, table), i, j, b);
+
+ if ((i + 1) % ((partitions + 3) / 4) == 0)
+ cfs.forceBlockingFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS);
+ }
+ cfs.forceBlockingFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS);
+ operations = new ArrayList<>();
+ Set newSSTables = new HashSet<>(cfs.getLiveSSTables());
+ newSSTables.removeAll(before);
+ long uncompressedSize = newSSTables.stream().mapToLong(SSTableReader::uncompressedLength).sum();
+
+ cfs.enableAutoCompaction(true); // since the trigger is hit, this initiates an L0 compaction
+ cfs.disableAutoCompaction();
+
+ // Check that the background compactions state is correct during the compaction
+ Assert.assertTrue("Byteman rule did not fire", !operations.isEmpty());
+ printStats();
+ int tasks = parallelize ? shards : 1;
+ assertEquals(tasks, operations.size());
+ TimeUUID mainOpId = null;
+ for (int i = 0; i < operations.size(); ++i)
+ {
+ BitSet seqs = new BitSet(shards);
+ int expectedSize = tasks - i;
+ final List ops = getCompactionOps(i, cfs);
+ final int size = ops.size();
+ int finished = tasks - size;
+ assertTrue(size >= expectedSize); // some task may have not managed to close
+ for (var op : ops)
+ {
+ final TimeUUID opIdSeq0 = op.getTaskId().withSequence(0);
+ if (mainOpId == null)
+ mainOpId = opIdSeq0;
+ else
+ assertEquals(mainOpId, opIdSeq0);
+ seqs.set(op.getTaskId().sequence());
+
+ if (i == 0)
+ Assert.assertEquals(uncompressedSize * 1.0 / tasks, op.getTotal(), uncompressedSize * 0.03);
+ assertTrue(op.getCompleted() <= op.getTotal() * 1.03);
+ if (op.getCompleted() >= op.getTotal() * 0.97)
+ ++finished;
+ }
+ assertTrue(finished > i);
+ assertEquals(size, seqs.cardinality());
+ }
+ assertEquals(iter * shards, cfs.getLiveSSTables().size());
+
+ // Check that the background compactions state is correct after the compaction
+ operations.clear();
+ getStats();
+ printStats();
+ final List ops = getCompactionOps(0, cfs);
+ assertEquals(0, ops.size());
+ }
+ }
+
+ private static List getCompactionOps(int i, ColumnFamilyStore cfs)
+ {
+ return operations.get(i)
+ .stream()
+ .filter(op -> op.getTableMetadata() == cfs.metadata())
+ .collect(Collectors.toList());
+ }
+
+ private void printStats()
+ {
+ for (int i = 0; i < operations.size(); ++i)
+ {
+ System.out.println(operations.get(i).stream().map(Object::toString).collect(Collectors.joining("\n")));
+ }
+ }
+
+ public static synchronized void getStats()
+ {
+ operations.add(CompactionManager.instance.getSSTableTasks());
+ }
+
+ static List> operations;
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/unified/ParallelizedTasksTest.java b/test/unit/org/apache/cassandra/db/compaction/unified/ParallelizedTasksTest.java
new file mode 100644
index 0000000000..7e20e2d5a6
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/unified/ParallelizedTasksTest.java
@@ -0,0 +1,128 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.unified;
+
+import java.util.Collection;
+import java.util.List;
+import java.util.concurrent.ForkJoinPool;
+import java.util.concurrent.Future;
+import java.util.stream.Collectors;
+
+import org.junit.AfterClass;
+import org.junit.Before;
+import org.junit.Test;
+
+import org.apache.cassandra.cql3.QueryProcessor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.Keyspace;
+import org.apache.cassandra.db.compaction.AbstractCompactionTask;
+import org.apache.cassandra.db.compaction.ActiveCompactionsTracker;
+import org.apache.cassandra.db.compaction.OperationType;
+import org.apache.cassandra.db.compaction.ShardManager;
+import org.apache.cassandra.db.compaction.ShardManagerNoDisks;
+import org.apache.cassandra.db.compaction.UnifiedCompactionStrategy;
+import org.apache.cassandra.db.lifecycle.CompositeLifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.PartialLifecycleTransaction;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.FBUtilities;
+import org.apache.cassandra.utils.concurrent.Transactional;
+import org.mockito.Mockito;
+
+import static org.apache.cassandra.db.ColumnFamilyStore.RING_VERSION_IRRELEVANT;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+public class ParallelizedTasksTest extends ShardingTestBase
+{
+ @Before
+ public void before()
+ {
+ // Disabling durable write since we don't care
+ schemaChange("CREATE KEYSPACE IF NOT EXISTS " + KEYSPACE + " WITH replication = {'class': 'SimpleStrategy', 'replication_factor': '1'} AND durable_writes=false");
+ schemaChange(String.format("CREATE TABLE IF NOT EXISTS %s.%s (k int, t int, v blob, PRIMARY KEY (k, t))", KEYSPACE, TABLE));
+ }
+
+ @AfterClass
+ public static void tearDownClass()
+ {
+ QueryProcessor.executeInternal("DROP KEYSPACE IF EXISTS " + KEYSPACE);
+ }
+
+ private ColumnFamilyStore getColumnFamilyStore()
+ {
+ return Keyspace.open(KEYSPACE).getColumnFamilyStore(TABLE);
+ }
+
+ @Test
+ public void testOneSSTablePerShard() throws Throwable
+ {
+ int numShards = 5;
+ testParallelized(numShards, PARTITIONS, numShards, true);
+ }
+
+ @Test
+ public void testMultipleInputSSTables() throws Throwable
+ {
+ int numShards = 3;
+ testParallelized(numShards, PARTITIONS, numShards, false);
+ }
+
+ private void testParallelized(int numShards, int rowCount, int numOutputSSTables, boolean compact) throws Throwable
+ {
+ ColumnFamilyStore cfs = getColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ populate(rowCount, compact);
+
+ LifecycleTransaction transaction = cfs.getTracker().tryModify(cfs.getLiveSSTables(), OperationType.COMPACTION);
+
+ ShardManager shardManager = new ShardManagerNoDisks(ColumnFamilyStore.fullWeightedRange(RING_VERSION_IRRELEVANT, cfs.getPartitioner()));
+
+ Controller mockController = Mockito.mock(Controller.class);
+ UnifiedCompactionStrategy mockStrategy = Mockito.mock(UnifiedCompactionStrategy.class, Mockito.CALLS_REAL_METHODS);
+ Mockito.when(mockStrategy.getController()).thenReturn(mockController);
+ Mockito.when(mockController.getNumShards(Mockito.anyDouble())).thenReturn(numShards);
+
+ Collection sstables = transaction.originals();
+ CompositeLifecycleTransaction compositeTransaction = new CompositeLifecycleTransaction(transaction);
+
+ List tasks = shardManager.splitSSTablesInShards(
+ sstables,
+ numShards,
+ (rangeSSTables, range) ->
+ new UnifiedCompactionTask(cfs,
+ mockStrategy,
+ new PartialLifecycleTransaction(compositeTransaction),
+ 0,
+ shardManager,
+ range,
+ rangeSSTables)
+ );
+ compositeTransaction.completeInitialization();
+ assertEquals(numOutputSSTables, tasks.size());
+
+ List> futures = tasks.stream().map(t -> ForkJoinPool.commonPool().submit(() -> t.execute(ActiveCompactionsTracker.NOOP))).collect(Collectors.toList());
+ FBUtilities.waitOnFutures(futures);
+ assertTrue(transaction.state() == Transactional.AbstractTransactional.State.COMMITTED);
+
+ verifySharding(numShards, rowCount, numOutputSSTables, cfs);
+ cfs.truncateBlocking();
+ }
+}
\ No newline at end of file
diff --git a/test/unit/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriterTest.java b/test/unit/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriterTest.java
index 2ebe1bdf83..50f9179743 100644
--- a/test/unit/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriterTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/unified/ShardedCompactionWriterTest.java
@@ -18,10 +18,7 @@
package org.apache.cassandra.db.compaction.unified;
-import java.nio.ByteBuffer;
-import java.util.Collection;
import java.util.List;
-import java.util.Random;
import java.util.Set;
import java.util.stream.Collectors;
@@ -29,35 +26,27 @@ import org.junit.AfterClass;
import org.junit.Before;
import org.junit.Test;
-import org.apache.cassandra.cql3.CQLTester;
import org.apache.cassandra.cql3.QueryProcessor;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.Keyspace;
-import org.apache.cassandra.db.compaction.AbstractCompactionStrategy;
-import org.apache.cassandra.db.compaction.CompactionController;
-import org.apache.cassandra.db.compaction.CompactionIterator;
import org.apache.cassandra.db.compaction.OperationType;
import org.apache.cassandra.db.compaction.ShardManager;
import org.apache.cassandra.db.compaction.ShardManagerDiskAware;
import org.apache.cassandra.db.compaction.ShardManagerNoDisks;
-import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
import org.apache.cassandra.dht.Token;
import org.apache.cassandra.io.sstable.format.SSTableReader;
-import org.apache.cassandra.io.sstable.format.SSTableReaderWithFilter;
-import org.apache.cassandra.utils.FBUtilities;
-import org.apache.cassandra.utils.TimeUUID;
import static org.apache.cassandra.db.ColumnFamilyStore.RING_VERSION_IRRELEVANT;
import static org.junit.Assert.assertEquals;
-import static org.junit.Assert.assertTrue;
-public class ShardedCompactionWriterTest extends CQLTester
+public class ShardedCompactionWriterTest extends ShardingTestBase
{
private static final String KEYSPACE = "cawt_keyspace";
private static final String TABLE = "cawt_table";
private static final int ROW_PER_PARTITION = 10;
+ private static final int PARTITIONS = 50000;
@Before
public void before()
@@ -84,8 +73,7 @@ public class ShardedCompactionWriterTest extends CQLTester
// If we set the minSSTableSize ratio to 0.5, because this gets multiplied by the shard size to give the min sstable size,
// assuming evenly distributed data, it should split at each boundary and so we should end up with numShards sstables
int numShards = 5;
- int rowCount = 5000;
- testShardedCompactionWriter(numShards, rowCount, numShards, true);
+ testShardedCompactionWriter(numShards, PARTITIONS, numShards, true);
}
@@ -93,8 +81,7 @@ public class ShardedCompactionWriterTest extends CQLTester
public void testMultipleInputSSTables() throws Throwable
{
int numShards = 3;
- int rowCount = 5000;
- testShardedCompactionWriter(numShards, rowCount, numShards, false);
+ testShardedCompactionWriter(numShards, PARTITIONS, numShards, false);
}
private void testShardedCompactionWriter(int numShards, int rowCount, int numOutputSSTables, boolean majorCompaction) throws Throwable
@@ -107,33 +94,15 @@ public class ShardedCompactionWriterTest extends CQLTester
LifecycleTransaction txn = cfs.getTracker().tryModify(cfs.getLiveSSTables(), OperationType.COMPACTION);
ShardManager boundaries = new ShardManagerNoDisks(ColumnFamilyStore.fullWeightedRange(RING_VERSION_IRRELEVANT, cfs.getPartitioner()));
- ShardedCompactionWriter writer = new ShardedCompactionWriter(cfs, cfs.getDirectories(), txn, txn.originals(), false, boundaries.boundaries(numShards));
+ ShardedCompactionWriter writer = new ShardedCompactionWriter(cfs, cfs.getDirectories(), txn, txn.originals(), false, true, boundaries.boundaries(numShards));
int rows = compact(cfs, txn, writer);
- assertEquals(numOutputSSTables, cfs.getLiveSSTables().size());
assertEquals(rowCount, rows);
- long totalOnDiskLength = cfs.getLiveSSTables().stream().mapToLong(SSTableReader::onDiskLength).sum();
- long totalBFSize = cfs.getLiveSSTables().stream().mapToLong(ShardedCompactionWriterTest::getFilterSize).sum();
- assert totalBFSize > 16 * numOutputSSTables : "Bloom Filter is empty"; // 16 is the size of empty bloom filter
- for (SSTableReader rdr : cfs.getLiveSSTables())
- {
- assertEquals((double) rdr.onDiskLength() / totalOnDiskLength,
- (double) getFilterSize(rdr) / totalBFSize, 0.1);
- assertEquals(1.0 / numOutputSSTables, rdr.tokenSpaceCoverage(), 0.05);
- }
-
- validateData(cfs, rowCount);
+ verifySharding(numShards, rowCount, numOutputSSTables, cfs);
cfs.truncateBlocking();
}
- static long getFilterSize(SSTableReader rdr)
- {
- if (!(rdr instanceof SSTableReaderWithFilter))
- return 0;
- return ((SSTableReaderWithFilter) rdr).getFilterSerializedSize();
- }
-
@Test
public void testDiskAdvance() throws Throwable
{
@@ -200,94 +169,4 @@ public class ShardedCompactionWriterTest extends CQLTester
validateData(cfs, rowCount);
cfs.truncateBlocking();
}
-
- private int compact(int numShards, ColumnFamilyStore cfs, ShardManager shardManager, Collection selection)
- {
- int rows;
- LifecycleTransaction txn = cfs.getTracker().tryModify(selection, OperationType.COMPACTION);
- ShardedCompactionWriter writer = new ShardedCompactionWriter(cfs,
- cfs.getDirectories(),
- txn,
- txn.originals(),
- false,
- shardManager.boundaries(numShards));
-
- rows = compact(cfs, txn, writer);
- return rows;
- }
-
- private static void verifyNoSpannedBoundaries(List diskBoundaries, SSTableReader rdr)
- {
- for (int i = 0; i < diskBoundaries.size(); ++i)
- {
- Token boundary = diskBoundaries.get(i);
- // rdr cannot span a boundary. I.e. it must be either fully before (last <= boundary) or fully after
- // (first > boundary).
- assertTrue(rdr.getFirst().getToken().compareTo(boundary) > 0 ||
- rdr.getLast().getToken().compareTo(boundary) <= 0);
- }
- }
-
- private int compact(ColumnFamilyStore cfs, LifecycleTransaction txn, CompactionAwareWriter writer)
- {
- //assert txn.originals().size() == 1;
- int rowsWritten = 0;
- long nowInSec = FBUtilities.nowInSeconds();
- try (AbstractCompactionStrategy.ScannerList scanners = cfs.getCompactionStrategyManager().getScanners(txn.originals());
- CompactionController controller = new CompactionController(cfs, txn.originals(), cfs.gcBefore(nowInSec));
- CompactionIterator ci = new CompactionIterator(OperationType.COMPACTION, scanners.scanners, controller, nowInSec, TimeUUID.minAtUnixMillis(System.currentTimeMillis())))
- {
- while (ci.hasNext())
- {
- if (writer.append(ci.next()))
- rowsWritten++;
- }
- }
- writer.finish();
- return rowsWritten;
- }
-
- private void populate(int count, boolean compact) throws Throwable
- {
- byte [] payload = new byte[5000];
- new Random(42).nextBytes(payload);
- ByteBuffer b = ByteBuffer.wrap(payload);
-
- ColumnFamilyStore cfs = getColumnFamilyStore();
- for (int i = 0; i < count; i++)
- {
- for (int j = 0; j < ROW_PER_PARTITION; j++)
- execute(String.format("INSERT INTO %s.%s(k, t, v) VALUES (?, ?, ?)", KEYSPACE, TABLE), i, j, b);
-
- if (i % (count / 4) == 0)
- cfs.forceBlockingFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS);
- }
-
- cfs.forceBlockingFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS);
- if (compact && cfs.getLiveSSTables().size() > 1)
- {
- // we want just one big sstable to avoid doing actual compaction in compact() above
- try
- {
- cfs.forceMajorCompaction();
- }
- catch (Throwable t)
- {
- throw new RuntimeException(t);
- }
- assert cfs.getLiveSSTables().size() == 1 : cfs.getLiveSSTables();
- }
- }
-
- private void validateData(ColumnFamilyStore cfs, int rowCount) throws Throwable
- {
- for (int i = 0; i < rowCount; i++)
- {
- Object[][] expected = new Object[ROW_PER_PARTITION][];
- for (int j = 0; j < ROW_PER_PARTITION; j++)
- expected[j] = row(i, j);
-
- assertRows(execute(String.format("SELECT k, t FROM %s.%s WHERE k = :i", KEYSPACE, TABLE), i), expected);
- }
- }
}
\ No newline at end of file
diff --git a/test/unit/org/apache/cassandra/db/compaction/unified/ShardingTestBase.java b/test/unit/org/apache/cassandra/db/compaction/unified/ShardingTestBase.java
new file mode 100644
index 0000000000..fc2da63d00
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/unified/ShardingTestBase.java
@@ -0,0 +1,205 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.unified;
+
+import java.nio.ByteBuffer;
+import java.util.Collection;
+import java.util.List;
+import java.util.Random;
+
+import org.junit.AfterClass;
+import org.junit.Before;
+
+import org.apache.cassandra.cql3.CQLTester;
+import org.apache.cassandra.cql3.QueryProcessor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.Keyspace;
+import org.apache.cassandra.db.compaction.AbstractCompactionStrategy;
+import org.apache.cassandra.db.compaction.CompactionController;
+import org.apache.cassandra.db.compaction.CompactionIterator;
+import org.apache.cassandra.db.compaction.OperationType;
+import org.apache.cassandra.db.compaction.ShardManager;
+import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
+import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
+import org.apache.cassandra.dht.Token;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.io.sstable.format.SSTableReaderWithFilter;
+import org.apache.cassandra.utils.FBUtilities;
+import org.apache.cassandra.utils.FilterFactory;
+import org.apache.cassandra.utils.TimeUUID;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+public class ShardingTestBase extends CQLTester
+{
+ static final String KEYSPACE = "cawt_keyspace";
+ static final String TABLE = "cawt_table";
+
+ static final int ROW_PER_PARTITION = 10;
+ static final int PARTITIONS = 50000;
+
+ @Before
+ public void before()
+ {
+ // Disabling durable write since we don't care
+ schemaChange("CREATE KEYSPACE IF NOT EXISTS " + KEYSPACE + " WITH replication = {'class': 'SimpleStrategy', 'replication_factor': '1'} AND durable_writes=false");
+ schemaChange(String.format("CREATE TABLE IF NOT EXISTS %s.%s (k int, t int, v blob, PRIMARY KEY (k, t))", KEYSPACE, TABLE));
+ }
+
+ @AfterClass
+ public static void tearDownClass()
+ {
+ QueryProcessor.executeInternal("DROP KEYSPACE IF EXISTS " + KEYSPACE);
+ }
+
+ private ColumnFamilyStore getColumnFamilyStore()
+ {
+ return Keyspace.open(KEYSPACE).getColumnFamilyStore(TABLE);
+ }
+
+ void verifySharding(int numShards, int rowCount, int numOutputSSTables, ColumnFamilyStore cfs) throws Throwable
+ {
+ assertEquals(numOutputSSTables, cfs.getLiveSSTables().size());
+
+ long totalOnDiskLength = cfs.getLiveSSTables().stream().mapToLong(SSTableReader::onDiskLength).sum();
+ long totalBFSize = cfs.getLiveSSTables().stream().mapToLong(ShardingTestBase::getFilterSize).sum();
+ long totalKeyCount = cfs.getLiveSSTables().stream().mapToLong(SSTableReader::estimatedKeys).sum();
+ assert totalBFSize > 16 * numOutputSSTables : "Bloom Filter is empty"; // 16 is the size of empty bloom filter
+ for (SSTableReader rdr : cfs.getLiveSSTables())
+ {
+ assertEquals((double) rdr.onDiskLength() / totalOnDiskLength,
+ (double) getFilterSize(rdr) / totalBFSize, 0.1);
+ assertEquals(1.0 / numOutputSSTables, rdr.tokenSpaceCoverage(), 0.05);
+ }
+
+ System.out.println("Total on disk length: " + FBUtilities.prettyPrintMemory(totalOnDiskLength));
+ System.out.println("Total BF size: " + FBUtilities.prettyPrintMemory(totalBFSize));
+ System.out.println("Total key count: " + FBUtilities.prettyPrintDecimal(totalKeyCount, "", ""));
+ try (var filter = FilterFactory.getFilter(totalKeyCount, 0.01))
+ {
+ System.out.println("Optimal total BF size: " + FBUtilities.prettyPrintMemory(filter.serializedSize(false)));
+ }
+ try (var filter = FilterFactory.getFilter(totalKeyCount / numShards, 0.01))
+ {
+ System.out.println("Sharded optimal total BF size: " + FBUtilities.prettyPrintMemory(filter.serializedSize(false) * numShards));
+ }
+
+ cfs.getLiveSSTables().forEach(s -> System.out.println("SSTable: " + s.toString() + " covers " + s.getFirst() + " to " + s.getLast()));
+
+ validateData(cfs, rowCount);
+ }
+
+ static long getFilterSize(SSTableReader rdr)
+ {
+ if (!(rdr instanceof SSTableReaderWithFilter))
+ return 0;
+ return ((SSTableReaderWithFilter) rdr).getFilterSerializedSize();
+ }
+
+ int compact(int numShards, ColumnFamilyStore cfs, ShardManager shardManager, Collection selection)
+ {
+ int rows;
+ LifecycleTransaction txn = cfs.getTracker().tryModify(selection, OperationType.COMPACTION);
+ ShardedCompactionWriter writer = new ShardedCompactionWriter(cfs,
+ cfs.getDirectories(),
+ txn,
+ txn.originals(),
+ false,
+ true,
+ shardManager.boundaries(numShards));
+
+ rows = compact(cfs, txn, writer);
+ return rows;
+ }
+
+ static void verifyNoSpannedBoundaries(List diskBoundaries, SSTableReader rdr)
+ {
+ for (int i = 0; i < diskBoundaries.size(); ++i)
+ {
+ Token boundary = diskBoundaries.get(i);
+ // rdr cannot span a boundary. I.e. it must be either fully before (last <= boundary) or fully after
+ // (first > boundary).
+ assertTrue(rdr.getFirst().getToken().compareTo(boundary) > 0 ||
+ rdr.getLast().getToken().compareTo(boundary) <= 0);
+ }
+ }
+
+ int compact(ColumnFamilyStore cfs, LifecycleTransaction txn, CompactionAwareWriter writer)
+ {
+ //assert txn.originals().size() == 1;
+ int rowsWritten = 0;
+ long nowInSec = FBUtilities.nowInSeconds();
+ try (AbstractCompactionStrategy.ScannerList scanners = cfs.getCompactionStrategyManager().getScanners(txn.originals());
+ CompactionController controller = new CompactionController(cfs, txn.originals(), cfs.gcBefore(nowInSec));
+ CompactionIterator ci = new CompactionIterator(OperationType.COMPACTION, scanners.scanners, controller, nowInSec, TimeUUID.minAtUnixMillis(System.currentTimeMillis())))
+ {
+ while (ci.hasNext())
+ {
+ if (writer.append(ci.next()))
+ rowsWritten++;
+ }
+ }
+ writer.finish();
+ return rowsWritten;
+ }
+
+ void populate(int count, boolean compact) throws Throwable
+ {
+ byte [] payload = new byte[5000];
+ new Random(42).nextBytes(payload);
+ ByteBuffer b = ByteBuffer.wrap(payload);
+
+ ColumnFamilyStore cfs = getColumnFamilyStore();
+ for (int i = 0; i < count; i++)
+ {
+ for (int j = 0; j < ROW_PER_PARTITION; j++)
+ execute(String.format("INSERT INTO %s.%s(k, t, v) VALUES (?, ?, ?)", KEYSPACE, TABLE), i, j, b);
+
+ if (i % (count / 4) == 0)
+ cfs.forceBlockingFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS);
+ }
+
+ cfs.forceBlockingFlush(ColumnFamilyStore.FlushReason.UNIT_TESTS);
+ if (compact && cfs.getLiveSSTables().size() > 1)
+ {
+ // we want no overlapping sstables to avoid doing actual compaction in compact() above
+ try
+ {
+ cfs.forceMajorCompaction();
+ }
+ catch (Throwable t)
+ {
+ throw new RuntimeException(t);
+ }
+ }
+ }
+
+ void validateData(ColumnFamilyStore cfs, int rowCount) throws Throwable
+ {
+ for (int i = 0; i < rowCount; i++)
+ {
+ Object[][] expected = new Object[ROW_PER_PARTITION][];
+ for (int j = 0; j < ROW_PER_PARTITION; j++)
+ expected[j] = row(i, j);
+
+ assertRows(execute(String.format("SELECT k, t FROM %s.%s WHERE k = :i", KEYSPACE, TABLE), i), expected);
+ }
+ }
+}
\ No newline at end of file
diff --git a/test/unit/org/apache/cassandra/db/streaming/CassandraEntireSSTableStreamWriterTest.java b/test/unit/org/apache/cassandra/db/streaming/CassandraEntireSSTableStreamWriterTest.java
index 89a3e9ed36..c29f059000 100644
--- a/test/unit/org/apache/cassandra/db/streaming/CassandraEntireSSTableStreamWriterTest.java
+++ b/test/unit/org/apache/cassandra/db/streaming/CassandraEntireSSTableStreamWriterTest.java
@@ -104,7 +104,7 @@ public class CassandraEntireSSTableStreamWriterTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
sstable = store.getLiveSSTables().iterator().next();
descriptor = sstable.descriptor;
diff --git a/test/unit/org/apache/cassandra/db/streaming/CassandraOutgoingFileTest.java b/test/unit/org/apache/cassandra/db/streaming/CassandraOutgoingFileTest.java
index 2921b5ae17..8663c5f5eb 100644
--- a/test/unit/org/apache/cassandra/db/streaming/CassandraOutgoingFileTest.java
+++ b/test/unit/org/apache/cassandra/db/streaming/CassandraOutgoingFileTest.java
@@ -81,7 +81,7 @@ public class CassandraOutgoingFileTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
sstable = store.getLiveSSTables().iterator().next();
}
diff --git a/test/unit/org/apache/cassandra/db/streaming/CassandraStreamHeaderTest.java b/test/unit/org/apache/cassandra/db/streaming/CassandraStreamHeaderTest.java
index de953734ca..00e30938e2 100644
--- a/test/unit/org/apache/cassandra/db/streaming/CassandraStreamHeaderTest.java
+++ b/test/unit/org/apache/cassandra/db/streaming/CassandraStreamHeaderTest.java
@@ -82,7 +82,7 @@ public class CassandraStreamHeaderTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
sstable = store.getLiveSSTables().iterator().next();
}
diff --git a/test/unit/org/apache/cassandra/db/streaming/EntireSSTableStreamConcurrentComponentMutationTest.java b/test/unit/org/apache/cassandra/db/streaming/EntireSSTableStreamConcurrentComponentMutationTest.java
index 8107fd9789..7f0fe2c9c6 100644
--- a/test/unit/org/apache/cassandra/db/streaming/EntireSSTableStreamConcurrentComponentMutationTest.java
+++ b/test/unit/org/apache/cassandra/db/streaming/EntireSSTableStreamConcurrentComponentMutationTest.java
@@ -130,7 +130,7 @@ public class EntireSSTableStreamConcurrentComponentMutationTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
Token start = ByteOrderedPartitioner.instance.getTokenFactory().fromString(Long.toHexString(0));
Token end = ByteOrderedPartitioner.instance.getTokenFactory().fromString(Long.toHexString(100));
diff --git a/test/unit/org/apache/cassandra/io/sstable/LegacySSTableTest.java b/test/unit/org/apache/cassandra/io/sstable/LegacySSTableTest.java
index 96ab73f626..7162da1018 100644
--- a/test/unit/org/apache/cassandra/io/sstable/LegacySSTableTest.java
+++ b/test/unit/org/apache/cassandra/io/sstable/LegacySSTableTest.java
@@ -44,7 +44,6 @@ import org.apache.cassandra.cql3.UntypedResultSet;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.Keyspace;
import org.apache.cassandra.db.SinglePartitionSliceCommandTest;
-import org.apache.cassandra.db.compaction.AbstractCompactionTask;
import org.apache.cassandra.db.compaction.CompactionManager;
import org.apache.cassandra.db.repair.PendingAntiCompaction;
import org.apache.cassandra.db.rows.RangeTombstoneMarker;
@@ -398,9 +397,9 @@ public class LegacySSTableTest
truncateLegacyTables(legacyVersion);
loadLegacyTables(legacyVersion);
ColumnFamilyStore cfs = Keyspace.open(LEGACY_TABLES_KEYSPACE).getColumnFamilyStore(String.format("legacy_%s_simple", legacyVersion));
- AbstractCompactionTask act = cfs.getCompactionStrategyManager().getNextBackgroundTask(0);
+ var act = cfs.getCompactionStrategyManager().getNextBackgroundTasks(0);
// there should be no compactions to run with auto upgrades disabled:
- assertEquals(null, act);
+ assertEquals(0, act.size());
}
DatabaseDescriptor.setAutomaticSSTableUpgradeEnabled(true);
diff --git a/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java b/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java
index a87ec024b5..4997420233 100644
--- a/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java
+++ b/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java
@@ -159,7 +159,7 @@ public class SSTableReaderTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
List> ranges = new ArrayList<>();
// 1 key
@@ -397,7 +397,7 @@ public class SSTableReaderTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
// check that all our keys are found correctly
SSTableReader sstable = store.getLiveSSTables().iterator().next();
@@ -522,7 +522,7 @@ public class SSTableReaderTest
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
SSTableReader sstable = store.getLiveSSTables().iterator().next();
long p2 = sstable.getPosition(dk(2), SSTableReader.Operator.EQ);
@@ -576,7 +576,7 @@ public class SSTableReaderTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
SSTableReader sstable = store.getLiveSSTables().iterator().next();
KeyCache keyCache = ((KeyCacheSupport>) sstable).getKeyCache();
@@ -773,7 +773,7 @@ public class SSTableReaderTest
}
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
SSTableReaderWithFilter sstable = (SSTableReaderWithFilter) store.getLiveSSTables().iterator().next();
sstable = (SSTableReaderWithFilter) sstable.cloneWithNewStart(dk(3));
@@ -1114,7 +1114,7 @@ public class SSTableReaderTest
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
// construct a range which is present in the sstable, but whose
// keys are not found in the first segment of the index.
@@ -1151,7 +1151,7 @@ public class SSTableReaderTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
Collection sstables = store.getLiveSSTables();
assert sstables.size() == 1;
@@ -1228,7 +1228,7 @@ public class SSTableReaderTest
.applyUnsafe();
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
Collection sstables = ServerTestUtils.getLiveIndexSummarySupportingReaders(store);
assert sstables.size() == 1;
diff --git a/test/unit/org/apache/cassandra/io/sstable/SSTableWriterTestBase.java b/test/unit/org/apache/cassandra/io/sstable/SSTableWriterTestBase.java
index 2e410da5cf..e4c47c68b1 100644
--- a/test/unit/org/apache/cassandra/io/sstable/SSTableWriterTestBase.java
+++ b/test/unit/org/apache/cassandra/io/sstable/SSTableWriterTestBase.java
@@ -152,7 +152,7 @@ public class SSTableWriterTestBase extends SchemaLoader
assertTrue(cfs.getTracker().getCompacting().isEmpty());
if(cfs.getLiveSSTables().size() > 0)
- assertFalse(CompactionManager.instance.submitMaximal(cfs, cfs.gcBefore((int) (System.currentTimeMillis() / 1000)), false).isEmpty());
+ assertFalse(CompactionManager.instance.submitMaximal(cfs, cfs.gcBefore((int) (System.currentTimeMillis() / 1000)), false, 0).isEmpty());
}
public static SSTableWriter getWriter(ColumnFamilyStore cfs, File directory, LifecycleTransaction txn, long repairedAt, TimeUUID pendingRepair, boolean isTransient)
diff --git a/test/unit/org/apache/cassandra/io/sstable/VerifyTest.java b/test/unit/org/apache/cassandra/io/sstable/VerifyTest.java
index 5493356a2e..3b651fc36b 100644
--- a/test/unit/org/apache/cassandra/io/sstable/VerifyTest.java
+++ b/test/unit/org/apache/cassandra/io/sstable/VerifyTest.java
@@ -108,6 +108,7 @@ public class VerifyTest
public static final String COUNTER_CF4 = "Counter4";
public static final String CORRUPT_CF = "Corrupt1";
public static final String CORRUPT_CF2 = "Corrupt2";
+ public static final String CORRUPT_CF3 = "Corrupt3";
public static final String CORRUPTCOUNTER_CF = "CounterCorrupt1";
public static final String CORRUPTCOUNTER_CF2 = "CounterCorrupt2";
@@ -130,6 +131,7 @@ public class VerifyTest
standardCFMD(KEYSPACE, CF4),
standardCFMD(KEYSPACE, CORRUPT_CF),
standardCFMD(KEYSPACE, CORRUPT_CF2),
+ standardCFMD(KEYSPACE, CORRUPT_CF3),
counterCFMD(KEYSPACE, COUNTER_CF).compression(compressionParameters),
counterCFMD(KEYSPACE, COUNTER_CF2).compression(compressionParameters),
counterCFMD(KEYSPACE, COUNTER_CF3),
@@ -516,7 +518,7 @@ public class VerifyTest
{
CompactionManager.instance.disableAutoCompaction();
Keyspace keyspace = Keyspace.open(KEYSPACE);
- ColumnFamilyStore cfs = keyspace.getColumnFamilyStore(CORRUPT_CF2);
+ ColumnFamilyStore cfs = keyspace.getColumnFamilyStore(CORRUPT_CF3);
fillCF(cfs, 2);
diff --git a/test/unit/org/apache/cassandra/streaming/EntireSSTableStreamingCorrectFilesCountTest.java b/test/unit/org/apache/cassandra/streaming/EntireSSTableStreamingCorrectFilesCountTest.java
index a4022cb346..80dd71f273 100644
--- a/test/unit/org/apache/cassandra/streaming/EntireSSTableStreamingCorrectFilesCountTest.java
+++ b/test/unit/org/apache/cassandra/streaming/EntireSSTableStreamingCorrectFilesCountTest.java
@@ -104,7 +104,7 @@ public class EntireSSTableStreamingCorrectFilesCountTest
}
Util.flush(store);
- CompactionManager.instance.performMaximal(store, false);
+ CompactionManager.instance.performMaximal(store);
sstable = store.getLiveSSTables().iterator().next();
diff --git a/tools/stress/src/org/apache/cassandra/stress/CompactionStress.java b/tools/stress/src/org/apache/cassandra/stress/CompactionStress.java
index eb741c6915..4fdb9c076f 100644
--- a/tools/stress/src/org/apache/cassandra/stress/CompactionStress.java
+++ b/tools/stress/src/org/apache/cassandra/stress/CompactionStress.java
@@ -239,7 +239,7 @@ public abstract class CompactionStress implements Runnable
List> futures = new ArrayList<>(threads);
if (maximal)
{
- futures = CompactionManager.instance.submitMaximal(cfs, FBUtilities.nowInSeconds(), false);
+ futures = CompactionManager.instance.submitMaximal(cfs, FBUtilities.nowInSeconds(), false, Integer.MAX_VALUE);
}
else
{