mirror of https://github.com/apache/cassandra
Accord/non-Accord interoperability and support for live migration
Patch by Ariel Weisberg; Reviewed by Blake Eggleston for CASSANDRA-18129 Co-authored-by: Blake Eggleston <beggleston@apple.com>
This commit is contained in:
parent
9e766f46a3
commit
c10c84b9cd
|
|
@ -1,4 +1,4 @@
|
||||||
[submodule "modules/accord"]
|
[submodule "modules/accord"]
|
||||||
path = modules/accord
|
path = modules/accord
|
||||||
url = https://github.com/apache/cassandra-accord.git
|
url = https://github.com/apache/cassandra-accord
|
||||||
branch = trunk
|
branch = trunk
|
||||||
|
|
|
||||||
12
build.xml
12
build.xml
|
|
@ -328,6 +328,7 @@
|
||||||
<resources id="_jvm11_test_arg_items">
|
<resources id="_jvm11_test_arg_items">
|
||||||
<string>-XX:-CMSClassUnloadingEnabled</string>
|
<string>-XX:-CMSClassUnloadingEnabled</string>
|
||||||
<string>-Dio.netty.tryReflectionSetAccessible=true</string>
|
<string>-Dio.netty.tryReflectionSetAccessible=true</string>
|
||||||
|
<string>-XX:MaxMetaspaceSize=2G</string>
|
||||||
</resources>
|
</resources>
|
||||||
<pathconvert property="_jvm11_test_arg_items_concat" refid="_jvm11_test_arg_items" pathsep=" "/>
|
<pathconvert property="_jvm11_test_arg_items_concat" refid="_jvm11_test_arg_items" pathsep=" "/>
|
||||||
<resources id="_jvm17_test_arg_items">
|
<resources id="_jvm17_test_arg_items">
|
||||||
|
|
@ -1155,6 +1156,7 @@
|
||||||
<attribute name="testtag" default=""/>
|
<attribute name="testtag" default=""/>
|
||||||
<attribute name="usejacoco" default="no"/>
|
<attribute name="usejacoco" default="no"/>
|
||||||
<attribute name="showoutput" default="false"/>
|
<attribute name="showoutput" default="false"/>
|
||||||
|
<attribute name="maxmemory" default="1024m"/>
|
||||||
|
|
||||||
<sequential>
|
<sequential>
|
||||||
<fail message="testing with build.test.dir (${build.test.dir}) not pointing to 'build/test/' will fail, test configurations are hardcoded.">
|
<fail message="testing with build.test.dir (${build.test.dir}) not pointing to 'build/test/' will fail, test configurations are hardcoded.">
|
||||||
|
|
@ -1174,7 +1176,7 @@
|
||||||
<mkdir dir="${build.test.output.dir}"/>
|
<mkdir dir="${build.test.output.dir}"/>
|
||||||
<mkdir dir="${build.test.output.dir}/@{testtag}"/>
|
<mkdir dir="${build.test.output.dir}/@{testtag}"/>
|
||||||
<mkdir dir="${tmp.dir}"/>
|
<mkdir dir="${tmp.dir}"/>
|
||||||
<junit-timeout fork="on" forkmode="@{forkmode}" failureproperty="testfailed" maxmemory="1024m" timeout="@{timeout}" showoutput="@{showoutput}">
|
<junit-timeout fork="on" forkmode="@{forkmode}" failureproperty="testfailed" maxmemory="@{maxmemory}" timeout="@{timeout}" showoutput="@{showoutput}">
|
||||||
<formatter classname="org.apache.cassandra.CassandraXMLJUnitResultFormatter" extension=".xml" usefile="true"/>
|
<formatter classname="org.apache.cassandra.CassandraXMLJUnitResultFormatter" extension=".xml" usefile="true"/>
|
||||||
<formatter classname="org.apache.cassandra.CassandraBriefJUnitResultFormatter" usefile="false"/>
|
<formatter classname="org.apache.cassandra.CassandraBriefJUnitResultFormatter" usefile="false"/>
|
||||||
<jvmarg value="-Dstorage-config=${test.conf}"/>
|
<jvmarg value="-Dstorage-config=${test.conf}"/>
|
||||||
|
|
@ -1353,7 +1355,7 @@
|
||||||
<fileset file="${test.conf}/cassandra.yaml"/>
|
<fileset file="${test.conf}/cassandra.yaml"/>
|
||||||
<fileset file="${test.conf}/storage_compatibility_mode_none.yaml"/>
|
<fileset file="${test.conf}/storage_compatibility_mode_none.yaml"/>
|
||||||
</concat>
|
</concat>
|
||||||
<testmacrohelper inputdir="${test.dir}/${test.classlistprefix}" filelist="@{test.file.list}"
|
<testmacrohelper inputdir="${test.dir}/${test.classlistprefix}" filelist="@{test.file.list}"
|
||||||
exclude="**/*.java" timeout="${test.timeout}" testtag="oa">
|
exclude="**/*.java" timeout="${test.timeout}" testtag="oa">
|
||||||
<jvmarg value="-Dlegacy-sstable-root=${test.data}/legacy-sstables"/>
|
<jvmarg value="-Dlegacy-sstable-root=${test.data}/legacy-sstables"/>
|
||||||
<jvmarg value="-Dinvalid-legacy-sstable-root=${test.data}/invalid-legacy-sstables"/>
|
<jvmarg value="-Dinvalid-legacy-sstable-root=${test.data}/invalid-legacy-sstables"/>
|
||||||
|
|
@ -1664,12 +1666,14 @@
|
||||||
<attribute name="exclude" default="" />
|
<attribute name="exclude" default="" />
|
||||||
<attribute name="filelist" default="" />
|
<attribute name="filelist" default="" />
|
||||||
<attribute name="testtag" default=""/>
|
<attribute name="testtag" default=""/>
|
||||||
|
<attribute name="maxmemory" default="1024m"/>
|
||||||
|
|
||||||
<sequential>
|
<sequential>
|
||||||
<testmacrohelper inputdir="@{inputdir}" timeout="@{timeout}"
|
<testmacrohelper inputdir="@{inputdir}" timeout="@{timeout}"
|
||||||
forkmode="@{forkmode}" filter="@{filter}"
|
forkmode="@{forkmode}" filter="@{filter}"
|
||||||
exclude="@{exclude}" filelist="@{filelist}"
|
exclude="@{exclude}" filelist="@{filelist}"
|
||||||
testtag="@{testtag}" showoutput="false" >
|
testtag="@{testtag}" showoutput="@{showoutput}"
|
||||||
|
maxmemory="@{maxmemory}">
|
||||||
<optjvmargs/>
|
<optjvmargs/>
|
||||||
</testmacrohelper>
|
</testmacrohelper>
|
||||||
<fail message="Some test(s) failed.">
|
<fail message="Some test(s) failed.">
|
||||||
|
|
@ -1804,7 +1808,7 @@
|
||||||
|
|
||||||
<property name="simulator.asm.print" value="none"/> <!-- Supports: NONE, CLASS_SUMMARY, CLASS_DETAIL, METHOD_SUMMARY, METHOD_DETAIL, ASM; see org.apache.cassandra.simulator.asm.MethodLogger.Level -->
|
<property name="simulator.asm.print" value="none"/> <!-- Supports: NONE, CLASS_SUMMARY, CLASS_DETAIL, METHOD_SUMMARY, METHOD_DETAIL, ASM; see org.apache.cassandra.simulator.asm.MethodLogger.Level -->
|
||||||
<target name="test-simulator-dtest" depends="maybe-build-test" description="Execute simulator dtests">
|
<target name="test-simulator-dtest" depends="maybe-build-test" description="Execute simulator dtests">
|
||||||
<testmacro inputdir="${test.simulator-test.src}" timeout="${test.simulation.timeout}" forkmode="perTest" showoutput="true" filter="**/test/${test.name}.java">
|
<testmacro inputdir="${test.simulator-test.src}" timeout="${test.simulation.timeout}" forkmode="perTest" showoutput="true" filter="**/test/${test.name}.java" maxmemory="8g">
|
||||||
<jvmarg value="-Dlogback.configurationFile=test/conf/logback-simulator.xml"/>
|
<jvmarg value="-Dlogback.configurationFile=test/conf/logback-simulator.xml"/>
|
||||||
<jvmarg value="-Dcassandra.ring_delay_ms=10000"/>
|
<jvmarg value="-Dcassandra.ring_delay_ms=10000"/>
|
||||||
<jvmarg value="-Dcassandra.tolerate_sstable_size=true"/>
|
<jvmarg value="-Dcassandra.tolerate_sstable_size=true"/>
|
||||||
|
|
|
||||||
|
|
@ -213,7 +213,6 @@
|
||||||
-XX:CICompilerCount=1
|
-XX:CICompilerCount=1
|
||||||
-XX:HeapDumpPath=build/test
|
-XX:HeapDumpPath=build/test
|
||||||
-XX:MaxMetaspaceSize=2G
|
-XX:MaxMetaspaceSize=2G
|
||||||
-Xmx4G
|
|
||||||
-XX:ReservedCodeCacheSize=256M
|
-XX:ReservedCodeCacheSize=256M
|
||||||
-XX:Tier4CompileThreshold=1000
|
-XX:Tier4CompileThreshold=1000
|
||||||
-ea" />
|
-ea" />
|
||||||
|
|
|
||||||
|
|
@ -1 +1 @@
|
||||||
Subproject commit 3056d13bc8c45a22ec794e0979d02f469cc4e209
|
Subproject commit 6c6872270e16d2e777f1fa2c510b8f15396be3f3
|
||||||
|
|
@ -47,6 +47,7 @@ public enum Stage
|
||||||
MUTATION (true, "MutationStage", "request", DatabaseDescriptor::getConcurrentWriters, DatabaseDescriptor::setConcurrentWriters, Stage::multiThreadedLowSignalStage),
|
MUTATION (true, "MutationStage", "request", DatabaseDescriptor::getConcurrentWriters, DatabaseDescriptor::setConcurrentWriters, Stage::multiThreadedLowSignalStage),
|
||||||
COUNTER_MUTATION (true, "CounterMutationStage", "request", DatabaseDescriptor::getConcurrentCounterWriters, DatabaseDescriptor::setConcurrentCounterWriters, Stage::multiThreadedLowSignalStage),
|
COUNTER_MUTATION (true, "CounterMutationStage", "request", DatabaseDescriptor::getConcurrentCounterWriters, DatabaseDescriptor::setConcurrentCounterWriters, Stage::multiThreadedLowSignalStage),
|
||||||
VIEW_MUTATION (true, "ViewMutationStage", "request", DatabaseDescriptor::getConcurrentViewWriters, DatabaseDescriptor::setConcurrentViewWriters, Stage::multiThreadedLowSignalStage),
|
VIEW_MUTATION (true, "ViewMutationStage", "request", DatabaseDescriptor::getConcurrentViewWriters, DatabaseDescriptor::setConcurrentViewWriters, Stage::multiThreadedLowSignalStage),
|
||||||
|
ACCORD_MIGRATION (false, "AccordMigrationReadStage", "request", DatabaseDescriptor::getConcurrentAccordOps, DatabaseDescriptor::setConcurrentAccordOps, Stage::multiThreadedLowSignalStage),
|
||||||
GOSSIP (true, "GossipStage", "internal", () -> 1, null, Stage::singleThreadedStage),
|
GOSSIP (true, "GossipStage", "internal", () -> 1, null, Stage::singleThreadedStage),
|
||||||
REQUEST_RESPONSE (false, "RequestResponseStage", "request", FBUtilities::getAvailableProcessors, null, Stage::multiThreadedLowSignalStage),
|
REQUEST_RESPONSE (false, "RequestResponseStage", "request", FBUtilities::getAvailableProcessors, null, Stage::multiThreadedLowSignalStage),
|
||||||
ANTI_ENTROPY (false, "AntiEntropyStage", "internal", () -> 1, null, Stage::singleThreadedStage),
|
ANTI_ENTROPY (false, "AntiEntropyStage", "internal", () -> 1, null, Stage::singleThreadedStage),
|
||||||
|
|
|
||||||
|
|
@ -71,7 +71,11 @@ public class SyncFutureTask<T> extends SyncFuture<T> implements RunnableFuture<T
|
||||||
catch (Throwable t)
|
catch (Throwable t)
|
||||||
{
|
{
|
||||||
tryFailure(t);
|
tryFailure(t);
|
||||||
ExecutionFailure.handle(t);
|
// A lot of exceptions are expected and will be handled by Cassandra
|
||||||
|
// by consuming the result of the future task so only treat Error
|
||||||
|
// as uncaught
|
||||||
|
if (t instanceof Error)
|
||||||
|
ExecutionFailure.handle(t);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -603,6 +603,7 @@ public enum CassandraRelevantProperties
|
||||||
TEST_ORG_CAFFINITAS_OHC_SEGMENTCOUNT("org.caffinitas.ohc.segmentCount"),
|
TEST_ORG_CAFFINITAS_OHC_SEGMENTCOUNT("org.caffinitas.ohc.segmentCount"),
|
||||||
TEST_PRESERVE_THREAD_CREATION_STACKTRACE("cassandra.test.preserve_thread_creation_stacktrace", "false"),
|
TEST_PRESERVE_THREAD_CREATION_STACKTRACE("cassandra.test.preserve_thread_creation_stacktrace", "false"),
|
||||||
TEST_RANDOM_SEED("cassandra.test.random.seed"),
|
TEST_RANDOM_SEED("cassandra.test.random.seed"),
|
||||||
|
TEST_RANGE_EXPENSIVE_CHECKS("cassandra.test.range_expensive_checks"),
|
||||||
TEST_READ_ITERATION_DELAY_MS("cassandra.test.read_iteration_delay_ms", "0"),
|
TEST_READ_ITERATION_DELAY_MS("cassandra.test.read_iteration_delay_ms", "0"),
|
||||||
TEST_REUSE_PREPARED("cassandra.test.reuse_prepared", "true"),
|
TEST_REUSE_PREPARED("cassandra.test.reuse_prepared", "true"),
|
||||||
TEST_ROW_CACHE_SIZE("cassandra.test.row_cache_size"),
|
TEST_ROW_CACHE_SIZE("cassandra.test.row_cache_size"),
|
||||||
|
|
|
||||||
|
|
@ -44,6 +44,7 @@ import org.apache.cassandra.io.compress.BufferType;
|
||||||
import org.apache.cassandra.io.sstable.format.big.BigFormat;
|
import org.apache.cassandra.io.sstable.format.big.BigFormat;
|
||||||
import org.apache.cassandra.service.StartupChecks.StartupCheckType;
|
import org.apache.cassandra.service.StartupChecks.StartupCheckType;
|
||||||
import org.apache.cassandra.utils.StorageCompatibilityMode;
|
import org.apache.cassandra.utils.StorageCompatibilityMode;
|
||||||
|
import org.apache.cassandra.service.accord.IAccordService;
|
||||||
|
|
||||||
import static org.apache.cassandra.config.CassandraRelevantProperties.AUTOCOMPACTION_ON_STARTUP_ENABLED;
|
import static org.apache.cassandra.config.CassandraRelevantProperties.AUTOCOMPACTION_ON_STARTUP_ENABLED;
|
||||||
import static org.apache.cassandra.config.CassandraRelevantProperties.CASSANDRA_AVAILABLE_PROCESSORS;
|
import static org.apache.cassandra.config.CassandraRelevantProperties.CASSANDRA_AVAILABLE_PROCESSORS;
|
||||||
|
|
@ -191,6 +192,7 @@ public class Config
|
||||||
|
|
||||||
public int concurrent_reads = 32;
|
public int concurrent_reads = 32;
|
||||||
public int concurrent_writes = 32;
|
public int concurrent_writes = 32;
|
||||||
|
public int concurrent_accord_operations = 32;
|
||||||
public int concurrent_counter_writes = 32;
|
public int concurrent_counter_writes = 32;
|
||||||
public int concurrent_materialized_view_writes = 32;
|
public int concurrent_materialized_view_writes = 32;
|
||||||
public OptionaldPositiveInt available_processors = new OptionaldPositiveInt(CASSANDRA_AVAILABLE_PROCESSORS.getInt(OptionaldPositiveInt.UNDEFINED_VALUE));
|
public OptionaldPositiveInt available_processors = new OptionaldPositiveInt(CASSANDRA_AVAILABLE_PROCESSORS.getInt(OptionaldPositiveInt.UNDEFINED_VALUE));
|
||||||
|
|
@ -500,6 +502,8 @@ public class Config
|
||||||
|
|
||||||
public DataStorageSpec.LongMebibytesBound paxos_cache_size = null;
|
public DataStorageSpec.LongMebibytesBound paxos_cache_size = null;
|
||||||
|
|
||||||
|
public DataStorageSpec.LongMebibytesBound consensus_migration_cache_size = null;
|
||||||
|
|
||||||
@Replaces(oldName = "cache_load_timeout_seconds", converter = Converters.NEGATIVE_SECONDS_DURATION, deprecated = true)
|
@Replaces(oldName = "cache_load_timeout_seconds", converter = Converters.NEGATIVE_SECONDS_DURATION, deprecated = true)
|
||||||
public DurationSpec.IntSecondsBound cache_load_timeout = new DurationSpec.IntSecondsBound("30s");
|
public DurationSpec.IntSecondsBound cache_load_timeout = new DurationSpec.IntSecondsBound("30s");
|
||||||
|
|
||||||
|
|
@ -1163,7 +1167,23 @@ public class Config
|
||||||
|
|
||||||
public volatile boolean client_request_size_metrics_enabled = true;
|
public volatile boolean client_request_size_metrics_enabled = true;
|
||||||
|
|
||||||
public LegacyPaxosStrategy legacy_paxos_strategy = LegacyPaxosStrategy.migration;
|
public LWTStrategy lwt_strategy = LWTStrategy.migration;
|
||||||
|
public NonSerialWriteStrategy non_serial_write_strategy = NonSerialWriteStrategy.normal;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* When a barrier transaction is requested how many times to repeat attempting the barrier before giving up
|
||||||
|
*/
|
||||||
|
public int accord_barrier_retry_attempts = 5;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* When a barrier transaction fails how long the initial backoff should be before being increased
|
||||||
|
* as part of exponential backoff on each attempt
|
||||||
|
*/
|
||||||
|
public DurationSpec.IntMillisecondsBound accord_barrier_retry_inital_backoff_millis = new DurationSpec.IntMillisecondsBound("1s");
|
||||||
|
|
||||||
|
public DurationSpec.IntMillisecondsBound accord_barrier_max_backoff = new DurationSpec.IntMillisecondsBound("10m");
|
||||||
|
|
||||||
|
public DurationSpec.IntMillisecondsBound accord_range_barrier_timeout = new DurationSpec.IntMillisecondsBound("2m");
|
||||||
|
|
||||||
public volatile int max_top_size_partition_count = 10;
|
public volatile int max_top_size_partition_count = 10;
|
||||||
public volatile int max_top_tombstone_partition_count = 10;
|
public volatile int max_top_tombstone_partition_count = 10;
|
||||||
|
|
@ -1387,7 +1407,7 @@ public class Config
|
||||||
* and serial read operations. Transaction statements
|
* and serial read operations. Transaction statements
|
||||||
* will always run on Accord. Legacy in this context includes PaxosV2.
|
* will always run on Accord. Legacy in this context includes PaxosV2.
|
||||||
*/
|
*/
|
||||||
public enum LegacyPaxosStrategy
|
public enum LWTStrategy
|
||||||
{
|
{
|
||||||
/**
|
/**
|
||||||
* Allow both Accord and PaxosV1/V2 to run on the same cluster
|
* Allow both Accord and PaxosV1/V2 to run on the same cluster
|
||||||
|
|
@ -1405,6 +1425,95 @@ public class Config
|
||||||
accord
|
accord
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Configure how non-serial writes should be executed. For Accord transactions to function correctly
|
||||||
|
* when mixed with non-SERIAL writes it's necessary for the writes to occur through Accord.
|
||||||
|
*
|
||||||
|
* Accord will also use this configuration to determine what consistency level to perform its reads
|
||||||
|
* at since it will need to be able to read data written at non-SERIAL consistency levels.
|
||||||
|
*
|
||||||
|
* BlockingReadRepair will also use this configuration to determine how BRR mutations are applied. For migration
|
||||||
|
* and accord the BRR mutations will be applied as Accord transactions so that BRR doesn't expose Accord to
|
||||||
|
* uncommitted Accord data that is being RRed. This can occur when Accord has applied a transaction at some, but not
|
||||||
|
* all replica since Accord defaults to asynchronous commit.
|
||||||
|
*
|
||||||
|
* By routing repairs through Accord it is guaranteed that the Accord derived contents of the repair have already been applied at any
|
||||||
|
* replica where Accord applies the transaction. This also prevents BRR from breaking atomicity of Accord writes.
|
||||||
|
*
|
||||||
|
* If they are not written through Accord then reads through Accord will be required to occur at
|
||||||
|
* consistency level compatible with the non-serial writes preventing single replica reads from being performed
|
||||||
|
* by Accord. It will also require Accord to perform read repair of non-serial writes.
|
||||||
|
*
|
||||||
|
* Even then there is the potential for Accord to inconsistently execute transactions at different replicas
|
||||||
|
* because different coordinators for an Accord transaction may encounter different non-SERIAL write state and
|
||||||
|
* race to commit different outcomes for the transaction.
|
||||||
|
*
|
||||||
|
* This is different from Paxos because Paxos performs consensus on the actual values to be applied so recovery
|
||||||
|
* coordinators will always produce a consistent state when applying a transaction. Accord performs consensus on
|
||||||
|
* the execution order of transaction and different coordinators witnessing different states not managed by Accord
|
||||||
|
* can produce multiple outcomes for a transaction.
|
||||||
|
*
|
||||||
|
* // TODO (maybe): To safely migrate you would have to route all writes through Accord with the current implementation
|
||||||
|
* // We could do it by range instead in the migration version, but then we need to know when all in flight writes
|
||||||
|
* // are done before marking a range as migrated. Would waiting out the timeout be enough (timeout bugs!)?
|
||||||
|
*/
|
||||||
|
public enum NonSerialWriteStrategy
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* Execute writes through Cassandra via StorageProxy's normal write path. This can lead Accord to compute
|
||||||
|
* multiple outcomes for a transaction that depends on data written by non-SERIAL writes.
|
||||||
|
*/
|
||||||
|
normal(false, false, false),
|
||||||
|
/*
|
||||||
|
* Allow mixing of non-SERIAL writes and Accord, but still force BRR through Accord
|
||||||
|
*/
|
||||||
|
mixed(false, false, true),
|
||||||
|
/*
|
||||||
|
* Execute writes through Accord skipping StorageProxy's normal write path, but commit
|
||||||
|
* writes at the provided consistency level so they can be read via non-SERIAL consistency levels.
|
||||||
|
*/
|
||||||
|
migration(false, true, true),
|
||||||
|
/*
|
||||||
|
* Execute writes through Accord skipping StorageProxy's normal write path. Ignores the provided consistency level
|
||||||
|
* which makes Accord commit writes at ANY similar to Paxos with commit consistency level ANY.
|
||||||
|
*/
|
||||||
|
accord(true, true, true);
|
||||||
|
|
||||||
|
public final boolean ignoresSuppliedConsistencyLevel;
|
||||||
|
public final boolean writesThroughAccord;
|
||||||
|
|
||||||
|
public final boolean blockingReadRepairThroughAccord;
|
||||||
|
|
||||||
|
NonSerialWriteStrategy(boolean ignoresSuppliedConsistencyLevel, boolean writesThroughAccord, boolean blockingReadRepairThroughAccord)
|
||||||
|
{
|
||||||
|
this.ignoresSuppliedConsistencyLevel = ignoresSuppliedConsistencyLevel;
|
||||||
|
this.writesThroughAccord = writesThroughAccord;
|
||||||
|
this.blockingReadRepairThroughAccord = blockingReadRepairThroughAccord;
|
||||||
|
}
|
||||||
|
|
||||||
|
public ConsistencyLevel commitCLForStrategy(ConsistencyLevel consistencyLevel)
|
||||||
|
{
|
||||||
|
if (ignoresSuppliedConsistencyLevel)
|
||||||
|
return null;
|
||||||
|
|
||||||
|
if (!IAccordService.SUPPORTED_COMMIT_CONSISTENCY_LEVELS.contains(consistencyLevel))
|
||||||
|
throw new UnsupportedOperationException("Consistency level " + consistencyLevel + " is unsupported with Accord for write/commit, supported are ANY, ONE, QUORUM, and ALL");
|
||||||
|
|
||||||
|
return consistencyLevel;
|
||||||
|
}
|
||||||
|
|
||||||
|
public ConsistencyLevel readCLForStrategy(ConsistencyLevel consistencyLevel)
|
||||||
|
{
|
||||||
|
if (ignoresSuppliedConsistencyLevel)
|
||||||
|
return null;
|
||||||
|
|
||||||
|
if (!IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS.contains(consistencyLevel))
|
||||||
|
throw new UnsupportedOperationException("Consistency level " + consistencyLevel + " is unsupported with Accord for read, supported are ONE, QUORUM, and SERIAL");
|
||||||
|
|
||||||
|
return consistencyLevel;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
private static final Set<String> SENSITIVE_KEYS = new HashSet<String>() {{
|
private static final Set<String> SENSITIVE_KEYS = new HashSet<String>() {{
|
||||||
add("client_encryption_options");
|
add("client_encryption_options");
|
||||||
add("server_encryption_options");
|
add("server_encryption_options");
|
||||||
|
|
|
||||||
|
|
@ -78,6 +78,8 @@ import org.apache.cassandra.auth.INetworkAuthorizer;
|
||||||
import org.apache.cassandra.auth.IRoleManager;
|
import org.apache.cassandra.auth.IRoleManager;
|
||||||
import org.apache.cassandra.config.Config.CommitLogSync;
|
import org.apache.cassandra.config.Config.CommitLogSync;
|
||||||
import org.apache.cassandra.config.Config.DiskAccessMode;
|
import org.apache.cassandra.config.Config.DiskAccessMode;
|
||||||
|
import org.apache.cassandra.config.Config.LWTStrategy;
|
||||||
|
import org.apache.cassandra.config.Config.NonSerialWriteStrategy;
|
||||||
import org.apache.cassandra.config.Config.PaxosOnLinearizabilityViolation;
|
import org.apache.cassandra.config.Config.PaxosOnLinearizabilityViolation;
|
||||||
import org.apache.cassandra.config.Config.PaxosStatePurging;
|
import org.apache.cassandra.config.Config.PaxosStatePurging;
|
||||||
import org.apache.cassandra.db.ConsistencyLevel;
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
|
|
@ -167,7 +169,7 @@ import static org.apache.cassandra.utils.Clock.Global.logInitializationOutcome;
|
||||||
public class DatabaseDescriptor
|
public class DatabaseDescriptor
|
||||||
{
|
{
|
||||||
public static final String NO_ACCORD_PAXOS_STRATEGY_WITH_ACCORD_DISABLED_MESSAGE =
|
public static final String NO_ACCORD_PAXOS_STRATEGY_WITH_ACCORD_DISABLED_MESSAGE =
|
||||||
"Cannot use legacy_paxos_strategy \"accord\" while Accord transactions are disabled.";
|
"Cannot use lwt_strategy \"accord\" while Accord transactions are disabled.";
|
||||||
|
|
||||||
static
|
static
|
||||||
{
|
{
|
||||||
|
|
@ -226,6 +228,7 @@ public class DatabaseDescriptor
|
||||||
|
|
||||||
private static long keyCacheSizeInMiB;
|
private static long keyCacheSizeInMiB;
|
||||||
private static long paxosCacheSizeInMiB;
|
private static long paxosCacheSizeInMiB;
|
||||||
|
private static long consensusMigrationCacheSizeInMiB;
|
||||||
private static long counterCacheSizeInMiB;
|
private static long counterCacheSizeInMiB;
|
||||||
private static long indexSummaryCapacityInMiB;
|
private static long indexSummaryCapacityInMiB;
|
||||||
|
|
||||||
|
|
@ -651,6 +654,9 @@ public class DatabaseDescriptor
|
||||||
if (conf.concurrent_counter_writes < 2)
|
if (conf.concurrent_counter_writes < 2)
|
||||||
throw new ConfigurationException("concurrent_counter_writes must be at least 2, but was " + conf.concurrent_counter_writes, false);
|
throw new ConfigurationException("concurrent_counter_writes must be at least 2, but was " + conf.concurrent_counter_writes, false);
|
||||||
|
|
||||||
|
if (conf.concurrent_accord_operations < 1)
|
||||||
|
throw new ConfigurationException("concurrent_accord_operations must be at least 1, but was " + conf.concurrent_accord_operations, false);
|
||||||
|
|
||||||
if (conf.networking_cache_size == null)
|
if (conf.networking_cache_size == null)
|
||||||
conf.networking_cache_size = new DataStorageSpec.IntMebibytesBound(Math.min(128, (int) (Runtime.getRuntime().maxMemory() / (16 * 1048576))));
|
conf.networking_cache_size = new DataStorageSpec.IntMebibytesBound(Math.min(128, (int) (Runtime.getRuntime().maxMemory() / (16 * 1048576))));
|
||||||
|
|
||||||
|
|
@ -718,7 +724,6 @@ public class DatabaseDescriptor
|
||||||
conf.commitlog_directory = storagedirFor("commitlog");
|
conf.commitlog_directory = storagedirFor("commitlog");
|
||||||
}
|
}
|
||||||
|
|
||||||
if (conf.accord.journal_directory == null)
|
|
||||||
initializeCommitLogDiskAccessMode();
|
initializeCommitLogDiskAccessMode();
|
||||||
if (commitLogWriteDiskAccessMode != conf.commitlog_disk_access_mode)
|
if (commitLogWriteDiskAccessMode != conf.commitlog_disk_access_mode)
|
||||||
logger.info("commitlog_disk_access_mode resolved to: {}", commitLogWriteDiskAccessMode);
|
logger.info("commitlog_disk_access_mode resolved to: {}", commitLogWriteDiskAccessMode);
|
||||||
|
|
@ -959,6 +964,22 @@ public class DatabaseDescriptor
|
||||||
+ conf.paxos_cache_size + "', supported values are <integer> >= 0.", false);
|
+ conf.paxos_cache_size + "', supported values are <integer> >= 0.", false);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
try
|
||||||
|
{
|
||||||
|
// if consensusMigrationCacheSizeInMiB option was set to "auto" then size of the cache should be "min(1% of Heap (in MB), 50MB)
|
||||||
|
consensusMigrationCacheSizeInMiB = (conf.consensus_migration_cache_size == null)
|
||||||
|
? Math.min(Math.max(1, (int) (Runtime.getRuntime().totalMemory() * 0.01 / 1024 / 1024)), 50)
|
||||||
|
: conf.consensus_migration_cache_size.toMebibytes();
|
||||||
|
|
||||||
|
if (consensusMigrationCacheSizeInMiB < 0)
|
||||||
|
throw new NumberFormatException(); // to escape duplicating error message
|
||||||
|
}
|
||||||
|
catch (NumberFormatException e)
|
||||||
|
{
|
||||||
|
throw new ConfigurationException("consensus_migration_cache_size option was set incorrectly to '"
|
||||||
|
+ conf.consensus_migration_cache_size + "', supported values are <integer> >= 0.", false);
|
||||||
|
}
|
||||||
|
|
||||||
// we need this assignment for the Settings virtual table - CASSANDRA-17735
|
// we need this assignment for the Settings virtual table - CASSANDRA-17735
|
||||||
conf.counter_cache_size = new DataStorageSpec.LongMebibytesBound(counterCacheSizeInMiB);
|
conf.counter_cache_size = new DataStorageSpec.LongMebibytesBound(counterCacheSizeInMiB);
|
||||||
|
|
||||||
|
|
@ -1146,8 +1167,13 @@ public class DatabaseDescriptor
|
||||||
if (conf.audit_logging_options != null)
|
if (conf.audit_logging_options != null)
|
||||||
setAuditLoggingOptions(conf.audit_logging_options);
|
setAuditLoggingOptions(conf.audit_logging_options);
|
||||||
|
|
||||||
if (conf.legacy_paxos_strategy == Config.LegacyPaxosStrategy.accord && !conf.accord.enabled)
|
if (conf.lwt_strategy == LWTStrategy.accord)
|
||||||
throw new ConfigurationException(NO_ACCORD_PAXOS_STRATEGY_WITH_ACCORD_DISABLED_MESSAGE);
|
{
|
||||||
|
if (!conf.accord.enabled)
|
||||||
|
throw new ConfigurationException(NO_ACCORD_PAXOS_STRATEGY_WITH_ACCORD_DISABLED_MESSAGE);
|
||||||
|
if (conf.non_serial_write_strategy == Config.NonSerialWriteStrategy.normal)
|
||||||
|
throw new ConfigurationException("If Accord is used for LWTs then regular writes needs to be routed through Accord for interoperability by setting non_serial_write_strategy to \"accord\" or \"migration\"");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
|
|
@ -2697,6 +2723,20 @@ public class DatabaseDescriptor
|
||||||
conf.concurrent_materialized_view_writes = concurrent_materialized_view_writes;
|
conf.concurrent_materialized_view_writes = concurrent_materialized_view_writes;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static int getConcurrentAccordOps()
|
||||||
|
{
|
||||||
|
return conf.concurrent_accord_operations;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static void setConcurrentAccordOps(int concurrent_operations)
|
||||||
|
{
|
||||||
|
if (concurrent_operations < 0)
|
||||||
|
{
|
||||||
|
throw new IllegalArgumentException("Concurrent accord operations must be non-negative");
|
||||||
|
}
|
||||||
|
conf.concurrent_accord_operations = concurrent_operations;
|
||||||
|
}
|
||||||
|
|
||||||
public static int getFlushWriters()
|
public static int getFlushWriters()
|
||||||
{
|
{
|
||||||
return conf.memtable_flush_writers;
|
return conf.memtable_flush_writers;
|
||||||
|
|
@ -3613,9 +3653,45 @@ public class DatabaseDescriptor
|
||||||
return conf.paxos_topology_repair_strict_each_quorum;
|
return conf.paxos_topology_repair_strict_each_quorum;
|
||||||
}
|
}
|
||||||
|
|
||||||
public static Config.LegacyPaxosStrategy getLegacyPaxosStrategy()
|
// TODO (desired): This configuration should come out of TrM to force the cluster to agree on it
|
||||||
|
public static LWTStrategy getLWTStrategy()
|
||||||
{
|
{
|
||||||
return conf.legacy_paxos_strategy;
|
return conf.lwt_strategy;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static void setLWTStrategy(LWTStrategy lwtStrategy)
|
||||||
|
{
|
||||||
|
conf.lwt_strategy = lwtStrategy;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static Config.NonSerialWriteStrategy getNonSerialWriteStrategy()
|
||||||
|
{
|
||||||
|
return conf.non_serial_write_strategy;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static void setNonSerialWriteStrategy(NonSerialWriteStrategy nonSerialWriteStrategy)
|
||||||
|
{
|
||||||
|
conf.non_serial_write_strategy = nonSerialWriteStrategy;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static int getAccordBarrierRetryAttempts()
|
||||||
|
{
|
||||||
|
return conf.accord_barrier_retry_attempts;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static long getAccordBarrierRetryInitialBackoffMillis()
|
||||||
|
{
|
||||||
|
return conf.accord_barrier_retry_inital_backoff_millis.toMilliseconds();
|
||||||
|
}
|
||||||
|
|
||||||
|
public static long getAccordBarrierRetryMaxBackoffMillis()
|
||||||
|
{
|
||||||
|
return conf.accord_barrier_max_backoff.toMilliseconds();
|
||||||
|
}
|
||||||
|
|
||||||
|
public static long getAccordRangeBarrierTimeoutNanos()
|
||||||
|
{
|
||||||
|
return conf.accord_range_barrier_timeout.to(TimeUnit.NANOSECONDS);
|
||||||
}
|
}
|
||||||
|
|
||||||
public static void setNativeTransportMaxRequestDataInFlightPerIpInBytes(long maxRequestDataInFlightInBytes)
|
public static void setNativeTransportMaxRequestDataInFlightPerIpInBytes(long maxRequestDataInFlightInBytes)
|
||||||
|
|
@ -4205,6 +4281,11 @@ public class DatabaseDescriptor
|
||||||
return paxosCacheSizeInMiB;
|
return paxosCacheSizeInMiB;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static long getConsensusMigrationCacheSizeInMiB()
|
||||||
|
{
|
||||||
|
return consensusMigrationCacheSizeInMiB;
|
||||||
|
}
|
||||||
|
|
||||||
public static long getCounterCacheSizeInMiB()
|
public static long getCounterCacheSizeInMiB()
|
||||||
{
|
{
|
||||||
return counterCacheSizeInMiB;
|
return counterCacheSizeInMiB;
|
||||||
|
|
|
||||||
|
|
@ -20,14 +20,26 @@ package org.apache.cassandra.cql3;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
|
|
||||||
|
import org.apache.cassandra.db.Clustering;
|
||||||
|
import org.apache.cassandra.db.ClusteringComparator;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DeletionPurger;
|
||||||
|
import org.apache.cassandra.db.DeletionTime;
|
||||||
|
import org.apache.cassandra.db.LivenessInfo;
|
||||||
|
import org.apache.cassandra.db.RangeTombstone;
|
||||||
|
import org.apache.cassandra.db.Slice;
|
||||||
|
import org.apache.cassandra.db.context.CounterContext;
|
||||||
import org.apache.cassandra.db.guardrails.Guardrails;
|
import org.apache.cassandra.db.guardrails.Guardrails;
|
||||||
|
import org.apache.cassandra.db.partitions.Partition;
|
||||||
|
import org.apache.cassandra.db.rows.BTreeRow;
|
||||||
|
import org.apache.cassandra.db.rows.BufferCell;
|
||||||
|
import org.apache.cassandra.db.rows.Cell;
|
||||||
|
import org.apache.cassandra.db.rows.CellPath;
|
||||||
|
import org.apache.cassandra.db.rows.Row;
|
||||||
|
import org.apache.cassandra.db.rows.Rows;
|
||||||
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.db.*;
|
|
||||||
import org.apache.cassandra.db.context.CounterContext;
|
|
||||||
import org.apache.cassandra.db.partitions.Partition;
|
|
||||||
import org.apache.cassandra.db.rows.*;
|
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
|
||||||
import org.apache.cassandra.service.ClientState;
|
import org.apache.cassandra.service.ClientState;
|
||||||
import org.apache.cassandra.utils.TimeUUID;
|
import org.apache.cassandra.utils.TimeUUID;
|
||||||
|
|
||||||
|
|
@ -39,6 +51,7 @@ public class UpdateParameters
|
||||||
public final TableMetadata metadata;
|
public final TableMetadata metadata;
|
||||||
public final ClientState clientState;
|
public final ClientState clientState;
|
||||||
public final QueryOptions options;
|
public final QueryOptions options;
|
||||||
|
public final boolean constructingAccordBaseUpdate;
|
||||||
|
|
||||||
private final long nowInSec;
|
private final long nowInSec;
|
||||||
private final long timestamp;
|
private final long timestamp;
|
||||||
|
|
@ -62,6 +75,18 @@ public class UpdateParameters
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
int ttl,
|
int ttl,
|
||||||
Map<DecoratedKey, Partition> prefetchedRows) throws InvalidRequestException
|
Map<DecoratedKey, Partition> prefetchedRows) throws InvalidRequestException
|
||||||
|
{
|
||||||
|
this(metadata, clientState, options, timestamp, nowInSec, ttl, prefetchedRows, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
public UpdateParameters(TableMetadata metadata,
|
||||||
|
ClientState clientState,
|
||||||
|
QueryOptions options,
|
||||||
|
long timestamp,
|
||||||
|
long nowInSec,
|
||||||
|
int ttl,
|
||||||
|
Map<DecoratedKey, Partition> prefetchedRows,
|
||||||
|
boolean constructingAccordBaseUpdate) throws InvalidRequestException
|
||||||
{
|
{
|
||||||
this.metadata = metadata;
|
this.metadata = metadata;
|
||||||
this.clientState = clientState;
|
this.clientState = clientState;
|
||||||
|
|
@ -79,6 +104,8 @@ public class UpdateParameters
|
||||||
// it to avoid potential confusion.
|
// it to avoid potential confusion.
|
||||||
if (timestamp == Long.MIN_VALUE)
|
if (timestamp == Long.MIN_VALUE)
|
||||||
throw new InvalidRequestException(String.format("Out of bound timestamp, must be in [%d, %d]", Long.MIN_VALUE + 1, Long.MAX_VALUE));
|
throw new InvalidRequestException(String.format("Out of bound timestamp, must be in [%d, %d]", Long.MIN_VALUE + 1, Long.MAX_VALUE));
|
||||||
|
|
||||||
|
this.constructingAccordBaseUpdate = constructingAccordBaseUpdate;
|
||||||
}
|
}
|
||||||
|
|
||||||
public <V> void newRow(Clustering<V> clustering) throws InvalidRequestException
|
public <V> void newRow(Clustering<V> clustering) throws InvalidRequestException
|
||||||
|
|
|
||||||
|
|
@ -343,7 +343,7 @@ public class BatchStatement implements CQLStatement.CompositeCQLStatement
|
||||||
}
|
}
|
||||||
QueryOptions statementOptions = options.forStatement(i);
|
QueryOptions statementOptions = options.forStatement(i);
|
||||||
long timestamp = attrs.getTimestamp(batchTimestamp, statementOptions);
|
long timestamp = attrs.getTimestamp(batchTimestamp, statementOptions);
|
||||||
statement.addUpdates(collector, partitionKeys.get(i), state, statementOptions, local, timestamp, nowInSeconds, requestTime);
|
statement.addUpdates(collector, partitionKeys.get(i), state, statementOptions, local, timestamp, nowInSeconds, requestTime, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (tablesWithZeroGcGs != null)
|
if (tablesWithZeroGcGs != null)
|
||||||
|
|
|
||||||
|
|
@ -223,7 +223,7 @@ final class BatchUpdatesCollector implements UpdatesCollector
|
||||||
PartitionUpdate update = updateEntry.getValue().build();
|
PartitionUpdate update = updateEntry.getValue().build();
|
||||||
updates.put(updateEntry.getKey(), update);
|
updates.put(updateEntry.getKey(), update);
|
||||||
}
|
}
|
||||||
return new Mutation(keyspaceName, key, updates.build(), createdAt);
|
return new Mutation(keyspaceName, key, updates.build(), createdAt, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
public PartitionUpdate.Builder get(TableId tableId)
|
public PartitionUpdate.Builder get(TableId tableId)
|
||||||
|
|
|
||||||
|
|
@ -31,14 +31,18 @@ import java.util.TreeMap;
|
||||||
|
|
||||||
import org.apache.commons.lang3.builder.ToStringBuilder;
|
import org.apache.commons.lang3.builder.ToStringBuilder;
|
||||||
import org.apache.commons.lang3.builder.ToStringStyle;
|
import org.apache.commons.lang3.builder.ToStringStyle;
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import accord.api.Update;
|
import accord.api.Update;
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
import org.apache.cassandra.cql3.QueryOptions;
|
import org.apache.cassandra.cql3.QueryOptions;
|
||||||
import org.apache.cassandra.cql3.UpdateParameters;
|
import org.apache.cassandra.cql3.UpdateParameters;
|
||||||
import org.apache.cassandra.cql3.conditions.ColumnCondition;
|
import org.apache.cassandra.cql3.conditions.ColumnCondition;
|
||||||
import org.apache.cassandra.db.Clustering;
|
import org.apache.cassandra.db.Clustering;
|
||||||
import org.apache.cassandra.db.Columns;
|
import org.apache.cassandra.db.Columns;
|
||||||
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.RegularAndStaticColumns;
|
import org.apache.cassandra.db.RegularAndStaticColumns;
|
||||||
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||||
|
|
@ -53,7 +57,6 @@ import org.apache.cassandra.db.partitions.FilteredPartition;
|
||||||
import org.apache.cassandra.db.partitions.Partition;
|
import org.apache.cassandra.db.partitions.Partition;
|
||||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||||
import org.apache.cassandra.db.rows.Row;
|
import org.apache.cassandra.db.rows.Row;
|
||||||
import org.apache.cassandra.db.rows.RowIterator;
|
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
import org.apache.cassandra.service.CASRequest;
|
import org.apache.cassandra.service.CASRequest;
|
||||||
import org.apache.cassandra.service.ClientState;
|
import org.apache.cassandra.service.ClientState;
|
||||||
|
|
@ -63,6 +66,7 @@ import org.apache.cassandra.service.accord.txn.TxnDataName;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnRead;
|
import org.apache.cassandra.service.accord.txn.TxnRead;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnReference;
|
import org.apache.cassandra.service.accord.txn.TxnReference;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnUpdate;
|
import org.apache.cassandra.service.accord.txn.TxnUpdate;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
||||||
import org.apache.cassandra.service.paxos.Ballot;
|
import org.apache.cassandra.service.paxos.Ballot;
|
||||||
|
|
@ -70,13 +74,21 @@ import org.apache.cassandra.transport.Dispatcher;
|
||||||
import org.apache.cassandra.utils.TimeUUID;
|
import org.apache.cassandra.utils.TimeUUID;
|
||||||
|
|
||||||
import static com.google.common.base.Preconditions.checkState;
|
import static com.google.common.base.Preconditions.checkState;
|
||||||
import static org.apache.cassandra.service.accord.txn.TxnDataName.Kind.USER;
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult;
|
||||||
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.RETRY_NEW_PROTOCOL;
|
||||||
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.casResult;
|
||||||
|
import static org.apache.cassandra.service.accord.txn.TxnDataName.Kind.CAS_READ;
|
||||||
|
import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol;
|
||||||
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Processed CAS conditions and update on potentially multiple rows of the same partition.
|
* Processed CAS conditions and update on potentially multiple rows of the same partition.
|
||||||
*/
|
*/
|
||||||
public class CQL3CasRequest implements CASRequest
|
public class CQL3CasRequest implements CASRequest
|
||||||
{
|
{
|
||||||
|
@SuppressWarnings("unused")
|
||||||
|
private static final Logger logger = LoggerFactory.getLogger(CQL3CasRequest.class);
|
||||||
|
|
||||||
public final TableMetadata metadata;
|
public final TableMetadata metadata;
|
||||||
public final DecoratedKey key;
|
public final DecoratedKey key;
|
||||||
private final RegularAndStaticColumns conditionColumns;
|
private final RegularAndStaticColumns conditionColumns;
|
||||||
|
|
@ -410,7 +422,7 @@ public class CQL3CasRequest implements CASRequest
|
||||||
|
|
||||||
public TxnCondition asTxnCondition()
|
public TxnCondition asTxnCondition()
|
||||||
{
|
{
|
||||||
TxnDataName txnDataName = new TxnDataName(USER, clustering, TxnRead.SERIAL_READ_NAME);
|
TxnDataName txnDataName = new TxnDataName(CAS_READ, clustering, TxnRead.CAS_READ_NAME);
|
||||||
TxnReference txnReference = new TxnReference(txnDataName, null);
|
TxnReference txnReference = new TxnReference(txnDataName, null);
|
||||||
return new TxnCondition.Exists(txnReference, TxnCondition.Kind.IS_NULL);
|
return new TxnCondition.Exists(txnReference, TxnCondition.Kind.IS_NULL);
|
||||||
}
|
}
|
||||||
|
|
@ -436,7 +448,7 @@ public class CQL3CasRequest implements CASRequest
|
||||||
|
|
||||||
public TxnCondition asTxnCondition()
|
public TxnCondition asTxnCondition()
|
||||||
{
|
{
|
||||||
TxnDataName txnDataName = new TxnDataName(USER, clustering, TxnRead.SERIAL_READ_NAME);
|
TxnDataName txnDataName = new TxnDataName(CAS_READ, clustering, TxnRead.CAS_READ_NAME);
|
||||||
TxnReference txnReference = new TxnReference(txnDataName, null);
|
TxnReference txnReference = new TxnReference(txnDataName, null);
|
||||||
return new TxnCondition.Exists(txnReference, TxnCondition.Kind.IS_NOT_NULL);
|
return new TxnCondition.Exists(txnReference, TxnCondition.Kind.IS_NOT_NULL);
|
||||||
}
|
}
|
||||||
|
|
@ -484,20 +496,26 @@ public class CQL3CasRequest implements CASRequest
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Txn toAccordTxn(ClientState clientState, long nowInSecs)
|
public Txn toAccordTxn(ConsistencyLevel consistencyLevel, ConsistencyLevel commitConsistencyLevel, ClientState clientState, long nowInSecs)
|
||||||
{
|
{
|
||||||
SinglePartitionReadCommand readCommand = readCommand(nowInSecs);
|
SinglePartitionReadCommand readCommand = readCommand(nowInSecs);
|
||||||
Update update = createUpdate(clientState);
|
Update update = createUpdate(clientState, commitConsistencyLevel);
|
||||||
// In a CAS request only one key is supported and writes
|
// If the write strategy is sending all writes through Accord there is no need to use the supplied consistency
|
||||||
|
// level since Accord will manage reading safely
|
||||||
|
consistencyLevel = DatabaseDescriptor.getNonSerialWriteStrategy().readCLForStrategy(consistencyLevel);
|
||||||
|
TxnRead read = TxnRead.createCasRead(readCommand, consistencyLevel);
|
||||||
|
// In a CAS requesting only one key is supported and writes
|
||||||
// can't be dependent on any data that is read (only conditions)
|
// can't be dependent on any data that is read (only conditions)
|
||||||
// so the only relevant keys are the read key
|
// so the only relevant keys are the read key
|
||||||
TxnRead read = TxnRead.createSerialRead(readCommand);
|
|
||||||
return new Txn.InMemory(read.keys(), read, TxnQuery.CONDITION, update);
|
return new Txn.InMemory(read.keys(), read, TxnQuery.CONDITION, update);
|
||||||
}
|
}
|
||||||
|
|
||||||
private Update createUpdate(ClientState clientState)
|
private Update createUpdate(ClientState clientState, ConsistencyLevel commitConsistencyLevel)
|
||||||
{
|
{
|
||||||
return new TxnUpdate(createWriteFragments(clientState), createCondition());
|
// Potentially ignore commit consistency level if non-SERIAL write strategy is Accord
|
||||||
|
// since it is safe to match what non-SERIAL writes do
|
||||||
|
commitConsistencyLevel = DatabaseDescriptor.getNonSerialWriteStrategy().commitCLForStrategy(commitConsistencyLevel);
|
||||||
|
return new TxnUpdate(createWriteFragments(clientState), createCondition(), commitConsistencyLevel);
|
||||||
}
|
}
|
||||||
|
|
||||||
private TxnCondition createCondition()
|
private TxnCondition createCondition()
|
||||||
|
|
@ -528,13 +546,23 @@ public class CQL3CasRequest implements CASRequest
|
||||||
TxnWrite.Fragment fragment = modification.getTxnWriteFragment(idx++, state, options);
|
TxnWrite.Fragment fragment = modification.getTxnWriteFragment(idx++, state, options);
|
||||||
fragments.add(fragment);
|
fragments.add(fragment);
|
||||||
}
|
}
|
||||||
|
for (RangeDeletion rangeDeletion : rangeDeletions)
|
||||||
|
{
|
||||||
|
ModificationStatement modification = rangeDeletion.stmt;
|
||||||
|
QueryOptions options = rangeDeletion.options;
|
||||||
|
TxnWrite.Fragment fragment = modification.getTxnWriteFragment(idx++, state, options);
|
||||||
|
fragments.add(fragment);
|
||||||
|
}
|
||||||
return fragments;
|
return fragments;
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public RowIterator toCasResult(TxnData txnData)
|
public ConsensusAttemptResult toCasResult(TxnResult txnResult)
|
||||||
{
|
{
|
||||||
FilteredPartition partition = txnData.get(TxnRead.SERIAL_READ);
|
if (txnResult.kind() == retry_new_protocol)
|
||||||
return partition != null ? partition.rowIterator() : null;
|
return RETRY_NEW_PROTOCOL;
|
||||||
|
TxnData txnData = (TxnData)txnResult;
|
||||||
|
FilteredPartition partition = txnData.get(TxnRead.CAS_READ);
|
||||||
|
return casResult(partition != null ? partition.rowIterator(false) : null);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -57,19 +57,10 @@ import org.apache.cassandra.cql3.UpdateParameters;
|
||||||
import org.apache.cassandra.cql3.Validation;
|
import org.apache.cassandra.cql3.Validation;
|
||||||
import org.apache.cassandra.cql3.VariableSpecifications;
|
import org.apache.cassandra.cql3.VariableSpecifications;
|
||||||
import org.apache.cassandra.cql3.WhereClause;
|
import org.apache.cassandra.cql3.WhereClause;
|
||||||
import org.apache.cassandra.cql3.constraints.ConstraintViolationException;
|
|
||||||
import org.apache.cassandra.db.guardrails.Guardrails;
|
|
||||||
import org.apache.cassandra.dht.Token;
|
|
||||||
import org.apache.cassandra.locator.Replica;
|
|
||||||
import org.apache.cassandra.locator.ReplicaLayout;
|
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
|
||||||
import org.apache.cassandra.schema.Schema;
|
|
||||||
import org.apache.cassandra.schema.SchemaConstants;
|
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
|
||||||
import org.apache.cassandra.schema.ViewMetadata;
|
|
||||||
import org.apache.cassandra.cql3.conditions.ColumnCondition;
|
import org.apache.cassandra.cql3.conditions.ColumnCondition;
|
||||||
import org.apache.cassandra.cql3.conditions.ColumnConditions;
|
import org.apache.cassandra.cql3.conditions.ColumnConditions;
|
||||||
import org.apache.cassandra.cql3.conditions.Conditions;
|
import org.apache.cassandra.cql3.conditions.Conditions;
|
||||||
|
import org.apache.cassandra.cql3.constraints.ConstraintViolationException;
|
||||||
import org.apache.cassandra.cql3.functions.Function;
|
import org.apache.cassandra.cql3.functions.Function;
|
||||||
import org.apache.cassandra.cql3.restrictions.StatementRestrictions;
|
import org.apache.cassandra.cql3.restrictions.StatementRestrictions;
|
||||||
import org.apache.cassandra.cql3.selection.ResultSetBuilder;
|
import org.apache.cassandra.cql3.selection.ResultSetBuilder;
|
||||||
|
|
@ -94,6 +85,7 @@ import org.apache.cassandra.db.filter.ClusteringIndexSliceFilter;
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
import org.apache.cassandra.db.filter.DataLimits;
|
import org.apache.cassandra.db.filter.DataLimits;
|
||||||
import org.apache.cassandra.db.filter.RowFilter;
|
import org.apache.cassandra.db.filter.RowFilter;
|
||||||
|
import org.apache.cassandra.db.guardrails.Guardrails;
|
||||||
import org.apache.cassandra.db.marshal.BooleanType;
|
import org.apache.cassandra.db.marshal.BooleanType;
|
||||||
import org.apache.cassandra.db.partitions.FilteredPartition;
|
import org.apache.cassandra.db.partitions.FilteredPartition;
|
||||||
import org.apache.cassandra.db.partitions.Partition;
|
import org.apache.cassandra.db.partitions.Partition;
|
||||||
|
|
@ -102,11 +94,19 @@ import org.apache.cassandra.db.partitions.PartitionIterators;
|
||||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||||
import org.apache.cassandra.db.rows.RowIterator;
|
import org.apache.cassandra.db.rows.RowIterator;
|
||||||
import org.apache.cassandra.db.view.View;
|
import org.apache.cassandra.db.view.View;
|
||||||
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
import org.apache.cassandra.exceptions.RequestExecutionException;
|
import org.apache.cassandra.exceptions.RequestExecutionException;
|
||||||
import org.apache.cassandra.exceptions.RequestValidationException;
|
import org.apache.cassandra.exceptions.RequestValidationException;
|
||||||
import org.apache.cassandra.exceptions.UnauthorizedException;
|
import org.apache.cassandra.exceptions.UnauthorizedException;
|
||||||
|
import org.apache.cassandra.locator.Replica;
|
||||||
|
import org.apache.cassandra.locator.ReplicaLayout;
|
||||||
import org.apache.cassandra.metrics.ClientRequestSizeMetrics;
|
import org.apache.cassandra.metrics.ClientRequestSizeMetrics;
|
||||||
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
|
import org.apache.cassandra.schema.Schema;
|
||||||
|
import org.apache.cassandra.schema.SchemaConstants;
|
||||||
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
|
import org.apache.cassandra.schema.ViewMetadata;
|
||||||
import org.apache.cassandra.service.ClientState;
|
import org.apache.cassandra.service.ClientState;
|
||||||
import org.apache.cassandra.service.QueryState;
|
import org.apache.cassandra.service.QueryState;
|
||||||
import org.apache.cassandra.service.StorageProxy;
|
import org.apache.cassandra.service.StorageProxy;
|
||||||
|
|
@ -641,7 +641,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
false,
|
false,
|
||||||
options.getTimestamp(queryState),
|
options.getTimestamp(queryState),
|
||||||
options.getNowInSeconds(queryState),
|
options.getNowInSeconds(queryState),
|
||||||
requestTime);
|
requestTime,
|
||||||
|
false);
|
||||||
if (!mutations.isEmpty())
|
if (!mutations.isEmpty())
|
||||||
{
|
{
|
||||||
StorageProxy.mutateWithTriggers(mutations, cl, false, requestTime);
|
StorageProxy.mutateWithTriggers(mutations, cl, false, requestTime);
|
||||||
|
|
@ -806,7 +807,7 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
{
|
{
|
||||||
long timestamp = options.getTimestamp(queryState);
|
long timestamp = options.getTimestamp(queryState);
|
||||||
long nowInSeconds = options.getNowInSeconds(queryState);
|
long nowInSeconds = options.getNowInSeconds(queryState);
|
||||||
for (IMutation mutation : getMutations(queryState.getClientState(), options, true, timestamp, nowInSeconds, requestTime))
|
for (IMutation mutation : getMutations(queryState.getClientState(), options, true, timestamp, nowInSeconds, requestTime, false))
|
||||||
mutation.apply();
|
mutation.apply();
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
@ -834,7 +835,7 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!request.appliesTo(current))
|
if (!request.appliesTo(current))
|
||||||
return current.rowIterator();
|
return current.rowIterator(false);
|
||||||
|
|
||||||
PartitionUpdate updates = request.makeUpdates(current, state, ballot);
|
PartitionUpdate updates = request.makeUpdates(current, state, ballot);
|
||||||
updates = TriggerExecutor.instance.execute(updates);
|
updates = TriggerExecutor.instance.execute(updates);
|
||||||
|
|
@ -859,19 +860,22 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
boolean local,
|
boolean local,
|
||||||
long timestamp,
|
long timestamp,
|
||||||
long nowInSeconds,
|
long nowInSeconds,
|
||||||
Dispatcher.RequestTime requestTime)
|
Dispatcher.RequestTime requestTime,
|
||||||
|
boolean constructingAccordBaseUpdate)
|
||||||
{
|
{
|
||||||
List<ByteBuffer> keys = buildPartitionKeyNames(options, state);
|
List<ByteBuffer> keys = buildPartitionKeyNames(options, state);
|
||||||
if(keys.size() == 1)
|
|
||||||
|
if (keys.size() == 1)
|
||||||
{
|
{
|
||||||
SingleTableSinglePartitionUpdatesCollector collector = new SingleTableSinglePartitionUpdatesCollector(metadata, updatedColumns);
|
SingleTableSinglePartitionUpdatesCollector collector = new SingleTableSinglePartitionUpdatesCollector(metadata, updatedColumns);
|
||||||
addUpdates(collector, keys, state, options, local, timestamp, nowInSeconds, requestTime);
|
addUpdates(collector, keys, state, options, local, timestamp, nowInSeconds, requestTime, constructingAccordBaseUpdate);
|
||||||
return collector.toMutations(state);
|
return collector.toMutations(state);
|
||||||
} else
|
}
|
||||||
|
else
|
||||||
{
|
{
|
||||||
HashMultiset<ByteBuffer> perPartitionKeyCounts = HashMultiset.create(keys);
|
HashMultiset<ByteBuffer> perPartitionKeyCounts = HashMultiset.create(keys);
|
||||||
SingleTableUpdatesCollector collector = new SingleTableUpdatesCollector(metadata, updatedColumns, perPartitionKeyCounts);
|
SingleTableUpdatesCollector collector = new SingleTableUpdatesCollector(metadata, updatedColumns, perPartitionKeyCounts);
|
||||||
addUpdates(collector, keys, state, options, local, timestamp, nowInSeconds, requestTime);
|
addUpdates(collector, keys, state, options, local, timestamp, nowInSeconds, requestTime, constructingAccordBaseUpdate);
|
||||||
return collector.toMutations(state);
|
return collector.toMutations(state);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -879,7 +883,7 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
public PartitionUpdate getTxnUpdate(ClientState state, QueryOptions options)
|
public PartitionUpdate getTxnUpdate(ClientState state, QueryOptions options)
|
||||||
{
|
{
|
||||||
List<? extends IMutation> mutations = getMutations(state, options, false, 0, 0, new Dispatcher.RequestTime(0, 0));
|
List<? extends IMutation> mutations = getMutations(state, options, false, 0, 0, new Dispatcher.RequestTime(0, 0), true);
|
||||||
if (mutations.size() != 1)
|
if (mutations.size() != 1)
|
||||||
throw new IllegalArgumentException("When running withing a transaction, modification statements may only mutate a single partition");
|
throw new IllegalArgumentException("When running withing a transaction, modification statements may only mutate a single partition");
|
||||||
return Iterables.getOnlyElement(mutations.get(0).getPartitionUpdates());
|
return Iterables.getOnlyElement(mutations.get(0).getPartitionUpdates());
|
||||||
|
|
@ -942,7 +946,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
boolean local,
|
boolean local,
|
||||||
long timestamp,
|
long timestamp,
|
||||||
long nowInSeconds,
|
long nowInSeconds,
|
||||||
Dispatcher.RequestTime requestTime)
|
Dispatcher.RequestTime requestTime,
|
||||||
|
boolean constructingAccordBaseUpdate)
|
||||||
{
|
{
|
||||||
if (hasSlices())
|
if (hasSlices())
|
||||||
{
|
{
|
||||||
|
|
@ -960,7 +965,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
local,
|
local,
|
||||||
timestamp,
|
timestamp,
|
||||||
nowInSeconds,
|
nowInSeconds,
|
||||||
requestTime);
|
requestTime,
|
||||||
|
constructingAccordBaseUpdate);
|
||||||
for (ByteBuffer key : keys)
|
for (ByteBuffer key : keys)
|
||||||
{
|
{
|
||||||
Validation.validateKey(metadata(), key);
|
Validation.validateKey(metadata(), key);
|
||||||
|
|
@ -984,7 +990,7 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
if (restrictions.hasClusteringColumnsRestrictions() && clusterings.isEmpty())
|
if (restrictions.hasClusteringColumnsRestrictions() && clusterings.isEmpty())
|
||||||
return;
|
return;
|
||||||
|
|
||||||
UpdateParameters params = makeUpdateParameters(keys, clusterings, state, options, local, timestamp, nowInSeconds, requestTime);
|
UpdateParameters params = makeUpdateParameters(keys, clusterings, state, options, local, timestamp, nowInSeconds, requestTime, constructingAccordBaseUpdate);
|
||||||
|
|
||||||
for (ByteBuffer key : keys)
|
for (ByteBuffer key : keys)
|
||||||
{
|
{
|
||||||
|
|
@ -1042,7 +1048,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
boolean local,
|
boolean local,
|
||||||
long timestamp,
|
long timestamp,
|
||||||
long nowInSeconds,
|
long nowInSeconds,
|
||||||
Dispatcher.RequestTime requestTime)
|
Dispatcher.RequestTime requestTime,
|
||||||
|
boolean constructingAccordBaseUpdate)
|
||||||
{
|
{
|
||||||
if (clusterings.contains(Clustering.STATIC_CLUSTERING))
|
if (clusterings.contains(Clustering.STATIC_CLUSTERING))
|
||||||
return makeUpdateParameters(keys,
|
return makeUpdateParameters(keys,
|
||||||
|
|
@ -1053,7 +1060,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
local,
|
local,
|
||||||
timestamp,
|
timestamp,
|
||||||
nowInSeconds,
|
nowInSeconds,
|
||||||
requestTime);
|
requestTime,
|
||||||
|
constructingAccordBaseUpdate);
|
||||||
|
|
||||||
return makeUpdateParameters(keys,
|
return makeUpdateParameters(keys,
|
||||||
new ClusteringIndexNamesFilter(clusterings, false),
|
new ClusteringIndexNamesFilter(clusterings, false),
|
||||||
|
|
@ -1063,7 +1071,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
local,
|
local,
|
||||||
timestamp,
|
timestamp,
|
||||||
nowInSeconds,
|
nowInSeconds,
|
||||||
requestTime);
|
requestTime,
|
||||||
|
constructingAccordBaseUpdate);
|
||||||
}
|
}
|
||||||
|
|
||||||
private UpdateParameters makeUpdateParameters(Collection<ByteBuffer> keys,
|
private UpdateParameters makeUpdateParameters(Collection<ByteBuffer> keys,
|
||||||
|
|
@ -1074,7 +1083,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
boolean local,
|
boolean local,
|
||||||
long timestamp,
|
long timestamp,
|
||||||
long nowInSeconds,
|
long nowInSeconds,
|
||||||
Dispatcher.RequestTime requestTime)
|
Dispatcher.RequestTime requestTime,
|
||||||
|
boolean constructingAccordBaseUpdate)
|
||||||
{
|
{
|
||||||
// Some lists operation requires reading
|
// Some lists operation requires reading
|
||||||
Map<DecoratedKey, Partition> lists =
|
Map<DecoratedKey, Partition> lists =
|
||||||
|
|
@ -1092,7 +1102,8 @@ public abstract class ModificationStatement implements CQLStatement.SingleKeyspa
|
||||||
getTimestamp(timestamp, options),
|
getTimestamp(timestamp, options),
|
||||||
nowInSeconds,
|
nowInSeconds,
|
||||||
getTimeToLive(options),
|
getTimeToLive(options),
|
||||||
lists);
|
lists,
|
||||||
|
constructingAccordBaseUpdate);
|
||||||
}
|
}
|
||||||
|
|
||||||
public static abstract class Parsed extends QualifiedStatement
|
public static abstract class Parsed extends QualifiedStatement
|
||||||
|
|
|
||||||
|
|
@ -44,7 +44,6 @@ import org.slf4j.LoggerFactory;
|
||||||
import accord.api.Key;
|
import accord.api.Key;
|
||||||
import accord.primitives.Keys;
|
import accord.primitives.Keys;
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
import accord.utils.Invariants;
|
|
||||||
import org.apache.cassandra.audit.AuditLogContext;
|
import org.apache.cassandra.audit.AuditLogContext;
|
||||||
import org.apache.cassandra.audit.AuditLogEntryType;
|
import org.apache.cassandra.audit.AuditLogEntryType;
|
||||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
|
|
@ -63,12 +62,11 @@ import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||||
import org.apache.cassandra.db.SinglePartitionReadQuery;
|
import org.apache.cassandra.db.SinglePartitionReadQuery;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.partitions.FilteredPartition;
|
import org.apache.cassandra.db.partitions.FilteredPartition;
|
||||||
import org.apache.cassandra.exceptions.ExceptionCode;
|
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.service.ClientState;
|
import org.apache.cassandra.service.ClientState;
|
||||||
import org.apache.cassandra.service.QueryState;
|
import org.apache.cassandra.service.QueryState;
|
||||||
import org.apache.cassandra.service.accord.AccordService;
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
import org.apache.cassandra.service.accord.api.AccordRoutableKey;
|
import org.apache.cassandra.service.accord.txn.AccordUpdate;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnCondition;
|
import org.apache.cassandra.service.accord.txn.TxnCondition;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnData;
|
import org.apache.cassandra.service.accord.txn.TxnData;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnDataName;
|
import org.apache.cassandra.service.accord.txn.TxnDataName;
|
||||||
|
|
@ -76,18 +74,18 @@ import org.apache.cassandra.service.accord.txn.TxnNamedRead;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnRead;
|
import org.apache.cassandra.service.accord.txn.TxnRead;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnReference;
|
import org.apache.cassandra.service.accord.txn.TxnReference;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnUpdate;
|
import org.apache.cassandra.service.accord.txn.TxnUpdate;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
||||||
import org.apache.cassandra.transport.Dispatcher;
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
|
||||||
import org.apache.cassandra.tcm.ClusterMetadataService;
|
|
||||||
import org.apache.cassandra.tcm.transformations.AddAccordKeyspace;
|
|
||||||
import org.apache.cassandra.transport.messages.ResultMessage;
|
import org.apache.cassandra.transport.messages.ResultMessage;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
|
|
||||||
import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse;
|
import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse;
|
||||||
import static org.apache.cassandra.cql3.statements.RequestValidations.checkNotNull;
|
import static org.apache.cassandra.cql3.statements.RequestValidations.checkNotNull;
|
||||||
import static org.apache.cassandra.cql3.statements.RequestValidations.checkTrue;
|
import static org.apache.cassandra.cql3.statements.RequestValidations.checkTrue;
|
||||||
|
import static org.apache.cassandra.service.accord.txn.TxnRead.createTxnRead;
|
||||||
|
import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol;
|
||||||
|
|
||||||
public class TransactionStatement implements CQLStatement.CompositeCQLStatement, CQLStatement.ReturningCQLStatement
|
public class TransactionStatement implements CQLStatement.CompositeCQLStatement, CQLStatement.ReturningCQLStatement
|
||||||
{
|
{
|
||||||
|
|
@ -302,9 +300,9 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement,
|
||||||
return fragments;
|
return fragments;
|
||||||
}
|
}
|
||||||
|
|
||||||
TxnUpdate createUpdate(ClientState state, QueryOptions options, Map<TxnDataName, NamedSelect> autoReads, Consumer<Key> keyConsumer)
|
AccordUpdate createUpdate(ClientState state, QueryOptions options, Map<TxnDataName, NamedSelect> autoReads, Consumer<Key> keyConsumer)
|
||||||
{
|
{
|
||||||
return new TxnUpdate(createWriteFragments(state, options, autoReads, keyConsumer), createCondition(options));
|
return new TxnUpdate(createWriteFragments(state, options, autoReads, keyConsumer), createCondition(options), null);
|
||||||
}
|
}
|
||||||
|
|
||||||
Keys toKeys(SortedSet<Key> keySet)
|
Keys toKeys(SortedSet<Key> keySet)
|
||||||
|
|
@ -323,16 +321,16 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement,
|
||||||
Preconditions.checkState(conditions.isEmpty(), "No condition should exist without updates present");
|
Preconditions.checkState(conditions.isEmpty(), "No condition should exist without updates present");
|
||||||
List<TxnNamedRead> reads = createNamedReads(options, state, ImmutableMap.of(), keySet::add);
|
List<TxnNamedRead> reads = createNamedReads(options, state, ImmutableMap.of(), keySet::add);
|
||||||
Keys txnKeys = toKeys(keySet);
|
Keys txnKeys = toKeys(keySet);
|
||||||
TxnRead read = new TxnRead(reads, txnKeys);
|
TxnRead read = createTxnRead(reads, txnKeys, null);
|
||||||
return new Txn.InMemory(txnKeys, read, TxnQuery.ALL);
|
return new Txn.InMemory(txnKeys, read, TxnQuery.ALL);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
Map<TxnDataName, NamedSelect> autoReads = new HashMap<>();
|
Map<TxnDataName, NamedSelect> autoReads = new HashMap<>();
|
||||||
TxnUpdate update = createUpdate(state, options, autoReads, keySet::add);
|
AccordUpdate update = createUpdate(state, options, autoReads, keySet::add);
|
||||||
List<TxnNamedRead> reads = createNamedReads(options, state, autoReads, keySet::add);
|
List<TxnNamedRead> reads = createNamedReads(options, state, autoReads, keySet::add);
|
||||||
Keys txnKeys = toKeys(keySet);
|
Keys txnKeys = toKeys(keySet);
|
||||||
TxnRead read = new TxnRead(reads, txnKeys);
|
TxnRead read = createTxnRead(reads, txnKeys, null);
|
||||||
return new Txn.InMemory(txnKeys, read, TxnQuery.ALL, update);
|
return new Txn.InMemory(txnKeys, read, TxnQuery.ALL, update);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -357,40 +355,6 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement,
|
||||||
return select.getLimit(options) != 1;
|
return select.getLimit(options) != 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
private void maybeConvertTablesToAccord(Txn txn)
|
|
||||||
{
|
|
||||||
Set<String> allKeyspaces = new HashSet<>();
|
|
||||||
Set<String> newKeyspaces = new HashSet<>();
|
|
||||||
txn.keys().forEach(key -> {
|
|
||||||
String keyspace = ((AccordRoutableKey) key).keyspace();
|
|
||||||
if (allKeyspaces.add(keyspace) && !AccordService.instance().isAccordManagedKeyspace(keyspace))
|
|
||||||
newKeyspaces.add(keyspace);
|
|
||||||
});
|
|
||||||
|
|
||||||
if (newKeyspaces.isEmpty())
|
|
||||||
return;
|
|
||||||
|
|
||||||
for (String keyspace : newKeyspaces)
|
|
||||||
{
|
|
||||||
ClusterMetadataService.instance().commit(new AddAccordKeyspace(keyspace),
|
|
||||||
metadata -> null,
|
|
||||||
(code, message) -> {
|
|
||||||
Invariants.checkState(code == ExceptionCode.ALREADY_EXISTS,
|
|
||||||
"Expected %s, got %s", ExceptionCode.ALREADY_EXISTS, code);
|
|
||||||
return null;
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
// we need to avoid creating a txnId in an epoch when no one has any ranges
|
|
||||||
FBUtilities.waitOnFuture(AccordService.instance().epochReady(ClusterMetadata.current().epoch));
|
|
||||||
|
|
||||||
for (String keyspace : allKeyspaces)
|
|
||||||
{
|
|
||||||
if (!AccordService.instance().isAccordManagedKeyspace(keyspace))
|
|
||||||
throw new IllegalStateException(keyspace + " is not an accord managed keyspace");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public ResultMessage execute(QueryState state, QueryOptions options, Dispatcher.RequestTime requestTime)
|
public ResultMessage execute(QueryState state, QueryOptions options, Dispatcher.RequestTime requestTime)
|
||||||
{
|
{
|
||||||
|
|
@ -407,9 +371,12 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement,
|
||||||
|
|
||||||
Txn txn = createTxn(state.getClientState(), options);
|
Txn txn = createTxn(state.getClientState(), options);
|
||||||
|
|
||||||
maybeConvertTablesToAccord(txn);
|
AccordService.instance().maybeConvertKeyspacesToAccord(txn);
|
||||||
|
|
||||||
TxnData data = AccordService.instance().coordinate(txn, options.getConsistency());
|
TxnResult txnResult = AccordService.instance().coordinate(txn, options.getConsistency(), requestTime);
|
||||||
|
if (txnResult.kind() == retry_new_protocol)
|
||||||
|
throw new IllegalStateException("Transaction statement should never be required to switch consensus protocols");
|
||||||
|
TxnData data = (TxnData)txnResult;
|
||||||
|
|
||||||
if (returningSelect != null)
|
if (returningSelect != null)
|
||||||
{
|
{
|
||||||
|
|
@ -420,8 +387,9 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement,
|
||||||
if (selectQuery.queries.size() == 1)
|
if (selectQuery.queries.size() == 1)
|
||||||
{
|
{
|
||||||
FilteredPartition partition = data.get(TxnDataName.returning());
|
FilteredPartition partition = data.get(TxnDataName.returning());
|
||||||
|
boolean reversed = selectQuery.queries.get(0).isReversed();
|
||||||
if (partition != null)
|
if (partition != null)
|
||||||
returningSelect.select.processPartition(partition.rowIterator(), options, result, FBUtilities.nowInSeconds());
|
returningSelect.select.processPartition(partition.rowIterator(reversed), options, result, FBUtilities.nowInSeconds());
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
|
@ -429,8 +397,9 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement,
|
||||||
for (int i = 0; i < selectQuery.queries.size(); i++)
|
for (int i = 0; i < selectQuery.queries.size(); i++)
|
||||||
{
|
{
|
||||||
FilteredPartition partition = data.get(TxnDataName.returning(i));
|
FilteredPartition partition = data.get(TxnDataName.returning(i));
|
||||||
|
boolean reversed = selectQuery.queries.get(i).isReversed();
|
||||||
if (partition != null)
|
if (partition != null)
|
||||||
returningSelect.select.processPartition(partition.rowIterator(), options, result, nowInSec);
|
returningSelect.select.processPartition(partition.rowIterator(reversed), options, result, nowInSec);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return new ResultMessage.Rows(result.build());
|
return new ResultMessage.Rows(result.build());
|
||||||
|
|
|
||||||
|
|
@ -27,6 +27,10 @@ import java.util.concurrent.atomic.AtomicReference;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
import java.util.stream.StreamSupport;
|
import java.util.stream.StreamSupport;
|
||||||
|
|
||||||
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import org.apache.cassandra.cql3.AssignmentTestable;
|
import org.apache.cassandra.cql3.AssignmentTestable;
|
||||||
import org.apache.cassandra.cql3.ColumnIdentifier;
|
import org.apache.cassandra.cql3.ColumnIdentifier;
|
||||||
import org.apache.cassandra.cql3.ColumnSpecification;
|
import org.apache.cassandra.cql3.ColumnSpecification;
|
||||||
|
|
@ -34,21 +38,25 @@ import org.apache.cassandra.cql3.Operation;
|
||||||
import org.apache.cassandra.cql3.QueryOptions;
|
import org.apache.cassandra.cql3.QueryOptions;
|
||||||
import org.apache.cassandra.cql3.UpdateParameters;
|
import org.apache.cassandra.cql3.UpdateParameters;
|
||||||
import org.apache.cassandra.cql3.VariableSpecifications;
|
import org.apache.cassandra.cql3.VariableSpecifications;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.guardrails.Guardrails;
|
import org.apache.cassandra.db.guardrails.Guardrails;
|
||||||
import org.apache.cassandra.db.marshal.MultiElementType;
|
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
|
||||||
import org.apache.cassandra.db.*;
|
|
||||||
import org.apache.cassandra.db.rows.*;
|
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.marshal.Int32Type;
|
import org.apache.cassandra.db.marshal.Int32Type;
|
||||||
import org.apache.cassandra.db.marshal.ListType;
|
import org.apache.cassandra.db.marshal.ListType;
|
||||||
|
import org.apache.cassandra.db.marshal.MultiElementType;
|
||||||
|
import org.apache.cassandra.db.marshal.TimeUUIDType;
|
||||||
|
import org.apache.cassandra.db.rows.Cell;
|
||||||
|
import org.apache.cassandra.db.rows.CellPath;
|
||||||
|
import org.apache.cassandra.db.rows.ComplexColumnData;
|
||||||
|
import org.apache.cassandra.db.rows.Row;
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
|
import org.apache.cassandra.utils.TimeUUID;
|
||||||
|
|
||||||
import static org.apache.cassandra.cql3.terms.Constants.UNSET_VALUE;
|
|
||||||
import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse;
|
import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse;
|
||||||
import static org.apache.cassandra.cql3.statements.RequestValidations.invalidRequest;
|
import static org.apache.cassandra.cql3.statements.RequestValidations.invalidRequest;
|
||||||
|
import static org.apache.cassandra.cql3.terms.Constants.UNSET_VALUE;
|
||||||
import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis;
|
import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis;
|
||||||
import static org.apache.cassandra.utils.TimeUUID.Generator.atUnixMillisAsBytes;
|
import static org.apache.cassandra.utils.TimeUUID.Generator.atUnixMillisAsBytes;
|
||||||
|
|
||||||
|
|
@ -57,6 +65,16 @@ import static org.apache.cassandra.utils.TimeUUID.Generator.atUnixMillisAsBytes;
|
||||||
*/
|
*/
|
||||||
public abstract class Lists
|
public abstract class Lists
|
||||||
{
|
{
|
||||||
|
@SuppressWarnings("unused")
|
||||||
|
private static final Logger logger = LoggerFactory.getLogger(Lists.class);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Sentinel value indicating the cell path should be replaced by Accord with one based on the transaction executeAt
|
||||||
|
*/
|
||||||
|
private static final TimeUUID ACCORD_CELL_PATH_SENTINEL_UUID = TimeUUID.atUnixMicrosWithLsb(0, 0);
|
||||||
|
public static final CellPath ACCORD_DUMMY_CELL_PATH = CellPath.create(ACCORD_CELL_PATH_SENTINEL_UUID.toBytes());
|
||||||
|
private static final long ACCORD_CELL_PATH_SENTINEL_MSB = ACCORD_CELL_PATH_SENTINEL_UUID.msb();
|
||||||
|
|
||||||
private Lists() {}
|
private Lists() {}
|
||||||
|
|
||||||
public static ColumnSpecification indexSpecOf(ColumnSpecification column)
|
public static ColumnSpecification indexSpecOf(ColumnSpecification column)
|
||||||
|
|
@ -142,6 +160,33 @@ public abstract class Lists
|
||||||
return type == null ? null : ListType.getInstance(type, false);
|
return type == null ? null : ListType.getInstance(type, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Return a function that given a cell with an ACCORD_CELL_PATH_SENTINEL_MSB will
|
||||||
|
* return a new CellPath with a TimeUUID that increases monotonically every time it is called or
|
||||||
|
* the existing cell path if path does not contain ACCORD_CELL_PATH_SENTINEL_MSB.
|
||||||
|
*
|
||||||
|
* Only intended to work with list cell paths where list append needs a timestamp based on the executeAt
|
||||||
|
* of the Accord transaction appending the cell.
|
||||||
|
* @param timestampMicros executeAt timestamp to use as the MSB for generated cell paths
|
||||||
|
*/
|
||||||
|
public static com.google.common.base.Function<Cell, CellPath> accordListPathSupplier(long timestampMicros)
|
||||||
|
{
|
||||||
|
return new com.google.common.base.Function<Cell, CellPath>()
|
||||||
|
{
|
||||||
|
final long timeUuidMsb = TimeUUID.unixMicrosToMsb(timestampMicros);
|
||||||
|
long cellIndex = 0;
|
||||||
|
@Override
|
||||||
|
public CellPath apply(Cell cell)
|
||||||
|
{
|
||||||
|
CellPath path = cell.path();
|
||||||
|
if (ACCORD_CELL_PATH_SENTINEL_MSB == path.get(0).getLong(0))
|
||||||
|
return CellPath.create(ByteBuffer.wrap(TimeUUID.toBytes(timeUuidMsb, TimeUUIDType.signedBytesToNativeLong(cellIndex++))));
|
||||||
|
else
|
||||||
|
return path;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
public static class Literal extends Term.Raw
|
public static class Literal extends Term.Raw
|
||||||
{
|
{
|
||||||
private final List<Term.Raw> elements;
|
private final List<Term.Raw> elements;
|
||||||
|
|
@ -406,11 +451,18 @@ public abstract class Lists
|
||||||
// during SSTable write.
|
// during SSTable write.
|
||||||
Guardrails.itemsPerCollection.guard(type.collectionSize(elements), column.name.toString(), false, params.clientState);
|
Guardrails.itemsPerCollection.guard(type.collectionSize(elements), column.name.toString(), false, params.clientState);
|
||||||
|
|
||||||
|
long cellIndex = 0;
|
||||||
int dataSize = 0;
|
int dataSize = 0;
|
||||||
for (ByteBuffer buffer : elements)
|
for (ByteBuffer buffer : elements)
|
||||||
{
|
{
|
||||||
ByteBuffer uuid = ByteBuffer.wrap(params.nextTimeUUIDAsBytes());
|
ByteBuffer cellPath;
|
||||||
Cell<?> cell = params.addCell(column, CellPath.create(uuid), buffer);
|
// Accord will need to replace this value later once it knows the executeAt timestamp
|
||||||
|
// so just put a TimeUUID with MSB sentinel for now
|
||||||
|
if (params.constructingAccordBaseUpdate)
|
||||||
|
cellPath = TimeUUID.atUnixMicrosWithLsb(0, cellIndex++).toBytes();
|
||||||
|
else
|
||||||
|
cellPath = ByteBuffer.wrap(params.nextTimeUUIDAsBytes());
|
||||||
|
Cell<?> cell = params.addCell(column, CellPath.create(cellPath), buffer);
|
||||||
dataSize += cell.dataSize();
|
dataSize += cell.dataSize();
|
||||||
}
|
}
|
||||||
Guardrails.collectionListSize.guard(dataSize, column.name.toString(), false, params.clientState);
|
Guardrails.collectionListSize.guard(dataSize, column.name.toString(), false, params.clientState);
|
||||||
|
|
|
||||||
|
|
@ -85,7 +85,9 @@ public abstract class AbstractMutationVerbHandler<T extends IMutation> implement
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!forToken.get().containsSelf())
|
// Mutations may intentionally be sent against an older Epoch so out of range checking doesn't work
|
||||||
|
// and could cause data to not end up where it needs to be for future operations
|
||||||
|
if (!message.payload.allowsOutOfRangeMutations() && !forToken.get().containsSelf())
|
||||||
{
|
{
|
||||||
StorageService.instance.incOutOfRangeOperationCount();
|
StorageService.instance.incOutOfRangeOperationCount();
|
||||||
Keyspace.open(message.payload.getKeyspaceName()).metric.outOfRangeTokenWrites.inc();
|
Keyspace.open(message.payload.getKeyspaceName()).metric.outOfRangeTokenWrites.inc();
|
||||||
|
|
@ -93,7 +95,7 @@ public abstract class AbstractMutationVerbHandler<T extends IMutation> implement
|
||||||
throw InvalidRoutingException.forWrite(respondTo, key.getToken(), metadata.epoch, message.payload);
|
throw InvalidRoutingException.forWrite(respondTo, key.getToken(), metadata.epoch, message.payload);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (forToken.lastModified().isAfter(message.epoch()))
|
if (!message.payload.allowsOutOfRangeMutations() && forToken.lastModified().isAfter(message.epoch()))
|
||||||
{
|
{
|
||||||
TCMMetrics.instance.coordinatorBehindPlacements.mark();
|
TCMMetrics.instance.coordinatorBehindPlacements.mark();
|
||||||
throw new CoordinatorBehindException(String.format("Routing is correct, but coordinator needs to catch-up at least to epoch %s to maintain consistency. Current coordinator epoch is %s",
|
throw new CoordinatorBehindException(String.format("Routing is correct, but coordinator needs to catch-up at least to epoch %s to maintain consistency. Current coordinator epoch is %s",
|
||||||
|
|
|
||||||
|
|
@ -3339,4 +3339,9 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean, Memtable.Owner
|
||||||
{
|
{
|
||||||
return metric;
|
return metric;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public TableId getTableId()
|
||||||
|
{
|
||||||
|
return metadata().id;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -70,4 +70,9 @@ public interface IMutation
|
||||||
}
|
}
|
||||||
return size;
|
return size;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
default boolean allowsOutOfRangeMutations()
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -18,7 +18,13 @@
|
||||||
package org.apache.cassandra.db;
|
package org.apache.cassandra.db;
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collection;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.HashSet;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.concurrent.atomic.AtomicLongFieldUpdater;
|
import java.util.concurrent.atomic.AtomicLongFieldUpdater;
|
||||||
import java.util.function.Supplier;
|
import java.util.function.Supplier;
|
||||||
|
|
@ -56,6 +62,7 @@ import static org.apache.cassandra.utils.MonotonicClock.Global.approxTime;
|
||||||
public class Mutation implements IMutation, Supplier<Mutation>
|
public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
{
|
{
|
||||||
public static final MutationSerializer serializer = new MutationSerializer();
|
public static final MutationSerializer serializer = new MutationSerializer();
|
||||||
|
public static final int ALLOW_OUT_OF_RANGE_MUTATIONS_FLAG = 0x01;
|
||||||
|
|
||||||
// todo this is redundant
|
// todo this is redundant
|
||||||
// when we remove it, also restore SerializationsTest.testMutationRead to not regenerate new Mutations each test
|
// when we remove it, also restore SerializationsTest.testMutationRead to not regenerate new Mutations each test
|
||||||
|
|
@ -84,23 +91,26 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
/** @see CassandraRelevantProperties#CACHEABLE_MUTATION_SIZE_LIMIT */
|
/** @see CassandraRelevantProperties#CACHEABLE_MUTATION_SIZE_LIMIT */
|
||||||
private static final long CACHEABLE_MUTATION_SIZE_LIMIT = CassandraRelevantProperties.CACHEABLE_MUTATION_SIZE_LIMIT.getLong();
|
private static final long CACHEABLE_MUTATION_SIZE_LIMIT = CassandraRelevantProperties.CACHEABLE_MUTATION_SIZE_LIMIT.getLong();
|
||||||
|
|
||||||
|
private boolean allowOutOfRangeMutations;
|
||||||
|
|
||||||
public Mutation(PartitionUpdate update)
|
public Mutation(PartitionUpdate update)
|
||||||
{
|
{
|
||||||
this(update.metadata().keyspace, update.partitionKey(), ImmutableMap.of(update.metadata().id, update), approxTime.now(), update.metadata().params.cdc);
|
this(update.metadata().keyspace, update.partitionKey(), ImmutableMap.of(update.metadata().id, update), approxTime.now(), update.metadata().params.cdc);
|
||||||
}
|
}
|
||||||
|
|
||||||
public Mutation(String keyspaceName, DecoratedKey key, ImmutableMap<TableId, PartitionUpdate> modifications, long approxCreatedAtNanos)
|
public Mutation(String keyspaceName, DecoratedKey key, ImmutableMap<TableId, PartitionUpdate> modifications, long approxCreatedAtNanos, boolean allowOutOfRangeMutations)
|
||||||
{
|
{
|
||||||
this(keyspaceName, key, modifications, approxCreatedAtNanos, cdcEnabled(modifications.values()));
|
this(keyspaceName, key, modifications, approxCreatedAtNanos, cdcEnabled(modifications.values()), allowOutOfRangeMutations);
|
||||||
}
|
}
|
||||||
|
|
||||||
public Mutation(String keyspaceName, DecoratedKey key, ImmutableMap<TableId, PartitionUpdate> modifications, long approxCreatedAtNanos, boolean cdcEnabled)
|
public Mutation(String keyspaceName, DecoratedKey key, ImmutableMap<TableId, PartitionUpdate> modifications, long approxCreatedAtNanos, boolean cdcEnabled, boolean allowOutOfRangeMutations)
|
||||||
{
|
{
|
||||||
this.keyspaceName = keyspaceName;
|
this.keyspaceName = keyspaceName;
|
||||||
this.key = key;
|
this.key = key;
|
||||||
this.modifications = modifications;
|
this.modifications = modifications;
|
||||||
this.cdcEnabled = cdcEnabled;
|
this.cdcEnabled = cdcEnabled;
|
||||||
this.approxCreatedAtNanos = approxCreatedAtNanos;
|
this.approxCreatedAtNanos = approxCreatedAtNanos;
|
||||||
|
this.allowOutOfRangeMutations = allowOutOfRangeMutations;
|
||||||
}
|
}
|
||||||
|
|
||||||
private static boolean cdcEnabled(Iterable<PartitionUpdate> modifications)
|
private static boolean cdcEnabled(Iterable<PartitionUpdate> modifications)
|
||||||
|
|
@ -125,7 +135,7 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return new Mutation(keyspaceName, key, builder.build(), approxCreatedAtNanos);
|
return new Mutation(keyspaceName, key, builder.build(), approxCreatedAtNanos, allowOutOfRangeMutations);
|
||||||
}
|
}
|
||||||
|
|
||||||
public Mutation without(TableId tableId)
|
public Mutation without(TableId tableId)
|
||||||
|
|
@ -201,18 +211,22 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
* @throws IllegalArgumentException if not all the mutations are on the same
|
* @throws IllegalArgumentException if not all the mutations are on the same
|
||||||
* keyspace and key.
|
* keyspace and key.
|
||||||
*/
|
*/
|
||||||
public static Mutation merge(List<Mutation> mutations)
|
public static Mutation merge(Collection<Mutation> mutations)
|
||||||
{
|
{
|
||||||
assert !mutations.isEmpty();
|
assert !mutations.isEmpty();
|
||||||
|
|
||||||
if (mutations.size() == 1)
|
if (mutations.size() == ALLOW_OUT_OF_RANGE_MUTATIONS_FLAG)
|
||||||
return mutations.get(0);
|
return mutations.iterator().next();
|
||||||
|
|
||||||
Set<TableId> updatedTables = new HashSet<>();
|
Set<TableId> updatedTables = new HashSet<>();
|
||||||
String ks = null;
|
String ks = null;
|
||||||
DecoratedKey key = null;
|
DecoratedKey key = null;
|
||||||
|
Boolean allowOutOfRangeMutations = null;
|
||||||
for (Mutation mutation : mutations)
|
for (Mutation mutation : mutations)
|
||||||
{
|
{
|
||||||
|
if (allowOutOfRangeMutations != null && allowOutOfRangeMutations != mutation.allowOutOfRangeMutations)
|
||||||
|
throw new IllegalArgumentException("Can't merge mutations with differing policies on allowing out of range mutations");
|
||||||
|
allowOutOfRangeMutations = mutation.allowOutOfRangeMutations;
|
||||||
updatedTables.addAll(mutation.modifications.keySet());
|
updatedTables.addAll(mutation.modifications.keySet());
|
||||||
if (ks != null && !ks.equals(mutation.keyspaceName))
|
if (ks != null && !ks.equals(mutation.keyspaceName))
|
||||||
throw new IllegalArgumentException();
|
throw new IllegalArgumentException();
|
||||||
|
|
@ -236,10 +250,10 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
if (updates.isEmpty())
|
if (updates.isEmpty())
|
||||||
continue;
|
continue;
|
||||||
|
|
||||||
modifications.put(table, updates.size() == 1 ? updates.get(0) : PartitionUpdate.merge(updates));
|
modifications.put(table, updates.size() == ALLOW_OUT_OF_RANGE_MUTATIONS_FLAG ? updates.get(0) : PartitionUpdate.merge(updates));
|
||||||
updates.clear();
|
updates.clear();
|
||||||
}
|
}
|
||||||
return new Mutation(ks, key, modifications.build(), approxTime.now());
|
return new Mutation(ks, key, modifications.build(), approxTime.now(), allowOutOfRangeMutations);
|
||||||
}
|
}
|
||||||
|
|
||||||
public Future<?> applyFuture()
|
public Future<?> applyFuture()
|
||||||
|
|
@ -296,6 +310,27 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
return cdcEnabled;
|
return cdcEnabled;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public Mutation allowOutOfRangeMutations()
|
||||||
|
{
|
||||||
|
allowOutOfRangeMutations = true;
|
||||||
|
return this;
|
||||||
|
}
|
||||||
|
|
||||||
|
public boolean allowsOutOfRangeMutations()
|
||||||
|
{
|
||||||
|
return allowOutOfRangeMutations;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static int allowsOutOfRangeMutationsFlag(boolean allowOutOfRangeMutations)
|
||||||
|
{
|
||||||
|
return allowOutOfRangeMutations ? ALLOW_OUT_OF_RANGE_MUTATIONS_FLAG : 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean allowsOutOfRangeMutations(int flags)
|
||||||
|
{
|
||||||
|
return (flags & ALLOW_OUT_OF_RANGE_MUTATIONS_FLAG) != 0;
|
||||||
|
}
|
||||||
|
|
||||||
public String toString()
|
public String toString()
|
||||||
{
|
{
|
||||||
return toString(false);
|
return toString(false);
|
||||||
|
|
@ -481,6 +516,9 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
{
|
{
|
||||||
Map<TableId, PartitionUpdate> modifications = mutation.modifications;
|
Map<TableId, PartitionUpdate> modifications = mutation.modifications;
|
||||||
|
|
||||||
|
if (version >= VERSION_51)
|
||||||
|
out.write(allowsOutOfRangeMutationsFlag(mutation.allowsOutOfRangeMutations()));
|
||||||
|
|
||||||
/* serialize the modifications in the mutation */
|
/* serialize the modifications in the mutation */
|
||||||
int size = modifications.size();
|
int size = modifications.size();
|
||||||
out.writeUnsignedVInt32(size);
|
out.writeUnsignedVInt32(size);
|
||||||
|
|
@ -500,6 +538,12 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
{
|
{
|
||||||
teeIn = new TeeDataInputPlus(in, dob, CACHEABLE_MUTATION_SIZE_LIMIT);
|
teeIn = new TeeDataInputPlus(in, dob, CACHEABLE_MUTATION_SIZE_LIMIT);
|
||||||
|
|
||||||
|
boolean allowsOutOfRangeMutations = false;
|
||||||
|
if (version >= VERSION_51)
|
||||||
|
{
|
||||||
|
int flags = in.readByte();
|
||||||
|
allowsOutOfRangeMutations = allowsOutOfRangeMutations(flags);
|
||||||
|
}
|
||||||
int size = teeIn.readUnsignedVInt32();
|
int size = teeIn.readUnsignedVInt32();
|
||||||
assert size > 0;
|
assert size > 0;
|
||||||
|
|
||||||
|
|
@ -519,7 +563,7 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
update = PartitionUpdate.serializer.deserialize(teeIn, version, flag);
|
update = PartitionUpdate.serializer.deserialize(teeIn, version, flag);
|
||||||
modifications.put(update.metadata().id, update);
|
modifications.put(update.metadata().id, update);
|
||||||
}
|
}
|
||||||
m = new Mutation(update.metadata().keyspace, dk, modifications.build(), approxTime.now());
|
m = new Mutation(update.metadata().keyspace, dk, modifications.build(), approxTime.now(), allowsOutOfRangeMutations);
|
||||||
}
|
}
|
||||||
|
|
||||||
//Only cache serializations that don't hit the limit
|
//Only cache serializations that don't hit the limit
|
||||||
|
|
@ -597,7 +641,9 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
long size = this.size;
|
long size = this.size;
|
||||||
if (size == 0L)
|
if (size == 0L)
|
||||||
{
|
{
|
||||||
size = TypeSizes.sizeofUnsignedVInt(mutation.modifications.size());
|
if (version >= VERSION_51)
|
||||||
|
size += ALLOW_OUT_OF_RANGE_MUTATIONS_FLAG; // flags
|
||||||
|
size += TypeSizes.sizeofUnsignedVInt(mutation.modifications.size());
|
||||||
for (PartitionUpdate partitionUpdate : mutation.modifications.values())
|
for (PartitionUpdate partitionUpdate : mutation.modifications.values())
|
||||||
size += serializer.serializedSize(partitionUpdate, version);
|
size += serializer.serializedSize(partitionUpdate, version);
|
||||||
this.size = size;
|
this.size = size;
|
||||||
|
|
@ -650,7 +696,7 @@ public class Mutation implements IMutation, Supplier<Mutation>
|
||||||
|
|
||||||
public Mutation build()
|
public Mutation build()
|
||||||
{
|
{
|
||||||
return new Mutation(keyspaceName, key, modifications.build(), approxCreatedAtNanos);
|
return new Mutation(keyspaceName, key, modifications.build(), approxCreatedAtNanos, false);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -71,6 +71,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -80,7 +81,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
Index.QueryPlan indexQueryPlan,
|
Index.QueryPlan indexQueryPlan,
|
||||||
boolean trackWarnings)
|
boolean trackWarnings)
|
||||||
{
|
{
|
||||||
super(serializedAtEpoch, Kind.PARTITION_RANGE, isDigest, digestVersion, acceptsTransient, metadata, nowInSec, columnFilter, rowFilter, limits, indexQueryPlan, trackWarnings, dataRange);
|
super(serializedAtEpoch, Kind.PARTITION_RANGE, isDigest, digestVersion, acceptsTransient, allowOutOfRangeReads, metadata, nowInSec, columnFilter, rowFilter, limits, indexQueryPlan, trackWarnings, dataRange);
|
||||||
this.requestedSlices = dataRange.clusteringIndexFilter.getSlices(metadata());
|
this.requestedSlices = dataRange.clusteringIndexFilter.getSlices(metadata());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -88,6 +89,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -115,6 +117,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
isDigest,
|
isDigest,
|
||||||
digestVersion,
|
digestVersion,
|
||||||
acceptsTransient,
|
acceptsTransient,
|
||||||
|
allowsOutOfRangeReads,
|
||||||
metadata,
|
metadata,
|
||||||
nowInSec,
|
nowInSec,
|
||||||
columnFilter,
|
columnFilter,
|
||||||
|
|
@ -136,6 +139,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
false,
|
false,
|
||||||
0,
|
0,
|
||||||
false,
|
false,
|
||||||
|
false,
|
||||||
metadata,
|
metadata,
|
||||||
nowInSec,
|
nowInSec,
|
||||||
columnFilter,
|
columnFilter,
|
||||||
|
|
@ -160,6 +164,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
false,
|
false,
|
||||||
0,
|
0,
|
||||||
false,
|
false,
|
||||||
|
false,
|
||||||
metadata,
|
metadata,
|
||||||
nowInSec,
|
nowInSec,
|
||||||
ColumnFilter.all(metadata),
|
ColumnFilter.all(metadata),
|
||||||
|
|
@ -206,6 +211,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -222,6 +228,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -239,6 +246,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
true,
|
true,
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
false,
|
false,
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -256,6 +264,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
false,
|
false,
|
||||||
0,
|
0,
|
||||||
true,
|
true,
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -273,6 +282,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -290,6 +300,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -525,6 +536,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -534,7 +546,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
throws IOException
|
throws IOException
|
||||||
{
|
{
|
||||||
DataRange range = DataRange.serializer.deserialize(in, version, metadata);
|
DataRange range = DataRange.serializer.deserialize(in, version, metadata);
|
||||||
return PartitionRangeReadCommand.create(serializedAtEpoch, isDigest, digestVersion, acceptsTransient, metadata, nowInSec, columnFilter, rowFilter, limits, range, indexQueryPlan, false);
|
return PartitionRangeReadCommand.create(serializedAtEpoch, isDigest, digestVersion, acceptsTransient, allowsOutOfRangeReads, metadata, nowInSec, columnFilter, rowFilter, limits, range, indexQueryPlan, false);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -552,7 +564,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR
|
||||||
Index.QueryPlan indexQueryPlan,
|
Index.QueryPlan indexQueryPlan,
|
||||||
boolean trackWarnings)
|
boolean trackWarnings)
|
||||||
{
|
{
|
||||||
super(metadata.epoch, isDigest, digestVersion, acceptsTransient, metadata, nowInSec, columnFilter, rowFilter, limits, dataRange, indexQueryPlan, trackWarnings);
|
super(metadata.epoch, isDigest, digestVersion, acceptsTransient, true, metadata, nowInSec, columnFilter, rowFilter, limits, dataRange, indexQueryPlan, trackWarnings);
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
|
|
|
||||||
|
|
@ -18,44 +18,62 @@
|
||||||
package org.apache.cassandra.db;
|
package org.apache.cassandra.db;
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Objects;
|
||||||
|
import java.util.Set;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.function.BiFunction;
|
import java.util.function.BiFunction;
|
||||||
import java.util.function.LongPredicate;
|
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
import java.util.function.LongPredicate;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.base.Preconditions;
|
import com.google.common.base.Preconditions;
|
||||||
import com.google.common.collect.Iterables;
|
import com.google.common.collect.Iterables;
|
||||||
import com.google.common.collect.Sets;
|
import com.google.common.collect.Sets;
|
||||||
|
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import io.netty.util.concurrent.FastThreadLocal;
|
import io.netty.util.concurrent.FastThreadLocal;
|
||||||
import org.apache.cassandra.config.*;
|
import org.apache.cassandra.config.CassandraRelevantProperties;
|
||||||
import org.apache.cassandra.db.filter.*;
|
import org.apache.cassandra.config.Config;
|
||||||
|
import org.apache.cassandra.config.DataStorageSpec;
|
||||||
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
|
import org.apache.cassandra.db.filter.ClusteringIndexFilter;
|
||||||
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
|
import org.apache.cassandra.db.filter.DataLimits;
|
||||||
|
import org.apache.cassandra.db.filter.LocalReadSizeTooLargeException;
|
||||||
|
import org.apache.cassandra.db.filter.RowFilter;
|
||||||
|
import org.apache.cassandra.db.filter.TombstoneOverwhelmingException;
|
||||||
|
import org.apache.cassandra.db.partitions.PurgeFunction;
|
||||||
|
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator;
|
||||||
|
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators;
|
||||||
|
import org.apache.cassandra.db.rows.Cell;
|
||||||
|
import org.apache.cassandra.db.rows.RangeTombstoneBoundMarker;
|
||||||
|
import org.apache.cassandra.db.rows.RangeTombstoneMarker;
|
||||||
|
import org.apache.cassandra.db.rows.Row;
|
||||||
|
import org.apache.cassandra.db.rows.Rows;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterators;
|
||||||
import org.apache.cassandra.db.transform.BasePartitions;
|
import org.apache.cassandra.db.transform.BasePartitions;
|
||||||
import org.apache.cassandra.db.transform.BaseRows;
|
import org.apache.cassandra.db.transform.BaseRows;
|
||||||
|
import org.apache.cassandra.db.transform.RTBoundCloser;
|
||||||
|
import org.apache.cassandra.db.transform.RTBoundValidator;
|
||||||
|
import org.apache.cassandra.db.transform.RTBoundValidator.Stage;
|
||||||
|
import org.apache.cassandra.db.transform.StoppingTransformation;
|
||||||
|
import org.apache.cassandra.db.transform.Transformation;
|
||||||
import org.apache.cassandra.exceptions.CoordinatorBehindException;
|
import org.apache.cassandra.exceptions.CoordinatorBehindException;
|
||||||
import org.apache.cassandra.exceptions.QueryCancelledException;
|
import org.apache.cassandra.exceptions.QueryCancelledException;
|
||||||
|
import org.apache.cassandra.exceptions.UnknownIndexException;
|
||||||
import org.apache.cassandra.exceptions.UnknownTableException;
|
import org.apache.cassandra.exceptions.UnknownTableException;
|
||||||
import org.apache.cassandra.metrics.TCMMetrics;
|
import org.apache.cassandra.metrics.TCMMetrics;
|
||||||
import org.apache.cassandra.net.MessageFlag;
|
import org.apache.cassandra.net.MessageFlag;
|
||||||
import org.apache.cassandra.net.MessagingService;
|
import org.apache.cassandra.net.MessagingService;
|
||||||
import org.apache.cassandra.net.ParamType;
|
import org.apache.cassandra.net.ParamType;
|
||||||
import org.apache.cassandra.net.Verb;
|
import org.apache.cassandra.net.Verb;
|
||||||
import org.apache.cassandra.db.partitions.*;
|
|
||||||
import org.apache.cassandra.db.rows.*;
|
|
||||||
import org.apache.cassandra.db.transform.RTBoundCloser;
|
|
||||||
import org.apache.cassandra.db.transform.RTBoundValidator;
|
|
||||||
import org.apache.cassandra.db.transform.RTBoundValidator.Stage;
|
|
||||||
import org.apache.cassandra.db.transform.StoppingTransformation;
|
|
||||||
import org.apache.cassandra.db.transform.Transformation;
|
|
||||||
import org.apache.cassandra.exceptions.UnknownIndexException;
|
|
||||||
import org.apache.cassandra.index.Index;
|
import org.apache.cassandra.index.Index;
|
||||||
import org.apache.cassandra.io.IVersionedSerializer;
|
import org.apache.cassandra.io.IVersionedSerializer;
|
||||||
import org.apache.cassandra.io.sstable.format.SSTableReader;
|
import org.apache.cassandra.io.sstable.format.SSTableReader;
|
||||||
|
|
@ -67,16 +85,16 @@ import org.apache.cassandra.net.Message;
|
||||||
import org.apache.cassandra.schema.IndexMetadata;
|
import org.apache.cassandra.schema.IndexMetadata;
|
||||||
import org.apache.cassandra.schema.Schema;
|
import org.apache.cassandra.schema.Schema;
|
||||||
import org.apache.cassandra.schema.SchemaConstants;
|
import org.apache.cassandra.schema.SchemaConstants;
|
||||||
|
import org.apache.cassandra.schema.SchemaProvider;
|
||||||
import org.apache.cassandra.schema.TableId;
|
import org.apache.cassandra.schema.TableId;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.schema.SchemaProvider;
|
|
||||||
import org.apache.cassandra.service.ActiveRepairService;
|
import org.apache.cassandra.service.ActiveRepairService;
|
||||||
import org.apache.cassandra.service.ClientWarn;
|
import org.apache.cassandra.service.ClientWarn;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
import org.apache.cassandra.tcm.Epoch;
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
import org.apache.cassandra.tracing.Tracing;
|
import org.apache.cassandra.tracing.Tracing;
|
||||||
import org.apache.cassandra.utils.CassandraUInt;
|
|
||||||
import org.apache.cassandra.transport.Dispatcher;
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
|
import org.apache.cassandra.utils.CassandraUInt;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
import org.apache.cassandra.utils.NoSpamLogger;
|
import org.apache.cassandra.utils.NoSpamLogger;
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
|
|
@ -84,8 +102,8 @@ import org.apache.cassandra.utils.TimeUUID;
|
||||||
|
|
||||||
import static com.google.common.collect.Iterables.any;
|
import static com.google.common.collect.Iterables.any;
|
||||||
import static com.google.common.collect.Iterables.filter;
|
import static com.google.common.collect.Iterables.filter;
|
||||||
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
|
|
||||||
import static org.apache.cassandra.db.partitions.UnfilteredPartitionIterators.MergeListener.NOOP;
|
import static org.apache.cassandra.db.partitions.UnfilteredPartitionIterators.MergeListener.NOOP;
|
||||||
|
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
|
||||||
import static org.apache.cassandra.utils.MonotonicClock.Global.approxTime;
|
import static org.apache.cassandra.utils.MonotonicClock.Global.approxTime;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -110,6 +128,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
private final boolean isDigestQuery;
|
private final boolean isDigestQuery;
|
||||||
private final boolean acceptsTransient;
|
private final boolean acceptsTransient;
|
||||||
private final Epoch serializedAtEpoch;
|
private final Epoch serializedAtEpoch;
|
||||||
|
private boolean allowsOutOfRangeReads;
|
||||||
// if a digest query, the version for which the digest is expected. Ignored if not a digest.
|
// if a digest query, the version for which the digest is expected. Ignored if not a digest.
|
||||||
private int digestVersion;
|
private int digestVersion;
|
||||||
|
|
||||||
|
|
@ -128,6 +147,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -154,6 +174,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
boolean isDigestQuery,
|
boolean isDigestQuery,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -172,6 +193,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
this.digestVersion = digestVersion;
|
this.digestVersion = digestVersion;
|
||||||
this.acceptsTransient = acceptsTransient;
|
this.acceptsTransient = acceptsTransient;
|
||||||
this.indexQueryPlan = indexQueryPlan;
|
this.indexQueryPlan = indexQueryPlan;
|
||||||
|
this.allowsOutOfRangeReads = allowsOutOfRangeReads;
|
||||||
this.trackWarnings = trackWarnings;
|
this.trackWarnings = trackWarnings;
|
||||||
this.serializedAtEpoch = serializedAtEpoch;
|
this.serializedAtEpoch = serializedAtEpoch;
|
||||||
this.dataRange = dataRange;
|
this.dataRange = dataRange;
|
||||||
|
|
@ -528,6 +550,17 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
return ReadExecutionController.forCommand(this, false);
|
return ReadExecutionController.forCommand(this, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public ReadCommand allowOutOfRangeReads()
|
||||||
|
{
|
||||||
|
allowsOutOfRangeReads = true;
|
||||||
|
return this;
|
||||||
|
}
|
||||||
|
|
||||||
|
public boolean allowsOutOfRangeReads()
|
||||||
|
{
|
||||||
|
return allowsOutOfRangeReads;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Wraps the provided iterator so that metrics on what is scanned by the command are recorded.
|
* Wraps the provided iterator so that metrics on what is scanned by the command are recorded.
|
||||||
* This also log warning/trow TombstoneOverwhelmingException if appropriate.
|
* This also log warning/trow TombstoneOverwhelmingException if appropriate.
|
||||||
|
|
@ -874,7 +907,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
// Skip purgeable tombstones. We do this because it's safe to do (post-merge of the memtable and sstable at least), it
|
// Skip purgeable tombstones. We do this because it's safe to do (post-merge of the memtable and sstable at least), it
|
||||||
// can save us some bandwith, and avoid making us throw a TombstoneOverwhelmingException for purgeable tombstones (which
|
// can save us some bandwith, and avoid making us throw a TombstoneOverwhelmingException for purgeable tombstones (which
|
||||||
// are to some extend an artefact of compaction lagging behind and hence counting them is somewhat unintuitive).
|
// are to some extend an artefact of compaction lagging behind and hence counting them is somewhat unintuitive).
|
||||||
protected UnfilteredPartitionIterator withoutPurgeableTombstones(UnfilteredPartitionIterator iterator,
|
protected UnfilteredPartitionIterator withoutPurgeableTombstones(UnfilteredPartitionIterator iterator,
|
||||||
ColumnFamilyStore cfs,
|
ColumnFamilyStore cfs,
|
||||||
ReadExecutionController controller)
|
ReadExecutionController controller)
|
||||||
{
|
{
|
||||||
|
|
@ -1217,6 +1250,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
private static final int HAS_INDEX = 0x04;
|
private static final int HAS_INDEX = 0x04;
|
||||||
private static final int ACCEPTS_TRANSIENT = 0x08;
|
private static final int ACCEPTS_TRANSIENT = 0x08;
|
||||||
private static final int NEEDS_RECONCILIATION = 0x10;
|
private static final int NEEDS_RECONCILIATION = 0x10;
|
||||||
|
private static final int ALLOWS_OUT_OF_RANGE_READS = 0x20;
|
||||||
|
|
||||||
private final SchemaProvider schema;
|
private final SchemaProvider schema;
|
||||||
|
|
||||||
|
|
@ -1281,6 +1315,16 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
return (flags & NEEDS_RECONCILIATION) != 0;
|
return (flags & NEEDS_RECONCILIATION) != 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static int allowsOutOfRangeReadsFlag(boolean allowsOutOfRangeReads)
|
||||||
|
{
|
||||||
|
return allowsOutOfRangeReads ? ALLOWS_OUT_OF_RANGE_READS: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean allowsOutOfRangeReads(int flags)
|
||||||
|
{
|
||||||
|
return (flags & ALLOWS_OUT_OF_RANGE_READS) != 0;
|
||||||
|
}
|
||||||
|
|
||||||
public void serialize(ReadCommand command, DataOutputPlus out, int version) throws IOException
|
public void serialize(ReadCommand command, DataOutputPlus out, int version) throws IOException
|
||||||
{
|
{
|
||||||
out.writeByte(command.kind.ordinal());
|
out.writeByte(command.kind.ordinal());
|
||||||
|
|
@ -1289,6 +1333,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
| indexFlag(null != command.indexQueryPlan())
|
| indexFlag(null != command.indexQueryPlan())
|
||||||
| acceptsTransientFlag(command.acceptsTransient())
|
| acceptsTransientFlag(command.acceptsTransient())
|
||||||
| needsReconciliationFlag(command.rowFilter().needsReconciliation())
|
| needsReconciliationFlag(command.rowFilter().needsReconciliation())
|
||||||
|
| allowsOutOfRangeReadsFlag(command.allowsOutOfRangeReads)
|
||||||
);
|
);
|
||||||
if (command.isDigestQuery())
|
if (command.isDigestQuery())
|
||||||
out.writeUnsignedVInt32(command.digestVersion());
|
out.writeUnsignedVInt32(command.digestVersion());
|
||||||
|
|
@ -1314,6 +1359,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
int flags = in.readByte();
|
int flags = in.readByte();
|
||||||
boolean isDigest = isDigest(flags);
|
boolean isDigest = isDigest(flags);
|
||||||
boolean acceptsTransient = acceptsTransient(flags);
|
boolean acceptsTransient = acceptsTransient(flags);
|
||||||
|
boolean allowsOutOfRangeReads = allowsOutOfRangeReads(flags);
|
||||||
// Shouldn't happen or it's a user error (see comment above) but
|
// Shouldn't happen or it's a user error (see comment above) but
|
||||||
// better complain loudly than doing the wrong thing.
|
// better complain loudly than doing the wrong thing.
|
||||||
if (isForThrift(flags))
|
if (isForThrift(flags))
|
||||||
|
|
@ -1359,7 +1405,7 @@ public abstract class ReadCommand extends AbstractReadQuery
|
||||||
indexQueryPlan = indexGroup.queryPlanFor(rowFilter);
|
indexQueryPlan = indexGroup.queryPlanFor(rowFilter);
|
||||||
}
|
}
|
||||||
|
|
||||||
return kind.selectionDeserializer.deserialize(in, version, schemaVersion, isDigest, digestVersion, acceptsTransient, tableMetadata, nowInSec, columnFilter, rowFilter, limits, indexQueryPlan);
|
return kind.selectionDeserializer.deserialize(in, version, schemaVersion, isDigest, digestVersion, acceptsTransient, allowsOutOfRangeReads, tableMetadata, nowInSec, columnFilter, rowFilter, limits, indexQueryPlan);
|
||||||
}
|
}
|
||||||
|
|
||||||
private IndexMetadata deserializeIndexMetadata(DataInputPlus in, int version, TableMetadata metadata) throws IOException
|
private IndexMetadata deserializeIndexMetadata(DataInputPlus in, int version, TableMetadata metadata) throws IOException
|
||||||
|
|
|
||||||
|
|
@ -22,21 +22,21 @@ import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator;
|
import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator;
|
||||||
import org.apache.cassandra.exceptions.CoordinatorBehindException;
|
|
||||||
import org.apache.cassandra.exceptions.InvalidRoutingException;
|
|
||||||
import org.apache.cassandra.exceptions.QueryCancelledException;
|
|
||||||
import org.apache.cassandra.dht.AbstractBounds;
|
import org.apache.cassandra.dht.AbstractBounds;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
|
import org.apache.cassandra.exceptions.CoordinatorBehindException;
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
|
import org.apache.cassandra.exceptions.InvalidRoutingException;
|
||||||
|
import org.apache.cassandra.exceptions.QueryCancelledException;
|
||||||
import org.apache.cassandra.locator.Replica;
|
import org.apache.cassandra.locator.Replica;
|
||||||
import org.apache.cassandra.metrics.TCMMetrics;
|
import org.apache.cassandra.metrics.TCMMetrics;
|
||||||
import org.apache.cassandra.schema.SchemaConstants;
|
|
||||||
import org.apache.cassandra.service.StorageService;
|
|
||||||
import org.apache.cassandra.tcm.ClusterMetadataService;
|
|
||||||
import org.apache.cassandra.net.IVerbHandler;
|
import org.apache.cassandra.net.IVerbHandler;
|
||||||
import org.apache.cassandra.net.Message;
|
import org.apache.cassandra.net.Message;
|
||||||
import org.apache.cassandra.net.MessagingService;
|
import org.apache.cassandra.net.MessagingService;
|
||||||
|
import org.apache.cassandra.schema.SchemaConstants;
|
||||||
|
import org.apache.cassandra.service.StorageService;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
|
import org.apache.cassandra.tcm.ClusterMetadataService;
|
||||||
import org.apache.cassandra.tcm.Epoch;
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
import org.apache.cassandra.tracing.Tracing;
|
import org.apache.cassandra.tracing.Tracing;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
|
|
@ -49,6 +49,18 @@ public class ReadCommandVerbHandler implements IVerbHandler<ReadCommand>
|
||||||
|
|
||||||
private static final Logger logger = LoggerFactory.getLogger(ReadCommandVerbHandler.class);
|
private static final Logger logger = LoggerFactory.getLogger(ReadCommandVerbHandler.class);
|
||||||
|
|
||||||
|
public ReadResponse doRead(ReadCommand command, boolean trackRepairedData)
|
||||||
|
{
|
||||||
|
ReadResponse response;
|
||||||
|
try (ReadExecutionController controller = command.executionController(trackRepairedData);
|
||||||
|
UnfilteredPartitionIterator iterator = command.executeLocally(controller))
|
||||||
|
{
|
||||||
|
response = command.createResponse(iterator, controller.getRepairedDataInfo());
|
||||||
|
}
|
||||||
|
|
||||||
|
return response;
|
||||||
|
}
|
||||||
|
|
||||||
public void doVerb(Message<ReadCommand> message)
|
public void doVerb(Message<ReadCommand> message)
|
||||||
{
|
{
|
||||||
if (message.epoch().isAfter(Epoch.EMPTY))
|
if (message.epoch().isAfter(Epoch.EMPTY))
|
||||||
|
|
@ -68,10 +80,9 @@ public class ReadCommandVerbHandler implements IVerbHandler<ReadCommand>
|
||||||
command.trackWarnings();
|
command.trackWarnings();
|
||||||
|
|
||||||
ReadResponse response;
|
ReadResponse response;
|
||||||
try (ReadExecutionController controller = command.executionController(message.trackRepairedData());
|
try
|
||||||
UnfilteredPartitionIterator iterator = command.executeLocally(controller))
|
|
||||||
{
|
{
|
||||||
response = command.createResponse(iterator, controller.getRepairedDataInfo());
|
response = doRead(command, message.trackRepairedData());
|
||||||
}
|
}
|
||||||
catch (RejectException e)
|
catch (RejectException e)
|
||||||
{
|
{
|
||||||
|
|
@ -147,15 +158,21 @@ public class ReadCommandVerbHandler implements IVerbHandler<ReadCommand>
|
||||||
private ClusterMetadata checkTokenOwnership(ClusterMetadata metadata, Message<ReadCommand> message)
|
private ClusterMetadata checkTokenOwnership(ClusterMetadata metadata, Message<ReadCommand> message)
|
||||||
{
|
{
|
||||||
ReadCommand command = message.payload;
|
ReadCommand command = message.payload;
|
||||||
|
|
||||||
if (command.metadata().isVirtual())
|
if (command.metadata().isVirtual())
|
||||||
return metadata;
|
return metadata;
|
||||||
|
|
||||||
|
// Some read commands may be sent using an older Epoch intentionally so validating using the current Epoch
|
||||||
|
// doesn't work
|
||||||
|
if (command.allowsOutOfRangeReads())
|
||||||
|
return metadata;
|
||||||
|
|
||||||
if (command.isTopK())
|
if (command.isTopK())
|
||||||
return metadata;
|
return metadata;
|
||||||
|
|
||||||
if (command instanceof SinglePartitionReadCommand)
|
if (command instanceof SinglePartitionReadCommand)
|
||||||
{
|
{
|
||||||
Token token = ((SinglePartitionReadCommand) command).partitionKey().getToken();
|
Token token = ((SinglePartitionReadCommand)command).partitionKey().getToken();
|
||||||
Replica localReplica = getLocalReplica(metadata, token, command.metadata().keyspace);
|
Replica localReplica = getLocalReplica(metadata, token, command.metadata().keyspace);
|
||||||
if (localReplica == null)
|
if (localReplica == null)
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -25,9 +25,14 @@ public class ReadRepairVerbHandler extends AbstractMutationVerbHandler<Mutation>
|
||||||
{
|
{
|
||||||
public static final ReadRepairVerbHandler instance = new ReadRepairVerbHandler();
|
public static final ReadRepairVerbHandler instance = new ReadRepairVerbHandler();
|
||||||
|
|
||||||
|
public void applyMutation(Mutation mutation)
|
||||||
|
{
|
||||||
|
mutation.apply();
|
||||||
|
}
|
||||||
|
|
||||||
void applyMutation(Message<Mutation> message, InetAddressAndPort respondToAddress)
|
void applyMutation(Message<Mutation> message, InetAddressAndPort respondToAddress)
|
||||||
{
|
{
|
||||||
message.payload.apply();
|
applyMutation(message.payload);
|
||||||
MessagingService.instance().send(message.emptyResponse(), respondToAddress);
|
MessagingService.instance().send(message.emptyResponse(), respondToAddress);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -98,6 +98,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -109,7 +110,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
boolean trackWarnings,
|
boolean trackWarnings,
|
||||||
DataRange dataRange)
|
DataRange dataRange)
|
||||||
{
|
{
|
||||||
super(serializedAtEpoch, Kind.SINGLE_PARTITION, isDigest, digestVersion, acceptsTransient, metadata, nowInSec, columnFilter, rowFilter, limits, indexQueryPlan, trackWarnings, dataRange);
|
super(serializedAtEpoch, Kind.SINGLE_PARTITION, isDigest, digestVersion, acceptsTransient, allowsOutOfRangeReads, metadata, nowInSec, columnFilter, rowFilter, limits, indexQueryPlan, trackWarnings, dataRange);
|
||||||
assert partitionKey.getPartitioner() == metadata.partitioner;
|
assert partitionKey.getPartitioner() == metadata.partitioner;
|
||||||
this.partitionKey = partitionKey;
|
this.partitionKey = partitionKey;
|
||||||
this.clusteringIndexFilter = clusteringIndexFilter;
|
this.clusteringIndexFilter = clusteringIndexFilter;
|
||||||
|
|
@ -119,6 +120,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -152,6 +154,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
isDigest,
|
isDigest,
|
||||||
digestVersion,
|
digestVersion,
|
||||||
acceptsTransient,
|
acceptsTransient,
|
||||||
|
allowsOutOfRangeReads,
|
||||||
metadata,
|
metadata,
|
||||||
nowInSec,
|
nowInSec,
|
||||||
columnFilter,
|
columnFilter,
|
||||||
|
|
@ -191,6 +194,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
false,
|
false,
|
||||||
0,
|
0,
|
||||||
false,
|
false,
|
||||||
|
false,
|
||||||
metadata,
|
metadata,
|
||||||
nowInSec,
|
nowInSec,
|
||||||
columnFilter,
|
columnFilter,
|
||||||
|
|
@ -369,6 +373,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -387,6 +392,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
true,
|
true,
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -405,6 +411,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
false,
|
false,
|
||||||
0,
|
0,
|
||||||
true,
|
true,
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -423,6 +430,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec(),
|
nowInSec(),
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -440,6 +448,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
isDigestQuery(),
|
isDigestQuery(),
|
||||||
digestVersion(),
|
digestVersion(),
|
||||||
acceptsTransient(),
|
acceptsTransient(),
|
||||||
|
allowsOutOfRangeReads(),
|
||||||
metadata(),
|
metadata(),
|
||||||
nowInSec,
|
nowInSec,
|
||||||
columnFilter(),
|
columnFilter(),
|
||||||
|
|
@ -1342,6 +1351,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
boolean isDigest,
|
boolean isDigest,
|
||||||
int digestVersion,
|
int digestVersion,
|
||||||
boolean acceptsTransient,
|
boolean acceptsTransient,
|
||||||
|
boolean allowsOutOfRangeReads,
|
||||||
TableMetadata metadata,
|
TableMetadata metadata,
|
||||||
long nowInSec,
|
long nowInSec,
|
||||||
ColumnFilter columnFilter,
|
ColumnFilter columnFilter,
|
||||||
|
|
@ -1352,7 +1362,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
{
|
{
|
||||||
DecoratedKey key = metadata.partitioner.decorateKey(metadata.partitionKeyType.readBuffer(in, DatabaseDescriptor.getMaxValueSize()));
|
DecoratedKey key = metadata.partitioner.decorateKey(metadata.partitionKeyType.readBuffer(in, DatabaseDescriptor.getMaxValueSize()));
|
||||||
ClusteringIndexFilter filter = ClusteringIndexFilter.serializer.deserialize(in, version, metadata);
|
ClusteringIndexFilter filter = ClusteringIndexFilter.serializer.deserialize(in, version, metadata);
|
||||||
return SinglePartitionReadCommand.create(serializedAtEpoch, isDigest, digestVersion, acceptsTransient, metadata, nowInSec, columnFilter, rowFilter, limits, key, filter, indexQueryPlan, false);
|
return SinglePartitionReadCommand.create(serializedAtEpoch, isDigest, digestVersion, acceptsTransient, allowsOutOfRangeReads, metadata, nowInSec, columnFilter, rowFilter, limits, key, filter, indexQueryPlan, false);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -1400,7 +1410,7 @@ public class SinglePartitionReadCommand extends ReadCommand implements SinglePar
|
||||||
boolean trackWarnings,
|
boolean trackWarnings,
|
||||||
DataRange dataRange)
|
DataRange dataRange)
|
||||||
{
|
{
|
||||||
super(metadata.epoch, isDigest, digestVersion, acceptsTransient, metadata, nowInSec, columnFilter,
|
super(metadata.epoch, isDigest, digestVersion, acceptsTransient, true, metadata, nowInSec, columnFilter,
|
||||||
rowFilter, limits, partitionKey, clusteringIndexFilter, indexQueryPlan, trackWarnings, dataRange);
|
rowFilter, limits, partitionKey, clusteringIndexFilter, indexQueryPlan, trackWarnings, dataRange);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -136,6 +136,8 @@ import static org.apache.cassandra.config.DatabaseDescriptor.paxosStatePurging;
|
||||||
import static org.apache.cassandra.cql3.QueryProcessor.executeInternal;
|
import static org.apache.cassandra.cql3.QueryProcessor.executeInternal;
|
||||||
import static org.apache.cassandra.cql3.QueryProcessor.executeInternalWithNowInSec;
|
import static org.apache.cassandra.cql3.QueryProcessor.executeInternalWithNowInSec;
|
||||||
import static org.apache.cassandra.cql3.QueryProcessor.executeOnceInternal;
|
import static org.apache.cassandra.cql3.QueryProcessor.executeOnceInternal;
|
||||||
|
import static org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.ConsensusMigratedAt;
|
||||||
|
import static org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.ConsensusMigrationTarget;
|
||||||
import static org.apache.cassandra.gms.ApplicationState.DC;
|
import static org.apache.cassandra.gms.ApplicationState.DC;
|
||||||
import static org.apache.cassandra.gms.ApplicationState.HOST_ID;
|
import static org.apache.cassandra.gms.ApplicationState.HOST_ID;
|
||||||
import static org.apache.cassandra.gms.ApplicationState.INTERNAL_ADDRESS_AND_PORT;
|
import static org.apache.cassandra.gms.ApplicationState.INTERNAL_ADDRESS_AND_PORT;
|
||||||
|
|
@ -162,6 +164,7 @@ public final class SystemKeyspace
|
||||||
|
|
||||||
public static final String BATCHES = "batches";
|
public static final String BATCHES = "batches";
|
||||||
public static final String PAXOS = "paxos";
|
public static final String PAXOS = "paxos";
|
||||||
|
public static final String CONSENSUS_MIGRATION_STATE = "consensus_migration_state";
|
||||||
public static final String PAXOS_REPAIR_HISTORY = "paxos_repair_history";
|
public static final String PAXOS_REPAIR_HISTORY = "paxos_repair_history";
|
||||||
public static final String PAXOS_REPAIR_STATE = "_paxos_repair_state";
|
public static final String PAXOS_REPAIR_STATE = "_paxos_repair_state";
|
||||||
public static final String BUILT_INDEXES = "IndexInfo";
|
public static final String BUILT_INDEXES = "IndexInfo";
|
||||||
|
|
@ -190,6 +193,7 @@ public final class SystemKeyspace
|
||||||
*/
|
*/
|
||||||
public static final Set<String> TABLES_SPLIT_ACROSS_MULTIPLE_DISKS = ImmutableSet.of(BATCHES,
|
public static final Set<String> TABLES_SPLIT_ACROSS_MULTIPLE_DISKS = ImmutableSet.of(BATCHES,
|
||||||
PAXOS,
|
PAXOS,
|
||||||
|
CONSENSUS_MIGRATION_STATE,
|
||||||
COMPACTION_HISTORY,
|
COMPACTION_HISTORY,
|
||||||
PREPARED_STATEMENTS,
|
PREPARED_STATEMENTS,
|
||||||
REPAIRS);
|
REPAIRS);
|
||||||
|
|
@ -215,14 +219,14 @@ public final class SystemKeyspace
|
||||||
TABLE_ESTIMATES_TYPE_LOCAL_PRIMARY, AVAILABLE_RANGES_V2, TRANSFERRED_RANGES_V2, VIEW_BUILDS_IN_PROGRESS,
|
TABLE_ESTIMATES_TYPE_LOCAL_PRIMARY, AVAILABLE_RANGES_V2, TRANSFERRED_RANGES_V2, VIEW_BUILDS_IN_PROGRESS,
|
||||||
BUILT_VIEWS, PREPARED_STATEMENTS, REPAIRS, TOP_PARTITIONS, LEGACY_PEERS, LEGACY_PEER_EVENTS,
|
BUILT_VIEWS, PREPARED_STATEMENTS, REPAIRS, TOP_PARTITIONS, LEGACY_PEERS, LEGACY_PEER_EVENTS,
|
||||||
LEGACY_TRANSFERRED_RANGES, LEGACY_AVAILABLE_RANGES, LEGACY_SIZE_ESTIMATES, LEGACY_SSTABLE_ACTIVITY,
|
LEGACY_TRANSFERRED_RANGES, LEGACY_AVAILABLE_RANGES, LEGACY_SIZE_ESTIMATES, LEGACY_SSTABLE_ACTIVITY,
|
||||||
METADATA_LOG, SNAPSHOT_TABLE_NAME);
|
METADATA_LOG, SNAPSHOT_TABLE_NAME, CONSENSUS_MIGRATION_STATE);
|
||||||
|
|
||||||
public static final Set<String> TABLE_NAMES = ImmutableSet.of(
|
public static final Set<String> TABLE_NAMES = ImmutableSet.of(
|
||||||
BATCHES, PAXOS, PAXOS_REPAIR_HISTORY, BUILT_INDEXES, LOCAL, PEERS_V2, PEER_EVENTS_V2,
|
BATCHES, PAXOS, PAXOS_REPAIR_HISTORY, BUILT_INDEXES, LOCAL, PEERS_V2, PEER_EVENTS_V2,
|
||||||
COMPACTION_HISTORY, SSTABLE_ACTIVITY_V2, TABLE_ESTIMATES, AVAILABLE_RANGES_V2, TRANSFERRED_RANGES_V2, VIEW_BUILDS_IN_PROGRESS,
|
COMPACTION_HISTORY, SSTABLE_ACTIVITY_V2, TABLE_ESTIMATES, AVAILABLE_RANGES_V2, TRANSFERRED_RANGES_V2, VIEW_BUILDS_IN_PROGRESS,
|
||||||
BUILT_VIEWS, PREPARED_STATEMENTS, REPAIRS, TOP_PARTITIONS, LEGACY_PEERS, LEGACY_PEER_EVENTS,
|
BUILT_VIEWS, PREPARED_STATEMENTS, REPAIRS, TOP_PARTITIONS, LEGACY_PEERS, LEGACY_PEER_EVENTS,
|
||||||
LEGACY_TRANSFERRED_RANGES, LEGACY_AVAILABLE_RANGES, LEGACY_SIZE_ESTIMATES, LEGACY_SSTABLE_ACTIVITY,
|
LEGACY_TRANSFERRED_RANGES, LEGACY_AVAILABLE_RANGES, LEGACY_SIZE_ESTIMATES, LEGACY_SSTABLE_ACTIVITY,
|
||||||
METADATA_LOG, SNAPSHOT_TABLE_NAME);
|
METADATA_LOG, SNAPSHOT_TABLE_NAME, CONSENSUS_MIGRATION_STATE);
|
||||||
|
|
||||||
public static final TableMetadata Batches =
|
public static final TableMetadata Batches =
|
||||||
parse(BATCHES,
|
parse(BATCHES,
|
||||||
|
|
@ -255,6 +259,25 @@ public final class SystemKeyspace
|
||||||
.indexes(PaxosUncommittedIndex.indexes())
|
.indexes(PaxosUncommittedIndex.indexes())
|
||||||
.build();
|
.build();
|
||||||
|
|
||||||
|
private static final TableMetadata ConsensusMigrationState =
|
||||||
|
parse(CONSENSUS_MIGRATION_STATE,
|
||||||
|
"Keys that have been migrated to another consensus protocol",
|
||||||
|
"CREATE TABLE %s ("
|
||||||
|
+ "row_key blob, "
|
||||||
|
+ "cf_id UUID, "
|
||||||
|
+ "consensus_migrated_at_epoch bigint, "
|
||||||
|
+ "consensus_target tinyint, "
|
||||||
|
+ "PRIMARY KEY ((row_key), cf_id, consensus_migrated_at_epoch)) "
|
||||||
|
+ "WITH CLUSTERING ORDER BY (cf_id ASC, consensus_migrated_at_epoch DESC)")
|
||||||
|
.compaction(CompactionParams.twcs(
|
||||||
|
ImmutableMap.of(
|
||||||
|
"compaction_window_unit", "MINUTES",
|
||||||
|
"compaction_window_size",
|
||||||
|
// 7 days divided into 30 windows
|
||||||
|
String.valueOf((7 * 24 * 60) / 30))))
|
||||||
|
.defaultTimeToLive((int)TimeUnit.DAYS.toSeconds(7))
|
||||||
|
.build();
|
||||||
|
|
||||||
private static final TableMetadata BuiltIndexes =
|
private static final TableMetadata BuiltIndexes =
|
||||||
parse(BUILT_INDEXES,
|
parse(BUILT_INDEXES,
|
||||||
"built column indexes",
|
"built column indexes",
|
||||||
|
|
@ -602,7 +625,8 @@ public final class SystemKeyspace
|
||||||
Repairs,
|
Repairs,
|
||||||
TopPartitions,
|
TopPartitions,
|
||||||
LocalMetadataLog,
|
LocalMetadataLog,
|
||||||
Snapshots);
|
Snapshots,
|
||||||
|
ConsensusMigrationState);
|
||||||
}
|
}
|
||||||
|
|
||||||
private static volatile Map<TableId, Pair<CommitLogPosition, Long>> truncationRecords;
|
private static volatile Map<TableId, Pair<CommitLogPosition, Long>> truncationRecords;
|
||||||
|
|
@ -1598,6 +1622,27 @@ public final class SystemKeyspace
|
||||||
return PaxosRepairHistory.fromTupleBufferList(keyspace, table, points);
|
return PaxosRepairHistory.fromTupleBufferList(keyspace, table, points);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static void saveConsensusKeyMigrationState(ByteBuffer partitionKey, UUID cfId, ConsensusMigratedAt consensusMigratedAt)
|
||||||
|
{
|
||||||
|
String cql = "UPDATE system." + CONSENSUS_MIGRATION_STATE + " SET consensus_target = ? WHERE row_key = ? AND cf_id = ? AND consensus_migrated_at_epoch = ?";
|
||||||
|
executeInternal(cql, consensusMigratedAt.migratedAtTarget.value, partitionKey, cfId, consensusMigratedAt.migratedAtEpoch.getEpoch());
|
||||||
|
}
|
||||||
|
|
||||||
|
public static ConsensusMigratedAt loadConsensusKeyMigrationState(ByteBuffer partitionKey, UUID cfId)
|
||||||
|
{
|
||||||
|
String cql = "SELECT consensus_migrated_at_epoch, consensus_target FROM system." + CONSENSUS_MIGRATION_STATE + " WHERE row_key = ? AND cf_id = ? LIMIT 1";
|
||||||
|
UntypedResultSet results = executeInternal(cql, partitionKey, cfId);
|
||||||
|
|
||||||
|
if (results.isEmpty())
|
||||||
|
return null;
|
||||||
|
|
||||||
|
UntypedResultSet.Row row = results.one();
|
||||||
|
// TODO Period won't be necessary eventually
|
||||||
|
Epoch migratedAtEpoch = Epoch.create(row.getLong("consensus_migrated_at_epoch"));
|
||||||
|
ConsensusMigrationTarget target = ConsensusMigrationTarget.fromValue(row.getByte("consensus_target"));
|
||||||
|
return new ConsensusMigratedAt(migratedAtEpoch, target);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Returns a RestorableMeter tracking the average read rate of a particular SSTable, restoring the last-seen rate
|
* Returns a RestorableMeter tracking the average read rate of a particular SSTable, restoring the last-seen rate
|
||||||
* from values in system.sstable_activity if present.
|
* from values in system.sstable_activity if present.
|
||||||
|
|
|
||||||
|
|
@ -24,12 +24,32 @@ import java.util.NavigableSet;
|
||||||
|
|
||||||
import com.google.common.collect.Iterators;
|
import com.google.common.collect.Iterators;
|
||||||
|
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.db.Clustering;
|
||||||
import org.apache.cassandra.db.*;
|
import org.apache.cassandra.db.ClusteringBound;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DeletionInfo;
|
||||||
|
import org.apache.cassandra.db.DeletionTime;
|
||||||
|
import org.apache.cassandra.db.MutableDeletionInfo;
|
||||||
|
import org.apache.cassandra.db.RangeTombstone;
|
||||||
|
import org.apache.cassandra.db.RegularAndStaticColumns;
|
||||||
|
import org.apache.cassandra.db.Slice;
|
||||||
|
import org.apache.cassandra.db.Slices;
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
import org.apache.cassandra.db.rows.*;
|
import org.apache.cassandra.db.rows.AbstractUnfilteredRowIterator;
|
||||||
|
import org.apache.cassandra.db.rows.BTreeRow;
|
||||||
|
import org.apache.cassandra.db.rows.EncodingStats;
|
||||||
|
import org.apache.cassandra.db.rows.RangeTombstoneMarker;
|
||||||
|
import org.apache.cassandra.db.rows.Row;
|
||||||
|
import org.apache.cassandra.db.rows.RowAndDeletionMergeIterator;
|
||||||
|
import org.apache.cassandra.db.rows.RowIterator;
|
||||||
|
import org.apache.cassandra.db.rows.Rows;
|
||||||
|
import org.apache.cassandra.db.rows.Unfiltered;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterators;
|
||||||
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.utils.SearchIterator;
|
import org.apache.cassandra.utils.SearchIterator;
|
||||||
import org.apache.cassandra.utils.btree.BTree;
|
import org.apache.cassandra.utils.btree.BTree;
|
||||||
|
import org.apache.cassandra.utils.btree.BTree.Dir;
|
||||||
|
|
||||||
import static org.apache.cassandra.utils.btree.BTree.Dir.desc;
|
import static org.apache.cassandra.utils.btree.BTree.Dir.desc;
|
||||||
|
|
||||||
|
|
@ -403,9 +423,15 @@ public abstract class AbstractBTreePartition implements Partition, Iterable<Row>
|
||||||
return BTree.size(holder().tree);
|
return BTree.size(holder().tree);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
public Iterator<Row> iterator()
|
public Iterator<Row> iterator()
|
||||||
{
|
{
|
||||||
return BTree.<Row>iterator(holder().tree);
|
return iterator(false);
|
||||||
|
}
|
||||||
|
|
||||||
|
public Iterator<Row> iterator(boolean reverse)
|
||||||
|
{
|
||||||
|
return BTree.<Row>iterator(holder().tree, reverse ? Dir.DESC : Dir.ASC);
|
||||||
}
|
}
|
||||||
|
|
||||||
public Row lastRow()
|
public Row lastRow()
|
||||||
|
|
|
||||||
|
|
@ -25,12 +25,17 @@ import java.util.concurrent.atomic.AtomicReferenceFieldUpdater;
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
|
||||||
import org.apache.cassandra.index.transactions.UpdateTransaction;
|
import org.apache.cassandra.index.transactions.UpdateTransaction;
|
||||||
|
|
||||||
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
|
import org.apache.cassandra.db.Clustering;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DeletionInfo;
|
||||||
|
import org.apache.cassandra.db.Slices;
|
||||||
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
|
import org.apache.cassandra.db.rows.Row;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.schema.TableMetadataRef;
|
import org.apache.cassandra.schema.TableMetadataRef;
|
||||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
|
||||||
import org.apache.cassandra.db.*;
|
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
|
||||||
import org.apache.cassandra.db.rows.*;
|
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
import org.apache.cassandra.utils.concurrent.OpOrder;
|
import org.apache.cassandra.utils.concurrent.OpOrder;
|
||||||
import org.apache.cassandra.utils.memory.Cloner;
|
import org.apache.cassandra.utils.memory.Cloner;
|
||||||
|
|
@ -223,9 +228,9 @@ public final class AtomicBTreePartition extends AbstractBTreePartition
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Iterator<Row> iterator()
|
public Iterator<Row> iterator(boolean reverse)
|
||||||
{
|
{
|
||||||
return allocator.ensureOnHeap().applyToPartition(super.iterator());
|
return allocator.ensureOnHeap().applyToPartition(super.iterator(reverse));
|
||||||
}
|
}
|
||||||
|
|
||||||
private boolean shouldLock(OpOrder.Group writeOp)
|
private boolean shouldLock(OpOrder.Group writeOp)
|
||||||
|
|
|
||||||
|
|
@ -19,11 +19,12 @@ package org.apache.cassandra.db.partitions;
|
||||||
|
|
||||||
import java.util.Iterator;
|
import java.util.Iterator;
|
||||||
|
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.DeletionInfo;
|
import org.apache.cassandra.db.DeletionInfo;
|
||||||
import org.apache.cassandra.db.RegularAndStaticColumns;
|
import org.apache.cassandra.db.RegularAndStaticColumns;
|
||||||
import org.apache.cassandra.db.rows.*;
|
import org.apache.cassandra.db.rows.Row;
|
||||||
|
import org.apache.cassandra.db.rows.RowIterator;
|
||||||
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.utils.btree.BTree;
|
import org.apache.cassandra.utils.btree.BTree;
|
||||||
|
|
||||||
public class FilteredPartition extends ImmutableBTreePartition
|
public class FilteredPartition extends ImmutableBTreePartition
|
||||||
|
|
@ -49,9 +50,9 @@ public class FilteredPartition extends ImmutableBTreePartition
|
||||||
return BTree.findByIndex(holder.tree, idx);
|
return BTree.findByIndex(holder.tree, idx);
|
||||||
}
|
}
|
||||||
|
|
||||||
public RowIterator rowIterator()
|
public RowIterator rowIterator(boolean reverse)
|
||||||
{
|
{
|
||||||
final Iterator<Row> iter = iterator();
|
final Iterator<Row> iter = iterator(reverse);
|
||||||
return new RowIterator()
|
return new RowIterator()
|
||||||
{
|
{
|
||||||
public TableMetadata metadata()
|
public TableMetadata metadata()
|
||||||
|
|
@ -61,7 +62,7 @@ public class FilteredPartition extends ImmutableBTreePartition
|
||||||
|
|
||||||
public boolean isReverseOrder()
|
public boolean isReverseOrder()
|
||||||
{
|
{
|
||||||
return false;
|
return reverse;
|
||||||
}
|
}
|
||||||
|
|
||||||
public RegularAndStaticColumns columns()
|
public RegularAndStaticColumns columns()
|
||||||
|
|
|
||||||
|
|
@ -24,17 +24,45 @@ import java.util.ArrayList;
|
||||||
import java.util.HashSet;
|
import java.util.HashSet;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
import com.google.common.base.Function;
|
||||||
import com.google.common.collect.Iterables;
|
import com.google.common.collect.Iterables;
|
||||||
import com.google.common.collect.Lists;
|
import com.google.common.collect.Lists;
|
||||||
import com.google.common.primitives.Ints;
|
import com.google.common.primitives.Ints;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import org.apache.cassandra.db.*;
|
import org.apache.cassandra.db.Clustering;
|
||||||
|
import org.apache.cassandra.db.Columns;
|
||||||
|
import org.apache.cassandra.db.CounterMutation;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DeletionInfo;
|
||||||
|
import org.apache.cassandra.db.DeletionTime;
|
||||||
|
import org.apache.cassandra.db.MutableDeletionInfo;
|
||||||
|
import org.apache.cassandra.db.Mutation;
|
||||||
|
import org.apache.cassandra.db.RangeTombstone;
|
||||||
|
import org.apache.cassandra.db.RegularAndStaticColumns;
|
||||||
|
import org.apache.cassandra.db.SimpleBuilders;
|
||||||
|
import org.apache.cassandra.db.Slices;
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
import org.apache.cassandra.db.rows.*;
|
import org.apache.cassandra.db.rows.BTreeRow;
|
||||||
|
import org.apache.cassandra.db.rows.Cell;
|
||||||
|
import org.apache.cassandra.db.rows.CellPath;
|
||||||
|
import org.apache.cassandra.db.rows.ColumnData;
|
||||||
|
import org.apache.cassandra.db.rows.ComplexColumnData;
|
||||||
|
import org.apache.cassandra.db.rows.DeserializationHelper;
|
||||||
|
import org.apache.cassandra.db.rows.EncodingStats;
|
||||||
|
import org.apache.cassandra.db.rows.RangeTombstoneMarker;
|
||||||
|
import org.apache.cassandra.db.rows.Row;
|
||||||
|
import org.apache.cassandra.db.rows.RowIterator;
|
||||||
|
import org.apache.cassandra.db.rows.RowIterators;
|
||||||
|
import org.apache.cassandra.db.rows.Rows;
|
||||||
|
import org.apache.cassandra.db.rows.Unfiltered;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIteratorSerializer;
|
||||||
|
import org.apache.cassandra.db.rows.UnfilteredRowIterators;
|
||||||
import org.apache.cassandra.exceptions.CoordinatorBehindException;
|
import org.apache.cassandra.exceptions.CoordinatorBehindException;
|
||||||
import org.apache.cassandra.exceptions.UnknownTableException;
|
import org.apache.cassandra.exceptions.UnknownTableException;
|
||||||
import org.apache.cassandra.index.IndexRegistry;
|
import org.apache.cassandra.index.IndexRegistry;
|
||||||
|
|
@ -1109,11 +1137,11 @@ public class PartitionUpdate extends AbstractBTreePartition
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
public Builder updateAllTimestampAndLocalDeletionTime(long newTimestamp, int newLocalDeletionTime)
|
public Builder updateTimesAndPathsForAccord(@Nonnull Function<Cell, CellPath> cellToMaybeNewListPath, long newTimestamp, int newLocalDeletionTime)
|
||||||
{
|
{
|
||||||
deletionInfo.updateAllTimestampAndLocalDeletionTime(newTimestamp - 1, newLocalDeletionTime);
|
deletionInfo.updateAllTimestampAndLocalDeletionTime(newTimestamp - 1, newLocalDeletionTime);
|
||||||
tree = BTree.<Row, Row>transformAndFilter(tree, (x) -> x.updateAllTimestampAndLocalDeletionTime(newTimestamp, newLocalDeletionTime));
|
tree = BTree.<Row, Row>transformAndFilter(tree, (x) -> x.updateTimesAndPathsForAccord(cellToMaybeNewListPath, newTimestamp, newLocalDeletionTime));
|
||||||
staticRow = this.staticRow.updateAllTimestampAndLocalDeletionTime(newTimestamp, newLocalDeletionTime);
|
staticRow = this.staticRow.updateTimesAndPathsForAccord(cellToMaybeNewListPath, newTimestamp, newLocalDeletionTime);
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -1130,6 +1158,5 @@ public class PartitionUpdate extends AbstractBTreePartition
|
||||||
", isBuilt=" + isBuilt +
|
", isBuilt=" + isBuilt +
|
||||||
'}';
|
'}';
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -19,9 +19,12 @@ package org.apache.cassandra.db.rows;
|
||||||
|
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.util.Objects;
|
import java.util.Objects;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
|
import com.google.common.base.Function;
|
||||||
|
|
||||||
import org.apache.cassandra.db.Digest;
|
|
||||||
import org.apache.cassandra.db.DeletionPurger;
|
import org.apache.cassandra.db.DeletionPurger;
|
||||||
|
import org.apache.cassandra.db.Digest;
|
||||||
import org.apache.cassandra.db.TypeSizes;
|
import org.apache.cassandra.db.TypeSizes;
|
||||||
import org.apache.cassandra.db.context.CounterContext;
|
import org.apache.cassandra.db.context.CounterContext;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
|
|
@ -118,12 +121,19 @@ public abstract class AbstractCell<V> extends Cell<V>
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public ColumnData updateAllTimestampAndLocalDeletionTime(long newTimestamp, int newLocalDeletionTime)
|
public ColumnData updateTimesAndPathsForAccord(@Nonnull Function<Cell, CellPath> cellToMaybeNewListPath, long newTimestamp, int newLocalDeletionTime)
|
||||||
{
|
{
|
||||||
long localDeletionTime = localDeletionTime() != NO_DELETION_TIME ? newLocalDeletionTime : NO_DELETION_TIME;
|
long localDeletionTime = localDeletionTime() != NO_DELETION_TIME ? newLocalDeletionTime : NO_DELETION_TIME;
|
||||||
return new BufferCell(column, isTombstone() ? newTimestamp - 1 : newTimestamp, ttl(), localDeletionTime, buffer(), path());
|
return new BufferCell(column, isTombstone() ? newTimestamp - 1 : newTimestamp, ttl(), localDeletionTime, buffer(), path());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Cell<?> updateAllTimesWithNewCellPathForComplexColumnData(@Nonnull CellPath maybeNewPath, long newTimestamp, int newLocalDeletionTime)
|
||||||
|
{
|
||||||
|
long localDeletionTime = localDeletionTime() != NO_DELETION_TIME ? newLocalDeletionTime : NO_DELETION_TIME;
|
||||||
|
return new BufferCell(column, isTombstone() ? newTimestamp - 1 : newTimestamp, ttl(), localDeletionTime, buffer(), maybeNewPath);
|
||||||
|
}
|
||||||
|
|
||||||
public int dataSize()
|
public int dataSize()
|
||||||
{
|
{
|
||||||
CellPath path = path();
|
CellPath path = path();
|
||||||
|
|
|
||||||
|
|
@ -18,7 +18,6 @@
|
||||||
package org.apache.cassandra.db.rows;
|
package org.apache.cassandra.db.rows;
|
||||||
|
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
|
|
||||||
import java.util.AbstractCollection;
|
import java.util.AbstractCollection;
|
||||||
import java.util.Arrays;
|
import java.util.Arrays;
|
||||||
import java.util.Collection;
|
import java.util.Collection;
|
||||||
|
|
@ -28,9 +27,10 @@ import java.util.Iterator;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.function.BiConsumer;
|
import java.util.function.BiConsumer;
|
||||||
import java.util.function.Consumer;
|
import java.util.function.Consumer;
|
||||||
import java.util.function.Function;
|
|
||||||
import java.util.function.Predicate;
|
import java.util.function.Predicate;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
|
import com.google.common.base.Function;
|
||||||
import com.google.common.collect.Collections2;
|
import com.google.common.collect.Collections2;
|
||||||
import com.google.common.collect.Iterators;
|
import com.google.common.collect.Iterators;
|
||||||
import com.google.common.primitives.Ints;
|
import com.google.common.primitives.Ints;
|
||||||
|
|
@ -40,15 +40,13 @@ import org.apache.cassandra.db.Columns;
|
||||||
import org.apache.cassandra.db.DeletionPurger;
|
import org.apache.cassandra.db.DeletionPurger;
|
||||||
import org.apache.cassandra.db.DeletionTime;
|
import org.apache.cassandra.db.DeletionTime;
|
||||||
import org.apache.cassandra.db.LivenessInfo;
|
import org.apache.cassandra.db.LivenessInfo;
|
||||||
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.marshal.UTF8Type;
|
import org.apache.cassandra.db.marshal.UTF8Type;
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
|
||||||
|
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
|
||||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||||
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.schema.DroppedColumn;
|
import org.apache.cassandra.schema.DroppedColumn;
|
||||||
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.utils.AbstractIterator;
|
import org.apache.cassandra.utils.AbstractIterator;
|
||||||
import org.apache.cassandra.utils.BiLongAccumulator;
|
import org.apache.cassandra.utils.BiLongAccumulator;
|
||||||
import org.apache.cassandra.utils.BulkIterator;
|
import org.apache.cassandra.utils.BulkIterator;
|
||||||
|
|
@ -446,7 +444,7 @@ public class BTreeRow extends AbstractRow
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Row updateAllTimestampAndLocalDeletionTime(long newTimestamp, int newLocalDeletionTime)
|
public Row updateTimesAndPathsForAccord(@Nonnull Function<Cell, CellPath> cellToMaybeNewListPath, long newTimestamp, int newLocalDeletionTime)
|
||||||
{
|
{
|
||||||
LivenessInfo newInfo = primaryKeyLivenessInfo.isEmpty() ? primaryKeyLivenessInfo : primaryKeyLivenessInfo.withUpdatedTimestampAndLocalDeletionTime(newTimestamp, newLocalDeletionTime);
|
LivenessInfo newInfo = primaryKeyLivenessInfo.isEmpty() ? primaryKeyLivenessInfo : primaryKeyLivenessInfo.withUpdatedTimestampAndLocalDeletionTime(newTimestamp, newLocalDeletionTime);
|
||||||
// If the deletion is shadowable and the row has a timestamp, we'll forced the deletion timestamp to be less than the row one, so we
|
// If the deletion is shadowable and the row has a timestamp, we'll forced the deletion timestamp to be less than the row one, so we
|
||||||
|
|
@ -454,7 +452,7 @@ public class BTreeRow extends AbstractRow
|
||||||
Deletion newDeletion = deletion.isLive() || (deletion.isShadowable() && !primaryKeyLivenessInfo.isEmpty())
|
Deletion newDeletion = deletion.isLive() || (deletion.isShadowable() && !primaryKeyLivenessInfo.isEmpty())
|
||||||
? Deletion.LIVE
|
? Deletion.LIVE
|
||||||
: new Deletion(DeletionTime.build(newTimestamp - 1, newLocalDeletionTime), deletion.isShadowable());
|
: new Deletion(DeletionTime.build(newTimestamp - 1, newLocalDeletionTime), deletion.isShadowable());
|
||||||
return transformAndFilter(newInfo, newDeletion, (cd) -> cd.updateAllTimestampAndLocalDeletionTime(newTimestamp, newLocalDeletionTime));
|
return transformAndFilter(newInfo, newDeletion, (cd) -> cd.updateTimesAndPathsForAccord(cellToMaybeNewListPath, newTimestamp, newLocalDeletionTime));
|
||||||
}
|
}
|
||||||
|
|
||||||
public Row withRowDeletion(DeletionTime newDeletion)
|
public Row withRowDeletion(DeletionTime newDeletion)
|
||||||
|
|
|
||||||
|
|
@ -18,13 +18,16 @@
|
||||||
package org.apache.cassandra.db.rows;
|
package org.apache.cassandra.db.rows;
|
||||||
|
|
||||||
import java.util.Comparator;
|
import java.util.Comparator;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
|
import com.google.common.base.Function;
|
||||||
|
|
||||||
import org.apache.cassandra.cache.IMeasurableMemory;
|
import org.apache.cassandra.cache.IMeasurableMemory;
|
||||||
import org.apache.cassandra.db.Digest;
|
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
|
||||||
import org.apache.cassandra.db.DeletionPurger;
|
import org.apache.cassandra.db.DeletionPurger;
|
||||||
import org.apache.cassandra.db.DeletionTime;
|
import org.apache.cassandra.db.DeletionTime;
|
||||||
|
import org.apache.cassandra.db.Digest;
|
||||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||||
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.serializers.MarshalException;
|
import org.apache.cassandra.serializers.MarshalException;
|
||||||
import org.apache.cassandra.utils.btree.BTree;
|
import org.apache.cassandra.utils.btree.BTree;
|
||||||
import org.apache.cassandra.utils.btree.UpdateFunction;
|
import org.apache.cassandra.utils.btree.UpdateFunction;
|
||||||
|
|
@ -284,7 +287,19 @@ public abstract class ColumnData implements IMeasurableMemory
|
||||||
* This exists for the Paxos path, see {@link PartitionUpdate#updateAllTimestamp} for additional details.
|
* This exists for the Paxos path, see {@link PartitionUpdate#updateAllTimestamp} for additional details.
|
||||||
*/
|
*/
|
||||||
public abstract ColumnData updateAllTimestamp(long newTimestamp);
|
public abstract ColumnData updateAllTimestamp(long newTimestamp);
|
||||||
public abstract ColumnData updateAllTimestampAndLocalDeletionTime(long newTimestamp, int newLocalDeletionTime);
|
|
||||||
|
/**
|
||||||
|
* @param cellToMaybeNewListPath If the cell is a list append cell a new cell path is returned generated based on the Accord executeAt timestamp
|
||||||
|
*/
|
||||||
|
public abstract ColumnData updateTimesAndPathsForAccord(@Nonnull Function<Cell, CellPath> cellToMaybeNewListPath, long newTimestamp, int newLocalDeletionTime);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* List paths are time UUIDs that increment for each item in the list and for Accord and Paxos
|
||||||
|
* should be based on the transaction's ballot/timestamp.
|
||||||
|
*
|
||||||
|
* @param maybeNewPath If this cell is a list append for a non-frozen list (multi-cell) then it will be new path generated using the executeAt timestamp, otherwise it will be the existing path
|
||||||
|
*/
|
||||||
|
public abstract ColumnData updateAllTimesWithNewCellPathForComplexColumnData(@Nonnull CellPath maybeNewPath, long newTimestamp, int newLocalDeletionTime);
|
||||||
|
|
||||||
public abstract ColumnData markCounterLocalToBeCleared();
|
public abstract ColumnData markCounterLocalToBeCleared();
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -20,6 +20,7 @@ package org.apache.cassandra.db.rows;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.util.Iterator;
|
import java.util.Iterator;
|
||||||
import java.util.Objects;
|
import java.util.Objects;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.base.Function;
|
import com.google.common.base.Function;
|
||||||
|
|
@ -30,6 +31,7 @@ import org.apache.cassandra.db.Digest;
|
||||||
import org.apache.cassandra.db.LivenessInfo;
|
import org.apache.cassandra.db.LivenessInfo;
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
import org.apache.cassandra.db.marshal.ByteType;
|
import org.apache.cassandra.db.marshal.ByteType;
|
||||||
|
import org.apache.cassandra.db.marshal.ListType;
|
||||||
import org.apache.cassandra.db.marshal.SetType;
|
import org.apache.cassandra.db.marshal.SetType;
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.schema.DroppedColumn;
|
import org.apache.cassandra.schema.DroppedColumn;
|
||||||
|
|
@ -265,10 +267,21 @@ public class ComplexColumnData extends ColumnData implements Iterable<Cell<?>>
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public ColumnData updateAllTimestampAndLocalDeletionTime(long newTimestamp, int newLocalDeletionTime)
|
public ColumnData updateTimesAndPathsForAccord(@Nonnull Function<Cell, CellPath> cellToMaybeNewListPath, long newTimestamp, int newLocalDeletionTime)
|
||||||
{
|
{
|
||||||
DeletionTime newDeletion = complexDeletion.isLive() ? complexDeletion : DeletionTime.build(newTimestamp - 1, newLocalDeletionTime);
|
DeletionTime newDeletion = complexDeletion.isLive() ? complexDeletion : DeletionTime.build(newTimestamp - 1, newLocalDeletionTime);
|
||||||
return transformAndFilter(newDeletion, (cell) -> (Cell<?>) cell.updateAllTimestampAndLocalDeletionTime(newTimestamp, newLocalDeletionTime));
|
Function<Cell, CellPath> maybeNewListPath;
|
||||||
|
if (column.type instanceof ListType && column.type.isMultiCell())
|
||||||
|
maybeNewListPath = cellToMaybeNewListPath;
|
||||||
|
else
|
||||||
|
maybeNewListPath = cell -> cell.path();
|
||||||
|
return transformAndFilter(newDeletion, (cell) -> (Cell<?>) cell.updateAllTimesWithNewCellPathForComplexColumnData(maybeNewListPath.apply(cell), newTimestamp, newLocalDeletionTime));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ColumnData updateAllTimesWithNewCellPathForComplexColumnData(@Nonnull CellPath maybeNewPath, long newTimestamp, int newLocalDeletionTime)
|
||||||
|
{
|
||||||
|
throw new UnsupportedOperationException();
|
||||||
}
|
}
|
||||||
|
|
||||||
public long maxTimestamp()
|
public long maxTimestamp()
|
||||||
|
|
|
||||||
|
|
@ -17,13 +17,26 @@
|
||||||
*/
|
*/
|
||||||
package org.apache.cassandra.db.rows;
|
package org.apache.cassandra.db.rows;
|
||||||
|
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.Arrays;
|
||||||
|
import java.util.Collection;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.Comparator;
|
||||||
|
import java.util.Iterator;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Objects;
|
||||||
import java.util.function.BiConsumer;
|
import java.util.function.BiConsumer;
|
||||||
import java.util.function.Consumer;
|
import java.util.function.Consumer;
|
||||||
import java.util.function.Function;
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
|
import com.google.common.base.Function;
|
||||||
|
|
||||||
import org.apache.cassandra.cache.IMeasurableMemory;
|
import org.apache.cassandra.cache.IMeasurableMemory;
|
||||||
import org.apache.cassandra.db.*;
|
import org.apache.cassandra.db.Clustering;
|
||||||
|
import org.apache.cassandra.db.DeletionPurger;
|
||||||
|
import org.apache.cassandra.db.DeletionTime;
|
||||||
|
import org.apache.cassandra.db.Digest;
|
||||||
|
import org.apache.cassandra.db.LivenessInfo;
|
||||||
import org.apache.cassandra.db.filter.ColumnFilter;
|
import org.apache.cassandra.db.filter.ColumnFilter;
|
||||||
import org.apache.cassandra.schema.ColumnMetadata;
|
import org.apache.cassandra.schema.ColumnMetadata;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
|
|
@ -299,7 +312,7 @@ public interface Row extends Unfiltered, Iterable<ColumnData>, IMeasurableMemory
|
||||||
*/
|
*/
|
||||||
public Row updateAllTimestamp(long newTimestamp);
|
public Row updateAllTimestamp(long newTimestamp);
|
||||||
|
|
||||||
public Row updateAllTimestampAndLocalDeletionTime(long newTimestamp, int newLocalDeletionTime);
|
public Row updateTimesAndPathsForAccord(@Nonnull Function<Cell, CellPath> cellToMaybeNewListPath, long newTimestamp, int newLocalDeletionTime);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Returns a copy of this row with the new deletion as row deletion if it is more recent
|
* Returns a copy of this row with the new deletion as row deletion if it is more recent
|
||||||
|
|
|
||||||
|
|
@ -49,6 +49,7 @@ public class CassandraOutgoingFile implements OutgoingStream
|
||||||
private final boolean shouldStreamEntireSSTable;
|
private final boolean shouldStreamEntireSSTable;
|
||||||
private final StreamOperation operation;
|
private final StreamOperation operation;
|
||||||
private final CassandraStreamHeader header;
|
private final CassandraStreamHeader header;
|
||||||
|
private final List<Range<Token>> ranges;
|
||||||
|
|
||||||
public CassandraOutgoingFile(StreamOperation operation, Ref<SSTableReader> ref,
|
public CassandraOutgoingFile(StreamOperation operation, Ref<SSTableReader> ref,
|
||||||
List<SSTableReader.PartitionPositionBounds> sections, List<Range<Token>> normalizedRanges,
|
List<SSTableReader.PartitionPositionBounds> sections, List<Range<Token>> normalizedRanges,
|
||||||
|
|
@ -60,6 +61,7 @@ public class CassandraOutgoingFile implements OutgoingStream
|
||||||
this.ref = ref;
|
this.ref = ref;
|
||||||
this.estimatedKeys = estimatedKeys;
|
this.estimatedKeys = estimatedKeys;
|
||||||
this.sections = sections;
|
this.sections = sections;
|
||||||
|
this.ranges = normalizedRanges;
|
||||||
|
|
||||||
SSTableReader sstable = ref.get();
|
SSTableReader sstable = ref.get();
|
||||||
|
|
||||||
|
|
@ -131,6 +133,12 @@ public class CassandraOutgoingFile implements OutgoingStream
|
||||||
return shouldStreamEntireSSTable ? header.componentManifest.components().size() : 1;
|
return shouldStreamEntireSSTable ? header.componentManifest.components().size() : 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<Range<Token>> ranges()
|
||||||
|
{
|
||||||
|
return ranges;
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public long getRepairedAt()
|
public long getRepairedAt()
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -81,9 +81,9 @@ public class CassandraStreamManager implements TableStreamManager
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public StreamReceiver createStreamReceiver(StreamSession session, int totalStreams)
|
public StreamReceiver createStreamReceiver(StreamSession session, List<Range<Token>> ranges, int totalStreams)
|
||||||
{
|
{
|
||||||
return new CassandraStreamReceiver(cfs, session, totalStreams);
|
return new CassandraStreamReceiver(cfs, session, ranges, totalStreams);
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
|
|
|
||||||
|
|
@ -41,18 +41,24 @@ import org.apache.cassandra.db.rows.ThrottledUnfilteredIterator;
|
||||||
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
|
import org.apache.cassandra.db.rows.UnfilteredRowIterator;
|
||||||
import org.apache.cassandra.db.view.View;
|
import org.apache.cassandra.db.view.View;
|
||||||
import org.apache.cassandra.dht.Bounds;
|
import org.apache.cassandra.dht.Bounds;
|
||||||
|
import org.apache.cassandra.dht.Range;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.io.sstable.ISSTableScanner;
|
import org.apache.cassandra.io.sstable.ISSTableScanner;
|
||||||
import org.apache.cassandra.io.sstable.SSTable;
|
import org.apache.cassandra.io.sstable.SSTable;
|
||||||
import org.apache.cassandra.io.sstable.SSTableMultiWriter;
|
import org.apache.cassandra.io.sstable.SSTableMultiWriter;
|
||||||
import org.apache.cassandra.io.sstable.format.SSTableReader;
|
import org.apache.cassandra.io.sstable.format.SSTableReader;
|
||||||
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
|
import org.apache.cassandra.service.accord.IAccordService;
|
||||||
import org.apache.cassandra.streaming.IncomingStream;
|
import org.apache.cassandra.streaming.IncomingStream;
|
||||||
import org.apache.cassandra.streaming.StreamReceiver;
|
import org.apache.cassandra.streaming.StreamReceiver;
|
||||||
import org.apache.cassandra.streaming.StreamSession;
|
import org.apache.cassandra.streaming.StreamSession;
|
||||||
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
|
import org.apache.cassandra.utils.CassandraVersion;
|
||||||
import org.apache.cassandra.utils.CloseableIterator;
|
import org.apache.cassandra.utils.CloseableIterator;
|
||||||
import org.apache.cassandra.utils.Throwables;
|
import org.apache.cassandra.utils.Throwables;
|
||||||
import org.apache.cassandra.utils.concurrent.Refs;
|
import org.apache.cassandra.utils.concurrent.Refs;
|
||||||
|
|
||||||
|
import static com.google.common.base.Preconditions.checkNotNull;
|
||||||
import static org.apache.cassandra.config.CassandraRelevantProperties.REPAIR_MUTATION_REPAIR_ROWS_PER_BATCH;
|
import static org.apache.cassandra.config.CassandraRelevantProperties.REPAIR_MUTATION_REPAIR_ROWS_PER_BATCH;
|
||||||
|
|
||||||
public class CassandraStreamReceiver implements StreamReceiver
|
public class CassandraStreamReceiver implements StreamReceiver
|
||||||
|
|
@ -74,14 +80,17 @@ public class CassandraStreamReceiver implements StreamReceiver
|
||||||
|
|
||||||
private final boolean requiresWritePath;
|
private final boolean requiresWritePath;
|
||||||
|
|
||||||
|
private final List<Range<Token>> ranges;
|
||||||
|
|
||||||
public CassandraStreamReceiver(ColumnFamilyStore cfs, StreamSession session, int totalFiles)
|
|
||||||
|
public CassandraStreamReceiver(ColumnFamilyStore cfs, StreamSession session, List<Range<Token>> ranges, int totalFiles)
|
||||||
{
|
{
|
||||||
this.cfs = cfs;
|
this.cfs = cfs;
|
||||||
this.session = session;
|
this.session = session;
|
||||||
// this is an "offline" transaction, as we currently manually expose the sstables once done;
|
// this is an "offline" transaction, as we currently manually expose the sstables once done;
|
||||||
// this should be revisited at a later date, so that LifecycleTransaction manages all sstable state changes
|
// this should be revisited at a later date, so that LifecycleTransaction manages all sstable state changes
|
||||||
this.txn = LifecycleTransaction.offline(OperationType.STREAM);
|
this.txn = LifecycleTransaction.offline(OperationType.STREAM);
|
||||||
|
this.ranges = ranges;
|
||||||
this.sstables = new ArrayList<>(totalFiles);
|
this.sstables = new ArrayList<>(totalFiles);
|
||||||
this.requiresWritePath = requiresWritePath(cfs);
|
this.requiresWritePath = requiresWritePath(cfs);
|
||||||
}
|
}
|
||||||
|
|
@ -233,6 +242,14 @@ public class CassandraStreamReceiver implements StreamReceiver
|
||||||
@Override
|
@Override
|
||||||
public void finished()
|
public void finished()
|
||||||
{
|
{
|
||||||
|
CassandraVersion minVersion = ClusterMetadata.current().directory.clusterMinVersion.cassandraVersion;
|
||||||
|
checkNotNull(minVersion, "Unable to determine minimum cluster version");
|
||||||
|
IAccordService accordService = AccordService.instance();
|
||||||
|
if (session.streamOperation().requiresBarrierTransaction()
|
||||||
|
&& accordService.isAccordManagedKeyspace(cfs.keyspace.getName())
|
||||||
|
&& CassandraVersion.CASSANDRA_5_0.compareTo(minVersion) >= 0)
|
||||||
|
accordService.postStreamReceivingBarrier(cfs, ranges);
|
||||||
|
|
||||||
boolean requiresWritePath = requiresWritePath(cfs);
|
boolean requiresWritePath = requiresWritePath(cfs);
|
||||||
Collection<SSTableReader> readers = sstables;
|
Collection<SSTableReader> readers = sstables;
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -144,7 +144,7 @@ public class LocalRepairTables
|
||||||
result.column("options_primary_range", state.options.isPrimaryRange());
|
result.column("options_primary_range", state.options.isPrimaryRange());
|
||||||
result.column("options_trace", state.options.isTraced());
|
result.column("options_trace", state.options.isTraced());
|
||||||
result.column("options_job_threads", state.options.getJobThreads());
|
result.column("options_job_threads", state.options.getJobThreads());
|
||||||
result.column("options_subrange_repair", state.options.isSubrangeRepair());
|
result.column("options_subrange_repair", false);
|
||||||
result.column("options_pull_repair", state.options.isPullRepair());
|
result.column("options_pull_repair", state.options.isPullRepair());
|
||||||
result.column("options_force_repair", state.options.isForcedRepair());
|
result.column("options_force_repair", state.options.isForcedRepair());
|
||||||
result.column("options_preview_kind", state.options.getPreviewKind().name());
|
result.column("options_preview_kind", state.options.getPreviewKind().name());
|
||||||
|
|
@ -183,6 +183,10 @@ public class LocalRepairTables
|
||||||
default: throw new AssertionError("Unknown preview kind: " + state.options.getPreviewKind());
|
default: throw new AssertionError("Unknown preview kind: " + state.options.getPreviewKind());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
else if (state.options.accordRepair())
|
||||||
|
{
|
||||||
|
return "accord repair";
|
||||||
|
}
|
||||||
else if (state.options.isIncremental())
|
else if (state.options.isIncremental())
|
||||||
{
|
{
|
||||||
return "incremental";
|
return "incremental";
|
||||||
|
|
|
||||||
|
|
@ -45,7 +45,7 @@ public abstract class AccordSplitter implements ShardDistributor.EvenSplit.Split
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public accord.primitives.Range subRange(accord.primitives.Range range, BigInteger startOffset, BigInteger endOffset)
|
public TokenRange subRange(accord.primitives.Range range, BigInteger startOffset, BigInteger endOffset)
|
||||||
{
|
{
|
||||||
AccordRoutingKey startBound = (AccordRoutingKey)range.start();
|
AccordRoutingKey startBound = (AccordRoutingKey)range.start();
|
||||||
AccordRoutingKey endBound = (AccordRoutingKey)range.end();
|
AccordRoutingKey endBound = (AccordRoutingKey)range.end();
|
||||||
|
|
|
||||||
|
|
@ -17,26 +17,6 @@
|
||||||
*/
|
*/
|
||||||
package org.apache.cassandra.dht;
|
package org.apache.cassandra.dht;
|
||||||
|
|
||||||
import accord.primitives.Ranges;
|
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
|
||||||
import org.apache.cassandra.schema.Schema;
|
|
||||||
import org.apache.cassandra.db.BufferDecoratedKey;
|
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
|
||||||
import org.apache.cassandra.db.marshal.BytesType;
|
|
||||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
|
||||||
import org.apache.cassandra.service.StorageService;
|
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteSource;
|
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteSourceInverse;
|
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
|
||||||
import org.apache.cassandra.utils.Hex;
|
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
|
||||||
import org.apache.cassandra.utils.Pair;
|
|
||||||
|
|
||||||
import org.apache.commons.lang3.ArrayUtils;
|
|
||||||
|
|
||||||
import java.math.BigInteger;
|
import java.math.BigInteger;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.util.ArrayList;
|
import java.util.ArrayList;
|
||||||
|
|
@ -48,6 +28,25 @@ import java.util.concurrent.ThreadLocalRandom;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
||||||
import com.google.common.collect.Maps;
|
import com.google.common.collect.Maps;
|
||||||
|
import org.apache.commons.lang3.ArrayUtils;
|
||||||
|
|
||||||
|
import accord.primitives.Ranges;
|
||||||
|
import org.apache.cassandra.db.BufferDecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
|
import org.apache.cassandra.db.marshal.BytesType;
|
||||||
|
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||||
|
import org.apache.cassandra.schema.Schema;
|
||||||
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
|
import org.apache.cassandra.service.StorageService;
|
||||||
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
|
import org.apache.cassandra.utils.Hex;
|
||||||
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
|
import org.apache.cassandra.utils.Pair;
|
||||||
|
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
||||||
|
import org.apache.cassandra.utils.bytecomparable.ByteSource;
|
||||||
|
import org.apache.cassandra.utils.bytecomparable.ByteSourceInverse;
|
||||||
|
|
||||||
public class ByteOrderedPartitioner implements IPartitioner
|
public class ByteOrderedPartitioner implements IPartitioner
|
||||||
{
|
{
|
||||||
|
|
@ -194,6 +193,8 @@ public class ByteOrderedPartitioner implements IPartitioner
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private ByteOrderedPartitioner() {}
|
||||||
|
|
||||||
public BytesToken getToken(ByteBuffer key)
|
public BytesToken getToken(ByteBuffer key)
|
||||||
{
|
{
|
||||||
if (key.remaining() == 0)
|
if (key.remaining() == 0)
|
||||||
|
|
|
||||||
|
|
@ -21,12 +21,13 @@ import java.nio.ByteBuffer;
|
||||||
import java.util.Collections;
|
import java.util.Collections;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
|
import java.util.Objects;
|
||||||
import java.util.Random;
|
import java.util.Random;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
||||||
import accord.primitives.Ranges;
|
import accord.primitives.Ranges;
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
|
||||||
import org.apache.cassandra.db.CachedHashDecoratedKey;
|
import org.apache.cassandra.db.CachedHashDecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.marshal.ByteBufferAccessor;
|
import org.apache.cassandra.db.marshal.ByteBufferAccessor;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
|
|
@ -140,6 +141,21 @@ public class LocalPartitioner implements IPartitioner
|
||||||
return comparator;
|
return comparator;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean equals(Object o)
|
||||||
|
{
|
||||||
|
if (this == o) return true;
|
||||||
|
if (o == null || getClass() != o.getClass()) return false;
|
||||||
|
LocalPartitioner that = (LocalPartitioner) o;
|
||||||
|
return comparator.equals(that.comparator) && tokenFactory.equals(that.tokenFactory);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public int hashCode()
|
||||||
|
{
|
||||||
|
return Objects.hash(comparator, tokenFactory);
|
||||||
|
}
|
||||||
|
|
||||||
public class LocalToken extends ComparableObjectToken<ByteBuffer>
|
public class LocalToken extends ComparableObjectToken<ByteBuffer>
|
||||||
{
|
{
|
||||||
static final long serialVersionUID = 8437543776403014875L;
|
static final long serialVersionUID = 8437543776403014875L;
|
||||||
|
|
|
||||||
|
|
@ -21,28 +21,33 @@ import java.io.IOException;
|
||||||
import java.math.BigDecimal;
|
import java.math.BigDecimal;
|
||||||
import java.math.BigInteger;
|
import java.math.BigInteger;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.util.*;
|
import java.util.HashMap;
|
||||||
|
import java.util.Iterator;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Optional;
|
||||||
|
import java.util.Random;
|
||||||
import java.util.concurrent.ThreadLocalRandom;
|
import java.util.concurrent.ThreadLocalRandom;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
||||||
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
import com.google.common.primitives.Longs;
|
||||||
|
|
||||||
import accord.primitives.Ranges;
|
import accord.primitives.Ranges;
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.PreHashedDecoratedKey;
|
import org.apache.cassandra.db.PreHashedDecoratedKey;
|
||||||
import org.apache.cassandra.db.TypeSizes;
|
import org.apache.cassandra.db.TypeSizes;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.marshal.PartitionerDefinedOrder;
|
|
||||||
import org.apache.cassandra.db.marshal.LongType;
|
import org.apache.cassandra.db.marshal.LongType;
|
||||||
|
import org.apache.cassandra.db.marshal.PartitionerDefinedOrder;
|
||||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
|
import org.apache.cassandra.utils.MurmurHash;
|
||||||
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteSource;
|
import org.apache.cassandra.utils.bytecomparable.ByteSource;
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteSourceInverse;
|
import org.apache.cassandra.utils.bytecomparable.ByteSourceInverse;
|
||||||
import org.apache.cassandra.utils.MurmurHash;
|
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
|
||||||
import com.google.common.primitives.Longs;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* This class generates a BigIntegerToken using a Murmur3 hash.
|
* This class generates a BigIntegerToken using a Murmur3 hash.
|
||||||
|
|
@ -85,6 +90,8 @@ public class Murmur3Partitioner implements IPartitioner
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
protected Murmur3Partitioner() {}
|
||||||
|
|
||||||
public DecoratedKey decorateKey(ByteBuffer key)
|
public DecoratedKey decorateKey(ByteBuffer key)
|
||||||
{
|
{
|
||||||
long[] hash = getHash(key);
|
long[] hash = getHash(key);
|
||||||
|
|
|
||||||
|
|
@ -20,14 +20,18 @@ package org.apache.cassandra.dht;
|
||||||
import java.math.BigInteger;
|
import java.math.BigInteger;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.nio.charset.CharacterCodingException;
|
import java.nio.charset.CharacterCodingException;
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Random;
|
||||||
import java.util.concurrent.ThreadLocalRandom;
|
import java.util.concurrent.ThreadLocalRandom;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
||||||
import accord.api.RoutingKey;
|
import accord.api.RoutingKey;
|
||||||
import accord.primitives.Ranges;
|
import accord.primitives.Ranges;
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
|
||||||
import org.apache.cassandra.db.CachedHashDecoratedKey;
|
import org.apache.cassandra.db.CachedHashDecoratedKey;
|
||||||
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.marshal.UTF8Type;
|
import org.apache.cassandra.db.marshal.UTF8Type;
|
||||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||||
|
|
@ -70,6 +74,8 @@ public class OrderPreservingPartitioner implements IPartitioner
|
||||||
|
|
||||||
public static final OrderPreservingPartitioner instance = new OrderPreservingPartitioner();
|
public static final OrderPreservingPartitioner instance = new OrderPreservingPartitioner();
|
||||||
|
|
||||||
|
private OrderPreservingPartitioner() {}
|
||||||
|
|
||||||
public DecoratedKey decorateKey(ByteBuffer key)
|
public DecoratedKey decorateKey(ByteBuffer key)
|
||||||
{
|
{
|
||||||
return new CachedHashDecoratedKey(getToken(key), key);
|
return new CachedHashDecoratedKey(getToken(key), key);
|
||||||
|
|
|
||||||
|
|
@ -22,28 +22,33 @@ import java.math.BigDecimal;
|
||||||
import java.math.BigInteger;
|
import java.math.BigInteger;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.security.MessageDigest;
|
import java.security.MessageDigest;
|
||||||
import java.util.*;
|
import java.util.HashMap;
|
||||||
|
import java.util.Iterator;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Optional;
|
||||||
|
import java.util.Random;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
|
||||||
import accord.primitives.Ranges;
|
import accord.primitives.Ranges;
|
||||||
import org.apache.cassandra.db.CachedHashDecoratedKey;
|
import org.apache.cassandra.db.CachedHashDecoratedKey;
|
||||||
import org.apache.cassandra.db.marshal.ByteArrayAccessor;
|
|
||||||
import org.apache.cassandra.db.marshal.ByteBufferAccessor;
|
|
||||||
import org.apache.cassandra.exceptions.ConfigurationException;
|
|
||||||
import org.apache.cassandra.db.DecoratedKey;
|
import org.apache.cassandra.db.DecoratedKey;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
|
import org.apache.cassandra.db.marshal.ByteArrayAccessor;
|
||||||
|
import org.apache.cassandra.db.marshal.ByteBufferAccessor;
|
||||||
import org.apache.cassandra.db.marshal.IntegerType;
|
import org.apache.cassandra.db.marshal.IntegerType;
|
||||||
import org.apache.cassandra.db.marshal.PartitionerDefinedOrder;
|
import org.apache.cassandra.db.marshal.PartitionerDefinedOrder;
|
||||||
|
import org.apache.cassandra.exceptions.ConfigurationException;
|
||||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteSource;
|
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
import org.apache.cassandra.utils.GuidGenerator;
|
import org.apache.cassandra.utils.GuidGenerator;
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
import org.apache.cassandra.utils.Pair;
|
import org.apache.cassandra.utils.Pair;
|
||||||
|
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
||||||
|
import org.apache.cassandra.utils.bytecomparable.ByteSource;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* This class generates a BigIntegerToken using MD5 hash.
|
* This class generates a BigIntegerToken using MD5 hash.
|
||||||
|
|
@ -108,6 +113,8 @@ public class RandomPartitioner implements IPartitioner
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
private RandomPartitioner() {}
|
||||||
|
|
||||||
public DecoratedKey decorateKey(ByteBuffer key)
|
public DecoratedKey decorateKey(ByteBuffer key)
|
||||||
{
|
{
|
||||||
return new CachedHashDecoratedKey(getToken(key), key);
|
return new CachedHashDecoratedKey(getToken(key), key);
|
||||||
|
|
|
||||||
|
|
@ -19,13 +19,27 @@ package org.apache.cassandra.dht;
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.io.Serializable;
|
import java.io.Serializable;
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.Arrays;
|
||||||
|
import java.util.Collection;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.Comparator;
|
||||||
|
import java.util.HashSet;
|
||||||
|
import java.util.Iterator;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Set;
|
||||||
import java.util.function.Predicate;
|
import java.util.function.Predicate;
|
||||||
|
|
||||||
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
import com.google.common.collect.ImmutableList;
|
||||||
import com.google.common.collect.Iterables;
|
import com.google.common.collect.Iterables;
|
||||||
|
import com.google.common.collect.Iterators;
|
||||||
|
import com.google.common.collect.PeekingIterator;
|
||||||
import org.apache.commons.lang3.ObjectUtils;
|
import org.apache.commons.lang3.ObjectUtils;
|
||||||
|
|
||||||
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
import org.apache.cassandra.db.PartitionPosition;
|
import org.apache.cassandra.db.PartitionPosition;
|
||||||
|
import org.apache.cassandra.dht.Token.TokenFactory;
|
||||||
import org.apache.cassandra.io.util.DataInputPlus;
|
import org.apache.cassandra.io.util.DataInputPlus;
|
||||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||||
import org.apache.cassandra.net.MessagingService;
|
import org.apache.cassandra.net.MessagingService;
|
||||||
|
|
@ -34,6 +48,10 @@ import org.apache.cassandra.tcm.serialization.MetadataSerializer;
|
||||||
import org.apache.cassandra.tcm.serialization.Version;
|
import org.apache.cassandra.tcm.serialization.Version;
|
||||||
import org.apache.cassandra.utils.Pair;
|
import org.apache.cassandra.utils.Pair;
|
||||||
|
|
||||||
|
import static com.google.common.base.Preconditions.checkState;
|
||||||
|
import static java.util.Collections.emptyList;
|
||||||
|
import static org.apache.cassandra.config.CassandraRelevantProperties.TEST_RANGE_EXPENSIVE_CHECKS;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* A representation of the range that a node is responsible for on the DHT ring.
|
* A representation of the range that a node is responsible for on the DHT ring.
|
||||||
*
|
*
|
||||||
|
|
@ -48,6 +66,34 @@ public class Range<T extends RingPosition<T>> extends AbstractBounds<T> implemen
|
||||||
public static final Serializer serializer = new Serializer();
|
public static final Serializer serializer = new Serializer();
|
||||||
public static final long serialVersionUID = 1L;
|
public static final long serialVersionUID = 1L;
|
||||||
|
|
||||||
|
public static final boolean EXPENSIVE_CHECKS = TEST_RANGE_EXPENSIVE_CHECKS.getBoolean();
|
||||||
|
|
||||||
|
public static final IPartitionerDependentSerializer rangeSerializer = new RangeSerializer();
|
||||||
|
|
||||||
|
public static class RangeSerializer<T extends RingPosition<T>> implements IPartitionerDependentSerializer<Range<T>>
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
public void serialize(Range range, DataOutputPlus out, int version) throws IOException
|
||||||
|
{
|
||||||
|
Token.compactSerializer.serialize(range.left.getToken(), out, version);
|
||||||
|
Token.compactSerializer.serialize(range.right.getToken(), out, version);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Range deserialize(DataInputPlus in, IPartitioner p, int version) throws IOException
|
||||||
|
{
|
||||||
|
return new Range(Token.compactSerializer.deserialize(in, p, version),
|
||||||
|
Token.compactSerializer.deserialize(in, p, version));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long serializedSize(Range range, int version)
|
||||||
|
{
|
||||||
|
return Token.compactSerializer.serializedSize(range.left.getToken(), version)
|
||||||
|
+ Token.compactSerializer.serializedSize(range.right.getToken(), version);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
public Range(T left, T right)
|
public Range(T left, T right)
|
||||||
{
|
{
|
||||||
super(left, right);
|
super(left, right);
|
||||||
|
|
@ -349,6 +395,43 @@ public class Range<T extends RingPosition<T>> extends AbstractBounds<T> implemen
|
||||||
return right.compareTo(rhs.right);
|
return right.compareTo(rhs.right);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Compares ranges by right token. Used for intersecting normalized ranges.
|
||||||
|
*
|
||||||
|
* Assumes no wrap around ranges except for RHS = minValue which is essentialy synonymous with the maximal value.
|
||||||
|
* This shows up coming out of unwrap because Range is not left inclusive so the only way to include minValue
|
||||||
|
* in the range is by wrapping from maxValue.
|
||||||
|
*/
|
||||||
|
private int compareNormalized(Range<T> rhs)
|
||||||
|
{
|
||||||
|
// otherwise compare by right.
|
||||||
|
int cmp = right.compareTo(rhs.right);
|
||||||
|
// minValue on the RHS is maxValue, but doesn't work with compare so check for it explicitly
|
||||||
|
boolean rhsRMin = rhs.right.isMinimum();
|
||||||
|
boolean lhsRMin = right.isMinimum();
|
||||||
|
|
||||||
|
if (rhsRMin && lhsRMin)
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
if (cmp < 0)
|
||||||
|
{
|
||||||
|
if (lhsRMin)
|
||||||
|
{
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
else if (cmp > 0)
|
||||||
|
{
|
||||||
|
if (rhsRMin)
|
||||||
|
{
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Subtracts a portion of this range.
|
* Subtracts a portion of this range.
|
||||||
* @param contained The range to subtract from this. It must be totally
|
* @param contained The range to subtract from this. It must be totally
|
||||||
|
|
@ -361,7 +444,7 @@ public class Range<T extends RingPosition<T>> extends AbstractBounds<T> implemen
|
||||||
// both ranges cover the entire ring, their difference is an empty set
|
// both ranges cover the entire ring, their difference is an empty set
|
||||||
if(isFull(left, right) && isFull(contained.left, contained.right))
|
if(isFull(left, right) && isFull(contained.left, contained.right))
|
||||||
{
|
{
|
||||||
return Collections.emptyList();
|
return emptyList();
|
||||||
}
|
}
|
||||||
|
|
||||||
// a range is subtracted from another range that covers the entire ring
|
// a range is subtracted from another range that covers the entire ring
|
||||||
|
|
@ -472,6 +555,190 @@ public class Range<T extends RingPosition<T>> extends AbstractBounds<T> implemen
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static final Comparator NORMALIZED_TOKEN_RANGE_COMPARATOR = (o1, o2) -> {
|
||||||
|
Range range = (Range)o1;
|
||||||
|
RingPosition key = (RingPosition) o2;
|
||||||
|
boolean rangeRightIsMin = range.right.isMinimum();
|
||||||
|
boolean keyIsMinimum = key.isMinimum();
|
||||||
|
|
||||||
|
if (keyIsMinimum & rangeRightIsMin)
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
int lc = key.compareTo(range.left);
|
||||||
|
int rc = key.compareTo(range.right);
|
||||||
|
if ((lc < 0 & !keyIsMinimum) | lc == 0) return 1;
|
||||||
|
if (rc > 0 & !rangeRightIsMin) return -1;
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
|
||||||
|
public static <T extends RingPosition<T>> boolean isInNormalizedRanges(T token, List<Range<T>> ranges)
|
||||||
|
{
|
||||||
|
if (ranges.size() == 1 && ranges.get(0).isFull())
|
||||||
|
return true;
|
||||||
|
boolean isIn = Collections.binarySearch((List)ranges, token, NORMALIZED_TOKEN_RANGE_COMPARATOR) >= 0;
|
||||||
|
if (EXPENSIVE_CHECKS)
|
||||||
|
checkState(isInRanges(token, ranges) == isIn);
|
||||||
|
return isIn;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static <T extends RingPosition<T>> List<Range<T>> subtractNormalizedRanges(List<Range<T>> a, List<Range<T>> b)
|
||||||
|
{
|
||||||
|
if (b.size() == 1 && b.get(0).isFull())
|
||||||
|
return emptyList();
|
||||||
|
|
||||||
|
if (a.size() == 1 && a.get(0).isFull())
|
||||||
|
return invertNormalizedRanges(b);
|
||||||
|
|
||||||
|
List<Range<T>> remaining = new ArrayList<>();
|
||||||
|
Iterator<Range<T>> aIter = a.iterator();
|
||||||
|
Iterator<Range<T>> bIter = b.iterator();
|
||||||
|
Range<T> aRange = aIter.hasNext() ? aIter.next() : null;
|
||||||
|
Range<T> bRange = bIter.hasNext() ? bIter.next() : null;
|
||||||
|
while (aRange != null && bRange != null)
|
||||||
|
{
|
||||||
|
boolean aRMin = aRange.right.isMinimum();
|
||||||
|
boolean bRMin = bRange.right.isMinimum();
|
||||||
|
|
||||||
|
if (aRMin && bRMin)
|
||||||
|
{
|
||||||
|
if (aRange.left.compareTo(bRange.left) < 0)
|
||||||
|
remaining.add(new Range<>(aRange.left, bRange.left));
|
||||||
|
checkState(!aIter.hasNext() && !bIter.hasNext());
|
||||||
|
aRange = null;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!aRMin && aRange.right.compareTo(bRange.left) <= 0)
|
||||||
|
{
|
||||||
|
remaining.add(aRange);
|
||||||
|
aRange = aIter.hasNext() ? aIter.next() : null;
|
||||||
|
}
|
||||||
|
else if (!bRMin && aRange.left.compareTo(bRange.right) >= 0)
|
||||||
|
{
|
||||||
|
bRange = bIter.hasNext() ? bIter.next() : null;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// Handle what remains to the left of the intersection
|
||||||
|
if (aRange.left.compareTo(bRange.left) < 0)
|
||||||
|
{
|
||||||
|
remaining.add(new Range(aRange.left, bRange.left));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Handle what remains to the right of the intersection
|
||||||
|
if (!aRMin && (aRange.right.compareTo(bRange.right) <= 0 | bRMin))
|
||||||
|
aRange = aIter.hasNext() ? aIter.next() : null;
|
||||||
|
else
|
||||||
|
aRange = new Range(bRange.right, aRange.right);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
while (aRange != null)
|
||||||
|
{
|
||||||
|
remaining.add(aRange);
|
||||||
|
aRange = aIter.hasNext() ? aIter.next() : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
List<Range<T>> result = ImmutableList.copyOf(normalize(remaining));
|
||||||
|
if (EXPENSIVE_CHECKS)
|
||||||
|
checkState(result.equals(normalize(subtract(a, b))));
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
private boolean isFull()
|
||||||
|
{
|
||||||
|
return isFull(left, right);
|
||||||
|
}
|
||||||
|
|
||||||
|
@VisibleForTesting
|
||||||
|
static <T extends RingPosition<T>> List<Range<T>> invertNormalizedRanges(List<Range<T>> ranges)
|
||||||
|
{
|
||||||
|
if (ranges.isEmpty())
|
||||||
|
return ranges;
|
||||||
|
|
||||||
|
List<Range<T>> result = new ArrayList<>(ranges.size() + 2);
|
||||||
|
T minValue = ranges.get(0).left.minValue();
|
||||||
|
T left = minValue;
|
||||||
|
for (Range<T> r : ranges)
|
||||||
|
{
|
||||||
|
if (!r.left.equals(left))
|
||||||
|
{
|
||||||
|
result.add(new Range<>(left, r.left));
|
||||||
|
}
|
||||||
|
left = r.right;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Loop doesn't add the range to the right of the last one
|
||||||
|
Range<T> last = ranges.get(ranges.size() - 1);
|
||||||
|
if (!last.right.isMinimum())
|
||||||
|
result.add(new Range<>(last.right, minValue));
|
||||||
|
|
||||||
|
result = normalize(result);
|
||||||
|
if (EXPENSIVE_CHECKS)
|
||||||
|
checkState(result.equals(normalize(subtract(ImmutableList.of(new Range<>(minValue, minValue)), ranges))));
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static <T extends RingPosition<T>> List<Range<T>> intersectionOfNormalizedRanges(List<Range<T>> a, List<Range<T>> b)
|
||||||
|
{
|
||||||
|
if (a.size() == 1 && a.get(0).isFull())
|
||||||
|
return b;
|
||||||
|
if (b.size() == 1 && b.get(0).isFull())
|
||||||
|
return a;
|
||||||
|
|
||||||
|
List<Range<T>> merged = new ArrayList<>();
|
||||||
|
PeekingIterator<Range<T>> aIter = Iterators.peekingIterator(a.iterator());
|
||||||
|
PeekingIterator<Range<T>> bIter = Iterators.peekingIterator(b.iterator());
|
||||||
|
while (aIter.hasNext() && bIter.hasNext())
|
||||||
|
{
|
||||||
|
Range<T> aRange = aIter.peek();
|
||||||
|
Range<T> bRange = bIter.peek();
|
||||||
|
|
||||||
|
int cmp = aRange.compareNormalized(bRange);
|
||||||
|
if (aRange.intersects(bRange))
|
||||||
|
{
|
||||||
|
merged.addAll(aRange.intersectionWith(bRange));
|
||||||
|
if (cmp == 0)
|
||||||
|
{
|
||||||
|
aIter.next();
|
||||||
|
bIter.next();
|
||||||
|
}
|
||||||
|
else if(cmp < 0)
|
||||||
|
{
|
||||||
|
aIter.next();
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
bIter.next();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
if (cmp <= 0)
|
||||||
|
aIter.next();
|
||||||
|
if (cmp >= 0)
|
||||||
|
bIter.next();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
List<Range<T>> result = ImmutableList.copyOf(normalize(merged));
|
||||||
|
|
||||||
|
if (EXPENSIVE_CHECKS)
|
||||||
|
{
|
||||||
|
List<Range<T>> expensiveResult = new ArrayList<>();
|
||||||
|
for (Range<T> r1 : a)
|
||||||
|
{
|
||||||
|
for (Range<T> r2 : b)
|
||||||
|
{
|
||||||
|
expensiveResult.addAll(r1.intersectionWith(r2));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
checkState(result.equals(normalize(expensiveResult)));
|
||||||
|
}
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public boolean equals(Object o)
|
public boolean equals(Object o)
|
||||||
{
|
{
|
||||||
|
|
@ -670,6 +937,26 @@ public class Range<T extends RingPosition<T>> extends AbstractBounds<T> implemen
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static <T extends RingPosition<T>> boolean equals(Collection<Range<T>> a, Collection<Range<T>> b)
|
||||||
|
{
|
||||||
|
return normalize(a).equals(normalize(b));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Helper to convert a range string to POJO so you can copy toString from a debugger
|
||||||
|
public static Range<Token> fromString(String value)
|
||||||
|
{
|
||||||
|
return fromString(value, DatabaseDescriptor.getPartitioner());
|
||||||
|
}
|
||||||
|
|
||||||
|
public static Range<Token> fromString(String value, IPartitioner partitioner)
|
||||||
|
{
|
||||||
|
TokenFactory tokenFactory = partitioner.getTokenFactory();
|
||||||
|
String[] parts = value.split(",");
|
||||||
|
Token left = tokenFactory.fromString(parts[0].substring(1));
|
||||||
|
Token right = tokenFactory.fromString(parts[1].substring(0, parts[1].length() -1));
|
||||||
|
return new Range<>(left, right);
|
||||||
|
}
|
||||||
|
|
||||||
public static <T extends RingPosition<T>> void assertNormalized(List<Range<T>> ranges)
|
public static <T extends RingPosition<T>> void assertNormalized(List<Range<T>> ranges)
|
||||||
{
|
{
|
||||||
Range<T> lastRange = null;
|
Range<T> lastRange = null;
|
||||||
|
|
|
||||||
|
|
@ -67,7 +67,7 @@ public class RequestFailure
|
||||||
public void serialize(RequestFailure t, DataOutputPlus out, int version) throws IOException
|
public void serialize(RequestFailure t, DataOutputPlus out, int version) throws IOException
|
||||||
{
|
{
|
||||||
RequestFailureReason.serializer.serialize(t.reason, out, version);
|
RequestFailureReason.serializer.serialize(t.reason, out, version);
|
||||||
if (version >= MessagingService.VERSION_50)
|
if (version >= MessagingService.VERSION_51)
|
||||||
nullableRemoteExceptionSerializer.serialize(t.failure, out, version);
|
nullableRemoteExceptionSerializer.serialize(t.failure, out, version);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -76,7 +76,7 @@ public class RequestFailure
|
||||||
{
|
{
|
||||||
RequestFailureReason reason = RequestFailureReason.serializer.deserialize(in, version);
|
RequestFailureReason reason = RequestFailureReason.serializer.deserialize(in, version);
|
||||||
Throwable failure = null;
|
Throwable failure = null;
|
||||||
if (version >= MessagingService.VERSION_50)
|
if (version >= MessagingService.VERSION_51)
|
||||||
failure = nullableRemoteExceptionSerializer.deserialize(in, version);
|
failure = nullableRemoteExceptionSerializer.deserialize(in, version);
|
||||||
if (failure == null)
|
if (failure == null)
|
||||||
return forReason(reason);
|
return forReason(reason);
|
||||||
|
|
@ -88,7 +88,7 @@ public class RequestFailure
|
||||||
public long serializedSize(RequestFailure t, int version)
|
public long serializedSize(RequestFailure t, int version)
|
||||||
{
|
{
|
||||||
long size = RequestFailureReason.serializer.serializedSize(t.reason, version);
|
long size = RequestFailureReason.serializer.serializedSize(t.reason, version);
|
||||||
if (version >= MessagingService.VERSION_50)
|
if (version >= MessagingService.VERSION_51)
|
||||||
size += nullableRemoteExceptionSerializer.serializedSize(t.failure, version);
|
size += nullableRemoteExceptionSerializer.serializedSize(t.failure, version);
|
||||||
return size;
|
return size;
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -26,6 +26,7 @@ import org.apache.cassandra.db.PartitionPosition;
|
||||||
import org.apache.cassandra.dht.AbstractBounds;
|
import org.apache.cassandra.dht.AbstractBounds;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||||
|
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
|
|
||||||
|
|
@ -356,11 +357,11 @@ public abstract class ReplicaLayout<E extends Endpoints<E>>
|
||||||
* @return the read layout for a token - this includes natural replicas, i.e. those that are not pending.
|
* @return the read layout for a token - this includes natural replicas, i.e. those that are not pending.
|
||||||
* They are reverse sorted by the badness score of the configured snitch
|
* They are reverse sorted by the badness score of the configured snitch
|
||||||
*/
|
*/
|
||||||
static ReplicaLayout.ForTokenRead forTokenReadSorted(ClusterMetadata metadata, Keyspace keyspace, AbstractReplicationStrategy replicationStrategy, Token token)
|
static ReplicaLayout.ForTokenRead forTokenReadSorted(ClusterMetadata metadata, Keyspace keyspace, AbstractReplicationStrategy replicationStrategy, Token token, ReadCoordinator coordinator)
|
||||||
{
|
{
|
||||||
EndpointsForToken replicas = keyspace.getMetadata().params.replication.isLocal()
|
EndpointsForToken replicas = keyspace.getMetadata().params.replication.isLocal()
|
||||||
? forLocalStrategyToken(metadata, replicationStrategy, token)
|
? forLocalStrategyToken(metadata, replicationStrategy, token)
|
||||||
: forNonLocalStrategyTokenRead(metadata, keyspace.getMetadata(), token);
|
: coordinator.forNonLocalStrategyTokenRead(metadata, keyspace.getMetadata(), token);
|
||||||
|
|
||||||
replicas = DatabaseDescriptor.getNodeProximity().sortedByProximity(FBUtilities.getBroadcastAddressAndPort(), replicas);
|
replicas = DatabaseDescriptor.getNodeProximity().sortedByProximity(FBUtilities.getBroadcastAddressAndPort(), replicas);
|
||||||
|
|
||||||
|
|
@ -386,7 +387,7 @@ public abstract class ReplicaLayout<E extends Endpoints<E>>
|
||||||
return metadata.placements.get(keyspace.params.replication).reads.forRange(range.right.getToken()).get();
|
return metadata.placements.get(keyspace.params.replication).reads.forRange(range.right.getToken()).get();
|
||||||
}
|
}
|
||||||
|
|
||||||
static EndpointsForToken forNonLocalStrategyTokenRead(ClusterMetadata metadata, KeyspaceMetadata keyspace, Token token)
|
public static EndpointsForToken forNonLocalStrategyTokenRead(ClusterMetadata metadata, KeyspaceMetadata keyspace, Token token)
|
||||||
{
|
{
|
||||||
return metadata.placements.get(keyspace.params.replication).reads.forToken(token).get();
|
return metadata.placements.get(keyspace.params.replication).reads.forToken(token).get();
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -63,6 +63,7 @@ import org.apache.cassandra.index.Index;
|
||||||
import org.apache.cassandra.index.IndexStatusManager;
|
import org.apache.cassandra.index.IndexStatusManager;
|
||||||
import org.apache.cassandra.schema.SchemaConstants;
|
import org.apache.cassandra.schema.SchemaConstants;
|
||||||
import org.apache.cassandra.service.StorageService;
|
import org.apache.cassandra.service.StorageService;
|
||||||
|
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
import org.apache.cassandra.service.reads.AlwaysSpeculativeRetryPolicy;
|
import org.apache.cassandra.service.reads.AlwaysSpeculativeRetryPolicy;
|
||||||
import org.apache.cassandra.service.reads.SpeculativeRetryPolicy;
|
import org.apache.cassandra.service.reads.SpeculativeRetryPolicy;
|
||||||
|
|
@ -533,7 +534,7 @@ public class ReplicaPlans
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
public static ReplicaPlan.ForWrite forReadRepair(ReplicaPlan<?, ?> forRead, ClusterMetadata metadata, Keyspace keyspace, ConsistencyLevel consistencyLevel, Token token, Predicate<Replica> isAlive) throws UnavailableException
|
public static ReplicaPlan.ForWrite forReadRepair(ReplicaPlan<?, ?> forRead, ClusterMetadata metadata, Keyspace keyspace, ConsistencyLevel consistencyLevel, Token token, Predicate<Replica> isAlive, ReadCoordinator coordinator) throws UnavailableException
|
||||||
{
|
{
|
||||||
AbstractReplicationStrategy replicationStrategy = keyspace.getReplicationStrategy();
|
AbstractReplicationStrategy replicationStrategy = keyspace.getReplicationStrategy();
|
||||||
Selector selector = writeReadRepair(forRead);
|
Selector selector = writeReadRepair(forRead);
|
||||||
|
|
@ -550,7 +551,7 @@ public class ReplicaPlans
|
||||||
liveAndDown.all(),
|
liveAndDown.all(),
|
||||||
live.all(),
|
live.all(),
|
||||||
contacts,
|
contacts,
|
||||||
(newClusterMetadata) -> forReadRepair(forRead, newClusterMetadata, keyspace, consistencyLevel, token, isAlive),
|
(newClusterMetadata) -> forReadRepair(forRead, newClusterMetadata, keyspace, consistencyLevel, token, isAlive, coordinator),
|
||||||
metadata.epoch);
|
metadata.epoch);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -882,9 +883,10 @@ public class ReplicaPlans
|
||||||
Token token,
|
Token token,
|
||||||
@Nullable Index.QueryPlan indexQueryPlan,
|
@Nullable Index.QueryPlan indexQueryPlan,
|
||||||
ConsistencyLevel consistencyLevel,
|
ConsistencyLevel consistencyLevel,
|
||||||
SpeculativeRetryPolicy retry)
|
SpeculativeRetryPolicy retry,
|
||||||
|
ReadCoordinator coordinator)
|
||||||
{
|
{
|
||||||
return forRead(ClusterMetadata.current(), keyspace, token, indexQueryPlan, consistencyLevel, retry, false);
|
return forRead(ClusterMetadata.current(), keyspace, token, indexQueryPlan, consistencyLevel, retry, coordinator, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
public static ReplicaPlan.ForTokenRead forRead(ClusterMetadata metadata,
|
public static ReplicaPlan.ForTokenRead forRead(ClusterMetadata metadata,
|
||||||
|
|
@ -892,9 +894,10 @@ public class ReplicaPlans
|
||||||
Token token,
|
Token token,
|
||||||
@Nullable Index.QueryPlan indexQueryPlan,
|
@Nullable Index.QueryPlan indexQueryPlan,
|
||||||
ConsistencyLevel consistencyLevel,
|
ConsistencyLevel consistencyLevel,
|
||||||
SpeculativeRetryPolicy retry)
|
SpeculativeRetryPolicy retry,
|
||||||
|
ReadCoordinator coordinator)
|
||||||
{
|
{
|
||||||
return forRead(metadata, keyspace, token, indexQueryPlan, consistencyLevel, retry, true);
|
return forRead(metadata, keyspace, token, indexQueryPlan, consistencyLevel, retry, coordinator, true);
|
||||||
}
|
}
|
||||||
|
|
||||||
private static ReplicaPlan.ForTokenRead forRead(ClusterMetadata metadata,
|
private static ReplicaPlan.ForTokenRead forRead(ClusterMetadata metadata,
|
||||||
|
|
@ -903,10 +906,11 @@ public class ReplicaPlans
|
||||||
@Nullable Index.QueryPlan indexQueryPlan,
|
@Nullable Index.QueryPlan indexQueryPlan,
|
||||||
ConsistencyLevel consistencyLevel,
|
ConsistencyLevel consistencyLevel,
|
||||||
SpeculativeRetryPolicy retry,
|
SpeculativeRetryPolicy retry,
|
||||||
|
ReadCoordinator coordinator,
|
||||||
boolean throwOnInsufficientLiveReplicas)
|
boolean throwOnInsufficientLiveReplicas)
|
||||||
{
|
{
|
||||||
AbstractReplicationStrategy replicationStrategy = keyspace.getReplicationStrategy();
|
AbstractReplicationStrategy replicationStrategy = keyspace.getReplicationStrategy();
|
||||||
ReplicaLayout.ForTokenRead forTokenReadLiveAndDown = ReplicaLayout.forTokenReadSorted(metadata, keyspace, replicationStrategy, token);
|
ReplicaLayout.ForTokenRead forTokenReadLiveAndDown = ReplicaLayout.forTokenReadSorted(metadata, keyspace, replicationStrategy, token, coordinator);
|
||||||
ReplicaLayout.ForTokenRead forTokenReadLive = forTokenReadLiveAndDown.filter(FailureDetector.isReplicaAlive);
|
ReplicaLayout.ForTokenRead forTokenReadLive = forTokenReadLiveAndDown.filter(FailureDetector.isReplicaAlive);
|
||||||
EndpointsForToken candidates = candidatesForRead(keyspace, indexQueryPlan, consistencyLevel, forTokenReadLive.all());
|
EndpointsForToken candidates = candidatesForRead(keyspace, indexQueryPlan, consistencyLevel, forTokenReadLive.all());
|
||||||
EndpointsForToken contacts = contactForRead(metadata.locator, replicationStrategy, consistencyLevel, retry.equals(AlwaysSpeculativeRetryPolicy.INSTANCE), candidates);
|
EndpointsForToken contacts = contactForRead(metadata.locator, replicationStrategy, consistencyLevel, retry.equals(AlwaysSpeculativeRetryPolicy.INSTANCE), candidates);
|
||||||
|
|
@ -915,8 +919,8 @@ public class ReplicaPlans
|
||||||
assureSufficientLiveReplicasForRead(metadata.locator, replicationStrategy, consistencyLevel, contacts);
|
assureSufficientLiveReplicasForRead(metadata.locator, replicationStrategy, consistencyLevel, contacts);
|
||||||
|
|
||||||
return new ReplicaPlan.ForTokenRead(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, forTokenReadLiveAndDown.all(),
|
return new ReplicaPlan.ForTokenRead(keyspace, replicationStrategy, consistencyLevel, candidates, contacts, forTokenReadLiveAndDown.all(),
|
||||||
(newClusterMetadata) -> forRead(newClusterMetadata, keyspace, token, indexQueryPlan, consistencyLevel, retry, false),
|
(newClusterMetadata) -> forRead(newClusterMetadata, keyspace, token, indexQueryPlan, consistencyLevel, retry, coordinator, false),
|
||||||
(self) -> forReadRepair(self, metadata, keyspace, consistencyLevel, token, FailureDetector.isReplicaAlive),
|
(self) -> forReadRepair(self, metadata, keyspace, consistencyLevel, token, FailureDetector.isReplicaAlive, coordinator),
|
||||||
metadata.epoch);
|
metadata.epoch);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -962,7 +966,7 @@ public class ReplicaPlans
|
||||||
forRangeReadLiveAndDown.all(),
|
forRangeReadLiveAndDown.all(),
|
||||||
vnodeCount,
|
vnodeCount,
|
||||||
(newClusterMetadata) -> forRangeRead(newClusterMetadata, keyspace, indexQueryPlan, consistencyLevel, range, vnodeCount, false),
|
(newClusterMetadata) -> forRangeRead(newClusterMetadata, keyspace, indexQueryPlan, consistencyLevel, range, vnodeCount, false),
|
||||||
(self, token) -> forReadRepair(self, metadata, keyspace, consistencyLevel, token, FailureDetector.isReplicaAlive),
|
(self, token) -> forReadRepair(self, metadata, keyspace, consistencyLevel, token, FailureDetector.isReplicaAlive, ReadCoordinator.DEFAULT),
|
||||||
metadata.epoch);
|
metadata.epoch);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -1041,7 +1045,7 @@ public class ReplicaPlans
|
||||||
(self, token) -> {
|
(self, token) -> {
|
||||||
// It might happen that the ring has moved forward since the operation has started, but because we'll be recomputing a quorum
|
// It might happen that the ring has moved forward since the operation has started, but because we'll be recomputing a quorum
|
||||||
// after the operation is complete, we will catch inconsistencies either way.
|
// after the operation is complete, we will catch inconsistencies either way.
|
||||||
return forReadRepair(self, ClusterMetadata.current(), keyspace, consistencyLevel, token, FailureDetector.isReplicaAlive);
|
return forReadRepair(self, ClusterMetadata.current(), keyspace, consistencyLevel, token, FailureDetector.isReplicaAlive, ReadCoordinator.DEFAULT);
|
||||||
},
|
},
|
||||||
left.epoch);
|
left.epoch);
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -19,6 +19,7 @@
|
||||||
package org.apache.cassandra.metrics;
|
package org.apache.cassandra.metrics;
|
||||||
|
|
||||||
import com.codahale.metrics.Histogram;
|
import com.codahale.metrics.Histogram;
|
||||||
|
import com.codahale.metrics.Meter;
|
||||||
|
|
||||||
import static org.apache.cassandra.metrics.CassandraMetricsRegistry.Metrics;
|
import static org.apache.cassandra.metrics.CassandraMetricsRegistry.Metrics;
|
||||||
|
|
||||||
|
|
@ -26,11 +27,27 @@ public class AccordClientRequestMetrics extends ClientRequestMetrics
|
||||||
{
|
{
|
||||||
public final Histogram keySize;
|
public final Histogram keySize;
|
||||||
|
|
||||||
|
// During migration back to Paxos it's possible a transaction runs
|
||||||
|
// in an Epoch where Accord is no longer accepting transactions
|
||||||
|
// and we still run it to completion, but we do skip the read from Cassandra
|
||||||
|
// although it would be harmless. This should only occur briefly when coordinators
|
||||||
|
// start transactions on the wrong protocol due to temporarily out of data cluster metadata.
|
||||||
|
public final Meter migrationSkippedReads;
|
||||||
|
|
||||||
|
// Number of times a key had to be run through PaxosRepair for migration to Accord
|
||||||
|
public final Meter paxosKeyMigrations;
|
||||||
|
|
||||||
|
// Number of times a query was rejected by Accord in TxnQuery due to a migration back to Paxos
|
||||||
|
public final Meter accordMigrationRejects;
|
||||||
|
|
||||||
public AccordClientRequestMetrics(String scope)
|
public AccordClientRequestMetrics(String scope)
|
||||||
{
|
{
|
||||||
super(scope);
|
super(scope);
|
||||||
|
|
||||||
keySize = Metrics.histogram(factory.createMetricName("KeySizeHistogram"), false);
|
keySize = Metrics.histogram(factory.createMetricName("KeySizeHistogram"), false);
|
||||||
|
migrationSkippedReads = Metrics.meter(factory.createMetricName("MigrationSkippedReads"));
|
||||||
|
paxosKeyMigrations = Metrics.meter(factory.createMetricName("PaxosKeyMigrations"));
|
||||||
|
accordMigrationRejects = Metrics.meter(factory.createMetricName("AccordMigrationRejects"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
|
|
@ -38,5 +55,8 @@ public class AccordClientRequestMetrics extends ClientRequestMetrics
|
||||||
{
|
{
|
||||||
super.release();
|
super.release();
|
||||||
Metrics.remove(factory.createMetricName("KeySizeHistogram"));
|
Metrics.remove(factory.createMetricName("KeySizeHistogram"));
|
||||||
|
Metrics.remove(factory.createMetricName("MigrationSkippedReads"));
|
||||||
|
Metrics.remove(factory.createMetricName("PaxosKeyMigrations"));
|
||||||
|
Metrics.remove(factory.createMetricName("AccordMigrationRejects"));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -29,6 +29,12 @@ public class CASClientRequestMetrics extends ClientRequestMetrics
|
||||||
public final Histogram contention;
|
public final Histogram contention;
|
||||||
public final Counter unfinishedCommit;
|
public final Counter unfinishedCommit;
|
||||||
public final Meter unknownResult;
|
public final Meter unknownResult;
|
||||||
|
// CAS request rejected after Prepare/Promise due to migration from Paxos to Accord
|
||||||
|
public final Meter beginMigrationRejects;
|
||||||
|
// Number of times a CAS request was rejected after Propose/Accept due to migration from Paxos to Accord
|
||||||
|
public final Meter acceptMigrationRejects;
|
||||||
|
// Number of times a key was migrated from Accord to Paxos
|
||||||
|
public final Meter accordKeyMigrations;
|
||||||
|
|
||||||
public CASClientRequestMetrics(String scope)
|
public CASClientRequestMetrics(String scope)
|
||||||
{
|
{
|
||||||
|
|
@ -36,6 +42,9 @@ public class CASClientRequestMetrics extends ClientRequestMetrics
|
||||||
contention = Metrics.histogram(factory.createMetricName("ContentionHistogram"), false);
|
contention = Metrics.histogram(factory.createMetricName("ContentionHistogram"), false);
|
||||||
unfinishedCommit = Metrics.counter(factory.createMetricName("UnfinishedCommit"));
|
unfinishedCommit = Metrics.counter(factory.createMetricName("UnfinishedCommit"));
|
||||||
unknownResult = Metrics.meter(factory.createMetricName("UnknownResult"));
|
unknownResult = Metrics.meter(factory.createMetricName("UnknownResult"));
|
||||||
|
beginMigrationRejects = Metrics.meter(factory.createMetricName("PaxosBeginMigrationRejects"));
|
||||||
|
acceptMigrationRejects = Metrics.meter(factory.createMetricName("PaxosAcceptMigrationRejects"));
|
||||||
|
accordKeyMigrations = Metrics.meter(factory.createMetricName("AccordKeyMigrations"));
|
||||||
}
|
}
|
||||||
|
|
||||||
public void release()
|
public void release()
|
||||||
|
|
@ -44,5 +53,8 @@ public class CASClientRequestMetrics extends ClientRequestMetrics
|
||||||
Metrics.remove(factory.createMetricName("ContentionHistogram"));
|
Metrics.remove(factory.createMetricName("ContentionHistogram"));
|
||||||
Metrics.remove(factory.createMetricName("UnfinishedCommit"));
|
Metrics.remove(factory.createMetricName("UnfinishedCommit"));
|
||||||
Metrics.remove(factory.createMetricName("UnknownResult"));
|
Metrics.remove(factory.createMetricName("UnknownResult"));
|
||||||
|
Metrics.remove(factory.createMetricName("PaxosBeginMigrationRejects"));
|
||||||
|
Metrics.remove(factory.createMetricName("PaxosAcceptMigrationRejects"));
|
||||||
|
Metrics.remove(factory.createMetricName("AccordKeyMigrations"));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -29,6 +29,8 @@ public final class ClientRequestsMetricsHolder
|
||||||
public static final CASClientWriteRequestMetrics casWriteMetrics = new CASClientWriteRequestMetrics("CASWrite");
|
public static final CASClientWriteRequestMetrics casWriteMetrics = new CASClientWriteRequestMetrics("CASWrite");
|
||||||
public static final CASClientRequestMetrics casReadMetrics = new CASClientRequestMetrics("CASRead");
|
public static final CASClientRequestMetrics casReadMetrics = new CASClientRequestMetrics("CASRead");
|
||||||
public static final ViewWriteMetrics viewWriteMetrics = new ViewWriteMetrics("ViewWrite");
|
public static final ViewWriteMetrics viewWriteMetrics = new ViewWriteMetrics("ViewWrite");
|
||||||
|
public static final AccordClientRequestMetrics accordReadMetrics = new AccordClientRequestMetrics("AccordRead");
|
||||||
|
public static final AccordClientRequestMetrics accordWriteMetrics = new AccordClientRequestMetrics("AccordWrite");
|
||||||
|
|
||||||
public static final Map<ConsistencyLevel, ClientRequestMetrics> readMetricsMap = new EnumMap<>(ConsistencyLevel.class);
|
public static final Map<ConsistencyLevel, ClientRequestMetrics> readMetricsMap = new EnumMap<>(ConsistencyLevel.class);
|
||||||
public static final Map<ConsistencyLevel, ClientWriteRequestMetrics> writeMetricsMap = new EnumMap<>(ConsistencyLevel.class);
|
public static final Map<ConsistencyLevel, ClientWriteRequestMetrics> writeMetricsMap = new EnumMap<>(ConsistencyLevel.class);
|
||||||
|
|
|
||||||
|
|
@ -101,6 +101,12 @@ public class KeyspaceMetrics
|
||||||
public final LatencyMetrics casPropose;
|
public final LatencyMetrics casPropose;
|
||||||
/** CAS Commit metrics */
|
/** CAS Commit metrics */
|
||||||
public final LatencyMetrics casCommit;
|
public final LatencyMetrics casCommit;
|
||||||
|
/** Latency for locally run key migrations **/
|
||||||
|
public final LatencyMetrics keyMigration;
|
||||||
|
/** Latency for range migrations run by locally coordinated Accord repairs **/
|
||||||
|
public final LatencyMetrics rangeMigration;
|
||||||
|
public final Meter rangeMigrationUnexpectedFailures;
|
||||||
|
public final Meter rangeMigrationDependencyLimitFailures;
|
||||||
/** Writes failed ideal consistency **/
|
/** Writes failed ideal consistency **/
|
||||||
public final Counter writeFailedIdealCL;
|
public final Counter writeFailedIdealCL;
|
||||||
/** Ideal CL write latency metrics */
|
/** Ideal CL write latency metrics */
|
||||||
|
|
@ -247,6 +253,10 @@ public class KeyspaceMetrics
|
||||||
casPrepare = createLatencyMetrics("CasPrepare");
|
casPrepare = createLatencyMetrics("CasPrepare");
|
||||||
casPropose = createLatencyMetrics("CasPropose");
|
casPropose = createLatencyMetrics("CasPropose");
|
||||||
casCommit = createLatencyMetrics("CasCommit");
|
casCommit = createLatencyMetrics("CasCommit");
|
||||||
|
keyMigration = createLatencyMetrics("KeyMigration");
|
||||||
|
rangeMigration = createLatencyMetrics("RangeMigration");
|
||||||
|
rangeMigrationUnexpectedFailures = createKeyspaceMeter("RangeMigrationUnexpectedFailures");
|
||||||
|
rangeMigrationDependencyLimitFailures = createKeyspaceMeter("RangeMigratingDependencyLimitFailures");
|
||||||
writeFailedIdealCL = createKeyspaceCounter("WriteFailedIdealCL");
|
writeFailedIdealCL = createKeyspaceCounter("WriteFailedIdealCL");
|
||||||
idealCLWriteLatency = createLatencyMetrics("IdealCLWrite");
|
idealCLWriteLatency = createLatencyMetrics("IdealCLWrite");
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -87,6 +87,8 @@ public class TableMetrics
|
||||||
public final static LatencyMetrics GLOBAL_READ_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "Read");
|
public final static LatencyMetrics GLOBAL_READ_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "Read");
|
||||||
public final static LatencyMetrics GLOBAL_WRITE_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "Write");
|
public final static LatencyMetrics GLOBAL_WRITE_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "Write");
|
||||||
public final static LatencyMetrics GLOBAL_RANGE_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "Range");
|
public final static LatencyMetrics GLOBAL_RANGE_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "Range");
|
||||||
|
public final static LatencyMetrics GLOBAL_KEY_MIGRATION_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "KeyMigration");
|
||||||
|
public final static LatencyMetrics GLOBAL_RANGE_MIGRATION_LATENCY = new LatencyMetrics(GLOBAL_FACTORY, GLOBAL_ALIAS_FACTORY, "RangeMigration");
|
||||||
|
|
||||||
/** Total amount of data stored in the memtable that resides on-heap, including column related overhead and partitions overwritten. */
|
/** Total amount of data stored in the memtable that resides on-heap, including column related overhead and partitions overwritten. */
|
||||||
public final Gauge<Long> memtableOnHeapDataSize;
|
public final Gauge<Long> memtableOnHeapDataSize;
|
||||||
|
|
@ -188,6 +190,12 @@ public class TableMetrics
|
||||||
public final LatencyMetrics casPropose;
|
public final LatencyMetrics casPropose;
|
||||||
/** CAS Commit metrics */
|
/** CAS Commit metrics */
|
||||||
public final LatencyMetrics casCommit;
|
public final LatencyMetrics casCommit;
|
||||||
|
/** Latency for locally run key migrations **/
|
||||||
|
public final LatencyMetrics keyMigration;
|
||||||
|
/** Latency for range migrations run by locally coordinated Accord repairs **/
|
||||||
|
public final LatencyMetrics rangeMigration;
|
||||||
|
public final TableMeter rangeMigrationUnexpectedFailures;
|
||||||
|
public final TableMeter rangeMigrationDependencyLimitFailures;
|
||||||
/** percent of the data that is repaired */
|
/** percent of the data that is repaired */
|
||||||
public final Gauge<Double> percentRepaired;
|
public final Gauge<Double> percentRepaired;
|
||||||
/** Reports the size of sstables in repaired, unrepaired, and any ongoing repair buckets */
|
/** Reports the size of sstables in repaired, unrepaired, and any ongoing repair buckets */
|
||||||
|
|
@ -624,6 +632,7 @@ public class TableMetrics
|
||||||
readLatency = createLatencyMetrics("Read", cfs.keyspace.metric.readLatency, GLOBAL_READ_LATENCY);
|
readLatency = createLatencyMetrics("Read", cfs.keyspace.metric.readLatency, GLOBAL_READ_LATENCY);
|
||||||
writeLatency = createLatencyMetrics("Write", cfs.keyspace.metric.writeLatency, GLOBAL_WRITE_LATENCY);
|
writeLatency = createLatencyMetrics("Write", cfs.keyspace.metric.writeLatency, GLOBAL_WRITE_LATENCY);
|
||||||
rangeLatency = createLatencyMetrics("Range", cfs.keyspace.metric.rangeLatency, GLOBAL_RANGE_LATENCY);
|
rangeLatency = createLatencyMetrics("Range", cfs.keyspace.metric.rangeLatency, GLOBAL_RANGE_LATENCY);
|
||||||
|
|
||||||
pendingFlushes = createTableCounter("PendingFlushes");
|
pendingFlushes = createTableCounter("PendingFlushes");
|
||||||
bytesFlushed = createTableCounter("BytesFlushed");
|
bytesFlushed = createTableCounter("BytesFlushed");
|
||||||
flushSizeOnDisk = ExpMovingAverage.decayBy1000();
|
flushSizeOnDisk = ExpMovingAverage.decayBy1000();
|
||||||
|
|
@ -804,6 +813,10 @@ public class TableMetrics
|
||||||
casPrepare = createLatencyMetrics("CasPrepare", cfs.keyspace.metric.casPrepare);
|
casPrepare = createLatencyMetrics("CasPrepare", cfs.keyspace.metric.casPrepare);
|
||||||
casPropose = createLatencyMetrics("CasPropose", cfs.keyspace.metric.casPropose);
|
casPropose = createLatencyMetrics("CasPropose", cfs.keyspace.metric.casPropose);
|
||||||
casCommit = createLatencyMetrics("CasCommit", cfs.keyspace.metric.casCommit);
|
casCommit = createLatencyMetrics("CasCommit", cfs.keyspace.metric.casCommit);
|
||||||
|
keyMigration = createLatencyMetrics("KeyMigration", cfs.keyspace.metric.keyMigration, GLOBAL_KEY_MIGRATION_LATENCY);
|
||||||
|
rangeMigration = createLatencyMetrics("RangeMigration", cfs.keyspace.metric.rangeMigration, GLOBAL_RANGE_MIGRATION_LATENCY);
|
||||||
|
rangeMigrationUnexpectedFailures = createTableMeter("RangeMigrationUnexpectedFailures", cfs.keyspace.metric.rangeMigrationUnexpectedFailures);
|
||||||
|
rangeMigrationDependencyLimitFailures = createTableMeter("RangeMigrationDependencyLimitFaiures", cfs.keyspace.metric.rangeMigrationDependencyLimitFailures);
|
||||||
|
|
||||||
repairsStarted = createTableCounter("RepairJobsStarted");
|
repairsStarted = createTableCounter("RepairJobsStarted");
|
||||||
repairsCompleted = createTableCounter("RepairJobsCompleted");
|
repairsCompleted = createTableCounter("RepairJobsCompleted");
|
||||||
|
|
|
||||||
|
|
@ -303,6 +303,7 @@ public class Message<T> implements ReplyContext
|
||||||
* Used by the {@code MultiRangeReadCommand} to split multi-range responses from a replica
|
* Used by the {@code MultiRangeReadCommand} to split multi-range responses from a replica
|
||||||
* into single-range responses.
|
* into single-range responses.
|
||||||
*/
|
*/
|
||||||
|
@VisibleForTesting
|
||||||
public static <T> Message<T> remoteResponse(InetAddressAndPort from, Verb verb, T payload)
|
public static <T> Message<T> remoteResponse(InetAddressAndPort from, Verb verb, T payload)
|
||||||
{
|
{
|
||||||
assert verb.isResponse();
|
assert verb.isResponse();
|
||||||
|
|
@ -574,6 +575,11 @@ public class Message<T> implements ReplyContext
|
||||||
return MessageFlag.TRACK_WARNINGS.isIn(flags);
|
return MessageFlag.TRACK_WARNINGS.isIn(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
boolean isFinal()
|
||||||
|
{
|
||||||
|
return !MessageFlag.NOT_FINAL.isIn(flags);
|
||||||
|
}
|
||||||
|
|
||||||
@Nullable
|
@Nullable
|
||||||
ForwardingInfo forwardTo()
|
ForwardingInfo forwardTo()
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -31,7 +31,10 @@ public enum MessageFlag
|
||||||
/** allow creating warnings or aborting queries based off query - see CASSANDRA-16850 */
|
/** allow creating warnings or aborting queries based off query - see CASSANDRA-16850 */
|
||||||
TRACK_WARNINGS(2),
|
TRACK_WARNINGS(2),
|
||||||
/** whether this message should be sent on an URGENT channel despite its Verb default priority */
|
/** whether this message should be sent on an URGENT channel despite its Verb default priority */
|
||||||
URGENT(3);
|
URGENT(3),
|
||||||
|
/** Allow a single callback to receive multiple responses until a final response is received **/
|
||||||
|
NOT_FINAL(4)
|
||||||
|
;
|
||||||
|
|
||||||
private final int id;
|
private final int id;
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -500,9 +500,9 @@ public class MessagingService extends MessagingServiceMBeanImpl implements Messa
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public void onFailure(InetAddressAndPort from, RequestFailure failureReason)
|
public void onFailure(InetAddressAndPort from, RequestFailure failure)
|
||||||
{
|
{
|
||||||
future.setFailure(new RuntimeException(failureReason.toString()));
|
future.setFailure(new RuntimeException(failure.toString()));
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -58,7 +58,11 @@ class ResponseVerbHandler implements IVerbHandler
|
||||||
@Override
|
@Override
|
||||||
public void doVerb(Message message)
|
public void doVerb(Message message)
|
||||||
{
|
{
|
||||||
RequestCallbacks.CallbackInfo callbackInfo = MessagingService.instance().callbacks.remove(message.id(), message.from());
|
RequestCallbacks.CallbackInfo callbackInfo;
|
||||||
|
if (message.header.isFinal())
|
||||||
|
callbackInfo = MessagingService.instance().callbacks.remove(message.id(), message.from());
|
||||||
|
else
|
||||||
|
callbackInfo = MessagingService.instance().callbacks.get(message.id(), message.from());
|
||||||
if (callbackInfo == null)
|
if (callbackInfo == null)
|
||||||
{
|
{
|
||||||
String msg = "Callback already removed for {} (from {})";
|
String msg = "Callback already removed for {} (from {})";
|
||||||
|
|
|
||||||
|
|
@ -81,6 +81,10 @@ import org.apache.cassandra.service.SnapshotVerbHandler;
|
||||||
import org.apache.cassandra.service.accord.AccordService;
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
import org.apache.cassandra.service.accord.AccordSyncPropagator;
|
import org.apache.cassandra.service.accord.AccordSyncPropagator;
|
||||||
import org.apache.cassandra.service.accord.AccordSyncPropagator.Notification;
|
import org.apache.cassandra.service.accord.AccordSyncPropagator.Notification;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropApply;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropCommit;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropRead;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropReadRepair;
|
||||||
import org.apache.cassandra.service.accord.serializers.AcceptSerializers;
|
import org.apache.cassandra.service.accord.serializers.AcceptSerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.ApplySerializers;
|
import org.apache.cassandra.service.accord.serializers.ApplySerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.BeginInvalidationSerializers;
|
import org.apache.cassandra.service.accord.serializers.BeginInvalidationSerializers;
|
||||||
|
|
@ -98,6 +102,8 @@ import org.apache.cassandra.service.accord.serializers.ReadDataSerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.RecoverySerializers;
|
import org.apache.cassandra.service.accord.serializers.RecoverySerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.SetDurableSerializers;
|
import org.apache.cassandra.service.accord.serializers.SetDurableSerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.WaitOnCommitSerializer;
|
import org.apache.cassandra.service.accord.serializers.WaitOnCommitSerializer;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.ConsensusKeyMigrationFinished;
|
||||||
import org.apache.cassandra.service.paxos.Commit;
|
import org.apache.cassandra.service.paxos.Commit;
|
||||||
import org.apache.cassandra.service.paxos.Commit.Agreed;
|
import org.apache.cassandra.service.paxos.Commit.Agreed;
|
||||||
import org.apache.cassandra.service.paxos.PaxosCommit;
|
import org.apache.cassandra.service.paxos.PaxosCommit;
|
||||||
|
|
@ -260,7 +266,7 @@ public enum Verb
|
||||||
PAXOS2_PREPARE_REQ (40, P2, writeTimeout, MUTATION, () -> PaxosPrepare.requestSerializer, () -> PaxosPrepare.requestHandler, PAXOS2_PREPARE_RSP ),
|
PAXOS2_PREPARE_REQ (40, P2, writeTimeout, MUTATION, () -> PaxosPrepare.requestSerializer, () -> PaxosPrepare.requestHandler, PAXOS2_PREPARE_RSP ),
|
||||||
PAXOS2_PREPARE_REFRESH_RSP (51, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPrepareRefresh.responseSerializer, RESPONSE_HANDLER ),
|
PAXOS2_PREPARE_REFRESH_RSP (51, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPrepareRefresh.responseSerializer, RESPONSE_HANDLER ),
|
||||||
PAXOS2_PREPARE_REFRESH_REQ (41, P2, writeTimeout, MUTATION, () -> PaxosPrepareRefresh.requestSerializer, () -> PaxosPrepareRefresh.requestHandler, PAXOS2_PREPARE_REFRESH_RSP ),
|
PAXOS2_PREPARE_REFRESH_REQ (41, P2, writeTimeout, MUTATION, () -> PaxosPrepareRefresh.requestSerializer, () -> PaxosPrepareRefresh.requestHandler, PAXOS2_PREPARE_REFRESH_RSP ),
|
||||||
PAXOS2_PROPOSE_RSP (52, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPropose.responseSerializer, RESPONSE_HANDLER ),
|
PAXOS2_PROPOSE_RSP (52, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPropose.ACCEPT_RESULT_SERIALIZER, RESPONSE_HANDLER ),
|
||||||
PAXOS2_PROPOSE_REQ (42, P2, writeTimeout, MUTATION, () -> PaxosPropose.requestSerializer, () -> PaxosPropose.requestHandler, PAXOS2_PROPOSE_RSP ),
|
PAXOS2_PROPOSE_REQ (42, P2, writeTimeout, MUTATION, () -> PaxosPropose.requestSerializer, () -> PaxosPropose.requestHandler, PAXOS2_PROPOSE_RSP ),
|
||||||
PAXOS2_COMMIT_AND_PREPARE_RSP (53, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPrepare.responseSerializer, RESPONSE_HANDLER ),
|
PAXOS2_COMMIT_AND_PREPARE_RSP (53, P2, writeTimeout, REQUEST_RESPONSE, () -> PaxosPrepare.responseSerializer, RESPONSE_HANDLER ),
|
||||||
PAXOS2_COMMIT_AND_PREPARE_REQ (43, P2, writeTimeout, MUTATION, () -> PaxosCommitAndPrepare.requestSerializer, () -> PaxosCommitAndPrepare.requestHandler, PAXOS2_COMMIT_AND_PREPARE_RSP ),
|
PAXOS2_COMMIT_AND_PREPARE_REQ (43, P2, writeTimeout, MUTATION, () -> PaxosCommitAndPrepare.requestSerializer, () -> PaxosCommitAndPrepare.requestHandler, PAXOS2_COMMIT_AND_PREPARE_RSP ),
|
||||||
|
|
@ -300,40 +306,51 @@ public enum Verb
|
||||||
|
|
||||||
// accord
|
// accord
|
||||||
ACCORD_SIMPLE_RSP (119, P2, writeTimeout, REQUEST_RESPONSE, () -> EnumSerializer.simpleReply, RESPONSE_HANDLER ),
|
ACCORD_SIMPLE_RSP (119, P2, writeTimeout, REQUEST_RESPONSE, () -> EnumSerializer.simpleReply, RESPONSE_HANDLER ),
|
||||||
ACCORD_PRE_ACCEPT_RSP (121, P2, writeTimeout, REQUEST_RESPONSE, () -> PreacceptSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_PRE_ACCEPT_RSP (120, P2, writeTimeout, REQUEST_RESPONSE, () -> PreacceptSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_PRE_ACCEPT_REQ (120, P2, writeTimeout, IMMEDIATE, () -> PreacceptSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_PRE_ACCEPT_RSP ),
|
ACCORD_PRE_ACCEPT_REQ (121, P2, writeTimeout, IMMEDIATE, () -> PreacceptSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_PRE_ACCEPT_RSP ),
|
||||||
ACCORD_ACCEPT_RSP (124, P2, writeTimeout, REQUEST_RESPONSE, () -> AcceptSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_ACCEPT_RSP (122, P2, writeTimeout, REQUEST_RESPONSE, () -> AcceptSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_ACCEPT_REQ (122, P2, writeTimeout, IMMEDIATE, () -> AcceptSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_ACCEPT_RSP ),
|
ACCORD_ACCEPT_REQ (123, P2, writeTimeout, IMMEDIATE, () -> AcceptSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_ACCEPT_RSP ),
|
||||||
ACCORD_ACCEPT_INVALIDATE_REQ (123, P2, writeTimeout, IMMEDIATE, () -> AcceptSerializers.invalidate, AccordService::verbHandlerOrNoop, ACCORD_ACCEPT_RSP ),
|
ACCORD_ACCEPT_INVALIDATE_REQ (124, P2, writeTimeout, IMMEDIATE, () -> AcceptSerializers.invalidate, AccordService::verbHandlerOrNoop, ACCORD_ACCEPT_RSP ),
|
||||||
ACCORD_READ_RSP (126, P2, writeTimeout, REQUEST_RESPONSE, () -> ReadDataSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_READ_RSP (125, P2, writeTimeout, REQUEST_RESPONSE, () -> ReadDataSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_READ_REQ (125, P2, writeTimeout, IMMEDIATE, () -> ReadDataSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_READ_RSP ),
|
ACCORD_READ_REQ (126, P2, writeTimeout, IMMEDIATE, () -> ReadDataSerializers.readData, AccordService::verbHandlerOrNoop, ACCORD_READ_RSP ),
|
||||||
ACCORD_COMMIT_REQ (127, P2, writeTimeout, IMMEDIATE, () -> CommitSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_READ_RSP ),
|
ACCORD_COMMIT_REQ (127, P2, writeTimeout, IMMEDIATE, () -> CommitSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_READ_RSP ),
|
||||||
ACCORD_COMMIT_INVALIDATE_REQ (128, P2, writeTimeout, IMMEDIATE, () -> CommitSerializers.invalidate, AccordService::verbHandlerOrNoop ),
|
ACCORD_COMMIT_INVALIDATE_REQ (128, P2, writeTimeout, IMMEDIATE, () -> CommitSerializers.invalidate, AccordService::verbHandlerOrNoop ),
|
||||||
ACCORD_APPLY_RSP (130, P2, writeTimeout, REQUEST_RESPONSE, () -> ApplySerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_APPLY_RSP (129, P2, writeTimeout, REQUEST_RESPONSE, () -> ApplySerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_APPLY_REQ (129, P2, writeTimeout, IMMEDIATE, () -> ApplySerializers.request, AccordService::verbHandlerOrNoop, ACCORD_APPLY_RSP ),
|
ACCORD_APPLY_REQ (130, P2, writeTimeout, IMMEDIATE, () -> ApplySerializers.request, AccordService::verbHandlerOrNoop, ACCORD_APPLY_RSP ),
|
||||||
ACCORD_BEGIN_RECOVER_RSP (132, P2, writeTimeout, REQUEST_RESPONSE, () -> RecoverySerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_BEGIN_RECOVER_RSP (131, P2, writeTimeout, REQUEST_RESPONSE, () -> RecoverySerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_BEGIN_RECOVER_REQ (131, P2, writeTimeout, IMMEDIATE, () -> RecoverySerializers.request, AccordService::verbHandlerOrNoop, ACCORD_BEGIN_RECOVER_RSP ),
|
ACCORD_BEGIN_RECOVER_REQ (132, P2, writeTimeout, IMMEDIATE, () -> RecoverySerializers.request, AccordService::verbHandlerOrNoop, ACCORD_BEGIN_RECOVER_RSP ),
|
||||||
ACCORD_BEGIN_INVALIDATE_RSP (134, P2, writeTimeout, REQUEST_RESPONSE, () -> BeginInvalidationSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_BEGIN_INVALIDATE_RSP (133, P2, writeTimeout, REQUEST_RESPONSE, () -> BeginInvalidationSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_BEGIN_INVALIDATE_REQ (133, P2, writeTimeout, IMMEDIATE, () -> BeginInvalidationSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_BEGIN_INVALIDATE_RSP ),
|
ACCORD_BEGIN_INVALIDATE_REQ (134, P2, writeTimeout, IMMEDIATE, () -> BeginInvalidationSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_BEGIN_INVALIDATE_RSP ),
|
||||||
ACCORD_WAIT_ON_COMMIT_RSP (136, P2, writeTimeout, REQUEST_RESPONSE, () -> WaitOnCommitSerializer.reply, RESPONSE_HANDLER ),
|
ACCORD_WAIT_ON_COMMIT_RSP (136, P2, writeTimeout, REQUEST_RESPONSE, () -> WaitOnCommitSerializer.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_WAIT_ON_COMMIT_REQ (135, P2, writeTimeout, IMMEDIATE, () -> WaitOnCommitSerializer.request, AccordService::verbHandlerOrNoop, ACCORD_WAIT_ON_COMMIT_RSP ),
|
ACCORD_WAIT_ON_COMMIT_REQ (135, P2, writeTimeout, IMMEDIATE, () -> WaitOnCommitSerializer.request, AccordService::verbHandlerOrNoop, ACCORD_WAIT_ON_COMMIT_RSP ),
|
||||||
ACCORD_WAIT_ON_APPLY_REQ (137, P2, writeTimeout, IMMEDIATE, () -> ReadDataSerializers.waitOnApply, AccordService::verbHandlerOrNoop, ACCORD_READ_RSP ),
|
ACCORD_WAIT_UNTIL_APPLIED_REQ (137, P2, writeTimeout, IMMEDIATE, () -> ReadDataSerializers.waitUntilApplied, AccordService::verbHandlerOrNoop, ACCORD_READ_RSP ),
|
||||||
ACCORD_INFORM_OF_TXN_REQ (138, P2, writeTimeout, IMMEDIATE, () -> InformOfTxnIdSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
ACCORD_INFORM_OF_TXN_REQ (138, P2, writeTimeout, IMMEDIATE, () -> InformOfTxnIdSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
||||||
ACCORD_INFORM_HOME_DURABLE_REQ (139, P2, writeTimeout, IMMEDIATE, () -> InformHomeDurableSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
ACCORD_INFORM_HOME_DURABLE_REQ (139, P2, writeTimeout, IMMEDIATE, () -> InformHomeDurableSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
||||||
ACCORD_INFORM_DURABLE_REQ (140, P2, writeTimeout, IMMEDIATE, () -> InformDurableSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
ACCORD_INFORM_DURABLE_REQ (140, P2, writeTimeout, IMMEDIATE, () -> InformDurableSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
||||||
ACCORD_CHECK_STATUS_RSP (142, P2, writeTimeout, REQUEST_RESPONSE, () -> CheckStatusSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_CHECK_STATUS_RSP (141, P2, writeTimeout, REQUEST_RESPONSE, () -> CheckStatusSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_CHECK_STATUS_REQ (141, P2, writeTimeout, IMMEDIATE, () -> CheckStatusSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_CHECK_STATUS_RSP ),
|
ACCORD_CHECK_STATUS_REQ (142, P2, writeTimeout, IMMEDIATE, () -> CheckStatusSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_CHECK_STATUS_RSP ),
|
||||||
ACCORD_GET_DEPS_RSP (144, P2, writeTimeout, REQUEST_RESPONSE, () -> GetDepsSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_GET_DEPS_RSP (143, P2, writeTimeout, REQUEST_RESPONSE, () -> GetDepsSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_GET_DEPS_REQ (143, P2, writeTimeout, IMMEDIATE, () -> GetDepsSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_GET_DEPS_RSP ),
|
ACCORD_GET_DEPS_REQ (144, P2, writeTimeout, IMMEDIATE, () -> GetDepsSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_GET_DEPS_RSP ),
|
||||||
ACCORD_FETCH_DATA_RSP (146, P2, repairTimeout,REQUEST_RESPONSE, () -> FetchSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_FETCH_DATA_RSP (145, P2, repairTimeout,REQUEST_RESPONSE, () -> FetchSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_FETCH_DATA_REQ (145, P2, repairTimeout,IMMEDIATE, () -> FetchSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_FETCH_DATA_RSP ),
|
ACCORD_FETCH_DATA_REQ (146, P2, repairTimeout,IMMEDIATE, () -> FetchSerializers.request, AccordService::verbHandlerOrNoop, ACCORD_FETCH_DATA_RSP ),
|
||||||
ACCORD_SET_SHARD_DURABLE_REQ (147, P2, writeTimeout, IMMEDIATE, () -> SetDurableSerializers.shardDurable, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
ACCORD_SET_SHARD_DURABLE_REQ (147, P2, writeTimeout, IMMEDIATE, () -> SetDurableSerializers.shardDurable, AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
||||||
ACCORD_SET_GLOBALLY_DURABLE_REQ (148, P2, writeTimeout, IMMEDIATE, () -> SetDurableSerializers.globallyDurable,AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
ACCORD_SET_GLOBALLY_DURABLE_REQ (148, P2, writeTimeout, IMMEDIATE, () -> SetDurableSerializers.globallyDurable,AccordService::verbHandlerOrNoop, ACCORD_SIMPLE_RSP ),
|
||||||
ACCORD_QUERY_DURABLE_BEFORE_RSP (150, P2, writeTimeout, REQUEST_RESPONSE, () -> QueryDurableBeforeSerializers.reply, RESPONSE_HANDLER ),
|
ACCORD_QUERY_DURABLE_BEFORE_RSP (149, P2, writeTimeout, REQUEST_RESPONSE, () -> QueryDurableBeforeSerializers.reply, RESPONSE_HANDLER ),
|
||||||
ACCORD_QUERY_DURABLE_BEFORE_REQ (149, P2, writeTimeout, IMMEDIATE, () -> QueryDurableBeforeSerializers.request,AccordService::verbHandlerOrNoop, ACCORD_QUERY_DURABLE_BEFORE_RSP),
|
ACCORD_QUERY_DURABLE_BEFORE_REQ (150, P2, writeTimeout, IMMEDIATE, () -> QueryDurableBeforeSerializers.request,AccordService::verbHandlerOrNoop, ACCORD_QUERY_DURABLE_BEFORE_RSP ),
|
||||||
|
|
||||||
ACCORD_SYNC_NOTIFY_REQ (151, P2, writeTimeout, IMMEDIATE, () -> Notification.listSerializer, () -> AccordSyncPropagator.verbHandler, ACCORD_SIMPLE_RSP ),
|
ACCORD_SYNC_NOTIFY_REQ (151, P2, writeTimeout, IMMEDIATE, () -> Notification.listSerializer, () -> AccordSyncPropagator.verbHandler, ACCORD_SIMPLE_RSP ),
|
||||||
|
|
||||||
|
ACCORD_APPLY_AND_WAIT_UNTIL_APPLIED_REQ(152, P2, writeTimeout, IMMEDIATE, () -> ReadDataSerializers.readData,() -> AccordSyncPropagator.verbHandler, ACCORD_READ_RSP),
|
||||||
|
|
||||||
|
CONSENSUS_KEY_MIGRATION (153, P1, writeTimeout, MUTATION, () -> ConsensusKeyMigrationFinished.serializer,() -> ConsensusKeyMigrationState.consensusKeyMigrationFinishedHandler),
|
||||||
|
|
||||||
|
ACCORD_INTEROP_READ_RSP (154, P2, writeTimeout, IMMEDIATE, () -> AccordInteropRead.replySerializer, RESPONSE_HANDLER),
|
||||||
|
ACCORD_INTEROP_READ_REQ (155, P2, writeTimeout, IMMEDIATE, () -> AccordInteropRead.requestSerializer, () -> AccordService.instance().verbHandler(), ACCORD_INTEROP_READ_RSP),
|
||||||
|
ACCORD_INTEROP_COMMIT_REQ (156, P2, writeTimeout, IMMEDIATE, () -> AccordInteropCommit.serializer, () -> AccordService.instance().verbHandler(), ACCORD_INTEROP_READ_RSP),
|
||||||
|
ACCORD_INTEROP_READ_REPAIR_RSP (157, P2, writeTimeout, IMMEDIATE, () -> AccordInteropReadRepair.replySerializer, RESPONSE_HANDLER),
|
||||||
|
ACCORD_INTEROP_READ_REPAIR_REQ (158, P2, writeTimeout, IMMEDIATE, () -> AccordInteropReadRepair.requestSerializer, () -> AccordService.instance().verbHandler(), ACCORD_INTEROP_READ_REPAIR_RSP),
|
||||||
|
ACCORD_INTEROP_APPLY_REQ (160, P2, writeTimeout, IMMEDIATE, () -> AccordInteropApply.serializer, AccordService::verbHandlerOrNoop, ACCORD_APPLY_RSP),
|
||||||
|
|
||||||
// generic failure response
|
// generic failure response
|
||||||
FAILURE_RSP (99, P0, noTimeout, REQUEST_RESPONSE, () -> RequestFailure.serializer, RESPONSE_HANDLER ),
|
FAILURE_RSP (99, P0, noTimeout, REQUEST_RESPONSE, () -> RequestFailure.serializer, RESPONSE_HANDLER ),
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,66 @@
|
||||||
|
/*
|
||||||
|
* Licensed to the Apache Software Foundation (ASF) under one
|
||||||
|
* or more contributor license agreements. See the NOTICE file
|
||||||
|
* distributed with this work for additional information
|
||||||
|
* regarding copyright ownership. The ASF licenses this file
|
||||||
|
* to you under the Apache License, Version 2.0 (the
|
||||||
|
* "License"); you may not use this file except in compliance
|
||||||
|
* with the License. You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package org.apache.cassandra.repair;
|
||||||
|
|
||||||
|
import java.util.concurrent.Executor;
|
||||||
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
|
import org.apache.cassandra.db.ColumnFamilyStore;
|
||||||
|
import org.apache.cassandra.db.Keyspace;
|
||||||
|
import org.apache.cassandra.repair.state.JobState;
|
||||||
|
import org.apache.cassandra.utils.concurrent.AsyncFuture;
|
||||||
|
|
||||||
|
public abstract class AbstractRepairJob extends AsyncFuture<RepairResult> implements Runnable
|
||||||
|
{
|
||||||
|
private final SharedContext ctx;
|
||||||
|
public final JobState state;
|
||||||
|
protected final RepairJobDesc desc;
|
||||||
|
protected final RepairSession session;
|
||||||
|
protected final Executor taskExecutor;
|
||||||
|
|
||||||
|
protected final Keyspace ks;
|
||||||
|
protected final ColumnFamilyStore cfs;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Create repair job to run on specific columnfamily
|
||||||
|
* @param session RepairSession that this RepairJob belongs
|
||||||
|
* @param columnFamily name of the ColumnFamily to repair
|
||||||
|
*/
|
||||||
|
public AbstractRepairJob(RepairSession session, String columnFamily)
|
||||||
|
{
|
||||||
|
this.ctx = session.ctx;
|
||||||
|
this.session = session;
|
||||||
|
this.taskExecutor = session.taskExecutor;
|
||||||
|
this.desc = new RepairJobDesc(session.state.parentRepairSession, session.getId(), session.state.keyspace, columnFamily, session.state.commonRange.ranges);
|
||||||
|
this.state = new JobState(ctx.clock(), desc, session.state.commonRange.endpoints);
|
||||||
|
this.ks = Keyspace.open(desc.keyspace);
|
||||||
|
this.cfs = ks.getColumnFamilyStore(columnFamily);
|
||||||
|
}
|
||||||
|
|
||||||
|
public void run()
|
||||||
|
{
|
||||||
|
state.phase.start();
|
||||||
|
cfs.metric.repairsStarted.inc();
|
||||||
|
runRepair();
|
||||||
|
}
|
||||||
|
|
||||||
|
abstract protected void runRepair();
|
||||||
|
|
||||||
|
abstract void abort(@Nullable Throwable reason);
|
||||||
|
}
|
||||||
|
|
@ -24,7 +24,6 @@ import java.util.Objects;
|
||||||
|
|
||||||
import com.google.common.collect.Lists;
|
import com.google.common.collect.Lists;
|
||||||
import com.google.common.util.concurrent.FutureCallback;
|
import com.google.common.util.concurrent.FutureCallback;
|
||||||
|
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
|
@ -59,6 +58,7 @@ public abstract class AbstractRepairTask implements RepairTask
|
||||||
ExecutorPlus executor,
|
ExecutorPlus executor,
|
||||||
Scheduler validationScheduler,
|
Scheduler validationScheduler,
|
||||||
List<CommonRange> commonRanges,
|
List<CommonRange> commonRanges,
|
||||||
|
boolean excludedDeadNodes,
|
||||||
String... cfnames)
|
String... cfnames)
|
||||||
{
|
{
|
||||||
List<RepairSession> futures = new ArrayList<>(options.getRanges().size());
|
List<RepairSession> futures = new ArrayList<>(options.getRanges().size());
|
||||||
|
|
@ -68,6 +68,7 @@ public abstract class AbstractRepairTask implements RepairTask
|
||||||
logger.info("Starting RepairSession for {}", commonRange);
|
logger.info("Starting RepairSession for {}", commonRange);
|
||||||
RepairSession session = coordinator.ctx.repair().submitRepairSession(parentSession,
|
RepairSession session = coordinator.ctx.repair().submitRepairSession(parentSession,
|
||||||
commonRange,
|
commonRange,
|
||||||
|
excludedDeadNodes,
|
||||||
keyspace,
|
keyspace,
|
||||||
options.getParallelism(),
|
options.getParallelism(),
|
||||||
isIncremental,
|
isIncremental,
|
||||||
|
|
@ -77,6 +78,7 @@ public abstract class AbstractRepairTask implements RepairTask
|
||||||
options.repairPaxos(),
|
options.repairPaxos(),
|
||||||
options.paxosOnly(),
|
options.paxosOnly(),
|
||||||
options.dontPurgeTombstones(),
|
options.dontPurgeTombstones(),
|
||||||
|
options.accordRepair(),
|
||||||
executor,
|
executor,
|
||||||
validationScheduler,
|
validationScheduler,
|
||||||
cfnames);
|
cfnames);
|
||||||
|
|
@ -93,9 +95,10 @@ public abstract class AbstractRepairTask implements RepairTask
|
||||||
ExecutorPlus executor,
|
ExecutorPlus executor,
|
||||||
Scheduler validationScheduler,
|
Scheduler validationScheduler,
|
||||||
List<CommonRange> commonRanges,
|
List<CommonRange> commonRanges,
|
||||||
|
boolean excludedDeadNodes,
|
||||||
String... cfnames)
|
String... cfnames)
|
||||||
{
|
{
|
||||||
List<RepairSession> allSessions = submitRepairSessions(parentSession, isIncremental, executor, validationScheduler, commonRanges, cfnames);
|
List<RepairSession> allSessions = submitRepairSessions(parentSession, isIncremental, executor, validationScheduler, commonRanges, excludedDeadNodes, cfnames);
|
||||||
List<Collection<Range<Token>>> ranges = Lists.transform(allSessions, RepairSession::ranges);
|
List<Collection<Range<Token>>> ranges = Lists.transform(allSessions, RepairSession::ranges);
|
||||||
Future<List<RepairSessionResult>> f = FutureCombiner.successfulOf(allSessions);
|
Future<List<RepairSessionResult>> f = FutureCombiner.successfulOf(allSessions);
|
||||||
return f.map(results -> {
|
return f.map(results -> {
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,173 @@
|
||||||
|
/*
|
||||||
|
* Licensed to the Apache Software Foundation (ASF) under one
|
||||||
|
* or more contributor license agreements. See the NOTICE file
|
||||||
|
* distributed with this work for additional information
|
||||||
|
* regarding copyright ownership. The ASF licenses this file
|
||||||
|
* to you under the Apache License, Version 2.0 (the
|
||||||
|
* "License"); you may not use this file except in compliance
|
||||||
|
* with the License. You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package org.apache.cassandra.repair;
|
||||||
|
|
||||||
|
import java.math.BigInteger;
|
||||||
|
import java.util.List;
|
||||||
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
|
import accord.api.BarrierType;
|
||||||
|
import accord.api.RoutingKey;
|
||||||
|
import accord.primitives.Ranges;
|
||||||
|
import accord.primitives.Seekables;
|
||||||
|
import org.apache.cassandra.dht.AccordSplitter;
|
||||||
|
import org.apache.cassandra.dht.IPartitioner;
|
||||||
|
import org.apache.cassandra.dht.Range;
|
||||||
|
import org.apache.cassandra.dht.Token;
|
||||||
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
|
import org.apache.cassandra.service.accord.TokenRange;
|
||||||
|
import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.ConsensusMigrationRepairResult;
|
||||||
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
|
|
||||||
|
import static com.google.common.base.Preconditions.checkState;
|
||||||
|
import static java.util.Collections.emptyList;
|
||||||
|
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Accord repair consists of creating a barrier transaction for all the ranges which ensure that all Accord transactions
|
||||||
|
* before the Epoch and point in time at which the repair started have their side effects visible to Paxos and regular quorum reads.
|
||||||
|
*/
|
||||||
|
public class AccordRepairJob extends AbstractRepairJob
|
||||||
|
{
|
||||||
|
public static final BigInteger TWO = BigInteger.valueOf(2);
|
||||||
|
|
||||||
|
private final Ranges ranges;
|
||||||
|
|
||||||
|
private final AccordSplitter splitter;
|
||||||
|
|
||||||
|
private BigInteger rangeStep;
|
||||||
|
|
||||||
|
private Epoch minEpoch = ClusterMetadata.current().epoch;
|
||||||
|
|
||||||
|
public AccordRepairJob(RepairSession repairSession, String cfname)
|
||||||
|
{
|
||||||
|
super(repairSession, cfname);
|
||||||
|
List<Range<Token>> normalizedRanges = Range.normalize(desc.ranges);
|
||||||
|
IPartitioner partitioner = normalizedRanges.get(0).left.getPartitioner();
|
||||||
|
TokenRange[] tokenRanges = new TokenRange[normalizedRanges.size()];
|
||||||
|
for (int i = 0; i < normalizedRanges.size(); i++)
|
||||||
|
tokenRanges[i] = new TokenRange(new TokenKey(ks.getName(), normalizedRanges.get(i).left), new TokenKey(ks.getName(), normalizedRanges.get(i).right));
|
||||||
|
this.ranges = Ranges.of(tokenRanges);
|
||||||
|
this.splitter = partitioner.accordSplitter().apply(Ranges.of(tokenRanges));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
protected void runRepair()
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
for (accord.primitives.Range range : ranges)
|
||||||
|
repairRange((TokenRange)range);
|
||||||
|
state.phase.success();
|
||||||
|
cfs.metric.repairsCompleted.inc();
|
||||||
|
trySuccess(new RepairResult(desc, emptyList(), ConsensusMigrationRepairResult.fromAccordRepair(minEpoch)));
|
||||||
|
}
|
||||||
|
catch (Throwable t)
|
||||||
|
{
|
||||||
|
state.phase.fail(t);
|
||||||
|
cfs.metric.repairsCompleted.inc();
|
||||||
|
tryFailure(t);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
void abort(@Nullable Throwable reason)
|
||||||
|
{
|
||||||
|
throw new UnsupportedOperationException("Have not implemented this yet, and the job runs synchronously so it isn't abortable");
|
||||||
|
}
|
||||||
|
|
||||||
|
private void repairRange(TokenRange range)
|
||||||
|
{
|
||||||
|
RoutingKey remainingStart = range.start();
|
||||||
|
BigInteger rangeSize = splitter.sizeOf(range);
|
||||||
|
if (rangeStep == null)
|
||||||
|
rangeStep = BigInteger.ONE.max(splitter.divide(rangeSize, 1000));
|
||||||
|
|
||||||
|
BigInteger offset = BigInteger.ZERO;
|
||||||
|
|
||||||
|
TokenRange lastRepaired = null;
|
||||||
|
int iteration = 0;
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
iteration++;
|
||||||
|
if (iteration % 100 == 0)
|
||||||
|
rangeStep = rangeStep.multiply(TWO);
|
||||||
|
|
||||||
|
BigInteger remaining = rangeSize.subtract(offset);
|
||||||
|
BigInteger length = remaining.min(rangeStep);
|
||||||
|
|
||||||
|
long start = nanoTime();
|
||||||
|
boolean dependencyOverflow = false;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
// Splitter is approximate so it can't work right up to the end
|
||||||
|
TokenRange toRepair;
|
||||||
|
if (splitter.compare(offset, rangeSize) >= 0)
|
||||||
|
{
|
||||||
|
if (remainingStart.equals(range.end()))
|
||||||
|
return;
|
||||||
|
|
||||||
|
// Final repair is whatever remains
|
||||||
|
toRepair = range.newRange(remainingStart, range.end());
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
toRepair = splitter.subRange(range, offset, splitter.add(offset, length));
|
||||||
|
checkState(iteration > 1 || toRepair.start().equals(range.start()));
|
||||||
|
}
|
||||||
|
checkState(!toRepair.equals(lastRepaired), "Shouldn't repair the same range twice");
|
||||||
|
checkState(lastRepaired == null || toRepair.start().equals(lastRepaired.end()), "Next range should directly follow previous range");
|
||||||
|
lastRepaired = toRepair;
|
||||||
|
AccordService.instance().barrierWithRetries(Seekables.of(toRepair), minEpoch.getEpoch(), BarrierType.global_sync, false);
|
||||||
|
remainingStart = toRepair.end();
|
||||||
|
}
|
||||||
|
catch (RuntimeException e)
|
||||||
|
{
|
||||||
|
// TODO Placeholder for dependency limit overflow
|
||||||
|
// dependencyOverflow = true;
|
||||||
|
cfs.metric.rangeMigrationDependencyLimitFailures.mark();
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
catch (Throwable t)
|
||||||
|
{
|
||||||
|
// unexpected error
|
||||||
|
cfs.metric.rangeMigrationUnexpectedFailures.mark();
|
||||||
|
throw new RuntimeException(t);
|
||||||
|
}
|
||||||
|
finally
|
||||||
|
{
|
||||||
|
cfs.metric.rangeMigration.addNano(start);
|
||||||
|
}
|
||||||
|
|
||||||
|
// TODO when dependency limits are added to Accord need to test repair overflow
|
||||||
|
if (dependencyOverflow)
|
||||||
|
{
|
||||||
|
offset = offset.subtract(rangeStep);
|
||||||
|
if (rangeStep.equals(BigInteger.ONE))
|
||||||
|
throw new IllegalStateException("Unable to repair without overflowing with range step of 1");
|
||||||
|
rangeStep = BigInteger.ONE.max(rangeStep.divide(TWO));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
offset = offset.add(length);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -17,7 +17,14 @@
|
||||||
*/
|
*/
|
||||||
package org.apache.cassandra.repair;
|
package org.apache.cassandra.repair;
|
||||||
|
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collection;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.LinkedList;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Queue;
|
||||||
import java.util.concurrent.CopyOnWriteArrayList;
|
import java.util.concurrent.CopyOnWriteArrayList;
|
||||||
import java.util.concurrent.Executor;
|
import java.util.concurrent.Executor;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
@ -28,18 +35,11 @@ import javax.annotation.Nullable;
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.base.Preconditions;
|
import com.google.common.base.Preconditions;
|
||||||
import com.google.common.collect.ImmutableMap;
|
import com.google.common.collect.ImmutableMap;
|
||||||
import com.google.common.util.concurrent.*;
|
import com.google.common.util.concurrent.FutureCallback;
|
||||||
|
|
||||||
import org.apache.cassandra.schema.Schema;
|
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
|
||||||
import org.apache.cassandra.repair.state.JobState;
|
|
||||||
import org.apache.cassandra.utils.concurrent.AsyncFuture;
|
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
import org.apache.cassandra.db.ColumnFamilyStore;
|
|
||||||
import org.apache.cassandra.db.Keyspace;
|
|
||||||
import org.apache.cassandra.dht.Range;
|
import org.apache.cassandra.dht.Range;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
|
|
@ -47,9 +47,15 @@ import org.apache.cassandra.repair.asymmetric.DifferenceHolder;
|
||||||
import org.apache.cassandra.repair.asymmetric.HostDifferences;
|
import org.apache.cassandra.repair.asymmetric.HostDifferences;
|
||||||
import org.apache.cassandra.repair.asymmetric.PreferedNodeFilter;
|
import org.apache.cassandra.repair.asymmetric.PreferedNodeFilter;
|
||||||
import org.apache.cassandra.repair.asymmetric.ReduceHelper;
|
import org.apache.cassandra.repair.asymmetric.ReduceHelper;
|
||||||
|
import org.apache.cassandra.repair.state.JobState;
|
||||||
|
import org.apache.cassandra.schema.Schema;
|
||||||
import org.apache.cassandra.schema.SystemDistributedKeyspace;
|
import org.apache.cassandra.schema.SystemDistributedKeyspace;
|
||||||
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.ConsensusMigrationRepairResult;
|
||||||
import org.apache.cassandra.service.paxos.cleanup.PaxosCleanup;
|
import org.apache.cassandra.service.paxos.cleanup.PaxosCleanup;
|
||||||
import org.apache.cassandra.streaming.PreviewKind;
|
import org.apache.cassandra.streaming.PreviewKind;
|
||||||
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
import org.apache.cassandra.tracing.Tracing;
|
import org.apache.cassandra.tracing.Tracing;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
import org.apache.cassandra.utils.MerkleTrees;
|
import org.apache.cassandra.utils.MerkleTrees;
|
||||||
|
|
@ -65,9 +71,9 @@ import static org.apache.cassandra.service.paxos.Paxos.useV2;
|
||||||
/**
|
/**
|
||||||
* RepairJob runs repair on given ColumnFamily.
|
* RepairJob runs repair on given ColumnFamily.
|
||||||
*/
|
*/
|
||||||
public class RepairJob extends AsyncFuture<RepairResult> implements Runnable
|
public class CassandraRepairJob extends AbstractRepairJob
|
||||||
{
|
{
|
||||||
private static final Logger logger = LoggerFactory.getLogger(RepairJob.class);
|
private static final Logger logger = LoggerFactory.getLogger(CassandraRepairJob.class);
|
||||||
|
|
||||||
private final SharedContext ctx;
|
private final SharedContext ctx;
|
||||||
public final JobState state;
|
public final JobState state;
|
||||||
|
|
@ -87,8 +93,9 @@ public class RepairJob extends AsyncFuture<RepairResult> implements Runnable
|
||||||
* @param session RepairSession that this RepairJob belongs
|
* @param session RepairSession that this RepairJob belongs
|
||||||
* @param columnFamily name of the ColumnFamily to repair
|
* @param columnFamily name of the ColumnFamily to repair
|
||||||
*/
|
*/
|
||||||
public RepairJob(RepairSession session, String columnFamily)
|
public CassandraRepairJob(RepairSession session, String columnFamily)
|
||||||
{
|
{
|
||||||
|
super(session, columnFamily);
|
||||||
this.ctx = session.ctx;
|
this.ctx = session.ctx;
|
||||||
this.session = session;
|
this.session = session;
|
||||||
this.taskExecutor = session.taskExecutor;
|
this.taskExecutor = session.taskExecutor;
|
||||||
|
|
@ -116,17 +123,16 @@ public class RepairJob extends AsyncFuture<RepairResult> implements Runnable
|
||||||
* This sets up necessary task and runs them on given {@code taskExecutor}.
|
* This sets up necessary task and runs them on given {@code taskExecutor}.
|
||||||
* After submitting all tasks, waits until validation with replica completes.
|
* After submitting all tasks, waits until validation with replica completes.
|
||||||
*/
|
*/
|
||||||
public void run()
|
@Override
|
||||||
|
protected void runRepair()
|
||||||
{
|
{
|
||||||
state.phase.start();
|
|
||||||
Keyspace ks = Keyspace.open(desc.keyspace);
|
|
||||||
ColumnFamilyStore cfs = ks.getColumnFamilyStore(desc.columnFamily);
|
|
||||||
cfs.metric.repairsStarted.inc();
|
|
||||||
List<InetAddressAndPort> allEndpoints = new ArrayList<>(session.state.commonRange.endpoints);
|
List<InetAddressAndPort> allEndpoints = new ArrayList<>(session.state.commonRange.endpoints);
|
||||||
allEndpoints.add(ctx.broadcastAddressAndPort());
|
allEndpoints.add(ctx.broadcastAddressAndPort());
|
||||||
|
|
||||||
Future<Void> paxosRepair;
|
Future<Void> paxosRepair;
|
||||||
if (paxosRepairEnabled() && (((useV2() || isMetadataKeyspace()) && session.repairPaxos) || session.paxosOnly))
|
Epoch repairStartingEpoch = ClusterMetadata.current().epoch;
|
||||||
|
boolean doPaxosRepair = paxosRepairEnabled() && (((useV2() || isMetadataKeyspace()) && session.repairPaxos) || session.paxosOnly);
|
||||||
|
if (doPaxosRepair)
|
||||||
{
|
{
|
||||||
logger.info("{} {}.{} starting paxos repair", session.previewKind.logPrefix(session.getId()), desc.keyspace, desc.columnFamily);
|
logger.info("{} {}.{} starting paxos repair", session.previewKind.logPrefix(session.getId()), desc.keyspace, desc.columnFamily);
|
||||||
TableMetadata metadata = Schema.instance.getTableMetadata(desc.keyspace, desc.columnFamily);
|
TableMetadata metadata = Schema.instance.getTableMetadata(desc.keyspace, desc.columnFamily);
|
||||||
|
|
@ -142,10 +148,10 @@ public class RepairJob extends AsyncFuture<RepairResult> implements Runnable
|
||||||
{
|
{
|
||||||
paxosRepair.addCallback(new FutureCallback<>()
|
paxosRepair.addCallback(new FutureCallback<>()
|
||||||
{
|
{
|
||||||
public void onSuccess(Void v)
|
public void onSuccess(Void ignored)
|
||||||
{
|
{
|
||||||
logger.info("{} {}.{} paxos repair completed", session.previewKind.logPrefix(session.getId()), desc.keyspace, desc.columnFamily);
|
logger.info("{} {}.{} paxos repair completed", session.previewKind.logPrefix(session.getId()), desc.keyspace, desc.columnFamily);
|
||||||
trySuccess(new RepairResult(desc, Collections.emptyList()));
|
trySuccess(new RepairResult(desc, Collections.emptyList(), ConsensusMigrationRepairResult.fromCassandraRepair(repairStartingEpoch, false)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -211,7 +217,8 @@ public class RepairJob extends AsyncFuture<RepairResult> implements Runnable
|
||||||
SystemDistributedKeyspace.successfulRepairJob(session.getId(), desc.keyspace, desc.columnFamily);
|
SystemDistributedKeyspace.successfulRepairJob(session.getId(), desc.keyspace, desc.columnFamily);
|
||||||
}
|
}
|
||||||
cfs.metric.repairsCompleted.inc();
|
cfs.metric.repairsCompleted.inc();
|
||||||
trySuccess(new RepairResult(desc, stats));
|
logger.info("Completing repair with excludedDeadNodes {}", session.excludedDeadNodes);
|
||||||
|
trySuccess(new RepairResult(desc, stats, ConsensusMigrationRepairResult.fromCassandraRepair(repairStartingEpoch, doPaxosRepair && !session.excludedDeadNodes)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -64,7 +64,7 @@ public class IncrementalRepairTask extends AbstractRepairTask
|
||||||
|
|
||||||
CoordinatorSession coordinatorSession = coordinator.ctx.repair().consistent.coordinated.registerSession(parentSession, allParticipants, neighborsAndRanges.shouldExcludeDeadParticipants);
|
CoordinatorSession coordinatorSession = coordinator.ctx.repair().consistent.coordinated.registerSession(parentSession, allParticipants, neighborsAndRanges.shouldExcludeDeadParticipants);
|
||||||
|
|
||||||
return coordinatorSession.execute(() -> runRepair(parentSession, true, executor, validationScheduler, allRanges, cfnames));
|
return coordinatorSession.execute(() -> runRepair(parentSession, true, executor, validationScheduler, allRanges, neighborsAndRanges.shouldExcludeDeadParticipants, cfnames));
|
||||||
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -27,16 +27,19 @@ public class NormalRepairTask extends AbstractRepairTask
|
||||||
{
|
{
|
||||||
private final TimeUUID parentSession;
|
private final TimeUUID parentSession;
|
||||||
private final List<CommonRange> commonRanges;
|
private final List<CommonRange> commonRanges;
|
||||||
|
private final boolean excludedDeadNodes;
|
||||||
private final String[] cfnames;
|
private final String[] cfnames;
|
||||||
|
|
||||||
protected NormalRepairTask(RepairCoordinator coordinator,
|
protected NormalRepairTask(RepairCoordinator coordinator,
|
||||||
TimeUUID parentSession,
|
TimeUUID parentSession,
|
||||||
List<CommonRange> commonRanges,
|
List<CommonRange> commonRanges,
|
||||||
|
boolean excludedDeadNodes,
|
||||||
String[] cfnames)
|
String[] cfnames)
|
||||||
{
|
{
|
||||||
super(coordinator);
|
super(coordinator);
|
||||||
this.parentSession = parentSession;
|
this.parentSession = parentSession;
|
||||||
this.commonRanges = commonRanges;
|
this.commonRanges = commonRanges;
|
||||||
|
this.excludedDeadNodes = excludedDeadNodes;
|
||||||
this.cfnames = cfnames;
|
this.cfnames = cfnames;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -49,6 +52,6 @@ public class NormalRepairTask extends AbstractRepairTask
|
||||||
@Override
|
@Override
|
||||||
public Future<CoordinatedRepairResult> performUnsafe(ExecutorPlus executor, Scheduler validationScheduler)
|
public Future<CoordinatedRepairResult> performUnsafe(ExecutorPlus executor, Scheduler validationScheduler)
|
||||||
{
|
{
|
||||||
return runRepair(parentSession, false, executor, validationScheduler, commonRanges, cfnames);
|
return runRepair(parentSession, false, executor, validationScheduler, commonRanges, excludedDeadNodes, cfnames);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -42,14 +42,16 @@ public class PreviewRepairTask extends AbstractRepairTask
|
||||||
{
|
{
|
||||||
private final TimeUUID parentSession;
|
private final TimeUUID parentSession;
|
||||||
private final List<CommonRange> commonRanges;
|
private final List<CommonRange> commonRanges;
|
||||||
|
private final boolean excludedDeadNodes;
|
||||||
private final String[] cfnames;
|
private final String[] cfnames;
|
||||||
private volatile String successMessage = name() + " completed successfully";
|
private volatile String successMessage = name() + " completed successfully";
|
||||||
|
|
||||||
protected PreviewRepairTask(RepairCoordinator coordinator, TimeUUID parentSession, List<CommonRange> commonRanges, String[] cfnames)
|
protected PreviewRepairTask(RepairCoordinator coordinator, TimeUUID parentSession, List<CommonRange> commonRanges, boolean excludedDeadNodes, String[] cfnames)
|
||||||
{
|
{
|
||||||
super(coordinator);
|
super(coordinator);
|
||||||
this.parentSession = parentSession;
|
this.parentSession = parentSession;
|
||||||
this.commonRanges = commonRanges;
|
this.commonRanges = commonRanges;
|
||||||
|
this.excludedDeadNodes = excludedDeadNodes;
|
||||||
this.cfnames = cfnames;
|
this.cfnames = cfnames;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -68,7 +70,7 @@ public class PreviewRepairTask extends AbstractRepairTask
|
||||||
@Override
|
@Override
|
||||||
public Future<CoordinatedRepairResult> performUnsafe(ExecutorPlus executor, Scheduler validationScheduler)
|
public Future<CoordinatedRepairResult> performUnsafe(ExecutorPlus executor, Scheduler validationScheduler)
|
||||||
{
|
{
|
||||||
Future<CoordinatedRepairResult> f = runRepair(parentSession, false, executor, validationScheduler, commonRanges, cfnames);
|
Future<CoordinatedRepairResult> f = runRepair(parentSession, false, executor, validationScheduler, commonRanges, excludedDeadNodes, cfnames);
|
||||||
return f.map(result -> {
|
return f.map(result -> {
|
||||||
if (result.hasFailed())
|
if (result.hasFailed())
|
||||||
return result;
|
return result;
|
||||||
|
|
|
||||||
|
|
@ -478,7 +478,7 @@ public class RepairCoordinator implements Runnable, ProgressEventNotifier, Repai
|
||||||
RepairTask task;
|
RepairTask task;
|
||||||
if (state.options.isPreview())
|
if (state.options.isPreview())
|
||||||
{
|
{
|
||||||
task = new PreviewRepairTask(this, state.id, neighborsAndRanges.filterCommonRanges(state.keyspace, cfnames), cfnames);
|
task = new PreviewRepairTask(this, state.id, neighborsAndRanges.filterCommonRanges(state.keyspace, cfnames), neighborsAndRanges.shouldExcludeDeadParticipants, cfnames);
|
||||||
}
|
}
|
||||||
else if (state.options.isIncremental())
|
else if (state.options.isIncremental())
|
||||||
{
|
{
|
||||||
|
|
@ -486,7 +486,7 @@ public class RepairCoordinator implements Runnable, ProgressEventNotifier, Repai
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
task = new NormalRepairTask(this, state.id, neighborsAndRanges.filterCommonRanges(state.keyspace, cfnames), cfnames);
|
task = new NormalRepairTask(this, state.id, neighborsAndRanges.filterCommonRanges(state.keyspace, cfnames), neighborsAndRanges.shouldExcludeDeadParticipants, cfnames);
|
||||||
}
|
}
|
||||||
|
|
||||||
ExecutorPlus executor = createExecutor();
|
ExecutorPlus executor = createExecutor();
|
||||||
|
|
|
||||||
|
|
@ -17,7 +17,8 @@
|
||||||
*/
|
*/
|
||||||
package org.apache.cassandra.repair;
|
package org.apache.cassandra.repair;
|
||||||
|
|
||||||
import java.util.*;
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
import java.util.function.BiFunction;
|
import java.util.function.BiFunction;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
||||||
|
|
@ -28,7 +29,14 @@ import org.apache.cassandra.db.ColumnFamilyStore;
|
||||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
import org.apache.cassandra.net.IVerbHandler;
|
import org.apache.cassandra.net.IVerbHandler;
|
||||||
import org.apache.cassandra.net.Message;
|
import org.apache.cassandra.net.Message;
|
||||||
import org.apache.cassandra.repair.messages.*;
|
import org.apache.cassandra.repair.messages.CleanupMessage;
|
||||||
|
import org.apache.cassandra.repair.messages.FailSession;
|
||||||
|
import org.apache.cassandra.repair.messages.PrepareMessage;
|
||||||
|
import org.apache.cassandra.repair.messages.RepairMessage;
|
||||||
|
import org.apache.cassandra.repair.messages.StatusRequest;
|
||||||
|
import org.apache.cassandra.repair.messages.StatusResponse;
|
||||||
|
import org.apache.cassandra.repair.messages.SyncRequest;
|
||||||
|
import org.apache.cassandra.repair.messages.ValidationRequest;
|
||||||
import org.apache.cassandra.repair.state.AbstractCompletable;
|
import org.apache.cassandra.repair.state.AbstractCompletable;
|
||||||
import org.apache.cassandra.repair.state.AbstractState;
|
import org.apache.cassandra.repair.state.AbstractState;
|
||||||
import org.apache.cassandra.repair.state.Completable;
|
import org.apache.cassandra.repair.state.Completable;
|
||||||
|
|
@ -39,6 +47,7 @@ import org.apache.cassandra.schema.TableId;
|
||||||
import org.apache.cassandra.service.ActiveRepairService;
|
import org.apache.cassandra.service.ActiveRepairService;
|
||||||
import org.apache.cassandra.service.StorageService;
|
import org.apache.cassandra.service.StorageService;
|
||||||
import org.apache.cassandra.streaming.PreviewKind;
|
import org.apache.cassandra.streaming.PreviewKind;
|
||||||
|
import org.apache.cassandra.tcm.ClusterMetadataService;
|
||||||
import org.apache.cassandra.utils.JVMStabilityInspector;
|
import org.apache.cassandra.utils.JVMStabilityInspector;
|
||||||
import org.apache.cassandra.utils.TimeUUID;
|
import org.apache.cassandra.utils.TimeUUID;
|
||||||
|
|
||||||
|
|
@ -86,6 +95,7 @@ public class RepairMessageVerbHandler implements IVerbHandler<RepairMessage>
|
||||||
|
|
||||||
public void doVerb(final Message<RepairMessage> message)
|
public void doVerb(final Message<RepairMessage> message)
|
||||||
{
|
{
|
||||||
|
ClusterMetadataService.instance().fetchLogFromCMS(message.epoch());
|
||||||
// TODO add cancel/interrupt message
|
// TODO add cancel/interrupt message
|
||||||
RepairJobDesc desc = message.payload.desc;
|
RepairJobDesc desc = message.payload.desc;
|
||||||
try
|
try
|
||||||
|
|
|
||||||
|
|
@ -19,6 +19,8 @@ package org.apache.cassandra.repair;
|
||||||
|
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.ConsensusMigrationRepairResult;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* RepairJob's result
|
* RepairJob's result
|
||||||
*/
|
*/
|
||||||
|
|
@ -26,10 +28,12 @@ public class RepairResult
|
||||||
{
|
{
|
||||||
public final RepairJobDesc desc;
|
public final RepairJobDesc desc;
|
||||||
public final List<SyncStat> stats;
|
public final List<SyncStat> stats;
|
||||||
|
public final ConsensusMigrationRepairResult consensusMigrationRepairResult;
|
||||||
|
|
||||||
public RepairResult(RepairJobDesc desc, List<SyncStat> stats)
|
public RepairResult(RepairJobDesc desc, List<SyncStat> stats, ConsensusMigrationRepairResult consensusMigrationRepairResult)
|
||||||
{
|
{
|
||||||
this.desc = desc;
|
this.desc = desc;
|
||||||
this.stats = stats;
|
this.stats = stats;
|
||||||
|
this.consensusMigrationRepairResult = consensusMigrationRepairResult;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -30,13 +30,11 @@ import java.util.concurrent.ConcurrentMap;
|
||||||
import java.util.concurrent.Executor;
|
import java.util.concurrent.Executor;
|
||||||
import java.util.concurrent.RejectedExecutionException;
|
import java.util.concurrent.RejectedExecutionException;
|
||||||
import java.util.concurrent.atomic.AtomicBoolean;
|
import java.util.concurrent.atomic.AtomicBoolean;
|
||||||
|
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.collect.Lists;
|
import com.google.common.collect.Lists;
|
||||||
import com.google.common.util.concurrent.*;
|
import com.google.common.util.concurrent.FutureCallback;
|
||||||
|
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
|
@ -48,7 +46,9 @@ import org.apache.cassandra.db.Keyspace;
|
||||||
import org.apache.cassandra.dht.Range;
|
import org.apache.cassandra.dht.Range;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.exceptions.RepairException;
|
import org.apache.cassandra.exceptions.RepairException;
|
||||||
import org.apache.cassandra.gms.*;
|
import org.apache.cassandra.gms.EndpointState;
|
||||||
|
import org.apache.cassandra.gms.IEndpointStateChangeSubscriber;
|
||||||
|
import org.apache.cassandra.gms.IFailureDetectionEventListener;
|
||||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
import org.apache.cassandra.net.Message;
|
import org.apache.cassandra.net.Message;
|
||||||
import org.apache.cassandra.repair.consistent.ConsistentSession;
|
import org.apache.cassandra.repair.consistent.ConsistentSession;
|
||||||
|
|
@ -59,6 +59,7 @@ import org.apache.cassandra.repair.messages.ValidationResponse;
|
||||||
import org.apache.cassandra.repair.state.SessionState;
|
import org.apache.cassandra.repair.state.SessionState;
|
||||||
import org.apache.cassandra.schema.SystemDistributedKeyspace;
|
import org.apache.cassandra.schema.SystemDistributedKeyspace;
|
||||||
import org.apache.cassandra.schema.TableId;
|
import org.apache.cassandra.schema.TableId;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState;
|
||||||
import org.apache.cassandra.streaming.PreviewKind;
|
import org.apache.cassandra.streaming.PreviewKind;
|
||||||
import org.apache.cassandra.tracing.Tracing;
|
import org.apache.cassandra.tracing.Tracing;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
|
|
@ -73,7 +74,7 @@ import org.apache.cassandra.utils.concurrent.AsyncFuture;
|
||||||
*
|
*
|
||||||
* A given RepairSession repairs a set of replicas for a given set of ranges on a list
|
* A given RepairSession repairs a set of replicas for a given set of ranges on a list
|
||||||
* of column families. For each of the column family to repair, RepairSession
|
* of column families. For each of the column family to repair, RepairSession
|
||||||
* creates a {@link RepairJob} that handles the repair of that CF.
|
* creates a {@link AbstractRepairJob} that handles the repair of that CF.
|
||||||
*
|
*
|
||||||
* A given RepairJob has the 3 main phases:
|
* A given RepairJob has the 3 main phases:
|
||||||
* <ol>
|
* <ol>
|
||||||
|
|
@ -122,6 +123,7 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
public final boolean repairPaxos;
|
public final boolean repairPaxos;
|
||||||
public final boolean paxosOnly;
|
public final boolean paxosOnly;
|
||||||
public final boolean dontPurgeTombstones;
|
public final boolean dontPurgeTombstones;
|
||||||
|
public final boolean excludedDeadNodes;
|
||||||
|
|
||||||
private final AtomicBoolean isFailed = new AtomicBoolean(false);
|
private final AtomicBoolean isFailed = new AtomicBoolean(false);
|
||||||
|
|
||||||
|
|
@ -135,7 +137,8 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
public final boolean optimiseStreams;
|
public final boolean optimiseStreams;
|
||||||
public final SharedContext ctx;
|
public final SharedContext ctx;
|
||||||
public final Scheduler validationScheduler;
|
public final Scheduler validationScheduler;
|
||||||
private volatile List<RepairJob> jobs = Collections.emptyList();
|
private volatile List<AbstractRepairJob> jobs = Collections.emptyList();
|
||||||
|
private final boolean accordRepair;
|
||||||
|
|
||||||
private volatile boolean terminated = false;
|
private volatile boolean terminated = false;
|
||||||
|
|
||||||
|
|
@ -143,6 +146,7 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
* Create new repair session.
|
* Create new repair session.
|
||||||
* @param parentRepairSession the parent sessions id
|
* @param parentRepairSession the parent sessions id
|
||||||
* @param commonRange ranges to repair
|
* @param commonRange ranges to repair
|
||||||
|
* @param excludedDeadNodes Was the repair started for --force and were dead nodes excluded as a result
|
||||||
* @param keyspace name of keyspace
|
* @param keyspace name of keyspace
|
||||||
* @param parallelismDegree specifies the degree of parallelism when calculating the merkle trees
|
* @param parallelismDegree specifies the degree of parallelism when calculating the merkle trees
|
||||||
* @param pullRepair true if the repair should be one way (from remote host to this host and only applicable between two hosts--see RepairOption)
|
* @param pullRepair true if the repair should be one way (from remote host to this host and only applicable between two hosts--see RepairOption)
|
||||||
|
|
@ -154,6 +158,7 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
Scheduler validationScheduler,
|
Scheduler validationScheduler,
|
||||||
TimeUUID parentRepairSession,
|
TimeUUID parentRepairSession,
|
||||||
CommonRange commonRange,
|
CommonRange commonRange,
|
||||||
|
boolean excludedDeadNodes,
|
||||||
String keyspace,
|
String keyspace,
|
||||||
RepairParallelism parallelismDegree,
|
RepairParallelism parallelismDegree,
|
||||||
boolean isIncremental,
|
boolean isIncremental,
|
||||||
|
|
@ -163,6 +168,7 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
boolean repairPaxos,
|
boolean repairPaxos,
|
||||||
boolean paxosOnly,
|
boolean paxosOnly,
|
||||||
boolean dontPurgeTombstones,
|
boolean dontPurgeTombstones,
|
||||||
|
boolean accordRepair,
|
||||||
String... cfnames)
|
String... cfnames)
|
||||||
{
|
{
|
||||||
this.ctx = ctx;
|
this.ctx = ctx;
|
||||||
|
|
@ -178,6 +184,8 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
this.optimiseStreams = optimiseStreams;
|
this.optimiseStreams = optimiseStreams;
|
||||||
this.dontPurgeTombstones = dontPurgeTombstones;
|
this.dontPurgeTombstones = dontPurgeTombstones;
|
||||||
this.taskExecutor = new SafeExecutor(createExecutor(ctx));
|
this.taskExecutor = new SafeExecutor(createExecutor(ctx));
|
||||||
|
this.accordRepair = accordRepair;
|
||||||
|
this.excludedDeadNodes = excludedDeadNodes;
|
||||||
}
|
}
|
||||||
|
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
|
|
@ -338,10 +346,14 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
|
|
||||||
// Create and submit RepairJob for each ColumnFamily
|
// Create and submit RepairJob for each ColumnFamily
|
||||||
state.phase.jobsSubmitted();
|
state.phase.jobsSubmitted();
|
||||||
List<RepairJob> jobs = new ArrayList<>(state.cfnames.length);
|
List<AbstractRepairJob> jobs = new ArrayList<>(state.cfnames.length);
|
||||||
for (String cfname : state.cfnames)
|
for (String cfname : state.cfnames)
|
||||||
{
|
{
|
||||||
RepairJob job = new RepairJob(this, cfname);
|
AbstractRepairJob job = accordRepair ?
|
||||||
|
new AccordRepairJob(this, cfname) :
|
||||||
|
new CassandraRepairJob(this, cfname);
|
||||||
|
// Repairs can drive forward progress for consensus migration so always check
|
||||||
|
job.addCallback(ConsensusTableMigrationState.completedRepairJobHandler);
|
||||||
state.register(job.state);
|
state.register(job.state);
|
||||||
executor.execute(job);
|
executor.execute(job);
|
||||||
jobs.add(job);
|
jobs.add(job);
|
||||||
|
|
@ -381,10 +393,10 @@ public class RepairSession extends AsyncFuture<RepairSessionResult> implements I
|
||||||
public synchronized void terminate(@Nullable Throwable reason)
|
public synchronized void terminate(@Nullable Throwable reason)
|
||||||
{
|
{
|
||||||
terminated = true;
|
terminated = true;
|
||||||
List<RepairJob> jobs = this.jobs;
|
List<AbstractRepairJob> jobs = this.jobs;
|
||||||
if (jobs != null)
|
if (jobs != null)
|
||||||
{
|
{
|
||||||
for (RepairJob job : jobs)
|
for (AbstractRepairJob job : jobs)
|
||||||
job.abort(reason);
|
job.abort(reason);
|
||||||
}
|
}
|
||||||
this.jobs = null;
|
this.jobs = null;
|
||||||
|
|
|
||||||
|
|
@ -25,7 +25,6 @@ import java.util.Set;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.function.BiConsumer;
|
import java.util.function.BiConsumer;
|
||||||
import java.util.function.Supplier;
|
import java.util.function.Supplier;
|
||||||
|
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
|
@ -74,7 +73,7 @@ public abstract class RepairMessage
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public void onFailure(InetAddressAndPort from, RequestFailure failureReason)
|
public void onFailure(InetAddressAndPort from, RequestFailure failure)
|
||||||
{
|
{
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
|
||||||
|
|
@ -17,7 +17,13 @@
|
||||||
*/
|
*/
|
||||||
package org.apache.cassandra.repair.messages;
|
package org.apache.cassandra.repair.messages;
|
||||||
|
|
||||||
import java.util.*;
|
import java.util.Collection;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.HashSet;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.StringTokenizer;
|
||||||
|
|
||||||
import com.google.common.base.Joiner;
|
import com.google.common.base.Joiner;
|
||||||
import com.google.common.base.Preconditions;
|
import com.google.common.base.Preconditions;
|
||||||
|
|
@ -57,6 +63,8 @@ public class RepairOption
|
||||||
public static final String NO_TOMBSTONE_PURGING = "nopurge";
|
public static final String NO_TOMBSTONE_PURGING = "nopurge";
|
||||||
|
|
||||||
|
|
||||||
|
public static final String ACCORD_REPAIR_KEY = "accordRepair";
|
||||||
|
|
||||||
// we don't want to push nodes too much for repair
|
// we don't want to push nodes too much for repair
|
||||||
public static final int MAX_JOB_THREADS = 4;
|
public static final int MAX_JOB_THREADS = 4;
|
||||||
|
|
||||||
|
|
@ -86,6 +94,7 @@ public class RepairOption
|
||||||
}
|
}
|
||||||
return ranges;
|
return ranges;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Construct RepairOptions object from given map of Strings.
|
* Construct RepairOptions object from given map of Strings.
|
||||||
* <p>
|
* <p>
|
||||||
|
|
@ -167,6 +176,12 @@ public class RepairOption
|
||||||
* ranges to the same host multiple times</td>
|
* ranges to the same host multiple times</td>
|
||||||
* <td>false</td>
|
* <td>false</td>
|
||||||
* </tr>
|
* </tr>
|
||||||
|
* <tr>
|
||||||
|
* <td>accordRepair</td>
|
||||||
|
* <td>"true" if the repair should be of Accord in flight transactions. Will ensure
|
||||||
|
* that once repair completes all Accord transactions are replicated at quorum</td>
|
||||||
|
* <td>false</td>
|
||||||
|
* </tr>
|
||||||
* </tbody>
|
* </tbody>
|
||||||
* </table>
|
* </table>
|
||||||
*
|
*
|
||||||
|
|
@ -188,11 +203,21 @@ public class RepairOption
|
||||||
boolean repairPaxos = Boolean.parseBoolean(options.get(REPAIR_PAXOS_KEY));
|
boolean repairPaxos = Boolean.parseBoolean(options.get(REPAIR_PAXOS_KEY));
|
||||||
boolean paxosOnly = Boolean.parseBoolean(options.get(PAXOS_ONLY_KEY));
|
boolean paxosOnly = Boolean.parseBoolean(options.get(PAXOS_ONLY_KEY));
|
||||||
boolean dontPurgeTombstones = Boolean.parseBoolean(options.get(NO_TOMBSTONE_PURGING));
|
boolean dontPurgeTombstones = Boolean.parseBoolean(options.get(NO_TOMBSTONE_PURGING));
|
||||||
|
boolean accordRepair = Boolean.parseBoolean(options.get(ACCORD_REPAIR_KEY));
|
||||||
|
|
||||||
if (previewKind != PreviewKind.NONE)
|
if (previewKind != PreviewKind.NONE)
|
||||||
{
|
{
|
||||||
Preconditions.checkArgument(!repairPaxos, "repairPaxos must be set to false for preview repairs");
|
Preconditions.checkArgument(!repairPaxos, "repairPaxos must be set to false for preview repairs");
|
||||||
Preconditions.checkArgument(!paxosOnly, "paxosOnly must be set to false for preview repairs");
|
Preconditions.checkArgument(!paxosOnly, "paxosOnly must be set to false for preview repairs");
|
||||||
|
Preconditions.checkArgument(!accordRepair, "accordRepair must be set to false for preview repairs");
|
||||||
|
}
|
||||||
|
|
||||||
|
if (accordRepair)
|
||||||
|
{
|
||||||
|
Preconditions.checkArgument(!paxosOnly, "paxosOnly must be set to false for Accord repairs");
|
||||||
|
Preconditions.checkArgument(previewKind == PreviewKind.NONE, "Can't perform preview repair with an Accord repair");
|
||||||
|
Preconditions.checkArgument(!force, "Accord repair only requires a quorum to work so force is not supported");
|
||||||
|
incremental = false;
|
||||||
}
|
}
|
||||||
|
|
||||||
int jobThreads = 1;
|
int jobThreads = 1;
|
||||||
|
|
@ -212,7 +237,7 @@ public class RepairOption
|
||||||
|
|
||||||
boolean asymmetricSyncing = Boolean.parseBoolean(options.get(OPTIMISE_STREAMS_KEY));
|
boolean asymmetricSyncing = Boolean.parseBoolean(options.get(OPTIMISE_STREAMS_KEY));
|
||||||
|
|
||||||
RepairOption option = new RepairOption(parallelism, primaryRange, incremental, trace, jobThreads, ranges, !ranges.isEmpty(), pullRepair, force, previewKind, asymmetricSyncing, ignoreUnreplicatedKeyspaces, repairPaxos, paxosOnly, dontPurgeTombstones);
|
RepairOption option = new RepairOption(parallelism, primaryRange, incremental, trace, jobThreads, ranges, pullRepair, force, previewKind, asymmetricSyncing, ignoreUnreplicatedKeyspaces, repairPaxos, paxosOnly, dontPurgeTombstones, accordRepair);
|
||||||
|
|
||||||
// data centers
|
// data centers
|
||||||
String dataCentersStr = options.get(DATACENTERS_KEY);
|
String dataCentersStr = options.get(DATACENTERS_KEY);
|
||||||
|
|
@ -286,7 +311,6 @@ public class RepairOption
|
||||||
private final boolean incremental;
|
private final boolean incremental;
|
||||||
private final boolean trace;
|
private final boolean trace;
|
||||||
private final int jobThreads;
|
private final int jobThreads;
|
||||||
private final boolean isSubrangeRepair;
|
|
||||||
private final boolean pullRepair;
|
private final boolean pullRepair;
|
||||||
private final boolean forceRepair;
|
private final boolean forceRepair;
|
||||||
private final PreviewKind previewKind;
|
private final PreviewKind previewKind;
|
||||||
|
|
@ -296,12 +320,17 @@ public class RepairOption
|
||||||
private final boolean paxosOnly;
|
private final boolean paxosOnly;
|
||||||
private final boolean dontPurgeTombstones;
|
private final boolean dontPurgeTombstones;
|
||||||
|
|
||||||
|
private final boolean accordRepair;
|
||||||
|
|
||||||
private final Collection<String> columnFamilies = new HashSet<>();
|
private final Collection<String> columnFamilies = new HashSet<>();
|
||||||
private final Collection<String> dataCenters = new HashSet<>();
|
private final Collection<String> dataCenters = new HashSet<>();
|
||||||
private final Collection<String> hosts = new HashSet<>();
|
private final Collection<String> hosts = new HashSet<>();
|
||||||
private final Collection<Range<Token>> ranges = new HashSet<>();
|
private final Collection<Range<Token>> ranges = new HashSet<>();
|
||||||
|
|
||||||
public RepairOption(RepairParallelism parallelism, boolean primaryRange, boolean incremental, boolean trace, int jobThreads, Collection<Range<Token>> ranges, boolean isSubrangeRepair, boolean pullRepair, boolean forceRepair, PreviewKind previewKind, boolean optimiseStreams, boolean ignoreUnreplicatedKeyspaces, boolean repairPaxos, boolean paxosOnly, boolean dontPurgeTombstones)
|
public RepairOption(RepairParallelism parallelism, boolean primaryRange, boolean incremental, boolean trace, int jobThreads,
|
||||||
|
Collection<Range<Token>> ranges, boolean pullRepair, boolean forceRepair,
|
||||||
|
PreviewKind previewKind, boolean optimiseStreams, boolean ignoreUnreplicatedKeyspaces, boolean repairPaxos,
|
||||||
|
boolean paxosOnly, boolean dontPurgeTombstones, boolean accordRepair)
|
||||||
{
|
{
|
||||||
|
|
||||||
this.parallelism = parallelism;
|
this.parallelism = parallelism;
|
||||||
|
|
@ -310,7 +339,6 @@ public class RepairOption
|
||||||
this.trace = trace;
|
this.trace = trace;
|
||||||
this.jobThreads = jobThreads;
|
this.jobThreads = jobThreads;
|
||||||
this.ranges.addAll(ranges);
|
this.ranges.addAll(ranges);
|
||||||
this.isSubrangeRepair = isSubrangeRepair;
|
|
||||||
this.pullRepair = pullRepair;
|
this.pullRepair = pullRepair;
|
||||||
this.forceRepair = forceRepair;
|
this.forceRepair = forceRepair;
|
||||||
this.previewKind = previewKind;
|
this.previewKind = previewKind;
|
||||||
|
|
@ -319,6 +347,7 @@ public class RepairOption
|
||||||
this.repairPaxos = repairPaxos;
|
this.repairPaxos = repairPaxos;
|
||||||
this.paxosOnly = paxosOnly;
|
this.paxosOnly = paxosOnly;
|
||||||
this.dontPurgeTombstones = dontPurgeTombstones;
|
this.dontPurgeTombstones = dontPurgeTombstones;
|
||||||
|
this.accordRepair = accordRepair;
|
||||||
}
|
}
|
||||||
|
|
||||||
public RepairParallelism getParallelism()
|
public RepairParallelism getParallelism()
|
||||||
|
|
@ -381,11 +410,6 @@ public class RepairOption
|
||||||
return dataCenters.isEmpty() && hosts.isEmpty();
|
return dataCenters.isEmpty() && hosts.isEmpty();
|
||||||
}
|
}
|
||||||
|
|
||||||
public boolean isSubrangeRepair()
|
|
||||||
{
|
|
||||||
return isSubrangeRepair;
|
|
||||||
}
|
|
||||||
|
|
||||||
public PreviewKind getPreviewKind()
|
public PreviewKind getPreviewKind()
|
||||||
{
|
{
|
||||||
return previewKind;
|
return previewKind;
|
||||||
|
|
@ -439,6 +463,11 @@ public class RepairOption
|
||||||
return dontPurgeTombstones;
|
return dontPurgeTombstones;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public boolean accordRepair()
|
||||||
|
{
|
||||||
|
return accordRepair;
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public String toString()
|
public String toString()
|
||||||
{
|
{
|
||||||
|
|
@ -459,6 +488,7 @@ public class RepairOption
|
||||||
", repairPaxos: " + repairPaxos +
|
", repairPaxos: " + repairPaxos +
|
||||||
", paxosOnly: " + paxosOnly +
|
", paxosOnly: " + paxosOnly +
|
||||||
", dontPurgeTombstones: " + dontPurgeTombstones +
|
", dontPurgeTombstones: " + dontPurgeTombstones +
|
||||||
|
", accordRepair: " + accordRepair +
|
||||||
')';
|
')';
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -472,7 +502,6 @@ public class RepairOption
|
||||||
options.put(COLUMNFAMILIES_KEY, Joiner.on(",").join(columnFamilies));
|
options.put(COLUMNFAMILIES_KEY, Joiner.on(",").join(columnFamilies));
|
||||||
options.put(DATACENTERS_KEY, Joiner.on(",").join(dataCenters));
|
options.put(DATACENTERS_KEY, Joiner.on(",").join(dataCenters));
|
||||||
options.put(HOSTS_KEY, Joiner.on(",").join(hosts));
|
options.put(HOSTS_KEY, Joiner.on(",").join(hosts));
|
||||||
options.put(SUB_RANGE_REPAIR_KEY, Boolean.toString(isSubrangeRepair));
|
|
||||||
options.put(TRACE_KEY, Boolean.toString(trace));
|
options.put(TRACE_KEY, Boolean.toString(trace));
|
||||||
options.put(RANGES_KEY, Joiner.on(",").join(ranges));
|
options.put(RANGES_KEY, Joiner.on(",").join(ranges));
|
||||||
options.put(PULL_REPAIR_KEY, Boolean.toString(pullRepair));
|
options.put(PULL_REPAIR_KEY, Boolean.toString(pullRepair));
|
||||||
|
|
@ -482,6 +511,7 @@ public class RepairOption
|
||||||
options.put(REPAIR_PAXOS_KEY, Boolean.toString(repairPaxos));
|
options.put(REPAIR_PAXOS_KEY, Boolean.toString(repairPaxos));
|
||||||
options.put(PAXOS_ONLY_KEY, Boolean.toString(paxosOnly));
|
options.put(PAXOS_ONLY_KEY, Boolean.toString(paxosOnly));
|
||||||
options.put(NO_TOMBSTONE_PURGING, Boolean.toString(dontPurgeTombstones));
|
options.put(NO_TOMBSTONE_PURGING, Boolean.toString(dontPurgeTombstones));
|
||||||
|
options.put(ACCORD_REPAIR_KEY, Boolean.toString(accordRepair));
|
||||||
return options;
|
return options;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -23,12 +23,13 @@ import java.util.List;
|
||||||
import java.util.Objects;
|
import java.util.Objects;
|
||||||
|
|
||||||
import org.apache.cassandra.db.TypeSizes;
|
import org.apache.cassandra.db.TypeSizes;
|
||||||
|
import org.apache.cassandra.dht.IPartitioner;
|
||||||
import org.apache.cassandra.io.IVersionedSerializer;
|
import org.apache.cassandra.io.IVersionedSerializer;
|
||||||
import org.apache.cassandra.io.util.DataInputPlus;
|
import org.apache.cassandra.io.util.DataInputPlus;
|
||||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
import org.apache.cassandra.repair.SyncNodePair;
|
|
||||||
import org.apache.cassandra.repair.RepairJobDesc;
|
import org.apache.cassandra.repair.RepairJobDesc;
|
||||||
|
import org.apache.cassandra.repair.SyncNodePair;
|
||||||
import org.apache.cassandra.streaming.SessionSummary;
|
import org.apache.cassandra.streaming.SessionSummary;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -103,7 +104,7 @@ public class SyncResponse extends RepairMessage
|
||||||
List<SessionSummary> summaries = new ArrayList<>(numSummaries);
|
List<SessionSummary> summaries = new ArrayList<>(numSummaries);
|
||||||
for (int i=0; i<numSummaries; i++)
|
for (int i=0; i<numSummaries; i++)
|
||||||
{
|
{
|
||||||
summaries.add(SessionSummary.serializer.deserialize(in, version));
|
summaries.add(SessionSummary.serializer.deserialize(in, IPartitioner.global(), version));
|
||||||
}
|
}
|
||||||
|
|
||||||
return new SyncResponse(desc, nodes, success, summaries);
|
return new SyncResponse(desc, nodes, success, summaries);
|
||||||
|
|
|
||||||
|
|
@ -32,11 +32,10 @@ import java.util.concurrent.TimeUnit;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.base.Joiner;
|
import com.google.common.base.Joiner;
|
||||||
|
import com.google.common.collect.ImmutableMap;
|
||||||
import com.google.common.collect.ImmutableSet;
|
import com.google.common.collect.ImmutableSet;
|
||||||
import com.google.common.collect.Lists;
|
import com.google.common.collect.Lists;
|
||||||
import com.google.common.collect.ImmutableMap;
|
|
||||||
import com.google.common.collect.Sets;
|
import com.google.common.collect.Sets;
|
||||||
|
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
|
@ -57,7 +56,6 @@ import org.apache.cassandra.utils.FBUtilities;
|
||||||
import org.apache.cassandra.utils.TimeUUID;
|
import org.apache.cassandra.utils.TimeUUID;
|
||||||
|
|
||||||
import static java.lang.String.format;
|
import static java.lang.String.format;
|
||||||
|
|
||||||
import static org.apache.cassandra.utils.ByteBufferUtil.bytes;
|
import static org.apache.cassandra.utils.ByteBufferUtil.bytes;
|
||||||
|
|
||||||
public final class SystemDistributedKeyspace
|
public final class SystemDistributedKeyspace
|
||||||
|
|
|
||||||
|
|
@ -30,6 +30,11 @@ import org.apache.commons.lang3.ArrayUtils;
|
||||||
|
|
||||||
import org.apache.cassandra.db.TypeSizes;
|
import org.apache.cassandra.db.TypeSizes;
|
||||||
import org.apache.cassandra.db.marshal.ValueAccessor;
|
import org.apache.cassandra.db.marshal.ValueAccessor;
|
||||||
|
import org.apache.cassandra.io.IVersionedSerializer;
|
||||||
|
import org.apache.cassandra.io.util.DataInputPlus;
|
||||||
|
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||||
|
import org.apache.cassandra.tcm.serialization.MetadataSerializer;
|
||||||
|
import org.apache.cassandra.tcm.serialization.Version;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
import org.apache.cassandra.utils.Pair;
|
import org.apache.cassandra.utils.Pair;
|
||||||
|
|
||||||
|
|
@ -193,4 +198,46 @@ public class TableId implements Comparable<TableId>
|
||||||
{
|
{
|
||||||
return id.compareTo(o.id);
|
return id.compareTo(o.id);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static final IVersionedSerializer<TableId> serializer = new IVersionedSerializer<TableId>()
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
public void serialize(TableId t, DataOutputPlus out, int version) throws IOException
|
||||||
|
{
|
||||||
|
t.serialize(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public TableId deserialize(DataInputPlus in, int version) throws IOException
|
||||||
|
{
|
||||||
|
return TableId.deserialize(in);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long serializedSize(TableId t, int version)
|
||||||
|
{
|
||||||
|
return t.serializedSize();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
public static final MetadataSerializer<TableId> metadataSerializer = new MetadataSerializer<TableId>()
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
public void serialize(TableId t, DataOutputPlus out, Version version) throws IOException
|
||||||
|
{
|
||||||
|
t.serialize(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public TableId deserialize(DataInputPlus in, Version version) throws IOException
|
||||||
|
{
|
||||||
|
return TableId.deserialize(in);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long serializedSize(TableId t, Version version)
|
||||||
|
{
|
||||||
|
return t.serializedSize();
|
||||||
|
}
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -70,10 +70,10 @@ import org.apache.cassandra.exceptions.ConfigurationException;
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
import org.apache.cassandra.io.util.DataInputPlus;
|
import org.apache.cassandra.io.util.DataInputPlus;
|
||||||
import org.apache.cassandra.io.util.DataOutputPlus;
|
import org.apache.cassandra.io.util.DataOutputPlus;
|
||||||
|
import org.apache.cassandra.service.reads.SpeculativeRetryPolicy;
|
||||||
import org.apache.cassandra.tcm.Epoch;
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
import org.apache.cassandra.tcm.serialization.UDTAndFunctionsAwareMetadataSerializer;
|
import org.apache.cassandra.tcm.serialization.UDTAndFunctionsAwareMetadataSerializer;
|
||||||
import org.apache.cassandra.tcm.serialization.Version;
|
import org.apache.cassandra.tcm.serialization.Version;
|
||||||
import org.apache.cassandra.service.reads.SpeculativeRetryPolicy;
|
|
||||||
import org.apache.cassandra.utils.AbstractIterator;
|
import org.apache.cassandra.utils.AbstractIterator;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
import org.apache.cassandra.utils.FBUtilities;
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
|
|
@ -320,6 +320,11 @@ public class TableMetadata implements SchemaElement
|
||||||
return unbuild().indexes(indexes).build();
|
return unbuild().indexes(indexes).build();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public TableId id()
|
||||||
|
{
|
||||||
|
return id;
|
||||||
|
}
|
||||||
|
|
||||||
public boolean isView()
|
public boolean isView()
|
||||||
{
|
{
|
||||||
return kind == Kind.VIEW;
|
return kind == Kind.VIEW;
|
||||||
|
|
@ -344,7 +349,7 @@ public class TableMetadata implements SchemaElement
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
public boolean isIncrementalBackupsEnabled()
|
public boolean isIncrementalBackupsEnabled()
|
||||||
{
|
{
|
||||||
return params.incrementalBackups;
|
return params.incrementalBackups;
|
||||||
|
|
|
||||||
|
|
@ -448,6 +448,7 @@ public class ActiveRepairService implements IEndpointStateChangeSubscriber, IFai
|
||||||
*/
|
*/
|
||||||
public RepairSession submitRepairSession(TimeUUID parentRepairSession,
|
public RepairSession submitRepairSession(TimeUUID parentRepairSession,
|
||||||
CommonRange range,
|
CommonRange range,
|
||||||
|
boolean excludedDeadNodes,
|
||||||
String keyspace,
|
String keyspace,
|
||||||
RepairParallelism parallelismDegree,
|
RepairParallelism parallelismDegree,
|
||||||
boolean isIncremental,
|
boolean isIncremental,
|
||||||
|
|
@ -457,6 +458,7 @@ public class ActiveRepairService implements IEndpointStateChangeSubscriber, IFai
|
||||||
boolean repairPaxos,
|
boolean repairPaxos,
|
||||||
boolean paxosOnly,
|
boolean paxosOnly,
|
||||||
boolean dontPurgeTombstones,
|
boolean dontPurgeTombstones,
|
||||||
|
boolean accordRepair,
|
||||||
ExecutorPlus executor,
|
ExecutorPlus executor,
|
||||||
Scheduler validationScheduler,
|
Scheduler validationScheduler,
|
||||||
String... cfnames)
|
String... cfnames)
|
||||||
|
|
@ -470,9 +472,11 @@ public class ActiveRepairService implements IEndpointStateChangeSubscriber, IFai
|
||||||
if (cfnames.length == 0)
|
if (cfnames.length == 0)
|
||||||
return null;
|
return null;
|
||||||
|
|
||||||
final RepairSession session = new RepairSession(ctx, validationScheduler, parentRepairSession, range, keyspace,
|
final RepairSession session = new RepairSession(ctx, validationScheduler, parentRepairSession,
|
||||||
|
range, excludedDeadNodes, keyspace,
|
||||||
parallelismDegree, isIncremental, pullRepair,
|
parallelismDegree, isIncremental, pullRepair,
|
||||||
previewKind, optimiseStreams, repairPaxos, paxosOnly, dontPurgeTombstones, cfnames);
|
previewKind, optimiseStreams, repairPaxos, paxosOnly,
|
||||||
|
dontPurgeTombstones, accordRepair, cfnames);
|
||||||
repairs.getIfPresent(parentRepairSession).register(session.state);
|
repairs.getIfPresent(parentRepairSession).register(session.state);
|
||||||
|
|
||||||
sessions.put(session.getId(), session);
|
sessions.put(session.getId(), session);
|
||||||
|
|
|
||||||
|
|
@ -18,15 +18,17 @@
|
||||||
package org.apache.cassandra.service;
|
package org.apache.cassandra.service;
|
||||||
|
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||||
import org.apache.cassandra.db.partitions.FilteredPartition;
|
import org.apache.cassandra.db.partitions.FilteredPartition;
|
||||||
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||||
import org.apache.cassandra.db.rows.RowIterator;
|
|
||||||
import org.apache.cassandra.exceptions.InvalidRequestException;
|
import org.apache.cassandra.exceptions.InvalidRequestException;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnData;
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
import org.apache.cassandra.service.paxos.Ballot;
|
import org.apache.cassandra.service.paxos.Ballot;
|
||||||
import org.apache.cassandra.transport.Dispatcher;
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
|
|
||||||
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Abstract the conditions and updates for a CAS operation.
|
* Abstract the conditions and updates for a CAS operation.
|
||||||
*/
|
*/
|
||||||
|
|
@ -51,7 +53,7 @@ public interface CASRequest
|
||||||
*/
|
*/
|
||||||
PartitionUpdate makeUpdates(FilteredPartition current, ClientState clientState, Ballot ballot) throws InvalidRequestException;
|
PartitionUpdate makeUpdates(FilteredPartition current, ClientState clientState, Ballot ballot) throws InvalidRequestException;
|
||||||
|
|
||||||
Txn toAccordTxn(ClientState clientState, long nowInSecs);
|
Txn toAccordTxn(ConsistencyLevel consistencyLevel, ConsistencyLevel commitConsistencyLevel, ClientState clientState, long nowInSecs);
|
||||||
|
|
||||||
RowIterator toCasResult(TxnData data);
|
ConsensusAttemptResult toCasResult(TxnResult txnResult);
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -39,14 +39,16 @@ import java.util.concurrent.atomic.AtomicInteger;
|
||||||
import java.util.concurrent.atomic.AtomicLong;
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
import com.google.common.base.Preconditions;
|
|
||||||
import com.google.common.cache.CacheLoader;
|
import com.google.common.cache.CacheLoader;
|
||||||
import com.google.common.collect.Iterables;
|
import com.google.common.collect.Iterables;
|
||||||
import com.google.common.util.concurrent.Uninterruptibles;
|
import com.google.common.util.concurrent.Uninterruptibles;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import accord.primitives.Keys;
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
import org.apache.cassandra.batchlog.Batch;
|
import org.apache.cassandra.batchlog.Batch;
|
||||||
import org.apache.cassandra.batchlog.BatchlogManager;
|
import org.apache.cassandra.batchlog.BatchlogManager;
|
||||||
|
|
@ -54,6 +56,7 @@ import org.apache.cassandra.concurrent.DebuggableTask.RunnableDebuggableTask;
|
||||||
import org.apache.cassandra.concurrent.Stage;
|
import org.apache.cassandra.concurrent.Stage;
|
||||||
import org.apache.cassandra.config.CassandraRelevantProperties;
|
import org.apache.cassandra.config.CassandraRelevantProperties;
|
||||||
import org.apache.cassandra.config.Config;
|
import org.apache.cassandra.config.Config;
|
||||||
|
import org.apache.cassandra.config.Config.NonSerialWriteStrategy;
|
||||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
import org.apache.cassandra.db.ColumnFamilyStore;
|
import org.apache.cassandra.db.ColumnFamilyStore;
|
||||||
import org.apache.cassandra.db.ConsistencyLevel;
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
|
|
@ -90,8 +93,8 @@ import org.apache.cassandra.exceptions.QueryCancelledException;
|
||||||
import org.apache.cassandra.exceptions.ReadAbortException;
|
import org.apache.cassandra.exceptions.ReadAbortException;
|
||||||
import org.apache.cassandra.exceptions.ReadFailureException;
|
import org.apache.cassandra.exceptions.ReadFailureException;
|
||||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||||
import org.apache.cassandra.exceptions.RequestFailureException;
|
|
||||||
import org.apache.cassandra.exceptions.RequestFailure;
|
import org.apache.cassandra.exceptions.RequestFailure;
|
||||||
|
import org.apache.cassandra.exceptions.RequestFailureException;
|
||||||
import org.apache.cassandra.exceptions.RequestTimeoutException;
|
import org.apache.cassandra.exceptions.RequestTimeoutException;
|
||||||
import org.apache.cassandra.exceptions.UnavailableException;
|
import org.apache.cassandra.exceptions.UnavailableException;
|
||||||
import org.apache.cassandra.exceptions.WriteFailureException;
|
import org.apache.cassandra.exceptions.WriteFailureException;
|
||||||
|
|
@ -125,9 +128,18 @@ import org.apache.cassandra.schema.SchemaConstants;
|
||||||
import org.apache.cassandra.schema.TableId;
|
import org.apache.cassandra.schema.TableId;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.service.accord.AccordService;
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
|
import org.apache.cassandra.service.accord.IAccordService;
|
||||||
|
import org.apache.cassandra.service.accord.api.PartitionKey;
|
||||||
|
import org.apache.cassandra.service.accord.txn.AccordUpdate;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnCondition;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnData;
|
import org.apache.cassandra.service.accord.txn.TxnData;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnRead;
|
import org.apache.cassandra.service.accord.txn.TxnRead;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnReferenceOperations;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnUpdate;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter;
|
||||||
import org.apache.cassandra.service.paxos.Ballot;
|
import org.apache.cassandra.service.paxos.Ballot;
|
||||||
import org.apache.cassandra.service.paxos.Commit;
|
import org.apache.cassandra.service.paxos.Commit;
|
||||||
import org.apache.cassandra.service.paxos.ContentionStrategy;
|
import org.apache.cassandra.service.paxos.ContentionStrategy;
|
||||||
|
|
@ -137,6 +149,7 @@ import org.apache.cassandra.service.paxos.v1.PrepareCallback;
|
||||||
import org.apache.cassandra.service.paxos.v1.ProposeCallback;
|
import org.apache.cassandra.service.paxos.v1.ProposeCallback;
|
||||||
import org.apache.cassandra.service.reads.AbstractReadExecutor;
|
import org.apache.cassandra.service.reads.AbstractReadExecutor;
|
||||||
import org.apache.cassandra.service.reads.ReadCallback;
|
import org.apache.cassandra.service.reads.ReadCallback;
|
||||||
|
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||||
import org.apache.cassandra.service.reads.range.RangeCommands;
|
import org.apache.cassandra.service.reads.range.RangeCommands;
|
||||||
import org.apache.cassandra.service.reads.repair.ReadRepair;
|
import org.apache.cassandra.service.reads.repair.ReadRepair;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
|
|
@ -155,10 +168,10 @@ import org.apache.cassandra.utils.TimeUUID;
|
||||||
import org.apache.cassandra.utils.concurrent.CountDownLatch;
|
import org.apache.cassandra.utils.concurrent.CountDownLatch;
|
||||||
import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException;
|
import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException;
|
||||||
|
|
||||||
|
import static com.google.common.base.Preconditions.checkNotNull;
|
||||||
import static com.google.common.collect.Iterables.concat;
|
import static com.google.common.collect.Iterables.concat;
|
||||||
import static java.util.concurrent.TimeUnit.MILLISECONDS;
|
import static java.util.concurrent.TimeUnit.MILLISECONDS;
|
||||||
import static java.util.concurrent.TimeUnit.NANOSECONDS;
|
import static java.util.concurrent.TimeUnit.NANOSECONDS;
|
||||||
import static org.apache.cassandra.config.Config.LegacyPaxosStrategy.accord;
|
|
||||||
import static org.apache.cassandra.db.ConsistencyLevel.SERIAL;
|
import static org.apache.cassandra.db.ConsistencyLevel.SERIAL;
|
||||||
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.casReadMetrics;
|
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.casReadMetrics;
|
||||||
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.casWriteMetrics;
|
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.casWriteMetrics;
|
||||||
|
|
@ -177,6 +190,11 @@ import static org.apache.cassandra.net.Verb.PAXOS_PROPOSE_REQ;
|
||||||
import static org.apache.cassandra.net.Verb.SCHEMA_VERSION_REQ;
|
import static org.apache.cassandra.net.Verb.SCHEMA_VERSION_REQ;
|
||||||
import static org.apache.cassandra.net.Verb.TRUNCATE_REQ;
|
import static org.apache.cassandra.net.Verb.TRUNCATE_REQ;
|
||||||
import static org.apache.cassandra.service.BatchlogResponseHandler.BatchlogCleanup;
|
import static org.apache.cassandra.service.BatchlogResponseHandler.BatchlogCleanup;
|
||||||
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.RETRY_NEW_PROTOCOL;
|
||||||
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.casResult;
|
||||||
|
import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.serialReadResult;
|
||||||
|
import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol;
|
||||||
|
import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.ConsensusRoutingDecision;
|
||||||
import static org.apache.cassandra.service.paxos.Ballot.Flag.GLOBAL;
|
import static org.apache.cassandra.service.paxos.Ballot.Flag.GLOBAL;
|
||||||
import static org.apache.cassandra.service.paxos.Ballot.Flag.LOCAL;
|
import static org.apache.cassandra.service.paxos.Ballot.Flag.LOCAL;
|
||||||
import static org.apache.cassandra.service.paxos.BallotGenerator.Global.nextBallot;
|
import static org.apache.cassandra.service.paxos.BallotGenerator.Global.nextBallot;
|
||||||
|
|
@ -322,6 +340,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
Dispatcher.RequestTime requestTime)
|
Dispatcher.RequestTime requestTime)
|
||||||
throws UnavailableException, IsBootstrappingException, RequestFailureException, RequestTimeoutException, InvalidRequestException, CasWriteUnknownResultException
|
throws UnavailableException, IsBootstrappingException, RequestFailureException, RequestTimeoutException, InvalidRequestException, CasWriteUnknownResultException
|
||||||
{
|
{
|
||||||
|
TableMetadata metadata = Schema.instance.validateTable(keyspaceName, cfName);
|
||||||
if (DatabaseDescriptor.getPartitionDenylistEnabled() && DatabaseDescriptor.getDenylistWritesEnabled() && !partitionDenylist.isKeyPermitted(keyspaceName, cfName, key.getKey()))
|
if (DatabaseDescriptor.getPartitionDenylistEnabled() && DatabaseDescriptor.getDenylistWritesEnabled() && !partitionDenylist.isKeyPermitted(keyspaceName, cfName, key.getKey()))
|
||||||
{
|
{
|
||||||
denylistMetrics.incrementWritesRejected();
|
denylistMetrics.incrementWritesRejected();
|
||||||
|
|
@ -329,34 +348,61 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
key, keyspaceName, cfName));
|
key, keyspaceName, cfName));
|
||||||
}
|
}
|
||||||
|
|
||||||
if (DatabaseDescriptor.getLegacyPaxosStrategy() == accord)
|
ConsensusAttemptResult lastAttemptResult;
|
||||||
|
do
|
||||||
{
|
{
|
||||||
TxnData data = AccordService.instance().coordinate(request.toAccordTxn(clientState, nowInSeconds), consistencyForPaxos);
|
ConsensusRoutingDecision decision = consensusRouting(metadata, key, consistencyForPaxos, requestTime, true);
|
||||||
return request.toCasResult(data);
|
switch (decision)
|
||||||
}
|
{
|
||||||
else
|
case paxosV2:
|
||||||
{
|
lastAttemptResult = Paxos.cas(key,
|
||||||
return (Paxos.useV2() || keyspaceName.equals(SchemaConstants.METADATA_KEYSPACE_NAME))
|
request,
|
||||||
? Paxos.cas(key, request, consistencyForPaxos, consistencyForCommit, clientState)
|
consistencyForPaxos,
|
||||||
: legacyCas(keyspaceName, cfName, key, request, consistencyForPaxos, consistencyForCommit, clientState, nowInSeconds, requestTime);
|
consistencyForCommit,
|
||||||
}
|
clientState,
|
||||||
|
requestTime);
|
||||||
|
break;
|
||||||
|
case paxosV1:
|
||||||
|
lastAttemptResult = legacyCas(metadata,
|
||||||
|
key,
|
||||||
|
request,
|
||||||
|
consistencyForPaxos,
|
||||||
|
consistencyForCommit,
|
||||||
|
clientState,
|
||||||
|
nowInSeconds,
|
||||||
|
requestTime);
|
||||||
|
break;
|
||||||
|
case accord:
|
||||||
|
Txn txn = request.toAccordTxn(consistencyForPaxos,
|
||||||
|
consistencyForCommit,
|
||||||
|
clientState,
|
||||||
|
nowInSeconds);
|
||||||
|
IAccordService accordService = AccordService.instance();
|
||||||
|
accordService.maybeConvertKeyspacesToAccord(txn);
|
||||||
|
TxnResult txnResult = accordService.coordinate(txn,
|
||||||
|
consistencyForPaxos,
|
||||||
|
requestTime);
|
||||||
|
lastAttemptResult = request.toCasResult(txnResult);
|
||||||
|
break;
|
||||||
|
default:
|
||||||
|
throw new IllegalStateException("Unsupported consensus " + decision);
|
||||||
|
}
|
||||||
|
} while (lastAttemptResult.shouldRetryOnNewConsensusProtocol);
|
||||||
|
return lastAttemptResult.casResult;
|
||||||
}
|
}
|
||||||
|
|
||||||
public static RowIterator legacyCas(String keyspaceName,
|
private static ConsensusAttemptResult legacyCas(TableMetadata metadata,
|
||||||
String cfName,
|
DecoratedKey key,
|
||||||
DecoratedKey key,
|
CASRequest request,
|
||||||
CASRequest request,
|
ConsistencyLevel consistencyForPaxos,
|
||||||
ConsistencyLevel consistencyForPaxos,
|
ConsistencyLevel consistencyForCommit,
|
||||||
ConsistencyLevel consistencyForCommit,
|
ClientState clientState,
|
||||||
ClientState clientState,
|
long nowInSeconds,
|
||||||
long nowInSeconds,
|
Dispatcher.RequestTime requestTime)
|
||||||
Dispatcher.RequestTime requestTime)
|
|
||||||
throws UnavailableException, IsBootstrappingException, RequestFailureException, RequestTimeoutException, InvalidRequestException
|
throws UnavailableException, IsBootstrappingException, RequestFailureException, RequestTimeoutException, InvalidRequestException
|
||||||
{
|
{
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
TableMetadata metadata = Schema.instance.validateTable(keyspaceName, cfName);
|
|
||||||
|
|
||||||
Function<Ballot, Pair<PartitionUpdate, RowIterator>> updateProposer = ballot ->
|
Function<Ballot, Pair<PartitionUpdate, RowIterator>> updateProposer = ballot ->
|
||||||
{
|
{
|
||||||
// read the current values and check they validate the conditions
|
// read the current values and check they validate the conditions
|
||||||
|
|
@ -374,7 +420,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
{
|
{
|
||||||
Tracing.trace("CAS precondition does not match current values {}", current);
|
Tracing.trace("CAS precondition does not match current values {}", current);
|
||||||
casWriteMetrics.conditionNotMet.inc();
|
casWriteMetrics.conditionNotMet.inc();
|
||||||
return Pair.create(PartitionUpdate.emptyUpdate(metadata, key), current.rowIterator());
|
return Pair.create(PartitionUpdate.emptyUpdate(metadata, key), current.rowIterator(false));
|
||||||
}
|
}
|
||||||
|
|
||||||
// Create the desired updates
|
// Create the desired updates
|
||||||
|
|
@ -399,15 +445,14 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
return Pair.create(updates, null);
|
return Pair.create(updates, null);
|
||||||
};
|
};
|
||||||
|
|
||||||
return doPaxos(metadata,
|
return casResult(doPaxos(metadata,
|
||||||
key,
|
key,
|
||||||
consistencyForPaxos,
|
consistencyForPaxos,
|
||||||
consistencyForCommit,
|
consistencyForCommit,
|
||||||
consistencyForCommit,
|
consistencyForCommit,
|
||||||
requestTime,
|
requestTime,
|
||||||
casWriteMetrics,
|
casWriteMetrics,
|
||||||
updateProposer);
|
updateProposer));
|
||||||
|
|
||||||
}
|
}
|
||||||
catch (CasWriteUnknownResultException e)
|
catch (CasWriteUnknownResultException e)
|
||||||
{
|
{
|
||||||
|
|
@ -1165,6 +1210,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
|
|
||||||
Collection<Mutation> augmented = TriggerExecutor.instance.execute(mutations);
|
Collection<Mutation> augmented = TriggerExecutor.instance.execute(mutations);
|
||||||
|
|
||||||
|
String keyspaceName = mutations.iterator().next().getKeyspaceName();
|
||||||
boolean updatesView = Keyspace.open(mutations.iterator().next().getKeyspaceName())
|
boolean updatesView = Keyspace.open(mutations.iterator().next().getKeyspaceName())
|
||||||
.viewManager
|
.viewManager
|
||||||
.updatesAffectView(mutations, true);
|
.updatesAffectView(mutations, true);
|
||||||
|
|
@ -1172,8 +1218,10 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
long size = IMutation.dataSize(mutations);
|
long size = IMutation.dataSize(mutations);
|
||||||
writeMetrics.mutationSize.update(size);
|
writeMetrics.mutationSize.update(size);
|
||||||
writeMetricsForLevel(consistencyLevel).mutationSize.update(size);
|
writeMetricsForLevel(consistencyLevel).mutationSize.update(size);
|
||||||
|
NonSerialWriteStrategy nonSerialWriteStrategy = DatabaseDescriptor.getNonSerialWriteStrategy();
|
||||||
if (augmented != null)
|
if (nonSerialWriteStrategy.writesThroughAccord && !SchemaConstants.getSystemKeyspaces().contains(keyspaceName))
|
||||||
|
mutateWithAccord(augmented != null ? augmented : mutations, consistencyLevel, requestTime, nonSerialWriteStrategy);
|
||||||
|
else if (augmented != null)
|
||||||
mutateAtomically(augmented, consistencyLevel, updatesView, requestTime);
|
mutateAtomically(augmented, consistencyLevel, updatesView, requestTime);
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
|
@ -1184,6 +1232,29 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static void mutateWithAccord(Collection<? extends IMutation> iMutations, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, Config.NonSerialWriteStrategy nonSerialWriteStrategy)
|
||||||
|
{
|
||||||
|
int fragmentIndex = 0;
|
||||||
|
List<TxnWrite.Fragment> fragments = new ArrayList<>(iMutations.size());
|
||||||
|
List<PartitionKey> partitionKeys = new ArrayList<>(iMutations.size());
|
||||||
|
for (IMutation mutation : iMutations)
|
||||||
|
{
|
||||||
|
for (PartitionUpdate update : mutation.getPartitionUpdates())
|
||||||
|
{
|
||||||
|
PartitionKey pk = PartitionKey.of(update);
|
||||||
|
partitionKeys.add(pk);
|
||||||
|
fragments.add(new TxnWrite.Fragment(PartitionKey.of(update), fragmentIndex++, update, TxnReferenceOperations.empty()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Potentially ignore commit consistency level if the strategy specifies accord and not migration
|
||||||
|
ConsistencyLevel clForCommit = nonSerialWriteStrategy.commitCLForStrategy(consistencyLevel);
|
||||||
|
AccordUpdate update = new TxnUpdate(fragments, TxnCondition.none(), clForCommit);
|
||||||
|
Txn.InMemory txn = new Txn.InMemory(Keys.of(partitionKeys), TxnRead.EMPTY, TxnQuery.EMPTY, update);
|
||||||
|
IAccordService accordService = AccordService.instance();
|
||||||
|
accordService.maybeConvertKeyspacesToAccord(txn);
|
||||||
|
accordService.coordinate(txn, consistencyLevel, requestTime);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* See mutate. Adds additional steps before and after writing a batch.
|
* See mutate. Adds additional steps before and after writing a batch.
|
||||||
* Before writing the batch (but after doing availability check against the FD for the row replicas):
|
* Before writing the batch (but after doing availability check against the FD for the row replicas):
|
||||||
|
|
@ -1602,7 +1673,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
|
|
||||||
if (insertLocal)
|
if (insertLocal)
|
||||||
{
|
{
|
||||||
Preconditions.checkNotNull(localReplica);
|
checkNotNull(localReplica);
|
||||||
performLocally(stage, localReplica, mutation::apply, responseHandler, mutation, requestTime);
|
performLocally(stage, localReplica, mutation::apply, responseHandler, mutation, requestTime);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -1881,43 +1952,68 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
return metadata.myNodeState() == NodeState.JOINED;
|
return metadata.myNodeState() == NodeState.JOINED;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static ConsensusRoutingDecision consensusRouting(TableMetadata metadata, DecoratedKey partitionKey, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, boolean isForWrite)
|
||||||
|
{
|
||||||
|
if (metadata.keyspace.equals(SchemaConstants.METADATA_KEYSPACE_NAME))
|
||||||
|
return ConsensusRoutingDecision.paxosV2;
|
||||||
|
return ConsensusRequestRouter.instance.routeAndMaybeMigrate(partitionKey,
|
||||||
|
metadata.id,
|
||||||
|
consistencyLevel,
|
||||||
|
requestTime,
|
||||||
|
DatabaseDescriptor.getCasContentionTimeout(NANOSECONDS),
|
||||||
|
isForWrite);
|
||||||
|
}
|
||||||
|
|
||||||
private static PartitionIterator readWithConsensus(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
private static PartitionIterator readWithConsensus(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||||
throws InvalidRequestException, UnavailableException, ReadFailureException, ReadTimeoutException
|
throws InvalidRequestException, UnavailableException, ReadFailureException, ReadTimeoutException
|
||||||
{
|
{
|
||||||
// TCM explicitly relies on paxos and doesn't work with accord
|
ConsensusAttemptResult lastResult;
|
||||||
if (DatabaseDescriptor.getLegacyPaxosStrategy() == accord && !group.metadata().keyspace.equals(SchemaConstants.METADATA_KEYSPACE_NAME))
|
do
|
||||||
{
|
{
|
||||||
return readWithAccord(group, consistencyLevel);
|
SinglePartitionReadCommand command = group.queries.get(0);
|
||||||
}
|
ConsensusRoutingDecision decision = consensusRouting(group.metadata(), command.partitionKey(), consistencyLevel, requestTime, false);
|
||||||
else
|
switch (decision)
|
||||||
{
|
{
|
||||||
return readWithPaxos(group, consistencyLevel, requestTime);
|
case paxosV2:
|
||||||
}
|
lastResult = Paxos.read(group, consistencyLevel, requestTime);
|
||||||
|
break;
|
||||||
|
case paxosV1:
|
||||||
|
lastResult = legacyReadWithPaxos(group, consistencyLevel, requestTime);
|
||||||
|
break;
|
||||||
|
case accord:
|
||||||
|
lastResult = readWithAccord(group, consistencyLevel, requestTime);
|
||||||
|
break;
|
||||||
|
default:
|
||||||
|
throw new IllegalStateException("Unsupported consensus " + decision);
|
||||||
|
}
|
||||||
|
} while (lastResult.shouldRetryOnNewConsensusProtocol);
|
||||||
|
return lastResult.serialReadResult;
|
||||||
}
|
}
|
||||||
|
|
||||||
private static PartitionIterator readWithPaxos(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
private static ConsensusAttemptResult readWithAccord(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||||
throws InvalidRequestException, UnavailableException, ReadFailureException, ReadTimeoutException
|
|
||||||
{
|
|
||||||
return (Paxos.useV2() || group.metadata().keyspace.equals(SchemaConstants.METADATA_KEYSPACE_NAME))
|
|
||||||
? Paxos.read(group, consistencyLevel, requestTime)
|
|
||||||
: legacyReadWithPaxos(group, consistencyLevel, requestTime);
|
|
||||||
}
|
|
||||||
|
|
||||||
private static PartitionIterator readWithAccord(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel)
|
|
||||||
{
|
{
|
||||||
if (group.queries.size() > 1)
|
if (group.queries.size() > 1)
|
||||||
throw new InvalidRequestException("SERIAL/LOCAL_SERIAL consistency may only be requested for one partition at a time");
|
throw new InvalidRequestException("SERIAL/LOCAL_SERIAL consistency may only be requested for one partition at a time");
|
||||||
TxnRead read = TxnRead.createSerialRead(group.queries.get(0));
|
SinglePartitionReadCommand readCommand = group.queries.get(0);
|
||||||
|
// If the non-SERIAL write strategy is sending all writes through Accord there is no need to use the supplied consistency
|
||||||
|
// level since Accord will manage reading safely
|
||||||
|
consistencyLevel = DatabaseDescriptor.getNonSerialWriteStrategy().readCLForStrategy(consistencyLevel);
|
||||||
|
TxnRead read = TxnRead.createSerialRead(readCommand, consistencyLevel);
|
||||||
Txn txn = new Txn.InMemory(read.keys(), read, TxnQuery.ALL);
|
Txn txn = new Txn.InMemory(read.keys(), read, TxnQuery.ALL);
|
||||||
TxnData data = AccordService.instance().coordinate(txn, consistencyLevel);
|
IAccordService accordService = AccordService.instance();
|
||||||
|
accordService.maybeConvertKeyspacesToAccord(txn);
|
||||||
|
TxnResult txnResult = accordService.coordinate(txn, consistencyLevel, requestTime);
|
||||||
|
if (txnResult.kind() == retry_new_protocol)
|
||||||
|
return RETRY_NEW_PROTOCOL;
|
||||||
|
TxnData data = (TxnData)txnResult;
|
||||||
FilteredPartition partition = data.get(TxnRead.SERIAL_READ);
|
FilteredPartition partition = data.get(TxnRead.SERIAL_READ);
|
||||||
if (partition != null)
|
if (partition != null)
|
||||||
return PartitionIterators.singletonIterator(partition.rowIterator());
|
return serialReadResult(PartitionIterators.singletonIterator(partition.rowIterator(readCommand.isReversed())));
|
||||||
else
|
else
|
||||||
return EmptyIterators.partition();
|
return serialReadResult(EmptyIterators.partition());
|
||||||
}
|
}
|
||||||
|
|
||||||
private static PartitionIterator legacyReadWithPaxos(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
private static ConsensusAttemptResult legacyReadWithPaxos(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||||
throws InvalidRequestException, UnavailableException, ReadFailureException, ReadTimeoutException
|
throws InvalidRequestException, UnavailableException, ReadFailureException, ReadTimeoutException
|
||||||
{
|
{
|
||||||
long start = nanoTime();
|
long start = nanoTime();
|
||||||
|
|
@ -1930,7 +2026,6 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
// calculate the blockFor before repair any paxos round to avoid RS being altered in between.
|
// calculate the blockFor before repair any paxos round to avoid RS being altered in between.
|
||||||
int blockForRead = consistencyLevel.blockFor(Keyspace.open(metadata.keyspace).getReplicationStrategy());
|
int blockForRead = consistencyLevel.blockFor(Keyspace.open(metadata.keyspace).getReplicationStrategy());
|
||||||
|
|
||||||
PartitionIterator result = null;
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
final ConsistencyLevel consistencyForReplayCommitsOrFetch = consistencyLevel == ConsistencyLevel.LOCAL_SERIAL
|
final ConsistencyLevel consistencyForReplayCommitsOrFetch = consistencyLevel == ConsistencyLevel.LOCAL_SERIAL
|
||||||
|
|
@ -1967,7 +2062,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
throw new ReadFailureException(consistencyLevel, e.received, e.blockFor, false, e.failureReasonByEndpoint);
|
throw new ReadFailureException(consistencyLevel, e.received, e.blockFor, false, e.failureReasonByEndpoint);
|
||||||
}
|
}
|
||||||
|
|
||||||
result = fetchRows(group.queries, consistencyForReplayCommitsOrFetch, requestTime);
|
return serialReadResult(fetchRows(group.queries, consistencyForReplayCommitsOrFetch, ReadCoordinator.DEFAULT, requestTime));
|
||||||
}
|
}
|
||||||
catch (UnavailableException e)
|
catch (UnavailableException e)
|
||||||
{
|
{
|
||||||
|
|
@ -2011,18 +2106,16 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
readMetricsForLevel(consistencyLevel).addNano(latency);
|
readMetricsForLevel(consistencyLevel).addNano(latency);
|
||||||
Keyspace.open(metadata.keyspace).getColumnFamilyStore(metadata.name).metric.coordinatorReadLatency.update(latency, TimeUnit.NANOSECONDS);
|
Keyspace.open(metadata.keyspace).getColumnFamilyStore(metadata.name).metric.coordinatorReadLatency.update(latency, TimeUnit.NANOSECONDS);
|
||||||
}
|
}
|
||||||
|
|
||||||
return result;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
@SuppressWarnings("resource")
|
@SuppressWarnings("resource")
|
||||||
private static PartitionIterator readRegular(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
public static PartitionIterator readRegular(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, ReadCoordinator coordinator, Dispatcher.RequestTime requestTime)
|
||||||
throws UnavailableException, ReadFailureException, ReadTimeoutException
|
throws UnavailableException, ReadFailureException, ReadTimeoutException
|
||||||
{
|
{
|
||||||
long start = nanoTime();
|
long start = nanoTime();
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
PartitionIterator result = fetchRows(group.queries, consistencyLevel, requestTime);
|
PartitionIterator result = fetchRows(group.queries, consistencyLevel, coordinator, requestTime);
|
||||||
// Note that the only difference between the command in a group must be the partition key on which
|
// Note that the only difference between the command in a group must be the partition key on which
|
||||||
// they applied.
|
// they applied.
|
||||||
boolean enforceStrictLiveness = group.queries.get(0).metadata().enforceStrictLiveness();
|
boolean enforceStrictLiveness = group.queries.get(0).metadata().enforceStrictLiveness();
|
||||||
|
|
@ -2072,6 +2165,11 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static PartitionIterator readRegular(SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||||
|
{
|
||||||
|
return readRegular(group, consistencyLevel, ReadCoordinator.DEFAULT, requestTime);
|
||||||
|
}
|
||||||
|
|
||||||
public static void recordReadRegularAbort(ConsistencyLevel consistencyLevel, Throwable cause)
|
public static void recordReadRegularAbort(ConsistencyLevel consistencyLevel, Throwable cause)
|
||||||
{
|
{
|
||||||
readMetrics.markAbort(cause);
|
readMetrics.markAbort(cause);
|
||||||
|
|
@ -2119,6 +2217,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
*/
|
*/
|
||||||
private static PartitionIterator fetchRows(List<SinglePartitionReadCommand> commands,
|
private static PartitionIterator fetchRows(List<SinglePartitionReadCommand> commands,
|
||||||
ConsistencyLevel consistencyLevel,
|
ConsistencyLevel consistencyLevel,
|
||||||
|
ReadCoordinator coordinator,
|
||||||
Dispatcher.RequestTime requestTime)
|
Dispatcher.RequestTime requestTime)
|
||||||
throws UnavailableException, ReadFailureException, ReadTimeoutException
|
throws UnavailableException, ReadFailureException, ReadTimeoutException
|
||||||
{
|
{
|
||||||
|
|
@ -2131,7 +2230,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
// for type of speculation we'll use in this read
|
// for type of speculation we'll use in this read
|
||||||
for (int i=0; i<cmdCount; i++)
|
for (int i=0; i<cmdCount; i++)
|
||||||
{
|
{
|
||||||
reads[i] = AbstractReadExecutor.getReadExecutor(metadata, commands.get(i), consistencyLevel, requestTime);
|
reads[i] = AbstractReadExecutor.getReadExecutor(metadata, commands.get(i), consistencyLevel, coordinator, requestTime);
|
||||||
|
|
||||||
if (reads[i].hasLocalRead())
|
if (reads[i].hasLocalRead())
|
||||||
readMetrics.localRequests.mark();
|
readMetrics.localRequests.mark();
|
||||||
|
|
@ -3021,6 +3120,37 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static class ConsensusAttemptResult
|
||||||
|
{
|
||||||
|
public static final ConsensusAttemptResult RETRY_NEW_PROTOCOL = new ConsensusAttemptResult(null, null, true);
|
||||||
|
|
||||||
|
@Nullable
|
||||||
|
RowIterator casResult;
|
||||||
|
|
||||||
|
@Nonnull
|
||||||
|
PartitionIterator serialReadResult;
|
||||||
|
|
||||||
|
boolean shouldRetryOnNewConsensusProtocol;
|
||||||
|
|
||||||
|
private ConsensusAttemptResult(@Nullable RowIterator casResult, @Nullable PartitionIterator serialReadResult, boolean shouldRetryOnNewConsensusProtocol)
|
||||||
|
{
|
||||||
|
this.casResult = casResult;
|
||||||
|
this.serialReadResult = serialReadResult;
|
||||||
|
this.shouldRetryOnNewConsensusProtocol = shouldRetryOnNewConsensusProtocol;
|
||||||
|
}
|
||||||
|
|
||||||
|
public static ConsensusAttemptResult serialReadResult(@Nonnull PartitionIterator serialReadResult)
|
||||||
|
{
|
||||||
|
checkNotNull(serialReadResult, "serialReadResult should not be null");
|
||||||
|
return new ConsensusAttemptResult(null, serialReadResult, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
public static ConsensusAttemptResult casResult(@Nullable RowIterator casResult)
|
||||||
|
{
|
||||||
|
return new ConsensusAttemptResult(casResult, null, false);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public boolean getSnapshotOnDuplicateRowDetectionEnabled()
|
public boolean getSnapshotOnDuplicateRowDetectionEnabled()
|
||||||
{
|
{
|
||||||
|
|
@ -3207,7 +3337,7 @@ public class StorageProxy implements StorageProxyMBean
|
||||||
@Override
|
@Override
|
||||||
public void setPaxosVariant(String variant)
|
public void setPaxosVariant(String variant)
|
||||||
{
|
{
|
||||||
Preconditions.checkNotNull(variant);
|
checkNotNull(variant);
|
||||||
Paxos.setPaxosVariant(Config.PaxosVariant.valueOf(variant));
|
Paxos.setPaxosVariant(Config.PaxosVariant.valueOf(variant));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -50,6 +50,7 @@ import java.util.concurrent.atomic.AtomicReference;
|
||||||
import java.util.function.Supplier;
|
import java.util.function.Supplier;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
import java.util.stream.Stream;
|
import java.util.stream.Stream;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
import javax.management.ListenerNotFoundException;
|
import javax.management.ListenerNotFoundException;
|
||||||
import javax.management.NotificationBroadcasterSupport;
|
import javax.management.NotificationBroadcasterSupport;
|
||||||
|
|
@ -61,7 +62,6 @@ import javax.management.openmbean.TabularData;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.base.Joiner;
|
import com.google.common.base.Joiner;
|
||||||
import com.google.common.base.Preconditions;
|
|
||||||
import com.google.common.base.Predicate;
|
import com.google.common.base.Predicate;
|
||||||
import com.google.common.collect.ImmutableList;
|
import com.google.common.collect.ImmutableList;
|
||||||
import com.google.common.collect.ImmutableMap;
|
import com.google.common.collect.ImmutableMap;
|
||||||
|
|
@ -124,6 +124,7 @@ import org.apache.cassandra.gms.FailureDetector;
|
||||||
import org.apache.cassandra.gms.Gossiper;
|
import org.apache.cassandra.gms.Gossiper;
|
||||||
import org.apache.cassandra.gms.IEndpointStateChangeSubscriber;
|
import org.apache.cassandra.gms.IEndpointStateChangeSubscriber;
|
||||||
import org.apache.cassandra.gms.VersionedValue;
|
import org.apache.cassandra.gms.VersionedValue;
|
||||||
|
import org.apache.cassandra.gms.VersionedValue.VersionedValueFactory;
|
||||||
import org.apache.cassandra.hints.Hint;
|
import org.apache.cassandra.hints.Hint;
|
||||||
import org.apache.cassandra.hints.HintsService;
|
import org.apache.cassandra.hints.HintsService;
|
||||||
import org.apache.cassandra.index.IndexStatusManager;
|
import org.apache.cassandra.index.IndexStatusManager;
|
||||||
|
|
@ -167,6 +168,9 @@ import org.apache.cassandra.schema.TableId;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.schema.TableMetadataRef;
|
import org.apache.cassandra.schema.TableMetadataRef;
|
||||||
import org.apache.cassandra.schema.ViewMetadata;
|
import org.apache.cassandra.schema.ViewMetadata;
|
||||||
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.ConsensusMigrationState;
|
||||||
import org.apache.cassandra.service.disk.usage.DiskUsageBroadcaster;
|
import org.apache.cassandra.service.disk.usage.DiskUsageBroadcaster;
|
||||||
import org.apache.cassandra.service.paxos.Paxos;
|
import org.apache.cassandra.service.paxos.Paxos;
|
||||||
import org.apache.cassandra.service.paxos.PaxosCommit;
|
import org.apache.cassandra.service.paxos.PaxosCommit;
|
||||||
|
|
@ -226,6 +230,8 @@ import org.apache.cassandra.utils.progress.jmx.JMXBroadcastExecutor;
|
||||||
import org.apache.cassandra.utils.progress.jmx.JMXProgressSupport;
|
import org.apache.cassandra.utils.progress.jmx.JMXProgressSupport;
|
||||||
|
|
||||||
import static com.google.common.base.Preconditions.checkArgument;
|
import static com.google.common.base.Preconditions.checkArgument;
|
||||||
|
import static com.google.common.base.Preconditions.checkNotNull;
|
||||||
|
import static com.google.common.base.Preconditions.checkState;
|
||||||
import static java.util.Arrays.asList;
|
import static java.util.Arrays.asList;
|
||||||
import static java.util.Arrays.stream;
|
import static java.util.Arrays.stream;
|
||||||
import static java.util.concurrent.TimeUnit.MILLISECONDS;
|
import static java.util.concurrent.TimeUnit.MILLISECONDS;
|
||||||
|
|
@ -253,6 +259,8 @@ import static org.apache.cassandra.service.StorageService.Mode.JOINING_FAILED;
|
||||||
import static org.apache.cassandra.service.StorageService.Mode.LEAVING;
|
import static org.apache.cassandra.service.StorageService.Mode.LEAVING;
|
||||||
import static org.apache.cassandra.service.StorageService.Mode.MOVE_FAILED;
|
import static org.apache.cassandra.service.StorageService.Mode.MOVE_FAILED;
|
||||||
import static org.apache.cassandra.service.StorageService.Mode.NORMAL;
|
import static org.apache.cassandra.service.StorageService.Mode.NORMAL;
|
||||||
|
import static org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.finishMigrationToConsensusProtocol;
|
||||||
|
import static org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.startMigrationToConsensusProtocol;
|
||||||
import static org.apache.cassandra.tcm.membership.NodeState.BOOTSTRAPPING;
|
import static org.apache.cassandra.tcm.membership.NodeState.BOOTSTRAPPING;
|
||||||
import static org.apache.cassandra.tcm.membership.NodeState.BOOT_REPLACING;
|
import static org.apache.cassandra.tcm.membership.NodeState.BOOT_REPLACING;
|
||||||
import static org.apache.cassandra.tcm.membership.NodeState.JOINED;
|
import static org.apache.cassandra.tcm.membership.NodeState.JOINED;
|
||||||
|
|
@ -260,6 +268,7 @@ import static org.apache.cassandra.tcm.membership.NodeState.MOVING;
|
||||||
import static org.apache.cassandra.tcm.membership.NodeState.REGISTERED;
|
import static org.apache.cassandra.tcm.membership.NodeState.REGISTERED;
|
||||||
import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis;
|
import static org.apache.cassandra.utils.Clock.Global.currentTimeMillis;
|
||||||
import static org.apache.cassandra.utils.FBUtilities.getBroadcastAddressAndPort;
|
import static org.apache.cassandra.utils.FBUtilities.getBroadcastAddressAndPort;
|
||||||
|
import static org.apache.cassandra.utils.PojoToString.pojoMapToString;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* This abstraction contains the token/identifier of this node
|
* This abstraction contains the token/identifier of this node
|
||||||
|
|
@ -315,6 +324,9 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
|
||||||
@VisibleForTesting // this is used for dtests only, see CASSANDRA-18152
|
@VisibleForTesting // this is used for dtests only, see CASSANDRA-18152
|
||||||
public volatile boolean skipNotificationListeners = false;
|
public volatile boolean skipNotificationListeners = false;
|
||||||
|
|
||||||
|
// For tests that unsafely change the partitioner store the original here
|
||||||
|
private IPartitioner originalPartitioner;
|
||||||
|
|
||||||
private final java.util.function.Predicate<Keyspace> anyOutOfRangeOpsRecorded
|
private final java.util.function.Predicate<Keyspace> anyOutOfRangeOpsRecorded
|
||||||
= keyspace -> keyspace.metric.outOfRangeTokenReads.getCount() > 0
|
= keyspace -> keyspace.metric.outOfRangeTokenReads.getCount() > 0
|
||||||
|| keyspace.metric.outOfRangeTokenWrites.getCount() > 0
|
|| keyspace.metric.outOfRangeTokenWrites.getCount() > 0
|
||||||
|
|
@ -1664,6 +1676,49 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void migrateConsensusProtocol(@Nonnull String targetProtocol,
|
||||||
|
@Nonnull List<String> keyspaceNames,
|
||||||
|
@Nullable List<String> maybeTableNames,
|
||||||
|
@Nullable String maybeRangesStr)
|
||||||
|
{
|
||||||
|
checkNotNull(targetProtocol, "targetProtocol is null");
|
||||||
|
checkArgument(!keyspaceNames.contains(SchemaConstants.METADATA_KEYSPACE_NAME));
|
||||||
|
startMigrationToConsensusProtocol(targetProtocol, keyspaceNames, Optional.ofNullable(maybeTableNames), Optional.ofNullable(maybeRangesStr));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<Integer> finishConsensusMigration(@Nonnull String keyspace,
|
||||||
|
@Nullable List<String> maybeTableNames,
|
||||||
|
@Nullable String maybeRangesStr)
|
||||||
|
{
|
||||||
|
checkArgument(!keyspace.equals(SchemaConstants.METADATA_KEYSPACE_NAME));
|
||||||
|
return finishMigrationToConsensusProtocol(keyspace, Optional.ofNullable(maybeTableNames), Optional.ofNullable(maybeRangesStr));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void setConsensusMigrationTargetProtocol(@Nonnull String targetProtocol,
|
||||||
|
@Nullable List<String> keyspaceNames,
|
||||||
|
@Nullable List<String> maybeTableNames)
|
||||||
|
{
|
||||||
|
checkNotNull(targetProtocol, "targetProtocol is null");
|
||||||
|
checkNotNull(keyspaceNames, "keyspaceNames is null");
|
||||||
|
checkArgument(!keyspaceNames.contains(SchemaConstants.METADATA_KEYSPACE_NAME));
|
||||||
|
|
||||||
|
ConsensusTableMigrationState.setConsensusMigrationTargetProtocol(targetProtocol, keyspaceNames, Optional.ofNullable(maybeTableNames));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String listConsensusMigrations(@Nullable Set<String> keyspaceNames,
|
||||||
|
@Nullable Set<String> tableNames,
|
||||||
|
@Nonnull String format)
|
||||||
|
{
|
||||||
|
ClusterMetadata cm = ClusterMetadata.current();
|
||||||
|
ConsensusMigrationState snapshot = cm.consensusMigrationState;
|
||||||
|
Map<String, Object> snapshotAsMap = snapshot.toMap(keyspaceNames, tableNames);
|
||||||
|
return pojoMapToString(snapshotAsMap, format);
|
||||||
|
}
|
||||||
|
|
||||||
public Map<String,List<Integer>> getConcurrency(List<String> stageNames)
|
public Map<String,List<Integer>> getConcurrency(List<String> stageNames)
|
||||||
{
|
{
|
||||||
Stream<Stage> stageStream = stageNames.isEmpty() ? stream(Stage.values()) : stageNames.stream().map(Stage::fromPoolName);
|
Stream<Stage> stageStream = stageNames.isEmpty() ? stream(Stage.values()) : stageNames.stream().map(Stage::fromPoolName);
|
||||||
|
|
@ -3978,11 +4033,21 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
|
||||||
|
|
||||||
// Never ever do this at home. Used by tests.
|
// Never ever do this at home. Used by tests.
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
public IPartitioner setPartitionerUnsafe(IPartitioner newPartitioner)
|
public void setPartitionerUnsafe(IPartitioner newPartitioner)
|
||||||
{
|
{
|
||||||
IPartitioner oldPartitioner = DatabaseDescriptor.setPartitionerUnsafe(newPartitioner);
|
checkNotNull(newPartitioner, "newPartitioner is null");
|
||||||
|
checkState(originalPartitioner == null, "Already changed the partitioner without resetting");
|
||||||
|
originalPartitioner = DatabaseDescriptor.setPartitionerUnsafe(newPartitioner);
|
||||||
valueFactory = new VersionedValue.VersionedValueFactory(newPartitioner);
|
valueFactory = new VersionedValue.VersionedValueFactory(newPartitioner);
|
||||||
return oldPartitioner;
|
}
|
||||||
|
|
||||||
|
@VisibleForTesting
|
||||||
|
public void resetPartitionerUnsafe()
|
||||||
|
{
|
||||||
|
checkState(originalPartitioner != null, "Original partitioner was never changed");
|
||||||
|
DatabaseDescriptor.setPartitionerUnsafe(originalPartitioner);
|
||||||
|
valueFactory = new VersionedValueFactory(originalPartitioner);
|
||||||
|
originalPartitioner = null;
|
||||||
}
|
}
|
||||||
|
|
||||||
public void truncate(String keyspace, String table) throws TimeoutException, IOException
|
public void truncate(String keyspace, String table) throws TimeoutException, IOException
|
||||||
|
|
@ -4165,6 +4230,16 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
|
||||||
return Lists.newArrayList(Schema.instance.distributedKeyspaces().names());
|
return Lists.newArrayList(Schema.instance.distributedKeyspaces().names());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<String> getAccordManagedKeyspaces()
|
||||||
|
{
|
||||||
|
// TODO (review) These are really just the ones Accord is aware of not necessarily managed
|
||||||
|
Set<String> keyspaces = Schema.instance.getNonLocalStrategyKeyspaces().names();
|
||||||
|
return keyspaces.stream()
|
||||||
|
.filter(AccordService.instance()::isAccordManagedKeyspace)
|
||||||
|
.collect(toList());
|
||||||
|
}
|
||||||
|
|
||||||
public Map<String, String> getViewBuildStatuses(String keyspace, String view, boolean withPort)
|
public Map<String, String> getViewBuildStatuses(String keyspace, String view, boolean withPort)
|
||||||
{
|
{
|
||||||
Map<UUID, String> coreViewStatus = SystemDistributedKeyspace.viewStatus(keyspace, view);
|
Map<UUID, String> coreViewStatus = SystemDistributedKeyspace.viewStatus(keyspace, view);
|
||||||
|
|
@ -4994,7 +5069,7 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
|
||||||
archiveCommand = archiveCommand != null ? archiveCommand : fqlOptions.archive_command;
|
archiveCommand = archiveCommand != null ? archiveCommand : fqlOptions.archive_command;
|
||||||
maxArchiveRetries = maxArchiveRetries != Integer.MIN_VALUE ? maxArchiveRetries : fqlOptions.max_archive_retries;
|
maxArchiveRetries = maxArchiveRetries != Integer.MIN_VALUE ? maxArchiveRetries : fqlOptions.max_archive_retries;
|
||||||
|
|
||||||
Preconditions.checkNotNull(path, "cassandra.yaml did not set log_dir and not set as parameter");
|
checkNotNull(path, "cassandra.yaml did not set log_dir and not set as parameter");
|
||||||
FullQueryLogger.instance.enableWithoutClean(File.getPath(path), rollCycle, blocking, maxQueueWeight, maxLogSize, archiveCommand, maxArchiveRetries);
|
FullQueryLogger.instance.enableWithoutClean(File.getPath(path), rollCycle, blocking, maxQueueWeight, maxLogSize, archiveCommand, maxArchiveRetries);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -5359,7 +5434,7 @@ public class StorageService extends NotificationBroadcasterSupport implements IE
|
||||||
|
|
||||||
public void setRepairRpcTimeout(Long timeoutInMillis)
|
public void setRepairRpcTimeout(Long timeoutInMillis)
|
||||||
{
|
{
|
||||||
Preconditions.checkState(timeoutInMillis > 0);
|
checkState(timeoutInMillis > 0);
|
||||||
DatabaseDescriptor.setRepairRpcTimeout(timeoutInMillis);
|
DatabaseDescriptor.setRepairRpcTimeout(timeoutInMillis);
|
||||||
logger.info("RepairRpcTimeout set to {}ms via JMX", timeoutInMillis);
|
logger.info("RepairRpcTimeout set to {}ms via JMX", timeoutInMillis);
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -27,6 +27,7 @@ import java.util.Map;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
import java.util.concurrent.ExecutionException;
|
import java.util.concurrent.ExecutionException;
|
||||||
import java.util.concurrent.TimeoutException;
|
import java.util.concurrent.TimeoutException;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
import javax.management.NotificationEmitter;
|
import javax.management.NotificationEmitter;
|
||||||
import javax.management.openmbean.CompositeData;
|
import javax.management.openmbean.CompositeData;
|
||||||
|
|
@ -1141,6 +1142,23 @@ public interface StorageServiceMBean extends NotificationEmitter
|
||||||
public String getBootstrapState();
|
public String getBootstrapState();
|
||||||
void abortBootstrap(String nodeId, String endpoint);
|
void abortBootstrap(String nodeId, String endpoint);
|
||||||
|
|
||||||
|
void migrateConsensusProtocol(@Nonnull String targetProtocol,
|
||||||
|
@Nullable List<String> keyspaceNames,
|
||||||
|
@Nullable List<String> maybeTableNames,
|
||||||
|
@Nullable String maybeRangesStr);
|
||||||
|
|
||||||
|
List<Integer> finishConsensusMigration(@Nonnull String keyspace,
|
||||||
|
@Nullable List<String> maybeTableNames,
|
||||||
|
@Nullable String maybeRangesStr);
|
||||||
|
|
||||||
|
void setConsensusMigrationTargetProtocol(@Nonnull String targetProtocol,
|
||||||
|
@Nullable List<String> keyspaceNames,
|
||||||
|
@Nullable List<String> maybeTableNames);
|
||||||
|
|
||||||
|
String listConsensusMigrations(@Nullable Set<String> keyspaceNames, @Nullable Set<String> tableNames, @Nonnull String format);
|
||||||
|
|
||||||
|
List<String> getAccordManagedKeyspaces();
|
||||||
|
|
||||||
/** Gets the concurrency settings for processing stages*/
|
/** Gets the concurrency settings for processing stages*/
|
||||||
static class StageConcurrency implements Serializable
|
static class StageConcurrency implements Serializable
|
||||||
{
|
{
|
||||||
|
|
@ -1188,6 +1206,7 @@ public interface StorageServiceMBean extends NotificationEmitter
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Start the fully query logger.
|
* Start the fully query logger.
|
||||||
|
*
|
||||||
* @param path Path where the full query log will be stored. If null cassandra.yaml value is used.
|
* @param path Path where the full query log will be stored. If null cassandra.yaml value is used.
|
||||||
* @param rollCycle How often to create a new file for query data (MINUTELY, DAILY, HOURLY)
|
* @param rollCycle How often to create a new file for query data (MINUTELY, DAILY, HOURLY)
|
||||||
* @param blocking Whether threads submitting queries to the query log should block if they can't be drained to the filesystem or alternatively drops samples and log
|
* @param blocking Whether threads submitting queries to the query log should block if they can't be drained to the filesystem or alternatively drops samples and log
|
||||||
|
|
|
||||||
|
|
@ -17,8 +17,6 @@
|
||||||
*/
|
*/
|
||||||
package org.apache.cassandra.service.accord;
|
package org.apache.cassandra.service.accord;
|
||||||
|
|
||||||
import java.util.Collections;
|
|
||||||
import java.util.Set;
|
|
||||||
import java.util.concurrent.Callable;
|
import java.util.concurrent.Callable;
|
||||||
import java.util.function.BiFunction;
|
import java.util.function.BiFunction;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
|
@ -27,7 +25,7 @@ import java.util.function.ToLongFunction;
|
||||||
import com.google.common.primitives.Ints;
|
import com.google.common.primitives.Ints;
|
||||||
|
|
||||||
import accord.local.Command.TransientListener;
|
import accord.local.Command.TransientListener;
|
||||||
import accord.utils.DeterministicIdentitySet;
|
import accord.local.Listeners;
|
||||||
import accord.utils.IntrusiveLinkedListNode;
|
import accord.utils.IntrusiveLinkedListNode;
|
||||||
import accord.utils.async.AsyncChain;
|
import accord.utils.async.AsyncChain;
|
||||||
import accord.utils.async.AsyncResults.RunnableResult;
|
import accord.utils.async.AsyncResults.RunnableResult;
|
||||||
|
|
@ -35,14 +33,14 @@ import org.apache.cassandra.concurrent.ExecutorPlus;
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
|
|
||||||
import static java.lang.String.format;
|
import static java.lang.String.format;
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.UNINITIALIZED;
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.EVICTED;
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.LOADING;
|
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.LOADED;
|
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.FAILED_TO_LOAD;
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.FAILED_TO_LOAD;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.FAILED_TO_SAVE;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.LOADED;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.LOADING;
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.MODIFIED;
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.MODIFIED;
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.SAVING;
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.SAVING;
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.FAILED_TO_SAVE;
|
import static org.apache.cassandra.service.accord.AccordCachingState.Status.UNINITIALIZED;
|
||||||
import static org.apache.cassandra.service.accord.AccordCachingState.Status.EVICTED;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Global (per CommandStore) state of a cached entity (Command or CommandsForKey).
|
* Global (per CommandStore) state of a cached entity (Command or CommandsForKey).
|
||||||
|
|
@ -61,7 +59,7 @@ public class AccordCachingState<K, V> extends IntrusiveLinkedListNode
|
||||||
/**
|
/**
|
||||||
* Transient listeners aren't meant to survive process restart, but must survive cache eviction.
|
* Transient listeners aren't meant to survive process restart, but must survive cache eviction.
|
||||||
*/
|
*/
|
||||||
private Set<TransientListener> transientListeners;
|
private Listeners<TransientListener> transientListeners;
|
||||||
|
|
||||||
public AccordCachingState(K key)
|
public AccordCachingState(K key)
|
||||||
{
|
{
|
||||||
|
|
@ -140,7 +138,7 @@ public class AccordCachingState<K, V> extends IntrusiveLinkedListNode
|
||||||
public void addListener(TransientListener listener)
|
public void addListener(TransientListener listener)
|
||||||
{
|
{
|
||||||
if (transientListeners == null)
|
if (transientListeners == null)
|
||||||
transientListeners = new DeterministicIdentitySet<>();
|
transientListeners = new Listeners<>();
|
||||||
transientListeners.add(listener);
|
transientListeners.add(listener);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -149,14 +147,14 @@ public class AccordCachingState<K, V> extends IntrusiveLinkedListNode
|
||||||
return transientListeners != null && transientListeners.remove(listener);
|
return transientListeners != null && transientListeners.remove(listener);
|
||||||
}
|
}
|
||||||
|
|
||||||
public void listeners(Set<TransientListener> listeners)
|
public void listeners(Listeners<TransientListener> listeners)
|
||||||
{
|
{
|
||||||
transientListeners = listeners;
|
transientListeners = listeners;
|
||||||
}
|
}
|
||||||
|
|
||||||
public Set<TransientListener> listeners()
|
public Listeners<TransientListener> listeners()
|
||||||
{
|
{
|
||||||
return transientListeners == null ? Collections.emptySet() : transientListeners;
|
return transientListeners == null ? Listeners.EMPTY : transientListeners;
|
||||||
}
|
}
|
||||||
|
|
||||||
public boolean hasListeners()
|
public boolean hasListeners()
|
||||||
|
|
|
||||||
|
|
@ -30,6 +30,8 @@ import java.util.concurrent.TimeUnit;
|
||||||
import java.util.function.BiFunction;
|
import java.util.function.BiFunction;
|
||||||
import java.util.function.Consumer;
|
import java.util.function.Consumer;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
import java.util.function.Predicate;
|
||||||
|
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
|
@ -450,7 +452,7 @@ public class AccordCommandStore extends CommandStore implements CacheSize
|
||||||
current = null;
|
current = null;
|
||||||
}
|
}
|
||||||
|
|
||||||
<O> O mapReduceForRange(Routables<?> keysOrRanges, Ranges slice, BiFunction<CommandTimeseriesHolder, O, O> map, O accumulate, O terminalValue)
|
<O> O mapReduceForRange(Routables<?> keysOrRanges, Ranges slice, BiFunction<CommandTimeseriesHolder, O, O> map, O accumulate, Predicate<O> terminate)
|
||||||
{
|
{
|
||||||
keysOrRanges = keysOrRanges.slice(slice, Routables.Slice.Minimal);
|
keysOrRanges = keysOrRanges.slice(slice, Routables.Slice.Minimal);
|
||||||
switch (keysOrRanges.domain())
|
switch (keysOrRanges.domain())
|
||||||
|
|
@ -461,7 +463,7 @@ public class AccordCommandStore extends CommandStore implements CacheSize
|
||||||
for (CommandTimeseriesHolder summary : commandsForRanges.search(keys))
|
for (CommandTimeseriesHolder summary : commandsForRanges.search(keys))
|
||||||
{
|
{
|
||||||
accumulate = map.apply(summary, accumulate);
|
accumulate = map.apply(summary, accumulate);
|
||||||
if (accumulate.equals(terminalValue))
|
if (terminate.test(accumulate))
|
||||||
return accumulate;
|
return accumulate;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -475,7 +477,7 @@ public class AccordCommandStore extends CommandStore implements CacheSize
|
||||||
if (summary == null)
|
if (summary == null)
|
||||||
continue;
|
continue;
|
||||||
accumulate = map.apply(summary, accumulate);
|
accumulate = map.apply(summary, accumulate);
|
||||||
if (accumulate.equals(terminalValue))
|
if (terminate.test(accumulate))
|
||||||
return accumulate;
|
return accumulate;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -73,7 +73,7 @@ import static org.apache.cassandra.utils.CollectionSerializers.serializedMapSize
|
||||||
|
|
||||||
public class AccordFetchCoordinator extends AbstractFetchCoordinator implements StreamManager.StreamListener
|
public class AccordFetchCoordinator extends AbstractFetchCoordinator implements StreamManager.StreamListener
|
||||||
{
|
{
|
||||||
private static final Query noopQuery = (txnId, executeAt, data, read, update) -> null;
|
private static final Query noopQuery = (txnId, executeAt, keys, data, read, update) -> null;
|
||||||
|
|
||||||
public static class StreamData implements Data
|
public static class StreamData implements Data
|
||||||
{
|
{
|
||||||
|
|
@ -145,7 +145,7 @@ public class AccordFetchCoordinator extends AbstractFetchCoordinator implements
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Data merge(Data data)
|
public StreamData merge(Data data)
|
||||||
{
|
{
|
||||||
StreamData that = (StreamData) data;
|
StreamData that = (StreamData) data;
|
||||||
if (that.streams.keySet().stream().anyMatch(this.streams::containsKey))
|
if (that.streams.keySet().stream().anyMatch(this.streams::containsKey))
|
||||||
|
|
@ -258,7 +258,7 @@ public class AccordFetchCoordinator extends AbstractFetchCoordinator implements
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public AsyncChain<Data> read(Seekable key, Txn.Kind kind, SafeCommandStore commandStore, Timestamp executeAt, DataStore store)
|
public AsyncChain<Data> read(Seekable key, SafeCommandStore commandStore, Timestamp executeAt, DataStore store)
|
||||||
{
|
{
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -20,8 +20,8 @@ package org.apache.cassandra.service.accord;
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.nio.ByteBuffer;
|
import java.nio.ByteBuffer;
|
||||||
import java.util.Collections;
|
import java.util.Collections;
|
||||||
import java.util.EnumMap;
|
import java.util.HashMap;
|
||||||
import java.util.EnumSet;
|
import java.util.List;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
import java.util.concurrent.Executor;
|
import java.util.concurrent.Executor;
|
||||||
|
|
@ -30,8 +30,12 @@ import java.util.function.Predicate;
|
||||||
import java.util.zip.Checksum;
|
import java.util.zip.Checksum;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
import com.google.common.collect.ArrayListMultimap;
|
||||||
|
import com.google.common.collect.ImmutableListMultimap;
|
||||||
|
import com.google.common.collect.ImmutableMap;
|
||||||
|
import com.google.common.collect.ListMultimap;
|
||||||
|
import com.google.common.collect.Multimap;
|
||||||
import com.google.common.primitives.Ints;
|
import com.google.common.primitives.Ints;
|
||||||
|
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
|
@ -65,6 +69,8 @@ import org.apache.cassandra.journal.KeySupport;
|
||||||
import org.apache.cassandra.journal.Params;
|
import org.apache.cassandra.journal.Params;
|
||||||
import org.apache.cassandra.journal.ValueSerializer;
|
import org.apache.cassandra.journal.ValueSerializer;
|
||||||
import org.apache.cassandra.net.MessagingService;
|
import org.apache.cassandra.net.MessagingService;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropApply;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropCommit;
|
||||||
import org.apache.cassandra.service.accord.serializers.AcceptSerializers;
|
import org.apache.cassandra.service.accord.serializers.AcceptSerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.ApplySerializers;
|
import org.apache.cassandra.service.accord.serializers.ApplySerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.BeginInvalidationSerializers;
|
import org.apache.cassandra.service.accord.serializers.BeginInvalidationSerializers;
|
||||||
|
|
@ -78,10 +84,31 @@ import org.apache.cassandra.service.accord.serializers.RecoverySerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.SetDurableSerializers;
|
import org.apache.cassandra.service.accord.serializers.SetDurableSerializers;
|
||||||
import org.apache.cassandra.utils.ByteArrayUtil;
|
import org.apache.cassandra.utils.ByteArrayUtil;
|
||||||
|
|
||||||
import static accord.messages.MessageType.*;
|
import static accord.messages.MessageType.ACCEPT_INVALIDATE_REQ;
|
||||||
|
import static accord.messages.MessageType.ACCEPT_REQ;
|
||||||
|
import static accord.messages.MessageType.APPLY_MAXIMAL_REQ;
|
||||||
|
import static accord.messages.MessageType.APPLY_MINIMAL_REQ;
|
||||||
|
import static accord.messages.MessageType.BEGIN_INVALIDATE_REQ;
|
||||||
|
import static accord.messages.MessageType.BEGIN_RECOVER_REQ;
|
||||||
|
import static accord.messages.MessageType.COMMIT_INVALIDATE_REQ;
|
||||||
|
import static accord.messages.MessageType.COMMIT_MAXIMAL_REQ;
|
||||||
|
import static accord.messages.MessageType.COMMIT_MINIMAL_REQ;
|
||||||
|
import static accord.messages.MessageType.INFORM_DURABLE_REQ;
|
||||||
|
import static accord.messages.MessageType.INFORM_OF_TXN_REQ;
|
||||||
|
import static accord.messages.MessageType.PRE_ACCEPT_REQ;
|
||||||
|
import static accord.messages.MessageType.PROPAGATE_APPLY_MSG;
|
||||||
|
import static accord.messages.MessageType.PROPAGATE_COMMIT_MSG;
|
||||||
|
import static accord.messages.MessageType.PROPAGATE_OTHER_MSG;
|
||||||
|
import static accord.messages.MessageType.PROPAGATE_PRE_ACCEPT_MSG;
|
||||||
|
import static accord.messages.MessageType.SET_GLOBALLY_DURABLE_REQ;
|
||||||
|
import static accord.messages.MessageType.SET_SHARD_DURABLE_REQ;
|
||||||
import static org.apache.cassandra.db.TypeSizes.BYTE_SIZE;
|
import static org.apache.cassandra.db.TypeSizes.BYTE_SIZE;
|
||||||
import static org.apache.cassandra.db.TypeSizes.INT_SIZE;
|
import static org.apache.cassandra.db.TypeSizes.INT_SIZE;
|
||||||
import static org.apache.cassandra.db.TypeSizes.LONG_SIZE;
|
import static org.apache.cassandra.db.TypeSizes.LONG_SIZE;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordMessageSink.AccordMessageType.INTEROP_APPLY_MAXIMAL_REQ;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordMessageSink.AccordMessageType.INTEROP_APPLY_MINIMAL_REQ;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordMessageSink.AccordMessageType.INTEROP_COMMIT_MAXIMAL_REQ;
|
||||||
|
import static org.apache.cassandra.service.accord.AccordMessageSink.AccordMessageType.INTEROP_COMMIT_MINIMAL_REQ;
|
||||||
|
|
||||||
public class AccordJournal implements Shutdownable
|
public class AccordJournal implements Shutdownable
|
||||||
{
|
{
|
||||||
|
|
@ -487,45 +514,68 @@ public class AccordJournal implements Shutdownable
|
||||||
REPLAY (0, ReplayRecord.SERIALIZER),
|
REPLAY (0, ReplayRecord.SERIALIZER),
|
||||||
|
|
||||||
/* Accord protocol requests */
|
/* Accord protocol requests */
|
||||||
PRE_ACCEPT (64, PRE_ACCEPT_REQ, PreacceptSerializers.request, TXN ),
|
PRE_ACCEPT (64, PRE_ACCEPT_REQ, PreacceptSerializers.request, TXN ),
|
||||||
ACCEPT (65, ACCEPT_REQ, AcceptSerializers.request, TXN ),
|
ACCEPT (65, ACCEPT_REQ, AcceptSerializers.request, TXN ),
|
||||||
ACCEPT_INVALIDATE (66, ACCEPT_INVALIDATE_REQ, AcceptSerializers.invalidate, EPOCH),
|
ACCEPT_INVALIDATE (66, ACCEPT_INVALIDATE_REQ, AcceptSerializers.invalidate, EPOCH),
|
||||||
COMMIT_MINIMAL (67, COMMIT_MINIMAL_REQ, CommitSerializers.request, TXN ),
|
COMMIT_MINIMAL (67, COMMIT_MINIMAL_REQ, CommitSerializers.request, TXN ),
|
||||||
COMMIT_MAXIMAL (68, COMMIT_MAXIMAL_REQ, CommitSerializers.request, TXN ),
|
COMMIT_MAXIMAL (68, COMMIT_MAXIMAL_REQ, CommitSerializers.request, TXN ),
|
||||||
COMMIT_INVALIDATE (69, COMMIT_INVALIDATE_REQ, CommitSerializers.invalidate, INVL),
|
COMMIT_INVALIDATE (69, COMMIT_INVALIDATE_REQ, CommitSerializers.invalidate, INVL ),
|
||||||
APPLY_MINIMAL (70, APPLY_MINIMAL_REQ, ApplySerializers.request, TXN ),
|
APPLY_MINIMAL (70, APPLY_MINIMAL_REQ, ApplySerializers.request, TXN ),
|
||||||
APPLY_MAXIMAL (71, APPLY_MAXIMAL_REQ, ApplySerializers.request, TXN ),
|
APPLY_MAXIMAL (71, APPLY_MAXIMAL_REQ, ApplySerializers.request, TXN ),
|
||||||
BEGIN_RECOVER (72, BEGIN_RECOVER_REQ, RecoverySerializers.request, TXN ),
|
|
||||||
BEGIN_INVALIDATE (73, BEGIN_INVALIDATE_REQ, BeginInvalidationSerializers.request, EPOCH),
|
INTEROP_COMMIT_MINIMAL (90, INTEROP_COMMIT_MINIMAL_REQ, COMMIT_MINIMAL_REQ, AccordInteropCommit.serializer, TXN),
|
||||||
INFORM_OF_TXN (74, INFORM_OF_TXN_REQ, InformOfTxnIdSerializers.request, EPOCH),
|
INTEROP_COMMIT_MAXIMAL (91, INTEROP_COMMIT_MAXIMAL_REQ, COMMIT_MINIMAL_REQ, AccordInteropCommit.serializer, TXN),
|
||||||
INFORM_DURABLE (75, INFORM_DURABLE_REQ, InformDurableSerializers.request, TXN ),
|
INTEROP_APPLY_MINIMAL (92, INTEROP_APPLY_MINIMAL_REQ, COMMIT_MINIMAL_REQ, AccordInteropApply.serializer, TXN),
|
||||||
SET_SHARD_DURABLE (76, SET_SHARD_DURABLE_REQ, SetDurableSerializers.shardDurable, EPOCH),
|
INTEROP_APPLY_MAXIMAL (93, INTEROP_APPLY_MAXIMAL_REQ, COMMIT_MINIMAL_REQ, AccordInteropApply.serializer, TXN),
|
||||||
SET_GLOBALLY_DURABLE (77, SET_GLOBALLY_DURABLE_REQ, SetDurableSerializers.globallyDurable, EPOCH),
|
|
||||||
|
BEGIN_RECOVER (72, BEGIN_RECOVER_REQ, RecoverySerializers.request, TXN ),
|
||||||
|
BEGIN_INVALIDATE (73, BEGIN_INVALIDATE_REQ, BeginInvalidationSerializers.request, EPOCH),
|
||||||
|
INFORM_OF_TXN (74, INFORM_OF_TXN_REQ, InformOfTxnIdSerializers.request, EPOCH),
|
||||||
|
INFORM_DURABLE (75, INFORM_DURABLE_REQ, InformDurableSerializers.request, TXN ),
|
||||||
|
SET_SHARD_DURABLE (76, SET_SHARD_DURABLE_REQ, SetDurableSerializers.shardDurable, EPOCH),
|
||||||
|
SET_GLOBALLY_DURABLE (77, SET_GLOBALLY_DURABLE_REQ, SetDurableSerializers.globallyDurable, EPOCH),
|
||||||
|
|
||||||
/* Accord local messages */
|
/* Accord local messages */
|
||||||
PROPAGATE_PRE_ACCEPT (78, PROPAGATE_PRE_ACCEPT_MSG, FetchSerializers.propagate, LOCAL),
|
PROPAGATE_PRE_ACCEPT (78, PROPAGATE_PRE_ACCEPT_MSG, FetchSerializers.propagate, LOCAL),
|
||||||
PROPAGATE_COMMIT (79, PROPAGATE_COMMIT_MSG, FetchSerializers.propagate, LOCAL),
|
PROPAGATE_COMMIT (79, PROPAGATE_COMMIT_MSG, FetchSerializers.propagate, LOCAL),
|
||||||
PROPAGATE_APPLY (80, PROPAGATE_APPLY_MSG, FetchSerializers.propagate, LOCAL),
|
PROPAGATE_APPLY (80, PROPAGATE_APPLY_MSG, FetchSerializers.propagate, LOCAL),
|
||||||
PROPAGATE_OTHER (81, PROPAGATE_OTHER_MSG, FetchSerializers.propagate, LOCAL),
|
PROPAGATE_OTHER (81, PROPAGATE_OTHER_MSG, FetchSerializers.propagate, LOCAL),
|
||||||
;
|
;
|
||||||
|
|
||||||
final int id;
|
final int id;
|
||||||
final MessageType type;
|
/**
|
||||||
|
* An incoming message of a given type from Accord's perspective might have multiple
|
||||||
|
* concrete implementations some of which are supplied by the Cassandra integration.
|
||||||
|
* The incoming type specifies the handling for writing out a message to the journal.
|
||||||
|
*/
|
||||||
|
final MessageType incomingType;
|
||||||
|
/**
|
||||||
|
* The outgoing type is the type that will be returned to Accord and it must be a subclass of the incoming type.
|
||||||
|
*
|
||||||
|
* This type will always be from accord.messages.MessageType and never from the extended types in the integration.
|
||||||
|
*/
|
||||||
|
final MessageType outgoingType;
|
||||||
final TxnIdProvider txnIdProvider;
|
final TxnIdProvider txnIdProvider;
|
||||||
final ValueSerializer<Key, Object> serializer;
|
final ValueSerializer<Key, Object> serializer;
|
||||||
|
|
||||||
Type(int id, ValueSerializer<Key, ? extends AuxiliaryRecord> serializer)
|
Type(int id, ValueSerializer<Key, ? extends AuxiliaryRecord> serializer)
|
||||||
{
|
{
|
||||||
this(id, null, serializer, null);
|
this(id, null, null, serializer, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
Type(int id, MessageType incomingType, MessageType outgoingType, IVersionedSerializer<?> serializer, TxnIdProvider txnIdProvider)
|
||||||
|
{
|
||||||
|
//noinspection unchecked
|
||||||
|
this(id, incomingType, outgoingType, MessageSerializer.wrap((IVersionedSerializer<Message>) serializer), txnIdProvider);
|
||||||
}
|
}
|
||||||
|
|
||||||
Type(int id, MessageType type, IVersionedSerializer<?> serializer, TxnIdProvider txnIdProvider)
|
Type(int id, MessageType type, IVersionedSerializer<?> serializer, TxnIdProvider txnIdProvider)
|
||||||
{
|
{
|
||||||
//noinspection unchecked
|
//noinspection unchecked
|
||||||
this(id, type, MessageSerializer.wrap((IVersionedSerializer<Message>) serializer), txnIdProvider);
|
this(id, type, type, MessageSerializer.wrap((IVersionedSerializer<Message>) serializer), txnIdProvider);
|
||||||
}
|
}
|
||||||
|
|
||||||
Type(int id, MessageType type, ValueSerializer<Key, ?> serializer, TxnIdProvider txnIdProvider)
|
Type(int id, MessageType incomingType, MessageType outgoingType, ValueSerializer<Key, ?> serializer, TxnIdProvider txnIdProvider)
|
||||||
{
|
{
|
||||||
if (id < 0)
|
if (id < 0)
|
||||||
throw new IllegalArgumentException("Negative Type id " + id);
|
throw new IllegalArgumentException("Negative Type id " + id);
|
||||||
|
|
@ -533,7 +583,8 @@ public class AccordJournal implements Shutdownable
|
||||||
throw new IllegalArgumentException("Type id doesn't fit in a single byte: " + id);
|
throw new IllegalArgumentException("Type id doesn't fit in a single byte: " + id);
|
||||||
|
|
||||||
this.id = id;
|
this.id = id;
|
||||||
this.type = type;
|
this.incomingType = incomingType;
|
||||||
|
this.outgoingType = outgoingType;
|
||||||
//noinspection unchecked
|
//noinspection unchecked
|
||||||
this.serializer = (ValueSerializer<Key, Object>) serializer;
|
this.serializer = (ValueSerializer<Key, Object>) serializer;
|
||||||
this.txnIdProvider = txnIdProvider;
|
this.txnIdProvider = txnIdProvider;
|
||||||
|
|
@ -542,6 +593,8 @@ public class AccordJournal implements Shutdownable
|
||||||
private static final Type[] idToTypeMapping;
|
private static final Type[] idToTypeMapping;
|
||||||
private static final Map<MessageType, Type> msgTypeToTypeMap;
|
private static final Map<MessageType, Type> msgTypeToTypeMap;
|
||||||
|
|
||||||
|
private static final ListMultimap<MessageType, Type> msgTypeToSynonymousTypesMap;
|
||||||
|
|
||||||
static
|
static
|
||||||
{
|
{
|
||||||
Type[] types = values();
|
Type[] types = values();
|
||||||
|
|
@ -559,13 +612,26 @@ public class AccordJournal implements Shutdownable
|
||||||
}
|
}
|
||||||
idToTypeMapping = idToType;
|
idToTypeMapping = idToType;
|
||||||
|
|
||||||
EnumMap<MessageType, Type> msgTypeToType = new EnumMap<>(MessageType.class);
|
Map<MessageType, Type> msgTypeToType = new HashMap<>();
|
||||||
for (Type type : types)
|
for (Type type : types)
|
||||||
{
|
{
|
||||||
if (null != type.type && null != msgTypeToType.put(type.type, type))
|
if (null != type.incomingType && null != msgTypeToType.put(type.incomingType, type))
|
||||||
throw new IllegalStateException("Duplicate MessageType " + type.type);
|
throw new IllegalStateException("Duplicate MessageType " + type.incomingType);
|
||||||
}
|
}
|
||||||
msgTypeToTypeMap = msgTypeToType;
|
msgTypeToTypeMap = ImmutableMap.copyOf(msgTypeToType);
|
||||||
|
|
||||||
|
Multimap<MessageType, Type> msgTypeToSynonymousTypes = ArrayListMultimap.create();
|
||||||
|
for (Type type : types)
|
||||||
|
{
|
||||||
|
if (null != type.outgoingType)
|
||||||
|
{
|
||||||
|
Type incomingType = msgTypeToTypeMap.get(type.incomingType);
|
||||||
|
if (msgTypeToSynonymousTypes.get(type.outgoingType).contains(incomingType))
|
||||||
|
throw new IllegalStateException("Duplicate synonymous Type " + type.incomingType);
|
||||||
|
msgTypeToSynonymousTypes.put(type.outgoingType, incomingType);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
msgTypeToSynonymousTypesMap = ImmutableListMultimap.copyOf(msgTypeToSynonymousTypes);
|
||||||
}
|
}
|
||||||
|
|
||||||
static Type fromId(int id)
|
static Type fromId(int id)
|
||||||
|
|
@ -578,6 +644,14 @@ public class AccordJournal implements Shutdownable
|
||||||
return type;
|
return type;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static List<Type> synonymousTypesFromMessageType(MessageType msgType)
|
||||||
|
{
|
||||||
|
List<Type> synonymousTypes = msgTypeToSynonymousTypesMap.get(msgType);
|
||||||
|
if (null == synonymousTypes)
|
||||||
|
throw new IllegalArgumentException("Unsupported MessageType " + msgType);
|
||||||
|
return synonymousTypes;
|
||||||
|
}
|
||||||
|
|
||||||
static Type fromMessageType(MessageType msgType)
|
static Type fromMessageType(MessageType msgType)
|
||||||
{
|
{
|
||||||
Type type = msgTypeToTypeMap.get(msgType);
|
Type type = msgTypeToTypeMap.get(msgType);
|
||||||
|
|
@ -613,7 +687,7 @@ public class AccordJournal implements Shutdownable
|
||||||
static
|
static
|
||||||
{
|
{
|
||||||
// make noise early if we forget to update our version mappings
|
// make noise early if we forget to update our version mappings
|
||||||
Invariants.checkState(MessagingService.current_version == MessagingService.VERSION_50, "Expected current version to be %d but given %d", MessagingService.VERSION_50, MessagingService.current_version);
|
Invariants.checkState(MessagingService.current_version == MessagingService.VERSION_51, "Expected current version to be %d but given %d", MessagingService.VERSION_51, MessagingService.current_version);
|
||||||
}
|
}
|
||||||
|
|
||||||
private static int msVersion(int version)
|
private static int msVersion(int version)
|
||||||
|
|
@ -621,7 +695,7 @@ public class AccordJournal implements Shutdownable
|
||||||
switch (version)
|
switch (version)
|
||||||
{
|
{
|
||||||
default: throw new IllegalArgumentException();
|
default: throw new IllegalArgumentException();
|
||||||
case 1: return MessagingService.VERSION_50;
|
case 1: return MessagingService.VERSION_51;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -693,11 +767,12 @@ public class AccordJournal implements Shutdownable
|
||||||
{
|
{
|
||||||
Set<Key> keys = new ObjectHashSet<>(messages.size() + 1, 0.9f);
|
Set<Key> keys = new ObjectHashSet<>(messages.size() + 1, 0.9f);
|
||||||
for (MessageType message : messages)
|
for (MessageType message : messages)
|
||||||
keys.add(new Key(txnId, Type.fromMessageType(message)));
|
for (Type synonymousType : Type.synonymousTypesFromMessageType(message))
|
||||||
|
keys.add(new Key(txnId, synonymousType));
|
||||||
Set<Key> presentKeys = journal.test(keys);
|
Set<Key> presentKeys = journal.test(keys);
|
||||||
EnumSet<MessageType> presentMessages = EnumSet.noneOf(MessageType.class);
|
Set<MessageType> presentMessages = new ObjectHashSet<>(presentKeys.size() + 1, 0.9f);
|
||||||
for (Key key : presentKeys)
|
for (Key key : presentKeys)
|
||||||
presentMessages.add(key.type.type);
|
presentMessages.add(key.type.outgoingType);
|
||||||
return presentMessages;
|
return presentMessages;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -150,8 +150,8 @@ import org.apache.cassandra.service.accord.serializers.TopologySerializers;
|
||||||
import org.apache.cassandra.service.accord.serializers.WaitingOnSerializer;
|
import org.apache.cassandra.service.accord.serializers.WaitingOnSerializer;
|
||||||
import org.apache.cassandra.transport.Dispatcher;
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
import org.apache.cassandra.utils.Clock;
|
import org.apache.cassandra.utils.Clock;
|
||||||
import org.apache.cassandra.utils.btree.BTree;
|
|
||||||
import org.apache.cassandra.utils.Throwables;
|
import org.apache.cassandra.utils.Throwables;
|
||||||
|
import org.apache.cassandra.utils.btree.BTree;
|
||||||
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
import org.apache.cassandra.utils.bytecomparable.ByteComparable;
|
||||||
|
|
||||||
import static accord.utils.Invariants.checkArgument;
|
import static accord.utils.Invariants.checkArgument;
|
||||||
|
|
@ -1753,7 +1753,7 @@ public class AccordKeyspace
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
return Iterables.getOnlyElement(statement.getMutations(clientState, options, true, tsMicros, (int) TimeUnit.MICROSECONDS.toSeconds(tsMicros), Dispatcher.RequestTime.forImmediateExecution()));
|
return Iterables.getOnlyElement(statement.getMutations(clientState, options, true, tsMicros, (int) TimeUnit.MICROSECONDS.toSeconds(tsMicros), Dispatcher.RequestTime.forImmediateExecution(), false));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -18,14 +18,18 @@
|
||||||
|
|
||||||
package org.apache.cassandra.service.accord;
|
package org.apache.cassandra.service.accord;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.lang.reflect.Modifier;
|
||||||
import java.util.Collections;
|
import java.util.Collections;
|
||||||
import java.util.EnumMap;
|
|
||||||
import java.util.EnumSet;
|
import java.util.EnumSet;
|
||||||
|
import java.util.List;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
|
|
||||||
import com.google.common.base.Preconditions;
|
import com.google.common.base.Preconditions;
|
||||||
|
import com.google.common.collect.ImmutableList;
|
||||||
import com.google.common.collect.ImmutableMap;
|
import com.google.common.collect.ImmutableMap;
|
||||||
|
import com.google.common.collect.Iterables;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
|
@ -42,61 +46,122 @@ import org.apache.cassandra.exceptions.RequestFailureReason;
|
||||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
import org.apache.cassandra.net.Message;
|
import org.apache.cassandra.net.Message;
|
||||||
import org.apache.cassandra.net.MessageDelivery;
|
import org.apache.cassandra.net.MessageDelivery;
|
||||||
|
import org.apache.cassandra.net.MessageFlag;
|
||||||
import org.apache.cassandra.net.MessagingService;
|
import org.apache.cassandra.net.MessagingService;
|
||||||
import org.apache.cassandra.net.Verb;
|
import org.apache.cassandra.net.Verb;
|
||||||
|
|
||||||
|
import static accord.messages.MessageType.Kind.REMOTE;
|
||||||
|
|
||||||
public class AccordMessageSink implements MessageSink
|
public class AccordMessageSink implements MessageSink
|
||||||
{
|
{
|
||||||
private static final Logger logger = LoggerFactory.getLogger(AccordMessageSink.class);
|
private static final Logger logger = LoggerFactory.getLogger(AccordMessageSink.class);
|
||||||
|
|
||||||
|
public static final class AccordMessageType extends MessageType
|
||||||
|
{
|
||||||
|
public static final MessageType INTEROP_READ_REQ = amt(REMOTE, false);
|
||||||
|
public static final MessageType INTEROP_READ_RSP = amt(REMOTE, false);
|
||||||
|
public static final MessageType INTEROP_READ_REPAIR_REQ = amt(REMOTE, false);
|
||||||
|
public static final MessageType INTEROP_READ_REPAIR_RSP = amt(REMOTE, false);
|
||||||
|
public static final MessageType INTEROP_COMMIT_MINIMAL_REQ = amt(REMOTE, true );
|
||||||
|
public static final MessageType INTEROP_COMMIT_MAXIMAL_REQ = amt(REMOTE, true );
|
||||||
|
public static final MessageType INTEROP_APPLY_MINIMAL_REQ = amt(REMOTE, true );
|
||||||
|
public static final MessageType INTEROP_APPLY_MAXIMAL_REQ = amt(REMOTE, true );
|
||||||
|
|
||||||
|
|
||||||
|
public static final List<MessageType> values;
|
||||||
|
|
||||||
|
static
|
||||||
|
{
|
||||||
|
ImmutableList.Builder<MessageType> builder = ImmutableList.builder();
|
||||||
|
for (Field f : AccordMessageType.class.getDeclaredFields())
|
||||||
|
{
|
||||||
|
if (f.getType().equals(AccordMessageType.class) && Modifier.isStatic(f.getModifiers()))
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
builder.add((MessageType) f.get(null));
|
||||||
|
}
|
||||||
|
catch (IllegalAccessException e)
|
||||||
|
{
|
||||||
|
throw new RuntimeException(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
values = builder.build();
|
||||||
|
}
|
||||||
|
|
||||||
|
private static MessageType amt(MessageType.Kind kind, boolean hasSideEffects)
|
||||||
|
{
|
||||||
|
return new AccordMessageType(kind, hasSideEffects);
|
||||||
|
}
|
||||||
|
|
||||||
|
private AccordMessageType(MessageType.Kind kind, boolean hasSideEffects)
|
||||||
|
{
|
||||||
|
super(kind, hasSideEffects);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
private static class VerbMapping
|
private static class VerbMapping
|
||||||
{
|
{
|
||||||
private static final VerbMapping instance = new VerbMapping();
|
private static final VerbMapping instance = new VerbMapping();
|
||||||
|
|
||||||
private final Map<MessageType, Verb> mapping = new EnumMap<>(MessageType.class);
|
private final Map<MessageType, Verb> mapping;
|
||||||
private final Map<Verb, Set<Verb>> overrideReplyVerbs = ImmutableMap.<Verb, Set<Verb>>builder()
|
private final Map<Verb, Set<Verb>> overrideReplyVerbs = ImmutableMap.<Verb, Set<Verb>>builder()
|
||||||
// read takes Result | Nack
|
// read takes Result | Nack
|
||||||
.put(Verb.ACCORD_FETCH_DATA_REQ, EnumSet.of(Verb.ACCORD_FETCH_DATA_RSP, Verb.ACCORD_READ_RSP /* nack */))
|
.put(Verb.ACCORD_FETCH_DATA_REQ, EnumSet.of(Verb.ACCORD_FETCH_DATA_RSP, Verb.ACCORD_READ_RSP /* nack */))
|
||||||
.build();
|
.put(Verb.ACCORD_INTEROP_COMMIT_REQ, EnumSet.of(Verb.ACCORD_INTEROP_READ_RSP, Verb.ACCORD_READ_RSP))
|
||||||
|
.put(Verb.ACCORD_INTEROP_READ_REPAIR_REQ, EnumSet.of(Verb.ACCORD_INTEROP_READ_REPAIR_RSP, Verb.ACCORD_READ_RSP))
|
||||||
|
.build();
|
||||||
|
|
||||||
private VerbMapping()
|
private VerbMapping()
|
||||||
{
|
{
|
||||||
mapping.put(MessageType.PRE_ACCEPT_REQ, Verb.ACCORD_PRE_ACCEPT_REQ);
|
ImmutableMap.Builder<MessageType, Verb> builder = ImmutableMap.builder();
|
||||||
mapping.put(MessageType.PRE_ACCEPT_RSP, Verb.ACCORD_PRE_ACCEPT_RSP);
|
builder.put(MessageType.SIMPLE_RSP, Verb.ACCORD_SIMPLE_RSP);
|
||||||
mapping.put(MessageType.ACCEPT_REQ, Verb.ACCORD_ACCEPT_REQ);
|
builder.put(MessageType.PRE_ACCEPT_REQ, Verb.ACCORD_PRE_ACCEPT_REQ);
|
||||||
mapping.put(MessageType.ACCEPT_RSP, Verb.ACCORD_ACCEPT_RSP);
|
builder.put(MessageType.PRE_ACCEPT_RSP, Verb.ACCORD_PRE_ACCEPT_RSP);
|
||||||
mapping.put(MessageType.ACCEPT_INVALIDATE_REQ, Verb.ACCORD_ACCEPT_INVALIDATE_REQ);
|
builder.put(MessageType.ACCEPT_REQ, Verb.ACCORD_ACCEPT_REQ);
|
||||||
mapping.put(MessageType.COMMIT_MINIMAL_REQ, Verb.ACCORD_COMMIT_REQ);
|
builder.put(MessageType.ACCEPT_RSP, Verb.ACCORD_ACCEPT_RSP);
|
||||||
mapping.put(MessageType.COMMIT_MAXIMAL_REQ, Verb.ACCORD_COMMIT_REQ);
|
builder.put(MessageType.ACCEPT_INVALIDATE_REQ, Verb.ACCORD_ACCEPT_INVALIDATE_REQ);
|
||||||
mapping.put(MessageType.COMMIT_INVALIDATE_REQ, Verb.ACCORD_COMMIT_INVALIDATE_REQ);
|
builder.put(MessageType.GET_DEPS_REQ, Verb.ACCORD_GET_DEPS_REQ);
|
||||||
mapping.put(MessageType.APPLY_MINIMAL_REQ, Verb.ACCORD_APPLY_REQ);
|
builder.put(MessageType.GET_DEPS_RSP, Verb.ACCORD_GET_DEPS_RSP);
|
||||||
mapping.put(MessageType.APPLY_MAXIMAL_REQ, Verb.ACCORD_APPLY_REQ);
|
builder.put(MessageType.COMMIT_MINIMAL_REQ, Verb.ACCORD_COMMIT_REQ);
|
||||||
mapping.put(MessageType.APPLY_RSP, Verb.ACCORD_APPLY_RSP);
|
builder.put(MessageType.COMMIT_MAXIMAL_REQ, Verb.ACCORD_COMMIT_REQ);
|
||||||
mapping.put(MessageType.READ_REQ, Verb.ACCORD_READ_REQ);
|
builder.put(MessageType.COMMIT_INVALIDATE_REQ, Verb.ACCORD_COMMIT_INVALIDATE_REQ);
|
||||||
mapping.put(MessageType.READ_RSP, Verb.ACCORD_READ_RSP);
|
builder.put(MessageType.APPLY_MINIMAL_REQ, Verb.ACCORD_APPLY_REQ);
|
||||||
mapping.put(MessageType.BEGIN_RECOVER_REQ, Verb.ACCORD_BEGIN_RECOVER_REQ);
|
builder.put(MessageType.APPLY_MAXIMAL_REQ, Verb.ACCORD_APPLY_REQ);
|
||||||
mapping.put(MessageType.BEGIN_RECOVER_RSP, Verb.ACCORD_BEGIN_RECOVER_RSP);
|
builder.put(MessageType.APPLY_RSP, Verb.ACCORD_APPLY_RSP);
|
||||||
mapping.put(MessageType.BEGIN_INVALIDATE_REQ, Verb.ACCORD_BEGIN_INVALIDATE_REQ);
|
builder.put(MessageType.READ_REQ, Verb.ACCORD_READ_REQ);
|
||||||
mapping.put(MessageType.BEGIN_INVALIDATE_RSP, Verb.ACCORD_BEGIN_INVALIDATE_RSP);
|
builder.put(MessageType.READ_RSP, Verb.ACCORD_READ_RSP);
|
||||||
mapping.put(MessageType.WAIT_ON_COMMIT_REQ, Verb.ACCORD_WAIT_ON_COMMIT_REQ);
|
builder.put(MessageType.BEGIN_RECOVER_REQ, Verb.ACCORD_BEGIN_RECOVER_REQ);
|
||||||
mapping.put(MessageType.WAIT_ON_COMMIT_RSP, Verb.ACCORD_WAIT_ON_COMMIT_RSP);
|
builder.put(MessageType.BEGIN_RECOVER_RSP, Verb.ACCORD_BEGIN_RECOVER_RSP);
|
||||||
mapping.put(MessageType.WAIT_ON_APPLY_REQ, Verb.ACCORD_WAIT_ON_APPLY_REQ);
|
builder.put(MessageType.BEGIN_INVALIDATE_REQ, Verb.ACCORD_BEGIN_INVALIDATE_REQ);
|
||||||
mapping.put(MessageType.INFORM_OF_TXN_REQ, Verb.ACCORD_INFORM_OF_TXN_REQ);
|
builder.put(MessageType.BEGIN_INVALIDATE_RSP, Verb.ACCORD_BEGIN_INVALIDATE_RSP);
|
||||||
mapping.put(MessageType.INFORM_DURABLE_REQ, Verb.ACCORD_INFORM_DURABLE_REQ);
|
builder.put(MessageType.WAIT_ON_COMMIT_REQ, Verb.ACCORD_WAIT_ON_COMMIT_REQ);
|
||||||
mapping.put(MessageType.INFORM_HOME_DURABLE_REQ, Verb.ACCORD_INFORM_HOME_DURABLE_REQ);
|
builder.put(MessageType.WAIT_ON_COMMIT_RSP, Verb.ACCORD_WAIT_ON_COMMIT_RSP);
|
||||||
mapping.put(MessageType.CHECK_STATUS_REQ, Verb.ACCORD_CHECK_STATUS_REQ);
|
builder.put(MessageType.WAIT_UNTIL_APPLIED_REQ, Verb.ACCORD_WAIT_UNTIL_APPLIED_REQ);
|
||||||
mapping.put(MessageType.CHECK_STATUS_RSP, Verb.ACCORD_CHECK_STATUS_RSP);
|
builder.put(MessageType.APPLY_AND_WAIT_UNTIL_APPLIED_REQ, Verb.ACCORD_APPLY_AND_WAIT_UNTIL_APPLIED_REQ);
|
||||||
mapping.put(MessageType.GET_DEPS_REQ, Verb.ACCORD_GET_DEPS_REQ);
|
builder.put(MessageType.INFORM_OF_TXN_REQ, Verb.ACCORD_INFORM_OF_TXN_REQ);
|
||||||
mapping.put(MessageType.GET_DEPS_RSP, Verb.ACCORD_GET_DEPS_RSP);
|
builder.put(MessageType.INFORM_DURABLE_REQ, Verb.ACCORD_INFORM_DURABLE_REQ);
|
||||||
mapping.put(MessageType.SIMPLE_RSP, Verb.ACCORD_SIMPLE_RSP);
|
builder.put(MessageType.INFORM_HOME_DURABLE_REQ, Verb.ACCORD_INFORM_HOME_DURABLE_REQ);
|
||||||
mapping.put(MessageType.FETCH_DATA_REQ, Verb.ACCORD_FETCH_DATA_REQ);
|
builder.put(MessageType.CHECK_STATUS_REQ, Verb.ACCORD_CHECK_STATUS_REQ);
|
||||||
mapping.put(MessageType.FETCH_DATA_RSP, Verb.ACCORD_FETCH_DATA_RSP);
|
builder.put(MessageType.CHECK_STATUS_RSP, Verb.ACCORD_CHECK_STATUS_RSP);
|
||||||
mapping.put(MessageType.SET_SHARD_DURABLE_REQ, Verb.ACCORD_SET_SHARD_DURABLE_REQ);
|
builder.put(MessageType.FETCH_DATA_REQ, Verb.ACCORD_FETCH_DATA_REQ);
|
||||||
mapping.put(MessageType.SET_GLOBALLY_DURABLE_REQ, Verb.ACCORD_SET_GLOBALLY_DURABLE_REQ);
|
builder.put(MessageType.FETCH_DATA_RSP, Verb.ACCORD_FETCH_DATA_RSP);
|
||||||
mapping.put(MessageType.QUERY_DURABLE_BEFORE_REQ, Verb.ACCORD_QUERY_DURABLE_BEFORE_REQ);
|
builder.put(MessageType.SET_SHARD_DURABLE_REQ, Verb.ACCORD_SET_SHARD_DURABLE_REQ);
|
||||||
mapping.put(MessageType.QUERY_DURABLE_BEFORE_RSP, Verb.ACCORD_QUERY_DURABLE_BEFORE_RSP);
|
builder.put(MessageType.SET_GLOBALLY_DURABLE_REQ, Verb.ACCORD_SET_GLOBALLY_DURABLE_REQ);
|
||||||
|
builder.put(MessageType.QUERY_DURABLE_BEFORE_REQ, Verb.ACCORD_QUERY_DURABLE_BEFORE_REQ);
|
||||||
|
builder.put(MessageType.QUERY_DURABLE_BEFORE_RSP, Verb.ACCORD_QUERY_DURABLE_BEFORE_RSP);
|
||||||
|
builder.put(AccordMessageType.INTEROP_READ_REQ, Verb.ACCORD_INTEROP_READ_REQ);
|
||||||
|
builder.put(AccordMessageType.INTEROP_READ_RSP, Verb.ACCORD_INTEROP_READ_RSP);
|
||||||
|
builder.put(AccordMessageType.INTEROP_READ_REPAIR_REQ, Verb.ACCORD_INTEROP_READ_REPAIR_REQ);
|
||||||
|
builder.put(AccordMessageType.INTEROP_READ_REPAIR_RSP, Verb.ACCORD_INTEROP_READ_REPAIR_RSP);
|
||||||
|
builder.put(AccordMessageType.INTEROP_COMMIT_MINIMAL_REQ, Verb.ACCORD_INTEROP_COMMIT_REQ);
|
||||||
|
builder.put(AccordMessageType.INTEROP_COMMIT_MAXIMAL_REQ, Verb.ACCORD_INTEROP_COMMIT_REQ);
|
||||||
|
builder.put(AccordMessageType.INTEROP_APPLY_MINIMAL_REQ, Verb.ACCORD_APPLY_REQ);
|
||||||
|
builder.put(AccordMessageType.INTEROP_APPLY_MAXIMAL_REQ, Verb.ACCORD_APPLY_REQ);
|
||||||
|
mapping = builder.build();
|
||||||
|
|
||||||
for (MessageType type : MessageType.values())
|
for (MessageType type : Iterables.concat(AccordMessageType.values, MessageType.values))
|
||||||
{
|
{
|
||||||
// Any request can receive a generic failure response
|
// Any request can receive a generic failure response
|
||||||
if (type == MessageType.FAILURE_RSP)
|
if (type == MessageType.FAILURE_RSP)
|
||||||
|
|
@ -119,6 +184,15 @@ public class AccordMessageSink implements MessageSink
|
||||||
return VerbMapping.instance.mapping.get(type);
|
return VerbMapping.instance.mapping.get(type);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static Verb getVerb(Request request)
|
||||||
|
{
|
||||||
|
MessageType type = request.type();
|
||||||
|
if (type != null)
|
||||||
|
return getVerb(request.type());
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
private final Agent agent;
|
private final Agent agent;
|
||||||
private final MessageDelivery messaging;
|
private final MessageDelivery messaging;
|
||||||
private final AccordEndpointMapper endpointMapper;
|
private final AccordEndpointMapper endpointMapper;
|
||||||
|
|
@ -138,7 +212,7 @@ public class AccordMessageSink implements MessageSink
|
||||||
@Override
|
@Override
|
||||||
public void send(Node.Id to, Request request)
|
public void send(Node.Id to, Request request)
|
||||||
{
|
{
|
||||||
Verb verb = getVerb(request.type());
|
Verb verb = getVerb(request);
|
||||||
Preconditions.checkNotNull(verb, "Verb is null for type %s", request.type());
|
Preconditions.checkNotNull(verb, "Verb is null for type %s", request.type());
|
||||||
Message<Request> message = Message.out(verb, request);
|
Message<Request> message = Message.out(verb, request);
|
||||||
InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(to);
|
InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(to);
|
||||||
|
|
@ -149,7 +223,7 @@ public class AccordMessageSink implements MessageSink
|
||||||
@Override
|
@Override
|
||||||
public void send(Node.Id to, Request request, AgentExecutor executor, Callback callback)
|
public void send(Node.Id to, Request request, AgentExecutor executor, Callback callback)
|
||||||
{
|
{
|
||||||
Verb verb = getVerb(request.type());
|
Verb verb = getVerb(request);
|
||||||
Preconditions.checkNotNull(verb, "Verb is null for type %s", request.type());
|
Preconditions.checkNotNull(verb, "Verb is null for type %s", request.type());
|
||||||
Message<Request> message = Message.out(verb, request);
|
Message<Request> message = Message.out(verb, request);
|
||||||
InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(to);
|
InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(to);
|
||||||
|
|
@ -162,6 +236,8 @@ public class AccordMessageSink implements MessageSink
|
||||||
{
|
{
|
||||||
Message<?> replyTo = (Message<?>) replyContext;
|
Message<?> replyTo = (Message<?>) replyContext;
|
||||||
Message<?> replyMsg = replyTo.responseWith(reply);
|
Message<?> replyMsg = replyTo.responseWith(reply);
|
||||||
|
if (!reply.isFinal())
|
||||||
|
replyMsg = replyMsg.withFlag(MessageFlag.NOT_FINAL);
|
||||||
checkReplyType(reply, replyTo);
|
checkReplyType(reply, replyTo);
|
||||||
InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(replyingToNode);
|
InetAddressAndPort endpoint = endpointMapper.mappedEndpoint(replyingToNode);
|
||||||
logger.debug("Replying {} {} to {}", replyMsg.verb(), replyMsg.payload, endpoint);
|
logger.debug("Replying {} {} to {}", replyMsg.verb(), replyMsg.payload, endpoint);
|
||||||
|
|
|
||||||
|
|
@ -60,10 +60,10 @@ import org.apache.cassandra.service.accord.api.AccordRoutingKey;
|
||||||
import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey;
|
import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey;
|
||||||
import org.apache.cassandra.service.accord.api.PartitionKey;
|
import org.apache.cassandra.service.accord.api.PartitionKey;
|
||||||
import org.apache.cassandra.service.accord.serializers.WaitingOnSerializer;
|
import org.apache.cassandra.service.accord.serializers.WaitingOnSerializer;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnData;
|
import org.apache.cassandra.service.accord.txn.AccordUpdate;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnRead;
|
import org.apache.cassandra.service.accord.txn.TxnRead;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnUpdate;
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
import org.apache.cassandra.service.accord.txn.TxnWrite;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
import org.apache.cassandra.utils.ObjectSizes;
|
import org.apache.cassandra.utils.ObjectSizes;
|
||||||
|
|
@ -197,7 +197,7 @@ public class AccordObjectSizes
|
||||||
size += seekables(txn.keys());
|
size += seekables(txn.keys());
|
||||||
size += ((TxnRead) txn.read()).estimatedSizeOnHeap();
|
size += ((TxnRead) txn.read()).estimatedSizeOnHeap();
|
||||||
if (txn.update() != null)
|
if (txn.update() != null)
|
||||||
size += ((TxnUpdate) txn.update()).estimatedSizeOnHeap();
|
size += ((AccordUpdate) txn.update()).estimatedSizeOnHeap();
|
||||||
if (txn.query() != null)
|
if (txn.query() != null)
|
||||||
size += ((TxnQuery) txn.query()).estimatedSizeOnHeap();
|
size += ((TxnQuery) txn.query()).estimatedSizeOnHeap();
|
||||||
return size;
|
return size;
|
||||||
|
|
@ -250,7 +250,7 @@ public class AccordObjectSizes
|
||||||
|
|
||||||
public static long results(Result result)
|
public static long results(Result result)
|
||||||
{
|
{
|
||||||
return ((TxnData) result).estimatedSizeOnHeap();
|
return ((TxnResult) result).estimatedSizeOnHeap();
|
||||||
}
|
}
|
||||||
|
|
||||||
private static final long EMPTY_COMMAND_LISTENER = measure(new Command.ProxyListener(null));
|
private static final long EMPTY_COMMAND_LISTENER = measure(new Command.ProxyListener(null));
|
||||||
|
|
@ -331,7 +331,7 @@ public class AccordObjectSizes
|
||||||
size += sizeNullable(command.accepted(), AccordObjectSizes::timestamp);
|
size += sizeNullable(command.accepted(), AccordObjectSizes::timestamp);
|
||||||
size += sizeNullable(command.writes(), AccordObjectSizes::writes);
|
size += sizeNullable(command.writes(), AccordObjectSizes::writes);
|
||||||
|
|
||||||
if (command.result() instanceof TxnData)
|
if (command.result() instanceof TxnResult)
|
||||||
size += sizeNullable(command.result(), AccordObjectSizes::results);
|
size += sizeNullable(command.result(), AccordObjectSizes::results);
|
||||||
|
|
||||||
if (!(command instanceof Command.Committed))
|
if (!(command instanceof Command.Committed))
|
||||||
|
|
|
||||||
|
|
@ -18,12 +18,13 @@
|
||||||
|
|
||||||
package org.apache.cassandra.service.accord;
|
package org.apache.cassandra.service.accord;
|
||||||
|
|
||||||
import java.util.Collection;
|
|
||||||
import java.util.Objects;
|
import java.util.Objects;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
|
|
||||||
import accord.local.Command;
|
import accord.local.Command;
|
||||||
|
import accord.local.Command.TransientListener;
|
||||||
|
import accord.local.Listeners;
|
||||||
import accord.local.SafeCommand;
|
import accord.local.SafeCommand;
|
||||||
import accord.primitives.TxnId;
|
import accord.primitives.TxnId;
|
||||||
|
|
||||||
|
|
@ -138,7 +139,7 @@ public class AccordSafeCommand extends SafeCommand implements AccordSafeState<Tx
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Collection<Command.TransientListener> transientListeners()
|
public Listeners<TransientListener> transientListeners()
|
||||||
{
|
{
|
||||||
checkNotInvalidated();
|
checkNotInvalidated();
|
||||||
return global.listeners();
|
return global.listeners();
|
||||||
|
|
|
||||||
|
|
@ -21,8 +21,11 @@ package org.apache.cassandra.service.accord;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.NavigableMap;
|
import java.util.NavigableMap;
|
||||||
import java.util.function.BiFunction;
|
import java.util.function.BiFunction;
|
||||||
|
import java.util.function.Predicate;
|
||||||
import javax.annotation.Nullable;
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
|
import com.google.common.base.Predicates;
|
||||||
|
|
||||||
import accord.api.Agent;
|
import accord.api.Agent;
|
||||||
import accord.api.DataStore;
|
import accord.api.DataStore;
|
||||||
import accord.api.Key;
|
import accord.api.Key;
|
||||||
|
|
@ -156,7 +159,7 @@ public class AccordSafeCommandStore extends AbstractSafeCommandStore<AccordSafeC
|
||||||
@Override
|
@Override
|
||||||
public Timestamp maxConflict(Seekables<?, ?> keysOrRanges, Ranges slice)
|
public Timestamp maxConflict(Seekables<?, ?> keysOrRanges, Ranges slice)
|
||||||
{
|
{
|
||||||
Timestamp maxConflict = mapReduce(keysOrRanges, slice, (ts, accum) -> Timestamp.max(ts.max(), accum), Timestamp.NONE, null);
|
Timestamp maxConflict = mapReduce(keysOrRanges, slice, (ts, accum) -> Timestamp.max(ts.max(), accum), Timestamp.NONE, Predicates.isNull());
|
||||||
return Timestamp.nonNullOrMax(maxConflict, commandStore.commandsForRanges().maxRedundant());
|
return Timestamp.nonNullOrMax(maxConflict, commandStore.commandsForRanges().maxRedundant());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -198,15 +201,15 @@ public class AccordSafeCommandStore extends AbstractSafeCommandStore<AccordSafeC
|
||||||
{
|
{
|
||||||
}
|
}
|
||||||
|
|
||||||
private <O> O mapReduce(Routables<?> keysOrRanges, Ranges slice, BiFunction<CommandTimeseriesHolder, O, O> map, O accumulate, O terminalValue)
|
private <O> O mapReduce(Routables<?> keysOrRanges, Ranges slice, BiFunction<CommandTimeseriesHolder, O, O> map, O accumulate, Predicate<O> terminate)
|
||||||
{
|
{
|
||||||
accumulate = commandStore.mapReduceForRange(keysOrRanges, slice, map, accumulate, terminalValue);
|
accumulate = commandStore.mapReduceForRange(keysOrRanges, slice, map, accumulate, terminate);
|
||||||
if (accumulate.equals(terminalValue))
|
if (terminate.test(accumulate))
|
||||||
return accumulate;
|
return accumulate;
|
||||||
return mapReduceForKey(keysOrRanges, slice, map, accumulate, terminalValue);
|
return mapReduceForKey(keysOrRanges, slice, map, accumulate, terminate);
|
||||||
}
|
}
|
||||||
|
|
||||||
private <O> O mapReduceForKey(Routables<?> keysOrRanges, Ranges slice, BiFunction<CommandTimeseriesHolder, O, O> map, O accumulate, O terminalValue)
|
private <O> O mapReduceForKey(Routables<?> keysOrRanges, Ranges slice, BiFunction<CommandTimeseriesHolder, O, O> map, O accumulate, Predicate<O> terminate)
|
||||||
{
|
{
|
||||||
switch (keysOrRanges.domain())
|
switch (keysOrRanges.domain())
|
||||||
{
|
{
|
||||||
|
|
@ -221,7 +224,7 @@ public class AccordSafeCommandStore extends AbstractSafeCommandStore<AccordSafeC
|
||||||
if (!slice.contains(key)) continue;
|
if (!slice.contains(key)) continue;
|
||||||
SafeCommandsForKey forKey = commandsForKey(key);
|
SafeCommandsForKey forKey = commandsForKey(key);
|
||||||
accumulate = map.apply(forKey.current(), accumulate);
|
accumulate = map.apply(forKey.current(), accumulate);
|
||||||
if (accumulate.equals(terminalValue))
|
if (terminate.test((accumulate)))
|
||||||
return accumulate;
|
return accumulate;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -239,7 +242,7 @@ public class AccordSafeCommandStore extends AbstractSafeCommandStore<AccordSafeC
|
||||||
if (!sliced.contains(key)) continue;
|
if (!sliced.contains(key)) continue;
|
||||||
SafeCommandsForKey forKey = commandsForKey(key);
|
SafeCommandsForKey forKey = commandsForKey(key);
|
||||||
accumulate = map.apply(forKey.current(), accumulate);
|
accumulate = map.apply(forKey.current(), accumulate);
|
||||||
if (accumulate.equals(terminalValue))
|
if (terminate.test(accumulate))
|
||||||
return accumulate;
|
return accumulate;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -251,6 +254,12 @@ public class AccordSafeCommandStore extends AbstractSafeCommandStore<AccordSafeC
|
||||||
@Override
|
@Override
|
||||||
public <T> T mapReduce(Seekables<?, ?> keysOrRanges, Ranges slice, TestKind testKind, TestTimestamp testTimestamp, Timestamp timestamp, TestDep testDep, @Nullable TxnId depId, @Nullable Status minStatus, @Nullable Status maxStatus, CommandFunction<T, T> map, T accumulate, T terminalValue)
|
public <T> T mapReduce(Seekables<?, ?> keysOrRanges, Ranges slice, TestKind testKind, TestTimestamp testTimestamp, Timestamp timestamp, TestDep testDep, @Nullable TxnId depId, @Nullable Status minStatus, @Nullable Status maxStatus, CommandFunction<T, T> map, T accumulate, T terminalValue)
|
||||||
{
|
{
|
||||||
|
Predicate<T> terminate = Predicates.equalTo(terminalValue);
|
||||||
|
return mapReduceWithTerminate(keysOrRanges, slice, testKind, testTimestamp, timestamp, testDep, depId, minStatus, maxStatus, map, accumulate, terminate);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public <T> T mapReduceWithTerminate(Seekables<?, ?> keysOrRanges, Ranges slice, TestKind testKind, TestTimestamp testTimestamp, Timestamp timestamp, TestDep testDep, @Nullable TxnId depId, @Nullable Status minStatus, @Nullable Status maxStatus, CommandFunction<T, T> map, T accumulate, Predicate<T> terminate) {
|
||||||
accumulate = mapReduce(keysOrRanges, slice, (forKey, prev) -> {
|
accumulate = mapReduce(keysOrRanges, slice, (forKey, prev) -> {
|
||||||
CommandTimeseries<?> timeseries;
|
CommandTimeseries<?> timeseries;
|
||||||
switch (testTimestamp)
|
switch (testTimestamp)
|
||||||
|
|
@ -276,8 +285,8 @@ public class AccordSafeCommandStore extends AbstractSafeCommandStore<AccordSafeC
|
||||||
case MAY_EXECUTE_BEFORE:
|
case MAY_EXECUTE_BEFORE:
|
||||||
remapTestTimestamp = CommandTimeseries.TestTimestamp.BEFORE;
|
remapTestTimestamp = CommandTimeseries.TestTimestamp.BEFORE;
|
||||||
}
|
}
|
||||||
return timeseries.mapReduce(testKind, remapTestTimestamp, timestamp, testDep, depId, minStatus, maxStatus, map, prev, terminalValue);
|
return timeseries.mapReduceWithTerminate(testKind, remapTestTimestamp, timestamp, testDep, depId, minStatus, maxStatus, map, prev, terminate);
|
||||||
}, accumulate, terminalValue);
|
}, accumulate, terminate);
|
||||||
|
|
||||||
return accumulate;
|
return accumulate;
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -27,6 +27,7 @@ import org.apache.cassandra.cql3.terms.Term;
|
||||||
import org.apache.cassandra.db.ArrayClustering;
|
import org.apache.cassandra.db.ArrayClustering;
|
||||||
import org.apache.cassandra.db.Clustering;
|
import org.apache.cassandra.db.Clustering;
|
||||||
import org.apache.cassandra.db.ClusteringPrefix;
|
import org.apache.cassandra.db.ClusteringPrefix;
|
||||||
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
import org.apache.cassandra.db.marshal.AbstractType;
|
import org.apache.cassandra.db.marshal.AbstractType;
|
||||||
import org.apache.cassandra.db.marshal.CollectionType;
|
import org.apache.cassandra.db.marshal.CollectionType;
|
||||||
import org.apache.cassandra.db.marshal.ListType;
|
import org.apache.cassandra.db.marshal.ListType;
|
||||||
|
|
@ -46,6 +47,7 @@ import org.apache.cassandra.schema.Schema;
|
||||||
import org.apache.cassandra.schema.TableId;
|
import org.apache.cassandra.schema.TableId;
|
||||||
import org.apache.cassandra.schema.TableMetadata;
|
import org.apache.cassandra.schema.TableMetadata;
|
||||||
import org.apache.cassandra.utils.ByteBufferUtil;
|
import org.apache.cassandra.utils.ByteBufferUtil;
|
||||||
|
import org.apache.cassandra.utils.NullableSerializer;
|
||||||
|
|
||||||
import static org.apache.cassandra.db.TypeSizes.sizeof;
|
import static org.apache.cassandra.db.TypeSizes.sizeof;
|
||||||
import static org.apache.cassandra.db.TypeSizes.sizeofUnsignedVInt;
|
import static org.apache.cassandra.db.TypeSizes.sizeofUnsignedVInt;
|
||||||
|
|
@ -127,6 +129,8 @@ public class AccordSerializers
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
public static final IVersionedSerializer<PartitionUpdate> nullablePartitionUpdateSerializer = NullableSerializer.wrap(partitionUpdateSerializer);
|
||||||
|
|
||||||
public static final IVersionedSerializer<ColumnMetadata> columnMetadataSerializer = new IVersionedSerializer<ColumnMetadata>()
|
public static final IVersionedSerializer<ColumnMetadata> columnMetadataSerializer = new IVersionedSerializer<ColumnMetadata>()
|
||||||
{
|
{
|
||||||
@Override
|
@Override
|
||||||
|
|
@ -246,4 +250,25 @@ public class AccordSerializers
|
||||||
return size;
|
return size;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
public static final IVersionedSerializer<ConsistencyLevel> consistencyLevelSerializer = new IVersionedSerializer<ConsistencyLevel>()
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
public void serialize(ConsistencyLevel t, DataOutputPlus out, int version) throws IOException
|
||||||
|
{
|
||||||
|
out.writeByte(t.code);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ConsistencyLevel deserialize(DataInputPlus in, int version) throws IOException
|
||||||
|
{
|
||||||
|
return ConsistencyLevel.fromCode(in.readByte());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long serializedSize(ConsistencyLevel t, int version)
|
||||||
|
{
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
@ -24,14 +24,17 @@ import java.util.concurrent.ExecutionException;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.concurrent.TimeoutException;
|
import java.util.concurrent.TimeoutException;
|
||||||
import java.util.concurrent.atomic.AtomicReference;
|
import java.util.concurrent.atomic.AtomicReference;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
import com.google.common.annotations.VisibleForTesting;
|
import com.google.common.annotations.VisibleForTesting;
|
||||||
import com.google.common.base.Preconditions;
|
import com.google.common.base.Preconditions;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import accord.api.BarrierType;
|
||||||
import accord.api.Result;
|
import accord.api.Result;
|
||||||
import accord.config.LocalConfig;
|
import accord.config.LocalConfig;
|
||||||
|
import accord.coordinate.CoordinationFailed;
|
||||||
import accord.coordinate.Preempted;
|
import accord.coordinate.Preempted;
|
||||||
import accord.coordinate.Timeout;
|
import accord.coordinate.Timeout;
|
||||||
import accord.impl.AbstractConfigurationService;
|
import accord.impl.AbstractConfigurationService;
|
||||||
|
|
@ -39,12 +42,16 @@ import accord.impl.SimpleProgressLog;
|
||||||
import accord.impl.SizeOfIntersectionSorter;
|
import accord.impl.SizeOfIntersectionSorter;
|
||||||
import accord.local.DurableBefore;
|
import accord.local.DurableBefore;
|
||||||
import accord.local.Node;
|
import accord.local.Node;
|
||||||
|
import accord.local.Node.Id;
|
||||||
import accord.local.NodeTimeService;
|
import accord.local.NodeTimeService;
|
||||||
import accord.local.RedundantBefore;
|
import accord.local.RedundantBefore;
|
||||||
import accord.local.ShardDistributor.EvenSplit;
|
import accord.local.ShardDistributor.EvenSplit;
|
||||||
import accord.messages.LocalMessage;
|
import accord.messages.LocalMessage;
|
||||||
import accord.messages.Request;
|
import accord.messages.Request;
|
||||||
|
import accord.primitives.Seekables;
|
||||||
|
import accord.primitives.Timestamp;
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
|
import accord.primitives.Txn.Kind;
|
||||||
import accord.primitives.TxnId;
|
import accord.primitives.TxnId;
|
||||||
import accord.topology.TopologyManager;
|
import accord.topology.TopologyManager;
|
||||||
import accord.utils.DefaultRandom;
|
import accord.utils.DefaultRandom;
|
||||||
|
|
@ -59,7 +66,9 @@ import org.apache.cassandra.concurrent.Shutdownable;
|
||||||
import org.apache.cassandra.config.DatabaseDescriptor;
|
import org.apache.cassandra.config.DatabaseDescriptor;
|
||||||
import org.apache.cassandra.db.ConsistencyLevel;
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
import org.apache.cassandra.db.WriteType;
|
import org.apache.cassandra.db.WriteType;
|
||||||
|
import org.apache.cassandra.exceptions.ExceptionCode;
|
||||||
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
import org.apache.cassandra.exceptions.ReadTimeoutException;
|
||||||
|
import org.apache.cassandra.exceptions.RequestTimeoutException;
|
||||||
import org.apache.cassandra.exceptions.WriteTimeoutException;
|
import org.apache.cassandra.exceptions.WriteTimeoutException;
|
||||||
import org.apache.cassandra.journal.AsyncWriteCallback;
|
import org.apache.cassandra.journal.AsyncWriteCallback;
|
||||||
import org.apache.cassandra.metrics.AccordClientRequestMetrics;
|
import org.apache.cassandra.metrics.AccordClientRequestMetrics;
|
||||||
|
|
@ -75,29 +84,38 @@ import org.apache.cassandra.service.accord.api.AccordTopologySorter;
|
||||||
import org.apache.cassandra.service.accord.api.CompositeTopologySorter;
|
import org.apache.cassandra.service.accord.api.CompositeTopologySorter;
|
||||||
import org.apache.cassandra.service.accord.exceptions.ReadPreemptedException;
|
import org.apache.cassandra.service.accord.exceptions.ReadPreemptedException;
|
||||||
import org.apache.cassandra.service.accord.exceptions.WritePreemptedException;
|
import org.apache.cassandra.service.accord.exceptions.WritePreemptedException;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnData;
|
import org.apache.cassandra.service.accord.interop.AccordInteropApply;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropExecution;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropPersist;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadataService;
|
import org.apache.cassandra.tcm.ClusterMetadataService;
|
||||||
import org.apache.cassandra.tcm.Epoch;
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
import org.apache.cassandra.tcm.membership.NodeId;
|
import org.apache.cassandra.tcm.membership.NodeId;
|
||||||
|
import org.apache.cassandra.tcm.transformations.AddAccordKeyspace;
|
||||||
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
import org.apache.cassandra.utils.Clock;
|
import org.apache.cassandra.utils.Clock;
|
||||||
import org.apache.cassandra.utils.ExecutorUtils;
|
import org.apache.cassandra.utils.ExecutorUtils;
|
||||||
|
import org.apache.cassandra.utils.FBUtilities;
|
||||||
import org.apache.cassandra.utils.Pair;
|
import org.apache.cassandra.utils.Pair;
|
||||||
|
import org.apache.cassandra.utils.Throwables;
|
||||||
import org.apache.cassandra.utils.concurrent.AsyncPromise;
|
import org.apache.cassandra.utils.concurrent.AsyncPromise;
|
||||||
import org.apache.cassandra.utils.concurrent.Future;
|
import org.apache.cassandra.utils.concurrent.Future;
|
||||||
import org.apache.cassandra.utils.concurrent.ImmediateFuture;
|
import org.apache.cassandra.utils.concurrent.ImmediateFuture;
|
||||||
import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException;
|
import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException;
|
||||||
|
|
||||||
import static accord.messages.SimpleReply.Ok;
|
import static accord.messages.SimpleReply.Ok;
|
||||||
|
import static accord.utils.Invariants.checkState;
|
||||||
|
import static java.util.concurrent.TimeUnit.NANOSECONDS;
|
||||||
import static org.apache.cassandra.config.DatabaseDescriptor.getPartitioner;
|
import static org.apache.cassandra.config.DatabaseDescriptor.getPartitioner;
|
||||||
|
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.accordReadMetrics;
|
||||||
|
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.accordWriteMetrics;
|
||||||
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
|
import static org.apache.cassandra.utils.Clock.Global.nanoTime;
|
||||||
|
|
||||||
public class AccordService implements IAccordService, Shutdownable
|
public class AccordService implements IAccordService, Shutdownable
|
||||||
{
|
{
|
||||||
private static final Logger logger = LoggerFactory.getLogger(AccordService.class);
|
private static final Logger logger = LoggerFactory.getLogger(AccordService.class);
|
||||||
|
|
||||||
public static final AccordClientRequestMetrics readMetrics = new AccordClientRequestMetrics("AccordRead");
|
|
||||||
public static final AccordClientRequestMetrics writeMetrics = new AccordClientRequestMetrics("AccordWrite");
|
|
||||||
private static final Future<Void> BOOTSTRAP_SUCCESS = ImmediateFuture.success(null);
|
private static final Future<Void> BOOTSTRAP_SUCCESS = ImmediateFuture.success(null);
|
||||||
|
|
||||||
private final Node node;
|
private final Node node;
|
||||||
|
|
@ -119,7 +137,13 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public TxnData coordinate(Txn txn, ConsistencyLevel consistencyLevel)
|
public long barrier(@Nonnull Seekables keysOrRanges, long minEpoch, Dispatcher.RequestTime requestTime, long timeoutNanos, BarrierType barrierType, boolean isForWrite)
|
||||||
|
{
|
||||||
|
throw new UnsupportedOperationException("No accord barriers should be executed when accord_transactions_enabled = false in cassandra.yaml");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public @Nonnull TxnResult coordinate(@Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, @Nonnull Dispatcher.RequestTime requestTime)
|
||||||
{
|
{
|
||||||
throw new UnsupportedOperationException("No accord transaction should be executed when accord.enabled = false in cassandra.yaml");
|
throw new UnsupportedOperationException("No accord transaction should be executed when accord.enabled = false in cassandra.yaml");
|
||||||
}
|
}
|
||||||
|
|
@ -181,6 +205,9 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
{
|
{
|
||||||
return Pair.create(new Int2ObjectHashMap<>(), DurableBefore.EMPTY);
|
return Pair.create(new Int2ObjectHashMap<>(), DurableBefore.EMPTY);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void ensureKeyspaceIsAccordManaged(String keyspace) {}
|
||||||
};
|
};
|
||||||
|
|
||||||
private static volatile Node.Id localId = null;
|
private static volatile Node.Id localId = null;
|
||||||
|
|
@ -257,6 +284,9 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
new AccordTopologySorter.Supplier(configService, DatabaseDescriptor.getNodeProximity())),
|
new AccordTopologySorter.Supplier(configService, DatabaseDescriptor.getNodeProximity())),
|
||||||
SimpleProgressLog::new,
|
SimpleProgressLog::new,
|
||||||
AccordCommandStores.factory(journal),
|
AccordCommandStores.factory(journal),
|
||||||
|
new AccordInteropExecution.Factory(agent, configService),
|
||||||
|
AccordInteropPersist.FACTORY,
|
||||||
|
AccordInteropApply.FACTORY,
|
||||||
configuration);
|
configuration);
|
||||||
this.nodeShutdown = toShutdownable(node);
|
this.nodeShutdown = toShutdownable(node);
|
||||||
this.verbHandler = new AccordVerbHandler<>(node, configService, journal);
|
this.verbHandler = new AccordVerbHandler<>(node, configService, journal);
|
||||||
|
|
@ -277,34 +307,21 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public long currentEpoch()
|
public long barrier(@Nonnull Seekables keysOrRanges, long epoch, Dispatcher.RequestTime requestTime, long timeoutNanos, BarrierType barrierType, boolean isForWrite)
|
||||||
{
|
{
|
||||||
return configService.currentEpoch();
|
AccordClientRequestMetrics metrics = isForWrite ? accordWriteMetrics : accordReadMetrics;
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public TopologyManager topology()
|
|
||||||
{
|
|
||||||
return node.topology();
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Consistency level is just echoed back in timeouts, in the future it may be used for interoperability
|
|
||||||
* with non-Accord operations.
|
|
||||||
*/
|
|
||||||
@Override
|
|
||||||
public TxnData coordinate(Txn txn, ConsistencyLevel consistencyLevel)
|
|
||||||
{
|
|
||||||
AccordClientRequestMetrics metrics = txn.isWrite() ? writeMetrics : readMetrics;
|
|
||||||
TxnId txnId = null;
|
TxnId txnId = null;
|
||||||
final long startNanos = nanoTime();
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
metrics.keySize.update(txn.keys().size());
|
logger.debug("Starting barrier key: {} epoch: {} barrierType: {} isForWrite {}", keysOrRanges, epoch, barrierType, isForWrite);
|
||||||
txnId = node.nextTxnId(txn.kind(), txn.keys().domain());
|
txnId = node.nextTxnId(Kind.SyncPoint, keysOrRanges.domain());
|
||||||
AsyncResult<Result> asyncResult = node.coordinate(txnId, txn);
|
AsyncResult<Timestamp> asyncResult = node.barrier(keysOrRanges, epoch, barrierType);
|
||||||
Result result = AsyncChains.getBlocking(asyncResult, DatabaseDescriptor.getTransactionTimeout(TimeUnit.MILLISECONDS), TimeUnit.MILLISECONDS);
|
long deadlineNanos = requestTime.startedAtNanos() + timeoutNanos;
|
||||||
return (TxnData) result;
|
Timestamp barrierExecuteAt = AsyncChains.getBlocking(asyncResult, deadlineNanos - nanoTime(), NANOSECONDS);
|
||||||
|
logger.debug("Completed in {}ms barrier key: {} epoch: {} barrierType: {} isForWrite {}",
|
||||||
|
NANOSECONDS.toMillis(nanoTime() - requestTime.startedAtNanos()),
|
||||||
|
keysOrRanges, epoch, barrierType, isForWrite);
|
||||||
|
return barrierExecuteAt.epoch();
|
||||||
}
|
}
|
||||||
catch (ExecutionException e)
|
catch (ExecutionException e)
|
||||||
{
|
{
|
||||||
|
|
@ -312,14 +329,14 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
if (cause instanceof Timeout)
|
if (cause instanceof Timeout)
|
||||||
{
|
{
|
||||||
metrics.timeouts.mark();
|
metrics.timeouts.mark();
|
||||||
throw throwTimeout(txnId, txn, consistencyLevel);
|
throw newBarrierTimeout(txnId, barrierType.global);
|
||||||
}
|
}
|
||||||
if (cause instanceof Preempted)
|
if (cause instanceof Preempted)
|
||||||
{
|
{
|
||||||
//TODO need to improve
|
//TODO need to improve
|
||||||
// Coordinator "could" query the accord state to see whats going on but that doesn't exist yet.
|
// Coordinator "could" query the accord state to see whats going on but that doesn't exist yet.
|
||||||
// Protocol also doesn't have a way to denote "unknown" outcome, so using a timeout as the closest match
|
// Protocol also doesn't have a way to denote "unknown" outcome, so using a timeout as the closest match
|
||||||
throw throwPreempted(txnId, txn, consistencyLevel);
|
throw newBarrierPreempted(txnId, barrierType.global);
|
||||||
}
|
}
|
||||||
metrics.failures.mark();
|
metrics.failures.mark();
|
||||||
throw new RuntimeException(cause);
|
throw new RuntimeException(cause);
|
||||||
|
|
@ -332,11 +349,125 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
catch (TimeoutException e)
|
catch (TimeoutException e)
|
||||||
{
|
{
|
||||||
metrics.timeouts.mark();
|
metrics.timeouts.mark();
|
||||||
throw throwTimeout(txnId, txn, consistencyLevel);
|
throw newBarrierTimeout(txnId, barrierType.global);
|
||||||
}
|
}
|
||||||
finally
|
finally
|
||||||
{
|
{
|
||||||
metrics.addNano(nanoTime() - startNanos);
|
// TODO Should barriers have a dedicated latency metric? Should it be a read/write metric?
|
||||||
|
// What about counts for timeouts/failures/preempts?
|
||||||
|
metrics.addNano(nanoTime() - requestTime.startedAtNanos());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static ReadTimeoutException newBarrierTimeout(TxnId txnId, boolean global)
|
||||||
|
{
|
||||||
|
return new ReadTimeoutException(global ? ConsistencyLevel.ANY : ConsistencyLevel.QUORUM, 0, 0, false, txnId.toString());
|
||||||
|
}
|
||||||
|
|
||||||
|
private static ReadTimeoutException newBarrierPreempted(TxnId txnId, boolean global)
|
||||||
|
{
|
||||||
|
return new ReadPreemptedException(global ? ConsistencyLevel.ANY : ConsistencyLevel.QUORUM, 0, 0, false, txnId.toString());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long barrierWithRetries(Seekables keysOrRanges, long minEpoch, BarrierType barrierType, boolean isForWrite) throws InterruptedException
|
||||||
|
{
|
||||||
|
// Since we could end up having the barrier transaction or the transaction it listens to invalidated
|
||||||
|
CoordinationFailed existingFailures = null;
|
||||||
|
Long success = null;
|
||||||
|
long backoffMillis = 0;
|
||||||
|
for (int attempt = 0; attempt < DatabaseDescriptor.getAccordBarrierRetryAttempts(); attempt++)
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
Thread.sleep(backoffMillis);
|
||||||
|
}
|
||||||
|
catch (InterruptedException e)
|
||||||
|
{
|
||||||
|
if (existingFailures != null)
|
||||||
|
e.addSuppressed(existingFailures);
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
backoffMillis = backoffMillis == 0 ? DatabaseDescriptor.getAccordBarrierRetryInitialBackoffMillis() : Math.min(backoffMillis * 2, DatabaseDescriptor.getAccordBarrierRetryMaxBackoffMillis());
|
||||||
|
try
|
||||||
|
{
|
||||||
|
success = AccordService.instance().barrier(keysOrRanges, minEpoch, Dispatcher.RequestTime.forImmediateExecution(), DatabaseDescriptor.getAccordRangeBarrierTimeoutNanos(), barrierType, isForWrite);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
catch (CoordinationFailed newFailures)
|
||||||
|
{
|
||||||
|
existingFailures = Throwables.merge(existingFailures, newFailures);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (success == null)
|
||||||
|
{
|
||||||
|
checkState(existingFailures != null, "Didn't have success, but also didn't have failures");
|
||||||
|
throw existingFailures;
|
||||||
|
}
|
||||||
|
return success;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long currentEpoch()
|
||||||
|
{
|
||||||
|
return configService.currentEpoch();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public TopologyManager topology()
|
||||||
|
{
|
||||||
|
return node.topology();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Consistency level is just echoed back in timeouts, in the future it may be used for interoperability
|
||||||
|
* with non-Accord operations.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public @Nonnull TxnResult coordinate(@Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime)
|
||||||
|
{
|
||||||
|
AccordClientRequestMetrics metrics = txn.isWrite() ? accordWriteMetrics : accordReadMetrics;
|
||||||
|
TxnId txnId = null;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
metrics.keySize.update(txn.keys().size());
|
||||||
|
txnId = node.nextTxnId(txn.kind(), txn.keys().domain());
|
||||||
|
long deadlineNanos = requestTime.startedAtNanos() + DatabaseDescriptor.getTransactionTimeout(NANOSECONDS);
|
||||||
|
AsyncResult<Result> asyncResult = node.coordinate(txnId, txn);
|
||||||
|
Result result = AsyncChains.getBlocking(asyncResult, deadlineNanos - nanoTime(), NANOSECONDS);
|
||||||
|
return (TxnResult) result;
|
||||||
|
}
|
||||||
|
catch (ExecutionException e)
|
||||||
|
{
|
||||||
|
Throwable cause = e.getCause();
|
||||||
|
if (cause instanceof Timeout)
|
||||||
|
{
|
||||||
|
metrics.timeouts.mark();
|
||||||
|
throw newTimeout(txnId, txn, consistencyLevel);
|
||||||
|
}
|
||||||
|
if (cause instanceof Preempted)
|
||||||
|
{
|
||||||
|
//TODO need to improve
|
||||||
|
// Coordinator "could" query the accord state to see whats going on but that doesn't exist yet.
|
||||||
|
// Protocol also doesn't have a way to denote "unknown" outcome, so using a timeout as the closest match
|
||||||
|
throw newPreempted(txnId, txn, consistencyLevel);
|
||||||
|
}
|
||||||
|
metrics.failures.mark();
|
||||||
|
throw new RuntimeException(cause);
|
||||||
|
}
|
||||||
|
catch (InterruptedException e)
|
||||||
|
{
|
||||||
|
metrics.failures.mark();
|
||||||
|
throw new UncheckedInterruptedException(e);
|
||||||
|
}
|
||||||
|
catch (TimeoutException e)
|
||||||
|
{
|
||||||
|
metrics.timeouts.mark();
|
||||||
|
throw newTimeout(txnId, txn, consistencyLevel);
|
||||||
|
}
|
||||||
|
finally
|
||||||
|
{
|
||||||
|
metrics.addNano(nanoTime() - requestTime.startedAtNanos());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -371,13 +502,13 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
private static RuntimeException throwTimeout(TxnId txnId, Txn txn, ConsistencyLevel consistencyLevel)
|
private static RequestTimeoutException newTimeout(TxnId txnId, Txn txn, ConsistencyLevel consistencyLevel)
|
||||||
{
|
{
|
||||||
throw txn.isWrite() ? new WriteTimeoutException(WriteType.CAS, consistencyLevel, 0, 0, txnId.toString())
|
throw txn.isWrite() ? new WriteTimeoutException(WriteType.CAS, consistencyLevel, 0, 0, txnId.toString())
|
||||||
: new ReadTimeoutException(consistencyLevel, 0, 0, false, txnId.toString());
|
: new ReadTimeoutException(consistencyLevel, 0, 0, false, txnId.toString());
|
||||||
}
|
}
|
||||||
|
|
||||||
private static RuntimeException throwPreempted(TxnId txnId, Txn txn, ConsistencyLevel consistencyLevel)
|
private static RuntimeException newPreempted(TxnId txnId, Txn txn, ConsistencyLevel consistencyLevel)
|
||||||
{
|
{
|
||||||
throw txn.isWrite() ? new WritePreemptedException(WriteType.CAS, consistencyLevel, 0, 0, txnId.toString())
|
throw txn.isWrite() ? new WritePreemptedException(WriteType.CAS, consistencyLevel, 0, 0, txnId.toString())
|
||||||
: new ReadPreemptedException(consistencyLevel, 0, 0, false, txnId.toString());
|
: new ReadPreemptedException(consistencyLevel, 0, 0, false, txnId.toString());
|
||||||
|
|
@ -438,6 +569,11 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
return scheduler;
|
return scheduler;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public Id nodeId()
|
||||||
|
{
|
||||||
|
return node.id();
|
||||||
|
}
|
||||||
|
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
public Node node()
|
public Node node()
|
||||||
{
|
{
|
||||||
|
|
@ -523,6 +659,22 @@ public class AccordService implements IAccordService, Shutdownable
|
||||||
return ClusterMetadata.current().accordKeyspaces.contains(keyspace);
|
return ClusterMetadata.current().accordKeyspaces.contains(keyspace);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void ensureKeyspaceIsAccordManaged(String keyspace)
|
||||||
|
{
|
||||||
|
if (isAccordManagedKeyspace(keyspace))
|
||||||
|
return;
|
||||||
|
ClusterMetadataService.instance().commit(new AddAccordKeyspace(keyspace),
|
||||||
|
metadata -> null,
|
||||||
|
(code, message) -> {
|
||||||
|
Invariants.checkState(code == ExceptionCode.ALREADY_EXISTS,
|
||||||
|
"Expected %s, got %s", ExceptionCode.ALREADY_EXISTS, code);
|
||||||
|
return null;
|
||||||
|
});
|
||||||
|
// we need to avoid creating a txnId in an epoch when no one has any ranges
|
||||||
|
FBUtilities.waitOnFuture(AccordService.instance().epochReady(ClusterMetadata.current().epoch));
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Pair<Int2ObjectHashMap<RedundantBefore>, DurableBefore> getRedundantBeforesAndDurableBefore()
|
public Pair<Int2ObjectHashMap<RedundantBefore>, DurableBefore> getRedundantBeforesAndDurableBefore()
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -37,13 +37,11 @@ import org.apache.cassandra.dht.Range;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.locator.EndpointsForRange;
|
import org.apache.cassandra.locator.EndpointsForRange;
|
||||||
import org.apache.cassandra.locator.InetAddressAndPort;
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
import org.apache.cassandra.schema.DistributedSchema;
|
|
||||||
import org.apache.cassandra.schema.KeyspaceMetadata;
|
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||||
import org.apache.cassandra.schema.ReplicationParams;
|
import org.apache.cassandra.schema.ReplicationParams;
|
||||||
import org.apache.cassandra.service.accord.api.AccordRoutingKey.SentinelKey;
|
import org.apache.cassandra.service.accord.api.AccordRoutingKey.SentinelKey;
|
||||||
import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey;
|
import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey;
|
||||||
import org.apache.cassandra.tcm.ClusterMetadata;
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
import org.apache.cassandra.tcm.Epoch;
|
|
||||||
import org.apache.cassandra.tcm.membership.Directory;
|
import org.apache.cassandra.tcm.membership.Directory;
|
||||||
import org.apache.cassandra.tcm.membership.NodeId;
|
import org.apache.cassandra.tcm.membership.NodeId;
|
||||||
import org.apache.cassandra.tcm.ownership.DataPlacement;
|
import org.apache.cassandra.tcm.ownership.DataPlacement;
|
||||||
|
|
@ -120,17 +118,17 @@ public class AccordTopologyUtils
|
||||||
return shards;
|
return shards;
|
||||||
}
|
}
|
||||||
|
|
||||||
public static Topology createAccordTopology(Epoch epoch, DistributedSchema schema, DataPlacements placements, Directory directory, Predicate<String> keyspacePredicate)
|
public static Topology createAccordTopology(ClusterMetadata cm, Predicate<String> keyspacePredicate)
|
||||||
{
|
{
|
||||||
List<Shard> shards = new ArrayList<>();
|
List<Shard> shards = new ArrayList<>();
|
||||||
for (KeyspaceMetadata keyspace : schema.getKeyspaces())
|
for (KeyspaceMetadata keyspace : cm.schema.getKeyspaces())
|
||||||
{
|
{
|
||||||
if (!keyspacePredicate.test(keyspace.name))
|
if (!keyspacePredicate.test(keyspace.name))
|
||||||
continue;
|
continue;
|
||||||
shards.addAll(createShards(keyspace, placements, directory));
|
shards.addAll(createShards(keyspace, cm.placements, cm.directory));
|
||||||
}
|
}
|
||||||
shards.sort((a, b) -> a.range.compare(b.range));
|
shards.sort((a, b) -> a.range.compare(b.range));
|
||||||
return new Topology(epoch.getEpoch(), shards.toArray(new Shard[0]));
|
return new Topology(cm.epoch.getEpoch(), shards.toArray(new Shard[0]));
|
||||||
}
|
}
|
||||||
|
|
||||||
public static EndpointMapping directoryToMapping(EndpointMapping mapping, long epoch, Directory directory)
|
public static EndpointMapping directoryToMapping(EndpointMapping mapping, long epoch, Directory directory)
|
||||||
|
|
@ -146,11 +144,6 @@ public class AccordTopologyUtils
|
||||||
return builder.build();
|
return builder.build();
|
||||||
}
|
}
|
||||||
|
|
||||||
public static Topology createAccordTopology(ClusterMetadata metadata, Predicate<String> keyspacePredicate)
|
|
||||||
{
|
|
||||||
return createAccordTopology(metadata.epoch, metadata.schema, metadata.placements, metadata.directory, keyspacePredicate);
|
|
||||||
}
|
|
||||||
|
|
||||||
public static Topology createAccordTopology(ClusterMetadata metadata)
|
public static Topology createAccordTopology(ClusterMetadata metadata)
|
||||||
{
|
{
|
||||||
return createAccordTopology(metadata, metadata.accordKeyspaces::contains);
|
return createAccordTopology(metadata, metadata.accordKeyspaces::contains);
|
||||||
|
|
|
||||||
|
|
@ -18,30 +18,74 @@
|
||||||
|
|
||||||
package org.apache.cassandra.service.accord;
|
package org.apache.cassandra.service.accord;
|
||||||
|
|
||||||
|
import java.util.HashSet;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
import java.util.Set;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.concurrent.TimeoutException;
|
import java.util.concurrent.TimeoutException;
|
||||||
|
import java.util.stream.Collectors;
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
|
import com.google.common.collect.ImmutableSet;
|
||||||
|
|
||||||
|
import accord.api.BarrierType;
|
||||||
import accord.local.DurableBefore;
|
import accord.local.DurableBefore;
|
||||||
|
import accord.local.Node.Id;
|
||||||
import accord.local.RedundantBefore;
|
import accord.local.RedundantBefore;
|
||||||
import accord.messages.Request;
|
import accord.messages.Request;
|
||||||
|
import accord.primitives.Ranges;
|
||||||
|
import accord.primitives.Seekables;
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
import accord.topology.TopologyManager;
|
import accord.topology.TopologyManager;
|
||||||
import org.agrona.collections.Int2ObjectHashMap;
|
import org.agrona.collections.Int2ObjectHashMap;
|
||||||
|
import org.apache.cassandra.db.ColumnFamilyStore;
|
||||||
import org.apache.cassandra.db.ConsistencyLevel;
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
|
import org.apache.cassandra.dht.Range;
|
||||||
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.net.IVerbHandler;
|
import org.apache.cassandra.net.IVerbHandler;
|
||||||
import org.apache.cassandra.net.Message;
|
import org.apache.cassandra.net.Message;
|
||||||
|
import org.apache.cassandra.service.accord.api.AccordRoutableKey;
|
||||||
|
import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey;
|
||||||
import org.apache.cassandra.service.accord.api.AccordScheduler;
|
import org.apache.cassandra.service.accord.api.AccordScheduler;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnData;
|
import org.apache.cassandra.service.accord.txn.TxnResult;
|
||||||
import org.apache.cassandra.tcm.Epoch;
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
import org.apache.cassandra.utils.Pair;
|
import org.apache.cassandra.utils.Pair;
|
||||||
import org.apache.cassandra.utils.concurrent.Future;
|
import org.apache.cassandra.utils.concurrent.Future;
|
||||||
|
|
||||||
public interface IAccordService
|
public interface IAccordService
|
||||||
{
|
{
|
||||||
|
Set<ConsistencyLevel> SUPPORTED_COMMIT_CONSISTENCY_LEVELS = ImmutableSet.of(ConsistencyLevel.ANY, ConsistencyLevel.ONE, ConsistencyLevel.QUORUM, ConsistencyLevel.SERIAL, ConsistencyLevel.ALL);
|
||||||
|
Set<ConsistencyLevel> SUPPORTED_READ_CONSISTENCY_LEVELS = ImmutableSet.of(ConsistencyLevel.ONE, ConsistencyLevel.QUORUM, ConsistencyLevel.SERIAL);
|
||||||
|
|
||||||
IVerbHandler<? extends Request> verbHandler();
|
IVerbHandler<? extends Request> verbHandler();
|
||||||
|
|
||||||
TxnData coordinate(Txn txn, ConsistencyLevel consistencyLevel);
|
default long barrierWithRetries(Seekables keysOrRanges, long minEpoch, BarrierType barrierType, boolean isForWrite) throws InterruptedException
|
||||||
|
{
|
||||||
|
throw new UnsupportedOperationException();
|
||||||
|
}
|
||||||
|
|
||||||
|
long barrier(@Nonnull Seekables keysOrRanges, long minEpoch, Dispatcher.RequestTime requestTime, long timeoutNanos, BarrierType barrierType, boolean isForWrite);
|
||||||
|
|
||||||
|
default void postStreamReceivingBarrier(ColumnFamilyStore cfs, List<Range<Token>> ranges)
|
||||||
|
{
|
||||||
|
String ks = cfs.keyspace.getName();
|
||||||
|
Ranges accordRanges = Ranges.of(ranges
|
||||||
|
.stream()
|
||||||
|
.map(r -> new TokenRange(new TokenKey(ks, r.left), new TokenKey(ks, r.right)))
|
||||||
|
.collect(Collectors.toList())
|
||||||
|
.toArray(new accord.primitives.Range[0]));
|
||||||
|
try
|
||||||
|
{
|
||||||
|
barrierWithRetries(accordRanges, Epoch.FIRST.getEpoch(), BarrierType.global_async, true);
|
||||||
|
}
|
||||||
|
catch (InterruptedException e)
|
||||||
|
{
|
||||||
|
throw new RuntimeException(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Nonnull TxnResult coordinate(@Nonnull Txn txn, @Nonnull ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime);
|
||||||
|
|
||||||
long currentEpoch();
|
long currentEpoch();
|
||||||
|
|
||||||
|
|
@ -74,4 +118,26 @@ public interface IAccordService
|
||||||
* Fetch the redundnant befores for every command store
|
* Fetch the redundnant befores for every command store
|
||||||
*/
|
*/
|
||||||
Pair<Int2ObjectHashMap<RedundantBefore>, DurableBefore> getRedundantBeforesAndDurableBefore();
|
Pair<Int2ObjectHashMap<RedundantBefore>, DurableBefore> getRedundantBeforesAndDurableBefore();
|
||||||
|
|
||||||
|
default Id nodeId() { throw new UnsupportedOperationException(); }
|
||||||
|
|
||||||
|
default void maybeConvertKeyspacesToAccord(Txn txn)
|
||||||
|
{
|
||||||
|
Set<String> allKeyspaces = new HashSet<>();
|
||||||
|
txn.keys().forEach(key -> allKeyspaces.add(((AccordRoutableKey) key).keyspace()));
|
||||||
|
|
||||||
|
for (String keyspace : allKeyspaces)
|
||||||
|
{
|
||||||
|
|
||||||
|
ensureKeyspaceIsAccordManaged(keyspace);
|
||||||
|
}
|
||||||
|
|
||||||
|
for (String keyspace : allKeyspaces)
|
||||||
|
{
|
||||||
|
if (!AccordService.instance().isAccordManagedKeyspace(keyspace))
|
||||||
|
throw new IllegalStateException(keyspace + " is not an accord managed keyspace");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void ensureKeyspaceIsAccordManaged(String keyspace);
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -23,6 +23,8 @@ import java.util.concurrent.TimeUnit;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
|
||||||
import accord.api.Agent;
|
import accord.api.Agent;
|
||||||
import accord.api.EventsListener;
|
import accord.api.EventsListener;
|
||||||
import accord.api.Result;
|
import accord.api.Result;
|
||||||
|
|
@ -32,16 +34,20 @@ import accord.primitives.Ranges;
|
||||||
import accord.primitives.Seekables;
|
import accord.primitives.Seekables;
|
||||||
import accord.primitives.Timestamp;
|
import accord.primitives.Timestamp;
|
||||||
import accord.primitives.Txn;
|
import accord.primitives.Txn;
|
||||||
|
import accord.primitives.Txn.Kind;
|
||||||
import accord.primitives.TxnId;
|
import accord.primitives.TxnId;
|
||||||
import org.apache.cassandra.service.accord.AccordService;
|
import org.apache.cassandra.service.accord.AccordService;
|
||||||
import org.apache.cassandra.metrics.AccordMetrics;
|
import org.apache.cassandra.metrics.AccordMetrics;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
import org.apache.cassandra.service.accord.txn.TxnQuery;
|
||||||
import org.apache.cassandra.service.accord.txn.TxnRead;
|
import org.apache.cassandra.service.accord.txn.TxnRead;
|
||||||
|
import org.apache.cassandra.tcm.Epoch;
|
||||||
import org.apache.cassandra.utils.JVMStabilityInspector;
|
import org.apache.cassandra.utils.JVMStabilityInspector;
|
||||||
|
|
||||||
|
import static accord.primitives.Routable.Domain.Key;
|
||||||
import static java.util.concurrent.TimeUnit.MICROSECONDS;
|
import static java.util.concurrent.TimeUnit.MICROSECONDS;
|
||||||
import static java.util.concurrent.TimeUnit.SECONDS;
|
import static java.util.concurrent.TimeUnit.SECONDS;
|
||||||
import static org.apache.cassandra.config.DatabaseDescriptor.getReadRpcTimeout;
|
import static org.apache.cassandra.config.DatabaseDescriptor.getReadRpcTimeout;
|
||||||
|
import static org.apache.cassandra.service.consensus.migration.ConsensusKeyMigrationState.maybeSaveAccordKeyMigrationLocally;
|
||||||
|
|
||||||
// TODO (expected): merge with AccordService
|
// TODO (expected): merge with AccordService
|
||||||
public class AccordAgent implements Agent
|
public class AccordAgent implements Agent
|
||||||
|
|
@ -78,6 +84,16 @@ public class AccordAgent implements Agent
|
||||||
AccordService.instance().scheduler().once(retry, retryBootstrapDelayMicros, MICROSECONDS);
|
AccordService.instance().scheduler().once(retry, retryBootstrapDelayMicros, MICROSECONDS);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onLocalBarrier(@Nonnull Seekables<?, ?> keysOrRanges, @Nonnull Timestamp executeAt)
|
||||||
|
{
|
||||||
|
if (keysOrRanges.domain() == Key)
|
||||||
|
{
|
||||||
|
PartitionKey key = (PartitionKey)keysOrRanges.get(0);
|
||||||
|
maybeSaveAccordKeyMigrationLocally(key, Epoch.create(executeAt.epoch()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public void onUncaughtException(Throwable t)
|
public void onUncaughtException(Throwable t)
|
||||||
{
|
{
|
||||||
|
|
@ -99,9 +115,9 @@ public class AccordAgent implements Agent
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public Txn emptyTxn(Txn.Kind kind, Seekables<?, ?> keysOrRanges)
|
public Txn emptyTxn(Kind kind, Seekables<?, ?> seekables)
|
||||||
{
|
{
|
||||||
return new Txn.InMemory(kind, keysOrRanges, TxnRead.EMPTY, TxnQuery.ALL, null);
|
return new Txn.InMemory(kind, seekables, TxnRead.EMPTY, TxnQuery.EMPTY, null);
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
|
|
|
||||||
|
|
@ -33,6 +33,7 @@ import org.apache.cassandra.db.TypeSizes;
|
||||||
import org.apache.cassandra.db.marshal.ByteBufferAccessor;
|
import org.apache.cassandra.db.marshal.ByteBufferAccessor;
|
||||||
import org.apache.cassandra.db.marshal.ValueAccessor;
|
import org.apache.cassandra.db.marshal.ValueAccessor;
|
||||||
import org.apache.cassandra.db.partitions.Partition;
|
import org.apache.cassandra.db.partitions.Partition;
|
||||||
|
import org.apache.cassandra.db.partitions.PartitionUpdate;
|
||||||
import org.apache.cassandra.dht.Token;
|
import org.apache.cassandra.dht.Token;
|
||||||
import org.apache.cassandra.io.IVersionedSerializer;
|
import org.apache.cassandra.io.IVersionedSerializer;
|
||||||
import org.apache.cassandra.io.util.DataInputPlus;
|
import org.apache.cassandra.io.util.DataInputPlus;
|
||||||
|
|
@ -70,6 +71,11 @@ public final class PartitionKey extends AccordRoutableKey implements Key
|
||||||
return (PartitionKey) key;
|
return (PartitionKey) key;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public static PartitionKey of(PartitionUpdate update)
|
||||||
|
{
|
||||||
|
return new PartitionKey(update.metadata().keyspace, update.metadata().id, update.partitionKey());
|
||||||
|
}
|
||||||
|
|
||||||
public static PartitionKey of(Partition partition)
|
public static PartitionKey of(Partition partition)
|
||||||
{
|
{
|
||||||
return new PartitionKey(partition.metadata().keyspace, partition.metadata().id, partition.partitionKey());
|
return new PartitionKey(partition.metadata().keyspace, partition.metadata().id, partition.partitionKey());
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,269 @@
|
||||||
|
/*
|
||||||
|
* Licensed to the Apache Software Foundation (ASF) under one
|
||||||
|
* or more contributor license agreements. See the NOTICE file
|
||||||
|
* distributed with this work for additional information
|
||||||
|
* regarding copyright ownership. The ASF licenses this file
|
||||||
|
* to you under the Apache License, Version 2.0 (the
|
||||||
|
* "License"); you may not use this file except in compliance
|
||||||
|
* with the License. You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package org.apache.cassandra.service.accord.interop;
|
||||||
|
|
||||||
|
import java.util.BitSet;
|
||||||
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
|
import accord.api.Result;
|
||||||
|
import accord.local.Command;
|
||||||
|
import accord.local.Node.Id;
|
||||||
|
import accord.local.PreLoadContext;
|
||||||
|
import accord.local.SafeCommand;
|
||||||
|
import accord.local.SafeCommandStore;
|
||||||
|
import accord.local.Status;
|
||||||
|
import accord.messages.Apply;
|
||||||
|
import accord.messages.MessageType;
|
||||||
|
import accord.primitives.Deps;
|
||||||
|
import accord.primitives.Keys;
|
||||||
|
import accord.primitives.PartialDeps;
|
||||||
|
import accord.primitives.PartialRoute;
|
||||||
|
import accord.primitives.PartialTxn;
|
||||||
|
import accord.primitives.Route;
|
||||||
|
import accord.primitives.Seekables;
|
||||||
|
import accord.primitives.Timestamp;
|
||||||
|
import accord.primitives.Txn;
|
||||||
|
import accord.primitives.TxnId;
|
||||||
|
import accord.primitives.Writes;
|
||||||
|
import accord.topology.Topologies;
|
||||||
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
|
import org.apache.cassandra.io.IVersionedSerializer;
|
||||||
|
import org.apache.cassandra.service.accord.AccordMessageSink.AccordMessageType;
|
||||||
|
import org.apache.cassandra.service.accord.serializers.ApplySerializers.ApplySerializer;
|
||||||
|
import org.apache.cassandra.service.accord.txn.AccordUpdate;
|
||||||
|
|
||||||
|
import static accord.utils.Invariants.checkState;
|
||||||
|
import static accord.utils.MapReduceConsume.forEach;
|
||||||
|
import static com.google.common.base.Preconditions.checkArgument;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Apply that waits until the transaction is actually applied before sending a response
|
||||||
|
* // TODO (desired): At this point there are a plethora of do X to Command, then wait until state Y before maybe doing Z and returning a response, potentially returning insufficient along the way
|
||||||
|
* // and these all are a bit copy pasta in terms of managing things like waiting on, obsoletion, cancellation/listeners, insufficient etc. and it would be less fragile
|
||||||
|
* // in the long run to not duplicate these kind of difficult to get right mechanism and have a single pluggable framework to request each specific behavior
|
||||||
|
*/
|
||||||
|
public class AccordInteropApply extends Apply implements Command.TransientListener
|
||||||
|
{
|
||||||
|
public static final Apply.Factory FACTORY = new Apply.Factory()
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
public Apply create(Kind kind, Id to, Topologies participates, Topologies executes, TxnId txnId, Route<?> route, Txn txn, Timestamp executeAt, Deps deps, Writes writes, Result result)
|
||||||
|
{
|
||||||
|
checkArgument(kind != Kind.Maximal, "Shouldn't need to send a maximal commit with interop support");
|
||||||
|
ConsistencyLevel commitCL = txn.update() instanceof AccordUpdate ? ((AccordUpdate) txn.update()).cassandraCommitCL() : null;
|
||||||
|
// Any asynchronous apply option should use the regular Apply that doesn't wait for writes to complete
|
||||||
|
if (commitCL == null || commitCL == ConsistencyLevel.ANY)
|
||||||
|
return Apply.FACTORY.create(kind, to, participates, executes, txnId, route, txn, executeAt, deps, writes, result);
|
||||||
|
return new AccordInteropApply(kind, to, participates, executes, txnId, route, txn, executeAt, deps, writes, result);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
public static final IVersionedSerializer<AccordInteropApply> serializer = new ApplySerializer<AccordInteropApply>()
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
protected AccordInteropApply deserializeApply(TxnId txnId, PartialRoute<?> scope, long waitForEpoch, Apply.Kind kind, Seekables<?, ?> keys, Timestamp executeAt, PartialDeps deps, PartialTxn txn, Writes writes, Result result)
|
||||||
|
{
|
||||||
|
return new AccordInteropApply(kind, txnId, scope, waitForEpoch, keys, executeAt, deps, txn, writes, result);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
transient BitSet waitingOn;
|
||||||
|
transient int waitingOnCount;
|
||||||
|
|
||||||
|
private AccordInteropApply(Kind kind, TxnId txnId, PartialRoute<?> route, long waitForEpoch, Seekables<?, ?> keys, Timestamp executeAt, PartialDeps deps, @Nullable PartialTxn txn, Writes writes, Result result)
|
||||||
|
{
|
||||||
|
super(kind, txnId, route, waitForEpoch, keys, executeAt, deps, txn, writes, result);
|
||||||
|
}
|
||||||
|
|
||||||
|
private AccordInteropApply(Kind kind, Id to, Topologies participates, Topologies executes, TxnId txnId, Route<?> route, Txn txn, Timestamp executeAt, Deps deps, Writes writes, Result result)
|
||||||
|
{
|
||||||
|
super(kind, to, participates, executes, txnId, route, txn, executeAt, deps, writes, result);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void process()
|
||||||
|
{
|
||||||
|
waitingOn = new BitSet();
|
||||||
|
super.process();
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ApplyReply apply(SafeCommandStore safeStore)
|
||||||
|
{
|
||||||
|
ApplyReply reply = super.apply(safeStore);
|
||||||
|
checkState(reply == ApplyReply.Redundant || reply == ApplyReply.Applied || reply == ApplyReply.Insufficient, "Unexpected ApplyReply");
|
||||||
|
|
||||||
|
// Hasn't necessarily finished applying yet so need to check and maybe add a listener
|
||||||
|
// Redundant means we are competing with a recovery coordinator which is fine
|
||||||
|
// we don't need to return an error we can wait for the Apply
|
||||||
|
// Insufficient means it is safe to install the listener and wait for Apply to happen
|
||||||
|
// once the coordinator sends a maximal commit
|
||||||
|
// Applied doesn't actually mean the command is in the Applied state so we still need to check and maybe install
|
||||||
|
// the listener
|
||||||
|
SafeCommand safeCommand = safeStore.get(txnId, executeAt, scope);
|
||||||
|
Command current = safeCommand.current();
|
||||||
|
// Don't actually think it is possible for this to reach applied while we are stll running, but just to be safe
|
||||||
|
// check anyways
|
||||||
|
Status status = current.status();
|
||||||
|
switch (status)
|
||||||
|
{
|
||||||
|
default: throw new AssertionError();
|
||||||
|
case NotDefined:
|
||||||
|
case PreAccepted:
|
||||||
|
case Accepted:
|
||||||
|
case AcceptedInvalidate:
|
||||||
|
case PreCommitted:
|
||||||
|
case Committed:
|
||||||
|
case PreApplied:
|
||||||
|
case ReadyToExecute:
|
||||||
|
synchronized (this)
|
||||||
|
{
|
||||||
|
waitingOn.set(safeStore.commandStore().id());
|
||||||
|
++waitingOnCount;
|
||||||
|
}
|
||||||
|
safeCommand.addListener(this);
|
||||||
|
break;
|
||||||
|
|
||||||
|
case Applied:
|
||||||
|
case Invalidated:
|
||||||
|
case Truncated:
|
||||||
|
}
|
||||||
|
|
||||||
|
return reply;
|
||||||
|
}
|
||||||
|
|
||||||
|
private synchronized void ack()
|
||||||
|
{
|
||||||
|
// wait for -1 to ensure the setup phase has also completed. Setup calls ack in its callback
|
||||||
|
// and prevents races where we respond before dispatching all the required reads (if the reads are
|
||||||
|
// completing faster than the reads can be setup on all required shards)
|
||||||
|
if (-1 == --waitingOnCount)
|
||||||
|
{
|
||||||
|
node.reply(replyTo, replyContext, ApplyReply.Applied, null);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ApplyReply reduce(ApplyReply r1, ApplyReply r2)
|
||||||
|
{
|
||||||
|
return r1 == null || r2 == null
|
||||||
|
? r1 == null ? r2 : r1
|
||||||
|
: r1.compareTo(r2) >= 0 ? r1 : r2;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void accept(ApplyReply reply, Throwable failure)
|
||||||
|
{
|
||||||
|
if (reply == ApplyReply.Insufficient)
|
||||||
|
{
|
||||||
|
// Respond with insufficient which should make the coordinator send us the commit
|
||||||
|
// we need to respond
|
||||||
|
node.reply(replyTo, replyContext, reply, failure);
|
||||||
|
}
|
||||||
|
else if (failure != null)
|
||||||
|
{
|
||||||
|
node.reply(replyTo, replyContext, null, failure);
|
||||||
|
node.agent().onUncaughtException(failure);
|
||||||
|
cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unless failed always ack to indicate setup has completed otherwise the counter never gets to -1
|
||||||
|
if (failure == null)
|
||||||
|
ack();
|
||||||
|
}
|
||||||
|
|
||||||
|
private void cancel()
|
||||||
|
{
|
||||||
|
node.commandStores().mapReduceConsume(this, waitingOn.stream(), forEach(safeStore -> {
|
||||||
|
SafeCommand safeCommand = safeStore.ifInitialised(txnId);
|
||||||
|
if (safeCommand != null)
|
||||||
|
safeCommand.removeListener(this);
|
||||||
|
}, node.agent()));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public TxnId primaryTxnId()
|
||||||
|
{
|
||||||
|
return txnId;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Seekables<?, ?> keys()
|
||||||
|
{
|
||||||
|
if (txn == null) return Keys.EMPTY;
|
||||||
|
return txn.keys();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public MessageType type()
|
||||||
|
{
|
||||||
|
switch (kind)
|
||||||
|
{
|
||||||
|
case Minimal: return AccordMessageType.INTEROP_APPLY_MINIMAL_REQ;
|
||||||
|
case Maximal: return AccordMessageType.INTEROP_APPLY_MAXIMAL_REQ;
|
||||||
|
default: throw new IllegalStateException();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String toString()
|
||||||
|
{
|
||||||
|
return "AccordInteropApply{" +
|
||||||
|
"txnId:" + txnId +
|
||||||
|
", deps:" + deps +
|
||||||
|
", executeAt:" + executeAt +
|
||||||
|
", writes:" + writes +
|
||||||
|
", result:" + result +
|
||||||
|
'}';
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onChange(SafeCommandStore safeStore, SafeCommand safeCommand)
|
||||||
|
{
|
||||||
|
Command command = safeCommand.current();
|
||||||
|
|
||||||
|
switch (command.status())
|
||||||
|
{
|
||||||
|
default: throw new AssertionError();
|
||||||
|
case NotDefined:
|
||||||
|
case PreAccepted:
|
||||||
|
case Accepted:
|
||||||
|
case AcceptedInvalidate:
|
||||||
|
case PreCommitted:
|
||||||
|
case Committed:
|
||||||
|
case PreApplied:
|
||||||
|
case ReadyToExecute:
|
||||||
|
return;
|
||||||
|
|
||||||
|
case Applied:
|
||||||
|
case Invalidated:
|
||||||
|
case Truncated:
|
||||||
|
}
|
||||||
|
|
||||||
|
if (safeCommand.removeListener(this))
|
||||||
|
ack();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public PreLoadContext listenerPreLoadContext(TxnId caller)
|
||||||
|
{
|
||||||
|
return PreLoadContext.contextFor(txnId);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -0,0 +1,73 @@
|
||||||
|
/*
|
||||||
|
* Licensed to the Apache Software Foundation (ASF) under one
|
||||||
|
* or more contributor license agreements. See the NOTICE file
|
||||||
|
* distributed with this work for additional information
|
||||||
|
* regarding copyright ownership. The ASF licenses this file
|
||||||
|
* to you under the Apache License, Version 2.0 (the
|
||||||
|
* "License"); you may not use this file except in compliance
|
||||||
|
* with the License. You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package org.apache.cassandra.service.accord.interop;
|
||||||
|
|
||||||
|
import javax.annotation.Nonnull;
|
||||||
|
import javax.annotation.Nullable;
|
||||||
|
|
||||||
|
import accord.local.Node;
|
||||||
|
import accord.messages.Commit;
|
||||||
|
import accord.messages.MessageType;
|
||||||
|
import accord.messages.ReadData;
|
||||||
|
import accord.primitives.Deps;
|
||||||
|
import accord.primitives.FullRoute;
|
||||||
|
import accord.primitives.PartialDeps;
|
||||||
|
import accord.primitives.PartialRoute;
|
||||||
|
import accord.primitives.PartialTxn;
|
||||||
|
import accord.primitives.Timestamp;
|
||||||
|
import accord.primitives.Txn;
|
||||||
|
import accord.primitives.TxnId;
|
||||||
|
import accord.topology.Topologies;
|
||||||
|
import accord.topology.Topology;
|
||||||
|
import org.apache.cassandra.io.IVersionedSerializer;
|
||||||
|
import org.apache.cassandra.service.accord.AccordMessageSink.AccordMessageType;
|
||||||
|
import org.apache.cassandra.service.accord.serializers.CommitSerializers.CommitSerializer;
|
||||||
|
|
||||||
|
public class AccordInteropCommit extends Commit
|
||||||
|
{
|
||||||
|
public static final IVersionedSerializer<AccordInteropCommit> serializer = new CommitSerializer<AccordInteropCommit, AccordInteropRead>(AccordInteropRead.class, AccordInteropRead.requestSerializer)
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
protected AccordInteropCommit deserializeCommit(TxnId txnId, PartialRoute<?> scope, long waitForEpoch, Kind kind, Timestamp executeAt, @Nullable PartialTxn partialTxn, PartialDeps partialDeps, @Nullable FullRoute<?> fullRoute, @Nullable ReadData read)
|
||||||
|
{
|
||||||
|
return new AccordInteropCommit(kind, txnId, scope, waitForEpoch, executeAt, partialTxn, partialDeps, fullRoute, read);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
public AccordInteropCommit(Kind kind, TxnId txnId, PartialRoute<?> scope, long waitForEpoch, Timestamp executeAt, @Nullable PartialTxn partialTxn, PartialDeps partialDeps, @Nullable FullRoute<?> fullRoute, @Nonnull ReadData readData)
|
||||||
|
{
|
||||||
|
super(kind, txnId, scope, waitForEpoch, executeAt, partialTxn, partialDeps, fullRoute, readData);
|
||||||
|
}
|
||||||
|
|
||||||
|
public AccordInteropCommit(Kind kind, Node.Id to, Topology coordinateTopology, Topologies topologies, TxnId txnId, Txn txn, FullRoute<?> route, Timestamp executeAt, Deps deps, AccordInteropRead read)
|
||||||
|
{
|
||||||
|
super(kind, to, coordinateTopology, topologies, txnId, txn, route, executeAt, deps, (t, u, p) -> read);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public MessageType type()
|
||||||
|
{
|
||||||
|
switch (kind)
|
||||||
|
{
|
||||||
|
case Minimal: return AccordMessageType.INTEROP_COMMIT_MINIMAL_REQ;
|
||||||
|
case Maximal: return AccordMessageType.INTEROP_COMMIT_MAXIMAL_REQ;
|
||||||
|
default: throw new IllegalStateException();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -0,0 +1,412 @@
|
||||||
|
/*
|
||||||
|
* Licensed to the Apache Software Foundation (ASF) under one
|
||||||
|
* or more contributor license agreements. See the NOTICE file
|
||||||
|
* distributed with this work for additional information
|
||||||
|
* regarding copyright ownership. The ASF licenses this file
|
||||||
|
* to you under the Apache License, Version 2.0 (the
|
||||||
|
* "License"); you may not use this file except in compliance
|
||||||
|
* with the License. You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package org.apache.cassandra.service.accord.interop;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.Callable;
|
||||||
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicInteger;
|
||||||
|
import java.util.function.BiConsumer;
|
||||||
|
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import accord.api.Agent;
|
||||||
|
import accord.api.Data;
|
||||||
|
import accord.api.Result;
|
||||||
|
import accord.coordinate.Execute;
|
||||||
|
import accord.coordinate.Persist;
|
||||||
|
import accord.coordinate.TxnExecute;
|
||||||
|
import accord.local.AgentExecutor;
|
||||||
|
import accord.local.CommandStore;
|
||||||
|
import accord.local.Node;
|
||||||
|
import accord.local.Node.Id;
|
||||||
|
import accord.messages.Commit;
|
||||||
|
import accord.messages.Commit.Kind;
|
||||||
|
import accord.primitives.Deps;
|
||||||
|
import accord.primitives.FullRoute;
|
||||||
|
import accord.primitives.Participants;
|
||||||
|
import accord.primitives.Seekables;
|
||||||
|
import accord.primitives.Timestamp;
|
||||||
|
import accord.primitives.Txn;
|
||||||
|
import accord.primitives.TxnId;
|
||||||
|
import accord.topology.Shard;
|
||||||
|
import accord.topology.Topologies;
|
||||||
|
import accord.topology.Topology;
|
||||||
|
import accord.utils.async.AsyncChain;
|
||||||
|
import accord.utils.async.AsyncChains;
|
||||||
|
import org.apache.cassandra.concurrent.Stage;
|
||||||
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
|
import org.apache.cassandra.db.Mutation;
|
||||||
|
import org.apache.cassandra.db.ReadCommand;
|
||||||
|
import org.apache.cassandra.db.ReadResponse;
|
||||||
|
import org.apache.cassandra.db.SinglePartitionReadCommand;
|
||||||
|
import org.apache.cassandra.db.SinglePartitionReadCommand.Group;
|
||||||
|
import org.apache.cassandra.db.partitions.FilteredPartition;
|
||||||
|
import org.apache.cassandra.db.partitions.PartitionIterator;
|
||||||
|
import org.apache.cassandra.db.rows.RowIterator;
|
||||||
|
import org.apache.cassandra.dht.Range;
|
||||||
|
import org.apache.cassandra.dht.Token;
|
||||||
|
import org.apache.cassandra.locator.EndpointsForToken;
|
||||||
|
import org.apache.cassandra.locator.InetAddressAndPort;
|
||||||
|
import org.apache.cassandra.locator.Replica;
|
||||||
|
import org.apache.cassandra.metrics.AccordClientRequestMetrics;
|
||||||
|
import org.apache.cassandra.net.Message;
|
||||||
|
import org.apache.cassandra.net.RequestCallback;
|
||||||
|
import org.apache.cassandra.schema.KeyspaceMetadata;
|
||||||
|
import org.apache.cassandra.service.StorageProxy;
|
||||||
|
import org.apache.cassandra.service.accord.AccordEndpointMapper;
|
||||||
|
import org.apache.cassandra.service.accord.TokenRange;
|
||||||
|
import org.apache.cassandra.service.accord.api.AccordAgent;
|
||||||
|
import org.apache.cassandra.service.accord.api.AccordRoutingKey;
|
||||||
|
import org.apache.cassandra.service.accord.api.PartitionKey;
|
||||||
|
import org.apache.cassandra.service.accord.interop.AccordInteropReadCallback.MaximalCommitSender;
|
||||||
|
import org.apache.cassandra.service.accord.txn.AccordUpdate;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnData;
|
||||||
|
import org.apache.cassandra.service.accord.txn.TxnRead;
|
||||||
|
import org.apache.cassandra.service.accord.txn.UnrecoverableRepairUpdate;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState;
|
||||||
|
import org.apache.cassandra.service.consensus.migration.ConsensusTableMigrationState.TableMigrationState;
|
||||||
|
import org.apache.cassandra.service.reads.ReadCoordinator;
|
||||||
|
import org.apache.cassandra.tcm.ClusterMetadata;
|
||||||
|
import org.apache.cassandra.transport.Dispatcher;
|
||||||
|
|
||||||
|
import static accord.utils.Invariants.checkArgument;
|
||||||
|
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.accordReadMetrics;
|
||||||
|
import static org.apache.cassandra.metrics.ClientRequestsMetricsHolder.accordWriteMetrics;
|
||||||
|
|
||||||
|
/*
|
||||||
|
* The core interoperability problem between Accord and C* writes (regular, and read repair)
|
||||||
|
* is that when the writes don't go through Accord then Accord can read data that is not yet committed
|
||||||
|
* because Accord replicas can lag behind and multiple coordinators can be attempting to compute the result of a
|
||||||
|
* transaction and they can compute different results depending on what they consider to be the inputs to the Accord
|
||||||
|
* transaction.
|
||||||
|
*
|
||||||
|
* We generally solve this by forcing non-Accord writes through Accord as well as by having Accord perform read repair
|
||||||
|
* on its inputs.
|
||||||
|
*
|
||||||
|
*/
|
||||||
|
public class AccordInteropExecution implements Execute, ReadCoordinator, MaximalCommitSender
|
||||||
|
{
|
||||||
|
private static final Logger logger = LoggerFactory.getLogger(AccordInteropExecution.class);
|
||||||
|
|
||||||
|
private static class InteropExecutor implements AgentExecutor
|
||||||
|
{
|
||||||
|
private final AccordAgent agent;
|
||||||
|
|
||||||
|
public InteropExecutor(AccordAgent agent)
|
||||||
|
{
|
||||||
|
this.agent = agent;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Agent agent()
|
||||||
|
{
|
||||||
|
return agent;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public <T> AsyncChain<T> submit(Callable<T> task)
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
return AsyncChains.success(task.call());
|
||||||
|
}
|
||||||
|
catch (Throwable e)
|
||||||
|
{
|
||||||
|
return AsyncChains.failure(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
public static class Factory implements Execute.Factory
|
||||||
|
{
|
||||||
|
private final InteropExecutor executor;
|
||||||
|
private final AccordEndpointMapper endpointMapper;
|
||||||
|
|
||||||
|
public Factory(AccordAgent agent, AccordEndpointMapper endpointMapper)
|
||||||
|
{
|
||||||
|
this.executor = new InteropExecutor(agent);
|
||||||
|
this.endpointMapper = endpointMapper;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Execute create(Node node, TxnId txnId, Txn txn, FullRoute<?> route, Participants<?> readScope, Timestamp executeAt, Deps deps, BiConsumer<? super Result, Throwable> callback)
|
||||||
|
{
|
||||||
|
// Unrecoverable repair always needs to be run by AccordInteropExecution
|
||||||
|
AccordUpdate.Kind updateKind = AccordUpdate.kind(txn.update());
|
||||||
|
ConsistencyLevel consistencyLevel = txn.read() instanceof TxnRead ? ((TxnRead) txn.read()).cassandraConsistencyLevel() : null;
|
||||||
|
if (updateKind != AccordUpdate.Kind.UNRECOVERABLE_REPAIR && (consistencyLevel == null || consistencyLevel == ConsistencyLevel.ONE || txn.read().keys().isEmpty()))
|
||||||
|
return TxnExecute.FACTORY.create(node, txnId, txn, route, readScope, executeAt, deps, callback);
|
||||||
|
return new AccordInteropExecution(node, txnId, txn, updateKind, route, readScope, executeAt, deps, callback, executor, consistencyLevel, endpointMapper);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private final Node node;
|
||||||
|
private final TxnId txnId;
|
||||||
|
private final Txn txn;
|
||||||
|
private final FullRoute<?> route;
|
||||||
|
private final Participants<?> readScope;
|
||||||
|
private final Timestamp executeAt;
|
||||||
|
private final Deps deps;
|
||||||
|
private final BiConsumer<? super Result, Throwable> callback;
|
||||||
|
private final AgentExecutor executor;
|
||||||
|
private final ConsistencyLevel consistencyLevel;
|
||||||
|
private final AccordEndpointMapper endpointMapper;
|
||||||
|
|
||||||
|
private final Topologies executes;
|
||||||
|
private final Topologies allTopologies;
|
||||||
|
private final Topology executeTopology;
|
||||||
|
private final Topology coordinateTopology;
|
||||||
|
|
||||||
|
private final AtomicInteger readsCurrentlyUnderConstruction;
|
||||||
|
|
||||||
|
private final Set<InetAddressAndPort> contacted;
|
||||||
|
private final AccordUpdate.Kind updateKind;
|
||||||
|
|
||||||
|
public AccordInteropExecution(Node node, TxnId txnId, Txn txn, AccordUpdate.Kind updateKind, FullRoute<?> route, Participants<?> readScope, Timestamp executeAt, Deps deps, BiConsumer<? super Result, Throwable> callback,
|
||||||
|
AgentExecutor executor, ConsistencyLevel consistencyLevel, AccordEndpointMapper endpointMapper)
|
||||||
|
{
|
||||||
|
checkArgument(!txn.read().keys().isEmpty() || updateKind == AccordUpdate.Kind.UNRECOVERABLE_REPAIR);
|
||||||
|
this.node = node;
|
||||||
|
this.txnId = txnId;
|
||||||
|
this.txn = txn;
|
||||||
|
this.route = route;
|
||||||
|
this.readScope = readScope;
|
||||||
|
this.executeAt = executeAt;
|
||||||
|
this.deps = deps;
|
||||||
|
this.callback = callback;
|
||||||
|
this.executor = executor;
|
||||||
|
|
||||||
|
checkArgument(updateKind == AccordUpdate.Kind.UNRECOVERABLE_REPAIR || consistencyLevel == ConsistencyLevel.QUORUM || consistencyLevel == ConsistencyLevel.ALL || consistencyLevel == ConsistencyLevel.SERIAL);
|
||||||
|
this.consistencyLevel = consistencyLevel;
|
||||||
|
this.endpointMapper = endpointMapper;
|
||||||
|
|
||||||
|
this.executes = node.topology().forEpoch(route, executeAt.epoch());
|
||||||
|
this.allTopologies = txnId.epoch() != executeAt.epoch()
|
||||||
|
? node.topology().preciseEpochs(route, txnId.epoch(), executeAt.epoch())
|
||||||
|
: executes;
|
||||||
|
this.executeTopology = executes.forEpoch(executeAt.epoch());
|
||||||
|
this.coordinateTopology = allTopologies.forEpoch(txnId.epoch());
|
||||||
|
if (consistencyLevel != ConsistencyLevel.ALL)
|
||||||
|
{
|
||||||
|
readsCurrentlyUnderConstruction = new AtomicInteger(txn.read().keys().size());
|
||||||
|
contacted = Collections.newSetFromMap(new ConcurrentHashMap<>());
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
readsCurrentlyUnderConstruction = null;
|
||||||
|
contacted = null;
|
||||||
|
}
|
||||||
|
this.updateKind = updateKind;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean localReadSupported()
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public EndpointsForToken forNonLocalStrategyTokenRead(ClusterMetadata doNotUse, KeyspaceMetadata keyspace, Token token)
|
||||||
|
{
|
||||||
|
AccordRoutingKey.TokenKey key = new AccordRoutingKey.TokenKey(keyspace.name, token);
|
||||||
|
Shard shard = executeTopology.forKey(key);
|
||||||
|
Range<Token> range = ((TokenRange) shard.range).toKeyspaceRange();
|
||||||
|
|
||||||
|
Replica[] replicas = new Replica[shard.nodes.size()];
|
||||||
|
for (int i=0; i<replicas.length; i++)
|
||||||
|
{
|
||||||
|
Node.Id id = shard.nodes.get(i);
|
||||||
|
replicas[i] = new Replica(endpointMapper.mappedEndpoint(id), range, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
return EndpointsForToken.of(token, replicas);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void sendReadCommand(Message<ReadCommand> message, InetAddressAndPort to, RequestCallback<ReadResponse> callback)
|
||||||
|
{
|
||||||
|
Node.Id id = endpointMapper.mappedId(to);
|
||||||
|
SinglePartitionReadCommand command = (SinglePartitionReadCommand) message.payload;
|
||||||
|
AccordInteropRead read = new AccordInteropRead(id, executes, txnId, readScope, executeAt, command);
|
||||||
|
AccordInteropCommit commit = new AccordInteropCommit(Commit.Kind.Minimal, id, coordinateTopology, allTopologies,
|
||||||
|
txnId, txn, route, executeAt, deps, read);
|
||||||
|
node.send(id, commit, executor, new AccordInteropRead.ReadCallback(id, to, message, callback, this));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void sendReadRepairMutation(Message<Mutation> message, InetAddressAndPort to, RequestCallback<Object> callback)
|
||||||
|
{
|
||||||
|
Node.Id id = endpointMapper.mappedId(to);
|
||||||
|
Mutation mutation = message.payload;
|
||||||
|
AccordInteropReadRepair readRepair = new AccordInteropReadRepair(id, executes, txnId, readScope, executeAt, mutation);
|
||||||
|
node.send(id, readRepair, executor, new AccordInteropReadRepair.ReadRepairCallback(id, to, message, callback, this));
|
||||||
|
}
|
||||||
|
|
||||||
|
private AsyncChain<Data> readChains()
|
||||||
|
{
|
||||||
|
int nowInSeconds = (int) TimeUnit.MICROSECONDS.toSeconds(executeAt.hlc());
|
||||||
|
// TODO (expected): use normal query nano time
|
||||||
|
Dispatcher.RequestTime requestTime = Dispatcher.RequestTime.forImmediateExecution();
|
||||||
|
|
||||||
|
TxnRead read = (TxnRead) txn.read();
|
||||||
|
List<AsyncChain<Data>> results = new ArrayList<>();
|
||||||
|
Seekables<?, ?> keys = txn.read().keys();
|
||||||
|
keys.forEach(key -> {
|
||||||
|
read.forEachWithKey((PartitionKey) key, fragment -> {
|
||||||
|
SinglePartitionReadCommand command = (SinglePartitionReadCommand) fragment.command();
|
||||||
|
|
||||||
|
// This should only rarely occur when coordinators start a transaction in a migrating range
|
||||||
|
// because they haven't yet updated their cluster metadata.
|
||||||
|
// It would be harmless to do the read, but we can respond faster skipping it
|
||||||
|
// and getting the transaction on the correct protocol
|
||||||
|
TableMigrationState tms = ConsensusTableMigrationState.getTableMigrationState(command.metadata().id);
|
||||||
|
AccordClientRequestMetrics metrics = txn.kind().isWrite() ? accordWriteMetrics : accordReadMetrics;
|
||||||
|
if (ConsensusRequestRouter.instance.isKeyInMigratingOrMigratedRangeFromAccord(tms, command.partitionKey()))
|
||||||
|
{
|
||||||
|
metrics.migrationSkippedReads.mark();
|
||||||
|
results.add(AsyncChains.success(TxnData.emptyPartition(fragment.txnDataName(), command)));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
Group group = Group.one(command.withNowInSec(nowInSeconds));
|
||||||
|
results.add(AsyncChains.ofCallable(Stage.ACCORD_MIGRATION.executor(), () -> {
|
||||||
|
TxnData result = new TxnData();
|
||||||
|
try (PartitionIterator iterator = StorageProxy.readRegular(group, consistencyLevel, this, requestTime))
|
||||||
|
{
|
||||||
|
if (iterator.hasNext())
|
||||||
|
{
|
||||||
|
try (RowIterator partition = iterator.next())
|
||||||
|
{
|
||||||
|
FilteredPartition filtered = FilteredPartition.create(partition);
|
||||||
|
if (filtered.hasRows() || command.selectsFullPartition())
|
||||||
|
result.put(fragment.txnDataName(), filtered);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}));
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
if (results.isEmpty())
|
||||||
|
return AsyncChains.success(new TxnData());
|
||||||
|
|
||||||
|
if (results.size() == 1)
|
||||||
|
return results.get(0);
|
||||||
|
|
||||||
|
return AsyncChains.reduce(results, Data::merge);
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Any nodes not contacted for read need to be sent commits
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public void notifyOfInitialContacts(EndpointsForToken fullDataRequests, EndpointsForToken transientRequests, EndpointsForToken digestRequests)
|
||||||
|
{
|
||||||
|
if (readsCurrentlyUnderConstruction == null)
|
||||||
|
return;
|
||||||
|
|
||||||
|
for (int i = 0; i < fullDataRequests.size(); i++)
|
||||||
|
contacted.add(fullDataRequests.endpoint(i));
|
||||||
|
for (int i = 0; i < transientRequests.size(); i++)
|
||||||
|
contacted.add(transientRequests.endpoint(i));
|
||||||
|
for (int i = 0; i < digestRequests.size(); i++)
|
||||||
|
contacted.add(digestRequests.endpoint(i));
|
||||||
|
if (readsCurrentlyUnderConstruction.decrementAndGet() == 0)
|
||||||
|
sendCommitsToUncontacted();
|
||||||
|
}
|
||||||
|
|
||||||
|
private void sendCommitsToUncontacted()
|
||||||
|
{
|
||||||
|
for (Node.Id to : executeTopology.nodes())
|
||||||
|
if (!contacted.contains(endpointMapper.mappedEndpoint(to)))
|
||||||
|
node.send(to, new Commit(Kind.Minimal, to, coordinateTopology, allTopologies, txnId, txn, route, readScope, executeAt, deps, false));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void start()
|
||||||
|
{
|
||||||
|
if (coordinateTopology != executeTopology)
|
||||||
|
{
|
||||||
|
for (Node.Id to : allTopologies.nodes())
|
||||||
|
{
|
||||||
|
if (!executeTopology.contains(to))
|
||||||
|
node.send(to, new Commit(Commit.Kind.Minimal, to, coordinateTopology, allTopologies, txnId, txn, route, readScope, executeAt, deps, false));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
AsyncChain<Data> result;
|
||||||
|
if (updateKind == AccordUpdate.Kind.UNRECOVERABLE_REPAIR)
|
||||||
|
result = executeUnrecoverableRepairUpdate();
|
||||||
|
else
|
||||||
|
result = readChains();
|
||||||
|
|
||||||
|
CommandStore cs = node.commandStores().select(route.homeKey());
|
||||||
|
result.beginAsResult().withExecutor(cs).begin((data, failure) -> {
|
||||||
|
if (failure == null)
|
||||||
|
Persist.persist(node, executes, txnId, route, txn, executeAt, deps, txn.execute(txnId, executeAt, data), txn.result(txnId, executeAt, data), callback);
|
||||||
|
else
|
||||||
|
callback.accept(null, failure);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
private AsyncChain<Data> executeUnrecoverableRepairUpdate()
|
||||||
|
{
|
||||||
|
return AsyncChains.ofCallable(Stage.ACCORD_MIGRATION.executor(), () -> {
|
||||||
|
UnrecoverableRepairUpdate repairUpdate = (UnrecoverableRepairUpdate)txn.update();
|
||||||
|
// TODO (expected): We should send the read in the same message as the commit. This requires refactor ReadData.Kind so that it doesn't specify the ordinal encoding
|
||||||
|
// and can be extended similar to MessageType which allows additional types not from Accord to be added
|
||||||
|
for (Node.Id to : executeTopology.nodes())
|
||||||
|
node.send(to, new Commit(Kind.Minimal, to, coordinateTopology, allTopologies, txnId, txn, route, readScope, executeAt, deps, false));
|
||||||
|
repairUpdate.runBRR(AccordInteropExecution.this);
|
||||||
|
return new TxnData();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean isEventuallyConsistent()
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ReadCommand maybeAllowOutOfRangeReads(ReadCommand readCommand)
|
||||||
|
{
|
||||||
|
return readCommand.allowOutOfRangeReads();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Mutation maybeAllowOutOfRangeMutations(Mutation m)
|
||||||
|
{
|
||||||
|
return m.allowOutOfRangeMutations();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Prrovide request callbacks with a way to send maximal commits on Insufficient responses
|
||||||
|
@Override
|
||||||
|
public void sendMaximalCommit(Id to)
|
||||||
|
{
|
||||||
|
Commit.commitMaximal(node, to, txn, txnId, executeAt, route, deps, readScope);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -0,0 +1,167 @@
|
||||||
|
/*
|
||||||
|
* Licensed to the Apache Software Foundation (ASF) under one
|
||||||
|
* or more contributor license agreements. See the NOTICE file
|
||||||
|
* distributed with this work for additional information
|
||||||
|
* regarding copyright ownership. The ASF licenses this file
|
||||||
|
* to you under the Apache License, Version 2.0 (the
|
||||||
|
* "License"); you may not use this file except in compliance
|
||||||
|
* with the License. You may obtain a copy of the License at
|
||||||
|
*
|
||||||
|
* http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
*
|
||||||
|
* Unless required by applicable law or agreed to in writing, software
|
||||||
|
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
* See the License for the specific language governing permissions and
|
||||||
|
* limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package org.apache.cassandra.service.accord.interop;
|
||||||
|
|
||||||
|
import java.util.function.BiConsumer;
|
||||||
|
|
||||||
|
import accord.api.Result;
|
||||||
|
import accord.api.Update;
|
||||||
|
import accord.coordinate.Persist;
|
||||||
|
import accord.coordinate.TxnPersist;
|
||||||
|
import accord.coordinate.tracking.AppliedTracker;
|
||||||
|
import accord.coordinate.tracking.QuorumTracker;
|
||||||
|
import accord.coordinate.tracking.RequestStatus;
|
||||||
|
import accord.coordinate.tracking.ResponseTracker;
|
||||||
|
import accord.local.Node;
|
||||||
|
import accord.messages.Apply;
|
||||||
|
import accord.primitives.Deps;
|
||||||
|
import accord.primitives.FullRoute;
|
||||||
|
import accord.primitives.Timestamp;
|
||||||
|
import accord.primitives.Txn;
|
||||||
|
import accord.primitives.TxnId;
|
||||||
|
import accord.primitives.Writes;
|
||||||
|
import accord.topology.Topologies;
|
||||||
|
import accord.utils.Invariants;
|
||||||
|
import org.apache.cassandra.db.ConsistencyLevel;
|
||||||
|
import org.apache.cassandra.service.accord.txn.AccordUpdate;
|
||||||
|
import org.apache.cassandra.utils.Throwables;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Similar to Accord persist, but can wait on a configurable number of responses and sends AccordInteropApply messages
|
||||||
|
* that only return a response when the Apply has actually occurred. Regular Apply messages only get the transaction
|
||||||
|
* to PreApplied.
|
||||||
|
*/
|
||||||
|
public class AccordInteropPersist extends Persist
|
||||||
|
{
|
||||||
|
public static Persist.Factory FACTORY = new Persist.Factory()
|
||||||
|
{
|
||||||
|
@Override
|
||||||
|
public Persist create(Node node, Topologies topologies, TxnId txnId, FullRoute<?> route, Txn txn, Timestamp executeAt, Deps deps, Writes writes, Result result)
|
||||||
|
{
|
||||||
|
Update update = txn.update();
|
||||||
|
ConsistencyLevel consistencyLevel = update instanceof AccordUpdate ? ((AccordUpdate) update).cassandraCommitCL() : null;
|
||||||
|
if (consistencyLevel == null || consistencyLevel == ConsistencyLevel.ANY || writes.isEmpty())
|
||||||
|
return TxnPersist.FACTORY.create(node, topologies, txnId, route, txn, executeAt, deps, writes, result);
|
||||||
|
return new AccordInteropPersist(node, topologies, txnId, route, txn, executeAt, deps, writes, result, consistencyLevel);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
private static class CallbackHolder
|
||||||
|
{
|
||||||
|
private final ResponseTracker tracker;
|
||||||
|
private final Result result;
|
||||||
|
private final BiConsumer<? super Result, Throwable> clientCallback;
|
||||||
|
private Throwable failure = null;
|
||||||
|
|
||||||
|
public CallbackHolder(ResponseTracker tracker, Result result, BiConsumer<? super Result, Throwable> clientCallback)
|
||||||
|
{
|
||||||
|
this.tracker = tracker;
|
||||||
|
this.result = result;
|
||||||
|
this.clientCallback = clientCallback;
|
||||||
|
}
|
||||||
|
|
||||||
|
private void handleStatus(RequestStatus status)
|
||||||
|
{
|
||||||
|
switch (status)
|
||||||
|
{
|
||||||
|
default: throw new IllegalStateException("Unhandled request status " + status);
|
||||||
|
case Success:
|
||||||
|
clientCallback.accept(result, null);
|
||||||
|
return;
|
||||||
|
case Failed:
|
||||||
|
clientCallback.accept(null, failure);
|
||||||
|
return;
|
||||||
|
case NoChange:
|
||||||
|
// noop
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
public void recordSuccess(Node.Id node)
|
||||||
|
{
|
||||||
|
handleStatus(tracker.recordSuccess(node));
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
public void recordFailure(Node.Id node, Throwable throwable)
|
||||||
|
{
|
||||||
|
failure = Throwables.merge(failure, throwable);
|
||||||
|
handleStatus(tracker.recordFailure(node));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private final ConsistencyLevel consistencyLevel;
|
||||||
|
private CallbackHolder holder = null;
|
||||||
|
|
||||||
|
public AccordInteropPersist(Node node, Topologies topologies, TxnId txnId, FullRoute<?> route, Txn txn, Timestamp executeAt, Deps deps, Writes writes, Result result, ConsistencyLevel consistencyLevel)
|
||||||
|
{
|
||||||
|
super(node, topologies, txnId, route, txn, executeAt, deps, writes, result);
|
||||||
|
Invariants.checkArgument(consistencyLevel == ConsistencyLevel.QUORUM || consistencyLevel == ConsistencyLevel.ALL || consistencyLevel == ConsistencyLevel.SERIAL || consistencyLevel == ConsistencyLevel.ONE);
|
||||||
|
this.consistencyLevel = consistencyLevel;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void registerClientCallback(Writes writes, Result result, BiConsumer<? super Result, Throwable> clientCallback)
|
||||||
|
{
|
||||||
|
|
||||||
|
Invariants.checkState(holder == null);
|
||||||
|
switch (consistencyLevel)
|
||||||
|
{
|
||||||
|
case ONE: // Can safely upgrade ONE to QUORUM/SERIAL to get a synchronous commit
|
||||||
|
case SERIAL:
|
||||||
|
case QUORUM:
|
||||||
|
holder = new CallbackHolder(new QuorumTracker(topologies), result, clientCallback);
|
||||||
|
break;
|
||||||
|
case ALL:
|
||||||
|
holder = new CallbackHolder(new AppliedTracker(topologies), result, clientCallback);
|
||||||
|
break;
|
||||||
|
default:
|
||||||
|
throw new IllegalArgumentException("Unhandled consistency level: " + consistencyLevel);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onSuccess(Node.Id from, Apply.ApplyReply reply)
|
||||||
|
{
|
||||||
|
super.onSuccess(from, reply);
|
||||||
|
switch (reply)
|
||||||
|
{
|
||||||
|
case Redundant:
|
||||||
|
case Applied:
|
||||||
|
holder.recordSuccess(from);
|
||||||
|
return;
|
||||||
|
case Insufficient:
|
||||||
|
// On insufficient Persist will send a commit with the missing information
|
||||||
|
// which will allow a final response to be returned later that could be successful
|
||||||
|
return;
|
||||||
|
default: throw new IllegalArgumentException("Unhandled apply response " + reply);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onFailure(Node.Id from, Throwable failure)
|
||||||
|
{
|
||||||
|
holder.recordFailure(from, failure);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onCallbackFailure(Node.Id from, Throwable failure)
|
||||||
|
{
|
||||||
|
holder.recordFailure(from, failure);
|
||||||
|
}
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue