diff --git a/src/java/org/apache/cassandra/cql3/selection/Selectable.java b/src/java/org/apache/cassandra/cql3/selection/Selectable.java index c4e7dd6d73..301a6306b4 100644 --- a/src/java/org/apache/cassandra/cql3/selection/Selectable.java +++ b/src/java/org/apache/cassandra/cql3/selection/Selectable.java @@ -18,12 +18,27 @@ */ package org.apache.cassandra.cql3.selection; -import java.util.*; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.Collections; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; import java.util.function.Predicate; import java.util.stream.Collectors; -import org.apache.cassandra.cql3.*; -import org.apache.cassandra.cql3.functions.*; +import org.apache.cassandra.cql3.AssignmentTestable; +import org.apache.cassandra.cql3.CQL3Type; +import org.apache.cassandra.cql3.ColumnIdentifier; +import org.apache.cassandra.cql3.ColumnSpecification; +import org.apache.cassandra.cql3.FieldIdentifier; +import org.apache.cassandra.cql3.VariableSpecifications; +import org.apache.cassandra.cql3.functions.AggregateFcts; +import org.apache.cassandra.cql3.functions.CastFcts; +import org.apache.cassandra.cql3.functions.Function; +import org.apache.cassandra.cql3.functions.FunctionName; +import org.apache.cassandra.cql3.functions.FunctionResolver; +import org.apache.cassandra.cql3.functions.OperationFcts; import org.apache.cassandra.cql3.selection.Selector.Factory; import org.apache.cassandra.cql3.terms.Constants; import org.apache.cassandra.cql3.terms.Lists; @@ -33,7 +48,18 @@ import org.apache.cassandra.cql3.terms.Term; import org.apache.cassandra.cql3.terms.Tuples; import org.apache.cassandra.cql3.terms.UserTypes; import org.apache.cassandra.cql3.terms.Vectors; -import org.apache.cassandra.db.marshal.*; +import org.apache.cassandra.db.marshal.AbstractType; +import org.apache.cassandra.db.marshal.CollectionType; +import org.apache.cassandra.db.marshal.DurationType; +import org.apache.cassandra.db.marshal.Int32Type; +import org.apache.cassandra.db.marshal.ListType; +import org.apache.cassandra.db.marshal.LongType; +import org.apache.cassandra.db.marshal.MapType; +import org.apache.cassandra.db.marshal.ReversedType; +import org.apache.cassandra.db.marshal.SetType; +import org.apache.cassandra.db.marshal.TupleType; +import org.apache.cassandra.db.marshal.UserType; +import org.apache.cassandra.db.marshal.VectorType; import org.apache.cassandra.exceptions.InvalidRequestException; import org.apache.cassandra.schema.ColumnMetadata; import org.apache.cassandra.schema.TableMetadata; @@ -1402,6 +1428,11 @@ public interface Selectable extends AssignmentTestable : FieldIdentifier.forUnquoted(text); } + public String getText() + { + return text; + } + @Override public String toString() { diff --git a/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java b/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java index 6837e50711..c315bd04a8 100644 --- a/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java +++ b/src/java/org/apache/cassandra/cql3/statements/CQL3CasRequest.java @@ -62,9 +62,9 @@ import org.apache.cassandra.service.CASRequest; import org.apache.cassandra.service.ClientState; import org.apache.cassandra.service.accord.txn.TxnCondition; import org.apache.cassandra.service.accord.txn.TxnData; -import org.apache.cassandra.service.accord.txn.TxnDataName; +import org.apache.cassandra.service.accord.txn.TxnDataKeyValue; +import org.apache.cassandra.service.accord.txn.TxnKeyRead; import org.apache.cassandra.service.accord.txn.TxnQuery; -import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.service.accord.txn.TxnReference; import org.apache.cassandra.service.accord.txn.TxnResult; import org.apache.cassandra.service.accord.txn.TxnUpdate; @@ -78,7 +78,8 @@ import static com.google.common.base.Preconditions.checkState; import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult; import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.RETRY_NEW_PROTOCOL; import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.casResult; -import static org.apache.cassandra.service.accord.txn.TxnDataName.Kind.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataName; import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol; import static org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter.getTableMetadata; @@ -424,8 +425,7 @@ public class CQL3CasRequest implements CASRequest public TxnCondition asTxnCondition() { - TxnDataName txnDataName = new TxnDataName(CAS_READ, clustering, TxnRead.CAS_READ_NAME); - TxnReference txnReference = new TxnReference(txnDataName, null); + TxnReference txnReference = new TxnReference(txnDataName(CAS_READ), null); return new TxnCondition.Exists(txnReference, TxnCondition.Kind.IS_NULL); } } @@ -450,8 +450,7 @@ public class CQL3CasRequest implements CASRequest public TxnCondition asTxnCondition() { - TxnDataName txnDataName = new TxnDataName(CAS_READ, clustering, TxnRead.CAS_READ_NAME); - TxnReference txnReference = new TxnReference(txnDataName, null); + TxnReference txnReference = new TxnReference(txnDataName(CAS_READ), null); return new TxnCondition.Exists(txnReference, TxnCondition.Kind.IS_NOT_NULL); } } @@ -501,23 +500,25 @@ public class CQL3CasRequest implements CASRequest public Txn toAccordTxn(ClusterMetadata cm, ConsistencyLevel consistencyLevel, ConsistencyLevel commitConsistencyLevel, ClientState clientState, long nowInSecs) { SinglePartitionReadCommand readCommand = readCommand(nowInSecs); - Update update = createUpdate(clientState, commitConsistencyLevel); + Update update = createUpdate(cm, clientState, commitConsistencyLevel); // If the write strategy is sending all writes through Accord there is no need to use the supplied consistency // level since Accord will manage reading safely TableParams tableParams = getTableMetadata(cm, metadata.id).params; consistencyLevel = tableParams.transactionalMode.readCLForStrategy(tableParams.transactionalMigrationFrom, consistencyLevel, cm, metadata.id, readCommand.partitionKey().getToken()); - TxnRead read = TxnRead.createCasRead(readCommand, consistencyLevel); + TxnKeyRead read = TxnKeyRead.createCasRead(readCommand, consistencyLevel); // In a CAS requesting only one key is supported and writes // can't be dependent on any data that is read (only conditions) // so the only relevant keys are the read key return new Txn.InMemory(read.keys(), read, TxnQuery.CONDITION, update); } - private Update createUpdate(ClientState clientState, ConsistencyLevel commitConsistencyLevel) + private Update createUpdate(ClusterMetadata cm, ClientState clientState, ConsistencyLevel commitConsistencyLevel) { // Potentially ignore commit consistency level if TransactionalMode is full // since it is safe to match what non-SERIAL writes do - commitConsistencyLevel = metadata.params.transactionalMode.commitCLForStrategy(commitConsistencyLevel, metadata.id, key.getToken()); + TableMetadata tableMetadata = getTableMetadata(cm, metadata.id); + TableParams tableParams = tableMetadata.params; + commitConsistencyLevel = tableParams.transactionalMode.commitCLForStrategy(tableParams.transactionalMigrationFrom, commitConsistencyLevel, cm, metadata.id, key.getToken()); // CAS requires using the new txn timestamp to correctly linearize some kinds of updates return new TxnUpdate(createWriteFragments(clientState), createCondition(), commitConsistencyLevel, false); } @@ -566,7 +567,7 @@ public class CQL3CasRequest implements CASRequest if (txnResult.kind() == retry_new_protocol) return RETRY_NEW_PROTOCOL; TxnData txnData = (TxnData)txnResult; - FilteredPartition partition = txnData.get(TxnRead.CAS_READ); + TxnDataKeyValue partition = (TxnDataKeyValue)txnData.get(txnDataName(CAS_READ)); return casResult(partition != null ? partition.rowIterator(false) : null); } } diff --git a/src/java/org/apache/cassandra/cql3/statements/TransactionStatement.java b/src/java/org/apache/cassandra/cql3/statements/TransactionStatement.java index 3b5bf66976..d3fe56847a 100644 --- a/src/java/org/apache/cassandra/cql3/statements/TransactionStatement.java +++ b/src/java/org/apache/cassandra/cql3/statements/TransactionStatement.java @@ -36,12 +36,12 @@ import javax.annotation.Nullable; import com.google.common.annotations.VisibleForTesting; import com.google.common.base.Preconditions; import com.google.common.collect.ImmutableList; -import com.google.common.collect.ImmutableMap; import com.google.common.collect.Iterables; import accord.api.Key; import accord.primitives.Keys; import accord.primitives.Txn; +import org.agrona.collections.Int2ObjectHashMap; import org.apache.cassandra.audit.AuditLogContext; import org.apache.cassandra.audit.AuditLogEntryType; import org.apache.cassandra.config.DatabaseDescriptor; @@ -60,7 +60,6 @@ import org.apache.cassandra.db.SinglePartitionReadCommand; import org.apache.cassandra.db.SinglePartitionReadQuery; import org.apache.cassandra.db.filter.DataLimits; import org.apache.cassandra.db.marshal.AbstractType; -import org.apache.cassandra.db.partitions.FilteredPartition; import org.apache.cassandra.exceptions.InvalidRequestException; import org.apache.cassandra.schema.ColumnMetadata; import org.apache.cassandra.schema.Schema; @@ -71,10 +70,10 @@ import org.apache.cassandra.service.accord.api.AccordRoutableKey; import org.apache.cassandra.service.accord.txn.AccordUpdate; import org.apache.cassandra.service.accord.txn.TxnCondition; import org.apache.cassandra.service.accord.txn.TxnData; -import org.apache.cassandra.service.accord.txn.TxnDataName; +import org.apache.cassandra.service.accord.txn.TxnDataKeyValue; +import org.apache.cassandra.service.accord.txn.TxnKeyRead; import org.apache.cassandra.service.accord.txn.TxnNamedRead; import org.apache.cassandra.service.accord.txn.TxnQuery; -import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.service.accord.txn.TxnReference; import org.apache.cassandra.service.accord.txn.TxnResult; import org.apache.cassandra.service.accord.txn.TxnUpdate; @@ -90,7 +89,11 @@ import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse import static org.apache.cassandra.cql3.statements.RequestValidations.checkNotNull; import static org.apache.cassandra.cql3.statements.RequestValidations.checkTrue; import static org.apache.cassandra.cql3.statements.RequestValidations.invalidRequest; -import static org.apache.cassandra.service.accord.txn.TxnRead.createTxnRead; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.AUTO_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.RETURNING; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.USER; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataName; +import static org.apache.cassandra.service.accord.txn.TxnKeyRead.createTxnRead; import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol; public class TransactionStatement implements CQLStatement.CompositeCQLStatement, CQLStatement.ReturningCQLStatement @@ -116,10 +119,10 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, static class NamedSelect { - final TxnDataName name; + final int name; final SelectStatement select; - public NamedSelect(TxnDataName name, SelectStatement select) + public NamedSelect(int name, SelectStatement select) { this.name = name; this.select = select; @@ -245,11 +248,11 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, List list = new ArrayList<>(selectQuery.queries.size()); for (int i = 0; i < selectQuery.queries.size(); i++) - list.add(new TxnNamedRead(TxnDataName.returning(i), selectQuery.queries.get(i))); + list.add(new TxnNamedRead(txnDataName(RETURNING, i), selectQuery.queries.get(i))); return list; } - private List createNamedReads(QueryOptions options, ClientState state, Map autoReads, Consumer keyConsumer) + private List createNamedReads(QueryOptions options, ClientState state, @Nullable Int2ObjectHashMap autoReads, Consumer keyConsumer) { List reads = new ArrayList<>(assignments.size() + 1); @@ -269,9 +272,12 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, } } - for (NamedSelect select : autoReads.values()) - // don't need keyConsumer as the keys are known to exist due to Modification - reads.add(createNamedRead(select, options, state)); + if (autoReads != null) + { + for (NamedSelect select : autoReads.values()) + // don't need keyConsumer as the keys are known to exist due to Modification + reads.add(createNamedRead(select, options, state)); + } return reads; } @@ -291,7 +297,7 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, return new TxnCondition.BooleanGroup(TxnCondition.Kind.AND, result); } - List createWriteFragments(ClientState state, QueryOptions options, Map autoReads, Consumer keyConsumer) + List createWriteFragments(ClientState state, QueryOptions options, Map autoReads, Consumer keyConsumer) { List fragments = new ArrayList<>(updates.size()); int idx = 0; @@ -304,7 +310,7 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, if (modification.allReferenceOperations().stream().anyMatch(ReferenceOperation::requiresRead)) { // Reads are not merged by partition here due to potentially differing columns retrieved, etc. - TxnDataName partitionName = TxnDataName.partitionRead(modification.metadata(), fragment.key.partitionKey(), idx); + int partitionName = txnDataName(AUTO_READ, idx); if (!autoReads.containsKey(partitionName)) autoReads.put(partitionName, new NamedSelect(partitionName, modification.createSelectForTxn())); } @@ -314,7 +320,7 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, return fragments; } - AccordUpdate createUpdate(ClientState state, QueryOptions options, Map autoReads, Consumer keyConsumer) + AccordUpdate createUpdate(ClientState state, QueryOptions options, Map autoReads, Consumer keyConsumer) { return new TxnUpdate(createWriteFragments(state, options, autoReads, keyConsumer), createCondition(options), null, false); } @@ -338,9 +344,9 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, { // TODO: Test case around this... Preconditions.checkState(conditions.isEmpty(), "No condition should exist without updates present"); - List reads = createNamedReads(options, state, ImmutableMap.of(), keySet::add); + List reads = createNamedReads(options, state, null, keySet::add); Keys txnKeys = toKeys(keySet); - TxnRead read = createTxnRead(reads, txnKeys, null); + TxnKeyRead read = createTxnRead(reads, null); Txn.Kind kind = txnKeys.size() == 1 && transactionalModeForSingleKey(txnKeys) == TransactionalMode.full && DatabaseDescriptor.getAccordEphemeralReadEnabledEnabled() @@ -349,11 +355,11 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, } else { - Map autoReads = new HashMap<>(); + Int2ObjectHashMap autoReads = new Int2ObjectHashMap<>(); AccordUpdate update = createUpdate(state, options, autoReads, keySet::add); List reads = createNamedReads(options, state, autoReads, keySet::add); Keys txnKeys = toKeys(keySet); - TxnRead read = createTxnRead(reads, txnKeys, null); + TxnKeyRead read = createTxnRead(reads, null); return new Txn.InMemory(txnKeys, read, TxnQuery.ALL, update); } } @@ -402,7 +408,7 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, ResultSetBuilder result = new ResultSetBuilder(resultMetadata, selectors, false); if (selectQuery.queries.size() == 1) { - FilteredPartition partition = data.get(TxnDataName.returning()); + TxnDataKeyValue partition = (TxnDataKeyValue)data.get(txnDataName(RETURNING)); boolean reversed = selectQuery.queries.get(0).isReversed(); if (partition != null) returningSelect.select.processPartition(partition.rowIterator(reversed), options, result, FBUtilities.nowInSeconds()); @@ -412,7 +418,7 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, long nowInSec = FBUtilities.nowInSeconds(); for (int i = 0; i < selectQuery.queries.size(); i++) { - FilteredPartition partition = data.get(TxnDataName.returning(i)); + TxnDataKeyValue partition = (TxnDataKeyValue)data.get(txnDataName(RETURNING, i)); boolean reversed = selectQuery.queries.get(i).isReversed(); if (partition != null) returningSelect.select.processPartition(partition.rowIterator(reversed), options, result, nowInSec); @@ -536,28 +542,31 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, checkTrue(select != null ^ returning != null, "Cannot specify both a full SELECT and a SELECT w/ LET references."); List preparedAssignments = new ArrayList<>(assignments.size()); - Map refSources = new HashMap<>(); - Set selectNames = new HashSet<>(); + Map refSources = new HashMap<>(); + Set selectNames = new HashSet<>(); + int userReadIndex = 0; + Map nameToTxnDataName = new HashMap<>(); for (SelectStatement.RawStatement select : assignments) { checkNotNull(select.parameters.refName, "Assignments must be named"); - TxnDataName name = TxnDataName.user(select.parameters.refName); - checkTrue(selectNames.add(name), DUPLICATE_TUPLE_NAME_MESSAGE, name.name()); + int name = txnDataName(USER, userReadIndex++); + nameToTxnDataName.put(select.parameters.refName, name); + checkTrue(selectNames.add(select.parameters.refName), DUPLICATE_TUPLE_NAME_MESSAGE, select.parameters.refName); validate(select); SelectStatement prepared = select.prepare(bindVariables); validate(prepared); NamedSelect namedSelect = new NamedSelect(name, prepared); - checkAtMostOneRowSpecified(namedSelect.select, "LET assignment " + name.name()); + checkAtMostOneRowSpecified(namedSelect.select, "LET assignment " + select.parameters.refName); preparedAssignments.add(namedSelect); refSources.put(name, new SelectReferenceSource(prepared)); } if (dataReferences != null) for (RowDataReference.Raw reference : dataReferences) - reference.resolveReference(refSources); + reference.resolveReference(refSources, nameToTxnDataName, userReadIndex++); NamedSelect returningSelect = null; if (select != null) @@ -565,7 +574,7 @@ public class TransactionStatement implements CQLStatement.CompositeCQLStatement, validate(select); SelectStatement prepared = select.prepare(bindVariables); validate(prepared); - returningSelect = new NamedSelect(TxnDataName.returning(), prepared); + returningSelect = new NamedSelect(txnDataName(RETURNING), prepared); checkAtMostOnePartitionSpecified(returningSelect.select, "returning select"); } diff --git a/src/java/org/apache/cassandra/cql3/transactions/RowDataReference.java b/src/java/org/apache/cassandra/cql3/transactions/RowDataReference.java index bb5b410ea1..6295e44578 100644 --- a/src/java/org/apache/cassandra/cql3/transactions/RowDataReference.java +++ b/src/java/org/apache/cassandra/cql3/transactions/RowDataReference.java @@ -27,20 +27,19 @@ import org.apache.cassandra.cql3.AssignmentTestable; import org.apache.cassandra.cql3.ColumnIdentifier; import org.apache.cassandra.cql3.ColumnSpecification; import org.apache.cassandra.cql3.FieldIdentifier; -import org.apache.cassandra.cql3.terms.Lists; -import org.apache.cassandra.cql3.terms.Maps; import org.apache.cassandra.cql3.QueryOptions; -import org.apache.cassandra.cql3.terms.Sets; -import org.apache.cassandra.cql3.terms.Term; -import org.apache.cassandra.cql3.terms.UserTypes; import org.apache.cassandra.cql3.VariableSpecifications; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.functions.types.utils.Bytes; import org.apache.cassandra.cql3.selection.Selectable; -import org.apache.cassandra.db.marshal.ListType; -import org.apache.cassandra.service.accord.txn.TxnDataName; +import org.apache.cassandra.cql3.terms.Lists; +import org.apache.cassandra.cql3.terms.Maps; +import org.apache.cassandra.cql3.terms.Sets; +import org.apache.cassandra.cql3.terms.Term; +import org.apache.cassandra.cql3.terms.UserTypes; import org.apache.cassandra.db.marshal.AbstractType; import org.apache.cassandra.db.marshal.CollectionType; +import org.apache.cassandra.db.marshal.ListType; import org.apache.cassandra.db.marshal.MapType; import org.apache.cassandra.db.marshal.SetType; import org.apache.cassandra.db.marshal.UserType; @@ -57,16 +56,18 @@ public class RowDataReference extends Term.NonTerminal public static final String CANNOT_FIND_TUPLE_MESSAGE = "Cannot resolve reference to tuple '%s'."; public static final String COLUMN_NOT_IN_TUPLE_MESSAGE = "Column '%s' does not exist in tuple '%s'."; - private final TxnDataName selectName; + private final String selectName; + private final int txnDataName; private final ColumnMetadata column; private final Term elementPath; private final CellPath fieldPath; - public RowDataReference(TxnDataName selectName, ColumnMetadata column, Term elementPath, CellPath fieldPath) + public RowDataReference(String selectName, int txnDataName, ColumnMetadata column, Term elementPath, CellPath fieldPath) { Preconditions.checkArgument(elementPath == null || fieldPath == null, "Cannot specify both element and field paths"); this.selectName = selectName; + this.txnDataName = txnDataName; this.column = column; this.elementPath = elementPath; this.fieldPath = fieldPath; @@ -182,14 +183,14 @@ public class RowDataReference extends Term.NonTerminal { Preconditions.checkState(elementPath == null || column.isComplex() || column.type.isFrozenCollection()); Preconditions.checkState(fieldPath == null || column.isComplex() || column.type.isUDT()); - return new TxnReference(selectName, column, bindCellPath(options)); + return new TxnReference(txnDataName, column, bindCellPath(options)); } public ColumnIdentifier getFullyQualifiedName() { // TODO: Make this more user-friendly... String path = fieldPath != null ? '.' + Bytes.toHexString(fieldPath.get(0)) : (elementPath == null ? "" : "[0x" + elementPath + ']'); - String fullName = selectName.name() + '.' + column.name.toString() + path; + String fullName = selectName + '.' + column.name.toString() + path; return new ColumnIdentifier(fullName, true); } @@ -206,7 +207,7 @@ public class RowDataReference extends Term.NonTerminal private boolean isResolved = false; - private TxnDataName tupleName; + private int tupleName; private ColumnMetadata column; private Term elementPath = null; private CellPath fieldPath = null; @@ -250,15 +251,16 @@ public class RowDataReference extends Term.NonTerminal isResolved = true; } - public void resolveReference(Map sources) + public void resolveReference(Map sources, Map nameToTxnDataName, int userReadIndex) { if (isResolved) return; - // root level name - tupleName = TxnDataName.user(tuple.toString()); + // root level name, use the one that was already created if it exists otherwise generate a new one + String rawTupleName = tuple.getText(); + tupleName = nameToTxnDataName.getOrDefault(rawTupleName, userReadIndex); ReferenceSource source = sources.get(tupleName); - checkNotNull(source, CANNOT_FIND_TUPLE_MESSAGE, tupleName.name()); + checkNotNull(source, CANNOT_FIND_TUPLE_MESSAGE, rawTupleName); if (selected == null) { @@ -267,7 +269,7 @@ public class RowDataReference extends Term.NonTerminal } column = source.getColumn(selected.toString()); - checkNotNull(column, COLUMN_NOT_IN_TUPLE_MESSAGE, selected.toString(), tupleName.name()); + checkNotNull(column, COLUMN_NOT_IN_TUPLE_MESSAGE, selected.toString(), rawTupleName); // TODO: confirm update partition key terms don't contain column references. This can't be done in prepare // because there can be intermediate functions (ie: pk=row.v+1 or pk=_add(row.v, 5)). Need a recursive Term visitor @@ -339,12 +341,12 @@ public class RowDataReference extends Term.NonTerminal public RowDataReference prepareAsReceiver() { checkResolved(); - return new RowDataReference(tupleName, column, elementPath, fieldPath); + return new RowDataReference(tuple.toString(), tupleName, column, elementPath, fieldPath); } private RowDataReference prepare(String keyspace, ColumnSpecification receiver, - TxnDataName selectName, + int txnDataName, ColumnMetadata column, Term elementPath, CellPath fieldPath) @@ -353,7 +355,7 @@ public class RowDataReference extends Term.NonTerminal throw new InvalidRequestException(String.format("Invalid reference type %s (%s) for \"%s\" of type %s", column.type, column.name, receiver.name, receiver.type.asCQL3Type())); - return new RowDataReference(selectName, column, elementPath, fieldPath); + return new RowDataReference(tuple.toString(), txnDataName, column, elementPath, fieldPath); } @Override diff --git a/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java b/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java index 6a5301946a..1714bb3c6f 100644 --- a/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java +++ b/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java @@ -53,6 +53,7 @@ import org.apache.cassandra.net.Verb; import org.apache.cassandra.schema.TableMetadata; import org.apache.cassandra.service.ClientState; import org.apache.cassandra.service.StorageProxy; +import org.apache.cassandra.service.reads.ReadCoordinator; import org.apache.cassandra.tcm.Epoch; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.transport.Dispatcher; @@ -222,6 +223,45 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR isTrackingWarnings()); } + public PartitionRangeReadCommand forSubRangeWithNowInSeconds(long nowInSec, AbstractBounds range, boolean isRangeContinuation) + { + // If we're not a continuation of whatever range we've previously queried, we should ignore the states of the + // DataLimits as it's either useless, or misleading. This is particularly important for GROUP BY queries, where + // DataLimits.CQLGroupByLimits.GroupByAwareCounter assumes that if GroupingState.hasClustering(), then we're in + // the middle of a group, but we can't make that assumption if we query and range "in advance" of where we are + // on the ring. + return create(serializedAtEpoch(), + isDigestQuery(), + digestVersion(), + acceptsTransient(), + allowsOutOfRangeReads(), + metadata(), + nowInSec, + columnFilter(), + rowFilter(), + isRangeContinuation ? limits() : limits().withoutState(), + dataRange().forSubRange(range), + indexQueryPlan(), + isTrackingWarnings()); + } + + public PartitionRangeReadCommand withNowInSec(long nowInSec) + { + return create(serializedAtEpoch(), + isDigestQuery(), + digestVersion(), + acceptsTransient(), + allowsOutOfRangeReads(), + metadata(), + nowInSec, + columnFilter(), + rowFilter(), + limits(), + dataRange(), + indexQueryPlan(), + isTrackingWarnings()); + } + public PartitionRangeReadCommand copy() { return create(serializedAtEpoch(), @@ -323,7 +363,7 @@ public class PartitionRangeReadCommand extends ReadCommand implements PartitionR public PartitionIterator execute(ConsistencyLevel consistency, ClientState state, Dispatcher.RequestTime requestTime) throws RequestExecutionException { - return StorageProxy.getRangeSlice(this, consistency, requestTime); + return StorageProxy.getRangeSlice(this, consistency, ReadCoordinator.DEFAULT, requestTime); } protected void recordLatency(TableMetrics metric, long latencyNanos) diff --git a/src/java/org/apache/cassandra/db/transform/FilteredPartitions.java b/src/java/org/apache/cassandra/db/transform/FilteredPartitions.java index 0e6934f76e..d8bdbcefd3 100644 --- a/src/java/org/apache/cassandra/db/transform/FilteredPartitions.java +++ b/src/java/org/apache/cassandra/db/transform/FilteredPartitions.java @@ -25,7 +25,8 @@ import org.apache.cassandra.db.partitions.PartitionIterator; import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; import org.apache.cassandra.db.rows.RowIterator; -public final class FilteredPartitions extends BasePartitions> implements PartitionIterator +// TODO (review): Why was this final? +public class FilteredPartitions extends BasePartitions> implements PartitionIterator { // wrap basic iterator for transformation FilteredPartitions(PartitionIterator input) diff --git a/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java b/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java index 17102b6cc0..b15fc8fbd7 100644 --- a/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java +++ b/src/java/org/apache/cassandra/metrics/KeyspaceMetrics.java @@ -108,6 +108,7 @@ public class KeyspaceMetrics public final Meter rangeMigrationUnexpectedFailures; public final Meter rangeMigrationDependencyLimitFailures; public final Meter mutationsRejectedOnWrongSystem; + public final Meter readsRejectedOnWrongSystem; /** Writes failed ideal consistency **/ public final Counter writeFailedIdealCL; /** Ideal CL write latency metrics */ @@ -259,6 +260,7 @@ public class KeyspaceMetrics rangeMigrationUnexpectedFailures = createKeyspaceMeter("RangeMigrationUnexpectedFailures"); rangeMigrationDependencyLimitFailures = createKeyspaceMeter("RangeMigratingDependencyLimitFailures"); mutationsRejectedOnWrongSystem = createKeyspaceMeter("MutationsRejectedOnWrongSystem"); + readsRejectedOnWrongSystem = createKeyspaceMeter("ReadsRejectedOnWrongSystem"); writeFailedIdealCL = createKeyspaceCounter("WriteFailedIdealCL"); idealCLWriteLatency = createLatencyMetrics("IdealCLWrite"); diff --git a/src/java/org/apache/cassandra/metrics/ReadRepairMetrics.java b/src/java/org/apache/cassandra/metrics/ReadRepairMetrics.java index 1adb5dcd7f..67d3a56ca7 100644 --- a/src/java/org/apache/cassandra/metrics/ReadRepairMetrics.java +++ b/src/java/org/apache/cassandra/metrics/ReadRepairMetrics.java @@ -30,6 +30,24 @@ public class ReadRepairMetrics private static final MetricNameFactory factory = new DefaultNameFactory(TYPE_NAME); public static final Meter repairedBlocking = Metrics.meter(factory.createMetricName("RepairedBlocking")); + + /** + * Non-transactional read did a blocking read repair via an Accord transaction. This is expected/normal if non-transactional + * reads are interoperating with Accord. + */ + public static final Meter repairedBlockingViaAccord = Metrics.meter(factory.createMetricName("RepairedBlockingViaAccord")); + + /** + * This should be zero if you are trying to run Accord in a 100% correct way and interoperating with non-transactional writes. + * + * An Accord transaction read at QUORUM and ended up having to do BRR to make something it read monotonic. While it + * will be monotonic this is not 100% deterministic for transaction recovery because different Accord coordinators could + * read different things when computing a transaction's writes. + * + * If Accord is operating in TransactionalMode.full and the range is migrated then this metric will be zero just + * because Accord is reading at ONE not QUORUM and there are should be no non-transactional writes anywyas. + */ + public static final Meter repairedBlockingFromAccord = Metrics.meter(factory.createMetricName("RepairedBlockingFromAccord")); public static final Meter reconcileRead = Metrics.meter(factory.createMetricName("ReconcileRead")); /** @deprecated See CASSANDRA-13910 */ diff --git a/src/java/org/apache/cassandra/metrics/TableMetrics.java b/src/java/org/apache/cassandra/metrics/TableMetrics.java index b490a26299..380333058c 100644 --- a/src/java/org/apache/cassandra/metrics/TableMetrics.java +++ b/src/java/org/apache/cassandra/metrics/TableMetrics.java @@ -197,6 +197,7 @@ public class TableMetrics public final TableMeter accordRepairUnexpectedFailures; public final TableMeter accordRepairDependencyLimitFailures; public final TableMeter mutationsRejectedOnWrongSystem; + public final TableMeter readsRejectedOnWrongSystem; /** percent of the data that is repaired */ public final Gauge percentRepaired; /** Reports the size of sstables in repaired, unrepaired, and any ongoing repair buckets */ @@ -819,6 +820,7 @@ public class TableMetrics accordRepairUnexpectedFailures = createTableMeter("AccordRepairUnexpectedFailures", cfs.keyspace.metric.rangeMigrationUnexpectedFailures); accordRepairDependencyLimitFailures = createTableMeter("AccordRepairDependencyLimitFaiures", cfs.keyspace.metric.rangeMigrationDependencyLimitFailures); mutationsRejectedOnWrongSystem = createTableMeter("MutationsRejectedOnWrongSystem", cfs.keyspace.metric.mutationsRejectedOnWrongSystem); + readsRejectedOnWrongSystem = createTableMeter("ReadsRejectedOnWrongSystem", cfs.keyspace.metric.mutationsRejectedOnWrongSystem); repairsStarted = createTableCounter("RepairJobsStarted"); repairsCompleted = createTableCounter("RepairJobsCompleted"); diff --git a/src/java/org/apache/cassandra/schema/TableParams.java b/src/java/org/apache/cassandra/schema/TableParams.java index 569fc27b1c..d3d059ab79 100644 --- a/src/java/org/apache/cassandra/schema/TableParams.java +++ b/src/java/org/apache/cassandra/schema/TableParams.java @@ -35,18 +35,39 @@ import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.service.accord.fastpath.FastPathStrategy; import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; -import org.apache.cassandra.tcm.serialization.MetadataSerializer; -import org.apache.cassandra.tcm.serialization.Version; import org.apache.cassandra.service.reads.PercentileSpeculativeRetryPolicy; import org.apache.cassandra.service.reads.SpeculativeRetryPolicy; import org.apache.cassandra.service.reads.repair.ReadRepairStrategy; +import org.apache.cassandra.tcm.serialization.MetadataSerializer; +import org.apache.cassandra.tcm.serialization.Version; import org.apache.cassandra.utils.BloomCalculations; import org.apache.cassandra.utils.ByteBufferUtil; import static java.lang.String.format; import static java.util.stream.Collectors.toMap; -import static org.apache.cassandra.schema.TableParams.Option.*; +import static com.google.common.base.Preconditions.checkNotNull; import static org.apache.cassandra.db.TypeSizes.sizeof; +import static org.apache.cassandra.schema.TableParams.Option.ADDITIONAL_WRITE_POLICY; +import static org.apache.cassandra.schema.TableParams.Option.ALLOW_AUTO_SNAPSHOT; +import static org.apache.cassandra.schema.TableParams.Option.BLOOM_FILTER_FP_CHANCE; +import static org.apache.cassandra.schema.TableParams.Option.CACHING; +import static org.apache.cassandra.schema.TableParams.Option.CDC; +import static org.apache.cassandra.schema.TableParams.Option.COMMENT; +import static org.apache.cassandra.schema.TableParams.Option.COMPACTION; +import static org.apache.cassandra.schema.TableParams.Option.COMPRESSION; +import static org.apache.cassandra.schema.TableParams.Option.CRC_CHECK_CHANCE; +import static org.apache.cassandra.schema.TableParams.Option.DEFAULT_TIME_TO_LIVE; +import static org.apache.cassandra.schema.TableParams.Option.EXTENSIONS; +import static org.apache.cassandra.schema.TableParams.Option.FAST_PATH; +import static org.apache.cassandra.schema.TableParams.Option.GC_GRACE_SECONDS; +import static org.apache.cassandra.schema.TableParams.Option.INCREMENTAL_BACKUPS; +import static org.apache.cassandra.schema.TableParams.Option.MAX_INDEX_INTERVAL; +import static org.apache.cassandra.schema.TableParams.Option.MEMTABLE; +import static org.apache.cassandra.schema.TableParams.Option.MEMTABLE_FLUSH_PERIOD_IN_MS; +import static org.apache.cassandra.schema.TableParams.Option.MIN_INDEX_INTERVAL; +import static org.apache.cassandra.schema.TableParams.Option.PENDING_DROP; +import static org.apache.cassandra.schema.TableParams.Option.READ_REPAIR; +import static org.apache.cassandra.schema.TableParams.Option.SPECULATIVE_RETRY; import static org.apache.cassandra.utils.LocalizeString.toLowerCaseLocalized; public final class TableParams @@ -136,6 +157,7 @@ public final class TableParams transactionalMode = builder.transactionalMode != null ? builder.transactionalMode : TransactionalMode.off; transactionalMigrationFrom = builder.transactionalMigrationFrom; pendingDrop = builder.pendingDrop; + checkNotNull(transactionalMigrationFrom); } public static Builder builder() diff --git a/src/java/org/apache/cassandra/service/StorageProxy.java b/src/java/org/apache/cassandra/service/StorageProxy.java index 555937dd96..49fad364a5 100644 --- a/src/java/org/apache/cassandra/service/StorageProxy.java +++ b/src/java/org/apache/cassandra/service/StorageProxy.java @@ -82,6 +82,7 @@ import org.apache.cassandra.db.partitions.PartitionUpdate; import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; import org.apache.cassandra.db.rows.RowIterator; import org.apache.cassandra.db.view.ViewUtils; +import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; import org.apache.cassandra.exceptions.CasWriteTimeoutException; import org.apache.cassandra.exceptions.CasWriteUnknownResultException; @@ -133,8 +134,12 @@ import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.IAccordService; import org.apache.cassandra.service.accord.IAccordService.AsyncTxnResult; import org.apache.cassandra.service.accord.txn.TxnData; -import org.apache.cassandra.service.accord.txn.TxnDataName; +import org.apache.cassandra.service.accord.txn.TxnDataKeyValue; +import org.apache.cassandra.service.accord.txn.TxnDataValue; +import org.apache.cassandra.service.accord.txn.TxnKeyRead; import org.apache.cassandra.service.accord.txn.TxnQuery; +import org.apache.cassandra.service.accord.txn.TxnRangeRead; +import org.apache.cassandra.service.accord.txn.TxnRangeReadResult; import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.service.accord.txn.TxnResult; import org.apache.cassandra.service.consensus.TransactionalMode; @@ -201,6 +206,7 @@ import static org.apache.cassandra.service.BatchlogResponseHandler.BatchlogClean import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.RETRY_NEW_PROTOCOL; import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.casResult; import static org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult.serialReadResult; +import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.range_read; import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.retry_new_protocol; import static org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper.mutateWithAccordAsync; import static org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper.splitMutationsIntoAccordAndNormal; @@ -366,7 +372,7 @@ public class StorageProxy implements StorageProxyMBean { ClusterMetadata cm = ClusterMetadata.current(); TableMetadata metadata = Schema.instance.validateTable(keyspaceName, cfName); - ConsensusRoutingDecision decision = consensusRouting(metadata, key, consistencyForPaxos, requestTime, true); + ConsensusRoutingDecision decision = consensusRouting(cm, metadata, key, consistencyForPaxos, requestTime, true); switch (decision) { case paxosV2: @@ -2129,11 +2135,12 @@ public class StorageProxy implements StorageProxyMBean return metadata.myNodeState() == NodeState.JOINED; } - private static ConsensusRoutingDecision consensusRouting(TableMetadata metadata, DecoratedKey partitionKey, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, boolean isForWrite) + private static ConsensusRoutingDecision consensusRouting(ClusterMetadata cm, TableMetadata metadata, DecoratedKey partitionKey, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, boolean isForWrite) { if (metadata.keyspace.equals(SchemaConstants.METADATA_KEYSPACE_NAME)) return ConsensusRoutingDecision.paxosV2; - return ConsensusRequestRouter.instance.routeAndMaybeMigrate(partitionKey, + return ConsensusRequestRouter.instance.routeAndMaybeMigrate(cm, + partitionKey, metadata.id, consistencyLevel, requestTime, @@ -2149,7 +2156,7 @@ public class StorageProxy implements StorageProxyMBean { ClusterMetadata cm = ClusterMetadata.current(); SinglePartitionReadCommand command = group.queries.get(0); - ConsensusRoutingDecision decision = consensusRouting(group.metadata(), command.partitionKey(), consistencyLevel, requestTime, false); + ConsensusRoutingDecision decision = consensusRouting(cm, group.metadata(), command.partitionKey(), consistencyLevel, requestTime, false); switch (decision) { case paxosV2: @@ -2191,15 +2198,33 @@ public class StorageProxy implements StorageProxyMBean return null; } + + public static AsyncTxnResult readWithAccord(ClusterMetadata cm, PartitionRangeReadCommand command, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime) + { + if (consistencyLevel != null && !IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS.contains(consistencyLevel)) + throw new InvalidRequestException(consistencyLevel + " is not supported by Accord"); + + TableMetadata tableMetadata = getTableMetadata(cm, command.metadata().id); + TableParams tableParams = tableMetadata.params; + Range readRange = new Range<>(command.dataRange().startKey().getToken(), command.dataRange().stopKey().getToken()); + consistencyLevel = tableParams.transactionalMode.readCLForStrategy(tableParams.transactionalMigrationFrom, consistencyLevel, cm, tableMetadata.id, readRange); + TxnRead read = new TxnRangeRead(command, consistencyLevel); + Txn.Kind kind = EphemeralRead; + Txn txn = new Txn.InMemory(kind, read.keys(), read, TxnQuery.RANGE_QUERY, null); + IAccordService accordService = AccordService.instance(); + return accordService.coordinateAsync(txn, consistencyLevel, requestTime); + } + private static ConsensusAttemptResult readWithAccord(ClusterMetadata cm, SinglePartitionReadCommand.Group group, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime) { if (consistencyLevel != null && !IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS.contains(consistencyLevel)) throw new InvalidRequestException(consistencyLevel + " is not supported by Accord"); + // If the non-SERIAL write strategy is sending all writes through Accord there is no need to use the supplied consistency // level since Accord will manage reading safely TransactionalMode transactionalMode = group.metadata().params.transactionalMode; consistencyLevel = consistencyLevelForAccordRead(cm, group, consistencyLevel); - TxnRead read = TxnRead.createSerialRead(group.queries, consistencyLevel); + TxnKeyRead read = TxnKeyRead.createSerialRead(group.queries, consistencyLevel); Txn.Kind kind = Read; if (transactionalMode == TransactionalMode.full && getAccordEphemeralReadEnabledEnabled() && group.queries.size() == 1 && group.metadata().params.transactionalMigrationFrom == none) kind = EphemeralRead; @@ -2217,6 +2242,8 @@ public class StorageProxy implements StorageProxyMBean // TODO (required): Converge on a single approach to RETRY_NEW_PROTOCOL, this works for now because reads don't support it anyways if (txnResult.kind() == retry_new_protocol) return RETRY_NEW_PROTOCOL; + if (txnResult.kind() == range_read) + return serialReadResult(((TxnRangeReadResult)txnResult).partitions.get()); TxnData data = (TxnData) txnResult; if (data.isEmpty()) @@ -2225,7 +2252,7 @@ public class StorageProxy implements StorageProxyMBean } else if (data.size() == 1) { - FilteredPartition value = data.values().iterator().next(); + TxnDataKeyValue value = ((TxnDataKeyValue)data.values().iterator().next()); return serialReadResult(PartitionIterators.singletonIterator(value.rowIterator(isQueryReversed.test(0)))); } else @@ -2234,10 +2261,11 @@ public class StorageProxy implements StorageProxyMBean List partitionIterators = new ArrayList<>(numQueries); for (int i = 0; i < numQueries; i++) partitionIterators.add(null); - for (Map.Entry e : data.entrySet()) + for (Map.Entry e : data.entrySet()) { - int queryIndex = Integer.valueOf(e.getKey().part(0)); - partitionIterators.set(queryIndex, PartitionIterators.singletonIterator(e.getValue().rowIterator(isQueryReversed.test(queryIndex)))); + int queryIndex = e.getKey(); + TxnDataKeyValue value = ((TxnDataKeyValue)e.getValue()); + partitionIterators.set(queryIndex, PartitionIterators.singletonIterator(value.rowIterator(isQueryReversed.test(queryIndex)))); } return serialReadResult(partitionIterators.size() == 1 ? partitionIterators.get(0) : PartitionIterators.concat(partitionIterators)); } @@ -2360,7 +2388,7 @@ public class StorageProxy implements StorageProxyMBean // throw new UnsupportedOperationException("Live migration is not supported, can't safely read when migrating from " + transactionalMigrationFromMode + " to " + transactionalMode); PartitionIterator result; - if (transactionalMode.readsThroughAccord && coordinator.isEventuallyConsistent()) + if (transactionalMode.nonSerialReadsThroughAccord && coordinator.isEventuallyConsistent()) { ConsensusAttemptResult consensusAttemptResult = readWithAccord(cm, group, consistencyLevel, requestTime); checkState(!consensusAttemptResult.shouldRetryOnNewConsensusProtocol, "Live migration is not supported with non-SERIAL reads yet"); @@ -2368,9 +2396,10 @@ public class StorageProxy implements StorageProxyMBean } else result = fetchRows(group.queries, consistencyLevel, coordinator, requestTime); + // Note that the only difference between the command in a group must be the partition key on which // they applied. - boolean enforceStrictLiveness = group.queries.get(0).metadata().enforceStrictLiveness(); + boolean enforceStrictLiveness = tableMetadata.enforceStrictLiveness(); // If we have more than one command, then despite each read command honoring the limit, the total result // might not honor it and so we should enforce it if (group.queries.size() > 1) @@ -2640,6 +2669,7 @@ public class StorageProxy implements StorageProxyMBean public static PartitionIterator getRangeSlice(PartitionRangeReadCommand command, ConsistencyLevel consistencyLevel, + ReadCoordinator readCoordinator, Dispatcher.RequestTime requestTime) { if (DatabaseDescriptor.getPartitionDenylistEnabled() && DatabaseDescriptor.getDenylistRangeReadsEnabled()) @@ -2654,7 +2684,7 @@ public class StorageProxy implements StorageProxyMBean tokens)); } } - return RangeCommands.partitions(command, consistencyLevel, requestTime); + return RangeCommands.partitions(command, consistencyLevel, readCoordinator, requestTime); } public Map> getSchemaVersions() diff --git a/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java b/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java index 4394fbb29a..61b1a9eca7 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java +++ b/src/java/org/apache/cassandra/service/accord/AccordCommandStore.java @@ -72,8 +72,8 @@ import org.apache.cassandra.utils.concurrent.AsyncPromise; import org.apache.cassandra.utils.concurrent.Promise; import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException; -import static accord.primitives.SaveStatus.Applying; import static accord.local.KeyHistory.SYNC; +import static accord.primitives.SaveStatus.Applying; import static accord.primitives.Status.Committed; import static accord.primitives.Status.Invalidated; import static accord.primitives.Status.Truncated; diff --git a/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java b/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java index 1172b8ea5f..8ffbb3ccb3 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java +++ b/src/java/org/apache/cassandra/service/accord/AccordMessageSink.java @@ -177,8 +177,8 @@ public class AccordMessageSink implements MessageSink builder.put(AccordMessageType.INTEROP_READ_REPAIR_RSP, Verb.ACCORD_INTEROP_READ_REPAIR_RSP); builder.put(AccordMessageType.INTEROP_COMMIT_MINIMAL_REQ, Verb.ACCORD_INTEROP_COMMIT_REQ); builder.put(AccordMessageType.INTEROP_COMMIT_MAXIMAL_REQ, Verb.ACCORD_INTEROP_COMMIT_REQ); - builder.put(AccordMessageType.INTEROP_APPLY_MINIMAL_REQ, Verb.ACCORD_APPLY_REQ); - builder.put(AccordMessageType.INTEROP_APPLY_MAXIMAL_REQ, Verb.ACCORD_APPLY_REQ); + builder.put(AccordMessageType.INTEROP_APPLY_MINIMAL_REQ, Verb.ACCORD_INTEROP_APPLY_REQ); + builder.put(AccordMessageType.INTEROP_APPLY_MAXIMAL_REQ, Verb.ACCORD_INTEROP_APPLY_REQ); mapping = builder.build(); for (MessageType type : Iterables.concat(AccordMessageType.values, MessageType.values)) @@ -208,7 +208,7 @@ public class AccordMessageSink implements MessageSink { MessageType type = request.type(); if (type != null) - return getVerb(request.type()); + return getVerb(type); return null; } diff --git a/src/java/org/apache/cassandra/service/accord/AccordObjectSizes.java b/src/java/org/apache/cassandra/service/accord/AccordObjectSizes.java index 315649e304..8a7c42d97a 100644 --- a/src/java/org/apache/cassandra/service/accord/AccordObjectSizes.java +++ b/src/java/org/apache/cassandra/service/accord/AccordObjectSizes.java @@ -24,17 +24,15 @@ import java.util.function.ToLongFunction; import accord.api.Key; import accord.api.Result; import accord.api.RoutingKey; -import accord.local.StoreParticipants; -import accord.local.cfk.CommandsForKey; -import accord.local.cfk.CommandsForKey.TxnInfo; import accord.impl.TimestampsForKey; import accord.local.Command; import accord.local.Command.WaitingOn; -import accord.local.cfk.CommandsForKey.TxnInfoExtra; import accord.local.CommonAttributes; import accord.local.Node; -import accord.primitives.SaveStatus; -import accord.primitives.Status; +import accord.local.StoreParticipants; +import accord.local.cfk.CommandsForKey; +import accord.local.cfk.CommandsForKey.TxnInfo; +import accord.local.cfk.CommandsForKey.TxnInfoExtra; import accord.primitives.AbstractKeys; import accord.primitives.AbstractRanges; import accord.primitives.Ballot; @@ -52,7 +50,9 @@ import accord.primitives.RangeDeps; import accord.primitives.Ranges; import accord.primitives.Routable.Domain; import accord.primitives.RoutingKeys; +import accord.primitives.SaveStatus; import accord.primitives.Seekables; +import accord.primitives.Status; import accord.primitives.Timestamp; import accord.primitives.Txn.Kind; import accord.primitives.TxnId; @@ -62,6 +62,7 @@ import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.api.AccordRoutingKey; import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey; import org.apache.cassandra.service.accord.api.PartitionKey; +import org.apache.cassandra.service.accord.serializers.CommandSerializers; import org.apache.cassandra.service.accord.serializers.WaitingOnSerializer; import org.apache.cassandra.service.accord.txn.AccordUpdate; import org.apache.cassandra.service.accord.txn.TxnQuery; @@ -155,7 +156,7 @@ public class AccordObjectSizes + key(route.homeKey()); } - private static long rangesOnly(AbstractRanges ranges) + public static long ranges(AbstractRanges ranges) { long size = ObjectSizes.sizeOfReferenceArray(ranges.size()); for (int i=0, mi=ranges.size(); i> results = new ArrayList<>(); Seekables keys = txn.read().keys(); keys.forEach(key -> { @@ -280,9 +280,9 @@ public class AccordInteropExecution implements ReadCoordinator, MaximalCommitSen { try (RowIterator partition = iterator.next()) { - FilteredPartition filtered = FilteredPartition.create(partition); - if (filtered.hasRows() || command.selectsFullPartition()) - result.put(fragment.txnDataName(), filtered); + TxnDataKeyValue value = new TxnDataKeyValue(partition); + if (value.hasRows() || command.selectsFullPartition()) + result.put(fragment.txnDataName(), value); } } } @@ -357,6 +357,7 @@ public class AccordInteropExecution implements ReadCoordinator, MaximalCommitSen UnrecoverableRepairUpdate repairUpdate = (UnrecoverableRepairUpdate)txn.update(); // TODO (expected): We should send the read in the same message as the commit. This requires refactor ReadData.Kind so that it doesn't specify the ordinal encoding // and can be extended similar to MessageType which allows additional types not from Accord to be added + // This commit won't necessarily execute before the interop read repair message so there could be an insufficient which is fine for (Node.Id to : executeTopology.nodes()) node.send(to, new Commit(Kind.StableFastPath, to, coordinateTopology, allTopologies, txnId, txn, route, Ballot.ZERO, executeAt, deps, (ReadTxnData) null)); repairUpdate.runBRR(AccordInteropExecution.this); diff --git a/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializers.java b/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializers.java index 0f109bd83d..0c86ee7747 100644 --- a/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializers.java +++ b/src/java/org/apache/cassandra/service/accord/serializers/ReadDataSerializers.java @@ -41,7 +41,6 @@ import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.service.accord.txn.TxnData; -import org.apache.cassandra.service.accord.txn.TxnResult; import static org.apache.cassandra.db.TypeSizes.sizeof; import static org.apache.cassandra.utils.NullableSerializer.deserializeNullable; @@ -86,7 +85,6 @@ public class ReadDataSerializers CommandSerializers.partialTxn.serialize(msg.txn, out, version); DepsSerializer.partialDeps.serialize(msg.deps, out, version); CommandSerializers.writes.serialize(msg.writes, out, version); - TxnResult.serializer.serialize((TxnResult) msg.result, out, version); } @Override @@ -100,8 +98,8 @@ public class ReadDataSerializers CommandSerializers.partialTxn.deserialize(in, version), DepsSerializer.partialDeps.deserialize(in, version), CommandSerializers.writes.deserialize(in, version), - TxnResult.serializer.deserialize(in, version)); - } + CommandSerializers.APPLIED); + } @Override public long serializedSize(ApplyThenWaitUntilApplied msg, int version) @@ -112,8 +110,7 @@ public class ReadDataSerializers + KeySerializers.fullRoute.serializedSize(msg.route, version) + CommandSerializers.partialTxn.serializedSize(msg.txn, version) + DepsSerializer.partialDeps.serializedSize(msg.deps, version) - + CommandSerializers.writes.serializedSize(msg.writes, version) - + TxnResult.serializer.serializedSize((TxnData)msg.result, version); + + CommandSerializers.writes.serializedSize(msg.writes, version); } } diff --git a/src/java/org/apache/cassandra/service/accord/txn/AbstractKeySorted.java b/src/java/org/apache/cassandra/service/accord/txn/AbstractKeySorted.java index c02dc41a80..90cd946d98 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/AbstractKeySorted.java +++ b/src/java/org/apache/cassandra/service/accord/txn/AbstractKeySorted.java @@ -28,7 +28,6 @@ import java.util.stream.Collectors; import com.google.common.annotations.VisibleForTesting; import com.google.common.collect.Iterators; -import accord.api.Key; import accord.primitives.Keys; import org.apache.cassandra.service.accord.api.PartitionKey; @@ -45,7 +44,6 @@ public abstract class AbstractKeySorted implements Iterable public AbstractKeySorted(T[] items) { this.items = items; - validateOrder(); this.itemKeys = extractItemKeys(); } @@ -55,13 +53,12 @@ public abstract class AbstractKeySorted implements Iterable items.toArray(arr); Arrays.sort(arr, this::compare); this.items = arr; - validateOrder(); this.itemKeys = extractItemKeys(); } private Keys extractItemKeys() { - Key[] keys = new Key[size()]; + PartitionKey[] keys = new PartitionKey[items.length]; for (int i = 0 ; i < keys.length ; ++i) keys[i] = getKey(items[i]); return Keys.ofSorted(keys); diff --git a/src/java/org/apache/cassandra/service/accord/txn/AbstractSerialized.java b/src/java/org/apache/cassandra/service/accord/txn/AbstractSerialized.java index e790ff6062..85e90132ac 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/AbstractSerialized.java +++ b/src/java/org/apache/cassandra/service/accord/txn/AbstractSerialized.java @@ -20,7 +20,7 @@ package org.apache.cassandra.service.accord.txn; import java.nio.ByteBuffer; import java.util.Objects; - +import javax.annotation.Nullable; import javax.annotation.concurrent.NotThreadSafe; import org.apache.cassandra.io.IVersionedSerializer; @@ -32,17 +32,17 @@ import org.apache.cassandra.service.accord.AccordSerializers; @NotThreadSafe public abstract class AbstractSerialized { - private final ByteBuffer bytes; - private T memoized = null; + private @Nullable final ByteBuffer bytes; + private @Nullable T memoized = null; - public AbstractSerialized(ByteBuffer bytes) + public AbstractSerialized(@Nullable ByteBuffer bytes) { this.bytes = bytes; } - public AbstractSerialized(T value) + public AbstractSerialized(@Nullable T value) { - this.bytes = AccordSerializers.serialize(value, serializer()); + this.bytes = value != null ? AccordSerializers.serialize(value, serializer()) : null; this.memoized = value; } @@ -51,14 +51,16 @@ public abstract class AbstractSerialized { if (this == o) return true; if (o == null || getClass() != o.getClass()) return false; + AbstractSerialized that = (AbstractSerialized) o; - return bytes.equals(that.bytes); + + return Objects.equals(bytes, that.bytes); } @Override public int hashCode() { - return Objects.hash(bytes); + return bytes != null ? bytes.hashCode() : 0; } @Override @@ -72,7 +74,7 @@ public abstract class AbstractSerialized protected T get() { T result = memoized; - if (result == null) + if (result == null && bytes != null) memoized = result = AccordSerializers.deserialize(bytes, serializer()); return result; } diff --git a/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java b/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java index e6a977955a..5792b4a50b 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java +++ b/src/java/org/apache/cassandra/service/accord/txn/AccordUpdateParameters.java @@ -26,7 +26,6 @@ import com.google.common.collect.ImmutableMap; import org.apache.cassandra.cql3.QueryOptions; import org.apache.cassandra.cql3.UpdateParameters; import org.apache.cassandra.db.DecoratedKey; -import org.apache.cassandra.db.partitions.FilteredPartition; import org.apache.cassandra.db.partitions.Partition; import org.apache.cassandra.schema.TableMetadata; import org.apache.cassandra.service.ClientState; @@ -74,17 +73,19 @@ public class AccordUpdateParameters private Map prefetchRow(TableMetadata metadata, DecoratedKey dk, int index) { - for (Map.Entry e : data.entrySet()) + for (Map.Entry e : data.entrySet()) { - TxnDataName name = e.getKey(); - switch (name.getKind()) + int name = e.getKey(); + TxnDataKeyValue value = (TxnDataKeyValue)e.getValue(); + switch (TxnData.txnDataNameKind(name)) { case CAS_READ: checkState(data.entrySet().size() == 1, "CAS read should only have one entry"); - return ImmutableMap.of(dk, e.getValue()); + return ImmutableMap.of(dk, value); case AUTO_READ: - if (name.atIndex(index)) - return ImmutableMap.of(name.getDecoratedKey(metadata), e.getValue()); + // TODO (review): Is this the right DK being passed into that matches what we used to store in TxnDataName + if (TxnData.txnDataNameIndex(name) == index) + return ImmutableMap.of(dk, value); default: } } diff --git a/src/java/org/apache/cassandra/service/accord/txn/RetryWithNewProtocolResult.java b/src/java/org/apache/cassandra/service/accord/txn/RetryWithNewProtocolResult.java index d5c02a596b..6156b6a72f 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/RetryWithNewProtocolResult.java +++ b/src/java/org/apache/cassandra/service/accord/txn/RetryWithNewProtocolResult.java @@ -18,10 +18,6 @@ package org.apache.cassandra.service.accord.txn; -import java.io.IOException; - -import org.apache.cassandra.io.util.DataInputPlus; -import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.tcm.Epoch; import org.apache.cassandra.utils.ObjectSizes; @@ -29,9 +25,9 @@ import org.apache.cassandra.utils.ObjectSizes; * Potentially returned by any transaction that tries to execute in an Epoch * where the range has migrated away from Accord */ -public class RetryWithNewProtocolResult extends TxnResult +public class RetryWithNewProtocolResult implements TxnResult { - private static final long SIZE = ObjectSizes.measure(new RetryWithNewProtocolResult(Epoch.FIRST)); + private static final long EMPTY_SIZE = ObjectSizes.measure(new RetryWithNewProtocolResult(null)); public final Epoch epoch; @@ -49,27 +45,6 @@ public class RetryWithNewProtocolResult extends TxnResult @Override public long estimatedSizeOnHeap() { - return SIZE; + return EMPTY_SIZE + epoch.estimatedSizeOnHeap(); } - - public static final TxnResultSerializer serializer = new TxnResultSerializer() - { - @Override - public void serialize(RetryWithNewProtocolResult retry, DataOutputPlus out, int version) throws IOException - { - Epoch.messageSerializer.serialize(retry.epoch, out, version); - } - - @Override - public RetryWithNewProtocolResult deserialize(DataInputPlus in, int version) throws IOException - { - return new RetryWithNewProtocolResult(Epoch.messageSerializer.deserialize(in, version)); - } - - @Override - public long serializedSize(RetryWithNewProtocolResult retry, int version) - { - return Epoch.messageSerializer.serializedSize(retry.epoch, version); - } - }; } diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnCondition.java b/src/java/org/apache/cassandra/service/accord/txn/TxnCondition.java index 46f59f0936..c3780502be 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnCondition.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnCondition.java @@ -52,7 +52,8 @@ import org.apache.cassandra.utils.ByteBufferUtil; import static com.google.common.base.Preconditions.checkNotNull; import static org.apache.cassandra.service.accord.AccordSerializers.clusteringSerializer; -import static org.apache.cassandra.service.accord.txn.TxnRead.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataName; import static org.apache.cassandra.utils.CollectionSerializers.deserializeList; import static org.apache.cassandra.utils.CollectionSerializers.serializeCollection; import static org.apache.cassandra.utils.CollectionSerializers.serializeList; @@ -332,8 +333,8 @@ public abstract class TxnCondition public boolean applies(@Nonnull TxnData data) { checkNotNull(data); - FilteredPartition partition = data.get(CAS_READ); - Row row = partition != null ? partition.getRow(clustering) : null; + TxnDataKeyValue value = (TxnDataKeyValue)data.get(txnDataName(CAS_READ)); + Row row = value != null ? value.getRow(clustering) : null; for (Bound bound : bounds) { if (!bound.appliesTo(row)) diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnData.java b/src/java/org/apache/cassandra/service/accord/txn/TxnData.java index 6ca4fc5083..8eed778a99 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnData.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnData.java @@ -19,81 +19,111 @@ package org.apache.cassandra.service.accord.txn; import java.io.IOException; -import java.util.Collection; -import java.util.HashMap; -import java.util.Iterator; import java.util.Map; -import java.util.Set; - -import com.google.common.collect.Maps; import accord.api.Data; +import org.agrona.collections.Int2ObjectHashMap; import org.apache.cassandra.db.EmptyIterators; import org.apache.cassandra.db.SinglePartitionReadCommand; import org.apache.cassandra.db.TypeSizes; -import org.apache.cassandra.db.filter.ColumnFilter; -import org.apache.cassandra.db.partitions.FilteredPartition; import org.apache.cassandra.db.partitions.PartitionIterators; -import org.apache.cassandra.db.rows.DeserializationHelper; -import org.apache.cassandra.db.rows.Row; -import org.apache.cassandra.db.rows.UnfilteredRowIterator; -import org.apache.cassandra.db.rows.UnfilteredRowIteratorSerializer; -import org.apache.cassandra.db.rows.UnfilteredRowIterators; import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; -import org.apache.cassandra.schema.Schema; -import org.apache.cassandra.schema.TableId; -import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.utils.CollectionSerializers; +import org.apache.cassandra.utils.Int32Serializer; import org.apache.cassandra.utils.NullableSerializer; import org.apache.cassandra.utils.ObjectSizes; +import static accord.utils.Invariants.checkArgument; import static org.apache.cassandra.service.accord.txn.TxnResult.Kind.txn_data; -public class TxnData extends TxnResult implements Data, Iterable +/** + * Fairly generic holder for result values for Accord txns as well as data exchange during Accord txn execution + * when read results are returned to the coordinator to compute query results and writes. + */ +public class TxnData extends Int2ObjectHashMap implements TxnResult, Data { private static final long EMPTY_SIZE = ObjectSizes.measure(new TxnData()); - private final Map data; + private static final int TXN_DATA_NAME_INDEX_BITS = 32 - 6; + private static final int TXN_DATA_NAME_INDEX_MASK = ~(~0 << TXN_DATA_NAME_INDEX_BITS); + public static final int TXN_DATA_NAME_INDEX_MAX = ((1 << TXN_DATA_NAME_INDEX_BITS) - 1); - public TxnData(Map data) + public enum TxnDataNameKind { - this.data = data; + USER((byte) 0), + RETURNING((byte) 1), + AUTO_READ((byte) 2), + CAS_READ((byte) 3); + + private final byte value; + + TxnDataNameKind(byte value) + { + this.value = value; + } + + public static TxnDataNameKind from(byte b) + { + switch (b) + { + case 0: + return USER; + case 1: + return RETURNING; + case 2: + return AUTO_READ; + case 3: + return CAS_READ; + default: + throw new IllegalArgumentException("Unknown kind: " + b); + } + } } - public TxnData() + public static int txnDataName(TxnDataNameKind kind, int index) { - this(new HashMap<>()); + checkArgument(index >= 0 && index <= TXN_DATA_NAME_INDEX_MAX); + int kindInt = (int)(((long)kind.value) << TXN_DATA_NAME_INDEX_BITS); + return kindInt | index; } - public void put(TxnDataName name, FilteredPartition partition) + public static int txnDataName(TxnDataNameKind kind) { - data.put(name, partition); + return txnDataName(kind, 0); } - public FilteredPartition get(TxnDataName name) + public static TxnDataNameKind txnDataNameKind(int txnDataName) { - return data.get(name); + int kind = txnDataName >>> TXN_DATA_NAME_INDEX_BITS; + return TxnDataNameKind.from((byte)kind); } - public Set> entrySet() + public static int txnDataNameIndex(int txnDataName) { - return data.entrySet(); + return txnDataName & TXN_DATA_NAME_INDEX_MASK; } - public Collection values() + public TxnData() {} + + private TxnData(int size) { - return data.values(); + super(size, 0.65f); } - public int size() + public static TxnData of(int key, TxnDataValue value) { - return data.size(); + TxnData result = newWithExpectedSize(1); + result.put(key, value); + return result; } - public boolean isEmpty() + public static TxnData newWithExpectedSize(int size) { - return data.isEmpty(); + checkArgument(size >= 0, "size can't be negative"); + size = Math.max(4, size); + return new TxnData(size < 1073741824 ? (int)((float)size / 0.75F + 1.0F) : Integer.MAX_VALUE); } @Override @@ -101,8 +131,9 @@ public class TxnData extends TxnResult implements Data, Iterable e : that.entrySet()) + merged.merge(e.getKey(), e.getValue(), TxnDataValue::merge); return merged; } @@ -119,119 +150,44 @@ public class TxnData extends TxnResult implements Data, Iterable entry : data.entrySet()) - { - size += entry.getKey().estimatedSizeOnHeap(); - - for (Row row : entry.getValue()) - size += row.unsharedHeapSize(); - - // TODO: Include the other parts of FilteredPartition after we rebase to pull in BTreePartitionData? - } + long size = EMPTY_SIZE + (size() * TypeSizes.INT_SIZE); + for (TxnDataValue value : values()) + size += value.estimatedSizeOnHeap(); return size; } - @Override - public Iterator iterator() - { - return data.values().iterator(); - } - - @Override - public boolean equals(Object o) - { - if (this == o) return true; - if (o == null || getClass() != o.getClass()) return false; - TxnData that = (TxnData) o; - return data.equals(that.data); - } - - public static TxnData emptyPartition(TxnDataName name, SinglePartitionReadCommand command) + public static TxnData emptyPartition(int name, SinglePartitionReadCommand command) { TxnData result = new TxnData(); - FilteredPartition empty = FilteredPartition.create(PartitionIterators.getOnlyElement(EmptyIterators.partition(), command)); + TxnDataKeyValue empty = new TxnDataKeyValue(PartitionIterators.getOnlyElement(EmptyIterators.partition(), command)); result.put(name, empty); return result; } - private static final IVersionedSerializer partitionSerializer = new IVersionedSerializer() - { - @Override - public void serialize(FilteredPartition partition, DataOutputPlus out, int version) throws IOException - { - partition.metadata().id.serialize(out); - try (UnfilteredRowIterator iterator = partition.unfilteredIterator()) - { - UnfilteredRowIteratorSerializer.serializer.serialize(iterator, ColumnFilter.all(partition.metadata()), out, version, partition.rowCount()); - } - } - - @Override - public FilteredPartition deserialize(DataInputPlus in, int version) throws IOException - { - TableMetadata metadata = Schema.instance.getExistingTableMetadata(TableId.deserialize(in)); - try (UnfilteredRowIterator partition = UnfilteredRowIteratorSerializer.serializer.deserialize(in, version, metadata, ColumnFilter.all(metadata), DeserializationHelper.Flag.FROM_REMOTE)) - { - return new FilteredPartition(UnfilteredRowIterators.filter(partition, 0)); - } - } - - @Override - public long serializedSize(FilteredPartition partition, int version) - { - TableId tableId = partition.metadata().id; - long size = tableId.serializedSize(); - try (UnfilteredRowIterator iterator = partition.unfilteredIterator()) - { - return size + UnfilteredRowIteratorSerializer.serializer.serializedSize(iterator, ColumnFilter.all(partition.metadata()), version, partition.rowCount()); - } - } - }; - @Override public Kind kind() { return txn_data; } - public static final TxnResultSerializer serializer = new TxnResultSerializer() + public static final IVersionedSerializer serializer = new IVersionedSerializer() { @Override public void serialize(TxnData data, DataOutputPlus out, int version) throws IOException { - out.writeUnsignedVInt32(data.data.size()); - for (Map.Entry entry : data.data.entrySet()) - { - TxnDataName.serializer.serialize(entry.getKey(), out, version); - partitionSerializer.serialize(entry.getValue(), out, version); - } + CollectionSerializers.serializeMap(data, out, version, Int32Serializer.serializer, TxnDataValue.serializer); } @Override public TxnData deserialize(DataInputPlus in, int version) throws IOException { - int size = in.readUnsignedVInt32(); - Map data = Maps.newHashMapWithExpectedSize(size); - for (int i=0; i entry : data.data.entrySet()) - { - size += TxnDataName.serializer.serializedSize(entry.getKey(), version); - size += partitionSerializer.serializedSize(entry.getValue(), version); - } - return size; + return CollectionSerializers.serializedMapSize(data, version, Int32Serializer.serializer, TxnDataValue.serializer); } }; diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnDataKeyValue.java b/src/java/org/apache/cassandra/service/accord/txn/TxnDataKeyValue.java new file mode 100644 index 0000000000..95e6a9c637 --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnDataKeyValue.java @@ -0,0 +1,105 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.accord.txn; + +import java.io.IOException; + +import org.apache.cassandra.db.filter.ColumnFilter; +import org.apache.cassandra.db.partitions.FilteredPartition; +import org.apache.cassandra.db.rows.DeserializationHelper; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.RowIterator; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; +import org.apache.cassandra.db.rows.UnfilteredRowIteratorSerializer; +import org.apache.cassandra.db.rows.UnfilteredRowIterators; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.schema.Schema; +import org.apache.cassandra.schema.TableId; +import org.apache.cassandra.schema.TableMetadata; + +public class TxnDataKeyValue extends FilteredPartition implements TxnDataValue +{ + public TxnDataKeyValue(RowIterator rows) + { + super(rows); + } + + @Override + public TxnDataValue.Kind kind() + { + return Kind.key; + } + + @Override + public TxnDataValue merge(TxnDataValue other) + { + // TODO (review): In Accord, for keys, these should always be identical and really there shouldn't be duplicates + // so we could return either + return this; + } + + @Override + public long estimatedSizeOnHeap() + { + long size = 0; + Row staticRow = staticRow(); + if (staticRow != null) + size += staticRow.unsharedHeapSize(); + for (Row row : this) + size += row.unsharedHeapSize(); + // TODO: Include the other parts of FilteredPartition after we rebase to pull in BTreePartitionData? + return size; + } + + public static final TxnDataValueSerializer serializer = new TxnDataValueSerializer() + { + @Override + public void serialize(TxnDataKeyValue value, DataOutputPlus out, int version) throws IOException + { + value.metadata().id.serialize(out); + try (UnfilteredRowIterator iterator = value.unfilteredIterator()) + { + UnfilteredRowIteratorSerializer.serializer.serialize(iterator, ColumnFilter.all(value.metadata()), out, version, value.rowCount()); + } + } + + @Override + public TxnDataKeyValue deserialize(DataInputPlus in, int version) throws IOException + { + // TODO (required): This needs to use the correct cluster metadata for schema change + TableMetadata metadata = Schema.instance.getExistingTableMetadata(TableId.deserialize(in)); + try (UnfilteredRowIterator partition = UnfilteredRowIteratorSerializer.serializer.deserialize(in, version, metadata, ColumnFilter.all(metadata), DeserializationHelper.Flag.FROM_REMOTE)) + { + return new TxnDataKeyValue(UnfilteredRowIterators.filter(partition, 0)); + } + } + + @Override + public long serializedSize(TxnDataKeyValue value, int version) + { + TableId tableId = value.metadata().id; + long size = tableId.serializedSize(); + try (UnfilteredRowIterator iterator = value.unfilteredIterator()) + { + return size + UnfilteredRowIteratorSerializer.serializer.serializedSize(iterator, ColumnFilter.all(value.metadata()), version, value.rowCount()); + } + } + }; +} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnDataName.java b/src/java/org/apache/cassandra/service/accord/txn/TxnDataName.java deleted file mode 100644 index 5625473857..0000000000 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnDataName.java +++ /dev/null @@ -1,259 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.apache.cassandra.service.accord.txn; - -import java.io.IOException; -import java.nio.ByteBuffer; -import java.util.Arrays; -import java.util.Collections; -import java.util.List; -import java.util.Objects; -import javax.annotation.Nonnull; -import javax.annotation.Nullable; - -import org.apache.cassandra.db.Clustering; -import org.apache.cassandra.db.DecoratedKey; -import org.apache.cassandra.db.TypeSizes; -import org.apache.cassandra.io.IVersionedSerializer; -import org.apache.cassandra.io.util.DataInputPlus; -import org.apache.cassandra.io.util.DataOutputPlus; -import org.apache.cassandra.schema.TableMetadata; -import org.apache.cassandra.utils.ByteBufferUtil; -import org.apache.cassandra.utils.ObjectSizes; - -import static com.google.common.base.Preconditions.checkNotNull; -import static org.apache.cassandra.db.TypeSizes.sizeofUnsignedVInt; -import static org.apache.cassandra.service.accord.AccordSerializers.clusteringSerializer; -import static org.apache.cassandra.utils.NullableSerializer.deserializeNullable; -import static org.apache.cassandra.utils.NullableSerializer.serializeNullable; -import static org.apache.cassandra.utils.NullableSerializer.serializedNullableSize; - -public class TxnDataName implements Comparable -{ - private static final TxnDataName RETURNING = new TxnDataName(Kind.RETURNING); - private static final long EMPTY_SIZE = ObjectSizes.measure(RETURNING); - - public enum Kind - { - USER((byte) 1), - RETURNING((byte) 2), - AUTO_READ((byte) 3), - CAS_READ((byte) 4); - - private final byte value; - - Kind(byte value) - { - this.value = value; - } - - public static Kind from(byte b) - { - switch (b) - { - case 1: - return USER; - case 2: - return RETURNING; - case 3: - return AUTO_READ; - case 4: - return CAS_READ; - default: - throw new IllegalArgumentException("Unknown kind: " + b); - } - } - } - - @Nonnull - private final Kind kind; - - @Nonnull - private final String[] parts; - - @Nullable - private final Clustering clustering; - - public TxnDataName(@Nonnull Kind kind, @Nonnull String... parts) - { - this(kind, null, parts); - } - - public TxnDataName(@Nonnull Kind kind, @Nullable Clustering clustering, @Nonnull String... parts) - { - checkNotNull(kind); - checkNotNull(parts); - this.kind = kind; - this.parts = parts; - this.clustering = clustering; - } - - public static TxnDataName user(String name) - { - return new TxnDataName(Kind.USER, name); - } - - public static TxnDataName returning() - { - return RETURNING; - } - - public static TxnDataName returning(int index) - { - return new TxnDataName(Kind.RETURNING, Integer.toString(index)); - } - - public static TxnDataName partitionRead(TableMetadata metadata, DecoratedKey key, int index) - { - return new TxnDataName(Kind.AUTO_READ, metadata.keyspace, metadata.name, bytesToString(key.getKey()), String.valueOf(index)); - } - - private static String bytesToString(ByteBuffer bytes) - { - return ByteBufferUtil.bytesToHex(bytes); - } - - private static ByteBuffer stringToBytes(String string) - { - return ByteBufferUtil.hexToBytes(string); - } - - public Kind getKind() - { - return kind; - } - - public List getParts() - { - return Collections.unmodifiableList(Arrays.asList(parts)); - } - - public String part(int index) - { - return parts[index]; - } - - public DecoratedKey getDecoratedKey(TableMetadata metadata) - { - checkKind(Kind.AUTO_READ); - ByteBuffer data = stringToBytes(parts[2]); - return metadata.partitioner.decorateKey(data); - } - - public boolean atIndex(int index) - { - checkKind(Kind.AUTO_READ); - return Integer.parseInt(parts[3]) == index; - } - - private void checkKind(Kind expected) - { - if (kind != expected) - throw new IllegalStateException("Expected kind " + expected + " but is " + kind); - } - - public long estimatedSizeOnHeap() - { - long size = EMPTY_SIZE; - for (String part : parts) - size += ObjectSizes.sizeOf(part); - return size; - } - - @Override - public int compareTo(TxnDataName o) - { - int rc = kind.compareTo(o.kind); - if (rc != 0) - return rc; - // same kind has same length - int size = parts.length; - assert o.parts.length == size : String.format("Expected other.parts.length == %d but was %d", size, o.parts.length); - for (int i = 0; i < size; i++) - { - rc = parts[i].compareTo(o.parts[i]); - if (rc != 0) - return rc; - } - return 0; - } - - @Override - public boolean equals(Object o) - { - if (this == o) return true; - if (o == null || getClass() != o.getClass()) return false; - TxnDataName that = (TxnDataName) o; - return kind == that.kind && Arrays.equals(parts, that.parts); - } - - @Override - public int hashCode() - { - int result = Objects.hash(kind); - result = 31 * result + Arrays.hashCode(parts); - return result; - } - - public String name() - { - return String.join(":", parts); - } - - @Override - public String toString() - { - return kind.name() + ':' + name(); - } - - public static final IVersionedSerializer serializer = new IVersionedSerializer() - { - @Override - public void serialize(TxnDataName t, DataOutputPlus out, int version) throws IOException - { - out.writeByte(t.kind.value); - out.writeUnsignedVInt32(t.parts.length); - for (String part : t.parts) - out.writeUTF(part); - serializeNullable(t.clustering, out, version, clusteringSerializer); - } - - @Override - public TxnDataName deserialize(DataInputPlus in, int version) throws IOException - { - Kind kind = Kind.from(in.readByte()); - int length = in.readUnsignedVInt32(); - String[] parts = new String[length]; - for (int i = 0; i < length; i++) - parts[i] = in.readUTF(); - Clustering clustering = deserializeNullable(in, version, clusteringSerializer); - return new TxnDataName(kind, clustering, parts); - } - - @Override - public long serializedSize(TxnDataName t, int version) - { - int size = Byte.BYTES + sizeofUnsignedVInt(t.parts.length); - for (String part : t.parts) - size += TypeSizes.sizeof(part); - size += serializedNullableSize(t.clustering, version, clusteringSerializer); - return size; - } - }; -} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnDataRangeValue.java b/src/java/org/apache/cassandra/service/accord/txn/TxnDataRangeValue.java new file mode 100644 index 0000000000..9a3e8f5f58 --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnDataRangeValue.java @@ -0,0 +1,177 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.accord.txn; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.Comparator; +import java.util.RandomAccess; +import java.util.function.Supplier; +import javax.annotation.Nullable; + +import com.google.common.collect.Lists; + +import org.apache.cassandra.db.TypeSizes; +import org.apache.cassandra.db.filter.ColumnFilter; +import org.apache.cassandra.db.partitions.FilteredPartition; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.partitions.PartitionIterators; +import org.apache.cassandra.db.rows.DeserializationHelper; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; +import org.apache.cassandra.db.rows.UnfilteredRowIteratorSerializer; +import org.apache.cassandra.db.rows.UnfilteredRowIterators; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.schema.Schema; +import org.apache.cassandra.schema.TableId; +import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.utils.ObjectSizes; + +import static com.google.common.base.Preconditions.checkState; +import static org.apache.cassandra.utils.ObjectSizes.sizeOfReferenceArray; + +public class TxnDataRangeValue extends ArrayList implements TxnDataValue, RandomAccess +{ + private static final long EMPTY_SIZE = ObjectSizes.measure(new TxnDataRangeValue(0)) - sizeOfReferenceArray(0); + + public TxnDataRangeValue() {} + + public TxnDataRangeValue(int size) + { + super(size); + } + + @Override + public Kind kind() + { + return Kind.range; + } + + @Override + public TxnDataRangeValue merge(TxnDataValue other) + { + if (isEmpty()) + return (TxnDataRangeValue)other; + + TxnDataRangeValue otherRange = (TxnDataRangeValue)other; + if (otherRange.isEmpty()) + return this; + + TableId tableId = tableId(); + for (FilteredPartition partition : otherRange) + checkState(partition.metadata().id.equals(tableId), "All values should be for the same table"); + + addAll(((TxnDataRangeValue)other)); + return this; + } + + Supplier toPartitionIterator(boolean reversed) + { + // Sorting isn't preserved when merging TxnDataRangeValues together so sort here + sort(); + return () -> PartitionIterators.concat(Lists.transform(this, v -> PartitionIterators.singletonIterator(v.rowIterator(reversed)))); + } + + private void sort() + { + sort(Comparator.comparing(FilteredPartition::partitionKey)); + } + + private @Nullable TableId tableId() + { + if (isEmpty()) + return null; + return get(0).metadata().id; + } + + @Override + public long estimatedSizeOnHeap() + { + long size = EMPTY_SIZE + sizeOfReferenceArray(size()); + for (FilteredPartition partition : this) + { + Row staticRow = partition.staticRow(); + if (staticRow != null) + size += staticRow.unsharedHeapSize(); + for (Row row : partition) + size += row.unsharedHeapSize(); + } + + // TODO: Include the other parts of FilteredPartition after we rebase to pull in BTreePartitionData? + return size; + } + + public static final TxnDataValueSerializer serializer = new TxnDataValueSerializer() + { + @Override + public void serialize(TxnDataRangeValue value, DataOutputPlus out, int version) throws IOException + { + out.writeUnsignedVInt32(value.size()); + + if (value.isEmpty()) + return; + + TableId.serializer.serialize(value.tableId(), out, version); + for (FilteredPartition partition : value) + { + try (UnfilteredRowIterator iterator = partition.unfilteredIterator()) + { + UnfilteredRowIteratorSerializer.serializer.serialize(iterator, ColumnFilter.all(partition.metadata()), out, version, partition.rowCount()); + } + } + } + + @Override + public TxnDataRangeValue deserialize(DataInputPlus in, int version) throws IOException + { + int numPartitions = in.readUnsignedVInt32(); + TxnDataRangeValue value = new TxnDataRangeValue(numPartitions); + if (numPartitions == 0) + return value; + // TODO (required): This needs to be updated for schema change to use the correct cluster metadata + TableMetadata metadata = Schema.instance.getExistingTableMetadata(TableId.deserialize(in)); + for (int i = 0; i < numPartitions; i++) + { + try (UnfilteredRowIterator partition = UnfilteredRowIteratorSerializer.serializer.deserialize(in, version, metadata, ColumnFilter.all(metadata), DeserializationHelper.Flag.FROM_REMOTE)) + { + value.add(new FilteredPartition(UnfilteredRowIterators.filter(partition, 0))); + } + } + return value; + } + + @Override + public long serializedSize(TxnDataRangeValue value, int version) + { + long size = TypeSizes.sizeofUnsignedVInt(value.size()); + if (value.size() == 0) + return size; + size += TableId.serializer.serializedSize(value.tableId(), version); + for (FilteredPartition partition : value) + { + try (UnfilteredRowIterator iterator = partition.unfilteredIterator()) + { + size += UnfilteredRowIteratorSerializer.serializer.serializedSize(iterator, ColumnFilter.all(partition.metadata()), version, partition.rowCount()); + } + } + return size; + } + }; +} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnDataValue.java b/src/java/org/apache/cassandra/service/accord/txn/TxnDataValue.java new file mode 100644 index 0000000000..cc73b43ea8 --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnDataValue.java @@ -0,0 +1,91 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.accord.txn; + +import java.io.IOException; + +import org.apache.cassandra.io.IVersionedSerializer; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; + +import static org.apache.cassandra.db.TypeSizes.sizeof; + +/** + * The result of either a key or range read during Accord transaction execution or a transaction result + */ +public interface TxnDataValue +{ + interface TxnDataValueSerializer extends IVersionedSerializer {} + + enum Kind + { + key(0), + range(1); + int id; + + Kind(int id) + { + this.id = id; + } + + public TxnDataValueSerializer serializer() + { + switch (this) + { + case key: + return TxnDataKeyValue.serializer; + case range: + return TxnDataRangeValue.serializer; + default: + throw new IllegalStateException("Unrecognized kind " + this); + } + } + } + + TxnDataValue.Kind kind(); + + TxnDataValue merge(TxnDataValue other); + + long estimatedSizeOnHeap(); + + IVersionedSerializer serializer = new IVersionedSerializer() + { + @SuppressWarnings("unchecked") + @Override + public void serialize(TxnDataValue txnDataValue, DataOutputPlus out, int version) throws IOException + { + out.writeByte(txnDataValue.kind().ordinal()); + txnDataValue.kind().serializer().serialize(txnDataValue, out, version); + } + + @Override + public TxnDataValue deserialize(DataInputPlus in, int version) throws IOException + { + TxnDataValue.Kind kind = TxnDataValue.Kind.values()[in.readByte()]; + return (TxnDataValue)kind.serializer().deserialize(in, version); + } + + @SuppressWarnings("unchecked") + @Override + public long serializedSize(TxnDataValue txnDataValue, int version) + { + return sizeof((byte)txnDataValue.kind().ordinal()) + txnDataValue.kind().serializer().serializedSize(txnDataValue, version); + } + }; +} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnKeyRead.java b/src/java/org/apache/cassandra/service/accord/txn/TxnKeyRead.java new file mode 100644 index 0000000000..1accdff61e --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnKeyRead.java @@ -0,0 +1,251 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.accord.txn; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.Collections; +import java.util.Comparator; +import java.util.List; +import javax.annotation.Nonnull; +import javax.annotation.Nullable; + +import com.google.common.collect.ImmutableList; + +import accord.api.Data; +import accord.api.DataStore; +import accord.api.Read; +import accord.local.SafeCommandStore; +import accord.primitives.Keys; +import accord.primitives.Participants; +import accord.primitives.Ranges; +import accord.primitives.Seekable; +import accord.primitives.Timestamp; +import accord.utils.async.AsyncChain; +import accord.utils.async.AsyncChains; +import org.apache.cassandra.db.ConsistencyLevel; +import org.apache.cassandra.db.SinglePartitionReadCommand; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.service.accord.api.PartitionKey; +import org.apache.cassandra.utils.ObjectSizes; + +import static com.google.common.base.Preconditions.checkArgument; +import static com.google.common.base.Preconditions.checkNotNull; +import static org.apache.cassandra.service.accord.AccordSerializers.consistencyLevelSerializer; +import static org.apache.cassandra.service.accord.IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.USER; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataName; +import static org.apache.cassandra.utils.ArraySerializers.deserializeArray; +import static org.apache.cassandra.utils.ArraySerializers.serializeArray; +import static org.apache.cassandra.utils.ArraySerializers.serializedArraySize; +import static org.apache.cassandra.utils.NullableSerializer.deserializeNullable; +import static org.apache.cassandra.utils.NullableSerializer.serializeNullable; +import static org.apache.cassandra.utils.NullableSerializer.serializedNullableSize; + +public class TxnKeyRead extends AbstractKeySorted implements TxnRead +{ + public static final TxnKeyRead EMPTY = new TxnKeyRead(new TxnNamedRead[0], null); + private static final long EMPTY_SIZE = ObjectSizes.measure(EMPTY); + private static final Comparator TXN_NAMED_READ_KEY_COMPARATOR = Comparator.comparing(TxnNamedRead::key); + + // Cassandra's consistency level used by Accord to safely read data written outside of Accord + @Nullable + private final ConsistencyLevel cassandraConsistencyLevel; + + private TxnKeyRead(@Nonnull TxnNamedRead[] items, @Nullable ConsistencyLevel cassandraConsistencyLevel) + { + super(items); + checkNotNull(items, "items is null"); + checkArgument(cassandraConsistencyLevel == null || SUPPORTED_READ_CONSISTENCY_LEVELS.contains(cassandraConsistencyLevel), "Unsupported consistency level for read"); + this.cassandraConsistencyLevel = cassandraConsistencyLevel; + } + + private TxnKeyRead(@Nonnull List items, @Nullable ConsistencyLevel cassandraConsistencyLevel) + { + super(items); + checkNotNull(items, "items is null"); + checkArgument(cassandraConsistencyLevel == null || SUPPORTED_READ_CONSISTENCY_LEVELS.contains(cassandraConsistencyLevel), "Unsupported consistency level for read"); + this.cassandraConsistencyLevel = cassandraConsistencyLevel; + } + + public static TxnKeyRead createTxnRead(@Nonnull List items, @Nullable ConsistencyLevel consistencyLevel) + { + items.sort(Comparator.comparing(TxnNamedRead::key)); + return new TxnKeyRead(items, consistencyLevel); + } + + public static TxnKeyRead createSerialRead(List readCommands, ConsistencyLevel consistencyLevel) + { + List reads = new ArrayList<>(readCommands.size()); + for (int i = 0; i < readCommands.size(); i++) + reads.add(new TxnNamedRead(txnDataName(USER, i), readCommands.get(i))); + reads.sort(TXN_NAMED_READ_KEY_COMPARATOR); + return new TxnKeyRead(reads, consistencyLevel); + } + + public static TxnKeyRead createCasRead(SinglePartitionReadCommand readCommand, ConsistencyLevel consistencyLevel) + { + TxnNamedRead read = new TxnNamedRead(txnDataName(CAS_READ), readCommand); + return new TxnKeyRead(ImmutableList.of(read), consistencyLevel); + } + + // A read that declares it will read from keys but doesn't actually read any data so dependent transactions will + // still be applied first + public static TxnKeyRead createNoOpRead(Keys keys) + { + List reads = new ArrayList<>(keys.size()); + for (int i = 0; i < keys.size(); i++) + reads.add(new TxnNamedRead(txnDataName(USER, i), (PartitionKey)keys.get(i), null)); + return new TxnKeyRead(reads, null); + } + + @Override + public long estimatedSizeOnHeap() + { + long size = EMPTY_SIZE; + for (TxnNamedRead read : items) + size += read.estimatedSizeOnHeap(); + return size; + } + + @Override + int compareNonKeyFields(TxnNamedRead left, TxnNamedRead right) + { + return Integer.compare(left.txnDataName(), right.txnDataName()); + } + + @Override + PartitionKey getKey(TxnNamedRead read) + { + return read.key(); + } + + @Override + TxnNamedRead[] newArray(int size) + { + return new TxnNamedRead[size]; + } + + @Override + public Keys keys() + { + return itemKeys; + } + + @Override + public ConsistencyLevel cassandraConsistencyLevel() + { + return cassandraConsistencyLevel; + } + + @Override + public Read slice(Ranges ranges) + { + return intersecting(itemKeys.slice(ranges)); + } + + @Override + public Read intersecting(Participants participants) + { + return intersecting(itemKeys.intersecting(participants)); + } + + private Read intersecting(Keys select) + { + Keys keys = itemKeys.intersecting(select); + List reads = new ArrayList<>(keys.size()); + + for (TxnNamedRead read : items) + if (keys.contains(read.key())) + reads.add(read); + + return createTxnRead(reads, cassandraConsistencyLevel); + } + + @Override + public Read merge(Read read) + { + List reads = new ArrayList<>(items.length); + Collections.addAll(reads, items); + + for (TxnNamedRead namedRead : (TxnKeyRead) read) + if (!reads.contains(namedRead)) + reads.add(namedRead); + + return createTxnRead(reads, cassandraConsistencyLevel); + } + + @Override + public void unmemoize() + { + for (TxnNamedRead read : items) + read.unmemoize(); + } + + @Override + public AsyncChain read(Seekable key, SafeCommandStore safeStore, Timestamp executeAt, DataStore store) + { + List> results = new ArrayList<>(); + forEachWithKey((PartitionKey) key, read -> results.add(read.read(cassandraConsistencyLevel, executeAt))); + + if (results.isEmpty()) + // Result type must match everywhere + return AsyncChains.success(new TxnData()); + + if (results.size() == 1) + return results.get(0); + + return AsyncChains.reduce(results, Data::merge); + } + + @Override + public Kind kind() + { + return Kind.key; + } + + public static final TxnReadSerializer serializer = new TxnReadSerializer() + { + @Override + public void serialize(TxnKeyRead read, DataOutputPlus out, int version) throws IOException + { + serializeArray(read.items, out, version, TxnNamedRead.serializer); + serializeNullable(read.cassandraConsistencyLevel, out, version, consistencyLevelSerializer); + } + + @Override + public TxnKeyRead deserialize(DataInputPlus in, int version) throws IOException + { + TxnNamedRead[] items = deserializeArray(in, version, TxnNamedRead.serializer, TxnNamedRead[]::new); + ConsistencyLevel consistencyLevel = deserializeNullable(in, version, consistencyLevelSerializer); + return new TxnKeyRead(items, consistencyLevel); + } + + @Override + public long serializedSize(TxnKeyRead read, int version) + { + long size = 0; + size += serializedArraySize(read.items, version, TxnNamedRead.serializer); + size += serializedNullableSize(read.cassandraConsistencyLevel, version, consistencyLevelSerializer); + return size; + } + }; +} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java b/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java index d2c92583af..525be9f775 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnNamedRead.java @@ -23,6 +23,7 @@ import java.nio.ByteBuffer; import java.util.Objects; import java.util.concurrent.Callable; import java.util.concurrent.TimeUnit; +import javax.annotation.Nullable; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -31,13 +32,14 @@ import accord.api.Data; import accord.primitives.Timestamp; import accord.utils.async.AsyncChain; import accord.utils.async.AsyncChains; +import accord.utils.async.AsyncResults; import org.apache.cassandra.concurrent.DebuggableTask; import org.apache.cassandra.concurrent.Stage; import org.apache.cassandra.db.ConsistencyLevel; import org.apache.cassandra.db.ReadCommand; import org.apache.cassandra.db.ReadExecutionController; import org.apache.cassandra.db.SinglePartitionReadCommand; -import org.apache.cassandra.db.partitions.FilteredPartition; +import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.db.partitions.PartitionIterator; import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators; import org.apache.cassandra.io.IVersionedSerializer; @@ -57,19 +59,19 @@ public class TxnNamedRead extends AbstractSerialized @SuppressWarnings("unused") private static final Logger logger = LoggerFactory.getLogger(TxnNamedRead.class); - private static final long EMPTY_SIZE = ObjectSizes.measure(new TxnNamedRead(null, null, null)); + private static final long EMPTY_SIZE = ObjectSizes.measure(new TxnNamedRead(0, null, null)); - private final TxnDataName name; + private final int name; private final PartitionKey key; - public TxnNamedRead(TxnDataName name, SinglePartitionReadCommand value) + public TxnNamedRead(int name, @Nullable SinglePartitionReadCommand value) { super(value); this.name = name; this.key = new PartitionKey(value.metadata().id, value.partitionKey()); } - private TxnNamedRead(TxnDataName name, PartitionKey key, ByteBuffer bytes) + public TxnNamedRead(int name, PartitionKey key, ByteBuffer bytes) { super(bytes); this.name = name; @@ -78,7 +80,7 @@ public class TxnNamedRead extends AbstractSerialized public long estimatedSizeOnHeap() { - return EMPTY_SIZE + name.estimatedSizeOnHeap() + key.estimatedSizeOnHeap() + ByteBufferUtil.estimatedSizeOnHeap(bytes()); + return EMPTY_SIZE + key.estimatedSizeOnHeap() + (bytes() != null ? ByteBufferUtil.estimatedSizeOnHeap(bytes()) : 0); } @Override @@ -94,7 +96,7 @@ public class TxnNamedRead extends AbstractSerialized if (o == null || getClass() != o.getClass()) return false; if (!super.equals(o)) return false; TxnNamedRead namedRead = (TxnNamedRead) o; - return name.equals(namedRead.name) && key.equals(namedRead.key); + return name == namedRead.name && key.equals(namedRead.key); } @Override @@ -109,7 +111,7 @@ public class TxnNamedRead extends AbstractSerialized return "TxnNamedRead{name='" + name + '\'' + ", key=" + key + ", update=" + get() + '}'; } - public TxnDataName txnDataName() + public int txnDataName() { return name; } @@ -122,6 +124,8 @@ public class TxnNamedRead extends AbstractSerialized public AsyncChain read(ConsistencyLevel consistencyLevel, Timestamp executeAt) { SinglePartitionReadCommand command = (SinglePartitionReadCommand) get(); + if (command == null) + return AsyncResults.success(TxnData.NOOP_DATA); // TODO (required, safety): before release, double check reasoning that this is safe // AccordCommandsForKey cfk = ((SafeAccordCommandStore)safeStore).commandsForKey(key); // int nowInSeconds = cfk.nowInSecondsFor(executeAt, isForWriteTxn); @@ -151,9 +155,9 @@ public class TxnNamedRead extends AbstractSerialized TxnData result = new TxnData(); if (iterator.hasNext()) { - FilteredPartition filtered = FilteredPartition.create(iterator.next()); - if (filtered.hasRows() || read.selectsFullPartition()) - result.put(name, filtered); + TxnDataKeyValue value = new TxnDataKeyValue(iterator.next()); + if (value.hasRows() || read.selectsFullPartition()) + result.put(name, value); } return result; } @@ -208,17 +212,25 @@ public class TxnNamedRead extends AbstractSerialized @Override public void serialize(TxnNamedRead read, DataOutputPlus out, int version) throws IOException { - TxnDataName.serializer.serialize(read.name, out, version); + out.writeInt(read.name); PartitionKey.serializer.serialize(read.key, out, version); - writeWithVIntLength(read.bytes(), out); + if (read.bytes() != null) + { + out.write(0); + writeWithVIntLength(read.bytes(), out); + } + else + { + out.write(1); + } } @Override public TxnNamedRead deserialize(DataInputPlus in, int version) throws IOException { - TxnDataName name = TxnDataName.serializer.deserialize(in, version); + int name = in.readInt(); PartitionKey key = PartitionKey.serializer.deserialize(in, version); - ByteBuffer bytes = readWithVIntLength(in); + ByteBuffer bytes = in.readByte() == 1 ? null : readWithVIntLength(in); return new TxnNamedRead(name, key, bytes); } @@ -226,9 +238,11 @@ public class TxnNamedRead extends AbstractSerialized public long serializedSize(TxnNamedRead read, int version) { long size = 0; - size += TxnDataName.serializer.serializedSize(read.name, version); + size += TypeSizes.sizeof(read.name); size += PartitionKey.serializer.serializedSize(read.key, version); - size += serializedSizeWithVIntLength(read.bytes()); + size += TypeSizes.BOOL_SIZE; // is null + if (read.bytes() != null) + size += serializedSizeWithVIntLength(read.bytes()); return size; } }; diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnQuery.java b/src/java/org/apache/cassandra/service/accord/txn/TxnQuery.java index e5ae18f8e5..c18c10006d 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnQuery.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnQuery.java @@ -22,7 +22,6 @@ import java.io.IOException; import javax.annotation.Nullable; import com.google.common.base.Preconditions; -import com.google.common.collect.ImmutableMap; import accord.api.Data; import accord.api.Query; @@ -33,20 +32,22 @@ import accord.primitives.Seekables; import accord.primitives.Timestamp; import accord.primitives.TxnId; import org.apache.cassandra.db.EmptyIterators; +import org.apache.cassandra.db.PartitionRangeReadCommand; import org.apache.cassandra.db.SinglePartitionReadCommand; import org.apache.cassandra.db.TypeSizes; -import org.apache.cassandra.db.partitions.FilteredPartition; import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.metrics.ClientRequestsMetricsHolder; import org.apache.cassandra.service.accord.api.PartitionKey; +import org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind; import org.apache.cassandra.service.consensus.migration.ConsensusRequestRouter; import org.apache.cassandra.tcm.Epoch; import org.apache.cassandra.utils.ObjectSizes; import static com.google.common.base.Preconditions.checkNotNull; -import static org.apache.cassandra.service.accord.txn.TxnRead.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind.CAS_READ; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataName; public abstract class TxnQuery implements Query { @@ -115,10 +116,10 @@ public abstract class TxnQuery implements Query return new TxnData(); else if (txnData.isEmpty()) { - TxnRead txnRead = (TxnRead)read; - SinglePartitionReadCommand command = (SinglePartitionReadCommand)txnRead.iterator().next().get(); + TxnKeyRead txnKeyRead = (TxnKeyRead)read; + SinglePartitionReadCommand command = (SinglePartitionReadCommand) txnKeyRead.iterator().next().get(); // For CAS must return a non-empty result to indicate error even if there was no partition found - return new TxnData(ImmutableMap.of(CAS_READ, FilteredPartition.create(EmptyIterators.row(command.metadata(), command.partitionKey(), command.isReversed())))); + return TxnData.of(txnDataName(CAS_READ), new TxnDataKeyValue(EmptyIterators.row(command.metadata(), command.partitionKey(), command.isReversed()))); } else // If it failed to apply the partition contents are returned and it indicates failure @@ -156,6 +157,29 @@ public abstract class TxnQuery implements Query } }; + public static final TxnQuery RANGE_QUERY = new TxnQuery() + { + @Override + protected byte type() + { + return 5; + } + + @Override + public Result doCompute(TxnId txnId, Timestamp executeAt, Seekables keys, @Nullable Data data, @Nullable Read read, @Nullable Update update) + { + return data != null ? concat((TxnData) data, read) : new TxnData(); + } + + private Result concat(TxnData data, Read read) + { + TxnRangeRead txnRead = (TxnRangeRead)read; + PartitionRangeReadCommand command = (PartitionRangeReadCommand) txnRead.get(); + TxnDataRangeValue value = (TxnDataRangeValue)data.get(txnDataName(TxnDataNameKind.USER)); + return new TxnRangeReadResult(value.toPartitionIterator(command.isReversed())); + } + }; + private static final long SIZE = ObjectSizes.measure(ALL); private TxnQuery() {} @@ -189,7 +213,7 @@ public abstract class TxnQuery implements Query @Override public void serialize(TxnQuery query, DataOutputPlus out, int version) throws IOException { - Preconditions.checkArgument(query == null | query == ALL | query == NONE | query == CONDITION | query == UNSAFE_EMPTY); + Preconditions.checkArgument(query == null | query == ALL | query == NONE | query == CONDITION | query == UNSAFE_EMPTY | query == RANGE_QUERY); out.writeByte(query == null ? 0 : query.type()); } @@ -204,13 +228,14 @@ public abstract class TxnQuery implements Query case 2: return NONE; case 3: return CONDITION; case 4: return UNSAFE_EMPTY; + case 5: return RANGE_QUERY; } } @Override public long serializedSize(TxnQuery query, int version) { - Preconditions.checkArgument(query == null | query == ALL | query == NONE | query == CONDITION | query == UNSAFE_EMPTY); + Preconditions.checkArgument(query == null | query == ALL | query == NONE | query == CONDITION | query == UNSAFE_EMPTY | query == RANGE_QUERY); return TypeSizes.sizeof((byte)2); } }; diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnRangeRead.java b/src/java/org/apache/cassandra/service/accord/txn/TxnRangeRead.java new file mode 100644 index 0000000000..9dc61bce72 --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnRangeRead.java @@ -0,0 +1,345 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.accord.txn; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.Callable; +import java.util.concurrent.TimeUnit; +import javax.annotation.Nonnull; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import accord.api.Data; +import accord.api.DataStore; +import accord.api.Read; +import accord.local.SafeCommandStore; +import accord.primitives.AbstractRanges.UnionMode; +import accord.primitives.Participants; +import accord.primitives.Range; +import accord.primitives.Ranges; +import accord.primitives.Routable.Domain; +import accord.primitives.Routables.Slice; +import accord.primitives.Seekable; +import accord.primitives.Seekables; +import accord.primitives.Timestamp; +import accord.utils.async.AsyncChain; +import accord.utils.async.AsyncChains; +import org.apache.cassandra.concurrent.DebuggableTask; +import org.apache.cassandra.concurrent.Stage; +import org.apache.cassandra.db.ConsistencyLevel; +import org.apache.cassandra.db.PartitionPosition; +import org.apache.cassandra.db.PartitionRangeReadCommand; +import org.apache.cassandra.db.ReadCommand; +import org.apache.cassandra.db.ReadExecutionController; +import org.apache.cassandra.db.partitions.FilteredPartition; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators; +import org.apache.cassandra.db.rows.RowIterator; +import org.apache.cassandra.dht.AbstractBounds; +import org.apache.cassandra.dht.Token; +import org.apache.cassandra.io.IVersionedSerializer; +import org.apache.cassandra.io.util.DataInputPlus; +import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.schema.TableId; +import org.apache.cassandra.service.accord.AccordObjectSizes; +import org.apache.cassandra.service.accord.TokenRange; +import org.apache.cassandra.service.accord.api.AccordRoutingKey; +import org.apache.cassandra.service.accord.api.AccordRoutingKey.SentinelKey; +import org.apache.cassandra.service.accord.api.AccordRoutingKey.TokenKey; +import org.apache.cassandra.service.accord.serializers.KeySerializers; +import org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind; +import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.MonotonicClock; +import org.apache.cassandra.utils.ObjectSizes; + +import static com.google.common.base.Preconditions.checkArgument; +import static org.apache.cassandra.service.accord.AccordSerializers.consistencyLevelSerializer; +import static org.apache.cassandra.service.accord.IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS; +import static org.apache.cassandra.utils.ByteBufferUtil.readWithVIntLength; +import static org.apache.cassandra.utils.ByteBufferUtil.serializedSizeWithVIntLength; +import static org.apache.cassandra.utils.ByteBufferUtil.writeWithVIntLength; +import static org.apache.cassandra.utils.NullableSerializer.deserializeNullable; +import static org.apache.cassandra.utils.NullableSerializer.serializeNullable; +import static org.apache.cassandra.utils.NullableSerializer.serializedNullableSize; + +public class TxnRangeRead extends AbstractSerialized implements TxnRead +{ + private static final Logger logger = LoggerFactory.getLogger(TxnRangeRead.class); + + public static final TxnRangeRead EMPTY = new TxnRangeRead(null, null, null); + private static final long EMPTY_SIZE = ObjectSizes.measure(EMPTY); + + @Nonnull + private final ConsistencyLevel cassandraConsistencyLevel; + @Nonnull + private final Ranges covering; + + public TxnRangeRead(@Nonnull PartitionRangeReadCommand command, @Nonnull ConsistencyLevel cassandraConsistencyLevel) + { + super(command); + checkArgument(cassandraConsistencyLevel == null || SUPPORTED_READ_CONSISTENCY_LEVELS.contains(cassandraConsistencyLevel), "Unsupported consistency level for read"); + this.cassandraConsistencyLevel = cassandraConsistencyLevel; + TableId tableId = command.metadata().id; + + AbstractBounds range = command.dataRange().keyRange(); + + // Read commands can contain a mix of different kinds of bounds to facilitate paging + // and we need to communicate that to Accord as its own ranges. We will use + // TokenKey and Sentinel key and stick exclusively with left exclusive/right inclusive + // ranges rather add more types of ranges to the mix + boolean inclusiveLeft = range.inclusiveLeft(); + Token startToken = range.left.getToken(); + AccordRoutingKey startAccordRoutingKey; + if (startToken.isMinimum() && inclusiveLeft) + startAccordRoutingKey = SentinelKey.min(tableId); + else + startAccordRoutingKey = new TokenKey(tableId, startToken); + + boolean inclusiveRight = range.inclusiveRight(); + Token stopToken = range.right.getToken(); + AccordRoutingKey stopAccordRoutingKey; + if (inclusiveRight) + stopAccordRoutingKey = new TokenKey(tableId, stopToken); + else + stopAccordRoutingKey = new TokenKey(tableId, stopToken.decreaseSlightly()); + + covering = Ranges.of(new TokenRange(startAccordRoutingKey, stopAccordRoutingKey)); + } + + private TxnRangeRead(@Nonnull ByteBuffer commandBytes, @Nonnull ConsistencyLevel cassandraConsistencyLevel, @Nonnull Ranges covering) + { + super(commandBytes); + checkArgument(cassandraConsistencyLevel == null || SUPPORTED_READ_CONSISTENCY_LEVELS.contains(cassandraConsistencyLevel), "Unsupported consistency level for read"); + this.cassandraConsistencyLevel = cassandraConsistencyLevel; + this.covering = covering; + } + + @Override + public Seekables keys() + { + return covering; + } + + @Override + public AsyncChain read(Seekable range, SafeCommandStore commandStore, Timestamp executeAt, DataStore store) + { + // It's fine for our nowInSeconds to lag slightly our insertion timestamp, as to the user + // this simply looks like the transaction witnessed TTL'd data and the data then expired + // immediately after the transaction executed, and this simplifies things a great deal + int nowInSeconds = (int) TimeUnit.MICROSECONDS.toSeconds(executeAt.hlc()); + + List> results = new ArrayList<>(); + PartitionRangeReadCommand command = (PartitionRangeReadCommand)get(); + Ranges intersecting = covering.slice(Ranges.of(range.asRange()), Slice.Minimal); + for (Range subRange : intersecting) + results.add(performLocalRead(command, subRange, nowInSeconds)); + if (results.isEmpty()) + // Result type must match everywhere + return AsyncChains.success(new TxnData()); + + return AsyncChains.reduce(results, Data::merge); + } + + private AsyncChain performLocalRead(PartitionRangeReadCommand command, Range r, int nowInSeconds) + { + Callable readCallable = () -> + { + AbstractBounds bounds = command.dataRange().keyRange(); + PartitionPosition startPP = bounds.left; + PartitionPosition endPP = bounds.right; + TokenKey startTokenKey = new TokenKey(command.metadata().id, startPP.getToken()); + Token subRangeStartToken = ((AccordRoutingKey)r.start()).asTokenKey().token(); + Token subRangeEndToken = ((AccordRoutingKey)r.end()).asTokenKey().token(); + /* + * The way ranges/bounds work for range queries is that the beginning and ending bounds from the command + * could be tokens (and min/max key bounds) or actual keys depending on the bounds of the top level query we + * are running and where we are in paging. We need to preserve whatever is in the command in case it is a + * key and not a token, or it's a token but might be a min/max key bound. + * + * Then Accord will further subdivide the range in the command so need to inject additional bounds in the middle + * that match the range ownership of Accord. + * + * The command still contains the original bound and then the Accord range passed in determines what subset of + * that bound we want. We have to make sure to use the bounds from the command if it is the start or end instead + * of a key bound created from the Accord range since it could be a real key or min/max bound. + * + * When we are dealing with a bound created by Accord's further subdivision we use a maxKeyBound (exclusive) + * for both beginning and end because Bounds is left and right inclusive while Range is only left inclusive. + * We only use TokenRange with Accord which matches the left/right inclusivity of Cassandra's Range. + * + * That means the Range we get from Accord overlaps the previous Range on the left which when converted to a Bound + * would potentially read the same Token twice. So the left needs to be a maxKeyBound to exclude the data that isn't + * owned here and to avoid potentially reading the same data twice. The right bound also needs to be a maxKeyBound since Range + * is right inclusive so every partition we find needs to be < the right bound. + */ + boolean isFirstSubrange = startPP.getToken().equals(subRangeStartToken); + PartitionPosition subRangeStartPP = isFirstSubrange ? startPP : subRangeStartToken.maxKeyBound(); + PartitionPosition subRangeEndPP = endPP.getToken().equals(subRangeEndToken) ? endPP : subRangeEndToken.maxKeyBound(); + // Need to preserve the fact it is a bounds for paging to work, a range is not left inclusive and will not start from where we left off + AbstractBounds subRange = isFirstSubrange ? bounds.withNewRight(subRangeEndPP) : new org.apache.cassandra.dht.Range(subRangeStartPP, subRangeEndPP); + PartitionRangeReadCommand read = command.forSubRangeWithNowInSeconds(nowInSeconds, subRange, startTokenKey.equals(r.start())); + + try (ReadExecutionController controller = read.executionController(); + UnfilteredPartitionIterator partition = read.executeLocally(controller); + PartitionIterator iterator = UnfilteredPartitionIterators.filter(partition, read.nowInSec())) + { + TxnData result = new TxnData(); + TxnDataRangeValue value = new TxnDataRangeValue(); + while(iterator.hasNext()) + { + try (RowIterator rows = iterator.next()) + { + FilteredPartition filtered = FilteredPartition.create(rows); + if (filtered.hasRows() || read.selectsFullPartition()) + { + value.add(filtered); + } + } + } + result.put(TxnData.txnDataName(TxnDataNameKind.USER), value); + return result; + } + }; + + return AsyncChains.ofCallable(Stage.READ.executor(), readCallable, (callable, receiver) -> + new DebuggableTask.RunnableDebuggableTask() + { + private final long approxCreationTimeNanos = MonotonicClock.Global.approxTime.now(); + private volatile long approxStartTimeNanos; + + @Override + public void run() + { + approxStartTimeNanos = MonotonicClock.Global.approxTime.now(); + + try + { + Data call = callable.call(); + receiver.accept(call, null); + } + catch (Throwable t) + { + logger.debug("AsyncChain Callable threw an Exception", t); + receiver.accept(null, t); + } + } + + @Override + public long creationTimeNanos() + { + return approxCreationTimeNanos; + } + + @Override + public long startTimeNanos() + { + return approxStartTimeNanos; + } + + @Override + public String description() + { + return command.toCQLString(); + } + } + ); + } + + @Override + public Read slice(Ranges ranges) + { + return new TxnRangeRead(bytes(), cassandraConsistencyLevel, covering.slice(ranges, Slice.Minimal)); + } + + @Override + public Read intersecting(Participants participants) + { + checkArgument(participants.domain() == Domain.Range, "Can only intersect with ranges"); + return new TxnRangeRead(bytes(), cassandraConsistencyLevel, covering.intersecting(participants, Slice.Minimal)); + } + + @Override + public Read merge(Read other) + { + Ranges newCovering = ((TxnRangeRead)other).covering.union(UnionMode.MERGE_OVERLAPPING, covering); + return new TxnRangeRead(bytes(), cassandraConsistencyLevel, newCovering); + } + + @Override + public Kind kind() + { + return Kind.range; + } + + @Override + public long estimatedSizeOnHeap() + { + long size = EMPTY_SIZE; + size += ByteBufferUtil.estimatedSizeOnHeap(bytes()); + size += AccordObjectSizes.ranges(covering); + return size; + } + + public static final TxnReadSerializer serializer = new TxnReadSerializer() + { + @Override + public void serialize(TxnRangeRead read, DataOutputPlus out, int version) throws IOException + { + serializeNullable(read.cassandraConsistencyLevel, out, version, consistencyLevelSerializer); + KeySerializers.ranges.serialize(read.covering, out, version); + writeWithVIntLength(read.bytes(), out); + } + + @Override + public TxnRangeRead deserialize(DataInputPlus in, int version) throws IOException + { + ConsistencyLevel consistencyLevel = deserializeNullable(in, version, consistencyLevelSerializer); + Ranges covering = KeySerializers.ranges.deserialize(in, version); + ByteBuffer commandBytes = readWithVIntLength(in); + return new TxnRangeRead(commandBytes, consistencyLevel, covering); + } + + @Override + public long serializedSize(TxnRangeRead read, int version) + { + long size = 0; + size += KeySerializers.ranges.serializedSize(read.covering, version); + size += serializedNullableSize(read.cassandraConsistencyLevel, version, consistencyLevelSerializer); + size += serializedSizeWithVIntLength(read.bytes()); + return size; + } + }; + + @Override + protected IVersionedSerializer serializer() + { + return PartitionRangeReadCommand.serializer; + } + + @Override + public ConsistencyLevel cassandraConsistencyLevel() + { + return null; + } +} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnRangeReadResult.java b/src/java/org/apache/cassandra/service/accord/txn/TxnRangeReadResult.java new file mode 100644 index 0000000000..a70709cee9 --- /dev/null +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnRangeReadResult.java @@ -0,0 +1,62 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.accord.txn; + +import java.util.function.Supplier; + +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.RowIterator; +import org.apache.cassandra.utils.ObjectSizes; + +public class TxnRangeReadResult implements TxnResult +{ + private static final long EMPTY_SIZE = ObjectSizes.measure(new TxnRangeReadResult(null)); + + public final Supplier partitions; + + public TxnRangeReadResult(Supplier partitions) + { + this.partitions = partitions; + } + + @Override + public Kind kind() + { + return Kind.range_read; + } + + @Override + public long estimatedSizeOnHeap() + { + long size = EMPTY_SIZE; + PartitionIterator iterator = partitions.get(); + while (iterator.hasNext()) + { + RowIterator rowIterator = iterator.next(); + Row staticRow = rowIterator.staticRow(); + if (staticRow != null) + size += staticRow.unsharedHeapSize(); + while (rowIterator.hasNext()) + size += rowIterator.next().unsharedHeapSize(); + } + // TODO: Include the other parts of FilteredPartition after we rebase to pull in BTreePartitionData? + return size; + } +} diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnRead.java b/src/java/org/apache/cassandra/service/accord/txn/TxnRead.java index dd8a56d119..b908bdd2b5 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnRead.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnRead.java @@ -19,232 +19,75 @@ package org.apache.cassandra.service.accord.txn; import java.io.IOException; -import java.util.ArrayList; -import java.util.Collections; -import java.util.List; -import javax.annotation.Nonnull; -import javax.annotation.Nullable; -import com.google.common.collect.ImmutableList; - -import accord.api.Data; -import accord.api.DataStore; import accord.api.Read; -import accord.local.SafeCommandStore; -import accord.primitives.Keys; -import accord.primitives.Participants; -import accord.primitives.Ranges; -import accord.primitives.Seekable; -import accord.primitives.Timestamp; -import accord.utils.async.AsyncChain; -import accord.utils.async.AsyncChains; import org.apache.cassandra.db.ConsistencyLevel; -import org.apache.cassandra.db.SinglePartitionReadCommand; import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; -import org.apache.cassandra.service.accord.api.PartitionKey; -import org.apache.cassandra.service.accord.serializers.KeySerializers; -import org.apache.cassandra.service.accord.txn.TxnDataName.Kind; -import org.apache.cassandra.utils.ObjectSizes; -import org.apache.cassandra.utils.Simulate; -import static com.google.common.base.Preconditions.checkArgument; -import static org.apache.cassandra.service.accord.AccordSerializers.consistencyLevelSerializer; -import static org.apache.cassandra.service.accord.IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS; -import static org.apache.cassandra.utils.ArraySerializers.deserializeArray; -import static org.apache.cassandra.utils.ArraySerializers.serializeArray; -import static org.apache.cassandra.utils.ArraySerializers.serializedArraySize; -import static org.apache.cassandra.utils.NullableSerializer.deserializeNullable; -import static org.apache.cassandra.utils.NullableSerializer.serializeNullable; -import static org.apache.cassandra.utils.NullableSerializer.serializedNullableSize; -import static org.apache.cassandra.utils.Simulate.With.MONITORS; +import static org.apache.cassandra.db.TypeSizes.sizeof; -public class TxnRead extends AbstractKeySorted implements Read +public interface TxnRead extends Read { - // There is only potentially one partition in a CAS and SERIAL/LOCAL_SERIAL read - public static final String SERIAL_READ_NAME = "SERIAL_READ"; - public static final TxnDataName SERIAL_READ = TxnDataName.user(SERIAL_READ_NAME); + ConsistencyLevel cassandraConsistencyLevel(); - public static final TxnRead EMPTY = new TxnRead(new TxnNamedRead[0], Keys.EMPTY, null); - private static final long EMPTY_SIZE = ObjectSizes.measure(EMPTY); + public interface TxnReadSerializer extends IVersionedSerializer {} - public static final String CAS_READ_NAME = "CAS_READ"; - public static final TxnDataName CAS_READ = new TxnDataName(Kind.CAS_READ, CAS_READ_NAME); - - @Nonnull - private final Keys txnKeys; - - // Cassandra's consistency level used by Accord to safely read data written outside of Accord - @Nullable - private final ConsistencyLevel cassandraConsistencyLevel; - - public TxnRead(@Nonnull TxnNamedRead[] items, @Nonnull Keys txnKeys, @Nullable ConsistencyLevel cassandraConsistencyLevel) + enum Kind { - super(items); - checkArgument(cassandraConsistencyLevel == null || SUPPORTED_READ_CONSISTENCY_LEVELS.contains(cassandraConsistencyLevel), "Unsupported consistency level for read"); - this.txnKeys = txnKeys; - this.cassandraConsistencyLevel = cassandraConsistencyLevel; - } + key(0), + range(1); - public TxnRead(@Nonnull List items, @Nonnull Keys txnKeys, @Nullable ConsistencyLevel cassandraConsistencyLevel) - { - super(items); - checkArgument(cassandraConsistencyLevel == null || SUPPORTED_READ_CONSISTENCY_LEVELS.contains(cassandraConsistencyLevel), "Unsupported consistency level for read"); - this.txnKeys = txnKeys; - this.cassandraConsistencyLevel = cassandraConsistencyLevel; - } + int id; - public static TxnRead createTxnRead(@Nonnull List items, @Nonnull Keys txnKeys, @Nullable ConsistencyLevel consistencyLevel) - { - return new TxnRead(items, txnKeys, consistencyLevel); - } - - public static TxnRead createSerialRead(List readCommands, ConsistencyLevel consistencyLevel) - { - List reads = new ArrayList<>(readCommands.size()); - for (int i = 0; i < readCommands.size(); i++) - reads.add(new TxnNamedRead(TxnDataName.user(String.valueOf(i)), readCommands.get(i))); - Keys keys = Keys.of(reads, TxnNamedRead::key); - return new TxnRead(reads, keys, consistencyLevel); - } - - public static TxnRead createCasRead(SinglePartitionReadCommand readCommand, ConsistencyLevel consistencyLevel) - { - TxnNamedRead read = new TxnNamedRead(CAS_READ, readCommand); - return new TxnRead(ImmutableList.of(read), Keys.of(read.key()), consistencyLevel); - } - - // A read that declares it will read from keys but doesn't actually read any data so dependent transactions will - // still be applied first - public static TxnRead createNoOpRead(Keys keys) - { - return new TxnRead(ImmutableList.of(), keys, null); - } - - public long estimatedSizeOnHeap() - { - long size = EMPTY_SIZE; - for (TxnNamedRead read : items) - size += read.estimatedSizeOnHeap(); - return size; - } - - @Override - int compareNonKeyFields(TxnNamedRead left, TxnNamedRead right) - { - return left.txnDataName().compareTo(right.txnDataName()); - } - - @Override - PartitionKey getKey(TxnNamedRead read) - { - return read.key(); - } - - @Override - TxnNamedRead[] newArray(int size) - { - return new TxnNamedRead[size]; - } - - public void unmemoize() - { - for (int i = 0 ; i < size() ; ++i) - items[i].unmemoize(); - } - - @Override - public Keys keys() - { - return txnKeys; - } - - public ConsistencyLevel cassandraConsistencyLevel() - { - return cassandraConsistencyLevel; - } - - @Override - public Read slice(Ranges ranges) - { - return intersecting(itemKeys.slice(ranges)); - } - - @Override - public Read intersecting(Participants participants) - { - return intersecting(itemKeys.intersecting(participants)); - } - - private Read intersecting(Keys select) - { - Keys keys = itemKeys.intersecting(select); - List reads = new ArrayList<>(keys.size()); - - for (TxnNamedRead read : items) - if (keys.contains(read.key())) - reads.add(read); - - return createTxnRead(reads, txnKeys.intersecting(select), cassandraConsistencyLevel); - } - - @Override - public Read merge(Read read) - { - List reads = new ArrayList<>(items.length); - Collections.addAll(reads, items); - - for (TxnNamedRead namedRead : (TxnRead) read) - if (!reads.contains(namedRead)) - reads.add(namedRead); - - return createTxnRead(reads, txnKeys.with((Keys)read.keys()), cassandraConsistencyLevel); - } - - @Override - public AsyncChain read(Seekable key, SafeCommandStore safeStore, Timestamp executeAt, DataStore store) - { - List> results = new ArrayList<>(); - forEachWithKey((PartitionKey) key, read -> results.add(read.read(cassandraConsistencyLevel, executeAt))); - - if (results.isEmpty()) - // Result type must match everywhere - return AsyncChains.success(new TxnData()); - - if (results.size() == 1) - return results.get(0); - - return AsyncChains.reduce(results, Data::merge); - } - - @Simulate(with = MONITORS) - public static final IVersionedSerializer serializer = new IVersionedSerializer() - { - @Override - public void serialize(TxnRead read, DataOutputPlus out, int version) throws IOException + Kind(int id) { - KeySerializers.keys.serialize(read.txnKeys, out, version); - serializeArray(read.items, out, version, TxnNamedRead.serializer); - serializeNullable(read.cassandraConsistencyLevel, out, version, consistencyLevelSerializer); + this.id = id; + } + + public TxnReadSerializer serializer() + { + switch (this) + { + case key: + return TxnKeyRead.serializer; + case range: + return TxnRangeRead.serializer; + default: + throw new IllegalStateException("Unrecognized kind " + this); + } + } + } + + long estimatedSizeOnHeap(); + + Kind kind(); + + void unmemoize(); + + IVersionedSerializer serializer = new IVersionedSerializer() + { + @SuppressWarnings("unchecked") + @Override + public void serialize(TxnRead txnRead, DataOutputPlus out, int version) throws IOException + { + out.writeByte(txnRead.kind().ordinal()); + txnRead.kind().serializer().serialize(txnRead, out, version); } @Override public TxnRead deserialize(DataInputPlus in, int version) throws IOException { - Keys keys = KeySerializers.keys.deserialize(in, version); - TxnNamedRead[] items = deserializeArray(in, version, TxnNamedRead.serializer, TxnNamedRead[]::new); - ConsistencyLevel consistencyLevel = deserializeNullable(in, version, consistencyLevelSerializer); - return new TxnRead(items, keys, consistencyLevel); + TxnRead.Kind kind = TxnRead.Kind.values()[in.readByte()]; + return (TxnRead)kind.serializer().deserialize(in, version); } + @SuppressWarnings("unchecked") @Override - public long serializedSize(TxnRead read, int version) + public long serializedSize(TxnRead txnRead, int version) { - long size = KeySerializers.keys.serializedSize(read.txnKeys, version); - size += serializedArraySize(read.items, version, TxnNamedRead.serializer); - size += serializedNullableSize(read.cassandraConsistencyLevel, version, consistencyLevelSerializer); - return size; + return sizeof((byte)txnRead.kind().ordinal()) + txnRead.kind().serializer().serializedSize(txnRead, version); } }; } diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnReference.java b/src/java/org/apache/cassandra/service/accord/txn/TxnReference.java index 0da0079303..0f1e37e90c 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnReference.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnReference.java @@ -46,18 +46,18 @@ import static org.apache.cassandra.service.accord.AccordSerializers.columnMetada public class TxnReference { - private final TxnDataName tuple; + private final Integer tuple; private final ColumnMetadata column; private final CellPath path; - public TxnReference(TxnDataName tuple, ColumnMetadata column, CellPath path) + public TxnReference(Integer tuple, ColumnMetadata column, CellPath path) { this.tuple = tuple; this.column = column; this.path = path; } - public TxnReference(TxnDataName tuple, ColumnMetadata column) + public TxnReference(Integer tuple, ColumnMetadata column) { this(tuple, column, null); } @@ -135,9 +135,9 @@ public class TxnReference return row.clustering().bufferAt(column.position()); } - public FilteredPartition getPartition(TxnData data) + public TxnDataKeyValue getPartition(TxnData data) { - return data.get(tuple); + return (TxnDataKeyValue)data.get(tuple); } public Row getRow(TxnData data) @@ -292,7 +292,7 @@ public class TxnReference @Override public void serialize(TxnReference reference, DataOutputPlus out, int version) throws IOException { - TxnDataName.serializer.serialize(reference.tuple, out, version); + out.writeInt(reference.tuple); out.writeBoolean(reference.column != null); if (reference.column != null) columnMetadataSerializer.serialize(reference.column, out, version); @@ -304,7 +304,7 @@ public class TxnReference @Override public TxnReference deserialize(DataInputPlus in, int version) throws IOException { - TxnDataName name = TxnDataName.serializer.deserialize(in, version); + int name = in.readInt(); ColumnMetadata column = in.readBoolean() ? columnMetadataSerializer.deserialize(in, version) : null; CellPath path = in.readBoolean() ? CollectionType.cellPathSerializer.deserialize(in) : null; return new TxnReference(name, column, path); @@ -314,7 +314,7 @@ public class TxnReference public long serializedSize(TxnReference reference, int version) { long size = 0; - size += TxnDataName.serializer.serializedSize(reference.tuple, version); + size += TypeSizes.INT_SIZE; size += TypeSizes.BOOL_SIZE; if (reference.column != null) size += columnMetadataSerializer.serializedSize(reference.column, version); diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnResult.java b/src/java/org/apache/cassandra/service/accord/txn/TxnResult.java index 38eeb88aca..2b5af7c08b 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnResult.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnResult.java @@ -18,23 +18,15 @@ package org.apache.cassandra.service.accord.txn; -import java.io.IOException; - import accord.api.Result; -import org.apache.cassandra.io.IVersionedSerializer; -import org.apache.cassandra.io.util.DataInputPlus; -import org.apache.cassandra.io.util.DataOutputPlus; -import static org.apache.cassandra.db.TypeSizes.sizeof; - -public abstract class TxnResult implements Result +public interface TxnResult extends Result { - public interface TxnResultSerializer extends IVersionedSerializer {} - - public enum Kind + enum Kind { txn_data(0), - retry_new_protocol(1); + retry_new_protocol(1), + range_read(2); int id; @@ -42,47 +34,9 @@ public abstract class TxnResult implements Result { this.id = id; } - - public TxnResultSerializer serializer() - { - switch (this) - { - case txn_data: - return TxnData.serializer; - case retry_new_protocol: - return RetryWithNewProtocolResult.serializer; - default: - throw new IllegalStateException("Unrecognized kind " + this); - } - } } - public abstract Kind kind(); + Kind kind(); - public abstract long estimatedSizeOnHeap(); - - public static final IVersionedSerializer serializer = new IVersionedSerializer() - { - @SuppressWarnings("unchecked") - @Override - public void serialize(TxnResult txnResult, DataOutputPlus out, int version) throws IOException - { - out.writeByte(txnResult.kind().ordinal()); - txnResult.kind().serializer().serialize(txnResult, out, version); - } - - @Override - public TxnResult deserialize(DataInputPlus in, int version) throws IOException - { - TxnResult.Kind kind = TxnResult.Kind.values()[in.readByte()]; - return (TxnResult)kind.serializer().deserialize(in, version); - } - - @SuppressWarnings("unchecked") - @Override - public long serializedSize(TxnResult txnResult, int version) - { - return sizeof((byte)txnResult.kind().ordinal()) + txnResult.kind().serializer().serializedSize(txnResult, version); - } - }; + long estimatedSizeOnHeap(); } diff --git a/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java b/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java index f8f0dc4983..48260635e4 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java +++ b/src/java/org/apache/cassandra/service/accord/txn/TxnWrite.java @@ -380,6 +380,12 @@ public class TxnWrite extends AbstractKeySorted implements Writ @Override public AsyncChain apply(Seekable key, SafeCommandStore safeStore, TxnId txnId, Timestamp executeAt, DataStore store, PartialTxn txn) { + // UnrecoverableRepairUpdate will deserialize as null at other nodes + // Accord should skip the Update for a read transaction, but handle it here anyways + if (txn.update() == null) + return Writes.SUCCESS; + + TxnUpdate txnUpdate = ((TxnUpdate)txn.update()); // TODO (expected, efficiency): 99.9999% of the time we can just use executeAt.hlc(), so can avoid bringing // cfk into memory by retaining at all times in memory key ranges that are dirty and must use this logic; // any that aren't can just use executeAt.hlc @@ -390,7 +396,7 @@ public class TxnWrite extends AbstractKeySorted implements Writ List> results = new ArrayList<>(); - boolean preserveTimestamps = ((TxnUpdate)txn.update()).preserveTimestamps(); + boolean preserveTimestamps = txnUpdate.preserveTimestamps(); // Apply updates not specified fully by the client but built from fragments completed by data from reads. // This occurs, for example, when an UPDATE statement uses a value assigned by a LET statement. Function accordListPathSuppler = accordListPathSupplier(timestamp); @@ -401,7 +407,6 @@ public class TxnWrite extends AbstractKeySorted implements Writ // Apply updates that are fully specified by the client and not reliant on data from reads. // ex. INSERT INTO tbl (a, b, c) VALUES (1, 2, 3) // These updates are persisted only in TxnUpdate and not in TxnWrite to avoid duplication. - TxnUpdate txnUpdate = (TxnUpdate) txn.update(); assert txnUpdate != null : "PartialTxn should contain an update if we're applying a write!"; List updates = txnUpdate.completeUpdatesForKey((RoutableKey) key); updates.forEach(update -> results.add(update.write(preserveTimestamps, accordListPathSuppler, timestamp, nowInSeconds))); diff --git a/src/java/org/apache/cassandra/service/accord/txn/UnrecoverableRepairUpdate.java b/src/java/org/apache/cassandra/service/accord/txn/UnrecoverableRepairUpdate.java index 25bbc0e6f9..dfd35f93e2 100644 --- a/src/java/org/apache/cassandra/service/accord/txn/UnrecoverableRepairUpdate.java +++ b/src/java/org/apache/cassandra/service/accord/txn/UnrecoverableRepairUpdate.java @@ -18,9 +18,7 @@ package org.apache.cassandra.service.accord.txn; -import java.io.IOException; import java.util.Map; -import java.util.concurrent.ConcurrentHashMap; import java.util.concurrent.atomic.AtomicLong; import javax.annotation.Nonnull; import javax.annotation.Nullable; @@ -29,6 +27,7 @@ import accord.api.Data; import accord.api.Update; import accord.api.Write; import accord.local.Node; +import accord.primitives.Keys; import accord.primitives.Participants; import accord.primitives.Ranges; import accord.primitives.Seekables; @@ -36,7 +35,6 @@ import accord.primitives.Timestamp; import org.apache.cassandra.db.ConsistencyLevel; import org.apache.cassandra.db.DecoratedKey; import org.apache.cassandra.db.Mutation; -import org.apache.cassandra.db.TypeSizes; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.locator.Endpoints; @@ -60,16 +58,13 @@ import org.apache.cassandra.service.reads.repair.BlockingReadRepair; * coordinator fails to complete the transaction then the dependent Cassandra read that triggered the read repair will * also fail and it doesn't matter if the read repair is partially applied or not applied at all since it doesn't propose * new values. + * + * The reason we stash this in an Update that Accord won't actually use is that there isn't an explicit parameter + * passing mechanism in Accord we can only provide implementations of Read, Query, and Update and then Accord will hand them + * back via the Txn during execution. */ public class UnrecoverableRepairUpdate, P extends ReplicaPlan.ForRead> extends AccordUpdate { - private static final ConcurrentHashMap inflightUpdates = new ConcurrentHashMap<>(); - - public static UnrecoverableRepairUpdate removeInflightUpdate(Key updateKey) - { - return inflightUpdates.remove(updateKey); - } - private static class Key { final int nodeId; @@ -111,7 +106,7 @@ public class UnrecoverableRepairUpdate, P extends Replica public final ReplicaPlan.ForWrite writePlan; public final Key updateKey; - private UnrecoverableRepairUpdate(Node.Id nodeId, BlockingReadRepair parent, + public UnrecoverableRepairUpdate(Node.Id nodeId, BlockingReadRepair parent, Seekables keys, DecoratedKey dk, Map mutations, ReplicaPlan.ForWrite writePlan) { this.parent = parent; @@ -123,25 +118,16 @@ public class UnrecoverableRepairUpdate, P extends Replica this.updateKey = new Key(nodeId.id, nextCounter.getAndIncrement()); } - public static , P extends ReplicaPlan.ForRead> UnrecoverableRepairUpdate create(Node.Id nodeId, BlockingReadRepair parent, - Seekables keys, DecoratedKey dk, Map mutations, - ReplicaPlan.ForWrite writePlan) - { - UnrecoverableRepairUpdate update = new UnrecoverableRepairUpdate<>(nodeId, parent, keys, dk, mutations, writePlan); - inflightUpdates.put(update.updateKey, update); - return update; - } - @Override public Seekables keys() { - return keys; + return Keys.EMPTY; } @Override public Write apply(Timestamp executeAt, @Nullable Data data) { - return TxnWrite.EMPTY_CONDITION_FAILED; + return null; } @Override @@ -191,25 +177,24 @@ public class UnrecoverableRepairUpdate, P extends Replica parent.repairPartitionDirectly(readCoordinator, dk, mutations, writePlan); } - public static final AccordUpdateSerializer serializer = new AccordUpdateSerializer() + // Only the original coordinator ever needs to access an UnrecoverableRepairUpdate + public static final AccordUpdateSerializer serializer = new AccordUpdateSerializer<>() { @Override - public void serialize(UnrecoverableRepairUpdate update, DataOutputPlus out, int version) throws IOException + public void serialize(UnrecoverableRepairUpdate update, DataOutputPlus out, int version) { - out.writeUnsignedVInt32(update.updateKey.nodeId); - out.writeUnsignedVInt(update.updateKey.counter); } @Override - public UnrecoverableRepairUpdate deserialize(DataInputPlus in, int version) throws IOException + public UnrecoverableRepairUpdate deserialize(DataInputPlus in, int version) { - return inflightUpdates.get(new Key(in.readUnsignedVInt32(), in.readUnsignedVInt())); + return null; } @Override public long serializedSize(UnrecoverableRepairUpdate update, int version) { - return TypeSizes.sizeofUnsignedVInt(update.updateKey.nodeId) + TypeSizes.sizeofUnsignedVInt(update.updateKey.counter); + return 0; } }; } diff --git a/src/java/org/apache/cassandra/service/consensus/TransactionalMode.java b/src/java/org/apache/cassandra/service/consensus/TransactionalMode.java index efaa80ac15..15868afdcc 100644 --- a/src/java/org/apache/cassandra/service/consensus/TransactionalMode.java +++ b/src/java/org/apache/cassandra/service/consensus/TransactionalMode.java @@ -18,7 +18,11 @@ package org.apache.cassandra.service.consensus; +import com.google.common.collect.ImmutableList; + import org.apache.cassandra.db.ConsistencyLevel; +import org.apache.cassandra.dht.NormalizedRanges; +import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.service.accord.IAccordService; @@ -28,6 +32,7 @@ import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.utils.LocalizeString; import static com.google.common.base.Preconditions.checkState; +import static org.apache.cassandra.dht.NormalizedRanges.normalizedRanges; /* * Configure the transactional behavior of a table. Enables accord on a table and defines how it mixes with non-serial writes @@ -67,22 +72,36 @@ public enum TransactionalMode // Running on Paxos V1 or V2 with Accord disabled off(false, false, false, false, false), + // TODO (maybe): These unsafe modes don't have Accord do async commit and single replica reads so how useful are they besides preserving non-SERIAL performance? + // These modes are unsafe when Accord and non-SERIAL reads and writes interact with the same data + // They don't guarantee that non-SERIAL reads or writes will see the latest Accord writes or that + // Accord transactions will recover correctly + /* + * Enables Accord but does not allow non-SERIAL reads and writes to occur safely to data read/written by Accord + * * Execute non-SERIAL writes through Cassandra via StorageProxy's normal write path. This can lead Accord to compute * multiple outcomes for a transaction that depend on data written by non-SERIAL writes. * * SERIAL reads and CAS will run on Accord. Accord will honor provided consistency levels and do synchronous commit - * so the results can be read with non-SERIAL CLs. + * so the results can be read correctly with non-SERIAL CLs, but read repair could interfere with Accord. */ unsafe(true, false, false, false, false), /* + * Enables Accord and makes it safe to perform non-SERIAL reads of Accord data without guaranteeing that they will + * see the latest Accord writes. non-SERIAL writes to data read by Accord will make Accord txn recovery non-deterministic + * * Allow mixing of non-SERIAL writes and Accord, but still force BRR through Accord. * This mode makes it safe to perform non-SERIAL or SERIAL reads of Accord data, but unsafe * to write data that Accord may attempt to read. */ unsafe_writes(true, false, false, false, true), + // These modes always provide correct execution with mixed_reads allow non-transaction non-SERIAL operations + // at the expense of slower Accord transactions, and full allowing faster transaction execution, but forcing + // all reads and writes to occur transactionally + /* * Execute writes through Accord skipping StorageProxy's normal write path, but commit * writes at the provided consistency level so they can be read via non-SERIAL consistency levels. @@ -98,36 +117,34 @@ public enum TransactionalMode public final boolean accordIsEnabled; public final boolean ignoresSuppliedCommitCL; - public final boolean writesThroughAccord; - public final boolean readsThroughAccord; + public final boolean nonSerialWritesThroughAccord; + public final boolean nonSerialReadsThroughAccord; public final boolean blockingReadRepairThroughAccord; private final String cqlParam; - TransactionalMode(boolean accordIsEnabled, boolean ignoresSuppliedCommitCL, boolean writesThroughAccord, boolean readsThroughAccord, boolean blockingReadRepairThroughAccord) + TransactionalMode(boolean accordIsEnabled, boolean ignoreSuppliedCommitCL, boolean nonSerialWritesThroughAccord, boolean nonSerialReadsThroughAccord, boolean blockingReadRepairThroughAccord) { this.accordIsEnabled = accordIsEnabled; - this.ignoresSuppliedCommitCL = ignoresSuppliedCommitCL; - this.writesThroughAccord = writesThroughAccord; - this.readsThroughAccord = readsThroughAccord; + this.ignoresSuppliedCommitCL = ignoreSuppliedCommitCL; + this.nonSerialWritesThroughAccord = nonSerialWritesThroughAccord; + this.nonSerialReadsThroughAccord = nonSerialReadsThroughAccord; this.blockingReadRepairThroughAccord = blockingReadRepairThroughAccord; this.cqlParam = String.format("transactional_mode = '%s'", LocalizeString.toLowerCaseLocalized(this.name())); } - public ConsistencyLevel commitCLForStrategy(ConsistencyLevel consistencyLevel, TableId tableId, Token token) + public ConsistencyLevel commitCLForStrategy(TransactionalMigrationFromMode fromMode, ConsistencyLevel consistencyLevel, ClusterMetadata cm, TableId tableId, Token token) { if (ignoresSuppliedCommitCL) { - TableMigrationState tms = ClusterMetadata.current().consensusMigrationState.tableStates.get(tableId); - if (tms != null) - { - // Only ignore the supplied consistency level if the token is not migrating - // otherwise honor it since there could still be Paxos and non-SERIAL reads racing with migration. - // Migrating to Accord, Paxos continues reading during the first phase of migration - // Migrating to Paxos, this doesn't really matter since this transaction will get RetryOnDifferentSystemException - if (tms.migratingRanges.intersects(token)) - return consistencyLevel; - } - return null; + TableMigrationState tms = cm.consensusMigrationState.tableStates.get(tableId); + checkState(tms != null || fromMode == TransactionalMigrationFromMode.none); + + // Only ignore the supplied consistency level if the token is not migrating + // otherwise honor it since there could still be Paxos and non-SERIAL reads racing with migration. + // Migrating to Accord, Paxos continues reading during the first phase of migration + // Migrating to Paxos, this doesn't really matter since this transaction will get RetryOnDifferentSystemException + if (tms == null || tms.migratedRanges.intersects(token)) + return null; } if (!IAccordService.SUPPORTED_COMMIT_CONSISTENCY_LEVELS.contains(consistencyLevel)) @@ -136,11 +153,16 @@ public enum TransactionalMode return consistencyLevel; } + /** + * Infer whether Accord can ignore the read CL and bias towards correctness by reading from a quorum + * if it's needed due to how non-SERIAL writes are done + */ private boolean ignoresSuppliedReadCL() { - return writesThroughAccord && blockingReadRepairThroughAccord; + return nonSerialWritesThroughAccord && blockingReadRepairThroughAccord; } + public ConsistencyLevel readCLForStrategy(TransactionalMigrationFromMode fromMode, ConsistencyLevel consistencyLevel, ClusterMetadata cm, TableId tableId, Token token) { if (ignoresSuppliedReadCL()) @@ -162,11 +184,42 @@ public enum TransactionalMode return consistencyLevel; } + public ConsistencyLevel readCLForStrategy(TransactionalMigrationFromMode fromMode, ConsistencyLevel consistencyLevel, ClusterMetadata cm, TableId tableId, Range range) + { + if (ignoresSuppliedReadCL()) + { + TableMigrationState tms = cm.consensusMigrationState.tableStates.get(tableId); + checkState(tms != null || fromMode == TransactionalMigrationFromMode.none); + + NormalizedRanges ranges = normalizedRanges(ImmutableList.of(range)); + // Only ignore the supplied consistency level if none of the range is migrating + // otherwise honor it because we might read through Accord for non-SERIAL reads before repair is run + // this is OK to do because BRR still works and Accord isn't computing a write so recovery + // determinism isn't an issue + if (tms == null || tms.migratedRanges.intersection(ranges).equals(ranges)) + return null; + } + + if (!IAccordService.SUPPORTED_READ_CONSISTENCY_LEVELS.contains(consistencyLevel)) + throw new UnsupportedOperationException("Consistency level " + consistencyLevel + " is unsupported with Accord for read, supported are ONE, QUORUM, and SERIAL"); + return consistencyLevel; + } + public String asCqlParam() { return cqlParam; } + public boolean nonSerialWritesThroughAccord() + { + return nonSerialWritesThroughAccord; + } + + public boolean readRepairsThroughAccord() + { + return blockingReadRepairThroughAccord; + } + public static TransactionalMode fromOrdinal(int ordinal) { return values()[ordinal]; diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationMutationHelper.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationMutationHelper.java index fdc112936c..6f62e49e84 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationMutationHelper.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationMutationHelper.java @@ -48,13 +48,14 @@ import org.apache.cassandra.schema.Schema; import org.apache.cassandra.schema.SchemaConstants; import org.apache.cassandra.schema.TableId; import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.schema.TableParams; import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.IAccordService; import org.apache.cassandra.service.accord.IAccordService.AsyncTxnResult; import org.apache.cassandra.service.accord.api.PartitionKey; import org.apache.cassandra.service.accord.txn.TxnCondition; +import org.apache.cassandra.service.accord.txn.TxnKeyRead; import org.apache.cassandra.service.accord.txn.TxnQuery; -import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.service.accord.txn.TxnReferenceOperations; import org.apache.cassandra.service.accord.txn.TxnUpdate; import org.apache.cassandra.service.accord.txn.TxnWrite; @@ -110,11 +111,13 @@ public class ConsensusMigrationMutationHelper { for (TableId tableId : mutation.getTableIds()) { - TransactionalMode mode = getTableMetadata(cm, tableId).params.transactionalMode; + TableParams tableParams = getTableMetadata(cm, tableId).params; + TransactionalMode mode = tableParams.transactionalMode; + TransactionalMigrationFromMode migrationFromMode = tableParams.transactionalMigrationFrom; // commitCLForStrategy should return either null or the supplied consistency level // in which case we will commit everything at that CL since Accord doesn't support per table // commit consistency - ConsistencyLevel commitCL = mode.commitCLForStrategy(consistencyLevel, tableId, mutation.key().getToken()); + ConsistencyLevel commitCL = mode.commitCLForStrategy(migrationFromMode, consistencyLevel, cm, tableId, mutation.key().getToken()); if (commitCL != null) return commitCL; } @@ -217,7 +220,7 @@ public class ConsensusMigrationMutationHelper return new SplitMutation<>(null, mutation); Token token = mutation.key().getToken(); - Predicate isAccordUpdate = tableId -> tokenShouldBeWrittenThroughAccord(cm, tableId, token); + Predicate isAccordUpdate = tableId -> tokenShouldBeWrittenThroughAccord(cm, tableId, token, TransactionalMode::nonSerialWritesThroughAccord, TransactionalMigrationFromMode::nonSerialReadsThroughAccord); T accordMutation = (T)mutation.filter(isAccordUpdate); T normalMutation = (T)mutation.filter(not(isAccordUpdate)); @@ -252,7 +255,7 @@ public class ConsensusMigrationMutationHelper // Potentially ignore commit consistency level if the TransactionalMode specifies full ConsistencyLevel clForCommit = consistencyLevelForCommit(cm, mutations, consistencyLevel); TxnUpdate update = new TxnUpdate(fragments, TxnCondition.none(), clForCommit, true); - Txn.InMemory txn = new Txn.InMemory(Keys.of(partitionKeys), TxnRead.EMPTY, TxnQuery.NONE, update); + Txn.InMemory txn = new Txn.InMemory(Keys.of(partitionKeys), TxnKeyRead.EMPTY, TxnQuery.NONE, update); IAccordService accordService = AccordService.instance(); try { @@ -291,7 +294,7 @@ public class ConsensusMigrationMutationHelper { TableId tableId = pu.metadata().id; ColumnFamilyStore cfs = ColumnFamilyStore.getIfExists(tableId); - if (tokenShouldBeWrittenThroughAccord(cm, tableId, dk.getToken())) + if (tokenShouldBeWrittenThroughAccord(cm, tableId, dk.getToken(), TransactionalMode::nonSerialWritesThroughAccord, TransactionalMigrationFromMode::nonSerialReadsThroughAccord)) { throwRetryOnDifferentSystem = true; if (markedColumnFamilies == null) @@ -306,15 +309,19 @@ public class ConsensusMigrationMutationHelper throw new RetryOnDifferentSystemException(); } - public static boolean tokenShouldBeWrittenThroughAccord(@Nonnull ClusterMetadata cm, @Nonnull TableId tableId, @Nonnull Token token) + public static boolean tokenShouldBeWrittenThroughAccord(@Nonnull ClusterMetadata cm, + @Nonnull TableId tableId, + @Nonnull Token token, + Predicate nonSerialWritesThroughAccord, + Predicate nonSerialWritesThroughAccordFrom) { TableMetadata tm = getTableMetadata(cm, tableId); if (tm == null) return false; - boolean transactionalModeWritesThroughAccord = tm.params.transactionalMode.writesThroughAccord; + boolean transactionalModeWritesThroughAccord = nonSerialWritesThroughAccord.test(tm.params.transactionalMode); TransactionalMigrationFromMode transactionalMigrationFromMode = tm.params.transactionalMigrationFrom; - boolean migrationFromWritesThroughAccord = transactionalMigrationFromMode.writesThroughAccord(); + boolean migrationFromWritesThroughAccord = nonSerialWritesThroughAccordFrom.test(transactionalMigrationFromMode); if (transactionalModeWritesThroughAccord && migrationFromWritesThroughAccord) return true; diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationState.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationState.java index 4367737642..e01a796aff 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationState.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusMigrationState.java @@ -34,6 +34,7 @@ import com.google.common.base.Preconditions; import com.google.common.collect.ImmutableList; import com.google.common.collect.ImmutableMap; import com.google.common.collect.ImmutableSet; +import com.google.common.collect.Maps; import com.google.common.collect.Sets; import org.apache.cassandra.dht.Range; @@ -53,7 +54,6 @@ import static com.google.common.base.Preconditions.checkNotNull; import static com.google.common.base.Preconditions.checkState; import static org.apache.cassandra.service.consensus.migration.TableMigrationState.initialRepairPendingRanges; import static org.apache.cassandra.utils.CollectionSerializers.deserializeMap; -import static org.apache.cassandra.utils.CollectionSerializers.newHashMap; import static org.apache.cassandra.utils.CollectionSerializers.serializeMap; import static org.apache.cassandra.utils.CollectionSerializers.serializedMapSize; @@ -153,6 +153,7 @@ public class ConsensusMigrationState implements MetadataValue tableMigrationStates = deserializeMap(in, version, TableId.metadataSerializer, TableMigrationState.serializer, newHashMap()); + Map tableMigrationStates = deserializeMap(in, version, TableId.metadataSerializer, TableMigrationState.serializer, Maps::newHashMapWithExpectedSize); return new ConsensusMigrationState(lastModified, tableMigrationStates); } diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java index 38f65e19f4..ae4ff8e4cd 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusRequestRouter.java @@ -115,9 +115,8 @@ public class ConsensusRequestRouter return tbm; } - public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull DecoratedKey key, @Nonnull String keyspace, @Nonnull String table, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) + public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull ClusterMetadata cm, @Nonnull DecoratedKey key, @Nonnull String keyspace, @Nonnull String table, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) { - ClusterMetadata cm = ClusterMetadata.current(); TableMetadata metadata = metadata(cm, keyspace, table); // Non-distributed tables always take the Paxos path @@ -126,9 +125,8 @@ public class ConsensusRequestRouter return routeAndMaybeMigrate(cm, metadata, key, consistencyLevel, requestTime, timeoutNanos, isForWrite); } - public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull DecoratedKey key, @Nonnull TableId tableId, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) + public ConsensusRoutingDecision routeAndMaybeMigrate(@Nonnull ClusterMetadata cm, @Nonnull DecoratedKey key, @Nonnull TableId tableId, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) { - ClusterMetadata cm = ClusterMetadata.current(); TableMetadata metadata = getTableMetadata(cm, tableId); // Non-distributed tables always take the Paxos path if (metadata == null) @@ -152,35 +150,6 @@ public class ConsensusRequestRouter return tm; } - protected static boolean mayWriteThroughAccord(TableMetadata metadata) - { - return metadata.params.transactionalMode.writesThroughAccord || metadata.params.transactionalMigrationFrom.writesThroughAccord(); - } - - public boolean shouldWriteThroughAccordAndMaybeMigrate(@Nonnull DecoratedKey key, @Nonnull TableId tableId, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) - { - ClusterMetadata cm = ClusterMetadata.current(); - TableMetadata metadata = cm.schema.getTableMetadata(tableId); - if (metadata == null) - throw new IllegalStateException(String.format("Can't route consensus request for nonexistent table %s", tableId)); - - if (!mayWriteThroughAccord(metadata)) - return false; - - consistencyLevel = consistencyLevel.isDatacenterLocal() ? ConsistencyLevel.LOCAL_SERIAL : ConsistencyLevel.SERIAL; - ConsensusRoutingDecision decision = routeAndMaybeMigrate(cm, metadata, key, consistencyLevel, requestTime, timeoutNanos, isForWrite); - switch (decision) - { - case paxosV1: - case paxosV2: - return false; - case accord: - return true; - default: - throw new IllegalStateException("Unsupported consensus " + decision); - } - } - protected ConsensusRoutingDecision routeAndMaybeMigrate(ClusterMetadata cm, @Nonnull TableMetadata tmd, @Nonnull DecoratedKey key, ConsistencyLevel consistencyLevel, Dispatcher.RequestTime requestTime, long timeoutNanos, boolean isForWrite) { @@ -303,7 +272,7 @@ public class ConsensusRequestRouter { ClusterMetadata cm = ClusterMetadataService.instance().fetchLogFromCMS(epoch); TableMigrationState tms = cm.consensusMigrationState.tableStates.get(tableId); - return isKeyInMigratingOrMigratedRangeFromAccord(cm.schema.getTableMetadata(tableId), tms, key); + return isKeyInMigratingOrMigratedRangeFromAccord(getTableMetadata(cm, tableId), tms, key); } /* diff --git a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java index 4eb79b0761..e4ba971ffe 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/ConsensusTableMigration.java @@ -194,7 +194,7 @@ public abstract class ConsensusTableMigration if (metadata == null || !metadata.params.transactionalMigrationFrom.isMigrating()) continue; TransactionalMode transactionalMode = metadata.params.transactionalMode; - if (!transactionalMode.writesThroughAccord && transactionalMode != TransactionalMode.unsafe_writes) + if (!transactionalMode.nonSerialWritesThroughAccord && transactionalMode != TransactionalMode.unsafe_writes) throw new IllegalStateException("non-SERIAL writes need to be routed through Accord before attempting migration, or enable mixed mode"); tableIds.add(tableId); } diff --git a/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java b/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java index 6724aab221..95013b3e88 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/TableMigrationState.java @@ -37,6 +37,7 @@ import javax.annotation.Nullable; import com.google.common.collect.ImmutableList; import com.google.common.collect.ImmutableMap; import com.google.common.collect.ImmutableSortedMap; +import com.google.common.collect.Maps; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -62,7 +63,6 @@ import static org.apache.cassandra.dht.NormalizedRanges.normalizedRanges; import static org.apache.cassandra.dht.Range.subtract; import static org.apache.cassandra.utils.CollectionSerializers.deserializeMap; import static org.apache.cassandra.utils.CollectionSerializers.deserializeSet; -import static org.apache.cassandra.utils.CollectionSerializers.newHashMap; import static org.apache.cassandra.utils.CollectionSerializers.serializeCollection; import static org.apache.cassandra.utils.CollectionSerializers.serializeMap; import static org.apache.cassandra.utils.CollectionSerializers.serializedCollectionSize; @@ -398,7 +398,7 @@ public class TableMigrationState TableId tableId = TableId.deserialize(in); Set> migratedRanges = deserializeSet(in, version, Range.serializer); Set> repairPendingRanges = deserializeSet(in, version, Range.serializer); - Map> migratingRangesByEpoch = deserializeMap(in, version, Epoch.serializer, ConsensusTableMigration.rangesSerializer, newHashMap()); + Map> migratingRangesByEpoch = deserializeMap(in, version, Epoch.serializer, ConsensusTableMigration.rangesSerializer, Maps::newHashMapWithExpectedSize); return new TableMigrationState(keyspaceName, tableName, tableId, targetProtocol, migratedRanges, repairPendingRanges, migratingRangesByEpoch); } diff --git a/src/java/org/apache/cassandra/service/consensus/migration/TransactionalMigrationFromMode.java b/src/java/org/apache/cassandra/service/consensus/migration/TransactionalMigrationFromMode.java index 5b7bc0a87b..ba631378e7 100644 --- a/src/java/org/apache/cassandra/service/consensus/migration/TransactionalMigrationFromMode.java +++ b/src/java/org/apache/cassandra/service/consensus/migration/TransactionalMigrationFromMode.java @@ -73,14 +73,19 @@ public enum TransactionalMigrationFromMode return from != null && from.accordIsEnabled; } - public boolean writesThroughAccord() + public boolean nonSerialWritesThroughAccord() { - return from != null && from.writesThroughAccord; + return from != null && from.nonSerialWritesThroughAccord; } - public boolean readsThroughAccord() + public boolean readRepairsThroughAccord() { - return from != null && from.writesThroughAccord; + return from != null && from.blockingReadRepairThroughAccord; + } + + public boolean nonSerialReadsThroughAccord() + { + return from != null && from.nonSerialReadsThroughAccord; } public boolean isMigrating() diff --git a/src/java/org/apache/cassandra/service/reads/range/AccordRangeResponse.java b/src/java/org/apache/cassandra/service/reads/range/AccordRangeResponse.java new file mode 100644 index 0000000000..75dfb4b3b2 --- /dev/null +++ b/src/java/org/apache/cassandra/service/reads/range/AccordRangeResponse.java @@ -0,0 +1,75 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.reads.range; + +import java.util.function.IntPredicate; + +import org.apache.cassandra.db.ConsistencyLevel; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.rows.RowIterator; +import org.apache.cassandra.service.StorageProxy; +import org.apache.cassandra.service.StorageProxy.ConsensusAttemptResult; +import org.apache.cassandra.service.accord.IAccordService.AsyncTxnResult; +import org.apache.cassandra.transport.Dispatcher; +import org.apache.cassandra.utils.AbstractIterator; + +import static com.google.common.base.Preconditions.checkState; + +public class AccordRangeResponse extends AbstractIterator implements IRangeResponse +{ + private final AsyncTxnResult asyncTxnResult; + // Range queries don't support reverse, but dutifully threading it through anyways + private final boolean reversed; + private final ConsistencyLevel cl; + private final Dispatcher.RequestTime requestTime; + private PartitionIterator result; + + public AccordRangeResponse(AsyncTxnResult asyncTxnResult, boolean reversed, ConsistencyLevel cl, Dispatcher.RequestTime requestTime) + { + this.asyncTxnResult = asyncTxnResult; + this.cl = cl; + this.requestTime = requestTime; + this.reversed = reversed; + } + + private void waitForResponse() + { + if (result != null) + return; + IntPredicate alwaysTrue = ignored -> true; + IntPredicate alwaysFalse = ignored -> false; + // TODO (required): Handle retry on different system + ConsensusAttemptResult consensusAttemptResult = StorageProxy.getConsensusAttemptResultFromAsyncTxnResult(asyncTxnResult, 1, reversed ? alwaysTrue : alwaysFalse, cl, requestTime); + checkState(!consensusAttemptResult.shouldRetryOnNewConsensusProtocol, "Live migration is not supported yet"); + result = consensusAttemptResult.serialReadResult; + } + + @Override + protected RowIterator computeNext() + { + waitForResponse(); + return result.hasNext() ? result.next() : endOfData(); + } + + @Override + public void close() + { + // It's an in-memory iterator so no need to close whatever might end up in TxnResult + } +} diff --git a/src/java/org/apache/cassandra/service/reads/range/SingleRangeResponse.java b/src/java/org/apache/cassandra/service/reads/range/CassandraRangeResponse.java similarity index 82% rename from src/java/org/apache/cassandra/service/reads/range/SingleRangeResponse.java rename to src/java/org/apache/cassandra/service/reads/range/CassandraRangeResponse.java index d318b41614..1d383b8017 100644 --- a/src/java/org/apache/cassandra/service/reads/range/SingleRangeResponse.java +++ b/src/java/org/apache/cassandra/service/reads/range/CassandraRangeResponse.java @@ -28,7 +28,7 @@ import org.apache.cassandra.service.reads.ReadCallback; import org.apache.cassandra.service.reads.repair.ReadRepair; import org.apache.cassandra.utils.AbstractIterator; -class SingleRangeResponse extends AbstractIterator implements PartitionIterator +class CassandraRangeResponse extends AbstractIterator implements IRangeResponse { private final DataResolver resolver; private final ReadCallback handler; @@ -36,16 +36,17 @@ class SingleRangeResponse extends AbstractIterator implements Parti private PartitionIterator result; - SingleRangeResponse(DataResolver resolver, - ReadCallback handler, - ReadRepair readRepair) + CassandraRangeResponse(DataResolver resolver, + ReadCallback handler, + ReadRepair readRepair) { this.resolver = resolver; this.handler = handler; this.readRepair = readRepair; } - ReadRepair getReadRepair() + @Override + public ReadRepair getReadRepair() { return readRepair; } diff --git a/src/java/org/apache/cassandra/service/reads/range/IRangeResponse.java b/src/java/org/apache/cassandra/service/reads/range/IRangeResponse.java new file mode 100644 index 0000000000..e3a869866a --- /dev/null +++ b/src/java/org/apache/cassandra/service/reads/range/IRangeResponse.java @@ -0,0 +1,33 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.service.reads.range; + +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.locator.EndpointsForRange; +import org.apache.cassandra.locator.ReplicaPlan.ForRangeRead; +import org.apache.cassandra.service.reads.repair.NoopReadRepair; +import org.apache.cassandra.service.reads.repair.ReadRepair; + +public interface IRangeResponse extends PartitionIterator +{ + default ReadRepair getReadRepair() + { + return NoopReadRepair.instance; + } +} diff --git a/src/java/org/apache/cassandra/service/reads/range/RangeCommandIterator.java b/src/java/org/apache/cassandra/service/reads/range/RangeCommandIterator.java index 98f0399d3c..762ea06fbc 100644 --- a/src/java/org/apache/cassandra/service/reads/range/RangeCommandIterator.java +++ b/src/java/org/apache/cassandra/service/reads/range/RangeCommandIterator.java @@ -45,11 +45,17 @@ import org.apache.cassandra.locator.ReplicaPlan; import org.apache.cassandra.metrics.ClientRangeRequestMetrics; import org.apache.cassandra.net.Message; import org.apache.cassandra.net.MessagingService; +import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.schema.TableParams; import org.apache.cassandra.service.StorageProxy; +import org.apache.cassandra.service.accord.IAccordService.AsyncTxnResult; +import org.apache.cassandra.service.consensus.TransactionalMode; +import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; import org.apache.cassandra.service.reads.DataResolver; import org.apache.cassandra.service.reads.ReadCallback; import org.apache.cassandra.service.reads.ReadCoordinator; import org.apache.cassandra.service.reads.repair.ReadRepair; +import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.transport.Dispatcher; import org.apache.cassandra.utils.AbstractIterator; @@ -69,11 +75,11 @@ public class RangeCommandIterator extends AbstractIterator implemen final PartitionRangeReadCommand command; final boolean enforceStrictLiveness; final Dispatcher.RequestTime requestTime; + final ReadCoordinator readCoordinator; int rangesQueried; int batchesRequested = 0; - private DataLimits.Counter counter; private PartitionIterator sentQueryIterator; @@ -85,6 +91,7 @@ public class RangeCommandIterator extends AbstractIterator implemen RangeCommandIterator(CloseableIterator replicaPlans, PartitionRangeReadCommand command, + ReadCoordinator readCoordinator, int concurrencyFactor, int maxConcurrencyFactor, int totalRangeCount, @@ -92,6 +99,7 @@ public class RangeCommandIterator extends AbstractIterator implemen { this.replicaPlans = replicaPlans; this.command = command; + this.readCoordinator = readCoordinator; this.concurrencyFactor = concurrencyFactor; this.maxConcurrencyFactor = maxConcurrencyFactor; this.totalRangeCount = totalRangeCount; @@ -187,7 +195,7 @@ public class RangeCommandIterator extends AbstractIterator implemen * {@code DataLimits}) may have "state" information and that state may only be valid for the first query (in * that it's the query that "continues" whatever we're previously queried). */ - private SingleRangeResponse query(ReplicaPlan.ForRangeRead replicaPlan, boolean isFirst) + private IRangeResponse query(ReplicaPlan.ForRangeRead replicaPlan, ReadCoordinator readCoordinator, boolean isFirst) { PartitionRangeReadCommand rangeCommand = command.forSubRange(replicaPlan.range(), isFirst); @@ -196,30 +204,47 @@ public class RangeCommandIterator extends AbstractIterator implemen boolean trackRepairedStatus = DatabaseDescriptor.getRepairedDataTrackingForRangeReadsEnabled() && replicaPlan.contacts().filter(Replica::isFull).size() > 1; - ReplicaPlan.SharedForRangeRead sharedReplicaPlan = ReplicaPlan.shared(replicaPlan); - ReadRepair readRepair = - ReadRepair.create(ReadCoordinator.DEFAULT, command, sharedReplicaPlan, requestTime); - DataResolver resolver = - new DataResolver<>(ReadCoordinator.DEFAULT, rangeCommand, sharedReplicaPlan, readRepair, requestTime, trackRepairedStatus); - ReadCallback handler = - new ReadCallback<>(resolver, rangeCommand, sharedReplicaPlan, requestTime); - - if (replicaPlan.contacts().size() == 1 && replicaPlan.contacts().get(0).isSelf()) + ClusterMetadata cm = ClusterMetadata.current(); + TableMetadata metadata = command.metadata(); + TableParams tableParams = metadata.params; + TransactionalMode transactionalMode = tableParams.transactionalMode; + TransactionalMigrationFromMode transactionalMigrationFromMode = tableParams.transactionalMigrationFrom; + if (transactionalMigrationFromMode != TransactionalMigrationFromMode.none && transactionalMode.nonSerialReadsThroughAccord && transactionalMigrationFromMode.nonSerialWritesThroughAccord() && transactionalMigrationFromMode.nonSerialReadsThroughAccord()) + throw new UnsupportedOperationException("Live migration is not supported, can't safely read when migrating from " + transactionalMigrationFromMode + " to " + transactionalMode); + if (transactionalMode.nonSerialReadsThroughAccord && readCoordinator.isEventuallyConsistent()) { - Stage.READ.execute(new StorageProxy.LocalReadRunnable(rangeCommand, handler, requestTime, trackRepairedStatus)); + //TODO (nicetohave): This is very inefficient because it will not map the the command store owned ranges + // so every command store will return results and most will be discarded due to the limit + // Really we want to split the ranges by command stores owned ranges and then query one at a time + AsyncTxnResult result = StorageProxy.readWithAccord(cm, rangeCommand, replicaPlan.consistencyLevel(), requestTime); + return new AccordRangeResponse(result, rangeCommand.isReversed(), replicaPlan.consistencyLevel(), requestTime); } else { - for (Replica replica : replicaPlan.contacts()) - { - Tracing.trace("Enqueuing request to {}", replica); - ReadCommand command = replica.isFull() ? rangeCommand : rangeCommand.copyAsTransientQuery(replica); - Message message = command.createMessage(trackRepairedStatus && replica.isFull(), requestTime); - MessagingService.instance().sendWithCallback(message, replica.endpoint(), handler); - } - } + ReplicaPlan.SharedForRangeRead sharedReplicaPlan = ReplicaPlan.shared(replicaPlan); + ReadRepair readRepair = + ReadRepair.create(ReadCoordinator.DEFAULT, command, sharedReplicaPlan, requestTime); + DataResolver resolver = + new DataResolver<>(ReadCoordinator.DEFAULT, rangeCommand, sharedReplicaPlan, readRepair, requestTime, trackRepairedStatus); + ReadCallback handler = + new ReadCallback<>(resolver, rangeCommand, sharedReplicaPlan, requestTime); - return new SingleRangeResponse(resolver, handler, readRepair); + if (replicaPlan.contacts().size() == 1 && replicaPlan.contacts().get(0).isSelf()) + { + Stage.READ.execute(new StorageProxy.LocalReadRunnable(rangeCommand, handler, requestTime, trackRepairedStatus)); + } + else + { + for (Replica replica : replicaPlan.contacts()) + { + Tracing.trace("Enqueuing request to {}", replica); + ReadCommand command = replica.isFull() ? rangeCommand : rangeCommand.copyAsTransientQuery(replica); + Message message = command.createMessage(trackRepairedStatus && replica.isFull(), requestTime); + MessagingService.instance().sendWithCallback(message, replica.endpoint(), handler); + } + } + return new CassandraRangeResponse(resolver, handler, readRepair); + } } PartitionIterator sendNextRequests() @@ -233,7 +258,7 @@ public class RangeCommandIterator extends AbstractIterator implemen { ReplicaPlan.ForRangeRead replicaPlan = replicaPlans.next(); - SingleRangeResponse response = query(replicaPlan, i == 0); + IRangeResponse response = query(replicaPlan, readCoordinator, i == 0); concurrentQueries.add(response); readRepairs.add(response.getReadRepair()); // due to RangeMerger, coordinator may fetch more ranges than required by concurrency factor. diff --git a/src/java/org/apache/cassandra/service/reads/range/RangeCommands.java b/src/java/org/apache/cassandra/service/reads/range/RangeCommands.java index bd2acfbd6f..ad6dfc150a 100644 --- a/src/java/org/apache/cassandra/service/reads/range/RangeCommands.java +++ b/src/java/org/apache/cassandra/service/reads/range/RangeCommands.java @@ -34,6 +34,7 @@ import org.apache.cassandra.exceptions.UnavailableException; import org.apache.cassandra.index.Index; import org.apache.cassandra.locator.ReplicaPlans; import org.apache.cassandra.schema.TableMetadata; +import org.apache.cassandra.service.reads.ReadCoordinator; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.transport.Dispatcher; import org.apache.cassandra.utils.FBUtilities; @@ -55,10 +56,11 @@ public class RangeCommands public static PartitionIterator partitions(PartitionRangeReadCommand command, ConsistencyLevel consistencyLevel, + ReadCoordinator readCoordinator, Dispatcher.RequestTime requestTime) { // Note that in general, a RangeCommandIterator will honor the command limit for each range, but will not enforce it globally. - RangeCommandIterator rangeCommands = rangeCommandIterator(command, consistencyLevel, requestTime); + RangeCommandIterator rangeCommands = rangeCommandIterator(command, consistencyLevel, readCoordinator, requestTime); return command.limits().filter(command.postReconciliationProcessing(rangeCommands), command.nowInSec(), command.selectsFullPartition(), @@ -68,6 +70,7 @@ public class RangeCommands @VisibleForTesting static RangeCommandIterator rangeCommandIterator(PartitionRangeReadCommand command, ConsistencyLevel consistencyLevel, + ReadCoordinator readCoordinator, Dispatcher.RequestTime requestTime) { Tracing.trace("Computing ranges to query"); @@ -80,7 +83,7 @@ public class RangeCommands consistencyLevel); if (command.isTopK()) - return new ScanAllRangesCommandIterator(keyspace, replicaPlans, command, replicaPlans.size(), requestTime); + return new ScanAllRangesCommandIterator(keyspace, replicaPlans, command, readCoordinator, replicaPlans.size(), requestTime); int maxConcurrencyFactor = Math.min(replicaPlans.size(), MAX_CONCURRENT_RANGE_REQUESTS); int concurrencyFactor = maxConcurrencyFactor; @@ -111,6 +114,7 @@ public class RangeCommands ReplicaPlanMerger mergedReplicaPlans = new ReplicaPlanMerger(replicaPlans, keyspace, command.metadata().id(), consistencyLevel); return new RangeCommandIterator(mergedReplicaPlans, command, + readCoordinator, concurrencyFactor, maxConcurrencyFactor, replicaPlans.size(), diff --git a/src/java/org/apache/cassandra/service/reads/range/ScanAllRangesCommandIterator.java b/src/java/org/apache/cassandra/service/reads/range/ScanAllRangesCommandIterator.java index c866911e1a..e2a877048f 100644 --- a/src/java/org/apache/cassandra/service/reads/range/ScanAllRangesCommandIterator.java +++ b/src/java/org/apache/cassandra/service/reads/range/ScanAllRangesCommandIterator.java @@ -63,10 +63,11 @@ public class ScanAllRangesCommandIterator extends RangeCommandIterator ScanAllRangesCommandIterator(Keyspace keyspace, CloseableIterator replicaPlans, PartitionRangeReadCommand command, + ReadCoordinator readCoordinator, int totalRangeCount, Dispatcher.RequestTime requestTime) { - super(replicaPlans, command, totalRangeCount, totalRangeCount, totalRangeCount, requestTime); + super(replicaPlans, command, readCoordinator, totalRangeCount, totalRangeCount, totalRangeCount, requestTime); Preconditions.checkState(command.isTopK()); this.keyspace = keyspace; @@ -111,6 +112,6 @@ public class ScanAllRangesCommandIterator extends RangeCommandIterator Tracing.trace("Submitted scanning all ranges requests to {} nodes", nodes); // skip read-repair for top-k query because data mismatch may be caused by top-k algorithm instead of actual inconsistency. - return new SingleRangeResponse(resolver, handler, NoopReadRepair.instance); + return new CassandraRangeResponse(resolver, handler, NoopReadRepair.instance); } } diff --git a/src/java/org/apache/cassandra/service/reads/repair/BlockingReadRepair.java b/src/java/org/apache/cassandra/service/reads/repair/BlockingReadRepair.java index 98b5dc99b2..66f0718dca 100644 --- a/src/java/org/apache/cassandra/service/reads/repair/BlockingReadRepair.java +++ b/src/java/org/apache/cassandra/service/reads/repair/BlockingReadRepair.java @@ -47,11 +47,13 @@ import org.apache.cassandra.locator.ReplicaPlan.ForWrite; import org.apache.cassandra.metrics.ReadRepairMetrics; import org.apache.cassandra.service.accord.AccordService; import org.apache.cassandra.service.accord.api.PartitionKey; +import org.apache.cassandra.service.accord.txn.TxnKeyRead; import org.apache.cassandra.service.accord.txn.TxnQuery; -import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.service.accord.txn.TxnResult; import org.apache.cassandra.service.accord.txn.UnrecoverableRepairUpdate; +import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.ConsensusMigrationMutationHelper; +import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; import org.apache.cassandra.service.reads.ReadCoordinator; import org.apache.cassandra.tcm.ClusterMetadata; import org.apache.cassandra.tracing.Tracing; @@ -59,7 +61,6 @@ import org.apache.cassandra.transport.Dispatcher; import org.apache.cassandra.utils.concurrent.Future; import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException; -import static com.google.common.base.Preconditions.checkNotNull; import static com.google.common.base.Preconditions.checkState; import static java.util.concurrent.TimeUnit.MICROSECONDS; import static java.util.concurrent.TimeUnit.NANOSECONDS; @@ -191,7 +192,7 @@ public class BlockingReadRepair, P extends ReplicaPlan.Fo } @Override - public void repairPartition(DecoratedKey dk, Map mutations, ReplicaPlan.ForWrite writePlan) + public void repairPartition(DecoratedKey dk, Map mutations, ReplicaPlan.ForWrite writePlan, ReadRepairSource rrSource) { // non-Accord reads only ever touch one table and key so all mutations need to be applied either transactionally // or non-transactionally (not a mix). There is no retry loop here because read repair is relatively rare so it racing @@ -203,42 +204,45 @@ public class BlockingReadRepair, P extends ReplicaPlan.Fo // then we take the non-transactional path and the mutations are intercepted in ReadCoordinator.sendRepairMutation // which will ensure the repair mutation runs in the command store thread after preceding transactions are done ClusterMetadata cm = ClusterMetadata.current(); - if (coordinator.isEventuallyConsistent() && ConsensusMigrationMutationHelper.tokenShouldBeWrittenThroughAccord(cm, command.metadata().id, dk.getToken())) - repairTransactionally(dk, mutations, writePlan); + if (coordinator.isEventuallyConsistent() && ConsensusMigrationMutationHelper.tokenShouldBeWrittenThroughAccord(cm, command.metadata().id, dk.getToken(), TransactionalMode::readRepairsThroughAccord, TransactionalMigrationFromMode::readRepairsThroughAccord)) + repairViaAccordTransaction(dk, mutations, writePlan); else - repairNonTransactionally(dk, mutations, writePlan); + repairViaReadCoordinator(dk, mutations, writePlan, rrSource); } - private void repairTransactionally(DecoratedKey dk, Map accordMutations, ForWrite writePlan) + /* + * Create a new Accord transaction to apply this blocking read repair ensuring that any data being written + * consists of already committed Accord writes just by virtue of creating a new transaction which must occur + * after any already partially applied transactions whose writes might be present in the repair mutation. + */ + private void repairViaAccordTransaction(DecoratedKey dk, Map accordMutations, ForWrite writePlan) { checkState(coordinator.isEventuallyConsistent(), "Should only repair transactionally for an eventually consistent read coordinator"); + ReadRepairMetrics.repairedBlockingViaAccord.mark(); PartitionKey partitionKey = new PartitionKey(command.metadata().id, dk); Keys key = Keys.of(partitionKey); // This is going create a new BlockingReadRepair inside an Accord transaction which will go down // the !isEventuallyConsistent path and apply the repairs through Accord command stores using AccordInteropExecution - UnrecoverableRepairUpdate repairUpdate = UnrecoverableRepairUpdate.create(AccordService.instance().nodeId(), this, key, dk, accordMutations, writePlan); - Future repairFuture; - try - { - Txn txn = new Txn.InMemory(Txn.Kind.Read, key, TxnRead.createNoOpRead(key), TxnQuery.NONE, repairUpdate); - repairFuture = Stage.ACCORD_MIGRATION.submit(() -> { - try - { - return AccordService.instance().coordinate(txn, ConsistencyLevel.ANY, requestTime); - } - finally - { - // If we successfully ran the repair txn then the update should definitely - // be there for us to clear which means we are sure it was there to be sent - checkNotNull(UnrecoverableRepairUpdate.removeInflightUpdate(repairUpdate.updateKey)); - } - }); - } - catch (Throwable t) - { - UnrecoverableRepairUpdate.removeInflightUpdate(repairUpdate.updateKey); - throw t; - } + UnrecoverableRepairUpdate repairUpdate = new UnrecoverableRepairUpdate(AccordService.instance().nodeId(), this, key, dk, accordMutations, writePlan); + + /* + * The motivation for using a read to apply read repair is that we want to apply the writes in the execute phase + * so it takes fewer roundtrips and re-use a lot of the AccordInteropExecution code. We don't want to wait for + * the extra roundtrip for apply since this is blocking a read. + * + * The reason this is safe/correct even though read transactions commute with each other is that read transactions + * don't return a result when they are recovered so there is no race with recovery coordinators to worry about. + * The remaining concern of a Read transaction seeing a torn write from an Accord transaction can't happen because + * this RR mutation only contains already applied Accord writes and possibly some non-transactional writes + * that need to be read repaired. + * + * Really the partialy applied Accord writes could just be barriered instead of read repaired, but we use this + * approach so we can read repair non-transactional writes as well. This doesn't make that any more deterministic + * since overlapping non-transactional writes with transactional reads will never be deterministic, but it combines + * the two things into the same mechanism and we can't tell the origin of the writes needing read repair anyways. + */ + Txn txn = new Txn.InMemory(Txn.Kind.Read, key, TxnKeyRead.createNoOpRead(key), TxnQuery.NONE, repairUpdate); + Future repairFuture = Stage.ACCORD_MIGRATION.submit(() -> AccordService.instance().coordinate(txn, ConsistencyLevel.ANY, requestTime)); repairs.add(new PendingPartitionRepair() { @@ -265,8 +269,17 @@ public class BlockingReadRepair, P extends ReplicaPlan.Fo }); } - private void repairNonTransactionally(DecoratedKey dk, Map mutations, ForWrite writePlan) + /* + * ReadCoordinator could be an Accord transaction if this is already in an Accord transaction or a regular + * non-transactional read coordinator. We might take this path because transactional repair is not needed, or this + * is an Accord transaction and the Accord read coordinator will take care of proxying the mutations through command + * stores + */ + private void repairViaReadCoordinator(DecoratedKey dk, Map mutations, ForWrite writePlan, ReadRepairSource rrSource) { + // Accord read at QUORUM and found it needed to read repair, this means txn recovery is non-deterministic + if (rrSource == ReadRepairSource.OTHER && !coordinator.isEventuallyConsistent()) + ReadRepairMetrics.repairedBlockingFromAccord.mark(); BlockingPartitionRepair blockingRepair = new BlockingPartitionRepair(coordinator, dk, mutations, writePlan); blockingRepair.sendInitialRepairs(); repairs.add(blockingRepair); @@ -275,7 +288,7 @@ public class BlockingReadRepair, P extends ReplicaPlan.Fo public void repairPartitionDirectly(ReadCoordinator readCoordinator, DecoratedKey dk, Map mutations, ForWrite writePlan) { ReadRepair delegateRR = ReadRepairStrategy.BLOCKING.create(readCoordinator, command, replicaPlan, requestTime); - delegateRR.repairPartition(dk, mutations, writePlan); + delegateRR.repairPartition(dk, mutations, writePlan, ReadRepairSource.REPAIR_VIA_ACCORD); delegateRR.maybeSendAdditionalWrites(); delegateRR.awaitWrites(); } diff --git a/src/java/org/apache/cassandra/service/reads/repair/NoopReadRepair.java b/src/java/org/apache/cassandra/service/reads/repair/NoopReadRepair.java index 2aad00bc9d..c44a0b9ce0 100644 --- a/src/java/org/apache/cassandra/service/reads/repair/NoopReadRepair.java +++ b/src/java/org/apache/cassandra/service/reads/repair/NoopReadRepair.java @@ -76,7 +76,7 @@ public class NoopReadRepair, P extends ReplicaPlan.ForRea } @Override - public void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan) + public void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan, ReadRepairSource rrSource) { } diff --git a/src/java/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepair.java b/src/java/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepair.java index 3a91cf67fc..9560247ef0 100644 --- a/src/java/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepair.java +++ b/src/java/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepair.java @@ -64,7 +64,7 @@ public class ReadOnlyReadRepair, P extends ReplicaPlan.Fo } @Override - public void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan) + public void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan, ReadRepairSource rrSource) { throw new UnsupportedOperationException("ReadOnlyReadRepair shouldn't be trying to repair partitions"); } diff --git a/src/java/org/apache/cassandra/service/reads/repair/ReadRepair.java b/src/java/org/apache/cassandra/service/reads/repair/ReadRepair.java index 2b73ab83b1..66703bb56e 100644 --- a/src/java/org/apache/cassandra/service/reads/repair/ReadRepair.java +++ b/src/java/org/apache/cassandra/service/reads/repair/ReadRepair.java @@ -90,11 +90,20 @@ public interface ReadRepair, P extends ReplicaPlan.ForRea */ public void awaitWrites(); + // For metrics need to know the source of the repair + enum ReadRepairSource + { + // Running a dedicated repair transacation to do RR for a non-transactional reads + REPAIR_VIA_ACCORD, + // Read repair from a regular Accord transaction or non-transactional read + OTHER + } + /** * Repairs a partition _after_ receiving data responses. This method receives replica list, since * we will block repair only on the replicas that have responded. */ - void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan); + void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan, ReadRepairSource rrSource); /** * Repairs a partition using the provided read coordinator diff --git a/src/java/org/apache/cassandra/service/reads/repair/RowIteratorMergeListener.java b/src/java/org/apache/cassandra/service/reads/repair/RowIteratorMergeListener.java index be6718fc47..f279a290d0 100644 --- a/src/java/org/apache/cassandra/service/reads/repair/RowIteratorMergeListener.java +++ b/src/java/org/apache/cassandra/service/reads/repair/RowIteratorMergeListener.java @@ -52,6 +52,7 @@ import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.locator.Replica; import org.apache.cassandra.locator.ReplicaPlan; import org.apache.cassandra.schema.ColumnMetadata; +import org.apache.cassandra.service.reads.repair.ReadRepair.ReadRepairSource; public class RowIteratorMergeListener> implements UnfilteredRowIterators.MergeListener @@ -397,6 +398,6 @@ public class RowIteratorMergeListener> mutations.put(replica, mutation); } - readRepair.repairPartition(partitionKey, mutations, repairPlan); + readRepair.repairPartition(partitionKey, mutations, repairPlan, ReadRepairSource.OTHER); } } diff --git a/src/java/org/apache/cassandra/tcm/Epoch.java b/src/java/org/apache/cassandra/tcm/Epoch.java index eeeefac359..6ddc86b456 100644 --- a/src/java/org/apache/cassandra/tcm/Epoch.java +++ b/src/java/org/apache/cassandra/tcm/Epoch.java @@ -30,6 +30,7 @@ import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; import org.apache.cassandra.tcm.serialization.MetadataSerializer; import org.apache.cassandra.tcm.serialization.Version; +import org.apache.cassandra.utils.ObjectSizes; import org.apache.cassandra.utils.vint.VIntCoding; public class Epoch implements Comparable, Serializable @@ -62,6 +63,7 @@ public class Epoch implements Comparable, Serializable public static final Epoch UPGRADE_STARTUP = new Epoch(Long.MIN_VALUE); public static final Epoch UPGRADE_GOSSIP = new Epoch(Long.MIN_VALUE + 1); private static final Set beforeFirst = Sets.newHashSet(EMPTY, UPGRADE_GOSSIP, UPGRADE_STARTUP); + private static final long EMPTY_SIZE = ObjectSizes.measure(EMPTY); private final long epoch; @@ -207,4 +209,9 @@ public class Epoch implements Comparable, Serializable return VIntCoding.computeUnsignedVIntSize(t.epoch); } } + + public long estimatedSizeOnHeap() + { + return EMPTY_SIZE; + } } diff --git a/src/java/org/apache/cassandra/utils/CollectionSerializers.java b/src/java/org/apache/cassandra/utils/CollectionSerializers.java index 0cf7fce0d9..fe5cc78031 100644 --- a/src/java/org/apache/cassandra/utils/CollectionSerializers.java +++ b/src/java/org/apache/cassandra/utils/CollectionSerializers.java @@ -183,7 +183,7 @@ public class CollectionSerializers public static Map deserializeMap(DataInputPlus in, int version, IVersionedSerializer keySerializer, IVersionedSerializer valueSerializer) throws IOException { - return deserializeMap(in, version, keySerializer, valueSerializer, newHashMap()); + return deserializeMap(in, version, keySerializer, valueSerializer, Maps::newHashMapWithExpectedSize); } public static long serializedCollectionSize(Collection values, int version, IVersionedSerializer valueSerializer) diff --git a/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java b/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java index a4549788f8..9107848f74 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java +++ b/test/distributed/org/apache/cassandra/distributed/test/TestBaseImpl.java @@ -21,6 +21,7 @@ package org.apache.cassandra.distributed.test; import java.math.BigDecimal; import java.math.BigInteger; import java.net.InetAddress; +import java.net.InetSocketAddress; import java.nio.ByteBuffer; import java.util.ArrayList; import java.util.Arrays; @@ -29,6 +30,8 @@ import java.util.LinkedHashMap; import java.util.List; import java.util.Map; import java.util.UUID; +import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.atomic.AtomicLong; import java.util.function.Function; import java.util.stream.Collectors; @@ -37,6 +40,8 @@ import com.google.common.util.concurrent.ListenableFuture; import com.google.common.util.concurrent.ListenableFutureTask; import org.junit.After; import org.junit.BeforeClass; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CassandraRelevantProperties; import org.apache.cassandra.cql3.Duration; @@ -64,8 +69,11 @@ import org.apache.cassandra.distributed.api.ICoordinator; import org.apache.cassandra.distributed.api.IInstance; import org.apache.cassandra.distributed.api.IInstanceConfig; import org.apache.cassandra.distributed.api.IInvokableInstance; +import org.apache.cassandra.distributed.api.IMessage; +import org.apache.cassandra.distributed.api.IMessageSink; import org.apache.cassandra.distributed.api.NodeToolResult; import org.apache.cassandra.distributed.shared.DistributedTestBase; +import org.apache.cassandra.net.Verb; import org.apache.cassandra.service.accord.AccordCache; import static java.lang.System.currentTimeMillis; @@ -79,12 +87,43 @@ import static org.assertj.core.api.Assertions.fail; // checkstyle: suppress below 'blockSystemPropertyUsage' public class TestBaseImpl extends DistributedTestBase { + private static final Logger logger = LoggerFactory.getLogger(TestBaseImpl.class); + public static final Object[][] EMPTY_ROWS = new Object[0][]; public static final boolean[] BOOLEANS = new boolean[]{ false, true }; + private static final AtomicLong ZERO = new AtomicLong(); + protected static final Map messageCounts = new ConcurrentHashMap<>(); + + protected static class MessageCountingSink implements IMessageSink + { + private final Cluster cluster; + + public MessageCountingSink(Cluster cluster) + { + this.cluster = cluster; + } + + @Override + public void accept(InetSocketAddress to, IMessage message) + { + Verb verb = Verb.fromId(message.verb()); + logger.debug("verb {} to {} message {}", verb, to, message); + messageCounts.computeIfAbsent(verb, ignored -> new AtomicLong()).incrementAndGet(); + cluster.get(to).receiveMessage(message); + } + } + + // Only works if MessageCountingSink is set on the cluster + public static long messageCount(Verb verb) + { + return messageCounts.getOrDefault(verb, ZERO).get(); + } + @After public void afterEach() { super.afterEach(); + messageCounts.clear(); } @BeforeClass diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordCQLTestBase.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordCQLTestBase.java index 28cb069b1e..c7295517c2 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordCQLTestBase.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordCQLTestBase.java @@ -22,9 +22,12 @@ import java.io.IOException; import java.nio.ByteBuffer; import java.util.ArrayList; import java.util.Arrays; +import java.util.HashMap; +import java.util.Iterator; import java.util.List; import java.util.Map; import java.util.Objects; +import java.util.Random; import java.util.Set; import java.util.concurrent.ExecutorService; import java.util.concurrent.Executors; @@ -62,6 +65,7 @@ import org.apache.cassandra.service.consensus.TransactionalMode; import org.apache.cassandra.service.consensus.migration.TransactionalMigrationFromMode; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FailingConsumer; +import org.apache.cassandra.utils.Pair; import org.assertj.core.api.Assertions; import static java.lang.String.format; @@ -73,16 +77,15 @@ import static org.apache.cassandra.distributed.util.QueryResultUtil.assertThat; import static org.junit.Assert.assertArrayEquals; import static org.junit.Assert.assertEquals; import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertTrue; import static org.junit.Assert.fail; public abstract class AccordCQLTestBase extends AccordTestBase { private static final Logger logger = LoggerFactory.getLogger(AccordCQLTestBase.class); - private final TransactionalMode transactionalMode; - protected AccordCQLTestBase(TransactionalMode transactionalMode) { - this.transactionalMode = transactionalMode; + super(transactionalMode); } @Override @@ -296,6 +299,140 @@ public abstract class AccordCQLTestBase extends AccordTestBase }); } + @Test + public void testRangeReadPageOne() throws Exception + { + testRangeRead(1); + } + + @Test + public void testRangeReadSmallPage() throws Exception + { + testRangeRead(2); + } + + @Test + public void testRangeReadExactPage() throws Exception + { + testRangeRead(100); + } + + @Test + public void testRangeReadLargePage() throws Exception + { + testRangeRead(200); + } + + @Test + public void testRangeReadClosePageLT() throws Exception + { + testRangeRead(99); + } + + @Test + public void testRangeReadClosePageGT() throws Exception + { + testRangeRead(101); + } + + private void testRangeRead(int pageSize) throws Exception + { + test(cluster -> { + Random r = new Random(0); + Map, Object[]> insertedRows = new HashMap<>(); + for (int i = 0; i < 10; i++) + { + int k = r.nextInt(); + for (int j = 0; j < 10; j++) + { + cluster.coordinator(1).execute("INSERT INTO " + qualifiedAccordTableName + "(k, c, v) VALUES (?, ?, ?);", ConsistencyLevel.ALL, k, j, i + j); + insertedRows.put(Pair.create(k, j), new Object[] {k, j, i + j}); + } + } + + Iterator iterator = cluster.coordinator(1).executeWithPaging("SELECT * FROM " + qualifiedAccordTableName + " WHERE TOKEN(k) > " + Long.MIN_VALUE + " AND TOKEN(k) < " + Long.MAX_VALUE, ConsistencyLevel.ALL, pageSize); + List resultRows = ImmutableList.copyOf(iterator); + resultRows.forEach(row -> System.out.println(Arrays.toString(row))); + Integer lastPartitionKey = null; + int currentRowKey = 0; + for (Object[] row : resultRows) + { + assertEquals(currentRowKey, row[1]); + + if (lastPartitionKey == null) + lastPartitionKey = (Integer)row[0]; + else + assertEquals(lastPartitionKey, row[0]); + + if (currentRowKey == 9) + { + currentRowKey = 0; + lastPartitionKey = null; + } + else + currentRowKey++; + + Object[] expected = insertedRows.remove(Pair.create(row[0], row[1])); + assertEquals(expected, row); + } + assertTrue(insertedRows.isEmpty()); + }); + } + + @Test + public void testIN() throws Exception + { + test(cluster -> { + Random r = new Random(0); + Map, Object[]> insertedRows = new HashMap<>(); + List partitionKeys = new ArrayList<>(); + for (int i = 0; i < 10; i++) + { + int k = r.nextInt(); + for (int j = 0; j < 10; j++) + { + cluster.coordinator(1).execute("INSERT INTO " + qualifiedAccordTableName + "(k, c, v) VALUES (?, ?, ?);", ConsistencyLevel.ALL, k, j, i + j); + insertedRows.put(Pair.create(k, j), new Object[] {k, j, i + j}); + partitionKeys.add(k); + } + } + + String query = "SELECT * FROM " + qualifiedAccordTableName + " WHERE k IN ("; + for (Integer key : partitionKeys) + { + query = query + key + ", "; + } + query = query.substring(0, query.length() - 2); + query = query + ")"; + Iterator iterator = cluster.coordinator(1).executeWithPaging(query, ConsistencyLevel.ALL, 2); + List resultRows = ImmutableList.copyOf(iterator); + resultRows.forEach(row -> System.out.println(Arrays.toString(row))); + Integer lastPartitionKey = null; + int currentRowKey = 0; + for (Object[] row : resultRows) + { + assertEquals(currentRowKey, row[1]); + + if (lastPartitionKey == null) + lastPartitionKey = (Integer)row[0]; + else + assertEquals(lastPartitionKey, row[0]); + + if (currentRowKey == 9) + { + currentRowKey = 0; + lastPartitionKey = null; + } + else + currentRowKey++; + + Object[] expected = insertedRows.remove(Pair.create(row[0], row[1])); + assertEquals(expected, row); + } + assertTrue(insertedRows.isEmpty()); + }); + } + @Test public void testMultiPartitionReturn() throws Exception { @@ -2781,8 +2918,6 @@ public abstract class AccordCQLTestBase extends AccordTestBase }); } - // TODO: Re-enable when TrM integration is working - @Ignore @Test public void testCASAndSerialRead() throws Exception { @@ -2823,7 +2958,7 @@ public abstract class AccordCQLTestBase extends AccordTestBase assertEquals(1, rangeDeletionCheck.length); // Make sure all the consensus using queries actually were run on Accord - if (transactionalMode.writesThroughAccord) + if (transactionalMode.nonSerialWritesThroughAccord) assertEquals( 20, getAccordCoordinateCount() - startingAccordCoordinateCount); else // Non-serial writes don't go through Accord in these modes diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordInteroperabilityTest.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordInteroperabilityTest.java index fcd9b1dabe..7deb529191 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordInteroperabilityTest.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordInteroperabilityTest.java @@ -19,20 +19,37 @@ package org.apache.cassandra.distributed.test.accord; import java.io.IOException; +import java.util.ArrayList; +import java.util.List; import java.util.function.Function; +import org.junit.After; import org.junit.BeforeClass; import org.junit.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.ConsistencyLevel; +import org.apache.cassandra.db.DataRange; +import org.apache.cassandra.db.filter.ColumnFilter; +import org.apache.cassandra.db.memtable.Memtable; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; import org.apache.cassandra.distributed.api.ICoordinator; import org.apache.cassandra.distributed.shared.AssertUtils; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.io.sstable.SSTableReadsListener; +import org.apache.cassandra.net.Verb; import org.apache.cassandra.service.accord.IAccordService; +import static com.google.common.base.Throwables.getStackTraceAsString; +import static org.apache.cassandra.Util.dk; +import static org.apache.commons.collections.ListUtils.synchronizedList; import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertTrue; import static org.junit.Assert.fail; public class AccordInteroperabilityTest extends AccordTestBase @@ -51,6 +68,12 @@ public class AccordInteroperabilityTest extends AccordTestBase AccordTestBase.setupCluster(builder -> builder, 3); } + @After + public void tearDown() + { + SHARED_CLUSTER.setMessageSink(null); + } + @Test public void testSerialReadDescending() throws Throwable { @@ -67,6 +90,73 @@ public class AccordInteroperabilityTest extends AccordTestBase ); } + @Test + public void testApplyIsInteropApply() throws Throwable + { + test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, PRIMARY KEY(k, c)) WITH transactional_mode='mixed_reads'", + cluster -> { + MessageCountingSink messageCountingSink = new MessageCountingSink(SHARED_CLUSTER); + List failures = synchronizedList(new ArrayList<>()); + // Verify that the apply response is only sent after the row has been inserted + // TODO (required): Need to delay mutation stage/mutation to ensure this has time to catch it + SHARED_CLUSTER.setMessageSink((to, message) -> { + try + { + if (message.verb() == Verb.ACCORD_APPLY_RSP.id) + { + String currentThread = Thread.currentThread().getName(); + char nodeIndexChar = currentThread.charAt(4); + int nodeIndex = Integer.parseInt(String.valueOf(nodeIndexChar)); + try + { + String keyspace = KEYSPACE; + String tableName = accordTableName; + String fail = SHARED_CLUSTER.get(nodeIndex).callOnInstance(() -> { + ColumnFamilyStore cfs = ColumnFamilyStore.getIfExists(keyspace, tableName); + Memtable memtable = cfs.getCurrentMemtable(); + assertEquals(1, memtable.partitionCount()); + UnfilteredPartitionIterator partitions = memtable.partitionIterator(ColumnFilter.all(cfs.metadata()), DataRange.allData(cfs.getPartitioner()), SSTableReadsListener.NOOP_LISTENER); + assertTrue(partitions.hasNext()); + UnfilteredRowIterator rows = partitions.next(); + assertEquals(dk(42), rows.partitionKey()); + assertFalse(partitions.hasNext()); + assertTrue(rows.hasNext()); + Row row = (Row)rows.next(); + assertFalse(rows.hasNext()); + return null; + }); + if (fail != null) + failures.add(fail); + } + catch (Exception e) + { + failures.add(getStackTraceAsString(e)); + } + } + } + finally + { + messageCountingSink.accept(to, message); + } + }); + cluster.coordinator(1).execute("INSERT INTO " + qualifiedAccordTableName + " (k, c, v) VALUES (42, 2, 3)", org.apache.cassandra.distributed.api.ConsistencyLevel.QUORUM); + assertEquals(3, messageCounts.get(Verb.ACCORD_INTEROP_APPLY_REQ).get()); + assertTrue(failures.toString(), failures.isEmpty()); + }); + } + + @Test + public void testReadIsAtQuorum() throws Throwable + { + test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, PRIMARY KEY(k, c)) WITH transactional_mode='unsafe_writes'", + cluster -> { + SHARED_CLUSTER.setMessageSink(new MessageCountingSink(SHARED_CLUSTER)); + cluster.coordinator(1).execute("SELECT * FROM " + qualifiedAccordTableName + " WHERE k = 0", org.apache.cassandra.distributed.api.ConsistencyLevel.SERIAL); + assertEquals(2, messageCounts.get(Verb.ACCORD_INTEROP_COMMIT_REQ).get()); + assertEquals(2, messageCounts.get(Verb.ACCORD_INTEROP_READ_RSP).get()); + }); + } + private static Object[][] assertTargetAccordRead(Function query, int coordinatorIndex, int key, int expectedAccordReadCount) { int startingReadCount = getAccordReadCount(coordinatorIndex); diff --git a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordTestBase.java b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordTestBase.java index 79109cf202..6fda5270d3 100644 --- a/test/distributed/org/apache/cassandra/distributed/test/accord/AccordTestBase.java +++ b/test/distributed/org/apache/cassandra/distributed/test/accord/AccordTestBase.java @@ -124,6 +124,17 @@ public abstract class AccordTestBase extends TestBaseImpl protected String regularTableName; protected String qualifiedRegularTableName; + protected final TransactionalMode transactionalMode; + + protected AccordTestBase() + { + this.transactionalMode = TransactionalMode.full; + } + + protected AccordTestBase(TransactionalMode transactionalMode) { + this.transactionalMode = transactionalMode; + } + public static void setupCluster(Function options, int nodes) throws IOException { SHARED_CLUSTER = createCluster(nodes, options); @@ -207,7 +218,7 @@ public abstract class AccordTestBase extends TestBaseImpl protected void test(FailingConsumer fn) throws Exception { - test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, primary key (k, c)) WITH transactional_mode='full'", fn); + test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v int, primary key (k, c)) WITH transactional_mode='" + transactionalMode + "'", fn); } protected static ConsensusMigrationState getMigrationStateSnapshot(IInvokableInstance instance) throws IOException diff --git a/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java b/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java index 16233a5e8d..a1235f7f03 100644 --- a/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java +++ b/test/simulator/test/org/apache/cassandra/simulator/test/AccordJournalSimulationTest.java @@ -26,33 +26,30 @@ import java.util.zip.Checksum; import com.google.common.collect.ImmutableMap; import com.google.common.jimfs.Jimfs; +import org.junit.Assert; +import org.junit.Test; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; import org.apache.cassandra.concurrent.ExecutorFactory; import org.apache.cassandra.concurrent.ExecutorPlus; import org.apache.cassandra.config.AccordSpec; import org.apache.cassandra.config.Config; +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.config.ParameterizedClass; +import org.apache.cassandra.db.Keyspace; import org.apache.cassandra.config.DurationSpec; import org.apache.cassandra.db.TypeSizes; +import org.apache.cassandra.dht.Murmur3Partitioner; import org.apache.cassandra.io.filesystem.ListenableFileSystem; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.io.util.File; import org.apache.cassandra.journal.Journal; import org.apache.cassandra.journal.KeySupport; import org.apache.cassandra.journal.RecordPointer; import org.apache.cassandra.journal.SegmentCompactor; import org.apache.cassandra.journal.ValueSerializer; - -import org.junit.Assert; -import org.junit.Test; - -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; - -import org.apache.cassandra.config.DatabaseDescriptor; -import org.apache.cassandra.config.ParameterizedClass; -import org.apache.cassandra.db.Keyspace; -import org.apache.cassandra.dht.Murmur3Partitioner; -import org.apache.cassandra.io.util.File; import org.apache.cassandra.utils.Isolated; import org.apache.cassandra.utils.concurrent.CountDownLatch; diff --git a/test/unit/org/apache/cassandra/Util.java b/test/unit/org/apache/cassandra/Util.java index 7b4ab79a9f..a01311942f 100644 --- a/test/unit/org/apache/cassandra/Util.java +++ b/test/unit/org/apache/cassandra/Util.java @@ -725,6 +725,16 @@ public class Util } } + public static void spinAssertEquals(int expected, Supplier actualSupplier) + { + spinAssertEquals((long)expected, () -> ((Number)actualSupplier.get()).longValue()); + } + + public static void spinAssertEquals(Object expected, Supplier actualSupplier) + { + spinAssertEquals(null, expected, actualSupplier, 10, TimeUnit.SECONDS); + } + public static void spinAssertEquals(Object expected, Supplier actualSupplier, int timeoutInSeconds) { spinAssertEquals(null, expected, actualSupplier, timeoutInSeconds, TimeUnit.SECONDS); diff --git a/test/unit/org/apache/cassandra/cql3/statements/TxnDataNameTest.java b/test/unit/org/apache/cassandra/cql3/statements/TxnDataNameTest.java index d084082a08..22703609f5 100644 --- a/test/unit/org/apache/cassandra/cql3/statements/TxnDataNameTest.java +++ b/test/unit/org/apache/cassandra/cql3/statements/TxnDataNameTest.java @@ -20,52 +20,75 @@ package org.apache.cassandra.cql3.statements; import org.junit.Test; -import org.apache.cassandra.io.util.DataInputBuffer; -import org.apache.cassandra.io.util.DataOutputBuffer; -import org.apache.cassandra.service.accord.txn.TxnDataName; -import org.apache.cassandra.service.accord.txn.TxnRead; -import org.apache.cassandra.utils.Generators; -import org.assertj.core.api.Assertions; +import org.apache.cassandra.service.accord.txn.TxnData.TxnDataNameKind; import org.quicktheories.core.Gen; import org.quicktheories.generators.SourceDSL; +import org.quicktheories.impl.Constraint; +import static org.apache.cassandra.service.accord.txn.TxnData.TXN_DATA_NAME_INDEX_MAX; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataName; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataNameIndex; +import static org.apache.cassandra.service.accord.txn.TxnData.txnDataNameKind; import static org.apache.cassandra.utils.FailingConsumer.orFail; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.fail; import static org.quicktheories.QuickTheory.qt; public class TxnDataNameTest { @Test - public void serde() + public void buildAndAccess() { - try (DataOutputBuffer out = new DataOutputBuffer()) + qt().forAll(gen()).checkAssert(orFail(test -> { + if (test.index < 0 || test.index > TXN_DATA_NAME_INDEX_MAX) + { + try + { + txnDataName(test.kind, test.index); + fail("Expect IllegalArgumentException"); + } + catch (IllegalArgumentException e) + { + // expected + } + return; + } + + int txnDataName = txnDataName(test.kind, test.index); + assertEquals(test.kind, txnDataNameKind(txnDataName)); + assertEquals(test.index, txnDataNameIndex(txnDataName)); + })); + } + + @Test + public void testIndex() + { + TxnDataNameKind kind = TxnDataNameKind.values()[TxnDataNameKind.values().length - 1]; + int txnDataName = txnDataName(kind, 0); + assertEquals(0, txnDataNameIndex(txnDataName)); + txnDataName = txnDataName(kind, TXN_DATA_NAME_INDEX_MAX); + assertEquals(TXN_DATA_NAME_INDEX_MAX, txnDataNameIndex(txnDataName)); + } + + static class TestData + { + final TxnDataNameKind kind; + final int index; + + public TestData(TxnDataNameKind kind, int index) { - qt().forAll(gen()).checkAssert(orFail(name -> { - out.clear(); - - long expectedSize = TxnDataName.serializer.serializedSize(name, 12); - TxnDataName.serializer.serialize(name, out, 12); - Assertions.assertThat(out.getLength()).isEqualTo(expectedSize); - - TxnDataName read = TxnDataName.serializer.deserialize(new DataInputBuffer(out.toByteArray()), 12); - Assertions.assertThat(read).isEqualTo(name); - })); + this.kind = kind; + this.index = index; } } - public static Gen gen() + public static Gen gen() { - Gen kindGen = SourceDSL.arbitrary().enumValues(TxnDataName.Kind.class); - Gen symbolGen = Generators.SYMBOL_GEN; + Gen kindGen = SourceDSL.arbitrary().enumValues(TxnDataNameKind.class); return rnd -> { - TxnDataName.Kind kind = kindGen.generate(rnd); - switch (kind) - { - case USER: return TxnDataName.user(symbolGen.generate(rnd)); - case RETURNING: return TxnDataName.returning(); - case AUTO_READ: return new TxnDataName(kind, symbolGen.generate(rnd), symbolGen.generate(rnd), symbolGen.generate(rnd)); - case CAS_READ: return TxnRead.CAS_READ; - default: throw new IllegalArgumentException("Unknown kind: " + kind); - } + TxnDataNameKind kind = kindGen.generate(rnd); + int index = (int)(rnd.next(Constraint.zeroToOne()) == 0 ? rnd.next(Constraint.between(0, TXN_DATA_NAME_INDEX_MAX)) : rnd.next(Constraint.between(Integer.MIN_VALUE, Integer.MAX_VALUE))); + return new TestData(kind, index); }; } } \ No newline at end of file diff --git a/test/unit/org/apache/cassandra/db/virtual/AccordDebugKeyspaceTest.java b/test/unit/org/apache/cassandra/db/virtual/AccordDebugKeyspaceTest.java index d79a346c36..1d3ee7d2ad 100644 --- a/test/unit/org/apache/cassandra/db/virtual/AccordDebugKeyspaceTest.java +++ b/test/unit/org/apache/cassandra/db/virtual/AccordDebugKeyspaceTest.java @@ -31,9 +31,10 @@ import org.junit.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import accord.primitives.SaveStatus; +import accord.Utils; import accord.messages.TxnRequest; import accord.primitives.Routable; +import accord.primitives.SaveStatus; import accord.primitives.Txn; import accord.primitives.TxnId; import accord.utils.async.AsyncChains; @@ -89,8 +90,8 @@ public class AccordDebugKeyspaceTest extends CQLTester Txn txn = createTxn(wrapInTxn(String.format("INSERT INTO %s.%s(k, c, v) VALUES (?, ?, ?)", KEYSPACE, tableName)), 0, 0, 0); AsyncChains.getBlocking(accord.node().coordinate(id, txn)); - assertRows(execute(QUERY_TXN_BLOCKED_BY, id.toString()), - row(id.toString(), anyInt(), 0, ByteBufferUtil.EMPTY_BYTE_BUFFER, "Self", any(), null, anyOf(SaveStatus.ReadyToExecute.name(), SaveStatus.Applying.name(), SaveStatus.Applied.name()))); + Utils.spinUntilSuccess(() -> assertRows(execute(QUERY_TXN_BLOCKED_BY, id.toString()), + row(id.toString(), anyInt(), 0, ByteBufferUtil.EMPTY_BYTE_BUFFER, "Self", any(), null, anyOf(SaveStatus.ReadyToExecute.name(), SaveStatus.Applying.name(), SaveStatus.Applied.name())))); } @Test @@ -103,7 +104,13 @@ public class AccordDebugKeyspaceTest extends CQLTester String tableName = createTable("CREATE TABLE %s (k int, c int, v int, PRIMARY KEY (k, c)) WITH transactional_mode = 'full'"); var accord = accord(); TxnId id = accord.node().nextTxnId(Txn.Kind.Write, Routable.Domain.Key); - Txn txn = createTxn(wrapInTxn(String.format("INSERT INTO %s.%s(k, c, v) VALUES (?, ?, ?)", KEYSPACE, tableName)), 0, 0, 0); + String insertTxn = String.format("BEGIN TRANSACTION\n" + + " LET r = (SELECT * FROM %s.%s WHERE k = ? AND c = ?);\n" + + " IF r IS NULL THEN\n " + + " INSERT INTO %s.%s (k, c, v) VALUES (?, ?, ?);\n" + + " END IF\n" + + "COMMIT TRANSACTION", KEYSPACE, tableName, KEYSPACE, tableName); + Txn txn = createTxn(insertTxn, 0, 0, 0, 0, 0); accord.node().coordinate(id, txn); filter.preAccept.awaitThrowUncheckedOnInterrupt(); @@ -131,7 +138,13 @@ public class AccordDebugKeyspaceTest extends CQLTester String tableName = createTable("CREATE TABLE %s (k int, c int, v int, PRIMARY KEY (k, c)) WITH transactional_mode = 'full'"); var accord = accord(); TxnId first = accord.node().nextTxnId(Txn.Kind.Write, Routable.Domain.Key); - accord.node().coordinate(first, createTxn(wrapInTxn(String.format("INSERT INTO %s.%s(k, c, v) VALUES (?, ?, ?)", KEYSPACE, tableName)), 0, 0, 0)); + String insertTxn = String.format("BEGIN TRANSACTION\n" + + " LET r = (SELECT * FROM %s.%s WHERE k = ? AND c = ?);\n" + + " IF r IS NULL THEN\n " + + " INSERT INTO %s.%s (k, c, v) VALUES (?, ?, ?);\n" + + " END IF\n" + + "COMMIT TRANSACTION", KEYSPACE, tableName, KEYSPACE, tableName); + accord.node().coordinate(first, createTxn(insertTxn, 0, 0, 0, 0, 0)); filter.preAccept.awaitThrowUncheckedOnInterrupt(); assertRows(execute(QUERY_TXN_BLOCKED_BY, first.toString()), @@ -144,7 +157,7 @@ public class AccordDebugKeyspaceTest extends CQLTester filter.reset(); TxnId second = accord.node().nextTxnId(Txn.Kind.Write, Routable.Domain.Key); - accord.node().coordinate(second, createTxn(wrapInTxn(String.format("INSERT INTO %s.%s(k, c, v) VALUES (?, ?, ?)", KEYSPACE, tableName)), 0, 0, 0)); + accord.node().coordinate(second, createTxn(insertTxn, 0, 0, 0, 0, 0)); filter.commit.awaitThrowUncheckedOnInterrupt(); diff --git a/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java b/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java index 2711b20336..ea6587a17d 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordReadRepairTest.java @@ -29,12 +29,19 @@ import org.slf4j.LoggerFactory; import org.apache.cassandra.distributed.Cluster; import org.apache.cassandra.distributed.api.ConsistencyLevel; -import org.apache.cassandra.distributed.api.IMessageFilters.Filter; +import org.apache.cassandra.distributed.shared.Metrics; import org.apache.cassandra.distributed.test.accord.AccordTestBase; import org.apache.cassandra.net.Verb; +import org.apache.cassandra.service.consensus.TransactionalMode; +import static org.apache.cassandra.Util.spinAssertEquals; import static org.apache.cassandra.distributed.shared.AssertUtils.assertRows; import static org.apache.cassandra.distributed.util.QueryResultUtil.assertThat; +import static org.apache.cassandra.net.Verb.HINT_REQ; +import static org.apache.cassandra.net.Verb.MUTATION_REQ; +import static org.apache.cassandra.net.Verb.READ_REPAIR_REQ; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertTrue; public class AccordReadRepairTest extends AccordTestBase { @@ -50,7 +57,7 @@ public class AccordReadRepairTest extends AccordTestBase public static void setupClass() throws IOException { AccordTestBase.setupCluster(builder -> builder, 2); - SHARED_CLUSTER.schemaChange("CREATE TYPE " + KEYSPACE + ".person (height int, age int)"); + SHARED_CLUSTER.setMessageSink(new MessageCountingSink(SHARED_CLUSTER)); } /* @@ -61,7 +68,9 @@ public class AccordReadRepairTest extends AccordTestBase public void testSerialReadRepair() throws Exception { testReadRepair(cluster -> cluster.coordinator(1).execute("SELECT * FROM " + qualifiedAccordTableName + " WHERE k = 1 AND c = 1;", ConsistencyLevel.SERIAL), - new Object[][] {{1, 1, 1, 1}}); + new Object[][] {{1, 1, 1, 1}}, + TransactionalMode.unsafe_writes, + 0, 2, 1, 0); } @Test @@ -69,7 +78,9 @@ public class AccordReadRepairTest extends AccordTestBase { // Even if the condition fails to apply the data checked when applying the condition should be repaired testReadRepair(cluster -> cluster.coordinator(1).execute("INSERT INTO " + qualifiedAccordTableName + " (k, c, v1) VALUES (1, 1, 99) IF NOT EXISTS;", ConsistencyLevel.SERIAL), - new Object[][] {{false, 1, 1, 1, 1}}); + new Object[][] {{false, 1, 1, 1, 1}}, + TransactionalMode.unsafe_writes, + 2, 0, 1, 0); } @Test @@ -77,7 +88,9 @@ public class AccordReadRepairTest extends AccordTestBase { // If the condition applies the read repair should preserve the existing timestamp testReadRepair(cluster -> cluster.coordinator(1).execute("UPDATE " + qualifiedAccordTableName + " SET v2 = 99 WHERE k = 1 and c = 1 IF EXISTS;", ConsistencyLevel.SERIAL), - new Object[][] {{Boolean.TRUE}}); + new Object[][] {{Boolean.TRUE}}, + TransactionalMode.unsafe_writes, + 2, 0, 1, 0); } /* @@ -89,28 +102,50 @@ public class AccordReadRepairTest extends AccordTestBase { for (ConsistencyLevel cl : ImmutableList.of(ConsistencyLevel.QUORUM)) testReadRepair(cluster -> cluster.coordinator(1).execute("SELECT * FROM " + qualifiedAccordTableName + " WHERE k = 1 AND c = 1;", cl), - new Object[][] {{1, 1, 1, 1}}); + new Object[][] {{1, 1, 1, 1}}, + TransactionalMode.unsafe_writes, + 0, 2, 0, 1); } - void testReadRepair(Function accordTxn, Object[][] expected) throws Exception + @Test + public void testNonSerialRangeReadRepair() throws Exception { - test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v1 int, v2 int, PRIMARY KEY ((k), c)) WITH transactional_mode='unsafe_writes';", + for (ConsistencyLevel cl : ImmutableList.of(ConsistencyLevel.QUORUM)) + testReadRepair(cluster -> cluster.coordinator(1).execute("SELECT * FROM " + qualifiedAccordTableName + " WHERE TOKEN(k) > " + Long.MIN_VALUE + " AND TOKEN(k) < " + Long.MAX_VALUE, cl), + new Object[][] {{1, 1, 1, 1}}, + TransactionalMode.unsafe_writes, + 0, 2, 0, 1); + } + + void testReadRepair(Function accordTxn, Object[][] expected, TransactionalMode transactionalMode, int expectedInteropApply, int expectedRegularApply, int expectedReadRepairFromAccord, int expectedReadRepairViaAccord) throws Exception + { + test("CREATE TABLE " + qualifiedAccordTableName + " (k int, c int, v1 int, v2 int, PRIMARY KEY ((k), c)) WITH transactional_mode='" + transactionalMode + "';", cluster -> { - Filter mutationFilter = cluster.filters().verbs(Verb.MUTATION_REQ.id).to(2).drop().on(); - cluster.filters().verbs(Verb.HINT_REQ.id, Verb.HINT_RSP.id).drop().on(); - cluster.coordinator(1).execute("INSERT INTO " + qualifiedAccordTableName + " (k, c, v1, v2) VALUES (1, 1, 1, 1) USING TIMESTAMP 42;", ConsistencyLevel.ONE); - mutationFilter.off(); - Filter blockNodeOneReads = cluster.filters().verbs(Verb.READ_REQ.id).to(1).drop().on(); - assertThat(cluster.coordinator(2).executeWithResult("SELECT * FROM " + qualifiedAccordTableName + " WHERE k = 1 AND c = 1;", ConsistencyLevel.ONE)) - .isEmpty(); - blockNodeOneReads.off(); + Metrics metrics = cluster.get(1).metrics(); + String repairedFromAccordName = metrics.getNames().stream().filter(name -> name.contains("RepairedBlockingFromAccord")).findFirst().get(); + String repairedViaAccordName = metrics.getNames().stream().filter(name -> name.contains("RepairedBlockingViaAccord")).findFirst().get(); + long startingRepairedFromAccord = metrics.getCounter(repairedFromAccordName); + long startingRepairedViaAccord = metrics.getCounter(repairedViaAccordName); + + cluster.filters().verbs(READ_REPAIR_REQ.id, MUTATION_REQ.id, HINT_REQ.id).drop().on(); + cluster.get(1).executeInternal("INSERT INTO " + qualifiedAccordTableName + " (k, c, v1, v2) VALUES (1, 1, 1, 1) USING TIMESTAMP 42;"); + assertThat(cluster.get(2).executeInternalWithResult("SELECT * FROM " + qualifiedAccordTableName + " WHERE k = 1 AND c = 1;")).isEmpty(); // Should perform read repair Object[][] result = accordTxn.apply(cluster); assertRows(result, expected); - blockNodeOneReads.on(); // Side effect of the read repair should be visible now - assertThat(cluster.coordinator(2).executeWithResult("SELECT k, c, v1, WRITETIME(v1) FROM " + qualifiedAccordTableName + " WHERE k = 1 AND c = 1;", ConsistencyLevel.ONE)) + assertThat(cluster.get(2).executeInternalWithResult("SELECT k, c, v1, WRITETIME(v1) FROM " + qualifiedAccordTableName + " WHERE k = 1 AND c = 1;")) .isEqualTo(1, 1, 1, 42L); + assertEquals(expectedInteropApply, messageCount(Verb.ACCORD_INTEROP_APPLY_REQ)); + // Regular apply is async so need to spin + spinAssertEquals(expectedRegularApply, () -> messageCount(Verb.ACCORD_APPLY_REQ)); + assertEquals(1, messageCount(Verb.ACCORD_INTEROP_READ_REPAIR_REQ)); + long readRepairRspCount = messageCount(Verb.ACCORD_INTEROP_READ_REPAIR_RSP); + assertTrue("Should be 1-2 read repair responses depending on whether insufficient occurred", readRepairRspCount >= 1 && readRepairRspCount <= 2); + long repairedFromAccord = metrics.getCounter(repairedFromAccordName) - startingRepairedFromAccord; + assertEquals(expectedReadRepairFromAccord, repairedFromAccord); + long repairedViaAccord = metrics.getCounter(repairedViaAccordName) - startingRepairedViaAccord; + assertEquals(expectedReadRepairViaAccord, repairedViaAccord); }); } } diff --git a/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java b/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java index dbd42297cc..df40d36a1a 100644 --- a/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java +++ b/test/unit/org/apache/cassandra/service/accord/AccordTestUtils.java @@ -99,8 +99,8 @@ import org.apache.cassandra.service.ClientState; import org.apache.cassandra.service.accord.api.AccordAgent; import org.apache.cassandra.service.accord.api.PartitionKey; import org.apache.cassandra.service.accord.txn.TxnData; +import org.apache.cassandra.service.accord.txn.TxnKeyRead; import org.apache.cassandra.service.accord.txn.TxnQuery; -import org.apache.cassandra.service.accord.txn.TxnRead; import org.apache.cassandra.utils.Pair; import org.apache.cassandra.utils.concurrent.Condition; import org.apache.cassandra.utils.concurrent.UncheckedInterruptedException; @@ -235,7 +235,7 @@ public class AccordTestUtils public static Pair processTxnResultDirect(SafeCommandStore safeStore, TxnId txnId, PartialTxn txn, Timestamp executeAt) { - TxnRead read = (TxnRead) txn.read(); + TxnKeyRead read = (TxnKeyRead) txn.read(); Data readData = read.keys().stream().map(key -> { try { @@ -319,7 +319,7 @@ public class AccordTestUtils public static Txn createTxn(Txn.Kind kind, Seekables seekables) { - return new Txn.InMemory(kind, seekables, TxnRead.EMPTY, TxnQuery.NONE, null); + return new Txn.InMemory(kind, seekables, TxnKeyRead.EMPTY, TxnQuery.NONE, null); } public static Ranges fullRange(Txn txn) diff --git a/test/unit/org/apache/cassandra/service/reads/range/RangeCommandIteratorTest.java b/test/unit/org/apache/cassandra/service/reads/range/RangeCommandIteratorTest.java index b13c45bfbc..7cec4c405e 100644 --- a/test/unit/org/apache/cassandra/service/reads/range/RangeCommandIteratorTest.java +++ b/test/unit/org/apache/cassandra/service/reads/range/RangeCommandIteratorTest.java @@ -22,7 +22,6 @@ import java.util.ArrayList; import java.util.List; import com.google.common.collect.Iterators; -import org.apache.cassandra.schema.TableId; import org.junit.BeforeClass; import org.junit.Test; @@ -41,6 +40,8 @@ import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.locator.ReplicaPlan; import org.apache.cassandra.locator.ReplicaPlans; import org.apache.cassandra.schema.KeyspaceParams; +import org.apache.cassandra.schema.TableId; +import org.apache.cassandra.service.reads.ReadCoordinator; import org.apache.cassandra.transport.Dispatcher; import org.apache.cassandra.utils.CloseableIterator; @@ -107,27 +108,27 @@ public class RangeCommandIteratorTest // without range merger, there will be 2 batches requested: 1st batch with 1 range and 2nd batch with remaining ranges CloseableIterator replicaPlans = replicaPlanIterator(keyRange, keyspace, false); - RangeCommandIterator data = new RangeCommandIterator(replicaPlans, command, 1, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); + RangeCommandIterator data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); verifyRangeCommandIterator(data, rows, 2, vnodeCount); // without range merger and initial cf=5, there will be 1 batches requested: 5 vnode ranges for 1st batch replicaPlans = replicaPlanIterator(keyRange, keyspace, false); - data = new RangeCommandIterator(replicaPlans, command, vnodeCount, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); + data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, vnodeCount, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); verifyRangeCommandIterator(data, rows, 1, vnodeCount); // without range merger and max cf=1, there will be 5 batches requested: 1 vnode range per batch replicaPlans = replicaPlanIterator(keyRange, keyspace, false); - data = new RangeCommandIterator(replicaPlans, command, 1, 1, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); + data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); verifyRangeCommandIterator(data, rows, vnodeCount, vnodeCount); // with range merger, there will be only 1 batch requested, as all ranges share the same replica - localhost replicaPlans = replicaPlanIterator(keyRange, keyspace, true); - data = new RangeCommandIterator(replicaPlans, command, 1, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); + data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1000, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); verifyRangeCommandIterator(data, rows, 1, vnodeCount); // with range merger and max cf=1, there will be only 1 batch requested, as all ranges share the same replica - localhost replicaPlans = replicaPlanIterator(keyRange, keyspace, true); - data = new RangeCommandIterator(replicaPlans, command, 1, 1, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); + data = new RangeCommandIterator(replicaPlans, command, ReadCoordinator.DEFAULT, 1, 1, vnodeCount, Dispatcher.RequestTime.forImmediateExecution()); verifyRangeCommandIterator(data, rows, 1, vnodeCount); } diff --git a/test/unit/org/apache/cassandra/service/reads/range/RangeCommandsTest.java b/test/unit/org/apache/cassandra/service/reads/range/RangeCommandsTest.java index 466309fcb2..42ea24173e 100644 --- a/test/unit/org/apache/cassandra/service/reads/range/RangeCommandsTest.java +++ b/test/unit/org/apache/cassandra/service/reads/range/RangeCommandsTest.java @@ -38,6 +38,7 @@ import org.apache.cassandra.distributed.shared.WithProperties; import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.index.StubIndex; import org.apache.cassandra.schema.IndexMetadata; +import org.apache.cassandra.service.reads.ReadCoordinator; import org.apache.cassandra.transport.Dispatcher; import static org.apache.cassandra.config.CassandraRelevantProperties.MAX_CONCURRENT_RANGE_REQUESTS; @@ -78,7 +79,7 @@ public class RangeCommandsTest extends CQLTester // verify that a low concurrency factor is not capped by the max concurrency factor PartitionRangeReadCommand command = command(cfs, 50, 50); - try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, Dispatcher.RequestTime.forImmediateExecution()); + try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, ReadCoordinator.DEFAULT, Dispatcher.RequestTime.forImmediateExecution()); ReplicaPlanIterator ranges = new ReplicaPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE)) { assertEquals(2, partitions.concurrencyFactor()); @@ -88,7 +89,7 @@ public class RangeCommandsTest extends CQLTester // verify that a high concurrency factor is capped by the max concurrency factor command = command(cfs, 1000, 50); - try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, Dispatcher.RequestTime.forImmediateExecution()); + try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, ReadCoordinator.DEFAULT, Dispatcher.RequestTime.forImmediateExecution()); ReplicaPlanIterator ranges = new ReplicaPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE)) { assertEquals(MAX_CONCURRENCY_FACTOR, partitions.concurrencyFactor()); @@ -98,7 +99,7 @@ public class RangeCommandsTest extends CQLTester // with 0 estimated results per range the concurrency factor should be 1 command = command(cfs, 1000, 0); - try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, Dispatcher.RequestTime.forImmediateExecution()); + try (RangeCommandIterator partitions = RangeCommands.rangeCommandIterator(command, ONE, ReadCoordinator.DEFAULT, Dispatcher.RequestTime.forImmediateExecution()); ReplicaPlanIterator ranges = new ReplicaPlanIterator(command.dataRange().keyRange(), command.indexQueryPlan(), keyspace, command.metadata().id, ONE)) { assertEquals(1, partitions.concurrencyFactor()); diff --git a/test/unit/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepairTest.java b/test/unit/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepairTest.java index a094ebfa6b..1d4b9b745a 100644 --- a/test/unit/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepairTest.java +++ b/test/unit/org/apache/cassandra/service/reads/repair/ReadOnlyReadRepairTest.java @@ -22,7 +22,6 @@ import java.util.Collections; import java.util.HashSet; import java.util.Set; -import org.apache.cassandra.locator.ReplicaPlan; import org.junit.Assert; import org.junit.BeforeClass; import org.junit.Test; @@ -32,9 +31,11 @@ import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators; import org.apache.cassandra.locator.Endpoints; import org.apache.cassandra.locator.InetAddressAndPort; import org.apache.cassandra.locator.Replica; +import org.apache.cassandra.locator.ReplicaPlan; import org.apache.cassandra.service.reads.ReadCallback; import org.apache.cassandra.transport.Dispatcher; import org.apache.cassandra.service.reads.ReadCoordinator; +import org.apache.cassandra.service.reads.repair.ReadRepair.ReadRepairSource; public class ReadOnlyReadRepairTest extends AbstractReadRepairTest { @@ -96,6 +97,6 @@ public class ReadOnlyReadRepairTest extends AbstractReadRepairTest ReplicaPlan.SharedForRangeRead readPlan = ReplicaPlan.shared(replicaPlan(replicas, replicas)); ReplicaPlan.ForWrite writePlan = repairPlan(replicas, replicas); InstrumentedReadRepair repair = createInstrumentedReadRepair(readPlan); - repair.repairPartition(null, Collections.emptyMap(), writePlan); + repair.repairPartition(null, Collections.emptyMap(), writePlan, ReadRepairSource.OTHER); } } diff --git a/test/unit/org/apache/cassandra/service/reads/repair/TestableReadRepair.java b/test/unit/org/apache/cassandra/service/reads/repair/TestableReadRepair.java index 650fa73eb2..3bca1cd062 100644 --- a/test/unit/org/apache/cassandra/service/reads/repair/TestableReadRepair.java +++ b/test/unit/org/apache/cassandra/service/reads/repair/TestableReadRepair.java @@ -112,7 +112,7 @@ public class TestableReadRepair, P extends ReplicaPlan.Fo } @Override - public void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan) + public void repairPartition(DecoratedKey partitionKey, Map mutations, ReplicaPlan.ForWrite writePlan, ReadRepairSource rrSource) { for (Map.Entry entry: mutations.entrySet()) sent.put(entry.getKey().endpoint(), entry.getValue());