From 0626be8667aefdcf50a051471f83da90bbae9bcf Mon Sep 17 00:00:00 2001 From: Sam Tunnicliffe Date: Fri, 24 Jul 2015 19:19:06 +0100 Subject: [PATCH] New 2i API and implementations for built in indexes Patch by Sam Tunnicliffe; reviewed by Sergio Bossa, Sylvain Lebresne and Aleksey Yeschenko for CASSANDRA-9459 Conflicts: src/java/org/apache/cassandra/db/ColumnFamilyStore.java src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java src/java/org/apache/cassandra/db/compaction/CompactionIterator.java src/java/org/apache/cassandra/db/index/AbstractSimplePerColumnSecondaryIndex.java src/java/org/apache/cassandra/db/index/PerColumnSecondaryIndex.java src/java/org/apache/cassandra/db/index/SecondaryIndexManager.java src/java/org/apache/cassandra/db/index/composites/CompositesIndex.java src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnClusteringKey.java src/java/org/apache/cassandra/db/partitions/AtomicBTreePartition.java src/java/org/apache/cassandra/db/partitions/PartitionUpdate.java src/java/org/apache/cassandra/index/internal/ColumnIndexSearcher.java src/java/org/apache/cassandra/index/internal/composites/CompositesSearcher.java src/java/org/apache/cassandra/index/internal/keys/KeysSearcher.java src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java test/unit/org/apache/cassandra/schema/DefsTest.java test/unit/org/apache/cassandra/schema/LegacySchemaMigratorTest.java --- NEWS.txt | 4 + .../apache/cassandra/config/CFMetaData.java | 8 +- .../org/apache/cassandra/config/Schema.java | 10 +- .../cassandra/cql3/UpdateParameters.java | 26 +- .../ForwardingPrimaryKeyRestrictions.java | 9 +- .../restrictions/MultiColumnRestriction.java | 45 +- .../PrimaryKeyRestrictionSet.java | 2 +- .../cql3/restrictions/Restriction.java | 2 +- .../cql3/restrictions/RestrictionSet.java | 2 +- .../cql3/restrictions/Restrictions.java | 2 +- .../restrictions/SingleColumnRestriction.java | 36 +- .../restrictions/StatementRestrictions.java | 7 +- .../cql3/restrictions/TermSlice.java | 15 +- .../cql3/restrictions/TokenRestriction.java | 6 +- .../cql3/statements/AlterTableStatement.java | 22 +- .../cql3/statements/CreateIndexStatement.java | 62 +- .../cql3/statements/DeleteStatement.java | 16 +- .../cql3/statements/IndexPropDefs.java | 11 +- .../cql3/statements/IndexTarget.java | 50 +- .../cql3/statements/SelectStatement.java | 24 +- .../cql3/statements/UpdateStatement.java | 16 +- .../cassandra/db/ColumnFamilyStore.java | 111 ++- .../org/apache/cassandra/db/Keyspace.java | 104 +- .../org/apache/cassandra/db/Memtable.java | 30 +- .../db/PartitionRangeReadCommand.java | 15 +- .../org/apache/cassandra/db/ReadCommand.java | 25 +- .../apache/cassandra/db/ReadOrderGroup.java | 12 +- .../apache/cassandra/db/SystemKeyspace.java | 45 +- .../db/compaction/CompactionIterator.java | 59 +- .../db/compaction/CompactionManager.java | 46 +- .../apache/cassandra/db/filter/RowFilter.java | 14 +- ...AbstractSimplePerColumnSecondaryIndex.java | 251 ----- .../db/index/PerColumnSecondaryIndex.java | 116 --- .../db/index/PerRowSecondaryIndex.java | 56 -- .../cassandra/db/index/SecondaryIndex.java | 429 --------- .../db/index/SecondaryIndexManager.java | 792 --------------- .../db/index/SecondaryIndexSearcher.java | 244 ----- .../db/index/composites/CompositesIndex.java | 168 ---- .../CompositesIndexOnClusteringKey.java | 137 --- .../cassandra/db/index/keys/KeysIndex.java | 90 -- .../db/partitions/AtomicBTreePartition.java | 77 +- .../db/partitions/PartitionUpdate.java | 15 +- .../org/apache/cassandra/db/rows/Cells.java | 50 +- .../org/apache/cassandra/db/rows/Rows.java | 27 +- .../org/apache/cassandra/index/Index.java | 388 ++++++++ .../apache/cassandra/index/IndexRegistry.java | 22 + .../{db => }/index/SecondaryIndexBuilder.java | 13 +- .../index/SecondaryIndexManager.java | 900 ++++++++++++++++++ .../index/internal/CassandraIndex.java | 755 +++++++++++++++ .../internal/CassandraIndexFunctions.java | 186 ++++ .../internal/CassandraIndexSearcher.java | 172 ++++ .../cassandra/index/internal/IndexEntry.java | 34 + .../composites/ClusteringColumnIndex.java | 100 ++ .../composites/CollectionEntryIndex.java} | 31 +- .../composites/CollectionKeyIndex.java} | 40 +- .../composites/CollectionKeyIndexBase.java} | 32 +- .../composites/CollectionValueIndex.java} | 56 +- .../composites/CompositesSearcher.java | 91 +- .../composites/PartitionKeyIndex.java} | 71 +- .../composites/RegularColumnIndex.java} | 41 +- .../index/internal/keys/KeysIndex.java | 62 ++ .../internal}/keys/KeysSearcher.java | 89 +- .../transactions/CleanupTransaction.java | 52 + .../transactions/CompactionTransaction.java | 44 + .../index/transactions/IndexTransaction.java | 57 ++ .../index/transactions/UpdateTransaction.java | 77 ++ .../io/sstable/format/SSTableReader.java | 17 +- .../cassandra/schema/IndexMetadata.java | 100 +- .../org/apache/cassandra/schema/Indexes.java | 27 +- .../schema/LegacySchemaMigrator.java | 6 +- .../cassandra/schema/SchemaKeyspace.java | 3 +- .../cassandra/service/DataResolver.java | 12 +- .../cassandra/service/StorageProxy.java | 45 +- .../streaming/StreamReceiveTask.java | 15 +- .../cassandra/thrift/CassandraServer.java | 12 +- .../cassandra/thrift/ThriftConversion.java | 41 +- .../cassandra/thrift/ThriftValidation.java | 17 +- .../org/apache/cassandra/SchemaLoader.java | 40 +- test/unit/org/apache/cassandra/Util.java | 13 +- .../entities/SecondaryIndexTest.java | 97 +- .../miscellaneous/CrcCheckChanceTest.java | 6 +- .../org/apache/cassandra/db/CleanupTest.java | 14 +- .../apache/cassandra/db/DirectoriesTest.java | 27 +- .../cassandra/db/RangeTombstoneTest.java | 137 +-- .../org/apache/cassandra/db/ScrubTest.java | 19 +- .../cassandra/db/SecondaryIndexTest.java | 88 +- .../db/index/PerRowSecondaryIndexTest.java | 304 ------ .../org/apache/cassandra/index/StubIndex.java | 186 ++++ .../index/internal/CassandraIndexTest.java | 718 ++++++++++++++ .../io/sstable/SSTableReaderTest.java | 63 +- .../org/apache/cassandra/schema/DefsTest.java | 19 +- .../schema/LegacySchemaMigratorTest.java | 4 +- 92 files changed, 4982 insertions(+), 3649 deletions(-) delete mode 100644 src/java/org/apache/cassandra/db/index/AbstractSimplePerColumnSecondaryIndex.java delete mode 100644 src/java/org/apache/cassandra/db/index/PerColumnSecondaryIndex.java delete mode 100644 src/java/org/apache/cassandra/db/index/PerRowSecondaryIndex.java delete mode 100644 src/java/org/apache/cassandra/db/index/SecondaryIndex.java delete mode 100644 src/java/org/apache/cassandra/db/index/SecondaryIndexManager.java delete mode 100644 src/java/org/apache/cassandra/db/index/SecondaryIndexSearcher.java delete mode 100644 src/java/org/apache/cassandra/db/index/composites/CompositesIndex.java delete mode 100644 src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnClusteringKey.java delete mode 100644 src/java/org/apache/cassandra/db/index/keys/KeysIndex.java create mode 100644 src/java/org/apache/cassandra/index/Index.java create mode 100644 src/java/org/apache/cassandra/index/IndexRegistry.java rename src/java/org/apache/cassandra/{db => }/index/SecondaryIndexBuilder.java (88%) create mode 100644 src/java/org/apache/cassandra/index/SecondaryIndexManager.java create mode 100644 src/java/org/apache/cassandra/index/internal/CassandraIndex.java create mode 100644 src/java/org/apache/cassandra/index/internal/CassandraIndexFunctions.java create mode 100644 src/java/org/apache/cassandra/index/internal/CassandraIndexSearcher.java create mode 100644 src/java/org/apache/cassandra/index/internal/IndexEntry.java create mode 100644 src/java/org/apache/cassandra/index/internal/composites/ClusteringColumnIndex.java rename src/java/org/apache/cassandra/{db/index/composites/CompositesIndexOnCollectionKeyAndValue.java => index/internal/composites/CollectionEntryIndex.java} (58%) rename src/java/org/apache/cassandra/{db/index/composites/CompositesIndexOnCollectionKey.java => index/internal/composites/CollectionKeyIndex.java} (54%) rename src/java/org/apache/cassandra/{db/index/composites/CompositesIndexIncludingCollectionKey.java => index/internal/composites/CollectionKeyIndexBase.java} (65%) rename src/java/org/apache/cassandra/{db/index/composites/CompositesIndexOnCollectionValue.java => index/internal/composites/CollectionValueIndex.java} (60%) rename src/java/org/apache/cassandra/{db/index => index/internal}/composites/CompositesSearcher.java (75%) rename src/java/org/apache/cassandra/{db/index/composites/CompositesIndexOnPartitionKey.java => index/internal/composites/PartitionKeyIndex.java} (55%) rename src/java/org/apache/cassandra/{db/index/composites/CompositesIndexOnRegular.java => index/internal/composites/RegularColumnIndex.java} (60%) create mode 100644 src/java/org/apache/cassandra/index/internal/keys/KeysIndex.java rename src/java/org/apache/cassandra/{db/index => index/internal}/keys/KeysSearcher.java (70%) create mode 100644 src/java/org/apache/cassandra/index/transactions/CleanupTransaction.java create mode 100644 src/java/org/apache/cassandra/index/transactions/CompactionTransaction.java create mode 100644 src/java/org/apache/cassandra/index/transactions/IndexTransaction.java create mode 100644 src/java/org/apache/cassandra/index/transactions/UpdateTransaction.java delete mode 100644 test/unit/org/apache/cassandra/db/index/PerRowSecondaryIndexTest.java create mode 100644 test/unit/org/apache/cassandra/index/StubIndex.java create mode 100644 test/unit/org/apache/cassandra/index/internal/CassandraIndexTest.java diff --git a/NEWS.txt b/NEWS.txt index c7f0d77104..dc515171ea 100644 --- a/NEWS.txt +++ b/NEWS.txt @@ -94,6 +94,10 @@ Upgrading - JMX methods set/getCompactionStrategyClass have been removed, use set/getCompactionParameters or set/getCompactionParametersJson instead. - SizeTieredCompactionStrategy parameter cold_reads_to_omit has been removed. + - The secondary index API has been comprehensively reworked. This will be a breaking + change for any custom index implementations, which should now look to implement + the new org.apache.cassandra.index.Index interface. + 2.2 diff --git a/src/java/org/apache/cassandra/config/CFMetaData.java b/src/java/org/apache/cassandra/config/CFMetaData.java index e794901a27..37f1f4dd75 100644 --- a/src/java/org/apache/cassandra/config/CFMetaData.java +++ b/src/java/org/apache/cassandra/config/CFMetaData.java @@ -44,7 +44,6 @@ import org.apache.cassandra.cql3.statements.CFStatement; import org.apache.cassandra.cql3.statements.CreateTableStatement; import org.apache.cassandra.db.*; import org.apache.cassandra.db.compaction.AbstractCompactionStrategy; -import org.apache.cassandra.db.index.SecondaryIndex; import org.apache.cassandra.db.marshal.*; import org.apache.cassandra.dht.IPartitioner; import org.apache.cassandra.exceptions.ConfigurationException; @@ -787,6 +786,7 @@ public final class CFMetaData throw new ConfigurationException(String.format("Column family comparators do not match or are not compatible (found %s; expected %s).", cfm.comparator.getClass().getSimpleName(), comparator.getClass().getSimpleName())); } + public static Class createCompactionStrategy(String className) throws ConfigurationException { className = className.contains(".") ? className : "org.apache.cassandra.db.compaction." + className; @@ -875,21 +875,19 @@ public final class CFMetaData Set indexNames = ksm == null ? new HashSet<>() : ksm.existingIndexNames(cfName); for (IndexMetadata index : indexes) { - index.validate(); - // check index names against this CF _and_ globally if (indexNames.contains(index.name)) throw new ConfigurationException("Duplicate index name " + index.name); indexNames.add(index.name); - // This method validates any custom options in the index metadata but does not intialize the index - SecondaryIndex.validate(this, index); + index.validate(); } return this; } + // The comparator to validate the definition name with thrift. public AbstractType thriftColumnNameType() { diff --git a/src/java/org/apache/cassandra/config/Schema.java b/src/java/org/apache/cassandra/config/Schema.java index e1e7380719..7cc8394d35 100644 --- a/src/java/org/apache/cassandra/config/Schema.java +++ b/src/java/org/apache/cassandra/config/Schema.java @@ -27,12 +27,10 @@ import com.google.common.collect.Sets; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import org.apache.cassandra.cql3.functions.Function; -import org.apache.cassandra.cql3.functions.FunctionName; -import org.apache.cassandra.cql3.functions.UDAggregate; -import org.apache.cassandra.cql3.functions.UDFunction; -import org.apache.cassandra.db.*; +import org.apache.cassandra.cql3.functions.*; +import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.Keyspace; +import org.apache.cassandra.db.SystemKeyspace; import org.apache.cassandra.db.commitlog.CommitLog; import org.apache.cassandra.db.compaction.CompactionManager; import org.apache.cassandra.db.marshal.AbstractType; @@ -539,7 +537,7 @@ public class Schema assert cfs != null; // make sure all the indexes are dropped, or else. - cfs.indexManager.setIndexRemoved(new HashSet<>(cfs.indexManager.getBuiltIndexes())); + cfs.indexManager.markAllIndexesRemoved(); // reinitialize the keyspace. CFMetaData cfm = oldKsm.tables.get(tableName).get(); diff --git a/src/java/org/apache/cassandra/cql3/UpdateParameters.java b/src/java/org/apache/cassandra/cql3/UpdateParameters.java index 1c1dd43044..1cdb64d5ce 100644 --- a/src/java/org/apache/cassandra/cql3/UpdateParameters.java +++ b/src/java/org/apache/cassandra/cql3/UpdateParameters.java @@ -23,12 +23,13 @@ import java.util.Map; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.filter.ColumnFilter; -import org.apache.cassandra.db.rows.*; import org.apache.cassandra.db.context.CounterContext; -import org.apache.cassandra.db.index.SecondaryIndexManager; -import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.filter.ColumnFilter; +import org.apache.cassandra.db.partitions.Partition; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.rows.*; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; import org.apache.cassandra.utils.FBUtilities; /** @@ -78,7 +79,8 @@ public class UpdateParameters this.prefetchedRows = prefetchedRows; - // Index column validation triggers a call to Keyspace.open() which we want to be able to avoid in some case (e.g. when using CQLSSTableWriter) + // Index column validation triggers a call to Keyspace.open() which we want + // to be able to avoid in some case (e.g. when using CQLSSTableWriter) if (validateIndexedColumns) { SecondaryIndexManager manager = Keyspace.openAndGetStore(metadata).indexManager; @@ -95,17 +97,16 @@ public class UpdateParameters throw new InvalidRequestException(String.format("Out of bound timestamp, must be in [%d, %d]", Long.MIN_VALUE + 1, Long.MAX_VALUE)); } - public void newPartition(DecoratedKey partitionKey) throws InvalidRequestException + public void validateIndexedColumns(PartitionUpdate update) { - if (indexManager != null) - indexManager.validate(partitionKey); + if (indexManager == null) + return; + + indexManager.validate(update); } public void newRow(Clustering clustering) throws InvalidRequestException { - if (indexManager != null) - indexManager.validate(clustering); - if (metadata.isDense() && !metadata.isCompound()) { // If it's a COMPACT STORAGE table with a single clustering column, the clustering value is @@ -165,9 +166,6 @@ public class UpdateParameters public void addCell(ColumnDefinition column, CellPath path, ByteBuffer value) throws InvalidRequestException { - if (indexManager != null) - indexManager.validate(column, value, path); - Cell cell = ttl == LivenessInfo.NO_TTL ? BufferCell.live(metadata, column, timestamp, value, path) : BufferCell.expiring(column, timestamp, ttl, nowInSec, value, path); diff --git a/src/java/org/apache/cassandra/cql3/restrictions/ForwardingPrimaryKeyRestrictions.java b/src/java/org/apache/cassandra/cql3/restrictions/ForwardingPrimaryKeyRestrictions.java index 82ac78ee26..f82cb11eda 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/ForwardingPrimaryKeyRestrictions.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/ForwardingPrimaryKeyRestrictions.java @@ -18,17 +18,20 @@ package org.apache.cassandra.cql3.restrictions; import java.nio.ByteBuffer; -import java.util.*; +import java.util.Collection; +import java.util.List; +import java.util.NavigableSet; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.QueryOptions; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; -import org.apache.cassandra.db.*; +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.MultiCBuilder; import org.apache.cassandra.db.Slice; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; /** * A PrimaryKeyRestrictions which forwards all its method calls to another diff --git a/src/java/org/apache/cassandra/cql3/restrictions/MultiColumnRestriction.java b/src/java/org/apache/cassandra/cql3/restrictions/MultiColumnRestriction.java index b0f592cf97..b60930e430 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/MultiColumnRestriction.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/MultiColumnRestriction.java @@ -20,18 +20,17 @@ package org.apache.cassandra.cql3.restrictions; import java.nio.ByteBuffer; import java.util.*; -import org.apache.cassandra.cql3.Term.Terminal; - -import org.apache.cassandra.cql3.Term; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.*; +import org.apache.cassandra.cql3.Term.Terminal; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; -import org.apache.cassandra.db.*; +import org.apache.cassandra.db.MultiCBuilder; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.index.SecondaryIndexManager; + import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse; import static org.apache.cassandra.cql3.statements.RequestValidations.checkNotNull; import static org.apache.cassandra.cql3.statements.RequestValidations.checkTrue; @@ -113,23 +112,20 @@ public abstract class MultiColumnRestriction extends AbstractRestriction @Override public final boolean hasSupportingIndex(SecondaryIndexManager indexManager) { - for (ColumnDefinition columnDef : columnDefs) - { - SecondaryIndex index = indexManager.getIndexForColumn(columnDef); - if (index != null && isSupportedBy(index)) - return true; - } + for (Index index : indexManager.listIndexes()) + if (isSupportedBy(index)) + return true; return false; } /** * Check if this type of restriction is supported for by the specified index. - * @param index the Secondary index + * @param index the secondary index * * @return true this type of restriction is supported by the specified index, * false otherwise. */ - protected abstract boolean isSupportedBy(SecondaryIndex index); + protected abstract boolean isSupportedBy(Index index); public static class EQRestriction extends MultiColumnRestriction { @@ -161,9 +157,12 @@ public abstract class MultiColumnRestriction extends AbstractRestriction } @Override - protected boolean isSupportedBy(SecondaryIndex index) + protected boolean isSupportedBy(Index index) { - return index.supportsOperator(Operator.EQ); + for(ColumnDefinition column : columnDefs) + if (index.supportsExpression(column, Operator.EQ)) + return true; + return false; } @Override @@ -228,9 +227,12 @@ public abstract class MultiColumnRestriction extends AbstractRestriction } @Override - protected boolean isSupportedBy(SecondaryIndex index) + protected boolean isSupportedBy(Index index) { - return index.supportsOperator(Operator.IN); + for (ColumnDefinition column: columnDefs) + if (index.supportsExpression(column, Operator.IN)) + return true; + return false; } @Override @@ -368,9 +370,12 @@ public abstract class MultiColumnRestriction extends AbstractRestriction } @Override - protected boolean isSupportedBy(SecondaryIndex index) + protected boolean isSupportedBy(Index index) { - return slice.isSupportedBy(index); + for(ColumnDefinition def : columnDefs) + if (slice.isSupportedBy(def, index)) + return true; + return false; } @Override diff --git a/src/java/org/apache/cassandra/cql3/restrictions/PrimaryKeyRestrictionSet.java b/src/java/org/apache/cassandra/cql3/restrictions/PrimaryKeyRestrictionSet.java index 005a1698c6..107cbd122f 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/PrimaryKeyRestrictionSet.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/PrimaryKeyRestrictionSet.java @@ -27,8 +27,8 @@ import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; import org.apache.cassandra.db.*; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; import org.apache.cassandra.utils.btree.BTreeSet; import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse; diff --git a/src/java/org/apache/cassandra/cql3/restrictions/Restriction.java b/src/java/org/apache/cassandra/cql3/restrictions/Restriction.java index 370a0f20b1..a21087e41b 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/Restriction.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/Restriction.java @@ -25,8 +25,8 @@ import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; import org.apache.cassandra.db.MultiCBuilder; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; /** * A restriction/clause on a column. diff --git a/src/java/org/apache/cassandra/cql3/restrictions/RestrictionSet.java b/src/java/org/apache/cassandra/cql3/restrictions/RestrictionSet.java index 10fae1386b..700840db8d 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/RestrictionSet.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/RestrictionSet.java @@ -26,8 +26,8 @@ import org.apache.cassandra.cql3.QueryOptions; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.restrictions.SingleColumnRestriction.ContainsRestriction; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; /** * Sets of column restrictions. diff --git a/src/java/org/apache/cassandra/cql3/restrictions/Restrictions.java b/src/java/org/apache/cassandra/cql3/restrictions/Restrictions.java index 88d0bb6c6a..705d66d26a 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/Restrictions.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/Restrictions.java @@ -23,8 +23,8 @@ import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.QueryOptions; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; /** * Sets of restrictions diff --git a/src/java/org/apache/cassandra/cql3/restrictions/SingleColumnRestriction.java b/src/java/org/apache/cassandra/cql3/restrictions/SingleColumnRestriction.java index a6ed4b4c14..25146e5cd6 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/SingleColumnRestriction.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/SingleColumnRestriction.java @@ -29,9 +29,10 @@ import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; import org.apache.cassandra.db.MultiCBuilder; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.index.SecondaryIndexManager; + import static org.apache.cassandra.cql3.statements.RequestValidations.checkBindValueSet; import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse; import static org.apache.cassandra.cql3.statements.RequestValidations.checkNotNull; @@ -71,8 +72,11 @@ public abstract class SingleColumnRestriction extends AbstractRestriction @Override public boolean hasSupportingIndex(SecondaryIndexManager indexManager) { - SecondaryIndex index = indexManager.getIndexForColumn(columnDef); - return index != null && isSupportedBy(index); + for (Index index : indexManager.listIndexes()) + if (isSupportedBy(index)) + return true; + + return false; } @Override @@ -110,11 +114,11 @@ public abstract class SingleColumnRestriction extends AbstractRestriction /** * Check if this type of restriction is supported by the specified index. * - * @param index the Secondary index + * @param index the secondary index * @return true this type of restriction is supported by the specified index, * false otherwise. */ - protected abstract boolean isSupportedBy(SecondaryIndex index); + protected abstract boolean isSupportedBy(Index index); public static final class EQRestriction extends SingleColumnRestriction { @@ -174,9 +178,9 @@ public abstract class SingleColumnRestriction extends AbstractRestriction } @Override - protected boolean isSupportedBy(SecondaryIndex index) + protected boolean isSupportedBy(Index index) { - return index.supportsOperator(Operator.EQ); + return index.supportsExpression(columnDef, Operator.EQ); } } @@ -220,9 +224,9 @@ public abstract class SingleColumnRestriction extends AbstractRestriction } @Override - protected final boolean isSupportedBy(SecondaryIndex index) + protected final boolean isSupportedBy(Index index) { - return index.supportsOperator(Operator.IN); + return index.supportsExpression(columnDef, Operator.IN); } protected abstract List getValues(QueryOptions options) throws InvalidRequestException; @@ -387,9 +391,9 @@ public abstract class SingleColumnRestriction extends AbstractRestriction } @Override - protected boolean isSupportedBy(SecondaryIndex index) + protected boolean isSupportedBy(Index index) { - return slice.isSupportedBy(index); + return slice.isSupportedBy(columnDef, index); } @Override @@ -484,18 +488,18 @@ public abstract class SingleColumnRestriction extends AbstractRestriction } @Override - protected boolean isSupportedBy(SecondaryIndex index) + protected boolean isSupportedBy(Index index) { boolean supported = false; if (numberOfValues() > 0) - supported |= index.supportsOperator(Operator.CONTAINS); + supported |= index.supportsExpression(columnDef, Operator.CONTAINS); if (numberOfKeys() > 0) - supported |= index.supportsOperator(Operator.CONTAINS_KEY); + supported |= index.supportsExpression(columnDef, Operator.CONTAINS_KEY); if (numberOfEntries() > 0) - supported |= index.supportsOperator(Operator.EQ); + supported |= index.supportsExpression(columnDef, Operator.EQ); return supported; } diff --git a/src/java/org/apache/cassandra/cql3/restrictions/StatementRestrictions.java b/src/java/org/apache/cassandra/cql3/restrictions/StatementRestrictions.java index 36f11ad696..b0c81b8e9d 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/StatementRestrictions.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/StatementRestrictions.java @@ -30,10 +30,9 @@ import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; import org.apache.cassandra.db.*; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.dht.*; import org.apache.cassandra.exceptions.InvalidRequestException; -import org.apache.cassandra.service.StorageService; +import org.apache.cassandra.index.SecondaryIndexManager; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.btree.BTreeSet; @@ -47,6 +46,8 @@ import static org.apache.cassandra.cql3.statements.RequestValidations.invalidReq */ public final class StatementRestrictions { + public static final String NO_INDEX_FOUND_MESSAGE = + "No supported secondary index found for the non primary key columns restrictions"; /** * The Column Family meta data */ @@ -159,7 +160,7 @@ public final class StatementRestrictions if (hasQueriableIndex) usesSecondaryIndexing = true; else if (!useFiltering) - throw new InvalidRequestException("No supported secondary index found for the non primary key columns restrictions"); + throw new InvalidRequestException(NO_INDEX_FOUND_MESSAGE); indexRestrictions.add(nonPrimaryKeyRestrictions); } diff --git a/src/java/org/apache/cassandra/cql3/restrictions/TermSlice.java b/src/java/org/apache/cassandra/cql3/restrictions/TermSlice.java index d082cc3ddf..6423be37f9 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/TermSlice.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/TermSlice.java @@ -21,11 +21,12 @@ import java.util.Collections; import com.google.common.collect.Iterables; +import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.cql3.Term; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; -import org.apache.cassandra.db.index.SecondaryIndex; +import org.apache.cassandra.index.Index; final class TermSlice { @@ -152,20 +153,20 @@ final class TermSlice /** * Check if this TermSlice is supported by the specified index. * - * @param index the Secondary index + * @param index the secondary index * @return true this type of TermSlice is supported by the specified index, * false otherwise. */ - public boolean isSupportedBy(SecondaryIndex index) + public boolean isSupportedBy(ColumnDefinition column, Index index) { boolean supported = false; if (hasBound(Bound.START)) - supported |= isInclusive(Bound.START) ? index.supportsOperator(Operator.GTE) - : index.supportsOperator(Operator.GT); + supported |= isInclusive(Bound.START) ? index.supportsExpression(column, Operator.GTE) + : index.supportsExpression(column, Operator.GT); if (hasBound(Bound.END)) - supported |= isInclusive(Bound.END) ? index.supportsOperator(Operator.LTE) - : index.supportsOperator(Operator.LT); + supported |= isInclusive(Bound.END) ? index.supportsExpression(column, Operator.LTE) + : index.supportsExpression(column, Operator.LT); return supported; } diff --git a/src/java/org/apache/cassandra/cql3/restrictions/TokenRestriction.java b/src/java/org/apache/cassandra/cql3/restrictions/TokenRestriction.java index 56da6da317..63fe206887 100644 --- a/src/java/org/apache/cassandra/cql3/restrictions/TokenRestriction.java +++ b/src/java/org/apache/cassandra/cql3/restrictions/TokenRestriction.java @@ -28,10 +28,12 @@ import org.apache.cassandra.cql3.QueryOptions; import org.apache.cassandra.cql3.Term; import org.apache.cassandra.cql3.functions.Function; import org.apache.cassandra.cql3.statements.Bound; -import org.apache.cassandra.db.*; +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.MultiCBuilder; +import org.apache.cassandra.db.Slice; import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.SecondaryIndexManager; import static org.apache.cassandra.cql3.statements.RequestValidations.invalidRequest; diff --git a/src/java/org/apache/cassandra/cql3/statements/AlterTableStatement.java b/src/java/org/apache/cassandra/cql3/statements/AlterTableStatement.java index a0136fab1a..da42c96b90 100644 --- a/src/java/org/apache/cassandra/cql3/statements/AlterTableStatement.java +++ b/src/java/org/apache/cassandra/cql3/statements/AlterTableStatement.java @@ -17,9 +17,7 @@ */ package org.apache.cassandra.cql3.statements; -import java.util.ArrayList; -import java.util.List; -import java.util.Map; +import java.util.*; import org.apache.cassandra.auth.Permission; import org.apache.cassandra.config.*; @@ -30,6 +28,8 @@ import org.apache.cassandra.db.marshal.AbstractType; import org.apache.cassandra.db.marshal.CollectionType; import org.apache.cassandra.db.marshal.CounterColumnType; import org.apache.cassandra.exceptions.*; +import org.apache.cassandra.schema.IndexMetadata; +import org.apache.cassandra.schema.Indexes; import org.apache.cassandra.schema.TableParams; import org.apache.cassandra.service.ClientState; import org.apache.cassandra.service.MigrationManager; @@ -264,11 +264,17 @@ public class AlterTableStatement extends SchemaAlteringStatement break; } - // If a column is dropped which is the target of a secondary index - // we need to also drop the index. - cfm.getIndexes().get(def).ifPresent(indexMetadata -> { - cfm.indexes(cfm.getIndexes().without(indexMetadata.name)); - }); + // If the dropped column is the only target column of a secondary + // index (and it's only possible to create an index with TargetType.COLUMN + // and a single target right now) we need to also drop the index. + Indexes allIndexes = cfm.getIndexes(); + Collection indexes = allIndexes.get(def); + for (IndexMetadata index : indexes) + { + assert index.columns.size() == 1 : String.format("Can't drop column %s as it's a target of multi-column index %s", def.name, index.name); + allIndexes = allIndexes.without(index.name); + } + cfm.indexes(allIndexes); // If a column is dropped which is the target of a materialized view, // then we need to drop the view. diff --git a/src/java/org/apache/cassandra/cql3/statements/CreateIndexStatement.java b/src/java/org/apache/cassandra/cql3/statements/CreateIndexStatement.java index 64bb1dad56..19d89b01ef 100644 --- a/src/java/org/apache/cassandra/cql3/statements/CreateIndexStatement.java +++ b/src/java/org/apache/cassandra/cql3/statements/CreateIndexStatement.java @@ -20,8 +20,10 @@ package org.apache.cassandra.cql3.statements; import java.util.Collections; import java.util.Map; +import com.google.common.base.Optional; import com.google.common.base.Strings; import com.google.common.collect.ImmutableMap; +import com.google.common.collect.Iterables; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -96,32 +98,15 @@ public class CreateIndexStatement extends SchemaAlteringStatement validateTargetColumnIsMapIfIndexInvolvesKeys(isMap, target); } - Indexes existingIndexes = cfm.getIndexes(); - for (IndexMetadata index : existingIndexes) - { - if (index.indexedColumn(cfm).equals(cd)) - { - IndexTarget.TargetType prevType = IndexTarget.TargetType.fromIndexMetadata(index, cfm); - if (isMap && target.type != prevType) - { - String msg = "Cannot create index on %s(%s): an index on %s(%s) already exists and indexing " + - "a map on more than one dimension at the same time is not currently supported"; - throw new InvalidRequestException(String.format(msg, - target.type, target.column, - prevType, target.column)); - } - - if (ifNotExists) - return; - else - throw new InvalidRequestException("Index already existss"); - } - } - if (!Strings.isNullOrEmpty(indexName)) { if (Schema.instance.getKSMetaData(keyspace()).existingIndexNames(null).contains(indexName)) - throw new InvalidRequestException("Index already exists"); + { + if (ifNotExists) + return; + else + throw new InvalidRequestException(String.format("Index %s already exists", indexName)); + } } properties.validate(); @@ -150,26 +135,26 @@ public class CreateIndexStatement extends SchemaAlteringStatement private void validateForFrozenCollection(IndexTarget target) throws InvalidRequestException { - if (target.type != IndexTarget.TargetType.FULL) + if (target.type != IndexTarget.Type.FULL) throw new InvalidRequestException(String.format("Cannot create index on %s of frozen column %s", target.type, target.column)); } private void validateNotFullIndex(IndexTarget target) throws InvalidRequestException { - if (target.type == IndexTarget.TargetType.FULL) + if (target.type == IndexTarget.Type.FULL) throw new InvalidRequestException("full() indexes can only be created on frozen collections"); } private void validateIsValuesIndexIfTargetColumnNotCollection(ColumnDefinition cd, IndexTarget target) throws InvalidRequestException { - if (!cd.type.isCollection() && target.type != IndexTarget.TargetType.VALUES) + if (!cd.type.isCollection() && target.type != IndexTarget.Type.VALUES) throw new InvalidRequestException(String.format("Cannot create index on %s of column %s; only non-frozen collections support %s indexes", target.type, target.column, target.type)); } private void validateTargetColumnIsMapIfIndexInvolvesKeys(boolean isMap, IndexTarget target) throws InvalidRequestException { - if (target.type == IndexTarget.TargetType.KEYS || target.type == IndexTarget.TargetType.KEYS_AND_VALUES) + if (target.type == IndexTarget.Type.KEYS || target.type == IndexTarget.Type.KEYS_AND_VALUES) { if (!isMap) throw new InvalidRequestException(String.format("Cannot create index on %s of column %s with non-map type", @@ -187,12 +172,13 @@ public class CreateIndexStatement extends SchemaAlteringStatement if (Strings.isNullOrEmpty(acceptedName)) acceptedName = Indexes.getAvailableIndexName(keyspace(), columnFamily(), cd.name); - for (IndexMetadata existing : cfm.getIndexes()) - if (existing.indexedColumn(cfm).equals(cd) || existing.name.equals(acceptedName)) - if (ifNotExists) - return false; - else - throw new InvalidRequestException("Index already exists"); + if (Schema.instance.getKSMetaData(keyspace()).existingIndexNames(null).contains(acceptedName)) + { + if (ifNotExists) + return false; + else + throw new InvalidRequestException(String.format("Index %s already exists", acceptedName)); + } IndexMetadata.IndexType indexType; Map indexOptions; @@ -218,12 +204,18 @@ public class CreateIndexStatement extends SchemaAlteringStatement indexOptions = Collections.emptyMap(); } + IndexMetadata index = IndexMetadata.singleColumnIndex(cd, acceptedName, indexType, indexOptions); + // check to disallow creation of an index which duplicates an existing one in all but name + Optional existingIndex = Iterables.tryFind(cfm.getIndexes(), existing -> existing.equalsWithoutName(index)); + if (existingIndex.isPresent()) + throw new InvalidRequestException(String.format("Index %s is a duplicate of existing index %s", + index.name, + existingIndex.get().name)); logger.debug("Updating index definition for {}", indexName); - - IndexMetadata index = IndexMetadata.legacyIndex(cd, acceptedName, indexType, indexOptions); cfm.indexes(cfm.getIndexes().with(index)); + MigrationManager.announceColumnFamilyUpdate(cfm, false, isLocalOnly); return true; } diff --git a/src/java/org/apache/cassandra/cql3/statements/DeleteStatement.java b/src/java/org/apache/cassandra/cql3/statements/DeleteStatement.java index 5fdf023afb..a33696edf1 100644 --- a/src/java/org/apache/cassandra/cql3/statements/DeleteStatement.java +++ b/src/java/org/apache/cassandra/cql3/statements/DeleteStatement.java @@ -17,17 +17,19 @@ */ package org.apache.cassandra.cql3.statements; -import java.util.*; +import java.util.Iterator; +import java.util.List; import com.google.common.collect.Iterators; -import org.apache.cassandra.cql3.*; -import org.apache.cassandra.cql3.restrictions.Restriction; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.exceptions.*; +import org.apache.cassandra.cql3.*; +import org.apache.cassandra.cql3.restrictions.Restriction; +import org.apache.cassandra.db.CBuilder; +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.exceptions.InvalidRequestException; import org.apache.cassandra.utils.Pair; /** @@ -93,6 +95,8 @@ public class DeleteStatement extends ModificationStatement update.add(params.buildRow()); } } + + params.validateIndexedColumns(update); } protected void validateWhereClauseForConditions() throws InvalidRequestException diff --git a/src/java/org/apache/cassandra/cql3/statements/IndexPropDefs.java b/src/java/org/apache/cassandra/cql3/statements/IndexPropDefs.java index 6790611296..e067217f9b 100644 --- a/src/java/org/apache/cassandra/cql3/statements/IndexPropDefs.java +++ b/src/java/org/apache/cassandra/cql3/statements/IndexPropDefs.java @@ -19,8 +19,9 @@ package org.apache.cassandra.cql3.statements; import java.util.*; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.exceptions.*; +import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.exceptions.RequestValidationException; +import org.apache.cassandra.exceptions.SyntaxException; public class IndexPropDefs extends PropertyDefinitions { @@ -50,9 +51,9 @@ public class IndexPropDefs extends PropertyDefinitions if (!isCustom && !properties.isEmpty()) throw new InvalidRequestException("Cannot specify options for a non-CUSTOM index"); - if (getRawOptions().containsKey(SecondaryIndex.CUSTOM_INDEX_OPTION_NAME)) + if (getRawOptions().containsKey(IndexTarget.CUSTOM_INDEX_OPTION_NAME)) throw new InvalidRequestException(String.format("Cannot specify %s as a CUSTOM option", - SecondaryIndex.CUSTOM_INDEX_OPTION_NAME)); + IndexTarget.CUSTOM_INDEX_OPTION_NAME)); } public Map getRawOptions() throws SyntaxException @@ -64,7 +65,7 @@ public class IndexPropDefs extends PropertyDefinitions public Map getOptions() throws SyntaxException { Map options = new HashMap<>(getRawOptions()); - options.put(SecondaryIndex.CUSTOM_INDEX_OPTION_NAME, customClass); + options.put(IndexTarget.CUSTOM_INDEX_OPTION_NAME, customClass); return options; } } diff --git a/src/java/org/apache/cassandra/cql3/statements/IndexTarget.java b/src/java/org/apache/cassandra/cql3/statements/IndexTarget.java index 4abb0778cf..f948d8b130 100644 --- a/src/java/org/apache/cassandra/cql3/statements/IndexTarget.java +++ b/src/java/org/apache/cassandra/cql3/statements/IndexTarget.java @@ -22,15 +22,31 @@ import java.util.Map; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.ColumnIdentifier; -import org.apache.cassandra.db.index.SecondaryIndex; import org.apache.cassandra.schema.IndexMetadata; public class IndexTarget { - public final ColumnIdentifier column; - public final TargetType type; + public static final String CUSTOM_INDEX_OPTION_NAME = "class_name"; - private IndexTarget(ColumnIdentifier column, TargetType type) + /** + * The name of the option used to specify that the index is on the collection keys. + */ + public static final String INDEX_KEYS_OPTION_NAME = "index_keys"; + + /** + * The name of the option used to specify that the index is on the collection values. + */ + public static final String INDEX_VALUES_OPTION_NAME = "index_values"; + + /** + * The name of the option used to specify that the index is on the collection (map) entries. + */ + public static final String INDEX_ENTRIES_OPTION_NAME = "index_keys_and_values"; + + public final ColumnIdentifier column; + public final Type type; + + private IndexTarget(ColumnIdentifier column, Type type) { this.column = column; this.type = type; @@ -39,9 +55,9 @@ public class IndexTarget public static class Raw { private final ColumnIdentifier.Raw column; - private final TargetType type; + private final Type type; - private Raw(ColumnIdentifier.Raw column, TargetType type) + private Raw(ColumnIdentifier.Raw column, Type type) { this.column = column; this.type = type; @@ -49,22 +65,22 @@ public class IndexTarget public static Raw valuesOf(ColumnIdentifier.Raw c) { - return new Raw(c, TargetType.VALUES); + return new Raw(c, Type.VALUES); } public static Raw keysOf(ColumnIdentifier.Raw c) { - return new Raw(c, TargetType.KEYS); + return new Raw(c, Type.KEYS); } public static Raw keysAndValuesOf(ColumnIdentifier.Raw c) { - return new Raw(c, TargetType.KEYS_AND_VALUES); + return new Raw(c, Type.KEYS_AND_VALUES); } public static Raw fullCollection(ColumnIdentifier.Raw c) { - return new Raw(c, TargetType.FULL); + return new Raw(c, Type.FULL); } public IndexTarget prepare(CFMetaData cfm) @@ -73,7 +89,7 @@ public class IndexTarget } } - public static enum TargetType + public static enum Type { VALUES, KEYS, KEYS_AND_VALUES, FULL; @@ -92,21 +108,21 @@ public class IndexTarget { switch (this) { - case KEYS: return SecondaryIndex.INDEX_KEYS_OPTION_NAME; - case KEYS_AND_VALUES: return SecondaryIndex.INDEX_ENTRIES_OPTION_NAME; - case VALUES: return SecondaryIndex.INDEX_VALUES_OPTION_NAME; + case KEYS: return INDEX_KEYS_OPTION_NAME; + case KEYS_AND_VALUES: return INDEX_ENTRIES_OPTION_NAME; + case VALUES: return INDEX_VALUES_OPTION_NAME; default: throw new AssertionError(); } } - public static TargetType fromIndexMetadata(IndexMetadata index, CFMetaData cfm) + public static Type fromIndexMetadata(IndexMetadata index, CFMetaData cfm) { Map options = index.options; - if (options.containsKey(SecondaryIndex.INDEX_KEYS_OPTION_NAME)) + if (options.containsKey(INDEX_KEYS_OPTION_NAME)) { return KEYS; } - else if (options.containsKey(SecondaryIndex.INDEX_ENTRIES_OPTION_NAME)) + else if (options.containsKey(INDEX_ENTRIES_OPTION_NAME)) { return KEYS_AND_VALUES; } diff --git a/src/java/org/apache/cassandra/cql3/statements/SelectStatement.java b/src/java/org/apache/cassandra/cql3/statements/SelectStatement.java index 01ce8e6aa5..5d5dfea302 100644 --- a/src/java/org/apache/cassandra/cql3/statements/SelectStatement.java +++ b/src/java/org/apache/cassandra/cql3/statements/SelectStatement.java @@ -35,20 +35,24 @@ import org.apache.cassandra.cql3.restrictions.StatementRestrictions; import org.apache.cassandra.cql3.selection.RawSelector; import org.apache.cassandra.cql3.selection.Selection; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; import org.apache.cassandra.db.filter.*; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.db.marshal.CollectionType; import org.apache.cassandra.db.marshal.CompositeType; import org.apache.cassandra.db.marshal.Int32Type; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.rows.ComplexColumnData; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.RowIterator; import org.apache.cassandra.db.view.MaterializedView; import org.apache.cassandra.dht.AbstractBounds; import org.apache.cassandra.exceptions.*; +import org.apache.cassandra.index.SecondaryIndexManager; import org.apache.cassandra.serializers.MarshalException; -import org.apache.cassandra.service.*; -import org.apache.cassandra.service.pager.QueryPager; +import org.apache.cassandra.service.ClientState; +import org.apache.cassandra.service.ClientWarn; +import org.apache.cassandra.service.QueryState; import org.apache.cassandra.service.pager.PagingState; +import org.apache.cassandra.service.pager.QueryPager; import org.apache.cassandra.thrift.ThriftValidation; import org.apache.cassandra.transport.messages.ResultMessage; import org.apache.cassandra.utils.ByteBufferUtil; @@ -74,6 +78,10 @@ public class SelectStatement implements CQLStatement private static final Logger logger = LoggerFactory.getLogger(SelectStatement.class); private static final int DEFAULT_COUNT_PAGE_SIZE = 10000; + public static final String REQUIRES_ALLOW_FILTERING_MESSAGE = + "Cannot execute this query as it might involve data filtering and " + + "thus may have unpredictable performance. If you want to execute " + + "this query despite the performance unpredictability, use ALLOW FILTERING"; private final int boundTerms; public final CFMetaData cfm; @@ -581,7 +589,6 @@ public class SelectStatement implements CQLStatement ColumnFamilyStore cfs = Keyspace.open(keyspace()).getColumnFamilyStore(columnFamily()); SecondaryIndexManager secondaryIndexManager = cfs.indexManager; RowFilter filter = restrictions.getRowFilter(secondaryIndexManager, options); - secondaryIndexManager.validateFilter(filter); return filter; } @@ -937,10 +944,7 @@ public class SelectStatement implements CQLStatement // We will potentially filter data if either: // - Have more than one IndexExpression // - Have no index expression and the row filter is not the identity - checkFalse(restrictions.needFiltering(), - "Cannot execute this query as it might involve data filtering and " + - "thus may have unpredictable performance. If you want to execute " + - "this query despite the performance unpredictability, use ALLOW FILTERING"); + checkFalse(restrictions.needFiltering(), REQUIRES_ALLOW_FILTERING_MESSAGE); } } diff --git a/src/java/org/apache/cassandra/cql3/statements/UpdateStatement.java b/src/java/org/apache/cassandra/cql3/statements/UpdateStatement.java index c6bfc98be4..e19deaaf69 100644 --- a/src/java/org/apache/cassandra/cql3/statements/UpdateStatement.java +++ b/src/java/org/apache/cassandra/cql3/statements/UpdateStatement.java @@ -17,14 +17,18 @@ */ package org.apache.cassandra.cql3.statements; -import java.util.*; +import java.util.Collection; +import java.util.Collections; +import java.util.List; -import org.apache.cassandra.cql3.*; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; +import org.apache.cassandra.cql3.*; +import org.apache.cassandra.db.CBuilder; +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.CompactTables; import org.apache.cassandra.db.partitions.PartitionUpdate; -import org.apache.cassandra.exceptions.*; +import org.apache.cassandra.exceptions.InvalidRequestException; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.Pair; @@ -49,8 +53,6 @@ public class UpdateStatement extends ModificationStatement public void addUpdateForKey(PartitionUpdate update, CBuilder cbuilder, UpdateParameters params) throws InvalidRequestException { - params.newPartition(update.partitionKey()); - if (updatesRegularRows()) { params.newRow(cbuilder.build()); @@ -90,6 +92,8 @@ public class UpdateStatement extends ModificationStatement op.execute(update.partitionKey(), params); update.add(params.buildRow()); } + + params.validateIndexedColumns(update); } public static class ParsedInsert extends ModificationStatement.Parsed diff --git a/src/java/org/apache/cassandra/db/ColumnFamilyStore.java b/src/java/org/apache/cassandra/db/ColumnFamilyStore.java index b199c77f0a..b177c23674 100644 --- a/src/java/org/apache/cassandra/db/ColumnFamilyStore.java +++ b/src/java/org/apache/cassandra/db/ColumnFamilyStore.java @@ -17,7 +17,9 @@ */ package org.apache.cassandra.db; -import java.io.*; +import java.io.File; +import java.io.IOException; +import java.io.PrintStream; import java.lang.management.ManagementFactory; import java.nio.ByteBuffer; import java.nio.file.Files; @@ -44,33 +46,38 @@ import org.apache.cassandra.config.*; import org.apache.cassandra.db.commitlog.CommitLog; import org.apache.cassandra.db.commitlog.ReplayPosition; import org.apache.cassandra.db.compaction.*; -import org.apache.cassandra.db.filter.*; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexManager; -import org.apache.cassandra.db.view.MaterializedViewManager; +import org.apache.cassandra.db.filter.ClusteringIndexFilter; +import org.apache.cassandra.db.filter.DataLimits; import org.apache.cassandra.db.lifecycle.*; -import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.db.partitions.CachedPartition; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.view.MaterializedViewManager; import org.apache.cassandra.dht.*; import org.apache.cassandra.dht.Range; import org.apache.cassandra.exceptions.ConfigurationException; +import org.apache.cassandra.index.SecondaryIndexManager; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.transactions.UpdateTransaction; import org.apache.cassandra.io.FSWriteError; -import org.apache.cassandra.io.sstable.*; +import org.apache.cassandra.io.sstable.Component; import org.apache.cassandra.io.sstable.Descriptor; +import org.apache.cassandra.io.sstable.SSTableMultiWriter; import org.apache.cassandra.io.sstable.format.*; import org.apache.cassandra.io.sstable.metadata.MetadataCollector; import org.apache.cassandra.io.util.FileUtils; -import org.apache.cassandra.metrics.TableMetrics.Sampler; import org.apache.cassandra.metrics.TableMetrics; +import org.apache.cassandra.metrics.TableMetrics.Sampler; import org.apache.cassandra.schema.*; import org.apache.cassandra.service.CacheService; import org.apache.cassandra.service.StorageService; import org.apache.cassandra.utils.*; import org.apache.cassandra.utils.TopKSampler.SamplerResult; -import org.apache.cassandra.utils.concurrent.*; +import org.apache.cassandra.utils.concurrent.OpOrder; +import org.apache.cassandra.utils.concurrent.Refs; import org.apache.cassandra.utils.memory.MemtableAllocator; -import org.json.simple.*; - +import org.json.simple.JSONArray; +import org.json.simple.JSONObject; import static org.apache.cassandra.utils.Throwables.maybeFail; @@ -403,7 +410,7 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean // create the private ColumnFamilyStores for the secondary column indexes for (IndexMetadata info : metadata.getIndexes()) - indexManager.addIndexedColumn(info); + indexManager.addIndex(info); if (registerBookkeeping) { @@ -504,8 +511,7 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean data.dropSSTables(); TransactionLog.waitForDeletions(); - - indexManager.invalidate(); + indexManager.invalidateAllIndexesBlocking(); materializedViewManager.invalidate(); invalidateCaches(); @@ -615,9 +621,12 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean } // also clean out any index leftovers. - for (IndexMetadata def : metadata.getIndexes()) - if (!def.isCustom()) - scrubDataDirectories(SecondaryIndex.newIndexMetadata(metadata, def)); + for (IndexMetadata index : metadata.getIndexes()) + if (!index.isCustom()) + { + CFMetaData indexMetadata = CassandraIndex.indexCfsMetadata(metadata, index); + scrubDataDirectories(indexMetadata); + } } // must be called after all sstables are loaded since row cache merges all row versions @@ -746,7 +755,7 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean try (Refs refs = Refs.ref(newSSTables)) { data.addSSTables(newSSTables); - indexManager.maybeBuildSecondaryIndexes(newSSTables, indexManager.allIndexesNames()); + indexManager.buildAllIndexesBlocking(newSSTables); } logger.info("Done loading load new SSTables for {}/{}", keyspace.getName(), name); @@ -766,10 +775,8 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean Iterable sstables = cfs.getSSTables(SSTableSet.CANONICAL); try (Refs refs = Refs.ref(sstables)) { - cfs.indexManager.setIndexRemoved(indexes); logger.info(String.format("User Requested secondary index re-build for %s/%s indexes", ksName, cfName)); - cfs.indexManager.maybeBuildSecondaryIndexes(refs, indexes); - cfs.indexManager.setIndexBuilt(indexes); + cfs.indexManager.rebuildIndexesBlocking(refs, indexes); } } @@ -852,16 +859,13 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean onHeapTotal += memtable.getAllocator().onHeap().owns(); offHeapTotal += memtable.getAllocator().offHeap().owns(); - for (SecondaryIndex index : indexManager.getIndexes()) + for (ColumnFamilyStore indexCfs : indexManager.getAllIndexColumnFamilyStores()) { - if (index.getIndexCfs() != null) - { - MemtableAllocator allocator = index.getIndexCfs().getTracker().getView().getCurrentMemtable().getAllocator(); - onHeapRatio += allocator.onHeap().ownershipRatio(); - offHeapRatio += allocator.offHeap().ownershipRatio(); - onHeapTotal += allocator.onHeap().owns(); - offHeapTotal += allocator.offHeap().owns(); - } + MemtableAllocator allocator = indexCfs.getTracker().getView().getCurrentMemtable().getAllocator(); + onHeapRatio += allocator.onHeap().ownershipRatio(); + offHeapRatio += allocator.offHeap().ownershipRatio(); + onHeapTotal += allocator.onHeap().owns(); + offHeapTotal += allocator.offHeap().owns(); } logger.info("Enqueuing flush of {}: {}", name, String.format("%d (%.0f%%) on-heap, %d (%.0f%%) off-heap", @@ -949,14 +953,7 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean */ if (flushSecondaryIndexes) - { - for (SecondaryIndex index : indexManager.getIndexesNotBackedByCfs()) - { - // flush any non-cfs backed indexes - logger.info("Flushing SecondaryIndex {}", index); - index.forceBlockingFlush(); - } - } + indexManager.flushAllCustomIndexesBlocking(); try { @@ -1127,14 +1124,11 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean onHeap += current.getAllocator().onHeap().ownershipRatio(); offHeap += current.getAllocator().offHeap().ownershipRatio(); - for (SecondaryIndex index : cfs.indexManager.getIndexes()) + for (ColumnFamilyStore indexCfs : cfs.indexManager.getAllIndexColumnFamilyStores()) { - if (index.getIndexCfs() != null) - { - MemtableAllocator allocator = index.getIndexCfs().getTracker().getView().getCurrentMemtable().getAllocator(); - onHeap += allocator.onHeap().ownershipRatio(); - offHeap += allocator.offHeap().ownershipRatio(); - } + MemtableAllocator allocator = indexCfs.getTracker().getView().getCurrentMemtable().getAllocator(); + onHeap += allocator.onHeap().ownershipRatio(); + offHeap += allocator.offHeap().ownershipRatio(); } float ratio = Math.max(onHeap, offHeap); @@ -1185,7 +1179,8 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean * param @ key - key for update/insert * param @ columnFamily - columnFamily changes */ - public void apply(PartitionUpdate update, SecondaryIndexManager.Updater indexer, OpOrder.Group opGroup, ReplayPosition replayPosition) + public void apply(PartitionUpdate update, UpdateTransaction indexer, OpOrder.Group opGroup, ReplayPosition replayPosition) + { long start = System.nanoTime(); Memtable mt = data.getMemtableFor(opGroup, replayPosition); @@ -1390,22 +1385,25 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean if (!isIndex()) return false; - SecondaryIndex index = null; + boolean validIndex = false; + ColumnFamilyStore parentCfs = null; + String indexName = null; if (metadata.cfName.contains(Directories.SECONDARY_INDEX_NAME_SEPARATOR)) { String[] parts = metadata.cfName.split("\\" + Directories.SECONDARY_INDEX_NAME_SEPARATOR, 2); - ColumnFamilyStore parentCfs = keyspace.getColumnFamilyStore(parts[0]); - index = parentCfs.indexManager.getIndexByName(metadata.cfName); - assert index != null; + parentCfs = keyspace.getColumnFamilyStore(parts[0]); + assert parentCfs.indexManager.getAllIndexColumnFamilyStores().contains(this); + validIndex = true; + indexName = this.name; } - if (index == null) + if (! validIndex) return false; truncateBlocking(); logger.warn("Rebuilding index for {} because of <{}>", name, failure.getMessage()); - index.getBaseCfs().rebuildSecondaryIndex(index.getIndexName()); + parentCfs.rebuildSecondaryIndex(indexName); return true; } @@ -1946,8 +1944,7 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean ReplayPosition replayAfter = discardSSTables(truncatedAt); - for (SecondaryIndex index : indexManager.getIndexes()) - index.truncateBlocking(truncatedAt); + indexManager.truncateAllIndexesBlocking(truncatedAt); materializedViewManager.truncateBlocking(truncatedAt); @@ -2192,12 +2189,12 @@ public class ColumnFamilyStore implements ColumnFamilyStoreMBean { // we return the main CFS first, which we rely on for simplicity in switchMemtable(), for getting the // latest replay position - return Iterables.concat(Collections.singleton(this), indexManager.getIndexesBackedByCfs()); + return Iterables.concat(Collections.singleton(this), indexManager.getAllIndexColumnFamilyStores()); } public List getBuiltIndexes() { - return indexManager.getBuiltIndexes(); + return indexManager.getBuiltIndexNames(); } public int getUnleveledSSTables() diff --git a/src/java/org/apache/cassandra/db/Keyspace.java b/src/java/org/apache/cassandra/db/Keyspace.java index c2613fe1e2..199cd25b59 100644 --- a/src/java/org/apache/cassandra/db/Keyspace.java +++ b/src/java/org/apache/cassandra/db/Keyspace.java @@ -33,26 +33,26 @@ import org.slf4j.LoggerFactory; import org.apache.cassandra.concurrent.Stage; import org.apache.cassandra.concurrent.StageManager; import org.apache.cassandra.config.*; -import org.apache.cassandra.db.rows.UnfilteredRowIterator; import org.apache.cassandra.db.commitlog.CommitLog; import org.apache.cassandra.db.commitlog.ReplayPosition; import org.apache.cassandra.db.compaction.CompactionManager; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexManager; +import org.apache.cassandra.db.lifecycle.SSTableSet; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; import org.apache.cassandra.db.view.MaterializedViewManager; import org.apache.cassandra.exceptions.WriteTimeoutException; -import org.apache.cassandra.db.partitions.PartitionUpdate; -import org.apache.cassandra.db.lifecycle.SSTableSet; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.index.transactions.UpdateTransaction; import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.locator.AbstractReplicationStrategy; +import org.apache.cassandra.metrics.KeyspaceMetrics; import org.apache.cassandra.schema.KeyspaceMetadata; import org.apache.cassandra.service.StorageService; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.FBUtilities; import org.apache.cassandra.utils.JVMStabilityInspector; import org.apache.cassandra.utils.concurrent.OpOrder; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.metrics.KeyspaceMetrics; /** * It represents a Keyspace. @@ -469,10 +469,10 @@ public class Keyspace } Tracing.trace("Adding to {} memtable", upd.metadata().cfName); - SecondaryIndexManager.Updater updater = updateIndexes - ? cfs.indexManager.updaterFor(upd, opGroup, nowInSec) - : SecondaryIndexManager.nullUpdater; - cfs.apply(upd, updater, opGroup, replayPosition); + UpdateTransaction indexTransaction = updateIndexes + ? cfs.indexManager.newUpdateTransaction(upd, opGroup, nowInSec) + : UpdateTransaction.NO_OP; + cfs.apply(upd, indexTransaction, opGroup, replayPosition); } } finally @@ -490,15 +490,16 @@ public class Keyspace /** * @param key row to index * @param cfs ColumnFamily to index partition in - * @param idxNames columns to index, in comparator order + * @param indexes the indexes to submit the row to */ - public static void indexPartition(DecoratedKey key, ColumnFamilyStore cfs, Set idxNames) + public static void indexPartition(DecoratedKey key, ColumnFamilyStore cfs, Set indexes) { if (logger.isDebugEnabled()) logger.debug("Indexing partition {} ", cfs.metadata.getKeyValidator().getString(key.getKey())); - Set indexes = cfs.indexManager.getIndexesByNames(idxNames); - SinglePartitionReadCommand cmd = SinglePartitionReadCommand.fullPartitionRead(cfs.metadata, FBUtilities.nowInSeconds(), key); + SinglePartitionReadCommand cmd = SinglePartitionReadCommand.fullPartitionRead(cfs.metadata, + FBUtilities.nowInSeconds(), + key); try (OpOrder.Group opGroup = cfs.keyspace.writeOrder.start(); UnfilteredRowIterator partition = cmd.queryMemtableAndDisk(cfs, opGroup)) @@ -515,7 +516,9 @@ public class Keyspace return futures; } - public Iterable getValidColumnFamilies(boolean allowIndexes, boolean autoAddIndexes, String... cfNames) throws IOException + public Iterable getValidColumnFamilies(boolean allowIndexes, + boolean autoAddIndexes, + String... cfNames) throws IOException { Set valid = new HashSet<>(); @@ -526,65 +529,58 @@ public class Keyspace { valid.add(cfStore); if (autoAddIndexes) - { - for (SecondaryIndex si : cfStore.indexManager.getIndexes()) - { - if (si.getIndexCfs() != null) { - logger.info("adding secondary index {} to operation", si.getIndexName()); - valid.add(si.getIndexCfs()); - } - } - - } + valid.addAll(getIndexColumnFamilyStores(cfStore)); } return valid; } - // filter out interesting stores + + // include the specified stores and possibly the stores of any of their indexes for (String cfName : cfNames) { - //if the CF name is an index, just flush the CF that owns the index - String baseCfName = cfName; - String idxName = null; - if (cfName.contains(".")) // secondary index + int separatorPos = cfName.indexOf(Directories.SECONDARY_INDEX_NAME_SEPARATOR); + if (separatorPos > -1) { - if(!allowIndexes) + if (!allowIndexes) { logger.warn("Operation not allowed on secondary Index table ({})", cfName); continue; } + String baseName = cfName.substring(0, separatorPos); + String indexName = cfName.substring(separatorPos + Directories.SECONDARY_INDEX_NAME_SEPARATOR.length()); - String[] parts = cfName.split("\\.", 2); - baseCfName = parts[0]; - idxName = parts[1]; - } + ColumnFamilyStore baseCfs = getColumnFamilyStore(baseName); + Index index = baseCfs.indexManager.getIndexByName(indexName); + if (index == null) + throw new IllegalArgumentException(String.format("Invalid index specified: %s/%s.", + baseCfs.metadata.cfName, + indexName)); - ColumnFamilyStore cfStore = getColumnFamilyStore(baseCfName); - if (idxName != null) - { - Collection< SecondaryIndex > indexes = cfStore.indexManager.getIndexesByNames(new HashSet<>(Arrays.asList(cfName))); - if (indexes.isEmpty()) - throw new IllegalArgumentException(String.format("Invalid index specified: %s/%s.", baseCfName, idxName)); - else - valid.add(Iterables.get(indexes, 0).getIndexCfs()); + if (index.getBackingTable().isPresent()) + valid.add(index.getBackingTable().get()); } else { + ColumnFamilyStore cfStore = getColumnFamilyStore(cfName); valid.add(cfStore); - if(autoAddIndexes) - { - for(SecondaryIndex si : cfStore.indexManager.getIndexes()) - { - if (si.getIndexCfs() != null) { - logger.info("adding secondary index {} to operation", si.getIndexName()); - valid.add(si.getIndexCfs()); - } - } - } + if (autoAddIndexes) + valid.addAll(getIndexColumnFamilyStores(cfStore)); } } + return valid; } + private Set getIndexColumnFamilyStores(ColumnFamilyStore baseCfs) + { + Set stores = new HashSet<>(); + for (ColumnFamilyStore indexCfs : baseCfs.indexManager.getAllIndexColumnFamilyStores()) + { + logger.info("adding secondary index table {} to operation", indexCfs.metadata.cfName); + stores.add(indexCfs); + } + return stores; + } + public static Iterable all() { return Iterables.transform(Schema.instance.getKeyspaces(), keyspaceTransformer); diff --git a/src/java/org/apache/cassandra/db/Memtable.java b/src/java/org/apache/cassandra/db/Memtable.java index 4a54666864..7af65d1786 100644 --- a/src/java/org/apache/cassandra/db/Memtable.java +++ b/src/java/org/apache/cassandra/db/Memtable.java @@ -19,40 +19,42 @@ package org.apache.cassandra.db; import java.io.File; import java.util.*; -import java.util.concurrent.ConcurrentNavigableMap; -import java.util.concurrent.ConcurrentSkipListMap; -import java.util.concurrent.ConcurrentSkipListSet; -import java.util.concurrent.TimeUnit; +import java.util.concurrent.*; import java.util.concurrent.atomic.AtomicBoolean; import java.util.concurrent.atomic.AtomicLong; import java.util.concurrent.atomic.AtomicReference; import com.google.common.annotations.VisibleForTesting; -import org.apache.cassandra.db.compaction.OperationType; -import org.apache.cassandra.db.lifecycle.LifecycleTransaction; -import org.apache.cassandra.io.sstable.SSTableTxnWriter; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.config.DatabaseDescriptor; -import org.apache.cassandra.db.rows.*; import org.apache.cassandra.db.commitlog.CommitLog; import org.apache.cassandra.db.commitlog.ReplayPosition; -import org.apache.cassandra.db.index.SecondaryIndexManager; -import org.apache.cassandra.dht.Murmur3Partitioner.LongToken; -import org.apache.cassandra.db.filter.*; +import org.apache.cassandra.db.compaction.OperationType; +import org.apache.cassandra.db.filter.ClusteringIndexFilter; +import org.apache.cassandra.db.filter.ColumnFilter; +import org.apache.cassandra.db.lifecycle.LifecycleTransaction; import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.EncodingStats; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; import org.apache.cassandra.dht.*; +import org.apache.cassandra.dht.Murmur3Partitioner.LongToken; +import org.apache.cassandra.index.transactions.UpdateTransaction; import org.apache.cassandra.io.sstable.Descriptor; +import org.apache.cassandra.io.sstable.SSTableTxnWriter; import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.io.sstable.metadata.MetadataCollector; import org.apache.cassandra.io.util.DiskAwareRunnable; import org.apache.cassandra.service.ActiveRepairService; -import org.apache.cassandra.utils.*; +import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.FBUtilities; +import org.apache.cassandra.utils.ObjectSizes; import org.apache.cassandra.utils.concurrent.OpOrder; -import org.apache.cassandra.utils.memory.*; +import org.apache.cassandra.utils.memory.MemtableAllocator; +import org.apache.cassandra.utils.memory.MemtablePool; public class Memtable implements Comparable { @@ -208,7 +210,7 @@ public class Memtable implements Comparable * * replayPosition should only be null if this is a secondary index, in which case it is *expected* to be null */ - long put(PartitionUpdate update, SecondaryIndexManager.Updater indexer, OpOrder.Group opGroup) + long put(PartitionUpdate update, UpdateTransaction indexer, OpOrder.Group opGroup) { AtomicBTreePartition previous = partitions.get(update.partitionKey()); diff --git a/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java b/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java index 42d5425335..2ba45e753b 100644 --- a/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java +++ b/src/java/org/apache/cassandra/db/PartitionRangeReadCommand.java @@ -25,21 +25,22 @@ import com.google.common.collect.Iterables; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.db.filter.*; import org.apache.cassandra.db.lifecycle.SSTableSet; import org.apache.cassandra.db.lifecycle.View; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.filter.*; import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; import org.apache.cassandra.dht.AbstractBounds; import org.apache.cassandra.exceptions.RequestExecutionException; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; -import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.metrics.TableMetrics; import org.apache.cassandra.net.MessageOut; import org.apache.cassandra.net.MessagingService; -import org.apache.cassandra.service.*; +import org.apache.cassandra.service.ClientState; +import org.apache.cassandra.service.StorageProxy; import org.apache.cassandra.service.pager.*; import org.apache.cassandra.thrift.ThriftResultsMerger; import org.apache.cassandra.tracing.Tracing; @@ -273,8 +274,8 @@ public class PartitionRangeReadCommand extends ReadCommand public PartitionIterator postReconciliationProcessing(PartitionIterator result) { ColumnFamilyStore cfs = Keyspace.open(metadata().ksName).getColumnFamilyStore(metadata().cfName); - SecondaryIndexSearcher searcher = getIndexSearcher(cfs); - return searcher == null ? result : searcher.postReconciliationProcessing(rowFilter(), result); + Index index = getIndex(cfs, false); + return index == null ? result : index.postProcessorFor(this).apply(result, this); } @Override diff --git a/src/java/org/apache/cassandra/db/ReadCommand.java b/src/java/org/apache/cassandra/db/ReadCommand.java index d63a832702..1d5d4778da 100644 --- a/src/java/org/apache/cassandra/db/ReadCommand.java +++ b/src/java/org/apache/cassandra/db/ReadCommand.java @@ -22,20 +22,16 @@ import java.nio.ByteBuffer; import java.util.*; import com.google.common.collect.Lists; - import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.config.DatabaseDescriptor; -import org.apache.cassandra.config.Schema; +import org.apache.cassandra.config.*; import org.apache.cassandra.cql3.Operator; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; import org.apache.cassandra.db.filter.*; -import org.apache.cassandra.db.rows.*; import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.*; import org.apache.cassandra.dht.AbstractBounds; +import org.apache.cassandra.index.Index; import org.apache.cassandra.io.IVersionedSerializer; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; @@ -43,7 +39,6 @@ import org.apache.cassandra.metrics.TableMetrics; import org.apache.cassandra.net.MessageOut; import org.apache.cassandra.net.MessagingService; import org.apache.cassandra.service.ClientWarn; -import org.apache.cassandra.service.StorageService; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.Pair; @@ -261,14 +256,16 @@ public abstract class ReadCommand implements ReadQuery : ReadResponse.createDataResponse(iterator, selection); } - protected SecondaryIndexSearcher getIndexSearcher(ColumnFamilyStore cfs) + protected Index getIndex(ColumnFamilyStore cfs, boolean includeInTrace) { - return cfs.indexManager.getBestIndexSearcherFor(this); + return cfs.indexManager.getBestIndexFor(this, includeInTrace); } /** * Executes this command on the local host. * + * @param orderGroup the operation group spanning this command + * * @return an iterator over the result of executing this command locally. */ @SuppressWarnings("resource") // The result iterator is closed upon exceptions (we know it's fine to potentially not close the intermediary @@ -278,10 +275,12 @@ public abstract class ReadCommand implements ReadQuery long startTimeNanos = System.nanoTime(); ColumnFamilyStore cfs = Keyspace.openAndGetStore(metadata()); - SecondaryIndexSearcher searcher = getIndexSearcher(cfs); + Index index = getIndex(cfs, true); + Index.Searcher searcher = index == null ? null : index.searcherFor(this); + UnfilteredPartitionIterator resultIterator = searcher == null ? queryStorage(cfs, orderGroup) - : searcher.search(this, orderGroup); + : searcher.search(orderGroup); try { @@ -291,7 +290,7 @@ public abstract class ReadCommand implements ReadQuery // no point in checking it again. RowFilter updatedFilter = searcher == null ? rowFilter() - : rowFilter().without(searcher.primaryClause(this)); + : index.getPostIndexQueryFilter(rowFilter()); // TODO: We'll currently do filtering by the rowFilter here because it's convenient. However, // we'll probably want to optimize by pushing it down the layer (like for dropped columns) as it diff --git a/src/java/org/apache/cassandra/db/ReadOrderGroup.java b/src/java/org/apache/cassandra/db/ReadOrderGroup.java index 0a5bee8c64..44befa2625 100644 --- a/src/java/org/apache/cassandra/db/ReadOrderGroup.java +++ b/src/java/org/apache/cassandra/db/ReadOrderGroup.java @@ -17,7 +17,7 @@ */ package org.apache.cassandra.db; -import org.apache.cassandra.db.index.*; +import org.apache.cassandra.index.Index; import org.apache.cassandra.utils.concurrent.OpOrder; public class ReadOrderGroup implements AutoCloseable @@ -98,14 +98,8 @@ public class ReadOrderGroup implements AutoCloseable private static ColumnFamilyStore maybeGetIndexCfs(ColumnFamilyStore baseCfs, ReadCommand command) { - SecondaryIndexSearcher searcher = command.getIndexSearcher(baseCfs); - if (searcher == null) - return null; - - SecondaryIndex index = searcher.highestSelectivityIndex(command.rowFilter()); - return index == null || !(index instanceof AbstractSimplePerColumnSecondaryIndex) - ? null - : ((AbstractSimplePerColumnSecondaryIndex)index).getIndexCfs(); + Index index = baseCfs.indexManager.getBestIndexFor(command); + return index == null ? null : index.getBackingTable().orElse(null); } public void close() diff --git a/src/java/org/apache/cassandra/db/SystemKeyspace.java b/src/java/org/apache/cassandra/db/SystemKeyspace.java index 38cfed6f71..fb9eb48b3d 100644 --- a/src/java/org/apache/cassandra/db/SystemKeyspace.java +++ b/src/java/org/apache/cassandra/db/SystemKeyspace.java @@ -17,18 +17,22 @@ */ package org.apache.cassandra.db; -import java.io.*; +import java.io.File; +import java.io.IOError; +import java.io.IOException; import java.net.InetAddress; import java.nio.ByteBuffer; import java.util.*; import java.util.concurrent.TimeUnit; - +import java.util.stream.Collectors; +import java.util.stream.StreamSupport; import javax.management.openmbean.OpenDataException; import javax.management.openmbean.TabularData; -import com.google.common.collect.*; +import com.google.common.collect.HashMultimap; +import com.google.common.collect.ImmutableSet; +import com.google.common.collect.SetMultimap; import com.google.common.io.ByteStreams; - import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -37,25 +41,18 @@ import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.cql3.QueryProcessor; import org.apache.cassandra.cql3.UntypedResultSet; import org.apache.cassandra.cql3.functions.*; -import org.apache.cassandra.db.partitions.*; import org.apache.cassandra.db.commitlog.ReplayPosition; import org.apache.cassandra.db.compaction.CompactionHistoryTabularData; import org.apache.cassandra.db.marshal.*; -import org.apache.cassandra.dht.IPartitioner; -import org.apache.cassandra.dht.LocalPartitioner; -import org.apache.cassandra.dht.Range; -import org.apache.cassandra.dht.Token; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.dht.*; import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.io.sstable.Descriptor; -import org.apache.cassandra.io.util.DataInputBuffer; -import org.apache.cassandra.io.util.DataOutputBuffer; -import org.apache.cassandra.io.util.FileUtils; -import org.apache.cassandra.io.util.NIODataInputStream; +import org.apache.cassandra.io.util.*; import org.apache.cassandra.locator.IEndpointSnitch; import org.apache.cassandra.metrics.RestorableMeter; import org.apache.cassandra.net.MessagingService; import org.apache.cassandra.schema.*; -import org.apache.cassandra.schema.Tables; import org.apache.cassandra.service.StorageService; import org.apache.cassandra.service.paxos.Commit; import org.apache.cassandra.service.paxos.PaxosState; @@ -65,7 +62,6 @@ import org.apache.cassandra.utils.*; import static java.util.Collections.emptyMap; import static java.util.Collections.singletonMap; - import static org.apache.cassandra.cql3.QueryProcessor.executeInternal; import static org.apache.cassandra.cql3.QueryProcessor.executeOnceInternal; @@ -516,7 +512,14 @@ public final class SystemKeyspace if (ksname.equals("system") && cfname.equals(COMPACTION_HISTORY)) return; String req = "INSERT INTO system.%s (id, keyspace_name, columnfamily_name, compacted_at, bytes_in, bytes_out, rows_merged) VALUES (?, ?, ?, ?, ?, ?, ?)"; - executeInternal(String.format(req, COMPACTION_HISTORY), UUIDGen.getTimeUUID(), ksname, cfname, ByteBufferUtil.bytes(compactedAt), bytesIn, bytesOut, rowsMerged); + executeInternal(String.format(req, COMPACTION_HISTORY), + UUIDGen.getTimeUUID(), + ksname, + cfname, + ByteBufferUtil.bytes(compactedAt), + bytesIn, + bytesOut, + rowsMerged); } public static TabularData getCompactionHistory() throws OpenDataException @@ -1030,6 +1033,16 @@ public final class SystemKeyspace forceBlockingFlush(BUILT_INDEXES); } + public static List getBuiltIndexes(String keyspaceName, Set indexNames) + { + List names = new ArrayList<>(indexNames); + String req = "SELECT index_name from %s.\"%s\" WHERE table_name=? AND index_name IN ?"; + UntypedResultSet results = executeInternal(String.format(req, NAME, BUILT_INDEXES), keyspaceName, names); + return StreamSupport.stream(results.spliterator(), false) + .map(r -> r.getString("index_name")) + .collect(Collectors.toList()); + } + /** * Read the host ID from the system keyspace, creating (and storing) one if * none exists. diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionIterator.java b/src/java/org/apache/cassandra/db/compaction/CompactionIterator.java index cab96fb649..4aaa17a8a0 100644 --- a/src/java/org/apache/cassandra/db/compaction/CompactionIterator.java +++ b/src/java/org/apache/cassandra/db/compaction/CompactionIterator.java @@ -17,15 +17,19 @@ */ package org.apache.cassandra.db.compaction; -import java.util.UUID; import java.util.List; +import java.util.UUID; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.index.SecondaryIndexManager; -import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.partitions.PurgingPartitionIterator; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators; import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.index.transactions.CompactionTransaction; import org.apache.cassandra.io.sstable.ISSTableScanner; import org.apache.cassandra.metrics.CompactionMetrics; @@ -47,6 +51,7 @@ import org.apache.cassandra.metrics.CompactionMetrics; */ public class CompactionIterator extends CompactionInfo.Holder implements UnfilteredPartitionIterator { + private static final Logger logger = LoggerFactory.getLogger(CompactionIterator.class); private static final long UNFILTERED_TO_UPDATE_PROGRESS = 100; private final OperationType type; @@ -148,29 +153,33 @@ public class CompactionIterator extends CompactionInfo.Holder implements Unfilte if (type != OperationType.COMPACTION || !controller.cfs.indexManager.hasIndexes()) return null; - // If we have a 2ndary index, we must update it with deleted/shadowed cells. - // TODO: this should probably be done asynchronously and batched. - final SecondaryIndexManager.Updater indexer = controller.cfs.indexManager.gcUpdaterFor(partitionKey, nowInSec); - final RowDiffListener diffListener = new RowDiffListener() + Columns statics = Columns.NONE; + Columns regulars = Columns.NONE; + for (UnfilteredRowIterator iter : versions) { - public void onPrimaryKeyLivenessInfo(int i, Clustering clustering, LivenessInfo merged, LivenessInfo original) + if (iter != null) { + statics = statics.mergeTo(iter.columns().statics); + regulars = regulars.mergeTo(iter.columns().regulars); } + } + final PartitionColumns partitionColumns = new PartitionColumns(statics, regulars); - public void onDeletion(int i, Clustering clustering, DeletionTime merged, DeletionTime original) - { - } - - public void onComplexDeletion(int i, Clustering clustering, ColumnDefinition column, DeletionTime merged, DeletionTime original) - { - } - - public void onCell(int i, Clustering clustering, Cell merged, Cell original) - { - if (original != null && (merged == null || !merged.isLive(nowInSec))) - indexer.remove(clustering, original); - } - }; + // If we have a 2ndary index, we must update it with deleted/shadowed cells. + // we can reuse a single CleanupTransaction for the duration of a partition. + // Currently, it doesn't do any batching of row updates, so every merge event + // for a single partition results in a fresh cycle of: + // * Get new Indexer instances + // * Indexer::start + // * Indexer::onRowMerge (for every row being merged by the compaction) + // * Indexer::commit + // A new OpOrder.Group is opened in an ARM block wrapping the commits + // TODO: this should probably be done asynchronously and batched. + final CompactionTransaction indexTransaction = + controller.cfs.indexManager.newCompactionTransaction(partitionKey, + partitionColumns, + versions.size(), + nowInSec); return new UnfilteredRowIterators.MergeListener() { @@ -180,7 +189,9 @@ public class CompactionIterator extends CompactionInfo.Holder implements Unfilte public void onMergedRows(Row merged, Columns columns, Row[] versions) { - Rows.diff(merged, columns, versions, diffListener); + indexTransaction.start(); + indexTransaction.onRowMerge(columns, merged, versions); + indexTransaction.commit(); } public void onMergedRangeTombstoneMarkers(RangeTombstoneMarker mergedMarker, RangeTombstoneMarker[] versions) diff --git a/src/java/org/apache/cassandra/db/compaction/CompactionManager.java b/src/java/org/apache/cassandra/db/compaction/CompactionManager.java index 66f9ed5a4e..089034195b 100644 --- a/src/java/org/apache/cassandra/db/compaction/CompactionManager.java +++ b/src/java/org/apache/cassandra/db/compaction/CompactionManager.java @@ -20,22 +20,8 @@ package org.apache.cassandra.db.compaction; import java.io.File; import java.io.IOException; import java.lang.management.ManagementFactory; -import java.util.ArrayList; -import java.util.Collection; -import java.util.Collections; -import java.util.HashSet; -import java.util.Iterator; -import java.util.List; -import java.util.Map; -import java.util.Set; -import java.util.UUID; -import java.util.concurrent.BlockingQueue; -import java.util.concurrent.Callable; -import java.util.concurrent.ExecutionException; -import java.util.concurrent.Future; -import java.util.concurrent.LinkedBlockingQueue; -import java.util.concurrent.SynchronousQueue; -import java.util.concurrent.TimeUnit; +import java.util.*; +import java.util.concurrent.*; import javax.management.MBeanServer; import javax.management.ObjectName; import javax.management.openmbean.OpenDataException; @@ -56,33 +42,27 @@ import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.config.Schema; import org.apache.cassandra.db.*; import org.apache.cassandra.db.compaction.CompactionInfo.Holder; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.index.SecondaryIndexBuilder; -import org.apache.cassandra.db.view.MaterializedViewBuilder; import org.apache.cassandra.db.lifecycle.LifecycleTransaction; import org.apache.cassandra.db.lifecycle.SSTableSet; import org.apache.cassandra.db.lifecycle.View; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; +import org.apache.cassandra.db.view.MaterializedViewBuilder; import org.apache.cassandra.dht.Bounds; import org.apache.cassandra.dht.Range; import org.apache.cassandra.dht.Token; -import org.apache.cassandra.io.sstable.*; +import org.apache.cassandra.index.SecondaryIndexBuilder; +import org.apache.cassandra.io.sstable.Descriptor; +import org.apache.cassandra.io.sstable.ISSTableScanner; +import org.apache.cassandra.io.sstable.SSTableRewriter; import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.io.sstable.format.SSTableWriter; import org.apache.cassandra.io.sstable.metadata.MetadataCollector; -import org.apache.cassandra.io.sstable.ISSTableScanner; -import org.apache.cassandra.io.sstable.SSTableRewriter; import org.apache.cassandra.io.util.FileUtils; import org.apache.cassandra.metrics.CompactionMetrics; import org.apache.cassandra.repair.Validator; import org.apache.cassandra.service.ActiveRepairService; import org.apache.cassandra.service.StorageService; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.JVMStabilityInspector; -import org.apache.cassandra.utils.MerkleTrees; -import org.apache.cassandra.utils.Pair; -import org.apache.cassandra.utils.WrappedRunnable; -import org.apache.cassandra.utils.UUIDGen; -import org.apache.cassandra.utils.concurrent.OpOrder; +import org.apache.cassandra.utils.*; import org.apache.cassandra.utils.concurrent.Refs; import static java.util.Collections.singleton; @@ -847,7 +827,7 @@ public class CompactionManager implements CompactionManagerMBean } // flush to ensure we don't lose the tombstones on a restart, since they are not commitlog'd - cfs.indexManager.flushIndexesBlocking(); + cfs.indexManager.flushAllIndexesBlocking(); finished = writer.finish(); } @@ -939,11 +919,7 @@ public class CompactionManager implements CompactionManagerMBean cfs.invalidateCachedPartition(partition.partitionKey()); - // acquire memtable lock here because secondary index deletion may cause a race. See CASSANDRA-3712 - try (OpOrder.Group opGroup = cfs.keyspace.writeOrder.start()) - { - cfs.indexManager.deleteFromIndexes(partition, opGroup, nowInSec); - } + cfs.indexManager.deletePartition(partition, nowInSec); return null; } } diff --git a/src/java/org/apache/cassandra/db/filter/RowFilter.java b/src/java/org/apache/cassandra/db/filter/RowFilter.java index a5212aa641..bbec004cc3 100644 --- a/src/java/org/apache/cassandra/db/filter/RowFilter.java +++ b/src/java/org/apache/cassandra/db/filter/RowFilter.java @@ -27,10 +27,9 @@ import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; import org.apache.cassandra.db.marshal.*; -import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.*; import org.apache.cassandra.exceptions.InvalidRequestException; import org.apache.cassandra.io.util.DataInputPlus; import org.apache.cassandra.io.util.DataOutputPlus; @@ -38,7 +37,9 @@ import org.apache.cassandra.net.MessagingService; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FBUtilities; -import static org.apache.cassandra.cql3.statements.RequestValidations.*; +import static org.apache.cassandra.cql3.statements.RequestValidations.checkBindValueSet; +import static org.apache.cassandra.cql3.statements.RequestValidations.checkFalse; +import static org.apache.cassandra.cql3.statements.RequestValidations.checkNotNull; /** * A filter on which rows a given query should include or exclude. @@ -91,6 +92,11 @@ public abstract class RowFilter implements Iterable expressions.add(new ThriftExpression(metadata, name, op, value)); } + public List getExpressions() + { + return expressions; + } + /** * Filters the provided iterator so that only the row satisfying the expression of this filter * are included in the resulting iterator. diff --git a/src/java/org/apache/cassandra/db/index/AbstractSimplePerColumnSecondaryIndex.java b/src/java/org/apache/cassandra/db/index/AbstractSimplePerColumnSecondaryIndex.java deleted file mode 100644 index a631b9a7d4..0000000000 --- a/src/java/org/apache/cassandra/db/index/AbstractSimplePerColumnSecondaryIndex.java +++ /dev/null @@ -1,251 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.util.Collection; -import java.util.Collections; -import java.util.concurrent.Future; - -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.compaction.CompactionManager; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.db.marshal.AbstractType; -import org.apache.cassandra.exceptions.InvalidRequestException; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.concurrent.OpOrder; - -/** - * Implements a secondary index for a column family using a second column family - * in which the row keys are indexed values, and column names are base row keys. - */ -public abstract class AbstractSimplePerColumnSecondaryIndex extends PerColumnSecondaryIndex -{ - protected ColumnFamilyStore indexCfs; - - // SecondaryIndex "forces" a set of ColumnDefinition. However this class (and thus it's subclass) - // only support one def per index. So inline it in a field for 1) convenience and 2) avoid creating - // an iterator each time we need to access it. - // TODO: we should fix SecondaryIndex API - protected ColumnDefinition columnDef; - - public void init() - { - assert baseCfs != null && columnDefs != null && columnDefs.size() == 1; - - columnDef = columnDefs.iterator().next(); - - CFMetaData indexedCfMetadata = SecondaryIndex.newIndexMetadata(baseCfs.metadata, indexMetadata, getIndexKeyComparator()); - indexCfs = ColumnFamilyStore.createColumnFamilyStore(baseCfs.keyspace, - indexedCfMetadata.cfName, - indexedCfMetadata, - baseCfs.getTracker().loadsstables); - } - - protected AbstractType getIndexKeyComparator() - { - return columnDef.type; - } - - public ColumnDefinition indexedColumn() - { - return columnDef; - } - - @Override - String indexTypeForGrouping() - { - return "_internal_"; - } - - protected Clustering makeIndexClustering(ByteBuffer rowKey, Clustering clustering, Cell cell) - { - return makeIndexClustering(rowKey, clustering, cell == null ? null : cell.path()); - } - - protected Clustering makeIndexClustering(ByteBuffer rowKey, Clustering clustering, CellPath path) - { - return buildIndexClusteringPrefix(rowKey, clustering, path).build(); - } - - protected Slice.Bound makeIndexBound(ByteBuffer rowKey, Slice.Bound bound) - { - return buildIndexClusteringPrefix(rowKey, bound, null).buildBound(bound.isStart(), bound.isInclusive()); - } - - protected abstract CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path); - - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, Cell cell) - { - return cell == null - ? getIndexedValue(rowKey, clustering, null, null) - : getIndexedValue(rowKey, clustering, cell.value(), cell.path()); - } - - protected abstract ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath cellPath); - - public void delete(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - deleteForCleanup(rowKey, clustering, cell, opGroup, nowInSec); - } - - public void deleteForCleanup(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - delete(rowKey, clustering, cell.value(), cell.path(), new DeletionTime(cell.timestamp(), nowInSec), opGroup); - } - - public void delete(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path, DeletionTime deletion, OpOrder.Group opGroup) - { - DecoratedKey valueKey = getIndexKeyFor(getIndexedValue(rowKey, clustering, cellValue, path)); - - Row row = BTreeRow.emptyDeletedRow(makeIndexClustering(rowKey, clustering, path), deletion); - PartitionUpdate upd = PartitionUpdate.singleRowUpdate(indexCfs.metadata, valueKey, row); - - indexCfs.apply(upd, SecondaryIndexManager.nullUpdater, opGroup, null); - if (logger.isDebugEnabled()) - logger.debug("removed index entry for cleaned-up value {}:{}", valueKey, upd); - } - - public void insert(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup) - { - insert(rowKey, clustering, cell, LivenessInfo.create(cell.timestamp(), cell.ttl(), cell.localDeletionTime()), opGroup); - } - - public void insert(ByteBuffer rowKey, Clustering clustering, Cell cell, LivenessInfo info, OpOrder.Group opGroup) - { - DecoratedKey valueKey = getIndexKeyFor(getIndexedValue(rowKey, clustering, cell)); - - Row row = BTreeRow.noCellLiveRow(makeIndexClustering(rowKey, clustering, cell), info); - PartitionUpdate upd = PartitionUpdate.singleRowUpdate(indexCfs.metadata, valueKey, row); - - if (logger.isDebugEnabled()) - logger.debug("applying index row {} in {}", indexCfs.metadata.getKeyValidator().getString(valueKey.getKey()), upd); - - indexCfs.apply(upd, SecondaryIndexManager.nullUpdater, opGroup, null); - } - - public void update(ByteBuffer rowKey, Clustering clustering, Cell oldCell, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - // insert the new value before removing the old one, so we never have a period - // where the row is invisible to both queries (the opposite seems preferable); see CASSANDRA-5540 - insert(rowKey, clustering, cell, opGroup); - if (SecondaryIndexManager.shouldCleanupOldValue(oldCell, cell)) - delete(rowKey, clustering, oldCell, opGroup, nowInSec); - } - - public boolean indexes(ColumnDefinition column) - { - return column.name.equals(columnDef.name); - } - - public void removeIndex(ByteBuffer columnName) - { - // interrupt in-progress compactions - Collection cfss = Collections.singleton(indexCfs); - CompactionManager.instance.interruptCompactionForCFs(cfss, true); - CompactionManager.instance.waitForCessation(cfss); - - indexCfs.keyspace.writeOrder.awaitNewBarrier(); - indexCfs.forceBlockingFlush(); - - indexCfs.readOrdering.awaitNewBarrier(); - indexCfs.invalidate(); - } - - public void forceBlockingFlush() - { - Future wait; - // we synchronise on the baseCfs to make sure we are ordered correctly with other flushes to the base CFS - synchronized (baseCfs.getTracker()) - { - wait = indexCfs.forceFlush(); - } - FBUtilities.waitOnFuture(wait); - } - - public void invalidate() - { - indexCfs.invalidate(); - } - - public void truncateBlocking(long truncatedAt) - { - indexCfs.discardSSTables(truncatedAt); - } - - public ColumnFamilyStore getIndexCfs() - { - return indexCfs; - } - - protected ClusteringComparator getIndexComparator() - { - assert indexCfs != null; - return indexCfs.metadata.comparator; - } - - public String getIndexName() - { - return indexCfs.name; - } - - public void reload() - { - indexCfs.metadata.reloadIndexMetadataProperties(baseCfs.metadata); - indexCfs.reload(); - } - - public long estimateResultRows() - { - return getIndexCfs().getMeanColumns(); - } - - public void validate(DecoratedKey partitionKey) throws InvalidRequestException - { - if (columnDef.kind == ColumnDefinition.Kind.PARTITION_KEY) - validateIndexedValue(getIndexedValue(partitionKey.getKey(), null, null, null)); - } - - public void validate(Clustering clustering) throws InvalidRequestException - { - if (columnDef.kind == ColumnDefinition.Kind.CLUSTERING) - validateIndexedValue(getIndexedValue(null, clustering, null, null)); - } - - public void validate(ByteBuffer cellValue, CellPath path) throws InvalidRequestException - { - if (!columnDef.isPrimaryKeyColumn()) - validateIndexedValue(getIndexedValue(null, null, cellValue, path)); - } - - private void validateIndexedValue(ByteBuffer value) - { - if (value != null && value.remaining() >= FBUtilities.MAX_UNSIGNED_SHORT) - throw new InvalidRequestException(String.format("Cannot index value of size %d for index %s on %s.%s(%s) (maximum allowed size=%d)", - value.remaining(), getIndexName(), baseKeyspace(), baseTable(), columnDef.name, FBUtilities.MAX_UNSIGNED_SHORT)); - } - - @Override - public String toString() - { - return String.format("%s(%s)", baseTable(), columnDef.name); - } -} diff --git a/src/java/org/apache/cassandra/db/index/PerColumnSecondaryIndex.java b/src/java/org/apache/cassandra/db/index/PerColumnSecondaryIndex.java deleted file mode 100644 index 897aa9c117..0000000000 --- a/src/java/org/apache/cassandra/db/index/PerColumnSecondaryIndex.java +++ /dev/null @@ -1,116 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.util.Iterator; - -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.utils.concurrent.OpOrder; - -/** - * Base class for Secondary indexes that implement a unique index per column - * - */ -public abstract class PerColumnSecondaryIndex extends SecondaryIndex -{ - /** - * Called when a column has been tombstoned or replaced. - * - * @param rowKey the underlying row key which is indexed - * @param col all the column info - */ - public abstract void delete(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec); - - /** - * Called when a column has been removed due to a cleanup operation. - */ - public abstract void deleteForCleanup(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec); - - /** - * For indexes on the primary key, index the given PK. - */ - public void maybeIndex(ByteBuffer partitionKey, Clustering clustering, long timestamp, int ttl, OpOrder.Group opGroup, int nowInSec) - { - } - - /** - * For indexes on the primary key, delete the given PK. - */ - public void maybeDelete(ByteBuffer partitionKey, Clustering clustering, DeletionTime deletion, OpOrder.Group opGroup) - { - } - - /** - * insert a column to the index - * - * @param rowKey the underlying row key which is indexed - * @param col all the column info - */ - public abstract void insert(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup); - - /** - * update a column from the index - * - * @param rowKey the underlying row key which is indexed - * @param oldCol the previous column info - * @param col all the column info - */ - public abstract void update(ByteBuffer rowKey, Clustering clustering, Cell oldCell, Cell cell, OpOrder.Group opGroup, int nowInSec); - - protected boolean indexPrimaryKeyColumn() - { - return false; - } - - public void indexRow(DecoratedKey key, Row row, OpOrder.Group opGroup, int nowInSec) - { - Clustering clustering = row.clustering(); - if (indexPrimaryKeyColumn()) - { - // Same as in AtomicBTreePartition.maybeIndexPrimaryKeyColumn - long timestamp = row.primaryKeyLivenessInfo().timestamp(); - int ttl = row.primaryKeyLivenessInfo().ttl(); - - for (Cell cell : row.cells()) - { - if (cell.isLive(nowInSec) && cell.timestamp() > timestamp) - { - timestamp = cell.timestamp(); - ttl = cell.ttl(); - } - } - maybeIndex(key.getKey(), clustering, timestamp, ttl, opGroup, nowInSec); - } - - for (Cell cell : row.cells()) - { - if (!indexes(cell.column())) - continue; - - if (cell.isLive(nowInSec)) - insert(key.getKey(), clustering, cell, opGroup); - } - } - - public String getNameForSystemKeyspace(ByteBuffer column) - { - return getIndexName(); - } -} diff --git a/src/java/org/apache/cassandra/db/index/PerRowSecondaryIndex.java b/src/java/org/apache/cassandra/db/index/PerRowSecondaryIndex.java deleted file mode 100644 index 502b213f78..0000000000 --- a/src/java/org/apache/cassandra/db/index/PerRowSecondaryIndex.java +++ /dev/null @@ -1,56 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.nio.charset.CharacterCodingException; - -import org.apache.cassandra.db.rows.UnfilteredRowIterator; -import org.apache.cassandra.db.partitions.PartitionUpdate; -import org.apache.cassandra.utils.concurrent.OpOrder; -import org.apache.cassandra.utils.ByteBufferUtil; - -/** - * Base class for Secondary indexes that implement a unique index per row - */ -public abstract class PerRowSecondaryIndex extends SecondaryIndex -{ - /** - * Index the given partition. - */ - public abstract void index(ByteBuffer key, UnfilteredRowIterator atoms); - - /** - * cleans up deleted columns from cassandra cleanup compaction - * - * @param key - */ - public abstract void delete(ByteBuffer key, OpOrder.Group opGroup); - - public String getNameForSystemKeyspace(ByteBuffer columnName) - { - try - { - return getIndexName()+ByteBufferUtil.string(columnName); - } - catch (CharacterCodingException e) - { - throw new RuntimeException(e); - } - } -} diff --git a/src/java/org/apache/cassandra/db/index/SecondaryIndex.java b/src/java/org/apache/cassandra/db/index/SecondaryIndex.java deleted file mode 100644 index 430211215f..0000000000 --- a/src/java/org/apache/cassandra/db/index/SecondaryIndex.java +++ /dev/null @@ -1,429 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.util.*; -import java.util.concurrent.ConcurrentHashMap; -import java.util.concurrent.Future; -import java.util.concurrent.FutureTask; - -import com.google.common.base.Objects; -import org.apache.commons.lang3.StringUtils; -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; - -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.cql3.ColumnIdentifier; -import org.apache.cassandra.cql3.Operator; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.compaction.CompactionManager; -import org.apache.cassandra.db.index.composites.CompositesIndex; -import org.apache.cassandra.db.index.keys.KeysIndex; -import org.apache.cassandra.db.lifecycle.SSTableSet; -import org.apache.cassandra.db.lifecycle.View; -import org.apache.cassandra.db.marshal.AbstractType; -import org.apache.cassandra.dht.LocalPartitioner; -import org.apache.cassandra.exceptions.ConfigurationException; -import org.apache.cassandra.io.sstable.format.SSTableReader; -import org.apache.cassandra.exceptions.InvalidRequestException; -import org.apache.cassandra.io.sstable.ReducingKeyIterator; -import org.apache.cassandra.schema.IndexMetadata; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.concurrent.Refs; - -/** - * Abstract base class for different types of secondary indexes. - * - * Do not extend this directly, please pick from PerColumnSecondaryIndex or PerRowSecondaryIndex - */ -public abstract class SecondaryIndex -{ - protected static final Logger logger = LoggerFactory.getLogger(SecondaryIndex.class); - - public static final String CUSTOM_INDEX_OPTION_NAME = "class_name"; - - /** - * The name of the option used to specify that the index is on the collection keys. - */ - public static final String INDEX_KEYS_OPTION_NAME = "index_keys"; - - /** - * The name of the option used to specify that the index is on the collection values. - */ - public static final String INDEX_VALUES_OPTION_NAME = "index_values"; - - /** - * The name of the option used to specify that the index is on the collection (map) entries. - */ - public static final String INDEX_ENTRIES_OPTION_NAME = "index_keys_and_values"; - - /** - * Base CF that has many indexes - */ - protected ColumnFamilyStore baseCfs; - - - /** - * The column definitions which this index is responsible for - */ - protected final Set columnDefs = Collections.newSetFromMap(new ConcurrentHashMap()); - - protected IndexMetadata indexMetadata; - - /** - * Perform any initialization work - */ - public abstract void init(); - - /** - * Reload an existing index following a change to its configuration, - * or that of the indexed column(s). Differs from init() in that we expect - * expect new resources (such as CFS for a KEYS index) to be created by - * init() but not here - */ - public abstract void reload(); - - /** - * Validates the index_options passed in the IndexMetadata - * @throws ConfigurationException - */ - public abstract void validateOptions(CFMetaData baseCfm, IndexMetadata def) throws ConfigurationException; - - /** - * @return The name of the index - */ - abstract public String getIndexName(); - - /** - * All internal 2ndary indexes will return "_internal_" for this. Custom - * 2ndary indexes will return their class name. This only matter for - * SecondaryIndexManager.groupByIndexType. - */ - String indexTypeForGrouping() - { - // Our internal indexes overwrite this - return getClass().getCanonicalName(); - } - - /** - * Return the unique name for this index and column - * to be stored in the SystemKeyspace that tracks if each column is built - * - * @param columnName the name of the column - * @return the unique name - */ - abstract public String getNameForSystemKeyspace(ByteBuffer columnName); - - /** - * Checks if the index for specified column is fully built - * - * @param columnName the column - * @return true if the index is fully built - */ - public boolean isIndexBuilt(ByteBuffer columnName) - { - return SystemKeyspace.isIndexBuilt(baseCfs.keyspace.getName(), getNameForSystemKeyspace(columnName)); - } - - public void setIndexBuilt() - { - for (ColumnDefinition columnDef : columnDefs) - SystemKeyspace.setIndexBuilt(baseCfs.keyspace.getName(), getNameForSystemKeyspace(columnDef.name.bytes)); - } - - public void setIndexRemoved() - { - for (ColumnDefinition columnDef : columnDefs) - SystemKeyspace.setIndexRemoved(baseCfs.keyspace.getName(), getNameForSystemKeyspace(columnDef.name.bytes)); - } - - /** - * Called at query time - * Creates a implementation specific searcher instance for this index type - * @param columns the list of columns which belong to this index type - * @return the secondary index search impl - */ - protected abstract SecondaryIndexSearcher createSecondaryIndexSearcher(Set columns); - - /** - * Forces this indexes' in memory data to disk - */ - public abstract void forceBlockingFlush(); - - /** - * Allow access to the underlying column family store if there is one - * @return the underlying column family store or null - */ - public abstract ColumnFamilyStore getIndexCfs(); - - - /** - * Delete all files and references to this index - * @param columnName the indexed column to remove - */ - public abstract void removeIndex(ByteBuffer columnName); - - /** - * Remove the index and unregisters this index's mbean if one exists - */ - public abstract void invalidate(); - - /** - * Truncate all the data from the current index - * - * @param truncatedAt The truncation timestamp, all data before that timestamp should be rejected. - */ - public abstract void truncateBlocking(long truncatedAt); - - /** - * Builds the index using the data in the underlying CFS - * Blocks till it's complete - */ - protected void buildIndexBlocking() - { - logger.info(String.format("Submitting index build of %s for data in %s", - getIndexName(), StringUtils.join(baseCfs.getSSTables(SSTableSet.CANONICAL), ", "))); - - try (Refs sstables = baseCfs.selectAndReference(View.select(SSTableSet.CANONICAL)).refs) - { - SecondaryIndexBuilder builder = new SecondaryIndexBuilder(baseCfs, - Collections.singleton(getIndexName()), - new ReducingKeyIterator(sstables)); - Future future = CompactionManager.instance.submitIndexBuild(builder); - FBUtilities.waitOnFuture(future); - forceBlockingFlush(); - setIndexBuilt(); - } - logger.info("Index build of {} complete", getIndexName()); - } - - - /** - * Builds the index using the data in the underlying CF, non blocking - * - * - * @return A future object which the caller can block on (optional) - */ - public Future buildIndexAsync() - { - // if we're just linking in the index to indexedColumns on an already-built index post-restart, we're done - boolean allAreBuilt = true; - for (ColumnDefinition cdef : columnDefs) - { - if (!SystemKeyspace.isIndexBuilt(baseCfs.keyspace.getName(), getNameForSystemKeyspace(cdef.name.bytes))) - { - allAreBuilt = false; - break; - } - } - - if (allAreBuilt) - return null; - - // build it asynchronously; addIndex gets called by CFS open and schema update, neither of which - // we want to block for a long period. (actual build is serialized on CompactionManager.) - Runnable runnable = new Runnable() - { - public void run() - { - baseCfs.forceBlockingFlush(); - buildIndexBlocking(); - } - }; - FutureTask f = new FutureTask(runnable, null); - - new Thread(f, "Creating index: " + getIndexName()).start(); - return f; - } - - public ColumnFamilyStore getBaseCfs() - { - return baseCfs; - } - - private void setBaseCfs(ColumnFamilyStore baseCfs) - { - this.baseCfs = baseCfs; - } - - public Set getColumnDefs() - { - return columnDefs; - } - - void setIndexMetadata(IndexMetadata indexDef) - { - this.indexMetadata = indexDef; - for (ColumnIdentifier col : indexDef.columns) - this.columnDefs.add(baseCfs.metadata.getColumnDefinition(col)); - } - - void addColumnDef(ColumnDefinition columnDef) - { - columnDefs.add(columnDef); - } - - void removeColumnDef(ByteBuffer name) - { - Iterator it = columnDefs.iterator(); - while (it.hasNext()) - { - if (it.next().name.bytes.equals(name)) - it.remove(); - } - } - - /** Returns true if the index supports lookups for the given operator, false otherwise. */ - public boolean supportsOperator(Operator operator) - { - return operator == Operator.EQ; - } - - /** - * Returns the decoratedKey for a column value. Assumes an index CFS is present. - * @param value column value - * @return decorated key - */ - public DecoratedKey getIndexKeyFor(ByteBuffer value) - { - return getIndexCfs().decorateKey(value); - } - - /** - * Returns true if the provided column is indexed by this secondary index. - * - * The default implementation checks whether the name is one the columnDef name, - * but this should be overriden but subclass if needed. - */ - public abstract boolean indexes(ColumnDefinition column); - - /** - * This is the primary way to create a secondary index instance for a CF column. - * It will validate the index_options before initializing. - * - * @param baseCfs the source of data for the Index - * @param indexDef the meta information about this index (index_type, index_options, name, etc...) - * - * @return The secondary index instance for this column - * @throws ConfigurationException - */ - public static SecondaryIndex createInstance(ColumnFamilyStore baseCfs, - IndexMetadata indexDef) throws ConfigurationException - { - SecondaryIndex index = uninitializedInstance(baseCfs.metadata, indexDef); - index.validateOptions(baseCfs.metadata, indexDef); - index.setBaseCfs(baseCfs); - index.setIndexMetadata(indexDef); - - return index; - } - - public static void validate(CFMetaData baseMetadata, - IndexMetadata indexDef) throws ConfigurationException - { - SecondaryIndex index = uninitializedInstance(baseMetadata, indexDef); - index.validateOptions(baseMetadata, indexDef); - } - - private static SecondaryIndex uninitializedInstance(CFMetaData baseMetadata, - IndexMetadata indexDef) throws ConfigurationException - { - if (indexDef.isKeys()) - { - return new KeysIndex(); - } - else if (indexDef.isComposites()) - { - return CompositesIndex.create(indexDef, baseMetadata); - } - else if (indexDef.isCustom()) - { - assert indexDef.options != null; - String class_name = indexDef.options.get(CUSTOM_INDEX_OPTION_NAME); - assert class_name != null; - try - { - return (SecondaryIndex) Class.forName(class_name).newInstance(); - } - catch (Exception e) - { - throw new RuntimeException(e); - } - } - - throw new AssertionError("Unknown index type: " + indexDef.name); - } - - public abstract void validate(DecoratedKey partitionKey) throws InvalidRequestException; - public abstract void validate(Clustering clustering) throws InvalidRequestException; - public abstract void validate(ByteBuffer cellValue, CellPath path) throws InvalidRequestException; - - public abstract long estimateResultRows(); - - protected String baseKeyspace() - { - return baseCfs.metadata.ksName; - } - - protected String baseTable() - { - return baseCfs.metadata.cfName; - } - - /** - * Create the index metadata for the index on a given column of a given table. - */ - public static CFMetaData newIndexMetadata(CFMetaData baseMetadata, IndexMetadata def) - { - return newIndexMetadata(baseMetadata, def, def.indexedColumn(baseMetadata).type); - } - - /** - * Create the index metadata for the index on a given column of a given table. - */ - static CFMetaData newIndexMetadata(CFMetaData baseMetadata, IndexMetadata def, AbstractType comparator) - { - assert !def.isCustom(); - - CFMetaData.Builder builder = CFMetaData.Builder.create(baseMetadata.ksName, baseMetadata.indexColumnFamilyName(def)) - .withId(baseMetadata.cfId) - .withPartitioner(new LocalPartitioner(comparator)) - .addPartitionKey(def.indexedColumn(baseMetadata).name, comparator); - - if (def.isComposites()) - { - CompositesIndex.addIndexClusteringColumns(builder, baseMetadata, def); - } - else - { - assert def.isKeys(); - KeysIndex.addIndexClusteringColumns(builder, baseMetadata); - } - - return builder.build().reloadIndexMetadataProperties(baseMetadata); - } - - @Override - public String toString() - { - return Objects.toStringHelper(this).add("columnDefs", columnDefs).toString(); - } -} diff --git a/src/java/org/apache/cassandra/db/index/SecondaryIndexManager.java b/src/java/org/apache/cassandra/db/index/SecondaryIndexManager.java deleted file mode 100644 index 996c7301a0..0000000000 --- a/src/java/org/apache/cassandra/db/index/SecondaryIndexManager.java +++ /dev/null @@ -1,792 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.util.*; -import java.util.concurrent.ConcurrentHashMap; -import java.util.concurrent.ConcurrentMap; -import java.util.concurrent.ConcurrentNavigableMap; -import java.util.concurrent.ConcurrentSkipListMap; -import java.util.concurrent.Future; - -import org.apache.commons.lang3.StringUtils; -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; - -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.partitions.PartitionUpdate; -import org.apache.cassandra.db.compaction.CompactionManager; -import org.apache.cassandra.exceptions.InvalidRequestException; -import org.apache.cassandra.io.sstable.ReducingKeyIterator; -import org.apache.cassandra.io.sstable.format.SSTableReader; -import org.apache.cassandra.schema.IndexMetadata; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.concurrent.OpOrder; - -/** - * Manages all the indexes associated with a given CFS - * Different types of indexes can be created across the same CF - */ -public class SecondaryIndexManager -{ - private static final Logger logger = LoggerFactory.getLogger(SecondaryIndexManager.class); - - public static final Updater nullUpdater = new Updater() - { - public void maybeIndex(Clustering clustering, long timestamp, int ttl, DeletionTime deletion) {} - public void insert(Clustering clustering, Cell cell) {} - public void update(Clustering clustering, Cell oldCell, Cell cell) {} - public void remove(Clustering clustering, Cell current) {} - - public void updateRowLevelIndexes() {} - }; - - /** - * Organizes the indexes by column name - */ - private final ConcurrentNavigableMap indexesByColumn; - - - /** - * Keeps a single instance of a SecondaryIndex for many columns when the index type - * has isRowLevelIndex() == true - * - * This allows updates to happen to an entire row at once - */ - private final ConcurrentMap, SecondaryIndex> rowLevelIndexMap; - - - /** - * Keeps all secondary index instances, either per-column or per-row - */ - private final Set allIndexes; - - - /** - * The underlying column family containing the source data for these indexes - */ - public final ColumnFamilyStore baseCfs; - - public SecondaryIndexManager(ColumnFamilyStore baseCfs) - { - indexesByColumn = new ConcurrentSkipListMap<>(); - rowLevelIndexMap = new ConcurrentHashMap<>(); - allIndexes = Collections.newSetFromMap(new ConcurrentHashMap()); - - this.baseCfs = baseCfs; - } - - /** - * Drops and adds new indexes associated with the underlying CF - */ - public void reload() - { - // figure out what needs to be added and dropped. - // future: if/when we have modifiable settings for secondary indexes, - // they'll need to be handled here. - Collection indexedColumnNames = indexesByColumn.keySet(); - for (ByteBuffer indexedColumn : indexedColumnNames) - { - ColumnDefinition def = baseCfs.metadata.getColumnDefinition(indexedColumn); - if (def == null || !baseCfs.metadata.getIndexes().get(def).isPresent()) - removeIndexedColumn(indexedColumn); - } - - // TODO: allow all ColumnDefinition type - for (IndexMetadata indexDef : baseCfs.metadata.getIndexes()) - { - if (!indexedColumnNames.contains(indexDef.indexedColumn(baseCfs.metadata).name.bytes)) - addIndexedColumn(indexDef); - } - - for (SecondaryIndex index : allIndexes) - index.reload(); - } - - public Set allIndexesNames() - { - Set names = new HashSet<>(allIndexes.size()); - for (SecondaryIndex index : allIndexes) - names.add(index.getIndexName()); - return names; - } - - public Set perColumnIndexes() - { - Set s = new HashSet<>(); - for (SecondaryIndex index : allIndexes) - if (index instanceof PerColumnSecondaryIndex) - s.add((PerColumnSecondaryIndex)index); - return s; - } - - public Set perRowIndexes() - { - Set s = new HashSet<>(); - for (SecondaryIndex index : allIndexes) - if (index instanceof PerRowSecondaryIndex) - s.add((PerRowSecondaryIndex)index); - return s; - } - - /** - * Does a full, blocking rebuild of the indexes specified by columns from the sstables. - * Does nothing if columns is empty. - * - * Caller must acquire and release references to the sstables used here. - * - * @param sstables the data to build from - * @param idxNames the list of columns to index, ordered by comparator - */ - public void maybeBuildSecondaryIndexes(Collection sstables, Set idxNames) - { - idxNames = filterByColumn(idxNames); - if (idxNames.isEmpty()) - return; - - logger.info(String.format("Submitting index build of %s for data in %s", - idxNames, StringUtils.join(sstables, ", "))); - - SecondaryIndexBuilder builder = new SecondaryIndexBuilder(baseCfs, idxNames, new ReducingKeyIterator(sstables)); - Future future = CompactionManager.instance.submitIndexBuild(builder); - FBUtilities.waitOnFuture(future); - - flushIndexesBlocking(); - - logger.info("Index build of {} complete", idxNames); - } - - public boolean indexes(ColumnDefinition column) - { - for (SecondaryIndex index : allIndexes) - if (index.indexes(column)) - return true; - return false; - } - - private Set indexFor(ColumnDefinition column) - { - Set matching = null; - for (SecondaryIndex index : allIndexes) - { - if (index.indexes(column)) - { - if (matching == null) - matching = new HashSet<>(); - matching.add(index); - } - } - return matching == null ? Collections.emptySet() : matching; - } - - /** - * Removes a existing index - * @param column the indexed column to remove - */ - public void removeIndexedColumn(ByteBuffer column) - { - SecondaryIndex index = indexesByColumn.remove(column); - - if (index == null) - return; - - // Remove this column from from row level index map as well as all indexes set - if (index instanceof PerRowSecondaryIndex) - { - index.removeColumnDef(column); - - // If no columns left remove from row level lookup as well as all indexes set - if (index.getColumnDefs().isEmpty()) - { - allIndexes.remove(index); - rowLevelIndexMap.remove(index.getClass()); - } - } - else - { - allIndexes.remove(index); - } - - index.removeIndex(column); - SystemKeyspace.setIndexRemoved(baseCfs.metadata.ksName, index.getNameForSystemKeyspace(column)); - } - - /** - * Adds and builds a index for a column - * @param indexDef the index metadata - * @return a future which the caller can optionally block on signaling the index is built - */ - public synchronized Future addIndexedColumn(IndexMetadata indexDef) - { - ColumnDefinition cdef = indexDef.indexedColumn(baseCfs.metadata); - if (indexesByColumn.containsKey(cdef.name.bytes)) - return null; - - SecondaryIndex index = SecondaryIndex.createInstance(baseCfs, indexDef); - - // Keep a single instance of the index per-cf for row level indexes - // since we want all columns to be under the index - if (index instanceof PerRowSecondaryIndex) - { - SecondaryIndex currentIndex = rowLevelIndexMap.get(index.getClass()); - - if (currentIndex == null) - { - rowLevelIndexMap.put(index.getClass(), index); - index.init(); - } - else - { - index = currentIndex; - index.setIndexMetadata(indexDef); - logger.info("Creating new index : {}",indexDef.name); - } - } - else - { - // TODO: We sould do better than throw a RuntimeException - if (indexDef.isCustom() && index instanceof AbstractSimplePerColumnSecondaryIndex) - throw new RuntimeException("Cannot use a subclass of AbstractSimplePerColumnSecondaryIndex as a CUSTOM index, as they assume they are CFS backed"); - index.init(); - } - - // link in indexedColumns. this means that writes will add new data to - // the index immediately, - // so we don't have to lock everything while we do the build. it's up to - // the operator to wait - // until the index is actually built before using in queries. - indexesByColumn.put(cdef.name.bytes, index); - - // Add to all indexes set: - allIndexes.add(index); - - // if we're just linking in the index to indexedColumns on an - // already-built index post-restart, we're done - if (index.isIndexBuilt(cdef.name.bytes)) - return null; - - return index.buildIndexAsync(); - } - - /** - * - * @param column the name of indexes column - * @return the index - */ - public SecondaryIndex getIndexForColumn(ColumnDefinition column) - { - return indexesByColumn.get(column.name.bytes); - } - - /** - * Remove the index - */ - public void invalidate() - { - for (SecondaryIndex index : allIndexes) - index.invalidate(); - } - - /** - * Flush all indexes to disk - */ - public void flushIndexesBlocking() - { - // despatch flushes for all CFS backed indexes - List> wait = new ArrayList<>(); - synchronized (baseCfs.getTracker()) - { - for (SecondaryIndex index : allIndexes) - if (index.getIndexCfs() != null) - wait.add(index.getIndexCfs().forceFlush()); - } - - // blockingFlush any non-CFS-backed indexes - for (SecondaryIndex index : allIndexes) - if (index.getIndexCfs() == null) - index.forceBlockingFlush(); - - // wait for the CFS-backed index flushes to complete - FBUtilities.waitOnFutures(wait); - } - - /** - * @return all built indexes (ready to use) - */ - public List getBuiltIndexes() - { - List indexList = new ArrayList<>(); - - for (Map.Entry entry : indexesByColumn.entrySet()) - { - SecondaryIndex index = entry.getValue(); - - if (index.isIndexBuilt(entry.getKey())) - indexList.add(entry.getValue().getIndexName()); - } - - return indexList; - } - - /** - * @return all CFS from indexes which use a backing CFS internally (KEYS) - */ - public Set getIndexesBackedByCfs() - { - Set cfsList = new HashSet<>(); - - for (SecondaryIndex index: allIndexes) - { - ColumnFamilyStore cfs = index.getIndexCfs(); - if (cfs != null) - cfsList.add(cfs); - } - - return cfsList; - } - - /** - * @return all indexes which do *not* use a backing CFS internally - */ - public Set getIndexesNotBackedByCfs() - { - // we use identity map because per row indexes use same instance across many columns - Set indexes = Collections.newSetFromMap(new IdentityHashMap()); - for (SecondaryIndex index: allIndexes) - if (index.getIndexCfs() == null) - indexes.add(index); - return indexes; - } - - /** - * @return all of the secondary indexes without distinction to the (non-)backed by secondary ColumnFamilyStore. - */ - public Set getIndexes() - { - return allIndexes; - } - - /** - * @return if there are ANY indexes for this table.. - */ - public boolean hasIndexes() - { - return !indexesByColumn.isEmpty(); - } - - /** - * When building an index against existing data, add the given partition to the index - */ - public void indexPartition(UnfilteredRowIterator partition, OpOrder.Group opGroup, Set allIndexes, int nowInSec) - { - Set perRowIndexes = perRowIndexes(); - Set perColumnIndexes = perColumnIndexes(); - - if (!perRowIndexes.isEmpty()) - { - // TODO: This is passing the same partition iterator to all perRow index, which means this only - // work if there is only one of them. We should change the API so it doesn't work directly on the - // partition, but rather on individual rows, so we can do a single iteration on the partition in this - // method and pass the rows to index to all indexes. - - // Update entire partition only once per row level index - Set> appliedRowLevelIndexes = new HashSet<>(); - for (PerRowSecondaryIndex index : perRowIndexes) - { - if (appliedRowLevelIndexes.add(index.getClass())) - ((PerRowSecondaryIndex)index).index(partition.partitionKey().getKey(), partition); - } - } - - if (!perColumnIndexes.isEmpty()) - { - DecoratedKey key = partition.partitionKey(); - - if (!partition.staticRow().isEmpty()) - { - for (PerColumnSecondaryIndex index : perColumnIndexes) - index.indexRow(key, partition.staticRow(), opGroup, nowInSec); - } - - try (RowIterator filtered = UnfilteredRowIterators.filter(partition, nowInSec)) - { - while (filtered.hasNext()) - { - Row row = filtered.next(); - for (PerColumnSecondaryIndex index : perColumnIndexes) - index.indexRow(key, row, opGroup, nowInSec); - } - } - } - } - - /** - * Delete all data from all indexes for this partition. For when cleanup rips a partition out entirely. - */ - public void deleteFromIndexes(UnfilteredRowIterator partition, OpOrder.Group opGroup, int nowInSec) - { - ByteBuffer key = partition.partitionKey().getKey(); - - for (PerRowSecondaryIndex index : perRowIndexes()) - index.delete(key, opGroup); - - Set indexes = perColumnIndexes(); - - while (partition.hasNext()) - { - Unfiltered unfiltered = partition.next(); - if (unfiltered.kind() != Unfiltered.Kind.ROW) - continue; - - Row row = (Row) unfiltered; - Clustering clustering = row.clustering(); - if (!row.deletion().isLive()) - for (PerColumnSecondaryIndex index : indexes) - index.maybeDelete(key, clustering, row.deletion(), opGroup); - - for (Cell cell : row.cells()) - { - for (PerColumnSecondaryIndex index : indexes) - { - if (!index.indexes(cell.column())) - continue; - - ((PerColumnSecondaryIndex) index).deleteForCleanup(key, clustering, cell, opGroup, nowInSec); - } - } - } - } - - /** - * This helper acts as a closure around the indexManager and updated data - * to ensure that down in Memtable's ColumnFamily implementation, the index - * can get updated. - */ - public Updater updaterFor(PartitionUpdate update, OpOrder.Group opGroup, int nowInSec) - { - return (indexesByColumn.isEmpty() && rowLevelIndexMap.isEmpty()) - ? nullUpdater - : new StandardUpdater(update, opGroup, nowInSec); - } - - /** - * Updated closure with only the modified row key. - */ - public Updater gcUpdaterFor(DecoratedKey key, int nowInSec) - { - return new GCUpdater(key, nowInSec); - } - - /** - * Get a list of IndexSearchers from the union of expression index types - * @param command the query - * @return the searchers needed to query the index - */ - public List getIndexSearchersFor(ReadCommand command) - { - Map> groupByIndexType = new HashMap<>(); - - //Group columns by type - for (RowFilter.Expression e : command.rowFilter()) - { - SecondaryIndex index = getIndexForColumn(e.column()); - - if (index == null || !index.supportsOperator(e.operator())) - continue; - - Set columns = groupByIndexType.get(index.indexTypeForGrouping()); - - if (columns == null) - { - columns = new HashSet<>(); - groupByIndexType.put(index.indexTypeForGrouping(), columns); - } - - columns.add(e.column()); - } - - List indexSearchers = new ArrayList<>(groupByIndexType.size()); - - //create searcher per type - for (Set column : groupByIndexType.values()) - indexSearchers.add(getIndexForColumn(column.iterator().next()).createSecondaryIndexSearcher(column)); - - return indexSearchers; - } - - public SecondaryIndexSearcher getBestIndexSearcherFor(ReadCommand command) - { - List indexSearchers = getIndexSearchersFor(command); - - if (indexSearchers.isEmpty()) - return null; - - SecondaryIndexSearcher mostSelective = null; - long bestEstimate = Long.MAX_VALUE; - for (SecondaryIndexSearcher searcher : indexSearchers) - { - SecondaryIndex highestSelectivityIndex = searcher.highestSelectivityIndex(command.rowFilter()); - long estimate = highestSelectivityIndex.estimateResultRows(); - if (estimate <= bestEstimate) - { - bestEstimate = estimate; - mostSelective = searcher; - } - } - return mostSelective; - } - - /** - * Validates an union of expression index types. It will throw an {@link InvalidRequestException} if - * any of the expressions in the provided clause is not valid for its index implementation. - * @param filter the filter to check - * @throws org.apache.cassandra.exceptions.InvalidRequestException in case of validation errors - */ - public void validateFilter(RowFilter filter) throws InvalidRequestException - { - for (RowFilter.Expression expression : filter) - { - SecondaryIndex index = getIndexForColumn(expression.column()); - if (index != null && index.supportsOperator(expression.operator())) - expression.validateForIndexing(); - } - } - - public Set getIndexesByNames(Set idxNames) - { - Set result = new HashSet<>(); - for (SecondaryIndex index : allIndexes) - if (idxNames.contains(index.getIndexName())) - result.add(index); - return result; - } - - public SecondaryIndex getIndexByName(String idxName) - { - for (SecondaryIndex index : allIndexes) - if (idxName.equals(index.getIndexName())) - return index; - - return null; - } - - public void setIndexBuilt(Set idxNames) - { - for (SecondaryIndex index : getIndexesByNames(idxNames)) - index.setIndexBuilt(); - } - - public void setIndexRemoved(Set idxNames) - { - for (SecondaryIndex index : getIndexesByNames(idxNames)) - index.setIndexRemoved(); - } - - public void validate(DecoratedKey partitionKey) throws InvalidRequestException - { - for (SecondaryIndex index : perColumnIndexes()) - index.validate(partitionKey); - } - - public void validate(Clustering clustering) throws InvalidRequestException - { - for (SecondaryIndex index : perColumnIndexes()) - index.validate(clustering); - } - - public void validate(ColumnDefinition column, ByteBuffer value, CellPath path) throws InvalidRequestException - { - for (SecondaryIndex index : indexFor(column)) - index.validate(value, path); - } - - static boolean shouldCleanupOldValue(Cell oldCell, Cell newCell) - { - // If either the value or timestamp is different, then we - // should delete from the index. If not, then we can infer that - // at least one of the cells is an ExpiringColumn and that the - // difference is in the expiry time. In this case, we don't want to - // delete the old value from the index as the tombstone we insert - // will just hide the inserted value. - // Completely identical cells (including expiring columns with - // identical ttl & localExpirationTime) will not get this far due - // to the oldCell.equals(newCell) in StandardUpdater.update - return !oldCell.value().equals(newCell.value()) || oldCell.timestamp() != newCell.timestamp(); - } - - private Set filterByColumn(Set idxNames) - { - Set indexes = getIndexesByNames(idxNames); - Set filtered = new HashSet<>(idxNames.size()); - for (SecondaryIndex candidate : indexes) - { - for (ColumnDefinition column : baseCfs.metadata.allColumns()) - { - if (candidate.indexes(column)) - { - filtered.add(candidate.getIndexName()); - break; - } - } - } - return filtered; - } - - public static interface Updater - { - /** Called when a row with the provided clustering and row infos is inserted */ - public void maybeIndex(Clustering clustering, long timestamp, int ttl, DeletionTime deletion); - - /** called when constructing the index against pre-existing data */ - public void insert(Clustering clustering, Cell cell); - - /** called when updating the index from a memtable */ - public void update(Clustering clustering, Cell oldCell, Cell cell); - - /** called when lazy-updating the index during compaction (CASSANDRA-2897) */ - public void remove(Clustering clustering, Cell current); - - /** called after memtable updates are complete (CASSANDRA-5397) */ - public void updateRowLevelIndexes(); - } - - private final class GCUpdater implements Updater - { - private final DecoratedKey key; - private final int nowInSec; - - public GCUpdater(DecoratedKey key, int nowInSec) - { - this.key = key; - this.nowInSec = nowInSec; - } - - public void maybeIndex(Clustering clustering, long timestamp, int ttl, DeletionTime deletion) - { - throw new UnsupportedOperationException(); - } - - public void insert(Clustering clustering, Cell cell) - { - throw new UnsupportedOperationException(); - } - - public void update(Clustering clustering, Cell oldCell, Cell newCell) - { - throw new UnsupportedOperationException(); - } - - public void remove(Clustering clustering, Cell cell) - { - for (SecondaryIndex index : indexFor(cell.column())) - { - if (index instanceof PerColumnSecondaryIndex) - { - try (OpOrder.Group opGroup = baseCfs.keyspace.writeOrder.start()) - { - ((PerColumnSecondaryIndex) index).delete(key.getKey(), clustering, cell, opGroup, nowInSec); - } - } - } - } - - public void updateRowLevelIndexes() - { - for (SecondaryIndex index : rowLevelIndexMap.values()) - ((PerRowSecondaryIndex) index).index(key.getKey(), null); - } - } - - private final class StandardUpdater implements Updater - { - private final PartitionUpdate update; - private final OpOrder.Group opGroup; - private final int nowInSec; - - public StandardUpdater(PartitionUpdate update, OpOrder.Group opGroup, int nowInSec) - { - this.update = update; - this.opGroup = opGroup; - this.nowInSec = nowInSec; - } - - public void maybeIndex(Clustering clustering, long timestamp, int ttl, DeletionTime deletion) - { - for (PerColumnSecondaryIndex index : perColumnIndexes()) - { - if (timestamp != LivenessInfo.NO_TIMESTAMP) - index.maybeIndex(update.partitionKey().getKey(), clustering, timestamp, ttl, opGroup, nowInSec); - if (!deletion.isLive()) - index.maybeDelete(update.partitionKey().getKey(), clustering, deletion, opGroup); - } - } - - public void insert(Clustering clustering, Cell cell) - { - if (!cell.isLive(nowInSec)) - return; - - for (SecondaryIndex index : indexFor(cell.column())) - if (index instanceof PerColumnSecondaryIndex) - ((PerColumnSecondaryIndex) index).insert(update.partitionKey().getKey(), clustering, cell, opGroup); - } - - public void update(Clustering clustering, Cell oldCell, Cell cell) - { - if (oldCell.equals(cell)) - return; - - for (SecondaryIndex index : indexFor(cell.column())) - { - if (index instanceof PerColumnSecondaryIndex) - { - if (cell.isLive(nowInSec)) - { - ((PerColumnSecondaryIndex) index).update(update.partitionKey().getKey(), clustering, oldCell, cell, opGroup, nowInSec); - } - else - { - // Usually we want to delete the old value from the index, except when - // name/value/timestamp are all equal, but the columns themselves - // are not (as is the case when overwriting expiring columns with - // identical values and ttl) Then, we don't want to delete as the - // tombstone will hide the new value we just inserted; see CASSANDRA-7268 - if (shouldCleanupOldValue(oldCell, cell)) - ((PerColumnSecondaryIndex) index).delete(update.partitionKey().getKey(), clustering, oldCell, opGroup, nowInSec); - } - } - } - } - - public void remove(Clustering clustering, Cell cell) - { - throw new UnsupportedOperationException(); - } - - public void updateRowLevelIndexes() - { - for (SecondaryIndex index : rowLevelIndexMap.values()) - ((PerRowSecondaryIndex) index).index(update.partitionKey().getKey(), update.unfilteredIterator()); - } - - } -} diff --git a/src/java/org/apache/cassandra/db/index/SecondaryIndexSearcher.java b/src/java/org/apache/cassandra/db/index/SecondaryIndexSearcher.java deleted file mode 100644 index d4ca7077af..0000000000 --- a/src/java/org/apache/cassandra/db/index/SecondaryIndexSearcher.java +++ /dev/null @@ -1,244 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.util.*; - -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; - -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.filter.*; -import org.apache.cassandra.db.partitions.PartitionIterator; -import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.dht.AbstractBounds; -import org.apache.cassandra.tracing.Tracing; -import org.apache.cassandra.utils.btree.BTreeSet; -import org.apache.cassandra.utils.FBUtilities; - -public abstract class SecondaryIndexSearcher -{ - private static final Logger logger = LoggerFactory.getLogger(SecondaryIndexSearcher.class); - - protected final SecondaryIndexManager indexManager; - protected final Set columns; - protected final ColumnFamilyStore baseCfs; - - public SecondaryIndexSearcher(SecondaryIndexManager indexManager, Set columns) - { - this.indexManager = indexManager; - this.columns = columns; - this.baseCfs = indexManager.baseCfs; - } - - public SecondaryIndex highestSelectivityIndex(RowFilter filter) - { - RowFilter.Expression expr = highestSelectivityPredicate(filter, false); - return expr == null ? null : indexManager.getIndexForColumn(expr.column()); - } - - public RowFilter.Expression primaryClause(ReadCommand command) - { - return highestSelectivityPredicate(command.rowFilter(), false); - } - - @SuppressWarnings("resource") // Both the OpOrder and 'indexIter' are closed on exception, or through the closing of the result - // of this method. - public UnfilteredPartitionIterator search(ReadCommand command, ReadOrderGroup orderGroup) - { - RowFilter.Expression primary = highestSelectivityPredicate(command.rowFilter(), true); - assert primary != null; - - AbstractSimplePerColumnSecondaryIndex index = (AbstractSimplePerColumnSecondaryIndex)indexManager.getIndexForColumn(primary.column()); - assert index != null && index.getIndexCfs() != null; - - if (logger.isDebugEnabled()) - logger.debug("Most-selective indexed predicate is {}", primary); - - DecoratedKey indexKey = index.getIndexKeyFor(primary.getIndexValue()); - - UnfilteredRowIterator indexIter = queryIndex(index, indexKey, command, orderGroup); - try - { - return queryDataFromIndex(index, indexKey, UnfilteredRowIterators.filter(indexIter, command.nowInSec()), command, orderGroup); - } - catch (RuntimeException | Error e) - { - indexIter.close(); - throw e; - } - } - - private UnfilteredRowIterator queryIndex(AbstractSimplePerColumnSecondaryIndex index, DecoratedKey indexKey, ReadCommand command, ReadOrderGroup orderGroup) - { - ClusteringIndexFilter filter = makeIndexFilter(index, command); - CFMetaData indexMetadata = index.getIndexCfs().metadata; - return SinglePartitionReadCommand.create(indexMetadata, command.nowInSec(), indexKey, ColumnFilter.all(indexMetadata), filter) - .queryMemtableAndDisk(index.getIndexCfs(), orderGroup.indexReadOpOrderGroup()); - } - - private ClusteringIndexFilter makeIndexFilter(AbstractSimplePerColumnSecondaryIndex index, ReadCommand command) - { - if (command instanceof SinglePartitionReadCommand) - { - // Note: as yet there's no route to get here - a 2i query *always* uses a - // PartitionRangeReadCommand. This is here in preparation for coming changes - // in SelectStatement. - SinglePartitionReadCommand sprc = (SinglePartitionReadCommand)command; - ByteBuffer pk = sprc.partitionKey().getKey(); - ClusteringIndexFilter filter = sprc.clusteringIndexFilter(); - - if (filter instanceof ClusteringIndexNamesFilter) - { - NavigableSet requested = ((ClusteringIndexNamesFilter)filter).requestedRows(); - BTreeSet.Builder clusterings = BTreeSet.builder(index.getIndexComparator()); - for (Clustering c : requested) - clusterings.add(index.makeIndexClustering(pk, c, (Cell)null)); - return new ClusteringIndexNamesFilter(clusterings.build(), filter.isReversed()); - } - else - { - Slices requested = ((ClusteringIndexSliceFilter)filter).requestedSlices(); - Slices.Builder builder = new Slices.Builder(index.getIndexComparator()); - for (Slice slice : requested) - builder.add(index.makeIndexBound(pk, slice.start()), index.makeIndexBound(pk, slice.end())); - return new ClusteringIndexSliceFilter(builder.build(), filter.isReversed()); - } - } - else - { - - DataRange dataRange = ((PartitionRangeReadCommand)command).dataRange(); - AbstractBounds range = dataRange.keyRange(); - - Slice slice = Slice.ALL; - - /* - * XXX: If the range requested is a token range, we'll have to start at the beginning (and stop at the end) of - * the indexed row unfortunately (which will be inefficient), because we have no way to intuit the smallest possible - * key having a given token. A potential fix would be to actually store the token along the key in the indexed row. - */ - if (range.left instanceof DecoratedKey) - { - // the right hand side of the range may not be a DecoratedKey (for instance if we're paging), - // but if it is, we can optimise slightly by restricting the slice - if (range.right instanceof DecoratedKey) - { - - DecoratedKey startKey = (DecoratedKey) range.left; - DecoratedKey endKey = (DecoratedKey) range.right; - - Slice.Bound start = Slice.Bound.BOTTOM; - Slice.Bound end = Slice.Bound.TOP; - - /* - * For index queries over a range, we can't do a whole lot better than querying everything for the key range, though for - * slice queries where we can slightly restrict the beginning and end. - */ - if (!dataRange.isNamesQuery()) - { - ClusteringIndexSliceFilter startSliceFilter = ((ClusteringIndexSliceFilter) dataRange.clusteringIndexFilter(startKey)); - ClusteringIndexSliceFilter endSliceFilter = ((ClusteringIndexSliceFilter) dataRange.clusteringIndexFilter(endKey)); - - // We can't effectively support reversed queries when we have a range, so we don't support it - // (or through post-query reordering) and shouldn't get there. - assert !startSliceFilter.isReversed() && !endSliceFilter.isReversed(); - - Slices startSlices = startSliceFilter.requestedSlices(); - Slices endSlices = endSliceFilter.requestedSlices(); - - if (startSlices.size() > 0) - start = startSlices.get(0).start(); - - if (endSlices.size() > 0) - end = endSlices.get(endSlices.size() - 1).end(); - } - - slice = Slice.make(index.makeIndexBound(startKey.getKey(), start), - index.makeIndexBound(endKey.getKey(), end)); - } - else - { - // otherwise, just start the index slice from the key we do have - slice = Slice.make(index.makeIndexBound(((DecoratedKey)range.left).getKey(), Slice.Bound.BOTTOM), - Slice.Bound.TOP); - } - } - return new ClusteringIndexSliceFilter(Slices.with(index.getIndexComparator(), slice), false); - } - } - - protected abstract UnfilteredPartitionIterator queryDataFromIndex(AbstractSimplePerColumnSecondaryIndex index, - DecoratedKey indexKey, - RowIterator indexHits, - ReadCommand command, - ReadOrderGroup orderGroup); - - protected RowFilter.Expression highestSelectivityPredicate(RowFilter filter, boolean includeInTrace) - { - RowFilter.Expression best = null; - int bestMeanCount = Integer.MAX_VALUE; - Map candidates = new HashMap<>(); - - for (RowFilter.Expression expression : filter) - { - // skip columns belonging to a different index type - if (!columns.contains(expression.column())) - continue; - - SecondaryIndex index = indexManager.getIndexForColumn(expression.column()); - if (index == null || index.getIndexCfs() == null || !index.supportsOperator(expression.operator())) - continue; - - int columns = index.getIndexCfs().getMeanColumns(); - candidates.put(index, columns); - if (columns < bestMeanCount) - { - best = expression; - bestMeanCount = columns; - } - } - - if (includeInTrace) - { - if (best == null) - Tracing.trace("No applicable indexes found"); - else if (Tracing.isTracing()) - // pay for an additional threadlocal get() rather than build the strings unnecessarily - Tracing.trace("Candidate index mean cardinalities are {}. Scanning with {}.", FBUtilities.toString(candidates), indexManager.getIndexForColumn(best.column()).getIndexName()); - } - return best; - } - - /** - * Post-process the result of an index query. This is done by the coordinator node after it has reconciled - * the replica responses. - * - * @param command The {@code ReadCommand} use for the query. - * @param result The index query results to be post-processed - * @return The post-processed results - */ - public PartitionIterator postReconciliationProcessing(RowFilter filter, PartitionIterator result) - { - return result; - } -} diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndex.java b/src/java/org/apache/cassandra/db/index/composites/CompositesIndex.java deleted file mode 100644 index abc534444b..0000000000 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndex.java +++ /dev/null @@ -1,168 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index.composites; - -import java.nio.ByteBuffer; -import java.util.HashMap; -import java.util.Map; -import java.util.Set; - -import org.apache.cassandra.schema.IndexMetadata; -import org.apache.cassandra.utils.concurrent.OpOrder; -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.db.index.AbstractSimplePerColumnSecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexManager; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; -import org.apache.cassandra.db.marshal.CollectionType; -import org.apache.cassandra.exceptions.ConfigurationException; - -/** - * Base class for internal secondary indexes (this could be merged with AbstractSimplePerColumnSecondaryIndex). - */ -public abstract class CompositesIndex extends AbstractSimplePerColumnSecondaryIndex -{ - public static CompositesIndex create(IndexMetadata indexDef, CFMetaData baseMetadata) - { - ColumnDefinition cfDef = indexDef.indexedColumn(baseMetadata); - if (cfDef.type.isCollection() && cfDef.type.isMultiCell()) - { - switch (((CollectionType)cfDef.type).kind) - { - case LIST: - return new CompositesIndexOnCollectionValue(); - case SET: - return new CompositesIndexOnCollectionKey(); - case MAP: - if (indexDef.options.containsKey(SecondaryIndex.INDEX_KEYS_OPTION_NAME)) - return new CompositesIndexOnCollectionKey(); - else if (indexDef.options.containsKey(SecondaryIndex.INDEX_ENTRIES_OPTION_NAME)) - return new CompositesIndexOnCollectionKeyAndValue(); - else - return new CompositesIndexOnCollectionValue(); - } - } - - switch (cfDef.kind) - { - case CLUSTERING: - return new CompositesIndexOnClusteringKey(); - case REGULAR: - return new CompositesIndexOnRegular(); - case PARTITION_KEY: - return new CompositesIndexOnPartitionKey(); - //case COMPACT_VALUE: - // return new CompositesIndexOnCompactValue(); - } - throw new AssertionError(); - } - - public static void addIndexClusteringColumns(CFMetaData.Builder indexMetadata, CFMetaData baseMetadata, IndexMetadata indexDef) - { - ColumnDefinition cfDef = indexDef.indexedColumn(baseMetadata); - if (cfDef.type.isCollection() && cfDef.type.isMultiCell()) - { - CollectionType type = (CollectionType)cfDef.type; - if (type.kind == CollectionType.Kind.LIST - || (type.kind == CollectionType.Kind.MAP && indexDef.options.containsKey(SecondaryIndex.INDEX_VALUES_OPTION_NAME))) - { - CompositesIndexOnCollectionValue.addClusteringColumns(indexMetadata, baseMetadata, cfDef); - } - else - { - addGenericClusteringColumns(indexMetadata, baseMetadata, cfDef); - } - } - else if (cfDef.isClusteringColumn()) - { - CompositesIndexOnClusteringKey.addClusteringColumns(indexMetadata, baseMetadata, cfDef); - } - else - { - addGenericClusteringColumns(indexMetadata, baseMetadata, cfDef); - } - } - - protected static void addGenericClusteringColumns(CFMetaData.Builder indexMetadata, CFMetaData baseMetadata, ColumnDefinition columnDef) - { - indexMetadata.addClusteringColumn("partition_key", baseMetadata.partitioner.partitionOrdering()); - for (ColumnDefinition def : baseMetadata.clusteringColumns()) - indexMetadata.addClusteringColumn(def.name, def.type); - } - - public abstract IndexedEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry); - - public abstract boolean isStale(Row row, ByteBuffer indexValue, int nowInSec); - - public void delete(IndexedEntry entry, OpOrder.Group opGroup, int nowInSec) - { - Row row = BTreeRow.emptyDeletedRow(entry.indexClustering, new DeletionTime(entry.timestamp, nowInSec)); - PartitionUpdate upd = PartitionUpdate.singleRowUpdate(indexCfs.metadata, entry.indexValue, row); - indexCfs.apply(upd, SecondaryIndexManager.nullUpdater, opGroup, null); - - if (logger.isDebugEnabled()) - logger.debug("removed index entry for cleaned-up value {}:{}", entry.indexValue, upd); - } - - public SecondaryIndexSearcher createSecondaryIndexSearcher(Set columns) - { - return new CompositesSearcher(baseCfs.indexManager, columns); - } - - public void validateOptions(CFMetaData baseCfm, IndexMetadata indexMetadata) throws ConfigurationException - { - ColumnDefinition columnDef = indexMetadata.indexedColumn(baseCfm); - Map options = new HashMap(indexMetadata.options); - - // We used to have an option called "prefix_size" so skip it silently for backward compatibility sake. - options.remove("prefix_size"); - - if (columnDef.type.isCollection()) - { - options.remove(SecondaryIndex.INDEX_VALUES_OPTION_NAME); - options.remove(SecondaryIndex.INDEX_KEYS_OPTION_NAME); - options.remove(SecondaryIndex.INDEX_ENTRIES_OPTION_NAME); - } - - if (!options.isEmpty()) - throw new ConfigurationException("Unknown options provided for COMPOSITES index: " + options.keySet()); - } - - public static class IndexedEntry - { - public final DecoratedKey indexValue; - public final Clustering indexClustering; - public final long timestamp; - - public final ByteBuffer indexedKey; - public final Clustering indexedEntryClustering; - - public IndexedEntry(DecoratedKey indexValue, Clustering indexClustering, long timestamp, ByteBuffer indexedKey, Clustering indexedEntryClustering) - { - this.indexValue = indexValue; - this.indexClustering = indexClustering; - this.timestamp = timestamp; - this.indexedKey = indexedKey; - this.indexedEntryClustering = indexedEntryClustering; - } - } -} diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnClusteringKey.java b/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnClusteringKey.java deleted file mode 100644 index cd4aff9435..0000000000 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnClusteringKey.java +++ /dev/null @@ -1,137 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index.composites; - -import java.nio.ByteBuffer; -import java.util.List; - -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.utils.concurrent.OpOrder; - -/** - * Index on a CLUSTERING column definition. - * - * A cell indexed by this index will have the general form: - * ck_0 ... ck_n c_name : v - * where ck_i are the cluster keys, c_name the last component of the cell - * composite name (or second to last if collections are in use, but this - * has no impact) and v the cell value. - * - * Such a cell is always indexed by this index (or rather, it is indexed if - * n >= columnDef.componentIndex, which will always be the case in practice) - * and it will generate (makeIndexColumnName()) an index entry whose: - * - row key will be ck_i (getIndexedValue()) where i == columnDef.componentIndex. - * - cell name will - * rk ck_0 ... ck_{i-1} ck_{i+1} ck_n - * where rk is the row key of the initial cell and i == columnDef.componentIndex. - */ -public class CompositesIndexOnClusteringKey extends CompositesIndex -{ - public static void addClusteringColumns(CFMetaData.Builder indexMetadata, CFMetaData baseMetadata, ColumnDefinition columnDef) - { - indexMetadata.addClusteringColumn("partition_key", baseMetadata.partitioner.partitionOrdering()); - - List cks = baseMetadata.clusteringColumns(); - for (int i = 0; i < columnDef.position(); i++) - { - ColumnDefinition def = cks.get(i); - indexMetadata.addClusteringColumn(def.name, def.type); - } - for (int i = columnDef.position() + 1; i < cks.size(); i++) - { - ColumnDefinition def = cks.get(i); - indexMetadata.addClusteringColumn(def.name, def.type); - } - } - - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) - { - return clustering.get(columnDef.position()); - } - - protected CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path) - { - CBuilder builder = CBuilder.create(getIndexComparator()); - builder.add(rowKey); - for (int i = 0; i < Math.min(columnDef.position(), prefix.size()); i++) - builder.add(prefix.get(i)); - for (int i = columnDef.position() + 1; i < prefix.size(); i++) - builder.add(prefix.get(i)); - return builder; - } - - public IndexedEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) - { - int ckCount = baseCfs.metadata.clusteringColumns().size(); - - Clustering clustering = indexEntry.clustering(); - CBuilder builder = CBuilder.create(baseCfs.getComparator()); - for (int i = 0; i < columnDef.position(); i++) - builder.add(clustering.get(i + 1)); - - builder.add(indexedValue.getKey()); - - for (int i = columnDef.position() + 1; i < ckCount; i++) - builder.add(clustering.get(i)); - - return new IndexedEntry(indexedValue, clustering, indexEntry.primaryKeyLivenessInfo().timestamp(), clustering.get(0), builder.build()); - } - - @Override - protected boolean indexPrimaryKeyColumn() - { - return true; - } - - @Override - public boolean indexes(ColumnDefinition c) - { - // Actual indexing for this index type is done through maybeIndex - return false; - } - - public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) - { - return !data.hasLiveData(nowInSec); - } - - @Override - public void maybeIndex(ByteBuffer partitionKey, Clustering clustering, long timestamp, int ttl, OpOrder.Group opGroup, int nowInSec) - { - if (clustering != Clustering.STATIC_CLUSTERING && clustering.get(columnDef.position()) != null) - insert(partitionKey, clustering, null, LivenessInfo.create(indexCfs.metadata, timestamp, ttl, nowInSec), opGroup); - } - - @Override - public void maybeDelete(ByteBuffer partitionKey, Clustering clustering, DeletionTime deletion, OpOrder.Group opGroup) - { - if (clustering.get(columnDef.position()) != null && !deletion.isLive()) - delete(partitionKey, clustering, null, null, deletion, opGroup); - } - - @Override - public void delete(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - // We only know that one column of the CQL row has been updated/deleted, but we don't know if the - // full row has been deleted so we should not do anything. If it ends up that the whole row has - // been deleted, it will be eventually cleaned up on read because the entry will be detected stale. - } -} diff --git a/src/java/org/apache/cassandra/db/index/keys/KeysIndex.java b/src/java/org/apache/cassandra/db/index/keys/KeysIndex.java deleted file mode 100644 index 384b29d8f1..0000000000 --- a/src/java/org/apache/cassandra/db/index/keys/KeysIndex.java +++ /dev/null @@ -1,90 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index.keys; - -import java.nio.ByteBuffer; -import java.util.Set; - -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.index.AbstractSimplePerColumnSecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; -import org.apache.cassandra.exceptions.ConfigurationException; -import org.apache.cassandra.schema.IndexMetadata; -import org.apache.cassandra.utils.concurrent.OpOrder; - -/** - * Implements a secondary index for a column family using a second column family. - * The design uses inverted index http://en.wikipedia.org/wiki/Inverted_index. - * The row key is the indexed value. For example, if we're indexing a column named - * city, the index value of city is the row key. - * The column names are the keys of the records. To see a detailed example, please - * refer to wikipedia. - */ -public class KeysIndex extends AbstractSimplePerColumnSecondaryIndex -{ - public static void addIndexClusteringColumns(CFMetaData.Builder indexMetadata, CFMetaData baseMetadata) - { - indexMetadata.addClusteringColumn("partition_key", baseMetadata.partitioner.partitionOrdering()); - } - - @Override - public void indexRow(DecoratedKey key, Row row, OpOrder.Group opGroup, int nowInSec) - { - super.indexRow(key, row, opGroup, nowInSec); - - // This is used when building indexes, in particular when the index is first created. On thrift, this - // potentially means the column definition just got created, and so we need to check if's not a "dynamic" - // row that actually correspond to the index definition. - assert baseCfs.metadata.isCompactTable(); - if (!row.isStatic()) - { - Clustering clustering = row.clustering(); - if (clustering.get(0).equals(columnDef.name.bytes)) - { - Cell cell = row.getCell(baseCfs.metadata.compactValueColumn()); - if (cell != null && cell.isLive(nowInSec)) - insert(key.getKey(), clustering, cell, opGroup); - } - } - } - - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) - { - return cellValue; - } - - protected CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path) - { - CBuilder builder = CBuilder.create(getIndexComparator()); - builder.add(rowKey); - return builder; - } - - public SecondaryIndexSearcher createSecondaryIndexSearcher(Set columns) - { - return new KeysSearcher(baseCfs.indexManager, columns); - } - - public void validateOptions(CFMetaData baseCfm, IndexMetadata def) throws ConfigurationException - { - // no options used - } -} diff --git a/src/java/org/apache/cassandra/db/partitions/AtomicBTreePartition.java b/src/java/org/apache/cassandra/db/partitions/AtomicBTreePartition.java index 0b732928a0..d279a6bef9 100644 --- a/src/java/org/apache/cassandra/db/partitions/AtomicBTreePartition.java +++ b/src/java/org/apache/cassandra/db/partitions/AtomicBTreePartition.java @@ -19,7 +19,6 @@ package org.apache.cassandra.db.partitions; import java.nio.ByteBuffer; import java.util.ArrayList; -import java.util.Iterator; import java.util.List; import java.util.concurrent.atomic.AtomicIntegerFieldUpdater; import java.util.concurrent.atomic.AtomicReferenceFieldUpdater; @@ -27,17 +26,18 @@ import java.util.concurrent.atomic.AtomicReferenceFieldUpdater; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.index.SecondaryIndexManager; +import org.apache.cassandra.db.rows.EncodingStats; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.Rows; +import org.apache.cassandra.index.transactions.UpdateTransaction; +import org.apache.cassandra.utils.FBUtilities; +import org.apache.cassandra.utils.ObjectSizes; import org.apache.cassandra.utils.btree.BTree; import org.apache.cassandra.utils.btree.UpdateFunction; -import org.apache.cassandra.utils.ObjectSizes; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.concurrent.OpOrder; import org.apache.cassandra.utils.concurrent.Locks; -import org.apache.cassandra.utils.memory.MemtableAllocator; +import org.apache.cassandra.utils.concurrent.OpOrder; import org.apache.cassandra.utils.memory.HeapAllocator; -import static org.apache.cassandra.db.index.SecondaryIndexManager.Updater; +import org.apache.cassandra.utils.memory.MemtableAllocator; /** * A thread-safe and atomic Partition implementation. @@ -86,7 +86,7 @@ public class AtomicBTreePartition extends AbstractBTreePartition public AtomicBTreePartition(CFMetaData metadata, DecoratedKey partitionKey, MemtableAllocator allocator) { - // TODO: is this a potential bug? partition columns may be a subset if we alter columns while it's in memtable + // involved in potential bug? partition columns may be a subset if we alter columns while it's in memtable super(metadata, partitionKey, metadata.partitionColumns()); this.allocator = allocator; this.ref = EMPTY; @@ -108,11 +108,10 @@ public class AtomicBTreePartition extends AbstractBTreePartition * @return an array containing first the difference in size seen after merging the updates, and second the minimum * time detla between updates. */ - public long[] addAllWithSizeDelta(final PartitionUpdate update, OpOrder.Group writeOp, Updater indexer) + public long[] addAllWithSizeDelta(final PartitionUpdate update, OpOrder.Group writeOp, UpdateTransaction indexer) { RowUpdater updater = new RowUpdater(this, allocator, writeOp, indexer); DeletionInfo inputDeletionInfoCopy = null; - boolean monitorOwned = false; try { @@ -121,12 +120,21 @@ public class AtomicBTreePartition extends AbstractBTreePartition Locks.monitorEnterUnsafe(this); monitorOwned = true; } + + indexer.start(); + while (true) { Holder current = ref; updater.ref = current; updater.reset(); + if (!update.deletionInfo().isLive()) + indexer.onPartitionDeletion(update.deletionInfo().getPartitionDeletion()); + + if (update.deletionInfo().hasRanges()) + update.deletionInfo().rangeIterator(false).forEachRemaining(indexer::onRangeTombstone); + DeletionInfo deletionInfo; if (update.deletionInfo().mayModify(current.deletionInfo)) { @@ -150,7 +158,6 @@ public class AtomicBTreePartition extends AbstractBTreePartition if (tree != null && refUpdater.compareAndSet(this, current, new Holder(tree, deletionInfo, staticRow, newStats))) { - indexer.updateRowLevelIndexes(); updater.finish(); return new long[]{ updater.dataSize, updater.colUpdateTimeDelta }; } @@ -171,6 +178,7 @@ public class AtomicBTreePartition extends AbstractBTreePartition } finally { + indexer.commit(); if (monitorOwned) Locks.monitorExitUnsafe(this); } @@ -230,7 +238,7 @@ public class AtomicBTreePartition extends AbstractBTreePartition final AtomicBTreePartition updating; final MemtableAllocator allocator; final OpOrder.Group writeOp; - final Updater indexer; + final UpdateTransaction indexer; final int nowInSec; Holder ref; Row.Builder regularBuilder; @@ -240,7 +248,7 @@ public class AtomicBTreePartition extends AbstractBTreePartition final MemtableAllocator.DataReclaimer reclaimer; List inserted; // TODO: replace with walk of aborted BTree - private RowUpdater(AtomicBTreePartition updating, MemtableAllocator allocator, OpOrder.Group writeOp, Updater indexer) + private RowUpdater(AtomicBTreePartition updating, MemtableAllocator allocator, OpOrder.Group writeOp, UpdateTransaction indexer) { this.updating = updating; this.allocator = allocator; @@ -265,7 +273,7 @@ public class AtomicBTreePartition extends AbstractBTreePartition public Row apply(Row insert) { Row data = Rows.copy(insert, builder(insert.clustering())).build(); - insertIntoIndexes(data); + indexer.onInserted(insert); this.dataSize += data.dataSize(); this.heapSize += data.unsharedHeapSizeExcludingData(); @@ -280,10 +288,12 @@ public class AtomicBTreePartition extends AbstractBTreePartition Columns mergedColumns = existing.columns().mergeTo(update.columns()); Row.Builder builder = builder(existing.clustering()); - colUpdateTimeDelta = Math.min(colUpdateTimeDelta, Rows.merge(existing, update, mergedColumns, builder, nowInSec, indexer)); + colUpdateTimeDelta = Math.min(colUpdateTimeDelta, Rows.merge(existing, update, mergedColumns, builder, nowInSec)); Row reconciled = builder.build(); + indexer.onUpdated(existing, reconciled); + dataSize += reconciled.dataSize() - existing.dataSize(); heapSize += reconciled.unsharedHeapSizeExcludingData() - existing.unsharedHeapSizeExcludingData(); if (inserted == null) @@ -294,41 +304,6 @@ public class AtomicBTreePartition extends AbstractBTreePartition return reconciled; } - private void insertIntoIndexes(Row toInsert) - { - if (indexer == SecondaryIndexManager.nullUpdater) - return; - - maybeIndexPrimaryKeyColumns(toInsert); - Clustering clustering = toInsert.clustering(); - for (Cell cell : toInsert.cells()) - indexer.insert(clustering, cell); - } - - private void maybeIndexPrimaryKeyColumns(Row row) - { - // We want to update a primary key index with the most up to date info contains in that inserted row (if only for - // backward compatibility). Note that if there is an index but not a partition key or clustering column one, we've - // wasting this work. We might be able to avoid that if row indexing was pushed in the index updater. - long timestamp = row.primaryKeyLivenessInfo().timestamp(); - int ttl = row.primaryKeyLivenessInfo().ttl(); - - for (Cell cell : row.cells()) - { - long cellTimestamp = cell.timestamp(); - if (cell.isLive(nowInSec)) - { - if (cellTimestamp > timestamp) - { - timestamp = cellTimestamp; - ttl = cell.ttl(); - } - } - } - - indexer.maybeIndex(row.clustering(), timestamp, ttl, row.deletion()); - } - protected void reset() { this.dataSize = 0; diff --git a/src/java/org/apache/cassandra/db/partitions/PartitionUpdate.java b/src/java/org/apache/cassandra/db/partitions/PartitionUpdate.java index 2f0a4ecf61..3d2c94b537 100644 --- a/src/java/org/apache/cassandra/db/partitions/PartitionUpdate.java +++ b/src/java/org/apache/cassandra/db/partitions/PartitionUpdate.java @@ -19,30 +19,23 @@ package org.apache.cassandra.db.partitions; import java.io.IOException; import java.nio.ByteBuffer; -import java.util.*; +import java.util.ArrayList; +import java.util.Iterator; +import java.util.List; import com.google.common.collect.Iterables; -import com.google.common.collect.Iterators; import com.google.common.collect.Lists; - import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.context.CounterContext; import org.apache.cassandra.db.filter.ColumnFilter; import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.io.util.DataInputBuffer; -import org.apache.cassandra.io.util.DataInputPlus; -import org.apache.cassandra.io.util.DataOutputBuffer; -import org.apache.cassandra.io.util.DataOutputPlus; +import org.apache.cassandra.io.util.*; import org.apache.cassandra.net.MessagingService; -import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.MergeIterator; -import org.apache.cassandra.utils.Pair; import org.apache.cassandra.utils.btree.BTree; import org.apache.cassandra.utils.btree.UpdateFunction; diff --git a/src/java/org/apache/cassandra/db/rows/Cells.java b/src/java/org/apache/cassandra/db/rows/Cells.java index 080d6408a6..54df26eca5 100644 --- a/src/java/org/apache/cassandra/db/rows/Cells.java +++ b/src/java/org/apache/cassandra/db/rows/Cells.java @@ -22,9 +22,9 @@ import java.util.Comparator; import java.util.Iterator; import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.*; +import org.apache.cassandra.db.Conflicts; +import org.apache.cassandra.db.DeletionTime; import org.apache.cassandra.db.partitions.PartitionStatisticsCollector; -import org.apache.cassandra.db.index.SecondaryIndexManager; /** * Static methods to work on cells. @@ -58,9 +58,6 @@ public abstract class Cells * Also note that which cell is provided as {@code existing} and which is * provided as {@code update} matters for index updates. * - * @param clustering the clustering for the row the cells to merge originate from. - * This is only used for index updates, so this can be {@code null} if - * {@code indexUpdater == SecondaryIndexManager.nullUpdater}. * @param existing the pre-existing cell, the one that is updated. This can be * {@code null} if this reconciliation correspond to an insertion. * @param update the newly added cell, the update. This can be {@code null} out @@ -72,20 +69,15 @@ public abstract class Cells * @param nowInSec the current time in seconds (which plays a role during reconciliation * because deleted cells always have precedence on timestamp equality and deciding if a * cell is a live or not depends on the current time due to expiring cells). - * @param indexUpdater an index updater to which the result of the reconciliation is - * signaled (if relevant, that is if the update is not simply ignored by the reconciliation). - * This cannot be {@code null} but {@code SecondaryIndexManager.nullUpdater} can be passed. * * @return the timestamp delta between existing and update, or {@code Long.MAX_VALUE} if one * of them is {@code null} or deleted by {@code deletion}). */ - public static long reconcile(Clustering clustering, - Cell existing, + public static long reconcile(Cell existing, Cell update, DeletionTime deletion, Row.Builder builder, - int nowInSec, - SecondaryIndexManager.Updater indexUpdater) + int nowInSec) { existing = existing == null || deletion.deletes(existing) ? null : existing; update = update == null || deletion.deletes(update) ? null : update; @@ -93,10 +85,6 @@ public abstract class Cells { if (update != null) { - // It's inefficient that we call maybeIndex (which is for primary key indexes) on every cell, but - // we'll need to fix that damn 2ndary index API to avoid that. - updatePKIndexes(clustering, update, nowInSec, indexUpdater); - indexUpdater.insert(clustering, update); builder.addCell(update); } else if (existing != null) @@ -109,21 +97,9 @@ public abstract class Cells Cell reconciled = reconcile(existing, update, nowInSec); builder.addCell(reconciled); - // Note that this test rely on reconcile returning either 'existing' or 'update'. That's not true for counters but we don't index them - if (reconciled == update) - { - updatePKIndexes(clustering, update, nowInSec, indexUpdater); - indexUpdater.update(clustering, existing, reconciled); - } return Math.abs(existing.timestamp() - update.timestamp()); } - private static void updatePKIndexes(Clustering clustering, Cell cell, int nowInSec, SecondaryIndexManager.Updater indexUpdater) - { - if (indexUpdater != SecondaryIndexManager.nullUpdater && cell.isLive(nowInSec)) - indexUpdater.maybeIndex(clustering, cell.timestamp(), cell.ttl(), DeletionTime.LIVE); - } - /** * Reconciles/merge two cells. *

@@ -202,9 +178,6 @@ public abstract class Cells * Also note that which cells is provided as {@code existing} and which are * provided as {@code update} matters for index updates. * - * @param clustering the clustering for the row the cells to merge originate from. - * This is only used for index updates, so this can be {@code null} if - * {@code indexUpdater == SecondaryIndexManager.nullUpdater}. * @param column the complex column the cells are for. * @param existing the pre-existing cells, the ones that are updated. This can be * {@code null} if this reconciliation correspond to an insertion. @@ -217,9 +190,6 @@ public abstract class Cells * @param nowInSec the current time in seconds (which plays a role during reconciliation * because deleted cells always have precedence on timestamp equality and deciding if a * cell is a live or not depends on the current time due to expiring cells). - * @param indexUpdater an index updater to which the result of the reconciliation is - * signaled (if relevant, that is if the updates are not simply ignored by the reconciliation). - * This cannot be {@code null} but {@code SecondaryIndexManager.nullUpdater} can be passed. * * @return the smallest timestamp delta between corresponding cells from existing and update. A * timestamp delta being computed as the difference between a cell from {@code update} and the @@ -227,14 +197,12 @@ public abstract class Cells * of cells from {@code existing} and {@code update} having the same cell path is empty, this * returns {@code Long.MAX_VALUE}. */ - public static long reconcileComplex(Clustering clustering, - ColumnDefinition column, + public static long reconcileComplex(ColumnDefinition column, Iterator existing, Iterator update, DeletionTime deletion, Row.Builder builder, - int nowInSec, - SecondaryIndexManager.Updater indexUpdater) + int nowInSec) { Comparator comparator = column.cellPathComparator(); Cell nextExisting = getNext(existing); @@ -247,17 +215,17 @@ public abstract class Cells : comparator.compare(nextExisting.path(), nextUpdate.path())); if (cmp < 0) { - reconcile(clustering, nextExisting, null, deletion, builder, nowInSec, indexUpdater); + reconcile(nextExisting, null, deletion, builder, nowInSec); nextExisting = getNext(existing); } else if (cmp > 0) { - reconcile(clustering, null, nextUpdate, deletion, builder, nowInSec, indexUpdater); + reconcile(null, nextUpdate, deletion, builder, nowInSec); nextUpdate = getNext(update); } else { - timeDelta = Math.min(timeDelta, reconcile(clustering, nextExisting, nextUpdate, deletion, builder, nowInSec, indexUpdater)); + timeDelta = Math.min(timeDelta, reconcile(nextExisting, nextUpdate, deletion, builder, nowInSec)); nextExisting = getNext(existing); nextUpdate = getNext(update); } diff --git a/src/java/org/apache/cassandra/db/rows/Rows.java b/src/java/org/apache/cassandra/db/rows/Rows.java index 0b739a81a5..ce177f2b98 100644 --- a/src/java/org/apache/cassandra/db/rows/Rows.java +++ b/src/java/org/apache/cassandra/db/rows/Rows.java @@ -17,7 +17,9 @@ */ package org.apache.cassandra.db.rows; -import java.util.*; +import java.util.ArrayList; +import java.util.Iterator; +import java.util.List; import com.google.common.collect.Iterators; import com.google.common.collect.PeekingIterator; @@ -25,7 +27,6 @@ import com.google.common.collect.PeekingIterator; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; import org.apache.cassandra.db.partitions.PartitionStatisticsCollector; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.utils.SearchIterator; /** @@ -127,7 +128,7 @@ public abstract class Rows * @param diffListener the listener to which to signal the differences between the inputs and the merged * result. */ - public static void diff(Row merged, Columns columns, Row[] inputs, RowDiffListener diffListener) + public static void diff(RowDiffListener diffListener, Row merged, Columns columns, Row...inputs) { Clustering clustering = merged.clustering(); LivenessInfo mergedInfo = merged.primaryKeyLivenessInfo().isEmpty() ? null : merged.primaryKeyLivenessInfo(); @@ -218,23 +219,17 @@ public abstract class Rows { Columns mergedColumns = row1.columns().mergeTo(row2.columns()); Row.Builder builder = BTreeRow.sortedBuilder(mergedColumns); - merge(row1, row2, mergedColumns, builder, nowInSec, SecondaryIndexManager.nullUpdater); + merge(row1, row2, mergedColumns, builder, nowInSec); return builder.build(); } - public static void merge(Row row1, Row row2, Columns mergedColumns, Row.Builder builder, int nowInSec) - { - merge(row1, row2, mergedColumns, builder, nowInSec, SecondaryIndexManager.nullUpdater); - } - // Merge rows in memtable // Return the minimum timestamp delta between existing and update public static long merge(Row existing, Row update, Columns mergedColumns, Row.Builder builder, - int nowInSec, - SecondaryIndexManager.Updater indexUpdater) + int nowInSec) { Clustering clustering = existing.clustering(); builder.newRow(clustering); @@ -253,20 +248,16 @@ public abstract class Rows builder.addPrimaryKeyLivenessInfo(mergedInfo); builder.addRowDeletion(deletion); - indexUpdater.maybeIndex(clustering, mergedInfo.timestamp(), mergedInfo.ttl(), deletion); - for (int i = 0; i < mergedColumns.simpleColumnCount(); i++) { ColumnDefinition c = mergedColumns.getSimple(i); Cell existingCell = existing.getCell(c); Cell updateCell = update.getCell(c); - timeDelta = Math.min(timeDelta, Cells.reconcile(clustering, - existingCell, + timeDelta = Math.min(timeDelta, Cells.reconcile(existingCell, updateCell, deletion, builder, - nowInSec, - indexUpdater)); + nowInSec)); } for (int i = 0; i < mergedColumns.complexColumnCount(); i++) @@ -285,7 +276,7 @@ public abstract class Rows Iterator existingCells = existingData == null ? null : existingData.iterator(); Iterator updateCells = updateData == null ? null : updateData.iterator(); - timeDelta = Math.min(timeDelta, Cells.reconcileComplex(clustering, c, existingCells, updateCells, maxDt, builder, nowInSec, indexUpdater)); + timeDelta = Math.min(timeDelta, Cells.reconcileComplex(c, existingCells, updateCells, maxDt, builder, nowInSec)); } return timeDelta; diff --git a/src/java/org/apache/cassandra/index/Index.java b/src/java/org/apache/cassandra/index/Index.java new file mode 100644 index 0000000000..8f126fe9e4 --- /dev/null +++ b/src/java/org/apache/cassandra/index/Index.java @@ -0,0 +1,388 @@ +package org.apache.cassandra.index; + +import java.util.Optional; +import java.util.concurrent.Callable; +import java.util.function.BiFunction; + +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.cql3.Operator; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.filter.RowFilter; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.transactions.IndexTransaction; +import org.apache.cassandra.schema.IndexMetadata; +import org.apache.cassandra.utils.concurrent.OpOrder; + +/** + * Consisting of a top level Index interface and two sub-interfaces which handle read and write operations, + * Searcher and Indexer respectively, this defines a secondary index implementation. + * Instantiation is done via reflection and implementations must provide a constructor which takes the base + * table's ColumnFamilyStore and the IndexMetadata which defines the Index as arguments. e.g: + * {@code MyCustomIndex( ColumnFamilyStore baseCfs, IndexMetadata indexDef )} + * + * The main interface defines methods for index management, index selection at both write and query time, + * as well as validation of values that will ultimately be indexed. + * Two sub-interfaces are also defined, which represent single use helpers for short lived tasks at read and write time. + * Indexer: an event listener which receives notifications at particular points during an update of a single partition + * in the base table. + * Searcher: performs queries against the index based on a predicate defined in a RowFilter. An instance + * is expected to be single use, being involved in the execution of a single ReadCommand. + * + * The main interface includes factory methods for obtaining instances of both of the sub-interfaces; + * + * The methods defined in the top level interface can be grouped into 3 categories: + * + * Management Tasks: + * This group of methods is primarily concerned with maintenance of secondary indexes are are mainly called from + * SecondaryIndexManager. It includes methods for registering and un-registering an index, performing maintenance + * tasks such as (re)building an index from SSTable data, flushing, invalidating and so forth, as well as some to + * retrieve general metadata about the index (index name, any internal tables used for persistence etc). + * Several of these maintenance functions have a return type of Callable; the expectation for these methods is + * that any work required to be performed by the method be done inside the Callable so that the responsibility for + * scheduling its execution can rest with SecondaryIndexManager. For instance, a task like reloading index metadata + * following potential updates caused by modifications to the base table may be performed in a blocking way. In + * contrast, adding a new index may require it to be built from existing SSTable data, a potentially expensive task + * which should be performed asyncronously. + * + * Index Selection: + * There are two facets to index selection, write time and read time selection. The former is concerned with + * identifying whether an index should be informed about a particular write operation. The latter is about providing + * means to use the index for search during query execution. + * + * Validation: + * Values that may be written to an index are checked as part of input validation, prior to an update or insert + * operation being accepted. + * + * + * Sub-interfaces: + * + * Update processing: + * Indexes are subscribed to the stream of events generated by modifications to the base table. Subscription is + * done via first registering the Index with the base table's SecondaryIndexManager. For each partition update, the set + * of registered indexes are then filtered based on the properties of the update using the selection methods on the main + * interface described above. Each of the indexes in the filtered set then provides an event listener to receive + * notifications about the update as it is processed. As such then, a event handler instance is scoped to a single + * partition update; SecondaryIndexManager obtains a new handler for every update it processes (via a call to the + * factory method, indexerFor. That handler will then receive all events for the update, before being + * discarded by the SecondaryIndexManager. Indexer instances are never re-used by SecondaryIndexManager and the + * expectation is that each call to indexerFor should return a unique instance, or at least if instances can + * be recycled, that a given instance is only used to process a single partition update at a time. + * + * Search: + * Each query (i.e. a single ReadCommand) that uses indexes will use a single instance of Index.Searcher. As with + * processing of updates, an Index must be registered with the primary table's SecondaryIndexManager to be able to + * support queries. During the processing of a ReadCommand, the Expressions in its RowFilter are examined to determine + * whether any of them are supported by a registered Index. supportsExpression is used to filter out Indexes which + * cannot support a given Expression. After filtering, the set of candidate indexes are ranked according to the result + * of getEstimatedResultRows and the most selective (i.e. the one expected to return the smallest number of results) is + * chosen. A Searcher instance is then obtained from the searcherFor method & used to perform the actual Index lookup. + * Finally, Indexes can define a post processing step to be performed on the coordinator, after results (partitions from + * the primary table) have been received from replicas and reconciled. This post processing is defined as a + * java.util.functions.BiFunction, that is a function which takes as + * arguments a PartitionIterator (containing the reconciled result rows) and a RowFilter (from the ReadCommand being + * executed) and returns another iterator of partitions, possibly having transformed the initial results in some way. + * The post processing function is obtained from the Index's postProcessorFor method; the built-in indexes which ship + * with Cassandra return a no-op function here. + * + * An optional static method may be provided to validate custom index options: + * + *

 {@code
+ * public static Map validateOptions(Map options);
+ * } 
+ * + * The input is the map of index options supplied in the WITH clause of a CREATE INDEX statement. The method should + * return a map containing any of the supplied options which are not valid for the implementation. If the returned + * map is not empty, validation is considered failed and an error is raised. Alternatively, the implementation may + * choose to throw an org.apache.cassandra.exceptions.ConfigurationException if invalid options are encountered. + * + */ +public interface Index +{ + + /* + * Management functions + */ + + /** + * Return a task to perform any initialization work when a new index instance is created. + * This may involve costly operations such as (re)building the index, and is performed asynchronously + * by SecondaryIndexManager + * @return a task to perform any necessary initialization work + */ + public Callable getInitializationTask(); + + /** + * Returns the IndexMetadata which configures and defines the index instance. This should be the same + * object passed as the argument to setIndexMetadata. + * @return the index's metadata + */ + public IndexMetadata getIndexMetadata(); + + /** + * Return a task to reload the internal metadata of an index. + * Called when the base table metadata is modified or when the configuration of the Index is updated + * Implementations should return a task which performs any necessary work to be done due to + * updating the configuration(s) such as (re)building etc. This task is performed asynchronously + * by SecondaryIndexManager + * @return task to be executed by the index manager during a reload + */ + public Callable getMetadataReloadTask(IndexMetadata indexMetadata); + + /** + * An index must be registered in order to be able to either subscribe to update events on the base + * table and/or to provide Searcher functionality for reads. The double dispatch involved here, where + * the Index actually performs its own registration by calling back to the supplied IndexRegistry's + * own registerIndex method, is to make the decision as to whether or not to register an index belong + * to the implementation, not the manager. + * @param registry the index registry to register the instance with + */ + public void register(IndexRegistry registry); + + /** + * Return an identifier for the index. This should be unique across all indexes on a given base table + * @return the name of the index + */ + public String getIndexName(); + + /** + * If the index implementation uses a local table to store its index data this method should return a + * handle to it. If not, an empty Optional should be returned. Typically, this is useful for the built-in + * Index implementations. + * @return an Optional referencing the Index's backing storage table if it has one, or Optional.empty() if not. + */ + public Optional getBackingTable(); + + /** + * Return a task which performs a blocking flush of the index's data to persistent storage. + * @return task to be executed by the index manager to perform the flush. + */ + public Callable getBlockingFlushTask(); + + /** + * Return a task which invalidates the index, indicating it should no longer be considered usable. + * This should include an clean up and releasing of resources required when dropping an index. + * @return task to be executed by the index manager to invalidate the index. + */ + public Callable getInvalidateTask(); + + /** + * Return a task to truncate the index with the specified truncation timestamp. + * Called when the base table is truncated. + * @param truncatedAt timestamp of the truncation operation. This will be the same timestamp used + * in the truncation of the base table. + * @return task to be executed by the index manager when the base table is truncated. + */ + public Callable getTruncateTask(long truncatedAt); + + /* + * Index selection + */ + + /** + * Called to determine whether this index should process a particular partition update. + * @param columns + * @return + */ + public boolean indexes(PartitionColumns columns); + + // TODO : this will change when we decouple indexes from specific columns for real per-row indexes + /** + * Called to determine whether this index can provide a searcher to execute a query on the + * supplied column using the specified operator. This forms part of the query validation done + * before a CQL select statement is executed. + * @param column the target column of a search query predicate + * @param operator the operator of a search query predicate + * @return true if this index is capable of supporting such expressions, false otherwise + */ + public boolean supportsExpression(ColumnDefinition column, Operator operator); + + /** + * Transform an initial RowFilter into the filter that will still need to applied + * to a set of Rows after the index has performed it's initial scan. + * Used in ReadCommand#executeLocal to reduce the amount of filtering performed on the + * results of the index query. + * + * @param filter the intial filter belonging to a ReadCommand + * @return the (hopefully) reduced filter that would still need to be applied after + * the index was used to narrow the initial result set + */ + public RowFilter getPostIndexQueryFilter(RowFilter filter); + + /** + * Return an estimate of the number of results this index is expected to return for any given + * query that it can be used to answer. Used in conjunction with indexes() and supportsExpression() + * to determine the most selective index for a given ReadCommand. Additionally, this is also used + * by StorageProxy.estimateResultsPerRange to calculate the initial concurrency factor for range requests + * + * @return the estimated average number of results a Searcher may return for any given query + */ + public long getEstimatedResultRows(); + + /* + * Input validation + */ + + /** + * Called at write time to ensure that values present in the update + * are valid according to the rules of all registered indexes which + * will process it. The partition key as well as the clustering and + * cell values for each row in the update may be checked by index + * implementations + * @param update PartitionUpdate containing the values to be validated by registered Index implementations + * @throws InvalidRequestException + */ + public void validate(PartitionUpdate update) throws InvalidRequestException; + + /* + * Update processing + */ + + /** + * Factory method for write time event handlers. + * Callers should check the indexes method first and only get a new + * handler when the index claims an interest in the specific update + * otherwise work may be done unnecessarily + * + * @param key key of the partition being modified + * @param nowInSec current time of the update operation + * @param opGroup operation group spanning the update operation + * @param transactionType indicates what kind of update is being performed on the base data + * i.e. a write time insert/update/delete or the result of compaction + * @return + */ + public Indexer indexerFor(DecoratedKey key, + int nowInSec, + OpOrder.Group opGroup, + IndexTransaction.Type transactionType); + + /** + * Listener for processing events emitted during a single partition update. + * Instances of this are responsible for applying modifications to the index in response to a single update + * operation on a particular partition of the base table. + * That update may be generated by the normal write path, by iterating SSTables during streaming operations or when + * building or rebuilding an index from source. Updates also occur during compaction when multiple versions of a + * source partition from different SSTables are merged. + */ + public interface Indexer + { + /** + * Notification of the start of a partition update. + * This event always occurs before any other during the update. + */ + public void begin(); + + /** + * Notification of a top level partition delete. + * @param deletionTime + */ + public void partitionDelete(DeletionTime deletionTime); + + /** + * Notification of a RangeTombstone. + * An update of a single partition may contain multiple RangeTombstones, + * and a notification will be passed for each of them. + * @param tombstone + */ + public void rangeTombstone(RangeTombstone tombstone); + + /** + * Notification that a new row was inserted into the Memtable holding the partition. + * This only implies that the inserted row was not already present in the Memtable, + * it *does not* guarantee that the row does not exist in an SSTable, potentially with + * additional column data. + * + * @param row the Row being inserted into the base table's Memtable. + */ + public void insertRow(Row row); + + /** + * Notification of a modification to a row in the base table's Memtable. + * This is allow an Index implementation to clean up entries for base data which is + * never flushed to disk (and so will not be purged during compaction). + * It's important to note that the old & new rows supplied here may not represent + * the totality of the data for the Row with this particular Clustering. There may be + * additional column data in SSTables which is not present in either the old or new row, + * so implementations should be aware of that. + * The supplied rows contain only column data which has actually been updated. + * oldRowData contains only the columns which have been removed from the Row's + * representation in the Memtable, while newRowData includes only new columns + * which were not previously present. Any column data which is unchanged by + * the update is not included. + * + * @param oldRowData data that was present in existing row and which has been removed from + * the base table's Memtable + * @param newRowData data that was not present in the existing row and is being inserted + * into the base table's Memtable + */ + public void updateRow(Row oldRowData, Row newRowData); + + /** + * Notification that a row was removed from the partition. + * Note that this is only called as part of either a compaction or a cleanup. + * This context is indicated by the TransactionType supplied to the indexerFor method. + * + * As with updateRow, it cannot be guaranteed that all data belonging to the Clustering + * of the supplied Row has been removed (although in the case of a cleanup, that is the + * ultimate intention). + * There may be data for the same row in other SSTables, so in this case Indexer implementations + * should *not* assume that all traces of the row have been removed. In particular, + * it is not safe to assert that all values associated with the Row's Clustering + * have been deleted, so implementations which index primary key columns should not + * purge those entries from their indexes. + * + * @param row data being removed from the base table + */ + public void removeRow(Row row); + + /** + * Notification of the end of the partition update. + * This event always occurs after all others for the particular update. + */ + public void finish(); + } + + /* + * Querying + */ + + /** + * Return a function which performs post processing on the results of a partition range read command. + * In future, this may be used as a generalized mechanism for transforming results on the coordinator prior + * to returning them to the caller. + * + * This is used on the coordinator during execution of a range command to perform post + * processing of merged results obtained from the necessary replicas. This is the only way in which results are + * transformed in this way but this may change over time as usage is generalized. + * See CASSANDRA-8717 for further discussion. + * + * The function takes a PartitionIterator of the results from the replicas which has already been collated + * & reconciled, along with the command being executed. It returns another PartitionIterator containing the results + * of the transformation (which may be the same as the input if the transformation is a no-op). + */ + public BiFunction postProcessorFor(ReadCommand command); + + /** + * Factory method for query time search helper. + * @param command the read command being executed + * @return an Searcher with which to perform the supplied command + */ + public Searcher searcherFor(ReadCommand command); + + /** + * Performs the actual index lookup during execution of a ReadCommand. + * An instance performs its query according to the RowFilter.Expression it was created for (see searcherFor) + * An Expression is a predicate of the form [column] [operator] [value]. + */ + public interface Searcher + { + /** + * @param orderGroup the collection of OpOrder.Groups which the ReadCommand is being performed under. + * @return partitions from the base table matching the criteria of the search. + */ + public UnfilteredPartitionIterator search(ReadOrderGroup orderGroup); + } +} diff --git a/src/java/org/apache/cassandra/index/IndexRegistry.java b/src/java/org/apache/cassandra/index/IndexRegistry.java new file mode 100644 index 0000000000..6a004fbc7b --- /dev/null +++ b/src/java/org/apache/cassandra/index/IndexRegistry.java @@ -0,0 +1,22 @@ +package org.apache.cassandra.index; + +import java.util.Collection; + +import org.apache.cassandra.schema.IndexMetadata; + +/** + * The collection of all Index instances for a base table. + * The SecondaryIndexManager for a ColumnFamilyStore contains an IndexRegistry + * (actually it implements this interface at present) and Index implementations + * register in order to: + * i) subscribe to the stream of updates being applied to partitions in the base table + * ii) provide searchers to support queries with the relevant search predicates + */ +public interface IndexRegistry +{ + void registerIndex(Index index); + void unregisterIndex(Index index); + + Index getIndex(IndexMetadata indexMetadata); + Collection listIndexes(); +} diff --git a/src/java/org/apache/cassandra/db/index/SecondaryIndexBuilder.java b/src/java/org/apache/cassandra/index/SecondaryIndexBuilder.java similarity index 88% rename from src/java/org/apache/cassandra/db/index/SecondaryIndexBuilder.java rename to src/java/org/apache/cassandra/index/SecondaryIndexBuilder.java index a117f6db0e..d6ae8e2b4a 100644 --- a/src/java/org/apache/cassandra/db/index/SecondaryIndexBuilder.java +++ b/src/java/org/apache/cassandra/index/SecondaryIndexBuilder.java @@ -15,7 +15,7 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index; +package org.apache.cassandra.index; import java.io.IOException; import java.util.Set; @@ -25,10 +25,9 @@ import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.DecoratedKey; import org.apache.cassandra.db.Keyspace; import org.apache.cassandra.db.compaction.CompactionInfo; -import org.apache.cassandra.db.compaction.OperationType; import org.apache.cassandra.db.compaction.CompactionInterruptedException; +import org.apache.cassandra.db.compaction.OperationType; import org.apache.cassandra.io.sstable.ReducingKeyIterator; -import org.apache.cassandra.utils.FBUtilities; import org.apache.cassandra.utils.UUIDGen; /** @@ -37,14 +36,14 @@ import org.apache.cassandra.utils.UUIDGen; public class SecondaryIndexBuilder extends CompactionInfo.Holder { private final ColumnFamilyStore cfs; - private final Set idxNames; + private final Set indexers; private final ReducingKeyIterator iter; private final UUID compactionId; - public SecondaryIndexBuilder(ColumnFamilyStore cfs, Set idxNames, ReducingKeyIterator iter) + public SecondaryIndexBuilder(ColumnFamilyStore cfs, Set indexers, ReducingKeyIterator iter) { this.cfs = cfs; - this.idxNames = idxNames; + this.indexers = indexers; this.iter = iter; this.compactionId = UUIDGen.getTimeUUID(); } @@ -67,7 +66,7 @@ public class SecondaryIndexBuilder extends CompactionInfo.Holder if (isStopRequested()) throw new CompactionInterruptedException(getCompactionInfo()); DecoratedKey key = iter.next(); - Keyspace.indexPartition(key, cfs, idxNames); + Keyspace.indexPartition(key, cfs, indexers); } } finally diff --git a/src/java/org/apache/cassandra/index/SecondaryIndexManager.java b/src/java/org/apache/cassandra/index/SecondaryIndexManager.java new file mode 100644 index 0000000000..5d32b70322 --- /dev/null +++ b/src/java/org/apache/cassandra/index/SecondaryIndexManager.java @@ -0,0 +1,900 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.index; + +import java.lang.reflect.Constructor; +import java.util.*; +import java.util.concurrent.*; +import java.util.function.Function; +import java.util.stream.Collectors; +import java.util.stream.Stream; + +import com.google.common.base.Strings; +import com.google.common.collect.ImmutableSet; +import com.google.common.collect.Maps; +import com.google.common.primitives.Longs; +import com.google.common.util.concurrent.Futures; +import com.google.common.util.concurrent.MoreExecutors; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import org.apache.cassandra.concurrent.JMXEnabledThreadPoolExecutor; +import org.apache.cassandra.concurrent.NamedThreadFactory; +import org.apache.cassandra.concurrent.StageManager; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.cql3.statements.IndexTarget; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.compaction.CompactionManager; +import org.apache.cassandra.db.filter.RowFilter; +import org.apache.cassandra.db.lifecycle.SSTableSet; +import org.apache.cassandra.db.lifecycle.View; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.transactions.*; +import org.apache.cassandra.io.sstable.ReducingKeyIterator; +import org.apache.cassandra.io.sstable.format.SSTableReader; +import org.apache.cassandra.schema.IndexMetadata; +import org.apache.cassandra.schema.Indexes; +import org.apache.cassandra.tracing.Tracing; +import org.apache.cassandra.utils.FBUtilities; +import org.apache.cassandra.utils.concurrent.OpOrder; +import org.apache.cassandra.utils.concurrent.Refs; + +/** + * Handles the core maintenance functionality associated with indexes: adding/removing them to or from + * a table, (re)building during bootstrap or other streaming operations, flushing, reloading metadata + * and so on. + * + * The Index interface defines a number of methods which return Callable. These are primarily the + * management tasks for an index implementation. Most of them are currently executed in a blocking + * fashion via submission to SIM's blockingExecutor. This provides the desired behaviour in pretty + * much all cases, as tasks like flushing an index needs to be executed synchronously to avoid potentially + * deadlocking on the FlushWriter or PostFlusher. Several of these Callable returning methods on Index could + * then be defined with as void and called directly from SIM (rather than being run via the executor service). + * Separating the task defintion from execution gives us greater flexibility though, so that in future, for example, + * if the flush process allows it we leave open the possibility of executing more of these tasks asynchronously. + * + * The primary exception to the above is the Callable returned from Index#addIndexedColumn. This may + * involve a significant effort, building a new index over any existing data. We perform this task asynchronously; + * as it is called as part of a schema update, which we do not want to block for a long period. Building non-custom + * indexes is performed on the CompactionManager. + * + * This class also provides instances of processors which listen to updates to the base table and forward to + * registered Indexes the info required to keep those indexes up to date. + * There are two variants of these processors, each with a factory method provided by SIM: + * IndexTransaction: deals with updates generated on the regular write path. + * CleanupTransaction: used when partitions are modified during compaction or cleanup operations. + * Further details on their usage and lifecycles can be found in the interface definitions below. + * + * Finally, the bestIndexFor method is used at query time to identify the most selective index of those able + * to satisfy any search predicates defined by a ReadCommand's RowFilter. It returns a thin IndexAccessor object + * which enables the ReadCommand to access the appropriate functions of the Index at various stages in its lifecycle. + * e.g. the getEstimatedResultRows is required when StorageProxy calculates the initial concurrency factor for + * distributing requests to replicas, whereas a Searcher instance is needed when the ReadCommand is executed locally on + * a target replica. + */ +public class SecondaryIndexManager implements IndexRegistry +{ + private static final Logger logger = LoggerFactory.getLogger(SecondaryIndexManager.class); + + + private Map indexes = Maps.newConcurrentMap(); + + // executes tasks returned by Indexer#addIndexColumn which may require index(es) to be (re)built + private static final ExecutorService asyncExecutor = + new JMXEnabledThreadPoolExecutor(1, + StageManager.KEEPALIVE, + TimeUnit.SECONDS, + new LinkedBlockingQueue<>(), + new NamedThreadFactory("SecondaryIndexManagement"), + "internal"); + + // executes all blocking tasks produced by Indexers e.g. getFlushTask, getMetadataReloadTask etc + private static final ExecutorService blockingExecutor = MoreExecutors.newDirectExecutorService(); + + /** + * The underlying column family containing the source data for these indexes + */ + public final ColumnFamilyStore baseCfs; + + public SecondaryIndexManager(ColumnFamilyStore baseCfs) + { + this.baseCfs = baseCfs; + } + + + /** + * Drops and adds new indexes associated with the underlying CF + */ + public void reload() + { + // figure out what needs to be added and dropped. + Indexes tableIndexes = baseCfs.metadata.getIndexes(); + indexes.keySet() + .stream() + .filter(indexName -> !tableIndexes.has(indexName)) + .forEach(this::removeIndex); + + // we call add for every index definition in the collection as + // some may not have been created here yet, only added to schema + for (IndexMetadata tableIndex : tableIndexes) + addIndex(tableIndex); + } + + private Future reloadIndex(IndexMetadata indexDef) + { + // if the index metadata has changed, reload the index + IndexMetadata registered = indexes.get(indexDef.name).getIndexMetadata(); + if (!registered.equals(indexDef)) + { + Index index = indexes.remove(registered.name); + index.register(this); + return blockingExecutor.submit(index.getMetadataReloadTask(indexDef)); + } + + // otherwise, nothing to do + return Futures.immediateFuture(null); + } + + private Future createIndex(IndexMetadata indexDef) + { + Index index = createInstance(indexDef); + index.register(this); + final Callable initialBuildTask = index.getInitializationTask(); + return initialBuildTask == null + ? Futures.immediateFuture(null) + : asyncExecutor.submit(initialBuildTask); + } + + /** + * Adds and builds a index + * @param indexDef the IndexMetadata describing the index + */ + public synchronized Future addIndex(IndexMetadata indexDef) + { + if (indexes.containsKey(indexDef.name)) + return reloadIndex(indexDef); + else + return createIndex(indexDef); + } + + public synchronized void removeIndex(String indexName) + { + Index index = indexes.remove(indexName); + if (null != index) + { + executeBlocking(index.getInvalidateTask()); + unregisterIndex(index); + } + } + + /** + * Called when dropping a Table + */ + public void markAllIndexesRemoved() + { + getBuiltIndexNames().forEach(this::markIndexRemoved); + } + + /** + * Does a full, blocking rebuild of the indexes specified by columns from the sstables. + * Caller must acquire and release references to the sstables used here. + * Note also that only this method of (re)building indexes: + * a) takes a set of index *names* rather than Indexers + * b) marks exsiting indexes removed prior to rebuilding + * + * @param sstables the data to build from + * @param indexNames the list of indexes to be rebuilt + */ + public void rebuildIndexesBlocking(Collection sstables, Set indexNames) + { + Set toRebuild = indexes.values().stream() + .filter(indexer -> indexNames.contains(indexer.getIndexName())) + .collect(Collectors.toSet()); + if (toRebuild.isEmpty()) + { + logger.info("No defined indexes with the supplied names"); + return; + } + + toRebuild.forEach(indexer -> markIndexRemoved(indexer.getIndexName())); + + buildIndexesBlocking(sstables, toRebuild); + + toRebuild.forEach(indexer -> markIndexBuilt(indexer.getIndexName())); + } + + public void buildAllIndexesBlocking(Collection sstables) + { + buildIndexesBlocking(sstables, ImmutableSet.copyOf(indexes.values())); + } + + // For convenience, may be called directly from Index impls + public void buildIndexBlocking(Index index) + { + try (ColumnFamilyStore.RefViewFragment viewFragment = baseCfs.selectAndReference(View.select(SSTableSet.CANONICAL)); + Refs sstables = viewFragment.refs) + { + buildIndexesBlocking(sstables, Collections.singleton(index)); + markIndexBuilt(index.getIndexName()); + } + } + + private void buildIndexesBlocking(Collection sstables, Set indexes) + { + if (indexes.isEmpty()) + return; + + logger.info("Submitting index build of {} for data in {}", + indexes.stream().map(Index::getIndexName).collect(Collectors.joining(",")), + sstables.stream().map(SSTableReader::toString).collect(Collectors.joining(","))); + + SecondaryIndexBuilder builder = new SecondaryIndexBuilder(baseCfs, + indexes, + new ReducingKeyIterator(sstables)); + Future future = CompactionManager.instance.submitIndexBuild(builder); + FBUtilities.waitOnFuture(future); + + flushIndexesBlocking(indexes); + logger.info("Index build of {} complete", + indexes.stream().map(Index::getIndexName).collect(Collectors.joining(","))); + } + + private void markIndexBuilt(String indexName) + { + SystemKeyspace.setIndexBuilt(baseCfs.name, indexName); + } + + private void markIndexRemoved(String indexName) + { + SystemKeyspace.setIndexRemoved(baseCfs.name, indexName); + } + + + public Index getIndexByName(String indexName) + { + return indexes.get(indexName); + } + + private Index createInstance(IndexMetadata indexDef) + { + Index newIndex; + if (indexDef.isCustom()) + { + assert indexDef.options != null; + String className = indexDef.options.get(IndexTarget.CUSTOM_INDEX_OPTION_NAME); + assert ! Strings.isNullOrEmpty(className); + try + { + Class indexClass = FBUtilities.classForName(className, "Index"); + Constructor ctor = indexClass.getConstructor(ColumnFamilyStore.class, IndexMetadata.class); + newIndex = (Index)ctor.newInstance(baseCfs, indexDef); + } + catch (Exception e) + { + throw new RuntimeException(e); + } + } + else + { + newIndex = CassandraIndex.newIndex(baseCfs, indexDef); + } + return newIndex; + } + + /** + * Truncate all indexes + */ + public void truncateAllIndexesBlocking(final long truncatedAt) + { + executeAllBlocking(indexes.values().stream(), (index) -> index.getTruncateTask(truncatedAt)); + } + + /** + * Remove all indexes + */ + public void invalidateAllIndexesBlocking() + { + executeAllBlocking(indexes.values().stream(), Index::getInvalidateTask); + } + + /** + * Perform a blocking flush all indexes + */ + public void flushAllIndexesBlocking() + { + flushIndexesBlocking(ImmutableSet.copyOf(indexes.values())); + } + + /** + * Perform a blocking flush of selected indexes + */ + public void flushIndexesBlocking(Set indexes) + { + if (indexes.isEmpty()) + return; + + List> wait = new ArrayList<>(); + List nonCfsIndexes = new ArrayList<>(); + + // for each CFS backed index, submit a flush task which we'll wait on for completion + // for the non-CFS backed indexes, we'll flush those while we wait. + synchronized (baseCfs.getTracker()) + { + indexes.forEach(index -> + index.getBackingTable() + .map(cfs -> wait.add(cfs.forceFlush())) + .orElse(nonCfsIndexes.add(index))); + } + executeAllBlocking(nonCfsIndexes.stream(), Index::getBlockingFlushTask); + FBUtilities.waitOnFutures(wait); + } + + /** + * Performs a blocking flush of all custom indexes + */ + public void flushAllCustomIndexesBlocking() + { + Set customIndexers = indexes.values().stream() + .filter(index -> !(index instanceof CassandraIndex)) + .collect(Collectors.toSet()); + flushIndexesBlocking(customIndexers); + } + + /** + * @return all indexes which are marked as built and ready to use + */ + public List getBuiltIndexNames() + { + Set allIndexNames = new HashSet<>(); + indexes.values().stream() + .map(Index::getIndexName) + .forEach(allIndexNames::add); + return SystemKeyspace.getBuiltIndexes(baseCfs.keyspace.getName(), allIndexNames); + } + + /** + * @return all backing Tables used by registered indexes + */ + public Set getAllIndexColumnFamilyStores() + { + Set backingTables = new HashSet<>(); + indexes.values().forEach(index -> index.getBackingTable().ifPresent(backingTables::add)); + return backingTables; + } + + /** + * @return if there are ANY indexes registered for this table + */ + public boolean hasIndexes() + { + return !indexes.isEmpty(); + } + + /** + * When building an index against existing data in sstables, add the given partition to the index + */ + public void indexPartition(UnfilteredRowIterator partition, OpOrder.Group opGroup, Set indexes, int nowInSec) + { + if (!indexes.isEmpty()) + { + DecoratedKey key = partition.partitionKey(); + Set indexers = indexes.stream() + .map(index -> index.indexerFor(key, + nowInSec, + opGroup, + IndexTransaction.Type.UPDATE)) + .collect(Collectors.toSet()); + + indexers.forEach(Index.Indexer::begin); + + try (RowIterator filtered = UnfilteredRowIterators.filter(partition, nowInSec)) + { + if (!filtered.staticRow().isEmpty()) + indexers.forEach(indexer -> indexer.insertRow(filtered.staticRow())); + + while (filtered.hasNext()) + { + Row row = filtered.next(); + indexers.forEach(indexer -> indexer.insertRow(row)); + } + } + + indexers.forEach(Index.Indexer::finish); + } + } + + /** + * Delete all data from all indexes for this partition. + * For when cleanup rips a partition out entirely. + * + * TODO : improve cleanup transaction to batch updates & perform them async + */ + public void deletePartition(UnfilteredRowIterator partition, int nowInSec) + { + // we need to acquire memtable lock because secondary index deletion may + // cause a race (see CASSANDRA-3712). This is done internally by the + // index transaction when it commits + CleanupTransaction indexTransaction = newCleanupTransaction(partition.partitionKey(), + partition.columns(), + nowInSec); + indexTransaction.start(); + indexTransaction.onPartitionDeletion(partition.partitionLevelDeletion()); + indexTransaction.commit(); + + while (partition.hasNext()) + { + Unfiltered unfiltered = partition.next(); + if (unfiltered.kind() != Unfiltered.Kind.ROW) + continue; + + indexTransaction = newCleanupTransaction(partition.partitionKey(), + partition.columns(), + nowInSec); + indexTransaction.start(); + indexTransaction.onRowDelete((Row)unfiltered); + indexTransaction.commit(); + } + } + + /** + * Called at query time to find the most selective of the registered index implementation + * (i.e. the one likely to return the fewest results) from those registered. + * Implementation specific validation of the target expression by the most selective + * index should be performed in the searcherFor method to ensure that we pick the right + * index regardless of the validity of the expression. + * + * This method is called at various points during the lifecycle of a ReadCommand (to obtain a Searcher, + * get the index's underlying CFS for ReadOrderGroup, or an estimate of the result size from an average index + * query). + * + * Ideally, we would do this relatively expensive operation only once, and attach the index to the + * ReadCommand for future reference. This requires the index be passed onto additional commands generated + * to process subranges etc. + * + * @param command ReadCommand to be executed + * @return an Index instance, ready to use during execution of the command, or null if none + * of the registered indexes can support the command. + */ + public Index getBestIndexFor(ReadCommand command, boolean includeInTrace) + { + if (indexes.isEmpty() || command.rowFilter().isEmpty()) + return null; + + Set searchableIndexes = new HashSet<>(); + for (RowFilter.Expression expression : command.rowFilter()) + { + indexes.values().stream() + .filter(index -> index.supportsExpression(expression.column(), expression.operator())) + .forEach(searchableIndexes::add); + } + + if (searchableIndexes.isEmpty()) + { + logger.debug("No applicable indexes found"); + if (includeInTrace) + Tracing.trace("No applicable indexes found"); + return null; + } + + Index selected = searchableIndexes.stream() + .max((a, b) -> Longs.compare(a.getEstimatedResultRows(), + b.getEstimatedResultRows())) + .orElseThrow(() -> new AssertionError("Could not select most selective index")); + + // pay for an additional threadlocal get() rather than build the strings unnecessarily + if (includeInTrace && Tracing.isTracing()) + { + Tracing.trace("Index mean cardinalities are {}. Scanning with {}.", + searchableIndexes.stream().map(i -> i.getIndexName() + ':' + i.getEstimatedResultRows()) + .collect(Collectors.joining(",")), + selected.getIndexName()); + } + return selected; + } + + // convenience method which doesn't emit tracing messages + public Index getBestIndexFor(ReadCommand command) + { + return getBestIndexFor(command, false); + } + + /** + * Called at write time to ensure that values present in the update + * are valid according to the rules of all registered indexes which + * will process it. The partition key as well as the clustering and + * cell values for each row in the update may be checked by index + * implementations + * @param update PartitionUpdate containing the values to be validated by registered Index implementations + * @throws InvalidRequestException + */ + public void validate(PartitionUpdate update) throws InvalidRequestException + { + indexes.values() + .stream() + .filter(i -> i.indexes(update.columns())) + .forEach(i -> i.validate(update)); + } + + /** + * IndexRegistry methods + */ + public void registerIndex(Index index) + { + indexes.put(index.getIndexMetadata().name, index); + logger.debug("Registered index {}", index.getIndexMetadata().name); + } + + public void unregisterIndex(Index index) + { + Index removed = indexes.remove(index.getIndexMetadata().name); + logger.debug(removed == null ? "Index {} was not registered" : "Removed index {} from registry", + index.getIndexMetadata().name); + } + + public Index getIndex(IndexMetadata metadata) + { + return indexes.get(metadata.name); + } + + public Collection listIndexes() + { + return ImmutableSet.copyOf(indexes.values()); + } + + /** + * Handling of index updates. + * Implementations of the various IndexTransaction interfaces, for keeping indexes in sync with base data + * during updates, compaction and cleanup. Plus factory methods for obtaining transaction instances. + */ + + /** + * Transaction for updates on the write path. + */ + public UpdateTransaction newUpdateTransaction(PartitionUpdate update, OpOrder.Group opGroup, int nowInSec) + { + if (!hasIndexes()) + return UpdateTransaction.NO_OP; + + // todo : optimize lookup, we can probably cache quite a bit of stuff, rather than doing + // a linear scan every time. Holding off that though until CASSANDRA-7771 to figure out + // exactly how indexes are to be identified & associated with a given partition update + Index.Indexer[] indexers = indexes.values().stream() + .filter(i -> i.indexes(update.columns())) + .map(i -> i.indexerFor(update.partitionKey(), + nowInSec, + opGroup, + IndexTransaction.Type.UPDATE)) + .toArray(Index.Indexer[]::new); + + return indexers.length == 0 ? UpdateTransaction.NO_OP : new WriteTimeTransaction(indexers); + } + + /** + * Transaction for use when merging rows during compaction + */ + public CompactionTransaction newCompactionTransaction(DecoratedKey key, + PartitionColumns partitionColumns, + int versions, + int nowInSec) + { + // the check for whether there are any registered indexes is already done in CompactionIterator + + Index[] interestedIndexes = indexes.values().stream() + .filter(i -> i.indexes(partitionColumns)) + .toArray(Index[]::new); + + return interestedIndexes.length == 0 + ? CompactionTransaction.NO_OP + : new IndexGCTransaction(key, versions, nowInSec, interestedIndexes); + } + + /** + * Transaction for use when removing partitions during cleanup + */ + public CleanupTransaction newCleanupTransaction(DecoratedKey key, + PartitionColumns partitionColumns, + int nowInSec) + { + // + if (!hasIndexes()) + return CleanupTransaction.NO_OP; + + Index[] interestedIndexes = indexes.values().stream() + .filter(i -> i.indexes(partitionColumns)) + .toArray(Index[]::new); + + return interestedIndexes.length == 0 + ? CleanupTransaction.NO_OP + : new CleanupGCTransaction(key, nowInSec, interestedIndexes); + } + + /** + * A single use transaction for processing a partition update on the regular write path + */ + private static final class WriteTimeTransaction implements UpdateTransaction + { + private final Index.Indexer[] indexers; + + private WriteTimeTransaction(Index.Indexer...indexers) + { + // don't allow null indexers, if we don't need any use a NullUpdater object + for (Index.Indexer indexer : indexers) assert indexer != null; + this.indexers = indexers; + } + + public void start() + { + Arrays.stream(indexers).forEach(Index.Indexer::begin); + } + + public void onPartitionDeletion(DeletionTime deletionTime) + { + Arrays.stream(indexers).forEach(h -> h.partitionDelete(deletionTime)); + } + + public void onRangeTombstone(RangeTombstone tombstone) + { + Arrays.stream(indexers) .forEach(h -> h.rangeTombstone(tombstone)); + } + + public void onInserted(Row row) + { + Arrays.stream(indexers).forEach(h -> h.insertRow(row)); + } + + public void onUpdated(Row existing, Row updated) + { + final Row.Builder toRemove = BTreeRow.sortedBuilder(existing.columns()); + toRemove.newRow(existing.clustering()); + final Row.Builder toInsert = BTreeRow.sortedBuilder(updated.columns()); + toInsert.newRow(updated.clustering()); + // diff listener collates the columns to be added & removed from the indexes + RowDiffListener diffListener = new RowDiffListener() + { + public void onPrimaryKeyLivenessInfo(int i, Clustering clustering, LivenessInfo merged, LivenessInfo original) + { + if (merged != null && merged != original) + toInsert.addPrimaryKeyLivenessInfo(merged); + } + + public void onDeletion(int i, Clustering clustering, DeletionTime merged, DeletionTime original) + { + } + + public void onComplexDeletion(int i, Clustering clustering, ColumnDefinition column, DeletionTime merged, DeletionTime original) + { + } + + public void onCell(int i, Clustering clustering, Cell merged, Cell original) + { + if (merged != null && merged != original) + toInsert.addCell(merged); + + if (merged == null || (original != null && shouldCleanupOldValue(original, merged))) + toRemove.addCell(original); + + } + }; + Rows.diff(diffListener, updated, updated.columns().mergeTo(existing.columns()), existing); + Row oldRow = toRemove.build(); + Row newRow = toInsert.build(); + Arrays.stream(indexers).forEach(i -> i.updateRow(oldRow, newRow)); + } + + public void commit() + { + Arrays.stream(indexers).forEach(Index.Indexer::finish); + } + + private boolean shouldCleanupOldValue(Cell oldCell, Cell newCell) + { + // If either the value or timestamp is different, then we + // should delete from the index. If not, then we can infer that + // at least one of the cells is an ExpiringColumn and that the + // difference is in the expiry time. In this case, we don't want to + // delete the old value from the index as the tombstone we insert + // will just hide the inserted value. + // Completely identical cells (including expiring columns with + // identical ttl & localExpirationTime) will not get this far due + // to the oldCell.equals(newCell) in StandardUpdater.update + return !oldCell.value().equals(newCell.value()) || oldCell.timestamp() != newCell.timestamp(); + } + } + + /** + * A single-use transaction for updating indexes for a single partition during compaction where the only + * operation is to merge rows + * TODO : make this smarter at batching updates so we can use a single transaction to process multiple rows in + * a single partition + */ + private static final class IndexGCTransaction implements CompactionTransaction + { + private final DecoratedKey key; + private final int versions; + private final int nowInSec; + private final Index[] indexes; + + private Row[] rows; + + private IndexGCTransaction(DecoratedKey key, + int versions, + int nowInSec, + Index...indexes) + { + // don't allow null indexers, if we don't have any, use a noop transaction + for (Index index : indexes) assert index != null; + + this.key = key; + this.versions = versions; + this.indexes = indexes; + this.nowInSec = nowInSec; + } + + public void start() + { + if (versions > 0) + rows = new Row[versions]; + } + + public void onRowMerge(Columns columns, Row merged, Row...versions) + { + // Diff listener constructs rows representing deltas between the merged and original versions + // These delta rows are then passed to registered indexes for removal processing + final Row.Builder[] builders = new Row.Builder[versions.length]; + RowDiffListener diffListener = new RowDiffListener() + { + public void onPrimaryKeyLivenessInfo(int i, Clustering clustering, LivenessInfo merged, LivenessInfo original) + { + } + + public void onDeletion(int i, Clustering clustering, DeletionTime merged, DeletionTime original) + { + } + + public void onComplexDeletion(int i, Clustering clustering, ColumnDefinition column, DeletionTime merged, DeletionTime original) + { + } + + public void onCell(int i, Clustering clustering, Cell merged, Cell original) + { + if (original != null && merged == null) + { + if (builders[i] == null) + { + builders[i] = BTreeRow.sortedBuilder(columns); + builders[i].newRow(clustering); + } + builders[i].addCell(original); + } + } + }; + + Rows.diff(diffListener, merged, columns, versions); + + for(int i = 0; i < builders.length; i++) + if (builders[i] != null) + rows[i] = builders[i].build(); + } + + public void commit() + { + if (rows == null) + return; + + try (OpOrder.Group opGroup = Keyspace.writeOrder.start()) + { + Index.Indexer[] indexers = Arrays.stream(indexes) + .map(i -> i.indexerFor(key, nowInSec, opGroup, Type.COMPACTION)) + .toArray(Index.Indexer[]::new); + + Arrays.stream(indexers).forEach(Index.Indexer::begin); + + for (Row row : rows) + if (row != null) + Arrays.stream(indexers).forEach(indexer -> indexer.removeRow(row)); + + Arrays.stream(indexers).forEach(Index.Indexer::finish); + } + } + } + + /** + * A single-use transaction for updating indexes for a single partition during cleanup, where + * partitions and rows are only removed + * TODO : make this smarter at batching updates so we can use a single transaction to process multiple rows in + * a single partition + */ + private static final class CleanupGCTransaction implements CleanupTransaction + { + private final DecoratedKey key; + private final int nowInSec; + private final Index[] indexes; + + private Row row; + private DeletionTime partitionDelete; + + private CleanupGCTransaction(DecoratedKey key, + int nowInSec, + Index...indexes) + { + // don't allow null indexers, if we don't have any, use a noop transaction + for (Index index : indexes) assert index != null; + + this.key = key; + this.indexes = indexes; + this.nowInSec = nowInSec; + } + + public void start() + { + } + + public void onPartitionDeletion(DeletionTime deletionTime) + { + partitionDelete = deletionTime; + } + + public void onRowDelete(Row row) + { + this.row = row; + } + + public void commit() + { + if (row == null && partitionDelete == null) + return; + + try (OpOrder.Group opGroup = Keyspace.writeOrder.start()) + { + Index.Indexer[] indexers = Arrays.stream(indexes) + .map(i -> i.indexerFor(key, nowInSec, opGroup, Type.CLEANUP)) + .toArray(Index.Indexer[]::new); + + Arrays.stream(indexers).forEach(Index.Indexer::begin); + + if (partitionDelete != null) + Arrays.stream(indexers).forEach(indexer -> indexer.partitionDelete(partitionDelete)); + + if (row != null) + Arrays.stream(indexers).forEach(indexer -> indexer.removeRow(row)); + + Arrays.stream(indexers).forEach(Index.Indexer::finish); + } + } + } + + private static void executeBlocking(Callable task) + { + if (null != task) + FBUtilities.waitOnFuture(blockingExecutor.submit(task)); + } + + private static void executeAllBlocking(Stream indexers, Function> function) + { + List> waitFor = new ArrayList<>(); + indexers.forEach(indexer -> { + Callable task = function.apply(indexer); + if (null != task) + waitFor.add(blockingExecutor.submit(task)); + }); + FBUtilities.waitOnFutures(waitFor); + } +} diff --git a/src/java/org/apache/cassandra/index/internal/CassandraIndex.java b/src/java/org/apache/cassandra/index/internal/CassandraIndex.java new file mode 100644 index 0000000000..b17ab4e6d4 --- /dev/null +++ b/src/java/org/apache/cassandra/index/internal/CassandraIndex.java @@ -0,0 +1,755 @@ +package org.apache.cassandra.index.internal; + +import java.nio.ByteBuffer; +import java.util.*; +import java.util.concurrent.Callable; +import java.util.concurrent.Future; +import java.util.function.BiFunction; +import java.util.stream.Collectors; +import java.util.stream.StreamSupport; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.cql3.Operator; +import org.apache.cassandra.cql3.statements.IndexTarget; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.compaction.CompactionManager; +import org.apache.cassandra.db.filter.RowFilter; +import org.apache.cassandra.db.lifecycle.SSTableSet; +import org.apache.cassandra.db.lifecycle.View; +import org.apache.cassandra.db.marshal.AbstractType; +import org.apache.cassandra.db.marshal.CollectionType; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.dht.LocalPartitioner; +import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.index.IndexRegistry; +import org.apache.cassandra.index.SecondaryIndexBuilder; +import org.apache.cassandra.index.internal.composites.CompositesSearcher; +import org.apache.cassandra.index.internal.keys.KeysSearcher; +import org.apache.cassandra.index.transactions.IndexTransaction; +import org.apache.cassandra.index.transactions.UpdateTransaction; +import org.apache.cassandra.io.sstable.ReducingKeyIterator; +import org.apache.cassandra.io.sstable.format.SSTableReader; +import org.apache.cassandra.schema.IndexMetadata; +import org.apache.cassandra.utils.FBUtilities; +import org.apache.cassandra.utils.concurrent.OpOrder; +import org.apache.cassandra.utils.concurrent.Refs; + +/** + * Index implementation which indexes the values for a single column in the base + * table and which stores its index data in a local, hidden table. + */ +public abstract class CassandraIndex implements Index +{ + private static final Logger logger = LoggerFactory.getLogger(CassandraIndex.class); + + public final ColumnFamilyStore baseCfs; + protected IndexMetadata metadata; + protected ColumnFamilyStore indexCfs; + protected ColumnDefinition indexedColumn; + protected CassandraIndexFunctions functions; + + public CassandraIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) + { + this.baseCfs = baseCfs; + setMetadata(indexDef); + } + + /** + * Returns true if an index of this type can support search predicates of the form [column] OPERATOR [value] + * @param indexedColumn + * @param operator + * @return + */ + protected boolean supportsOperator(ColumnDefinition indexedColumn, Operator operator) + { + return operator.equals(Operator.EQ); + } + + /** + * Used to construct an the clustering for an entry in the index table based on values from the base data. + * The clustering columns in the index table encode the values required to retrieve the correct data from the base + * table and varies depending on the kind of the indexed column. See indexCfsMetadata for more details + * Used whenever a row in the index table is written or deleted. + * @param metadata + * @param partitionKey from the base data being indexed + * @param prefix from the base data being indexed + * @param path from the base data being indexed + * @return a clustering prefix to be used to insert into the index table + */ + protected abstract CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path); + + /** + * Used at search time to convert a row in the index table into a simple struct containing the values required + * to retrieve the corresponding row from the base table. + * @param metadata + * @param indexedValue the partition key of the indexed table (i.e. the value that was indexed) + * @param indexEntry a row from the index table + * @return + */ + public abstract IndexEntry decodeEntry(DecoratedKey indexedValue, + Row indexEntry); + + /** + * Check whether a value retrieved from an index is still valid by comparing it to current row from the base table. + * Used at read time to identify out of date index entries so that they can be excluded from search results and + * repaired + * @param metadata required to get the indexed column definition + * @param row the current row from the primary data table + * @param indexValue the value we retrieved from the index + * @param nowInSec + * @return true if the index is out of date and the entry should be dropped + */ + public abstract boolean isStale(Row row, ByteBuffer indexValue, int nowInSec); + + /** + * Extract the value to be inserted into the index from the components of the base data + * @param metadata + * @param partitionKey from the primary data + * @param clustering from the primary data + * @param path from the primary data + * @param cellValue from the primary data + * @return a ByteBuffer containing the value to be inserted in the index. This will be used to make the partition + * key in the index table + */ + protected abstract ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, + ByteBuffer cellValue); + + public ColumnDefinition getIndexedColumn() + { + return indexedColumn; + } + + public ClusteringComparator getIndexComparator() + { + return indexCfs.metadata.comparator; + } + + public ColumnFamilyStore getIndexCfs() + { + return indexCfs; + } + + public void register(IndexRegistry registry) + { + registry.registerIndex(this); + } + + public Callable getInitializationTask() + { + // if we're just linking in the index on an already-built index post-restart + // we've nothing to do. Otherwise, submit for building via SecondaryIndexBuilder + return isBuilt() ? null : getBuildIndexTask(); + } + + public IndexMetadata getIndexMetadata() + { + return metadata; + } + + public String getIndexName() + { + // should return metadata.name, see CASSANDRA-10127 + return indexCfs.name; + } + + public Optional getBackingTable() + { + return indexCfs == null ? Optional.empty() : Optional.of(indexCfs); + } + + public Callable getBlockingFlushTask() + { + return () -> { + indexCfs.forceBlockingFlush(); + return null; + }; + } + + public Callable getInvalidateTask() + { + return () -> { + markRemoved(); + invalidate(); + return null; + }; + } + + public Callable getMetadataReloadTask(IndexMetadata indexDef) + { + setMetadata(indexDef); + return () -> { + indexCfs.metadata.reloadIndexMetadataProperties(baseCfs.metadata); + indexCfs.reload(); + return null; + }; + } + + private void setMetadata(IndexMetadata indexDef) + { + metadata = indexDef; + functions = getFunctions(baseCfs.metadata, indexDef); + CFMetaData cfm = indexCfsMetadata(baseCfs.metadata, indexDef); + indexCfs = ColumnFamilyStore.createColumnFamilyStore(baseCfs.keyspace, + cfm.cfName, + cfm, + baseCfs.getTracker().loadsstables); + assert indexDef.columns.size() == 1 : "Build in indexes on multiple target columns are not supported"; + indexedColumn = indexDef.indexedColumn(baseCfs.metadata); + } + + public Callable getTruncateTask(final long truncatedAt) + { + return () -> { + indexCfs.discardSSTables(truncatedAt); + return null; + }; + } + + public boolean indexes(PartitionColumns columns) + { + // if we have indexes on the partition key or clustering columns, return true + return isPrimaryKeyIndex() || columns.contains(indexedColumn); + } + + public boolean supportsExpression(ColumnDefinition column, Operator operator) + { + return indexedColumn.name.equals(column.name) + && supportsOperator(indexedColumn, operator); + } + + private boolean supportsExpression(RowFilter.Expression expression) + { + return supportsExpression(expression.column(), expression.operator()); + } + + public long getEstimatedResultRows() + { + return indexCfs.getMeanColumns(); + } + + /** + * No post processing of query results, just return them unchanged + */ + public BiFunction postProcessorFor(ReadCommand command) + { + return (partitionIterator, readCommand) -> partitionIterator; + } + + public RowFilter getPostIndexQueryFilter(RowFilter filter) + { + return getTargetExpression(filter.getExpressions()).map(filter::without) + .orElse(filter); + } + + private Optional getTargetExpression(List expressions) + { + return expressions.stream().filter(this::supportsExpression).findFirst(); + } + + public Index.Searcher searcherFor(ReadCommand command) + { + Optional target = getTargetExpression(command.rowFilter().getExpressions()); + + if (target.isPresent()) + { + target.get().validateForIndexing(); + switch (getIndexMetadata().indexType) + { + case COMPOSITES: + return new CompositesSearcher(command, target.get(), this); + case KEYS: + return new KeysSearcher(command, target.get(), this); + default: + throw new IllegalStateException(String.format("Unsupported index type %s for index %s on %s", + metadata.indexType, + metadata.name, + indexedColumn.name.toString())); + } + } + + return null; + + } + + public void validate(PartitionUpdate update) throws InvalidRequestException + { + switch (indexedColumn.kind) + { + case PARTITION_KEY: + validatePartitionKey(update.partitionKey()); + break; + case CLUSTERING: + validateClusterings(update); + break; + case REGULAR: + validateRows(update); + break; + case STATIC: + validateRows(Collections.singleton(update.staticRow())); + break; + } + } + + public Indexer indexerFor(final DecoratedKey key, + final int nowInSec, + final OpOrder.Group opGroup, + final IndexTransaction.Type transactionType) + { + return new Indexer() + { + public void begin() + { + } + + public void partitionDelete(DeletionTime deletionTime) + { + } + + public void rangeTombstone(RangeTombstone tombstone) + { + } + + public void insertRow(Row row) + { + if (isPrimaryKeyIndex()) + { + indexPrimaryKey(row.clustering(), + getPrimaryKeyIndexLiveness(row), + row.deletion()); + } + else + { + if (indexedColumn.isComplex()) + indexCells(row.clustering(), row.getComplexColumnData(indexedColumn)); + else + indexCell(row.clustering(), row.getCell(indexedColumn)); + } + } + + public void removeRow(Row row) + { + if (isPrimaryKeyIndex()) + indexPrimaryKey(row.clustering(), row.primaryKeyLivenessInfo(), row.deletion()); + + if (indexedColumn.isComplex()) + removeCells(row.clustering(), row.getComplexColumnData(indexedColumn)); + else + removeCell(row.clustering(), row.getCell(indexedColumn)); + } + + + public void updateRow(Row oldRow, Row newRow) + { + if (isPrimaryKeyIndex()) + indexPrimaryKey(newRow.clustering(), + newRow.primaryKeyLivenessInfo(), + newRow.deletion()); + + if (indexedColumn.isComplex()) + { + indexCells(newRow.clustering(), newRow.getComplexColumnData(indexedColumn)); + removeCells(oldRow.clustering(), oldRow.getComplexColumnData(indexedColumn)); + } + else + { + indexCell(newRow.clustering(), newRow.getCell(indexedColumn)); + removeCell(oldRow.clustering(), oldRow.getCell(indexedColumn)); + } + } + + public void finish() + { + } + + private void indexCells(Clustering clustering, Iterable cells) + { + if (cells == null) + return; + + for (Cell cell : cells) + indexCell(clustering, cell); + } + + private void indexCell(Clustering clustering, Cell cell) + { + if (cell == null || !cell.isLive(nowInSec)) + return; + + insert(key.getKey(), + clustering, + cell, + LivenessInfo.create(cell.timestamp(), cell.ttl(), cell.localDeletionTime()), + opGroup); + } + + private void removeCells(Clustering clustering, Iterable cells) + { + if (cells == null) + return; + + for (Cell cell : cells) + removeCell(clustering, cell); + } + + private void removeCell(Clustering clustering, Cell cell) + { + if (cell == null || !cell.isLive(nowInSec)) + return; + + delete(key.getKey(), clustering, cell, opGroup, nowInSec); + } + + private void indexPrimaryKey(final Clustering clustering, + final LivenessInfo liveness, + final DeletionTime deletion) + { + if (liveness.timestamp() != LivenessInfo.NO_TIMESTAMP) + insert(key.getKey(), clustering, null, liveness, opGroup); + + if (!deletion.isLive()) + delete(key.getKey(), clustering, deletion, opGroup); + } + + private LivenessInfo getPrimaryKeyIndexLiveness(Row row) + { + long timestamp = row.primaryKeyLivenessInfo().timestamp(); + int ttl = row.primaryKeyLivenessInfo().ttl(); + for (Cell cell : row.cells()) + { + long cellTimestamp = cell.timestamp(); + if (cell.isLive(nowInSec)) + { + if (cellTimestamp > timestamp) + { + timestamp = cellTimestamp; + ttl = cell.ttl(); + } + } + } + return LivenessInfo.create(baseCfs.metadata, timestamp, ttl, nowInSec); + } + }; + } + + /** + * Specific to internal indexes, this is called by a + * searcher when it encounters a stale entry in the index + * @param indexKey the partition key in the index table + * @param indexClustering the clustering in the index table + * @param deletion deletion timestamp etc + * @param opGroup the operation under which to perform the deletion + */ + public void deleteStaleEntry(DecoratedKey indexKey, + Clustering indexClustering, + DeletionTime deletion, + OpOrder.Group opGroup) + { + doDelete(indexKey, indexClustering, deletion, opGroup); + logger.debug("Removed index entry for stale value {}", indexKey); + } + + /** + * Called when adding a new entry to the index + */ + private void insert(ByteBuffer rowKey, + Clustering clustering, + Cell cell, + LivenessInfo info, + OpOrder.Group opGroup) + { + DecoratedKey valueKey = getIndexKeyFor(getIndexedValue(rowKey, + clustering, + cell)); + Row row = BTreeRow.noCellLiveRow(buildIndexClustering(rowKey, clustering, cell), info); + PartitionUpdate upd = partitionUpdate(valueKey, row); + indexCfs.apply(upd, UpdateTransaction.NO_OP, opGroup, null); + logger.debug("Inserted entry into index for value {}", valueKey); + } + + /** + * Called when deleting entries on non-primary key columns + */ + private void delete(ByteBuffer rowKey, + Clustering clustering, + Cell cell, + OpOrder.Group opGroup, + int nowInSec) + { + DecoratedKey valueKey = getIndexKeyFor(getIndexedValue(rowKey, + clustering, + cell)); + doDelete(valueKey, + buildIndexClustering(rowKey, clustering, cell), + new DeletionTime(cell.timestamp(), nowInSec), + opGroup); + } + + /** + * Called when deleting entries from indexes on primary key columns + */ + private void delete(ByteBuffer rowKey, + Clustering clustering, + DeletionTime deletion, + OpOrder.Group opGroup) + { + DecoratedKey valueKey = getIndexKeyFor(getIndexedValue(rowKey, + clustering, + null)); + doDelete(valueKey, + buildIndexClustering(rowKey, clustering, null), + deletion, + opGroup); + } + + private void doDelete(DecoratedKey indexKey, + Clustering indexClustering, + DeletionTime deletion, + OpOrder.Group opGroup) + { + Row row = BTreeRow.emptyDeletedRow(indexClustering, deletion); + PartitionUpdate upd = partitionUpdate(indexKey, row); + indexCfs.apply(upd, UpdateTransaction.NO_OP, opGroup, null); + logger.debug("Removed index entry for value {}", indexKey); + } + + private void validatePartitionKey(DecoratedKey partitionKey) throws InvalidRequestException + { + assert indexedColumn.isPartitionKey(); + validateIndexedValue(getIndexedValue(partitionKey.getKey(), null, null )); + } + + private void validateClusterings(PartitionUpdate update) throws InvalidRequestException + { + assert indexedColumn.isClusteringColumn(); + for (Row row : update) + validateIndexedValue(getIndexedValue(null, row.clustering(), null)); + } + + private void validateRows(Iterable rows) + { + assert !indexedColumn.isPrimaryKeyColumn(); + for (Row row : rows) + { + if (indexedColumn.isComplex()) + { + ComplexColumnData data = row.getComplexColumnData(indexedColumn); + if (data != null) + { + for (Cell cell : data) + { + validateIndexedValue(getIndexedValue(null, null, cell.path(), cell.value())); + } + } + } + else + { + validateIndexedValue(getIndexedValue(null, null, row.getCell(indexedColumn))); + } + } + } + + private void validateIndexedValue(ByteBuffer value) + { + if (value != null && value.remaining() >= FBUtilities.MAX_UNSIGNED_SHORT) + throw new InvalidRequestException(String.format( + "Cannot index value of size %d for index %s on %s.%s(%s) (maximum allowed size=%d)", + value.remaining(), + getIndexName(), + baseCfs.metadata.ksName, + baseCfs.metadata.cfName, + indexedColumn.name.toString(), + FBUtilities.MAX_UNSIGNED_SHORT)); + } + + private ByteBuffer getIndexedValue(ByteBuffer rowKey, + Clustering clustering, + Cell cell) + { + return getIndexedValue(rowKey, + clustering, + cell == null ? null : cell.path(), + cell == null ? null : cell.value() + ); + } + + private Clustering buildIndexClustering(ByteBuffer rowKey, + Clustering clustering, + Cell cell) + { + return buildIndexClusteringPrefix(rowKey, + clustering, + cell == null ? null : cell.path()).build(); + } + + private DecoratedKey getIndexKeyFor(ByteBuffer value) + { + return indexCfs.decorateKey(value); + } + + private PartitionUpdate partitionUpdate(DecoratedKey valueKey, Row row) + { + return PartitionUpdate.singleRowUpdate(indexCfs.metadata, valueKey, row); + } + + private void invalidate() + { + // interrupt in-progress compactions + Collection cfss = Collections.singleton(indexCfs); + CompactionManager.instance.interruptCompactionForCFs(cfss, true); + CompactionManager.instance.waitForCessation(cfss); + indexCfs.keyspace.writeOrder.awaitNewBarrier(); + indexCfs.forceBlockingFlush(); + indexCfs.readOrdering.awaitNewBarrier(); + indexCfs.invalidate(); + } + + private boolean isBuilt() + { + return SystemKeyspace.isIndexBuilt(baseCfs.keyspace.getName(), getIndexName()); + } + + private void markBuilt() + { + SystemKeyspace.setIndexBuilt(baseCfs.keyspace.getName(), getIndexName()); + } + + private void markRemoved() + { + SystemKeyspace.setIndexRemoved(baseCfs.keyspace.getName(), getIndexName()); + } + + private boolean isPrimaryKeyIndex() + { + return indexedColumn.isPrimaryKeyColumn(); + } + + private Callable getBuildIndexTask() + { + return () -> { + buildBlocking(); + return null; + }; + } + + private void buildBlocking() + { + baseCfs.forceBlockingFlush(); + + try (ColumnFamilyStore.RefViewFragment viewFragment = baseCfs.selectAndReference(View.select(SSTableSet.CANONICAL)); + Refs sstables = viewFragment.refs) + { + if (sstables.isEmpty()) + { + logger.info("No SSTable data for {}.{} to build index {} from, marking empty index as built", + baseCfs.metadata.ksName, + baseCfs.metadata.cfName, + getIndexName()); + markBuilt(); + return; + } + + logger.info("Submitting index build of {} for data in {}", + getIndexName(), + getSSTableNames(sstables)); + + SecondaryIndexBuilder builder = new SecondaryIndexBuilder(baseCfs, + Collections.singleton(this), + new ReducingKeyIterator(sstables)); + Future future = CompactionManager.instance.submitIndexBuild(builder); + FBUtilities.waitOnFuture(future); + indexCfs.forceBlockingFlush(); + markBuilt(); + } + logger.info("Index build of {} complete", getIndexName()); + } + + private static String getSSTableNames(Collection sstables) + { + return StreamSupport.stream(sstables.spliterator(), false) + .map(SSTableReader::toString) + .collect(Collectors.joining(", ")); + } + + /** + * Construct the CFMetadata for an index table, the clustering columns in the index table + * vary dependent on the kind of the indexed value. + * @param baseCfsMetadata + * @param indexMetadata + * @return + */ + public static final CFMetaData indexCfsMetadata(CFMetaData baseCfsMetadata, IndexMetadata indexMetadata) + { + CassandraIndexFunctions utils = getFunctions(baseCfsMetadata, indexMetadata); + ColumnDefinition indexedColumn = indexMetadata.indexedColumn(baseCfsMetadata); + AbstractType indexedValueType = utils.getIndexedValueType(indexedColumn); + CFMetaData.Builder builder = CFMetaData.Builder.create(baseCfsMetadata.ksName, + baseCfsMetadata.indexColumnFamilyName(indexMetadata)) + .withId(baseCfsMetadata.cfId) + .withPartitioner(new LocalPartitioner(indexedValueType)) + .addPartitionKey(indexedColumn.name, indexedColumn.type); + + builder.addClusteringColumn("partition_key", baseCfsMetadata.partitioner.partitionOrdering()); + builder = utils.addIndexClusteringColumns(builder, baseCfsMetadata, indexedColumn); + return builder.build().reloadIndexMetadataProperties(baseCfsMetadata); + } + + /** + * Factory method for new CassandraIndex instances + * @param baseCfs + * @param indexMetadata + * @return + */ + public static final CassandraIndex newIndex(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return getFunctions(baseCfs.metadata, indexMetadata).newIndexInstance(baseCfs, indexMetadata); + } + + private static CassandraIndexFunctions getFunctions(CFMetaData baseCfMetadata, IndexMetadata indexDef) + { + if (indexDef.isKeys()) + return CassandraIndexFunctions.KEYS_INDEX_FUNCTIONS; + + ColumnDefinition indexedColumn = indexDef.indexedColumn(baseCfMetadata); + if (indexedColumn.type.isCollection() && indexedColumn.type.isMultiCell()) + { + switch (((CollectionType)indexedColumn.type).kind) + { + case LIST: + return CassandraIndexFunctions.COLLECTION_VALUE_INDEX_FUNCTIONS; + case SET: + return CassandraIndexFunctions.COLLECTION_KEY_INDEX_FUNCTIONS; + case MAP: + if (indexDef.options.containsKey(IndexTarget.INDEX_KEYS_OPTION_NAME)) + return CassandraIndexFunctions.COLLECTION_KEY_INDEX_FUNCTIONS; + else if (indexDef.options.containsKey(IndexTarget.INDEX_ENTRIES_OPTION_NAME)) + return CassandraIndexFunctions.COLLECTION_ENTRY_INDEX_FUNCTIONS; + else + return CassandraIndexFunctions.COLLECTION_VALUE_INDEX_FUNCTIONS; + } + } + + switch (indexedColumn.kind) + { + case CLUSTERING: + return CassandraIndexFunctions.CLUSTERING_COLUMN_INDEX_FUNCTIONS; + case REGULAR: + return CassandraIndexFunctions.REGULAR_COLUMN_INDEX_FUNCTIONS; + case PARTITION_KEY: + return CassandraIndexFunctions.PARTITION_KEY_INDEX_FUNCTIONS; + //case COMPACT_VALUE: + // return new CompositesIndexOnCompactValue(); + } + throw new AssertionError(); + } +} diff --git a/src/java/org/apache/cassandra/index/internal/CassandraIndexFunctions.java b/src/java/org/apache/cassandra/index/internal/CassandraIndexFunctions.java new file mode 100644 index 0000000000..b7cb3a233d --- /dev/null +++ b/src/java/org/apache/cassandra/index/internal/CassandraIndexFunctions.java @@ -0,0 +1,186 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index.internal; + +import java.util.List; + +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.db.ColumnFamilyStore; +import org.apache.cassandra.db.marshal.AbstractType; +import org.apache.cassandra.db.marshal.CollectionType; +import org.apache.cassandra.db.marshal.CompositeType; +import org.apache.cassandra.index.internal.composites.*; +import org.apache.cassandra.index.internal.keys.KeysIndex; +import org.apache.cassandra.schema.IndexMetadata; + +public interface CassandraIndexFunctions +{ + /** + * + * @param baseCfs + * @param indexMetadata + * @return + */ + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata); + + /** + * Returns the type of the the values in the index. For most columns this is simply its type, but for collections + * it depends on whether the index is on the collection name/value element or on a frozen collection + * @param indexedColumn + * @return + */ + default AbstractType getIndexedValueType(ColumnDefinition indexedColumn) + { + return indexedColumn.type; + } + + /** + * Add the clustering columns for a specific type of index table to the a CFMetaData.Builder (which is being + * used to construct the index table's CFMetadata. In the default implementation, the clustering columns of the + * index table hold the partition key & clustering columns of the base table. This is overridden in several cases: + * * When the indexed value is itself a clustering column, in which case, we only need store the base table's + * *other* clustering values in the index - the indexed value being the index table's partition key + * * When the indexed value is a collection value, in which case we also need to capture the cell path from the base + * table + * * In a KEYS index (for thrift/compact storage/static column indexes), where only the base partition key is + * held in the index table. + * + * Called from indexCfsMetadata + * @param builder + * @param baseMetadata + * @param cfDef + * @return + */ + default CFMetaData.Builder addIndexClusteringColumns(CFMetaData.Builder builder, + CFMetaData baseMetadata, + ColumnDefinition cfDef) + { + for (ColumnDefinition def : baseMetadata.clusteringColumns()) + builder.addClusteringColumn(def.name, def.type); + return builder; + } + + /* + * implementations providing specializations for the built in index types + */ + + static final CassandraIndexFunctions KEYS_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new KeysIndex(baseCfs, indexMetadata); + } + }; + + static final CassandraIndexFunctions REGULAR_COLUMN_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new RegularColumnIndex(baseCfs, indexMetadata); + } + }; + + static final CassandraIndexFunctions CLUSTERING_COLUMN_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new ClusteringColumnIndex(baseCfs, indexMetadata); + } + + public CFMetaData.Builder addIndexClusteringColumns(CFMetaData.Builder builder, + CFMetaData baseMetadata, + ColumnDefinition columnDef) + { + List cks = baseMetadata.clusteringColumns(); + for (int i = 0; i < columnDef.position(); i++) + { + ColumnDefinition def = cks.get(i); + builder.addClusteringColumn(def.name, def.type); + } + for (int i = columnDef.position() + 1; i < cks.size(); i++) + { + ColumnDefinition def = cks.get(i); + builder.addClusteringColumn(def.name, def.type); + } + return builder; + } + }; + + static final CassandraIndexFunctions PARTITION_KEY_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new PartitionKeyIndex(baseCfs, indexMetadata); + } + }; + + static final CassandraIndexFunctions COLLECTION_KEY_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new CollectionKeyIndex(baseCfs, indexMetadata); + } + + public AbstractType getIndexedValueType(ColumnDefinition indexedColumn) + { + return ((CollectionType) indexedColumn.type).nameComparator(); + } + }; + + static final CassandraIndexFunctions COLLECTION_VALUE_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new CollectionValueIndex(baseCfs, indexMetadata); + } + + public AbstractType getIndexedValueType(ColumnDefinition indexedColumn) + { + return ((CollectionType)indexedColumn.type).valueComparator(); + } + + public CFMetaData.Builder addIndexClusteringColumns(CFMetaData.Builder builder, + CFMetaData baseMetadata, + ColumnDefinition columnDef) + { + for (ColumnDefinition def : baseMetadata.clusteringColumns()) + builder.addClusteringColumn(def.name, def.type); + + // collection key + builder.addClusteringColumn("cell_path", ((CollectionType)columnDef.type).nameComparator()); + return builder; + } + }; + + static final CassandraIndexFunctions COLLECTION_ENTRY_INDEX_FUNCTIONS = new CassandraIndexFunctions() + { + public CassandraIndex newIndexInstance(ColumnFamilyStore baseCfs, IndexMetadata indexMetadata) + { + return new CollectionEntryIndex(baseCfs, indexMetadata); + } + + public AbstractType getIndexedValueType(ColumnDefinition indexedColumn) + { + CollectionType colType = (CollectionType)indexedColumn.type; + return CompositeType.getInstance(colType.nameComparator(), colType.valueComparator()); + } + }; +} diff --git a/src/java/org/apache/cassandra/index/internal/CassandraIndexSearcher.java b/src/java/org/apache/cassandra/index/internal/CassandraIndexSearcher.java new file mode 100644 index 0000000000..72d252827a --- /dev/null +++ b/src/java/org/apache/cassandra/index/internal/CassandraIndexSearcher.java @@ -0,0 +1,172 @@ +package org.apache.cassandra.index.internal; + +import java.nio.ByteBuffer; +import java.util.NavigableSet; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.filter.*; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.rows.RowIterator; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; +import org.apache.cassandra.db.rows.UnfilteredRowIterators; +import org.apache.cassandra.dht.AbstractBounds; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.utils.btree.BTreeSet; + +public abstract class CassandraIndexSearcher implements Index.Searcher +{ + private static final Logger logger = LoggerFactory.getLogger(CassandraIndexSearcher.class); + + private final RowFilter.Expression expression; + protected final CassandraIndex index; + protected final ReadCommand command; + + public CassandraIndexSearcher(ReadCommand command, + RowFilter.Expression expression, + CassandraIndex index) + { + this.command = command; + this.expression = expression; + this.index = index; + } + + @SuppressWarnings("resource") // Both the OpOrder and 'indexIter' are closed on exception, or through the closing of the result + // of this method. + public UnfilteredPartitionIterator search(ReadOrderGroup orderGroup) + { + // the value of the index expression is the partition key in the index table + DecoratedKey indexKey = index.getBackingTable().get().decorateKey(expression.getIndexValue()); + UnfilteredRowIterator indexIter = queryIndex(indexKey, command, orderGroup); + try + { + return queryDataFromIndex(indexKey, UnfilteredRowIterators.filter(indexIter, command.nowInSec()), command, orderGroup); + } + catch (RuntimeException | Error e) + { + indexIter.close(); + throw e; + } + } + + private UnfilteredRowIterator queryIndex(DecoratedKey indexKey, ReadCommand command, ReadOrderGroup orderGroup) + { + ClusteringIndexFilter filter = makeIndexFilter(command); + ColumnFamilyStore indexCfs = index.getBackingTable().get(); + CFMetaData indexCfm = indexCfs.metadata; + return SinglePartitionReadCommand.create(indexCfm, command.nowInSec(), indexKey, ColumnFilter.all(indexCfm), filter) + .queryMemtableAndDisk(indexCfs, orderGroup.indexReadOpOrderGroup()); + } + + private ClusteringIndexFilter makeIndexFilter(ReadCommand command) + { + if (command instanceof SinglePartitionReadCommand) + { + // Note: as yet there's no route to get here - a 2i query *always* uses a + // PartitionRangeReadCommand. This is here in preparation for coming changes + // in SelectStatement. + SinglePartitionReadCommand sprc = (SinglePartitionReadCommand)command; + ByteBuffer pk = sprc.partitionKey().getKey(); + ClusteringIndexFilter filter = sprc.clusteringIndexFilter(); + + if (filter instanceof ClusteringIndexNamesFilter) + { + NavigableSet requested = ((ClusteringIndexNamesFilter)filter).requestedRows(); + BTreeSet.Builder clusterings = BTreeSet.builder(index.getIndexComparator()); + for (Clustering c : requested) + clusterings.add(makeIndexClustering(pk, c)); + return new ClusteringIndexNamesFilter(clusterings.build(), filter.isReversed()); + } + else + { + Slices requested = ((ClusteringIndexSliceFilter)filter).requestedSlices(); + Slices.Builder builder = new Slices.Builder(index.getIndexComparator()); + for (Slice slice : requested) + builder.add(makeIndexBound(pk, slice.start()), makeIndexBound(pk, slice.end())); + return new ClusteringIndexSliceFilter(builder.build(), filter.isReversed()); + } + } + else + { + + DataRange dataRange = ((PartitionRangeReadCommand)command).dataRange(); + AbstractBounds range = dataRange.keyRange(); + + Slice slice = Slice.ALL; + + /* + * XXX: If the range requested is a token range, we'll have to start at the beginning (and stop at the end) of + * the indexed row unfortunately (which will be inefficient), because we have no way to intuit the smallest possible + * key having a given token. A potential fix would be to actually store the token along the key in the indexed row. + */ + if (range.left instanceof DecoratedKey) + { + // the right hand side of the range may not be a DecoratedKey (for instance if we're paging), + // but if it is, we can optimise slightly by restricting the slice + if (range.right instanceof DecoratedKey) + { + + DecoratedKey startKey = (DecoratedKey) range.left; + DecoratedKey endKey = (DecoratedKey) range.right; + + Slice.Bound start = Slice.Bound.BOTTOM; + Slice.Bound end = Slice.Bound.TOP; + + /* + * For index queries over a range, we can't do a whole lot better than querying everything for the key range, though for + * slice queries where we can slightly restrict the beginning and end. + */ + if (!dataRange.isNamesQuery()) + { + ClusteringIndexSliceFilter startSliceFilter = ((ClusteringIndexSliceFilter) dataRange.clusteringIndexFilter( + startKey)); + ClusteringIndexSliceFilter endSliceFilter = ((ClusteringIndexSliceFilter) dataRange.clusteringIndexFilter( + endKey)); + + // We can't effectively support reversed queries when we have a range, so we don't support it + // (or through post-query reordering) and shouldn't get there. + assert !startSliceFilter.isReversed() && !endSliceFilter.isReversed(); + + Slices startSlices = startSliceFilter.requestedSlices(); + Slices endSlices = endSliceFilter.requestedSlices(); + + if (startSlices.size() > 0) + start = startSlices.get(0).start(); + + if (endSlices.size() > 0) + end = endSlices.get(endSlices.size() - 1).end(); + } + + slice = Slice.make(makeIndexBound(startKey.getKey(), start), + makeIndexBound(endKey.getKey(), end)); + } + else + { + // otherwise, just start the index slice from the key we do have + slice = Slice.make(makeIndexBound(((DecoratedKey)range.left).getKey(), Slice.Bound.BOTTOM), + Slice.Bound.TOP); + } + } + return new ClusteringIndexSliceFilter(Slices.with(index.getIndexComparator(), slice), false); + } + } + + private Slice.Bound makeIndexBound(ByteBuffer rowKey, Slice.Bound bound) + { + return index.buildIndexClusteringPrefix(rowKey, bound, null) + .buildBound(bound.isStart(), bound.isInclusive()); + } + + protected Clustering makeIndexClustering(ByteBuffer rowKey, Clustering clustering) + { + return index.buildIndexClusteringPrefix(rowKey, clustering, null).build(); + } + + protected abstract UnfilteredPartitionIterator queryDataFromIndex(DecoratedKey indexKey, + RowIterator indexHits, + ReadCommand command, + ReadOrderGroup orderGroup); +} diff --git a/src/java/org/apache/cassandra/index/internal/IndexEntry.java b/src/java/org/apache/cassandra/index/internal/IndexEntry.java new file mode 100644 index 0000000000..6f94ace57e --- /dev/null +++ b/src/java/org/apache/cassandra/index/internal/IndexEntry.java @@ -0,0 +1,34 @@ +package org.apache.cassandra.index.internal; + +import java.nio.ByteBuffer; + +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.DecoratedKey; + +/** + * Entries in indexes on non-compact tables (tables with composite comparators) + * can be encapsulated as IndexedEntry instances. These are not used when dealing + * with indexes on static/compact/thrift tables (i.e. KEYS indexes). + */ +public final class IndexEntry +{ + public final DecoratedKey indexValue; + public final Clustering indexClustering; + public final long timestamp; + + public final ByteBuffer indexedKey; + public final Clustering indexedEntryClustering; + + public IndexEntry(DecoratedKey indexValue, + Clustering indexClustering, + long timestamp, + ByteBuffer indexedKey, + Clustering indexedEntryClustering) + { + this.indexValue = indexValue; + this.indexClustering = indexClustering; + this.timestamp = timestamp; + this.indexedKey = indexedKey; + this.indexedEntryClustering = indexedEntryClustering; + } +} diff --git a/src/java/org/apache/cassandra/index/internal/composites/ClusteringColumnIndex.java b/src/java/org/apache/cassandra/index/internal/composites/ClusteringColumnIndex.java new file mode 100644 index 0000000000..b932602436 --- /dev/null +++ b/src/java/org/apache/cassandra/index/internal/composites/ClusteringColumnIndex.java @@ -0,0 +1,100 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.cassandra.index.internal.composites; + +import java.nio.ByteBuffer; + +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.IndexEntry; +import org.apache.cassandra.schema.IndexMetadata; + +/** + * Index on a CLUSTERING_COLUMN column definition. + * + * A cell indexed by this index will have the general form: + * ck_0 ... ck_n c_name : v + * where ck_i are the cluster keys, c_name the last component of the cell + * composite name (or second to last if collections are in use, but this + * has no impact) and v the cell value. + * + * Such a cell is always indexed by this index (or rather, it is indexed if + * n >= columnDef.componentIndex, which will always be the case in practice) + * and it will generate (makeIndexColumnName()) an index entry whose: + * - row key will be ck_i (getIndexedValue()) where i == columnDef.componentIndex. + * - cell name will + * rk ck_0 ... ck_{i-1} ck_{i+1} ck_n + * where rk is the row key of the initial cell and i == columnDef.componentIndex. + */ +public class ClusteringColumnIndex extends CassandraIndex +{ + public ClusteringColumnIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) + { + super(baseCfs, indexDef); + } + + + public ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, ByteBuffer cellValue) + { + return clustering.get(indexedColumn.position()); + } + + public CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path) + { + CBuilder builder = CBuilder.create(getIndexComparator()); + builder.add(partitionKey); + for (int i = 0; i < Math.min(indexedColumn.position(), prefix.size()); i++) + builder.add(prefix.get(i)); + for (int i = indexedColumn.position() + 1; i < prefix.size(); i++) + builder.add(prefix.get(i)); + return builder; + } + + public IndexEntry decodeEntry(DecoratedKey indexedValue, + Row indexEntry) + { + int ckCount = baseCfs.metadata.clusteringColumns().size(); + + Clustering clustering = indexEntry.clustering(); + CBuilder builder = CBuilder.create(baseCfs.getComparator()); + for (int i = 0; i < indexedColumn.position(); i++) + builder.add(clustering.get(i + 1)); + + builder.add(indexedValue.getKey()); + + for (int i = indexedColumn.position() + 1; i < ckCount; i++) + builder.add(clustering.get(i)); + + return new IndexEntry(indexedValue, + clustering, + indexEntry.primaryKeyLivenessInfo().timestamp(), + clustering.get(0), + builder.build()); + } + + public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) + { + return !data.hasLiveData(nowInSec); + } +} diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionKeyAndValue.java b/src/java/org/apache/cassandra/index/internal/composites/CollectionEntryIndex.java similarity index 58% rename from src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionKeyAndValue.java rename to src/java/org/apache/cassandra/index/internal/composites/CollectionEntryIndex.java index 766f8039ff..111360039b 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionKeyAndValue.java +++ b/src/java/org/apache/cassandra/index/internal/composites/CollectionEntryIndex.java @@ -15,13 +15,20 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.marshal.*; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.ColumnFamilyStore; +import org.apache.cassandra.db.marshal.AbstractType; +import org.apache.cassandra.db.marshal.CollectionType; +import org.apache.cassandra.db.marshal.CompositeType; +import org.apache.cassandra.db.rows.Cell; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.schema.IndexMetadata; /** * Index on the element and value of cells participating in a collection. @@ -29,26 +36,28 @@ import org.apache.cassandra.db.marshal.*; * The row keys for this index are a composite of the collection element * and value of indexed columns. */ -public class CompositesIndexOnCollectionKeyAndValue extends CompositesIndexIncludingCollectionKey +public class CollectionEntryIndex extends CollectionKeyIndexBase { - @Override - protected AbstractType getIndexKeyComparator() + public CollectionEntryIndex(ColumnFamilyStore baseCfs, + IndexMetadata indexDef) { - CollectionType colType = (CollectionType)columnDef.type; - return CompositeType.getInstance(colType.nameComparator(), colType.valueComparator()); + super(baseCfs, indexDef); } - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) + public ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, ByteBuffer cellValue) { return CompositeType.build(path.get(0), cellValue); } public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) { - ByteBuffer[] components = ((CompositeType)getIndexKeyComparator()).split(indexValue); + ByteBuffer[] components = ((CompositeType)functions.getIndexedValueType(indexedColumn)).split(indexValue); ByteBuffer mapKey = components[0]; ByteBuffer mapValue = components[1]; + ColumnDefinition columnDef = indexedColumn; Cell cell = data.getCell(columnDef, CellPath.create(mapKey)); if (cell == null || !cell.isLive(nowInSec)) return true; diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionKey.java b/src/java/org/apache/cassandra/index/internal/composites/CollectionKeyIndex.java similarity index 54% rename from src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionKey.java rename to src/java/org/apache/cassandra/index/internal/composites/CollectionKeyIndex.java index 50e81c426a..42c45e5552 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionKey.java +++ b/src/java/org/apache/cassandra/index/internal/composites/CollectionKeyIndex.java @@ -15,14 +15,19 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; +import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.Operator; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.marshal.*; +import org.apache.cassandra.db.Clustering; +import org.apache.cassandra.db.ColumnFamilyStore; +import org.apache.cassandra.db.marshal.SetType; +import org.apache.cassandra.db.rows.Cell; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.schema.IndexMetadata; /** * Index on the collection element of the cell name of a collection. @@ -30,29 +35,30 @@ import org.apache.cassandra.db.marshal.*; * The row keys for this index are given by the collection element for * indexed columns. */ -public class CompositesIndexOnCollectionKey extends CompositesIndexIncludingCollectionKey +public class CollectionKeyIndex extends CollectionKeyIndexBase { - @Override - protected AbstractType getIndexKeyComparator() + public CollectionKeyIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) { - return ((CollectionType)columnDef.type).nameComparator(); + super(baseCfs, indexDef); } - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) + public ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, + ByteBuffer cellValue) { return path.get(0); } - @Override - public boolean supportsOperator(Operator operator) - { - return operator == Operator.CONTAINS_KEY || - operator == Operator.CONTAINS && columnDef.type instanceof SetType; - } - public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) { - Cell cell = data.getCell(columnDef, CellPath.create(indexValue)); + Cell cell = data.getCell(indexedColumn, CellPath.create(indexValue)); return cell == null || !cell.isLive(nowInSec); } + + public boolean supportsOperator(ColumnDefinition indexedColumn, Operator operator) + { + return operator == Operator.CONTAINS_KEY || + operator == Operator.CONTAINS && indexedColumn.type instanceof SetType; + } } diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexIncludingCollectionKey.java b/src/java/org/apache/cassandra/index/internal/composites/CollectionKeyIndexBase.java similarity index 65% rename from src/java/org/apache/cassandra/db/index/composites/CompositesIndexIncludingCollectionKey.java rename to src/java/org/apache/cassandra/index/internal/composites/CollectionKeyIndexBase.java index 7624c1f74b..fe77c96c2b 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexIncludingCollectionKey.java +++ b/src/java/org/apache/cassandra/index/internal/composites/CollectionKeyIndexBase.java @@ -15,12 +15,16 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.IndexEntry; +import org.apache.cassandra.schema.IndexMetadata; /** * Common superclass for indexes that capture collection keys, including @@ -37,24 +41,38 @@ import org.apache.cassandra.db.rows.*; * - the row key is determined by subclasses of this type. * - the cell name will be 'rk ck_0 ... ck_n' where rk is the row key of the initial cell. */ -public abstract class CompositesIndexIncludingCollectionKey extends CompositesIndex +public abstract class CollectionKeyIndexBase extends CassandraIndex { - protected CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path) + public CollectionKeyIndexBase(ColumnFamilyStore baseCfs, IndexMetadata indexDef) + { + super(baseCfs, indexDef); + } + + public CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path) { CBuilder builder = CBuilder.create(getIndexComparator()); - builder.add(rowKey); + builder.add(partitionKey); for (int i = 0; i < prefix.size(); i++) builder.add(prefix.get(i)); + return builder; } - public IndexedEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) + public IndexEntry decodeEntry(DecoratedKey indexedValue, + Row indexEntry) { int count = 1 + baseCfs.metadata.clusteringColumns().size(); Clustering clustering = indexEntry.clustering(); CBuilder builder = CBuilder.create(baseCfs.getComparator()); for (int i = 0; i < count - 1; i++) builder.add(clustering.get(i + 1)); - return new IndexedEntry(indexedValue, clustering, indexEntry.primaryKeyLivenessInfo().timestamp(), clustering.get(0), builder.build()); + + return new IndexEntry(indexedValue, + clustering, + indexEntry.primaryKeyLivenessInfo().timestamp(), + clustering.get(0), + builder.build()); } } diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionValue.java b/src/java/org/apache/cassandra/index/internal/composites/CollectionValueIndex.java similarity index 60% rename from src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionValue.java rename to src/java/org/apache/cassandra/index/internal/composites/CollectionValueIndex.java index 30391cf990..95bd7e1121 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnCollectionValue.java +++ b/src/java/org/apache/cassandra/index/internal/composites/CollectionValueIndex.java @@ -15,17 +15,19 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; -import java.util.Iterator; -import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.db.*; +import org.apache.cassandra.db.marshal.CollectionType; +import org.apache.cassandra.db.marshal.SetType; import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.marshal.*; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.IndexEntry; +import org.apache.cassandra.schema.IndexMetadata; /** * Index the value of a collection cell. @@ -38,31 +40,26 @@ import org.apache.cassandra.db.marshal.*; * for each so that if we delete one of the value only we only delete the * entry corresponding to that value. */ -public class CompositesIndexOnCollectionValue extends CompositesIndex +public class CollectionValueIndex extends CassandraIndex { - public static void addClusteringColumns(CFMetaData.Builder indexMetadata, CFMetaData baseMetadata, ColumnDefinition columnDef) + public CollectionValueIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) { - addGenericClusteringColumns(indexMetadata, baseMetadata, columnDef); - - // collection key - indexMetadata.addClusteringColumn("cell_path", ((CollectionType)columnDef.type).nameComparator()); + super(baseCfs, indexDef); } - @Override - protected AbstractType getIndexKeyComparator() - { - return ((CollectionType)columnDef.type).valueComparator(); - } - - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) + public ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, ByteBuffer cellValue) { return cellValue; } - protected CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path) + public CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path) { CBuilder builder = CBuilder.create(getIndexComparator()); - builder.add(rowKey); + builder.add(partitionKey); for (int i = 0; i < prefix.size(); i++) builder.add(prefix.get(i)); @@ -73,27 +70,36 @@ public class CompositesIndexOnCollectionValue extends CompositesIndex return builder; } - public IndexedEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) + public IndexEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) { Clustering clustering = indexEntry.clustering(); CBuilder builder = CBuilder.create(baseCfs.getComparator()); for (int i = 0; i < baseCfs.getComparator().size(); i++) builder.add(clustering.get(i + 1)); - return new IndexedEntry(indexedValue, clustering, indexEntry.primaryKeyLivenessInfo().timestamp(), clustering.get(0), builder.build()); + + return new IndexEntry(indexedValue, + clustering, + indexEntry.primaryKeyLivenessInfo().timestamp(), + clustering.get(0), + builder.build()); } - @Override - public boolean supportsOperator(Operator operator) + public boolean supportsOperator(ColumnDefinition indexedColumn, Operator operator) { - return operator == Operator.CONTAINS && !(columnDef.type instanceof SetType); + return operator == Operator.CONTAINS && !(indexedColumn.type instanceof SetType); } public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) { + ColumnDefinition columnDef = indexedColumn; ComplexColumnData complexData = data.getComplexColumnData(columnDef); + if (complexData == null) + return true; + for (Cell cell : complexData) { - if (cell.isLive(nowInSec) && ((CollectionType) columnDef.type).valueComparator().compare(indexValue, cell.value()) == 0) + if (cell.isLive(nowInSec) && ((CollectionType) columnDef.type).valueComparator() + .compare(indexValue, cell.value()) == 0) return false; } return true; diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesSearcher.java b/src/java/org/apache/cassandra/index/internal/composites/CompositesSearcher.java similarity index 75% rename from src/java/org/apache/cassandra/db/index/composites/CompositesSearcher.java rename to src/java/org/apache/cassandra/index/internal/composites/CompositesSearcher.java index b76bf7eea1..77867fcbd8 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesSearcher.java +++ b/src/java/org/apache/cassandra/index/internal/composites/CompositesSearcher.java @@ -15,53 +15,55 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; -import java.util.*; +import java.util.ArrayList; +import java.util.List; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.filter.*; -import org.apache.cassandra.db.index.*; +import org.apache.cassandra.db.filter.ClusteringIndexNamesFilter; +import org.apache.cassandra.db.filter.DataLimits; +import org.apache.cassandra.db.filter.RowFilter; import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.CassandraIndexSearcher; +import org.apache.cassandra.index.internal.IndexEntry; import org.apache.cassandra.utils.btree.BTreeSet; import org.apache.cassandra.utils.concurrent.OpOrder; -public class CompositesSearcher extends SecondaryIndexSearcher +public class CompositesSearcher extends CassandraIndexSearcher { private static final Logger logger = LoggerFactory.getLogger(CompositesSearcher.class); - public CompositesSearcher(SecondaryIndexManager indexManager, Set columns) + public CompositesSearcher(ReadCommand command, + RowFilter.Expression expression, + CassandraIndex index) { - super(indexManager, columns); + super(command, expression, index); } - private boolean isMatchingEntry(DecoratedKey partitionKey, CompositesIndex.IndexedEntry entry, ReadCommand command) + private boolean isMatchingEntry(DecoratedKey partitionKey, IndexEntry entry, ReadCommand command) { return command.selects(partitionKey, entry.indexedEntryClustering); } - protected UnfilteredPartitionIterator queryDataFromIndex(AbstractSimplePerColumnSecondaryIndex secondaryIdx, - final DecoratedKey indexKey, + protected UnfilteredPartitionIterator queryDataFromIndex(final DecoratedKey indexKey, final RowIterator indexHits, final ReadCommand command, final ReadOrderGroup orderGroup) { assert indexHits.staticRow() == Rows.EMPTY_STATIC_ROW; - assert secondaryIdx instanceof CompositesIndex; - final CompositesIndex index = (CompositesIndex)secondaryIdx; - return new UnfilteredPartitionIterator() { - private CompositesIndex.IndexedEntry nextEntry; + private IndexEntry nextEntry; private UnfilteredRowIterator next; @@ -109,9 +111,9 @@ public class CompositesSearcher extends SecondaryIndexSearcher // in memory so we should consider adding some paging mechanism. However, index hits should // be relatively small so it's much better than the previous code that was materializing all // *data* for a given partition. - BTreeSet.Builder clusterings = BTreeSet.builder(baseCfs.getComparator()); - List entries = new ArrayList<>(); - DecoratedKey partitionKey = baseCfs.decorateKey(nextEntry.indexedKey); + BTreeSet.Builder clusterings = BTreeSet.builder(index.baseCfs.getComparator()); + List entries = new ArrayList<>(); + DecoratedKey partitionKey = index.baseCfs.decorateKey(nextEntry.indexedKey); while (nextEntry != null && partitionKey.getKey().equals(nextEntry.indexedKey)) { @@ -131,7 +133,7 @@ public class CompositesSearcher extends SecondaryIndexSearcher // Query the gathered index hits. We still need to filter stale hits from the resulting query. ClusteringIndexNamesFilter filter = new ClusteringIndexNamesFilter(clusterings.build(), false); - SinglePartitionReadCommand dataCmd = new SinglePartitionNamesCommand(metadata(), + SinglePartitionReadCommand dataCmd = new SinglePartitionNamesCommand(index.baseCfs.metadata, command.nowInSec(), command.columnFilter(), command.rowFilter(), @@ -140,12 +142,14 @@ public class CompositesSearcher extends SecondaryIndexSearcher filter); @SuppressWarnings("resource") // We close right away if empty, and if it's assign to next it will be called either // by the next caller of next, or through closing this iterator is this come before. - UnfilteredRowIterator dataIter = filterStaleEntries(dataCmd.queryMemtableAndDisk(baseCfs, orderGroup.baseReadOpOrderGroup()), - index, + UnfilteredRowIterator dataIter = filterStaleEntries(dataCmd.queryMemtableAndDisk(index.baseCfs, + orderGroup.baseReadOpOrderGroup()), indexKey.getKey(), entries, orderGroup.writeOpOrderGroup(), command.nowInSec()); + + if (dataIter.isEmpty()) { dataIter.close(); @@ -170,35 +174,62 @@ public class CompositesSearcher extends SecondaryIndexSearcher }; } + private void deleteAllEntries(final List entries, final OpOrder.Group writeOp, final int nowInSec) + { + entries.forEach(entry -> + index.deleteStaleEntry(entry.indexValue, + entry.indexClustering, + new DeletionTime(entry.timestamp, nowInSec), + writeOp)); + } + private UnfilteredRowIterator filterStaleEntries(UnfilteredRowIterator dataIter, - final CompositesIndex index, final ByteBuffer indexValue, - final List entries, + final List entries, final OpOrder.Group writeOp, final int nowInSec) { + // collect stale index entries and delete them when we close this iterator + final List staleEntries = new ArrayList<>(); + + // if there is a partition level delete in the base table, we need to filter + // any index entries which would be shadowed by it + if (!dataIter.partitionLevelDeletion().isLive()) + { + DeletionTime deletion = dataIter.partitionLevelDeletion(); + entries.forEach(e -> { + if (deletion.deletes(e.timestamp)) + staleEntries.add(e); + }); + } + return new AlteringUnfilteredRowIterator(dataIter) { private int entriesIdx; + public void close() + { + deleteAllEntries(staleEntries, writeOp, nowInSec); + super.close(); + } + @Override protected Row computeNext(Row row) { - CompositesIndex.IndexedEntry entry = findEntry(row.clustering(), writeOp, nowInSec); + IndexEntry entry = findEntry(row.clustering()); if (!index.isStale(row, indexValue, nowInSec)) return row; - // The entry is stale: delete the entry and ignore otherwise - index.delete(entry, writeOp, nowInSec); + staleEntries.add(entry); return null; } - private CompositesIndex.IndexedEntry findEntry(Clustering clustering, OpOrder.Group writeOp, int nowInSec) + private IndexEntry findEntry(Clustering clustering) { assert entriesIdx < entries.size(); while (entriesIdx < entries.size()) { - CompositesIndex.IndexedEntry entry = entries.get(entriesIdx++); + IndexEntry entry = entries.get(entriesIdx++); // The entries are in clustering order. So that the requested entry should be the // next entry, the one at 'entriesIdx'. However, we can have stale entries, entries // that have no corresponding row in the base table typically because of a range @@ -208,7 +239,7 @@ public class CompositesSearcher extends SecondaryIndexSearcher if (cmp == 0) return entry; else - index.delete(entry, writeOp, nowInSec); + staleEntries.add(entry); } // entries correspond to the rows we've queried, so we shouldn't have a row that has no corresponding entry. throw new AssertionError(); diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnPartitionKey.java b/src/java/org/apache/cassandra/index/internal/composites/PartitionKeyIndex.java similarity index 55% rename from src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnPartitionKey.java rename to src/java/org/apache/cassandra/index/internal/composites/PartitionKeyIndex.java index a93f8e1df2..2c0b5aa9d4 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnPartitionKey.java +++ b/src/java/org/apache/cassandra/index/internal/composites/PartitionKeyIndex.java @@ -15,15 +15,17 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; -import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.marshal.*; -import org.apache.cassandra.utils.concurrent.OpOrder; +import org.apache.cassandra.db.marshal.CompositeType; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.IndexEntry; +import org.apache.cassandra.schema.IndexMetadata; /** * Index on a PARTITION_KEY column definition. @@ -43,25 +45,35 @@ import org.apache.cassandra.utils.concurrent.OpOrder; * want to order the index cell name by partitioner first, and skipping a part * of the row key would change the order. */ -public class CompositesIndexOnPartitionKey extends CompositesIndex +public class PartitionKeyIndex extends CassandraIndex { - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) + public PartitionKeyIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) { - CompositeType keyComparator = (CompositeType)baseCfs.metadata.getKeyValidator(); - ByteBuffer[] components = keyComparator.split(rowKey); - return components[columnDef.position()]; + super(baseCfs, indexDef); } - protected CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path) + public ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, + ByteBuffer cellValue) + { + CompositeType keyComparator = (CompositeType)baseCfs.metadata.getKeyValidator(); + ByteBuffer[] components = keyComparator.split(partitionKey); + return components[indexedColumn.position()]; + } + + public CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path) { CBuilder builder = CBuilder.create(getIndexComparator()); - builder.add(rowKey); + builder.add(partitionKey); for (int i = 0; i < prefix.size(); i++) builder.add(prefix.get(i)); return builder; } - public IndexedEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) + public IndexEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) { int ckCount = baseCfs.metadata.clusteringColumns().size(); Clustering clustering = indexEntry.clustering(); @@ -69,38 +81,15 @@ public class CompositesIndexOnPartitionKey extends CompositesIndex for (int i = 0; i < ckCount; i++) builder.add(clustering.get(i + 1)); - return new IndexedEntry(indexedValue, clustering, indexEntry.primaryKeyLivenessInfo().timestamp(), clustering.get(0), builder.build()); - } - - @Override - protected boolean indexPrimaryKeyColumn() - { - return true; - } - - @Override - public boolean indexes(ColumnDefinition c) - { - // Actual indexing for this index type is done through maybeIndex - return false; + return new IndexEntry(indexedValue, + clustering, + indexEntry.primaryKeyLivenessInfo().timestamp(), + clustering.get(0), + builder.build()); } public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) { return !data.hasLiveData(nowInSec); } - - @Override - public void maybeIndex(ByteBuffer partitionKey, Clustering clustering, long timestamp, int ttl, OpOrder.Group opGroup, int nowInSec) - { - insert(partitionKey, clustering, null, LivenessInfo.create(indexCfs.metadata, timestamp, ttl, nowInSec), opGroup); - } - - @Override - public void delete(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - // We only know that one column of the CQL row has been updated/deleted, but we don't know if the - // full row has been deleted so we should not do anything. If it ends up that the whole row has - // been deleted, it will be eventually cleaned up on read because the entry will be detected stale. - } } diff --git a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnRegular.java b/src/java/org/apache/cassandra/index/internal/composites/RegularColumnIndex.java similarity index 60% rename from src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnRegular.java rename to src/java/org/apache/cassandra/index/internal/composites/RegularColumnIndex.java index a88502a1a6..f1dc3af6f7 100644 --- a/src/java/org/apache/cassandra/db/index/composites/CompositesIndexOnRegular.java +++ b/src/java/org/apache/cassandra/index/internal/composites/RegularColumnIndex.java @@ -15,12 +15,17 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.composites; +package org.apache.cassandra.index.internal.composites; import java.nio.ByteBuffer; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.db.rows.Cell; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.IndexEntry; +import org.apache.cassandra.schema.IndexMetadata; /** * Index on a REGULAR column definition on a composite type. @@ -39,37 +44,53 @@ import org.apache.cassandra.db.rows.*; * where rk is the row key of the initial cell. I.e. the index entry store * all the information require to locate back the indexed cell. */ -public class CompositesIndexOnRegular extends CompositesIndex +public class RegularColumnIndex extends CassandraIndex { - protected ByteBuffer getIndexedValue(ByteBuffer rowKey, Clustering clustering, ByteBuffer cellValue, CellPath path) + public RegularColumnIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) + { + super(baseCfs, indexDef); + } + + public ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, + ByteBuffer cellValue) { return cellValue; } - protected CBuilder buildIndexClusteringPrefix(ByteBuffer rowKey, ClusteringPrefix prefix, CellPath path) + public CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path) { CBuilder builder = CBuilder.create(getIndexComparator()); - builder.add(rowKey); + builder.add(partitionKey); for (int i = 0; i < prefix.size(); i++) builder.add(prefix.get(i)); + return builder; } - public IndexedEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) + public IndexEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) { Clustering clustering = indexEntry.clustering(); ClusteringComparator baseComparator = baseCfs.getComparator(); CBuilder builder = CBuilder.create(baseComparator); for (int i = 0; i < baseComparator.size(); i++) builder.add(clustering.get(i + 1)); - return new IndexedEntry(indexedValue, clustering, indexEntry.primaryKeyLivenessInfo().timestamp(), clustering.get(0), builder.build()); + + return new IndexEntry(indexedValue, + clustering, + indexEntry.primaryKeyLivenessInfo().timestamp(), + clustering.get(0), + builder.build()); } public boolean isStale(Row data, ByteBuffer indexValue, int nowInSec) { - Cell cell = data.getCell(columnDef); + Cell cell = data.getCell(indexedColumn); return cell == null || !cell.isLive(nowInSec) - || columnDef.type.compare(indexValue, cell.value()) != 0; + || indexedColumn.type.compare(indexValue, cell.value()) != 0; } } diff --git a/src/java/org/apache/cassandra/index/internal/keys/KeysIndex.java b/src/java/org/apache/cassandra/index/internal/keys/KeysIndex.java new file mode 100644 index 0000000000..53ecd01ee7 --- /dev/null +++ b/src/java/org/apache/cassandra/index/internal/keys/KeysIndex.java @@ -0,0 +1,62 @@ +package org.apache.cassandra.index.internal.keys; + +import java.nio.ByteBuffer; + +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.rows.Cell; +import org.apache.cassandra.db.rows.CellPath; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.IndexEntry; +import org.apache.cassandra.schema.IndexMetadata; + +public class KeysIndex extends CassandraIndex +{ + public KeysIndex(ColumnFamilyStore baseCfs, IndexMetadata indexDef) + { + super(baseCfs, indexDef); + } + + public CFMetaData.Builder addIndexClusteringColumns(CFMetaData.Builder builder, + CFMetaData baseMetadata, + ColumnDefinition cfDef) + { + // no additional clustering columns required + return builder; + } + + protected CBuilder buildIndexClusteringPrefix(ByteBuffer partitionKey, + ClusteringPrefix prefix, + CellPath path) + { + CBuilder builder = CBuilder.create(getIndexComparator()); + builder.add(partitionKey); + return builder; + } + + protected ByteBuffer getIndexedValue(ByteBuffer partitionKey, + Clustering clustering, + CellPath path, ByteBuffer cellValue) + { + return cellValue; + } + + public IndexEntry decodeEntry(DecoratedKey indexedValue, Row indexEntry) + { + throw new UnsupportedOperationException("KEYS indexes do not use a specialized index entry format"); + } + + public boolean isStale(Row row, ByteBuffer indexValue, int nowInSec) + { + if (row == null) + return true; + + Cell cell = row.getCell(indexedColumn); + + return (cell == null + || !cell.isLive(nowInSec) + || indexedColumn.type.compare(indexValue, cell.value()) != 0); + } +} diff --git a/src/java/org/apache/cassandra/db/index/keys/KeysSearcher.java b/src/java/org/apache/cassandra/index/internal/keys/KeysSearcher.java similarity index 70% rename from src/java/org/apache/cassandra/db/index/keys/KeysSearcher.java rename to src/java/org/apache/cassandra/index/internal/keys/KeysSearcher.java index 4b70dcfbe7..b60d2d98fa 100644 --- a/src/java/org/apache/cassandra/db/index/keys/KeysSearcher.java +++ b/src/java/org/apache/cassandra/index/internal/keys/KeysSearcher.java @@ -15,35 +15,36 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -package org.apache.cassandra.db.index.keys; +package org.apache.cassandra.index.internal.keys; import java.nio.ByteBuffer; -import java.util.Set; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.db.index.*; -import org.apache.cassandra.db.filter.*; +import org.apache.cassandra.db.filter.DataLimits; +import org.apache.cassandra.db.filter.RowFilter; +import org.apache.cassandra.db.partitions.ImmutableBTreePartition; import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; +import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.index.internal.CassandraIndexSearcher; import org.apache.cassandra.utils.concurrent.OpOrder; -public class KeysSearcher extends SecondaryIndexSearcher +public class KeysSearcher extends CassandraIndexSearcher { private static final Logger logger = LoggerFactory.getLogger(KeysSearcher.class); - public KeysSearcher(SecondaryIndexManager indexManager, Set columns) + public KeysSearcher(ReadCommand command, + RowFilter.Expression expression, + CassandraIndex indexer) { - super(indexManager, columns); + super(command, expression, indexer); } - protected UnfilteredPartitionIterator queryDataFromIndex(final AbstractSimplePerColumnSecondaryIndex index, - final DecoratedKey indexKey, + protected UnfilteredPartitionIterator queryDataFromIndex(final DecoratedKey indexKey, final RowIterator indexHits, final ReadCommand command, final ReadOrderGroup orderGroup) @@ -84,21 +85,22 @@ public class KeysSearcher extends SecondaryIndexSearcher while (next == null && indexHits.hasNext()) { Row hit = indexHits.next(); - DecoratedKey key = baseCfs.decorateKey(hit.clustering().get(0)); + DecoratedKey key = index.baseCfs.decorateKey(hit.clustering().get(0)); SinglePartitionReadCommand dataCmd = SinglePartitionReadCommand.create(isForThrift(), - baseCfs.metadata, + index.baseCfs.metadata, command.nowInSec(), command.columnFilter(), command.rowFilter(), DataLimits.NONE, key, command.clusteringIndexFilter(key)); + @SuppressWarnings("resource") // filterIfStale closes it's iterator if either it materialize it or if it returns null. // Otherwise, we close right away if empty, and if it's assigned to next it will be called either // by the next caller of next, or through closing this iterator is this come before. - UnfilteredRowIterator dataIter = filterIfStale(dataCmd.queryMemtableAndDisk(baseCfs, orderGroup.baseReadOpOrderGroup()), - index, + UnfilteredRowIterator dataIter = filterIfStale(dataCmd.queryMemtableAndDisk(index.baseCfs, + orderGroup.baseReadOpOrderGroup()), hit, indexKey.getKey(), orderGroup.writeOpOrderGroup(), @@ -131,7 +133,6 @@ public class KeysSearcher extends SecondaryIndexSearcher } private UnfilteredRowIterator filterIfStale(UnfilteredRowIterator iterator, - AbstractSimplePerColumnSecondaryIndex index, Row indexHit, ByteBuffer indexedValue, OpOrder.Group writeOp, @@ -144,19 +145,37 @@ public class KeysSearcher extends SecondaryIndexSearcher // is the indexed name. Ans so we need to materialize the partition. ImmutableBTreePartition result = ImmutableBTreePartition.create(iterator); iterator.close(); - Row data = result.getRow(new Clustering(index.indexedColumn().name.bytes)); - Cell cell = data == null ? null : data.getCell(baseCfs.metadata.compactValueColumn()); - return deleteIfStale(iterator.partitionKey(), cell, index, indexHit, indexedValue, writeOp, nowInSec) - ? null - : result.unfilteredIterator(); + Row data = result.getRow(new Clustering(index.getIndexedColumn().name.bytes)); + + // for thrift tables, we need to compare the index entry against the compact value column, + // not the column actually designated as the indexed column so we don't use the index function + // lib for the staleness check like we do in every other case + Cell baseData = data.getCell(index.baseCfs.metadata.compactValueColumn()); + if (baseData == null || !baseData.isLive(nowInSec) || index.getIndexedColumn().type.compare(indexedValue, baseData.value()) != 0) + { + // Index is stale, remove the index entry and ignore + index.deleteStaleEntry(index.getIndexCfs().decorateKey(indexedValue), + new Clustering(index.getIndexedColumn().name.bytes), + new DeletionTime(indexHit.primaryKeyLivenessInfo().timestamp(), nowInSec), + writeOp); + return null; + } + else + { + return result.unfilteredIterator(); + } } else { assert iterator.metadata().isCompactTable(); Row data = iterator.staticRow(); - Cell cell = data == null ? null : data.getCell(index.indexedColumn()); - if (deleteIfStale(iterator.partitionKey(), cell, index, indexHit, indexedValue, writeOp, nowInSec)) + if (index.isStale(data, indexedValue, nowInSec)) { + // Index is stale, remove the index entry and ignore + index.deleteStaleEntry(index.getIndexCfs().decorateKey(indexedValue), + makeIndexClustering(iterator.partitionKey().getKey(), Clustering.EMPTY), + new DeletionTime(indexHit.primaryKeyLivenessInfo().timestamp(), nowInSec), + writeOp); iterator.close(); return null; } @@ -166,26 +185,4 @@ public class KeysSearcher extends SecondaryIndexSearcher } } } - - private boolean deleteIfStale(DecoratedKey partitionKey, - Cell cell, - AbstractSimplePerColumnSecondaryIndex index, - Row indexHit, - ByteBuffer indexedValue, - OpOrder.Group writeOp, - int nowInSec) - { - if (cell == null || !cell.isLive(nowInSec) || index.indexedColumn().type.compare(indexedValue, cell.value()) != 0) - { - // Index is stale, remove the index entry and ignore - index.delete(partitionKey.getKey(), - new Clustering(index.indexedColumn().name.bytes), - indexedValue, - null, - new DeletionTime(indexHit.primaryKeyLivenessInfo().timestamp(), nowInSec), - writeOp); - return true; - } - return false; - } } diff --git a/src/java/org/apache/cassandra/index/transactions/CleanupTransaction.java b/src/java/org/apache/cassandra/index/transactions/CleanupTransaction.java new file mode 100644 index 0000000000..1d6ba5610f --- /dev/null +++ b/src/java/org/apache/cassandra/index/transactions/CleanupTransaction.java @@ -0,0 +1,52 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index.transactions; + +import org.apache.cassandra.db.DeletionTime; +import org.apache.cassandra.db.rows.Row; + +/** + * Performs garbage collection of index entries during a cleanup. + * + * Notifies registered indexers of each partition being removed and + * + * Compaction & Cleanup are somewhat simpler than dealing with incoming writes, + * being only concerned with cleaning up stale index entries. + * + * When multiple versions of a row are compacted, the CleanupTransaction is + * notified of the versions being merged, which it diffs against the merge result + * and forwards to the registered Index.Indexer instances when on commit. + * + * Instances are currently scoped to a single row within a partition, but this could be improved to batch process + * multiple rows within a single partition. + */ +public interface CleanupTransaction extends IndexTransaction +{ + + void onPartitionDeletion(DeletionTime deletionTime); + void onRowDelete(Row row); + + CleanupTransaction NO_OP = new CleanupTransaction() + { + public void start(){} + public void onPartitionDeletion(DeletionTime deletionTime){} + public void onRowDelete(Row row){} + public void commit(){} + }; +} diff --git a/src/java/org/apache/cassandra/index/transactions/CompactionTransaction.java b/src/java/org/apache/cassandra/index/transactions/CompactionTransaction.java new file mode 100644 index 0000000000..a9fbf413aa --- /dev/null +++ b/src/java/org/apache/cassandra/index/transactions/CompactionTransaction.java @@ -0,0 +1,44 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index.transactions; + +import org.apache.cassandra.db.Columns; +import org.apache.cassandra.db.rows.Row; + +/** + * Performs garbage collection of stale index entries during a regular compaction. + * + * A CompactionTransaction is concerned with cleaning up stale index entries. + * When multiple versions of a row are compacted, the CompactionTransaction is + * notified of the versions being merged, which it diffs against the merge result. + * + * Instances are currently scoped to a single row within a partition, but this could be improved to batch process + * multiple rows within a single partition. + */ +public interface CompactionTransaction extends IndexTransaction +{ + void onRowMerge(Columns columns, Row merged, Row...versions); + + CompactionTransaction NO_OP = new CompactionTransaction() + { + public void start(){} + public void onRowMerge(Columns columns, Row merged, Row...versions){} + public void commit(){} + }; +} diff --git a/src/java/org/apache/cassandra/index/transactions/IndexTransaction.java b/src/java/org/apache/cassandra/index/transactions/IndexTransaction.java new file mode 100644 index 0000000000..3fb8235385 --- /dev/null +++ b/src/java/org/apache/cassandra/index/transactions/IndexTransaction.java @@ -0,0 +1,57 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index.transactions; + +/** + * Base interface for the handling of index updates. + * There are 3 types of transaction where indexes are updated to stay in sync with the base table, each represented by + * a subinterface: + * * {@code UpdateTransaction} + * Used on the regular write path and when indexing newly acquired SSTables from streaming or sideloading. This type + * of transaction may include both row inserts and updates to rows previously existing in the base Memtable. Instances + * are scoped to a single partition update and are obtained from the factory method + * @{code SecondaryIndexManager#newUpdateTransaction} + * + * * {@code CompactionTransaction} + * Used during compaction when stale entries which have been superceded are cleaned up from the index. As rows in a + * partition are merged during the compaction, index entries for any purged rows are cleaned from the index to + * compensate for the fact that they may not have been removed at write time if the data in the base table had been + * already flushed to disk (and so was processed as an insert, not an update by the UpdateTransaction). These + * transactions are currently scoped to a single row within a partition, but this could be improved to batch process + * multiple rows within a single partition. + * + * * @{code CleanupTransaction} + * During cleanup no merging is required, the only thing to do is to notify indexes of the partitions being removed, + * along with the rows within those partitions. Like with compaction, these transactions are currently scoped to a + * single row within a partition, but this could be improved with batching. + */ +public interface IndexTransaction +{ + /** + * Used to differentiate between type of index transaction when obtaining + * a handler from Index implementations. + */ + public enum Type + { + UPDATE, COMPACTION, CLEANUP + } + + void start(); + void commit(); +} diff --git a/src/java/org/apache/cassandra/index/transactions/UpdateTransaction.java b/src/java/org/apache/cassandra/index/transactions/UpdateTransaction.java new file mode 100644 index 0000000000..c78304af9b --- /dev/null +++ b/src/java/org/apache/cassandra/index/transactions/UpdateTransaction.java @@ -0,0 +1,77 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index.transactions; + +import org.apache.cassandra.db.DeletionTime; +import org.apache.cassandra.db.RangeTombstone; +import org.apache.cassandra.db.rows.Row; + +/** + * Handling of index updates on the write path. + * + * Instances of an UpdateTransaction are scoped to a single partition update + * A new instance is used for every write, obtained from the + * newUpdateTransaction(PartitionUpdate) method. Likewise, a single + * CleanupTransaction instance is used for each partition processed during a + * compaction or cleanup. + * + * We make certain guarantees about the lifecycle of each UpdateTransaction + * instance. Namely that start() will be called before any other method, and + * commit() will be called at the end of the update. + * Each instance is initialized with 1..many Index.Indexer instances, one per + * registered Index. As with the transaction itself, these are scoped to a + * specific partition update, so implementations can be assured that all indexing + * events they receive relate to the same logical operation. + * + * onPartitionDelete(), onRangeTombstone(), onInserted() and onUpdated() + * calls may arrive in any order, but this should have no impact for the + * Indexers being notified as any events delivered to a single instance + * necessarily relate to a single partition. + * + * The typical sequence of events during a Memtable update would be: + * start() -- no-op, used to notify Indexers of the start of the transaction + * onPartitionDeletion(dt) -- if the PartitionUpdate implies one + * onRangeTombstone(rt)* -- for each in the PartitionUpdate, if any + * + * then: + * onInserted(row)* -- called for each Row not already present in the Memtable + * onUpdated(existing, updated)* -- called for any Row in the update for where a version was already present + * in the Memtable. It's important to note here that existing is the previous + * row from the Memtable & updated is the final version replacing it. It is + * *not* the incoming row, but the result of merging the incoming and existing + * rows. + * commit() -- finally, finish is called when the new Partition is swapped into the Memtable + */ +public interface UpdateTransaction extends IndexTransaction +{ + void onPartitionDeletion(DeletionTime deletionTime); + void onRangeTombstone(RangeTombstone rangeTombstone); + void onInserted(Row row); + void onUpdated(Row existing, Row updated); + + UpdateTransaction NO_OP = new UpdateTransaction() + { + public void start(){} + public void onPartitionDeletion(DeletionTime deletionTime){} + public void onRangeTombstone(RangeTombstone rangeTombstone){} + public void onInserted(Row row){} + public void onUpdated(Row existing, Row updated){} + public void commit(){} + }; +} diff --git a/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java b/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java index 59085942fb..9f68cea54a 100644 --- a/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java +++ b/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java @@ -40,14 +40,18 @@ import org.apache.cassandra.cache.InstrumentingCache; import org.apache.cassandra.cache.KeyCacheKey; import org.apache.cassandra.concurrent.DebuggableThreadPoolExecutor; import org.apache.cassandra.concurrent.ScheduledExecutors; -import org.apache.cassandra.config.*; +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.DatabaseDescriptor; +import org.apache.cassandra.config.Schema; import org.apache.cassandra.db.*; import org.apache.cassandra.db.commitlog.ReplayPosition; import org.apache.cassandra.db.filter.ColumnFilter; -import org.apache.cassandra.db.index.SecondaryIndex; import org.apache.cassandra.db.lifecycle.TransactionLog; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.dht.*; +import org.apache.cassandra.db.rows.SliceableUnfilteredRowIterator; +import org.apache.cassandra.dht.AbstractBounds; +import org.apache.cassandra.dht.Range; +import org.apache.cassandra.dht.Token; +import org.apache.cassandra.index.internal.CassandraIndex; import org.apache.cassandra.io.FSError; import org.apache.cassandra.io.compress.CompressionMetadata; import org.apache.cassandra.io.sstable.*; @@ -340,8 +344,9 @@ public abstract class SSTableReader extends SSTable implements SelfRefCounted new AssertionError("Could not find index metadata for index cf " + i)); - metadata = SecondaryIndex.newIndexMetadata(parent, def); + .orElseThrow(() -> new AssertionError( + "Could not find index metadata for index cf " + i)); + metadata = CassandraIndex.indexCfsMetadata(parent, def); } else { diff --git a/src/java/org/apache/cassandra/schema/IndexMetadata.java b/src/java/org/apache/cassandra/schema/IndexMetadata.java index af07ceea49..40a75c6f74 100644 --- a/src/java/org/apache/cassandra/schema/IndexMetadata.java +++ b/src/java/org/apache/cassandra/schema/IndexMetadata.java @@ -18,6 +18,7 @@ package org.apache.cassandra.schema; +import java.lang.reflect.InvocationTargetException; import java.util.Collections; import java.util.Map; import java.util.Set; @@ -25,20 +26,26 @@ import java.util.Set; import com.google.common.base.Objects; import com.google.common.collect.ImmutableMap; import com.google.common.collect.ImmutableSet; - +import com.google.common.collect.Maps; import org.apache.commons.lang3.builder.ToStringBuilder; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.ColumnIdentifier; -import org.apache.cassandra.db.index.SecondaryIndex; +import org.apache.cassandra.cql3.statements.IndexTarget; import org.apache.cassandra.exceptions.ConfigurationException; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.utils.FBUtilities; /** * An immutable representation of secondary index metadata. */ public final class IndexMetadata { + private static final Logger logger = LoggerFactory.getLogger(IndexMetadata.class); + public enum IndexType { KEYS, CUSTOM, COMPOSITES @@ -68,20 +75,20 @@ public final class IndexMetadata this.columns = columns == null ? ImmutableSet.of() : ImmutableSet.copyOf(columns); } - public static IndexMetadata legacyIndex(ColumnIdentifier column, - String name, - IndexType type, - Map options) + public static IndexMetadata singleColumnIndex(ColumnIdentifier column, + String name, + IndexType type, + Map options) { return new IndexMetadata(name, options, type, TargetType.COLUMN, Collections.singleton(column)); } - public static IndexMetadata legacyIndex(ColumnDefinition column, - String name, - IndexType type, - Map options) + public static IndexMetadata singleColumnIndex(ColumnDefinition column, + String name, + IndexType type, + Map options) { - return legacyIndex(column.name, name, type, options); + return singleColumnIndex(column.name, name, type, options); } public static boolean isNameValid(String name) @@ -107,11 +114,54 @@ public final class IndexMetadata throw new ConfigurationException("Target type is null for index " + name); if (indexType == IndexMetadata.IndexType.CUSTOM) - if (options == null || !options.containsKey(SecondaryIndex.CUSTOM_INDEX_OPTION_NAME)) + { + if (options == null || !options.containsKey(IndexTarget.CUSTOM_INDEX_OPTION_NAME)) throw new ConfigurationException(String.format("Required option missing for index %s : %s", - name, SecondaryIndex.CUSTOM_INDEX_OPTION_NAME)); + name, IndexTarget.CUSTOM_INDEX_OPTION_NAME)); + String className = options.get(IndexTarget.CUSTOM_INDEX_OPTION_NAME); + Class indexerClass = FBUtilities.classForName(className, "custom indexer"); + if(!Index.class.isAssignableFrom(indexerClass)) + throw new ConfigurationException(String.format("Specified Indexer class (%s) does not implement the Indexer interface", className)); + validateCustomIndexOptions(indexerClass, options); + } } + private void validateCustomIndexOptions(Class indexerClass, Map options) throws ConfigurationException + { + try + { + Map filteredOptions = + Maps.filterKeys(options,key -> !key.equals(IndexTarget.CUSTOM_INDEX_OPTION_NAME)); + + if (filteredOptions.isEmpty()) + return; + + Map unknownOptions = (Map) indexerClass.getMethod("validateOptions", Map.class).invoke(null, filteredOptions); + if (!unknownOptions.isEmpty()) + throw new ConfigurationException(String.format("Properties specified %s are not understood by %s", unknownOptions.keySet(), indexerClass.getSimpleName())); + } + catch (NoSuchMethodException e) + { + logger.info("Indexer {} does not have a static validateOptions method. Validation ignored", + indexerClass.getName()); + } + catch (InvocationTargetException e) + { + if (e.getTargetException() instanceof ConfigurationException) + throw (ConfigurationException) e.getTargetException(); + throw new ConfigurationException("Failed to validate custom indexer options: " + options); + } + catch (ConfigurationException e) + { + throw e; + } + catch (Exception e) + { + throw new ConfigurationException("Failed to validate custom indexer options: " + options); + } + } + + // to be removed in CASSANDRA-10124 with multi-target & row based indexes public ColumnDefinition indexedColumn(CFMetaData cfm) { return cfm.getColumnDefinition(columns.iterator().next()); @@ -132,11 +182,29 @@ public final class IndexMetadata return indexType == IndexType.COMPOSITES; } + public boolean isRowIndex() + { + return targetType == TargetType.ROW; + } + + public boolean isColumnIndex() + { + return targetType == TargetType.COLUMN; + } + public int hashCode() { return Objects.hashCode(name, indexType, targetType, options, columns); } + public boolean equalsWithoutName(IndexMetadata other) + { + return Objects.equal(indexType, other.indexType) + && Objects.equal(targetType, other.targetType) + && Objects.equal(columns, other.columns) + && Objects.equal(options, other.options); + } + public boolean equals(Object obj) { if (obj == this) @@ -147,11 +215,7 @@ public final class IndexMetadata IndexMetadata other = (IndexMetadata)obj; - return Objects.equal(name, other.name) - && Objects.equal(indexType, other.indexType) - && Objects.equal(targetType, other.targetType) - && Objects.equal(options, other.options) - && Objects.equal(columns, other.columns); + return Objects.equal(name, other.name) && equalsWithoutName(other); } public String toString() diff --git a/src/java/org/apache/cassandra/schema/Indexes.java b/src/java/org/apache/cassandra/schema/Indexes.java index 7c930b30e6..6227e0ba51 100644 --- a/src/java/org/apache/cassandra/schema/Indexes.java +++ b/src/java/org/apache/cassandra/schema/Indexes.java @@ -21,6 +21,7 @@ package org.apache.cassandra.schema; import java.util.*; import com.google.common.collect.ImmutableMap; +import com.google.common.collect.ImmutableMultimap; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.config.Schema; @@ -39,17 +40,13 @@ import static com.google.common.collect.Iterables.filter; */ public class Indexes implements Iterable { - // lookup for index by target column - private final ImmutableMap indexes; + private final ImmutableMap indexes; + private final ImmutableMultimap indexesByColumn; private Indexes(Builder builder) { - ImmutableMap.Builder internalBuilder = ImmutableMap.builder(); - builder.indexes.build() - .values() - .stream() - .forEach(def -> internalBuilder.put(def.columns.iterator().next(), def)); - indexes = internalBuilder.build(); + indexes = builder.indexes.build(); + indexesByColumn = builder.indexesByColumn.build(); } public static Builder builder() @@ -105,9 +102,9 @@ public class Indexes implements Iterable * @param column a column definition for which an {@link IndexMetadata} is being sought * @return an empty {@link Optional} if the named index is not found; a non-empty optional of {@link IndexMetadata} otherwise */ - public Optional get(ColumnDefinition column) + public Collection get(ColumnDefinition column) { - return Optional.ofNullable(indexes.get(column.name)); + return indexesByColumn.get(column.name); } /** @@ -117,7 +114,7 @@ public class Indexes implements Iterable */ public boolean hasIndexFor(ColumnDefinition column) { - return indexes.get(column.name) != null; + return !indexesByColumn.get(column.name).isEmpty(); } /** @@ -183,6 +180,7 @@ public class Indexes implements Iterable public static final class Builder { final ImmutableMap.Builder indexes = new ImmutableMap.Builder<>(); + final ImmutableMultimap.Builder indexesByColumn = new ImmutableMultimap.Builder<>(); private Builder() { @@ -196,6 +194,13 @@ public class Indexes implements Iterable public Builder add(IndexMetadata index) { indexes.put(index.name, index); + // All indexes are column indexes at the moment + if (index.isColumnIndex()) + { + for (ColumnIdentifier target : index.columns) + indexesByColumn.put(target, index); + + } return this; } diff --git a/src/java/org/apache/cassandra/schema/LegacySchemaMigrator.java b/src/java/org/apache/cassandra/schema/LegacySchemaMigrator.java index c8b566c3c5..7c0eadf72f 100644 --- a/src/java/org/apache/cassandra/schema/LegacySchemaMigrator.java +++ b/src/java/org/apache/cassandra/schema/LegacySchemaMigrator.java @@ -27,7 +27,9 @@ import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.cassandra.config.*; -import org.apache.cassandra.cql3.*; +import org.apache.cassandra.cql3.ColumnIdentifier; +import org.apache.cassandra.cql3.QueryProcessor; +import org.apache.cassandra.cql3.UntypedResultSet; import org.apache.cassandra.cql3.functions.FunctionName; import org.apache.cassandra.cql3.functions.UDAggregate; import org.apache.cassandra.cql3.functions.UDFunction; @@ -628,7 +630,7 @@ public final class LegacySchemaMigrator isStaticCompactTable, needsUpgrade); - indexes.add(IndexMetadata.legacyIndex(column, indexName, indexType, indexOptions)); + indexes.add(IndexMetadata.singleColumnIndex(column, indexName, indexType, indexOptions)); } return indexes.build(); diff --git a/src/java/org/apache/cassandra/schema/SchemaKeyspace.java b/src/java/org/apache/cassandra/schema/SchemaKeyspace.java index d35ef8226d..acd9e9347a 100644 --- a/src/java/org/apache/cassandra/schema/SchemaKeyspace.java +++ b/src/java/org/apache/cassandra/schema/SchemaKeyspace.java @@ -38,7 +38,6 @@ import org.apache.cassandra.cql3.ColumnIdentifier; import org.apache.cassandra.cql3.QueryProcessor; import org.apache.cassandra.cql3.UntypedResultSet; import org.apache.cassandra.cql3.functions.*; -import org.apache.cassandra.db.ClusteringComparator; import org.apache.cassandra.db.*; import org.apache.cassandra.db.marshal.*; import org.apache.cassandra.db.partitions.*; @@ -1427,7 +1426,7 @@ public final class SchemaKeyspace findColumnIdentifierWithName(targetColumnName, cfm.allColumns()).ifPresent(targetColumns::add); }); } - return IndexMetadata.legacyIndex(targetColumns.iterator().next(), name, type, options); + return IndexMetadata.singleColumnIndex(targetColumns.iterator().next(), name, type, options); } private static Optional findColumnIdentifierWithName(String name, diff --git a/src/java/org/apache/cassandra/service/DataResolver.java b/src/java/org/apache/cassandra/service/DataResolver.java index 3dc323e059..cd69ef3a7c 100644 --- a/src/java/org/apache/cassandra/service/DataResolver.java +++ b/src/java/org/apache/cassandra/service/DataResolver.java @@ -26,14 +26,12 @@ import org.apache.cassandra.concurrent.StageManager; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.filter.*; +import org.apache.cassandra.db.filter.ClusteringIndexFilter; +import org.apache.cassandra.db.filter.DataLimits; import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.*; import org.apache.cassandra.exceptions.ReadTimeoutException; -import org.apache.cassandra.net.AsyncOneResponse; -import org.apache.cassandra.net.MessageIn; -import org.apache.cassandra.net.MessageOut; -import org.apache.cassandra.net.MessagingService; +import org.apache.cassandra.net.*; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.utils.FBUtilities; @@ -237,7 +235,7 @@ public class DataResolver extends ResponseResolver if (merged.isEmpty()) return; - Rows.diff(merged, columns, versions, diffListener); + Rows.diff(diffListener, merged, columns, versions); for (int i = 0; i < currentRows.length; i++) { if (currentRows[i] != null) diff --git a/src/java/org/apache/cassandra/service/StorageProxy.java b/src/java/org/apache/cassandra/service/StorageProxy.java index 12c2c241fa..9d999ee1db 100644 --- a/src/java/org/apache/cassandra/service/StorageProxy.java +++ b/src/java/org/apache/cassandra/service/StorageProxy.java @@ -31,11 +31,6 @@ import com.google.common.base.Predicate; import com.google.common.cache.CacheLoader; import com.google.common.collect.*; import com.google.common.util.concurrent.Uninterruptibles; - -import org.apache.cassandra.db.view.MaterializedViewManager; -import org.apache.cassandra.db.view.MaterializedViewUtils; -import org.apache.cassandra.db.HintedHandOffManager; -import org.apache.cassandra.metrics.*; import org.apache.commons.lang3.StringUtils; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -46,31 +41,31 @@ import org.apache.cassandra.config.CFMetaData; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.config.Schema; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.filter.DataLimits; import org.apache.cassandra.db.filter.TombstoneOverwhelmingException; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; import org.apache.cassandra.db.marshal.UUIDType; -import org.apache.cassandra.dht.AbstractBounds; -import org.apache.cassandra.dht.Bounds; -import org.apache.cassandra.dht.RingPosition; -import org.apache.cassandra.dht.Token; +import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.RowIterator; +import org.apache.cassandra.db.view.MaterializedViewManager; +import org.apache.cassandra.db.view.MaterializedViewUtils; +import org.apache.cassandra.dht.*; import org.apache.cassandra.exceptions.*; import org.apache.cassandra.gms.FailureDetector; import org.apache.cassandra.gms.Gossiper; import org.apache.cassandra.hints.Hint; import org.apache.cassandra.hints.HintsService; +import org.apache.cassandra.index.Index; import org.apache.cassandra.io.util.DataOutputBuffer; -import org.apache.cassandra.locator.AbstractReplicationStrategy; -import org.apache.cassandra.locator.IEndpointSnitch; -import org.apache.cassandra.locator.LocalStrategy; -import org.apache.cassandra.locator.TokenMetadata; +import org.apache.cassandra.locator.*; +import org.apache.cassandra.metrics.*; import org.apache.cassandra.net.*; -import org.apache.cassandra.service.paxos.*; +import org.apache.cassandra.service.paxos.Commit; +import org.apache.cassandra.service.paxos.PrepareCallback; +import org.apache.cassandra.service.paxos.ProposeCallback; import org.apache.cassandra.tracing.Tracing; import org.apache.cassandra.triggers.TriggerExecutor; -import org.apache.cassandra.utils.*; import org.apache.cassandra.utils.AbstractIterator; +import org.apache.cassandra.utils.*; public class StorageProxy implements StorageProxyMBean { @@ -1698,11 +1693,10 @@ public class StorageProxy implements StorageProxyMBean private static float estimateResultsPerRange(PartitionRangeReadCommand command, Keyspace keyspace) { ColumnFamilyStore cfs = keyspace.getColumnFamilyStore(command.metadata().cfId); - SecondaryIndexSearcher searcher = cfs.indexManager.getBestIndexSearcherFor(command); - - float maxExpectedResults = searcher == null + Index index = cfs.indexManager.getBestIndexFor(command); + float maxExpectedResults = index == null ? command.limits().estimateTotalResults(cfs) - : searcher.highestSelectivityIndex(command.rowFilter()).estimateResultRows(); + : index.getEstimatedResultRows(); // adjust maxExpectedResults by the number of tokens this node has and the replication factor for this ks return (maxExpectedResults / DatabaseDescriptor.getNumTokens()) / keyspace.getReplicationStrategy().getReplicationFactor(); @@ -1962,7 +1956,9 @@ public class StorageProxy implements StorageProxyMBean } Tracing.trace("Submitted {} concurrent range requests", concurrentQueries.size()); - return new CountingPartitionIterator(PartitionIterators.concat(concurrentQueries), command.limits(), command.nowInSec()); + // We want to count the results for the sake of updating the concurrency factor (see updateConcurrencyFactor) but we don't want to + // enforce any particular limit at this point (this could break code than rely on postReconciliationProcessing), hence the DataLimits.NONE. + return new CountingPartitionIterator(PartitionIterators.concat(concurrentQueries), DataLimits.NONE, command.nowInSec()); } public void close() @@ -2003,7 +1999,8 @@ public class StorageProxy implements StorageProxyMBean Tracing.trace("Submitting range requests on {} ranges with a concurrency of {} ({} rows per range expected)", ranges.rangeCount(), concurrencyFactor, resultsPerRange); // Note that in general, a RangeCommandIterator will honor the command limit for each range, but will not enforce it globally. - return command.postReconciliationProcessing(command.limits().filter(new RangeCommandIterator(ranges, command, concurrencyFactor, keyspace, consistencyLevel), command.nowInSec())); + + return command.limits().filter(command.postReconciliationProcessing(new RangeCommandIterator(ranges, command, concurrencyFactor, keyspace, consistencyLevel)), command.nowInSec()); } public Map> getSchemaVersions() diff --git a/src/java/org/apache/cassandra/streaming/StreamReceiveTask.java b/src/java/org/apache/cassandra/streaming/StreamReceiveTask.java index 2bcbbc1c36..52c8884eb6 100644 --- a/src/java/org/apache/cassandra/streaming/StreamReceiveTask.java +++ b/src/java/org/apache/cassandra/streaming/StreamReceiveTask.java @@ -17,10 +17,7 @@ */ package org.apache.cassandra.streaming; -import java.util.ArrayList; -import java.util.Collection; -import java.util.List; -import java.util.UUID; +import java.util.*; import java.util.concurrent.ExecutorService; import java.util.concurrent.Executors; @@ -31,21 +28,17 @@ import org.apache.cassandra.concurrent.NamedThreadFactory; import org.apache.cassandra.config.Schema; import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.Keyspace; -import org.apache.cassandra.db.lifecycle.LifecycleTransaction; -import org.apache.cassandra.db.compaction.OperationType; import org.apache.cassandra.db.Mutation; +import org.apache.cassandra.db.compaction.OperationType; +import org.apache.cassandra.db.lifecycle.LifecycleTransaction; import org.apache.cassandra.db.partitions.PartitionUpdate; -import org.apache.cassandra.db.rows.RowIterators; -import org.apache.cassandra.db.rows.Unfiltered; import org.apache.cassandra.db.rows.UnfilteredRowIterator; -import org.apache.cassandra.db.rows.UnfilteredRowIterators; import org.apache.cassandra.io.sstable.ISSTableScanner; import org.apache.cassandra.io.sstable.SSTableMultiWriter; import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.io.sstable.format.SSTableWriter; import org.apache.cassandra.utils.JVMStabilityInspector; import org.apache.cassandra.utils.Pair; - import org.apache.cassandra.utils.concurrent.Refs; /** @@ -176,7 +169,7 @@ public class StreamReceiveTask extends StreamTask // add sstables and build secondary indexes cfs.addSSTables(readers); - cfs.indexManager.maybeBuildSecondaryIndexes(readers, cfs.indexManager.allIndexesNames()); + cfs.indexManager.buildAllIndexesBlocking(readers); } } catch (Throwable t) diff --git a/src/java/org/apache/cassandra/thrift/CassandraServer.java b/src/java/org/apache/cassandra/thrift/CassandraServer.java index 7a2126c0f2..fd391aa3be 100644 --- a/src/java/org/apache/cassandra/thrift/CassandraServer.java +++ b/src/java/org/apache/cassandra/thrift/CassandraServer.java @@ -38,12 +38,12 @@ import org.apache.cassandra.config.*; import org.apache.cassandra.cql3.QueryOptions; import org.apache.cassandra.cql3.statements.ParsedStatement; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; import org.apache.cassandra.db.context.CounterContext; import org.apache.cassandra.db.filter.*; import org.apache.cassandra.db.marshal.AbstractType; import org.apache.cassandra.db.marshal.TimeUUIDType; +import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.*; import org.apache.cassandra.dht.*; import org.apache.cassandra.dht.Range; import org.apache.cassandra.exceptions.*; @@ -826,6 +826,9 @@ public class CassandraServer implements Cassandra.Iface Cell cell = cellFromColumn(metadata, name, column); PartitionUpdate update = PartitionUpdate.singleRowUpdate(metadata, key, BTreeRow.singleCellRow(name.clustering, cell)); + // Indexed column values cannot be larger than 64K. See CASSANDRA-3057/4240 for more details + Keyspace.open(metadata.ksName).getColumnFamilyStore(metadata.cfName).indexManager.validate(update); + mutation = new org.apache.cassandra.db.Mutation(update); } catch (MarshalException|UnknownColumnException e) @@ -916,6 +919,8 @@ public class CassandraServer implements Cassandra.Iface int nowInSec = FBUtilities.nowInSeconds(); PartitionUpdate partitionUpdates = PartitionUpdate.fromIterator(LegacyLayout.toRowIterator(metadata, dk, toLegacyCells(metadata, updates, nowInSec).iterator(), nowInSec)); + // Indexed column values cannot be larger than 64K. See CASSANDRA-3057/4240 for more details + Keyspace.open(metadata.ksName).getColumnFamilyStore(metadata.cfName).indexManager.validate(partitionUpdates); FilteredPartition partitionExpected = null; if (!expected.isEmpty()) @@ -1121,6 +1126,9 @@ public class CassandraServer implements Cassandra.Iface DecoratedKey dk = metadata.decorateKey(key); PartitionUpdate update = PartitionUpdate.fromIterator(LegacyLayout.toUnfilteredRowIterator(metadata, dk, delInfo, cells.iterator())); + // Indexed column values cannot be larger than 64K. See CASSANDRA-3057/4240 for more details + Keyspace.open(metadata.ksName).getColumnFamilyStore(metadata.cfName).indexManager.validate(update); + org.apache.cassandra.db.Mutation mutation; if (metadata.isCounter()) { diff --git a/src/java/org/apache/cassandra/thrift/ThriftConversion.java b/src/java/org/apache/cassandra/thrift/ThriftConversion.java index a74bcea318..4c17a4b81c 100644 --- a/src/java/org/apache/cassandra/thrift/ThriftConversion.java +++ b/src/java/org/apache/cassandra/thrift/ThriftConversion.java @@ -23,21 +23,17 @@ import com.google.common.annotations.VisibleForTesting; import com.google.common.base.Strings; import com.google.common.collect.Maps; -import org.apache.cassandra.db.compaction.AbstractCompactionStrategy; -import org.apache.cassandra.io.compress.ICompressor; -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.config.DatabaseDescriptor; -import org.apache.cassandra.config.Schema; +import org.apache.cassandra.config.*; import org.apache.cassandra.cql3.ColumnIdentifier; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.db.CompactTables; import org.apache.cassandra.db.LegacyLayout; import org.apache.cassandra.db.WriteType; +import org.apache.cassandra.db.compaction.AbstractCompactionStrategy; import org.apache.cassandra.db.filter.RowFilter; import org.apache.cassandra.db.marshal.*; import org.apache.cassandra.exceptions.*; -import org.apache.cassandra.schema.CompressionParams; +import org.apache.cassandra.io.compress.ICompressor; import org.apache.cassandra.locator.AbstractReplicationStrategy; import org.apache.cassandra.locator.LocalStrategy; import org.apache.cassandra.schema.*; @@ -560,10 +556,10 @@ public class ThriftConversion Map indexOptions = def.getIndex_options(); IndexMetadata.IndexType indexType = IndexMetadata.IndexType.valueOf(def.index_type.name()); - indexes.add(IndexMetadata.legacyIndex(column, - indexName, - indexType, - indexOptions)); + indexes.add(IndexMetadata.singleColumnIndex(column, + indexName, + indexType, + indexOptions)); } } return indexes.build(); @@ -576,13 +572,22 @@ public class ThriftConversion cd.setName(ByteBufferUtil.clone(column.name.bytes)); cd.setValidation_class(column.type.toString()); - Optional index = cfMetaData.getIndexes().get(column); - index.ifPresent(def -> { - cd.setIndex_type(org.apache.cassandra.thrift.IndexType.valueOf(def.indexType.name())); - cd.setIndex_name(def.name); - cd.setIndex_options(def.options == null || def.options.isEmpty() ? null : Maps.newHashMap(def.options)); - }); - + Collection indexes = cfMetaData.getIndexes().get(column); + // we include the index in the ColumnDef iff + // * it is the only index on the column + // * it is the only target column for the index + if (indexes.size() == 1) + { + IndexMetadata index = indexes.iterator().next(); + if (index.columns.size() == 1) + { + cd.setIndex_type(org.apache.cassandra.thrift.IndexType.valueOf(index.indexType.name())); + cd.setIndex_name(index.name); + cd.setIndex_options(index.options == null || index.options.isEmpty() + ? null + : Maps.newHashMap(index.options)); + } + } return cd; } diff --git a/src/java/org/apache/cassandra/thrift/ThriftValidation.java b/src/java/org/apache/cassandra/thrift/ThriftValidation.java index 054b466af4..71aa335a1d 100644 --- a/src/java/org/apache/cassandra/thrift/ThriftValidation.java +++ b/src/java/org/apache/cassandra/thrift/ThriftValidation.java @@ -18,18 +18,24 @@ package org.apache.cassandra.thrift; import java.nio.ByteBuffer; -import java.util.*; +import java.util.Arrays; +import java.util.Comparator; +import java.util.List; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import org.apache.cassandra.config.*; +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.config.Schema; import org.apache.cassandra.cql3.Attributes; +import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.index.SecondaryIndexManager; import org.apache.cassandra.db.marshal.AbstractType; import org.apache.cassandra.dht.IPartitioner; import org.apache.cassandra.dht.Token; +import org.apache.cassandra.index.Index; +import org.apache.cassandra.index.SecondaryIndexManager; import org.apache.cassandra.serializers.MarshalException; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FBUtilities; @@ -449,8 +455,6 @@ public class ThriftValidation LegacyLayout.LegacyCellName cn = LegacyLayout.decodeCellName(metadata, scName, column.name); cn.column.validateCellValue(column.value); - // Indexed column values cannot be larger than 64K. See CASSANDRA-3057/4240 for more details - Keyspace.open(metadata.ksName).getColumnFamilyStore(metadata.cfName).indexManager.validate(cn.column, column.value, null); } catch (UnknownColumnException e) { @@ -609,7 +613,8 @@ public class ThriftValidation me.getMessage())); } - isIndexed |= (expression.op == IndexOperator.EQ) && idxManager.indexes(def); + for(Index index : idxManager.listIndexes()) + isIndexed |= index.supportsExpression(def, Operator.valueOf(expression.op.name())); } return isIndexed; diff --git a/test/unit/org/apache/cassandra/SchemaLoader.java b/test/unit/org/apache/cassandra/SchemaLoader.java index 1c0dd76ffc..8675d7fa74 100644 --- a/test/unit/org/apache/cassandra/SchemaLoader.java +++ b/test/unit/org/apache/cassandra/SchemaLoader.java @@ -27,15 +27,15 @@ import org.junit.BeforeClass; import org.apache.cassandra.config.*; import org.apache.cassandra.cql3.CQLTester; import org.apache.cassandra.cql3.ColumnIdentifier; -import org.apache.cassandra.db.*; +import org.apache.cassandra.cql3.statements.IndexTarget; +import org.apache.cassandra.db.RowUpdateBuilder; import org.apache.cassandra.db.commitlog.CommitLog; -import org.apache.cassandra.db.index.PerRowSecondaryIndexTest; -import org.apache.cassandra.db.index.SecondaryIndex; import org.apache.cassandra.db.marshal.*; import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.gms.Gossiper; -import org.apache.cassandra.schema.*; +import org.apache.cassandra.index.StubIndex; import org.apache.cassandra.io.util.FileUtils; +import org.apache.cassandra.schema.*; import org.apache.cassandra.service.MigrationManager; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FBUtilities; @@ -218,10 +218,6 @@ public class SchemaLoader schema.add(KeyspaceMetadata.create(ks_nocommit, KeyspaceParams.simpleTransient(1), Tables.of( standardCFMD(ks_nocommit, "Standard1")))); - // PerRowSecondaryIndexTest - schema.add(KeyspaceMetadata.create(ks_prsi, KeyspaceParams.simple(1), Tables.of( - perRowIndexedCFMD(ks_prsi, "Indexed1")))); - // CQLKeyspace schema.add(KeyspaceMetadata.create(ks_cql, KeyspaceParams.simple(1), Tables.of( @@ -291,8 +287,8 @@ public class SchemaLoader public static CFMetaData perRowIndexedCFMD(String ksName, String cfName) { final Map indexOptions = Collections.singletonMap( - SecondaryIndex.CUSTOM_INDEX_OPTION_NAME, - PerRowSecondaryIndexTest.TestIndex.class.getName()); + IndexTarget.CUSTOM_INDEX_OPTION_NAME, + StubIndex.class.getName()); CFMetaData cfm = CFMetaData.Builder.create(ksName, cfName) .addPartitionKey("key", AsciiType.instance) @@ -303,10 +299,10 @@ public class SchemaLoader cfm.indexes( cfm.getIndexes() - .with(IndexMetadata.legacyIndex(indexedColumn, - "indexe1", - IndexMetadata.IndexType.CUSTOM, - indexOptions))); + .with(IndexMetadata.singleColumnIndex(indexedColumn, + "indexe1", + IndexMetadata.IndexType.CUSTOM, + indexOptions))); return cfm; } @@ -414,10 +410,10 @@ public class SchemaLoader if (withIndex) cfm.indexes( cfm.getIndexes() - .with(IndexMetadata.legacyIndex(cfm.getColumnDefinition(new ColumnIdentifier("birthdate", true)), - "birthdate_key_index", - IndexMetadata.IndexType.COMPOSITES, - Collections.EMPTY_MAP))); + .with(IndexMetadata.singleColumnIndex(cfm.getColumnDefinition(new ColumnIdentifier("birthdate", true)), + "birthdate_key_index", + IndexMetadata.IndexType.COMPOSITES, + Collections.EMPTY_MAP))); return cfm.compression(getCompressionParameters()); } @@ -434,10 +430,10 @@ public class SchemaLoader if (withIndex) cfm.indexes( cfm.getIndexes() - .with(IndexMetadata.legacyIndex(cfm.getColumnDefinition(new ColumnIdentifier("birthdate", true)), - "birthdate_composite_index", - IndexMetadata.IndexType.KEYS, - Collections.EMPTY_MAP))); + .with(IndexMetadata.singleColumnIndex(cfm.getColumnDefinition(new ColumnIdentifier("birthdate", true)), + "birthdate_composite_index", + IndexMetadata.IndexType.KEYS, + Collections.EMPTY_MAP))); return cfm.compression(getCompressionParameters()); diff --git a/test/unit/org/apache/cassandra/Util.java b/test/unit/org/apache/cassandra/Util.java index e8451e0f33..5e19d5e2b0 100644 --- a/test/unit/org/apache/cassandra/Util.java +++ b/test/unit/org/apache/cassandra/Util.java @@ -32,7 +32,6 @@ import java.util.function.Supplier; import com.google.common.base.Function; import com.google.common.base.Preconditions; import com.google.common.collect.Iterators; - import org.apache.commons.lang3.StringUtils; import org.apache.cassandra.config.CFMetaData; @@ -40,13 +39,15 @@ import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.cql3.ColumnIdentifier; import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; import org.apache.cassandra.db.compaction.AbstractCompactionTask; import org.apache.cassandra.db.compaction.CompactionManager; import org.apache.cassandra.db.marshal.AbstractType; import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.dht.*; +import org.apache.cassandra.db.rows.*; +import org.apache.cassandra.dht.IPartitioner; import org.apache.cassandra.dht.RandomPartitioner.BigIntegerToken; +import org.apache.cassandra.dht.Range; +import org.apache.cassandra.dht.Token; import org.apache.cassandra.gms.ApplicationState; import org.apache.cassandra.gms.Gossiper; import org.apache.cassandra.gms.VersionedValue; @@ -531,4 +532,10 @@ public class Util { thread.join(10000); } + + // for use with Optional in tests, can be used as an argument to orElseThrow + public static Supplier throwAssert(final String message) + { + return () -> new AssertionError(message); + } } diff --git a/test/unit/org/apache/cassandra/cql3/validation/entities/SecondaryIndexTest.java b/test/unit/org/apache/cassandra/cql3/validation/entities/SecondaryIndexTest.java index be3568a149..02b2abd939 100644 --- a/test/unit/org/apache/cassandra/cql3/validation/entities/SecondaryIndexTest.java +++ b/test/unit/org/apache/cassandra/cql3/validation/entities/SecondaryIndexTest.java @@ -18,20 +18,24 @@ package org.apache.cassandra.cql3.validation.entities; import java.nio.ByteBuffer; -import java.util.HashMap; -import java.util.Locale; -import java.util.Map; -import java.util.UUID; +import java.util.*; import org.apache.commons.lang3.StringUtils; - import org.junit.Test; +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.CQLTester; +import org.apache.cassandra.cql3.ColumnIdentifier; +import org.apache.cassandra.db.ColumnFamilyStore; +import org.apache.cassandra.db.marshal.AbstractType; +import org.apache.cassandra.db.rows.Row; import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.exceptions.SyntaxException; -import org.apache.cassandra.utils.FBUtilities; +import org.apache.cassandra.index.StubIndex; +import static org.apache.cassandra.Util.throwAssert; +import static org.junit.Assert.assertEquals; import static org.junit.Assert.assertTrue; import static org.junit.Assert.fail; @@ -70,13 +74,23 @@ public class SecondaryIndexTest extends CQLTester private void testCreateAndDropIndex(String indexName, boolean addKeyspaceOnDrop) throws Throwable { execute("USE system"); - assertInvalidMessage("Index '" + removeQuotes(indexName.toLowerCase(Locale.US)) + "' could not be found", "DROP INDEX " + indexName + ";"); + assertInvalidMessage(String.format("Index '%s' could not be found", + removeQuotes(indexName.toLowerCase(Locale.US))), + "DROP INDEX " + indexName + ";"); createTable("CREATE TABLE %s (a int primary key, b int);"); createIndex("CREATE INDEX " + indexName + " ON %s(b);"); createIndex("CREATE INDEX IF NOT EXISTS " + indexName + " ON %s(b);"); - assertInvalidMessage("Index already exists", "CREATE INDEX " + indexName + " ON %s(b)"); + assertInvalidMessage(String.format("Index %s already exists", + removeQuotes(indexName.toLowerCase(Locale.US))), + "CREATE INDEX " + indexName + " ON %s(b)"); + + String otherIndexName = "index_" + System.nanoTime(); + assertInvalidMessage(String.format("Index %s is a duplicate of existing index %s", + removeQuotes(otherIndexName.toLowerCase(Locale.US)), + removeQuotes(indexName.toLowerCase(Locale.US))), + "CREATE INDEX " + otherIndexName + " ON %s(b)"); execute("INSERT INTO %s (a, b) values (?, ?);", 0, 0); execute("INSERT INTO %s (a, b) values (?, ?);", 1, 1); @@ -84,7 +98,8 @@ public class SecondaryIndexTest extends CQLTester execute("INSERT INTO %s (a, b) values (?, ?);", 3, 1); assertRows(execute("SELECT * FROM %s where b = ?", 1), row(1, 1), row(3, 1)); - assertInvalidMessage("Index '" + removeQuotes(indexName.toLowerCase(Locale.US)) + "' could not be found in any of the tables of keyspace 'system'", + assertInvalidMessage(String.format("Index '%s' could not be found in any of the tables of keyspace 'system'", + removeQuotes(indexName.toLowerCase(Locale.US))), "DROP INDEX " + indexName); if (addKeyspaceOnDrop) @@ -100,7 +115,9 @@ public class SecondaryIndexTest extends CQLTester assertInvalidMessage("No supported secondary index found for the non primary key columns restrictions", "SELECT * FROM %s where b = ?", 1); dropIndex("DROP INDEX IF EXISTS " + indexName); - assertInvalidMessage("Index '" + removeQuotes(indexName.toLowerCase(Locale.US)) + "' could not be found", "DROP INDEX " + indexName); + assertInvalidMessage(String.format("Index '%s' could not be found", + removeQuotes(indexName.toLowerCase(Locale.US))), + "DROP INDEX " + indexName); } /** @@ -189,12 +206,10 @@ public class SecondaryIndexTest extends CQLTester public void testUnknownCompressionOptions() throws Throwable { String tableName = createTableName(); - assertInvalidThrow(SyntaxException.class, String.format( - "CREATE TABLE %s (key varchar PRIMARY KEY, password varchar, gender varchar) WITH compression_parameters:sstable_compressor = 'DeflateCompressor'", tableName)); + assertInvalidThrow(SyntaxException.class, String.format("CREATE TABLE %s (key varchar PRIMARY KEY, password varchar, gender varchar) WITH compression_parameters:sstable_compressor = 'DeflateCompressor'", tableName)); - - assertInvalidThrow(ConfigurationException.class, String.format( - "CREATE TABLE %s (key varchar PRIMARY KEY, password varchar, gender varchar) WITH compression = { 'sstable_compressor': 'DeflateCompressor' }", tableName)); + assertInvalidThrow(ConfigurationException.class, String.format("CREATE TABLE %s (key varchar PRIMARY KEY, password varchar, gender varchar) WITH compression = { 'sstable_compressor': 'DeflateCompressor' }", + tableName)); } /** @@ -400,9 +415,6 @@ public class SecondaryIndexTest extends CQLTester assertRows(execute("SELECT k, v FROM %s WHERE k = 0 AND m CONTAINS KEY 'a'"), row(0, 0), row(0, 1)); assertRows(execute("SELECT k, v FROM %s WHERE m CONTAINS KEY 'c'"), row(0, 2)); assertEmpty(execute("SELECT k, v FROM %s WHERE m CONTAINS KEY 'd'")); - - // we're not allowed to create a value index if we already have a key one - assertInvalid("CREATE INDEX ON %s(m)"); } /** @@ -412,7 +424,7 @@ public class SecondaryIndexTest extends CQLTester @Test public void testIndexOnKeyWithReverseClustering() throws Throwable { - createTable(" CREATE TABLE %s (k1 int, k2 int, v int, PRIMARY KEY ((k1, k2), v) ) WITH CLUSTERING ORDER BY (v DESC)"); + createTable("CREATE TABLE %s (k1 int, k2 int, v int, PRIMARY KEY ((k1, k2), v) ) WITH CLUSTERING ORDER BY (v DESC)"); createIndex("CREATE INDEX ON %s (k2)"); @@ -591,4 +603,51 @@ public class SecondaryIndexTest extends CQLTester assertInvalid("CREATE INDEX ON %s (c)"); } + @Test + public void testMultipleIndexesOnOneColumn() throws Throwable + { + String indexClassName = StubIndex.class.getName(); + createTable("CREATE TABLE %s (a int, b int, c int, PRIMARY KEY ((a), b))"); + // uses different options otherwise the two indexes are considered duplicates + createIndex(String.format("CREATE CUSTOM INDEX c_idx_1 ON %%s(c) USING '%s' WITH OPTIONS = {'foo':'a'}", indexClassName)); + createIndex(String.format("CREATE CUSTOM INDEX c_idx_2 ON %%s(c) USING '%s' WITH OPTIONS = {'foo':'b'}", indexClassName)); + + ColumnFamilyStore cfs = getCurrentColumnFamilyStore(); + CFMetaData cfm = cfs.metadata; + StubIndex index1 = (StubIndex)cfs.indexManager.getIndex(cfm.getIndexes() + .get("c_idx_1") + .orElseThrow(throwAssert("index not found"))); + StubIndex index2 = (StubIndex)cfs.indexManager.getIndex(cfm.getIndexes() + .get("c_idx_2") + .orElseThrow(throwAssert("index not found"))); + Object[] row1a = row(0, 0, 0); + Object[] row1b = row(0, 0, 1); + Object[] row2 = row(2, 2, 2); + execute("INSERT INTO %s (a, b, c) VALUES (?, ?, ?)", row1a); + execute("INSERT INTO %s (a, b, c) VALUES (?, ?, ?)", row1b); + execute("INSERT INTO %s (a, b, c) VALUES (?, ?, ?)", row2); + + assertEquals(2, index1.rowsInserted.size()); + assertColumnValue(0, "c", index1.rowsInserted.get(0), cfm); + assertColumnValue(2, "c", index1.rowsInserted.get(1), cfm); + + assertEquals(2, index2.rowsInserted.size()); + assertColumnValue(0, "c", index2.rowsInserted.get(0), cfm); + assertColumnValue(2, "c", index2.rowsInserted.get(1), cfm); + + assertEquals(1, index1.rowsUpdated.size()); + assertColumnValue(0, "c", index1.rowsUpdated.get(0).left, cfm); + assertColumnValue(1, "c", index1.rowsUpdated.get(0).right, cfm); + + assertEquals(1, index2.rowsUpdated.size()); + assertColumnValue(0, "c", index2.rowsUpdated.get(0).left, cfm); + assertColumnValue(1, "c", index2.rowsUpdated.get(0).right, cfm); + } + + private static void assertColumnValue(int expected, String name, Row row, CFMetaData cfm) + { + ColumnDefinition col = cfm.getColumnDefinition(new ColumnIdentifier(name, true)); + AbstractType type = col.type; + assertEquals(expected, type.compose(row.getCell(col).value())); + } } diff --git a/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java b/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java index 8e1f43825e..e0879d24c1 100644 --- a/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java +++ b/test/unit/org/apache/cassandra/cql3/validation/miscellaneous/CrcCheckChanceTest.java @@ -21,6 +21,8 @@ import java.util.List; import java.util.concurrent.ExecutionException; import java.util.concurrent.Future; +import org.junit.Test; + import junit.framework.Assert; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.cql3.CQLTester; @@ -30,8 +32,6 @@ import org.apache.cassandra.db.compaction.CompactionInterruptedException; import org.apache.cassandra.db.compaction.CompactionManager; import org.apache.cassandra.utils.FBUtilities; -import org.junit.Test; - public class CrcCheckChanceTest extends CQLTester { @@ -49,7 +49,7 @@ public class CrcCheckChanceTest extends CQLTester ColumnFamilyStore cfs = Keyspace.open(CQLTester.KEYSPACE).getColumnFamilyStore(currentTable()); - ColumnFamilyStore indexCfs = cfs.indexManager.getIndexesBackedByCfs().iterator().next(); + ColumnFamilyStore indexCfs = cfs.indexManager.getAllIndexColumnFamilyStores().iterator().next(); cfs.forceBlockingFlush(); Assert.assertEquals(0.99, cfs.metadata.params.compression.getCrcCheckChance()); diff --git a/test/unit/org/apache/cassandra/db/CleanupTest.java b/test/unit/org/apache/cassandra/db/CleanupTest.java index 792aaa7b29..7b03640613 100644 --- a/test/unit/org/apache/cassandra/db/CleanupTest.java +++ b/test/unit/org/apache/cassandra/db/CleanupTest.java @@ -27,17 +27,15 @@ import java.util.List; import java.util.concurrent.ExecutionException; import java.util.concurrent.TimeUnit; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.db.filter.RowFilter; - import org.junit.BeforeClass; import org.junit.Test; import org.apache.cassandra.SchemaLoader; import org.apache.cassandra.Util; +import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.db.compaction.CompactionManager; -import org.apache.cassandra.db.index.SecondaryIndex; +import org.apache.cassandra.db.filter.RowFilter; import org.apache.cassandra.dht.ByteOrderedPartitioner.BytesToken; import org.apache.cassandra.exceptions.ConfigurationException; import org.apache.cassandra.io.sstable.format.SSTableReader; @@ -115,12 +113,14 @@ public class CleanupTest fillCF(cfs, "birthdate", LOOPS); assertEquals(LOOPS, Util.getAll(Util.cmd(cfs).build()).size()); - SecondaryIndex index = cfs.indexManager.getIndexForColumn(cfs.metadata.getColumnDefinition(COLUMN)); + ColumnDefinition cdef = cfs.metadata.getColumnDefinition(COLUMN); + String indexName = cfs.metadata.getIndexes() + .get(cdef) + .iterator().next().name; long start = System.nanoTime(); - while (!index.isIndexBuilt(COLUMN) && System.nanoTime() - start < TimeUnit.SECONDS.toNanos(10)) + while (!cfs.getBuiltIndexes().contains(indexName) && System.nanoTime() - start < TimeUnit.SECONDS.toNanos(10)) Thread.sleep(10); - ColumnDefinition cdef = cfs.metadata.getColumnDefinition(COLUMN); RowFilter cf = RowFilter.create(); cf.add(cdef, Operator.EQ, VALUE); assertEquals(LOOPS, Util.getAll(Util.cmd(cfs).filterOn("birthdate", Operator.EQ, VALUE).build()).size()); diff --git a/test/unit/org/apache/cassandra/db/DirectoriesTest.java b/test/unit/org/apache/cassandra/db/DirectoriesTest.java index cb38e37d1f..40093eacd4 100644 --- a/test/unit/org/apache/cassandra/db/DirectoriesTest.java +++ b/test/unit/org/apache/cassandra/db/DirectoriesTest.java @@ -17,33 +17,39 @@ */ package org.apache.cassandra.db; -import java.io.*; +import java.io.File; +import java.io.FileOutputStream; +import java.io.IOException; import java.util.*; import java.util.concurrent.Callable; import java.util.concurrent.Executors; import java.util.concurrent.Future; import org.apache.commons.lang3.StringUtils; - import org.junit.AfterClass; import org.junit.BeforeClass; import org.junit.Test; import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.config.Config.DiskFailurePolicy; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.db.Directories.DataDirectory; -import org.apache.cassandra.db.index.SecondaryIndex; import org.apache.cassandra.db.marshal.UTF8Type; +import org.apache.cassandra.index.internal.CassandraIndex; +import org.apache.cassandra.io.FSWriteError; import org.apache.cassandra.io.sstable.Component; import org.apache.cassandra.io.sstable.Descriptor; import org.apache.cassandra.io.util.FileUtils; import org.apache.cassandra.schema.IndexMetadata; import org.apache.cassandra.utils.ByteBufferUtil; -import org.apache.cassandra.io.FSWriteError; import org.apache.cassandra.utils.Pair; -import static org.junit.Assert.*; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertSame; +import static org.junit.Assert.assertTrue; +import static org.junit.Assert.fail; public class DirectoriesTest { @@ -161,12 +167,13 @@ public class DirectoriesTest .addPartitionKey("thekey", UTF8Type.instance) .addClusteringColumn("col", UTF8Type.instance) .build(); - IndexMetadata indexDef = IndexMetadata.legacyIndex(PARENT_CFM.getColumnDefinition(ByteBufferUtil.bytes("col")), - "idx", - IndexMetadata.IndexType.KEYS, - Collections.emptyMap()); + ColumnDefinition col = PARENT_CFM.getColumnDefinition(ByteBufferUtil.bytes("col")); + IndexMetadata indexDef = IndexMetadata.singleColumnIndex(col, + "idx", + IndexMetadata.IndexType.KEYS, + Collections.emptyMap()); PARENT_CFM.indexes(PARENT_CFM.getIndexes().with(indexDef)); - CFMetaData INDEX_CFM = SecondaryIndex.newIndexMetadata(PARENT_CFM, indexDef); + CFMetaData INDEX_CFM = CassandraIndex.indexCfsMetadata(PARENT_CFM, indexDef); Directories parentDirectories = new Directories(PARENT_CFM); Directories indexDirectories = new Directories(INDEX_CFM); // secondary index has its own directory diff --git a/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java b/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java index 70fbfe3887..ffc43c504d 100644 --- a/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java +++ b/test/unit/org/apache/cassandra/db/RangeTombstoneTest.java @@ -22,37 +22,33 @@ import java.nio.ByteBuffer; import java.util.ArrayList; import java.util.Collection; import java.util.List; -import java.util.Set; import java.util.concurrent.ExecutionException; import java.util.concurrent.Future; import com.google.common.collect.ImmutableMap; import com.google.common.collect.Iterators; - -import org.apache.cassandra.io.sstable.format.SSTableReader; import org.junit.BeforeClass; import org.junit.Test; import org.apache.cassandra.SchemaLoader; -import org.apache.cassandra.config.*; -import org.apache.cassandra.Util; import org.apache.cassandra.UpdateBuilder; +import org.apache.cassandra.Util; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.cql3.statements.IndexTarget; import org.apache.cassandra.db.compaction.CompactionManager; -import org.apache.cassandra.db.filter.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.partitions.*; -import org.apache.cassandra.db.index.PerColumnSecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; -import org.apache.cassandra.db.marshal.UTF8Type; +import org.apache.cassandra.db.filter.ColumnFilter; import org.apache.cassandra.db.marshal.Int32Type; +import org.apache.cassandra.db.marshal.UTF8Type; +import org.apache.cassandra.db.partitions.*; +import org.apache.cassandra.db.rows.*; import org.apache.cassandra.exceptions.ConfigurationException; +import org.apache.cassandra.index.StubIndex; +import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.io.sstable.metadata.StatsMetadata; import org.apache.cassandra.schema.IndexMetadata; import org.apache.cassandra.schema.KeyspaceParams; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.concurrent.OpOrder; import static org.junit.Assert.assertEquals; import static org.junit.Assert.assertFalse; @@ -71,7 +67,7 @@ public class RangeTombstoneTest KeyspaceParams.simple(1), SchemaLoader.standardCFMD(KSNAME, CFNAME, - 0, + 1, UTF8Type.instance, Int32Type.instance, Int32Type.instance)); @@ -468,22 +464,26 @@ public class RangeTombstoneTest cfs.disableAutoCompaction(); ColumnDefinition cd = cfs.metadata.getColumnDefinition(indexedColumnName).copy(); - IndexMetadata indexDef = IndexMetadata.legacyIndex(cd, - "test_index", - IndexMetadata.IndexType.CUSTOM, - ImmutableMap.of(SecondaryIndex.CUSTOM_INDEX_OPTION_NAME, - TestIndex.class.getName())); + IndexMetadata indexDef = IndexMetadata.singleColumnIndex(cd, + "test_index", + IndexMetadata.IndexType.CUSTOM, + ImmutableMap.of(IndexTarget.CUSTOM_INDEX_OPTION_NAME, + StubIndex.class.getName())); if (!cfs.metadata.getIndexes().get("test_index").isPresent()) cfs.metadata.indexes(cfs.metadata.getIndexes().with(indexDef)); - Future rebuild = cfs.indexManager.addIndexedColumn(indexDef); + Future rebuild = cfs.indexManager.addIndex(indexDef); // If rebuild there is, wait for the rebuild to finish so it doesn't race with the following insertions if (rebuild != null) rebuild.get(); - TestIndex index = ((TestIndex)cfs.indexManager.getIndexForColumn(cd)); - index.resetCounts(); + StubIndex index = (StubIndex)cfs.indexManager.listIndexes() + .stream() + .filter(i -> "test_index".equals(i.getIndexName())) + .findFirst() + .orElseThrow(() -> new RuntimeException(new AssertionError("Index not found"))); + index.reset(); UpdateBuilder builder = UpdateBuilder.create(cfs.metadata, key).withTimestamp(0); for (int i = 0; i < 10; i++) @@ -494,14 +494,14 @@ public class RangeTombstoneTest new RowUpdateBuilder(cfs.metadata, 0, key).addRangeTombstone(0, 7).build().applyUnsafe(); cfs.forceBlockingFlush(); - assertEquals(10, index.inserts.size()); + assertEquals(10, index.rowsInserted.size()); CompactionManager.instance.performMaximal(cfs, false); // compacted down to single sstable assertEquals(1, cfs.getLiveSSTables().size()); - assertEquals(8, index.deletes.size()); + assertEquals(8, index.rowsDeleted.size()); } @Test @@ -560,21 +560,26 @@ public class RangeTombstoneTest cfs.disableAutoCompaction(); ColumnDefinition cd = cfs.metadata.getColumnDefinition(indexedColumnName).copy(); - IndexMetadata indexDef = IndexMetadata.legacyIndex(cd, - "test_index", - IndexMetadata.IndexType.CUSTOM, - ImmutableMap.of(SecondaryIndex.CUSTOM_INDEX_OPTION_NAME, - TestIndex.class.getName())); + IndexMetadata indexDef = IndexMetadata.singleColumnIndex(cd, + "test_index", + IndexMetadata.IndexType.CUSTOM, + ImmutableMap.of(IndexTarget.CUSTOM_INDEX_OPTION_NAME, + StubIndex.class.getName())); if (!cfs.metadata.getIndexes().get("test_index").isPresent()) cfs.metadata.indexes(cfs.metadata.getIndexes().with(indexDef)); - Future rebuild = cfs.indexManager.addIndexedColumn(indexDef); + Future rebuild = cfs.indexManager.addIndex(indexDef); // If rebuild there is, wait for the rebuild to finish so it doesn't race with the following insertions if (rebuild != null) rebuild.get(); - TestIndex index = ((TestIndex)cfs.indexManager.getIndexForColumn(cd)); - index.resetCounts(); + + StubIndex index = (StubIndex)cfs.indexManager.listIndexes() + .stream() + .filter(i -> "test_index".equals(i.getIndexName())) + .findFirst() + .orElseThrow(() -> new RuntimeException(new AssertionError("Index not found"))); + index.reset(); UpdateBuilder.create(cfs.metadata, key).withTimestamp(0).newRow(1).add("val", 1).applyUnsafe(); @@ -588,8 +593,8 @@ public class RangeTombstoneTest // We should have 1 insert and 1 update to the indexed "1" column // CASSANDRA-6640 changed index update to just update, not insert then delete - assertEquals(1, index.inserts.size()); - assertEquals(1, index.updates.size()); + assertEquals(1, index.rowsInserted.size()); + assertEquals(1, index.rowsUpdated.size()); } private static ByteBuffer bb(int i) @@ -601,68 +606,4 @@ public class RangeTombstoneTest { return ByteBufferUtil.toInt(bb); } - - public static class TestIndex extends PerColumnSecondaryIndex - { - public List inserts = new ArrayList<>(); - public List deletes = new ArrayList<>(); - public List updates = new ArrayList<>(); - - public void resetCounts() - { - inserts.clear(); - deletes.clear(); - updates.clear(); - } - - public void delete(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - deletes.add(cell); - } - - @Override - public void deleteForCleanup(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup, int nowInSec) {} - - public void insert(ByteBuffer rowKey, Clustering clustering, Cell cell, OpOrder.Group opGroup) - { - inserts.add(cell); - } - - public void update(ByteBuffer rowKey, Clustering clustering, Cell oldCell, Cell cell, OpOrder.Group opGroup, int nowInSec) - { - updates.add(cell); - } - - public void init(){} - - public void reload(){} - - public void validateOptions(CFMetaData cfm, IndexMetadata def) throws ConfigurationException{} - - public String getIndexName(){ return "TestIndex";} - - protected SecondaryIndexSearcher createSecondaryIndexSearcher(Set columns) { return null; }; - - public void forceBlockingFlush(){} - - public ColumnFamilyStore getIndexCfs(){ return null; } - - public void removeIndex(ByteBuffer columnName){} - - public void invalidate(){} - - public void validate(DecoratedKey key) {} - public void validate(ByteBuffer value, CellPath path) {} - public void validate(Clustering clustering) {} - - public void truncateBlocking(long truncatedAt) { } - - public boolean indexes(ColumnDefinition name) { return true; } - - @Override - public long estimateResultRows() - { - return 0; - } - } } diff --git a/test/unit/org/apache/cassandra/db/ScrubTest.java b/test/unit/org/apache/cassandra/db/ScrubTest.java index 52b2aa8182..5887fd4001 100644 --- a/test/unit/org/apache/cassandra/db/ScrubTest.java +++ b/test/unit/org/apache/cassandra/db/ScrubTest.java @@ -18,10 +18,7 @@ */ package org.apache.cassandra.db; -import java.io.File; -import java.io.IOError; -import java.io.IOException; -import java.io.RandomAccessFile; +import java.io.*; import java.nio.ByteBuffer; import java.util.*; import java.util.concurrent.ExecutionException; @@ -31,10 +28,7 @@ import org.junit.BeforeClass; import org.junit.Test; import org.junit.runner.RunWith; -import org.apache.cassandra.OrderedJUnit4ClassRunner; -import org.apache.cassandra.SchemaLoader; -import org.apache.cassandra.UpdateBuilder; -import org.apache.cassandra.Util; +import org.apache.cassandra.*; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.cql3.Operator; import org.apache.cassandra.cql3.QueryProcessor; @@ -44,7 +38,7 @@ import org.apache.cassandra.db.compaction.OperationType; import org.apache.cassandra.db.compaction.Scrubber; import org.apache.cassandra.db.lifecycle.LifecycleTransaction; import org.apache.cassandra.db.lifecycle.TransactionLog; -import org.apache.cassandra.db.marshal.*; +import org.apache.cassandra.db.marshal.LongType; import org.apache.cassandra.db.marshal.UUIDType; import org.apache.cassandra.db.partitions.Partition; import org.apache.cassandra.db.partitions.PartitionUpdate; @@ -60,7 +54,10 @@ import org.apache.cassandra.io.util.FileUtils; import org.apache.cassandra.schema.KeyspaceParams; import org.apache.cassandra.utils.ByteBufferUtil; -import static org.junit.Assert.*; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertNotNull; +import static org.junit.Assert.assertTrue; +import static org.junit.Assert.fail; import static org.junit.Assume.assumeTrue; @RunWith(OrderedJUnit4ClassRunner.class) @@ -607,7 +604,7 @@ public class ScrubTest assertOrdered(Util.cmd(cfs).filterOn(colName, Operator.EQ, 1L).build(), numRows / 2); // scrub index - Set indexCfss = cfs.indexManager.getIndexesBackedByCfs(); + Set indexCfss = cfs.indexManager.getAllIndexColumnFamilyStores(); assertTrue(indexCfss.size() == 1); for(ColumnFamilyStore indexCfs : indexCfss) { diff --git a/test/unit/org/apache/cassandra/db/SecondaryIndexTest.java b/test/unit/org/apache/cassandra/db/SecondaryIndexTest.java index 115b13cbac..c017675773 100644 --- a/test/unit/org/apache/cassandra/db/SecondaryIndexTest.java +++ b/test/unit/org/apache/cassandra/db/SecondaryIndexTest.java @@ -33,18 +33,20 @@ import org.apache.cassandra.SchemaLoader; import org.apache.cassandra.Util; import org.apache.cassandra.config.ColumnDefinition; import org.apache.cassandra.cql3.Operator; -import org.apache.cassandra.db.index.SecondaryIndex; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; import org.apache.cassandra.db.marshal.AbstractType; import org.apache.cassandra.db.partitions.*; import org.apache.cassandra.db.rows.Row; import org.apache.cassandra.exceptions.ConfigurationException; +import org.apache.cassandra.index.Index; import org.apache.cassandra.schema.IndexMetadata; import org.apache.cassandra.schema.KeyspaceParams; import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.FBUtilities; import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertNotNull; +import static org.junit.Assert.assertNull; import static org.junit.Assert.assertTrue; public class SecondaryIndexTest @@ -111,9 +113,9 @@ public class SecondaryIndexTest .columns("birthdate") .filterOn("birthdate", Operator.EQ, 1L) .build(); - List searchers = cfs.indexManager.getIndexSearchersFor(rc); - assertEquals(searchers.size(), 1); - try (ReadOrderGroup orderGroup = rc.startOrderGroup(); UnfilteredPartitionIterator pi = searchers.get(0).search(rc, orderGroup)) + + Index.Searcher searcher = cfs.indexManager.getBestIndexFor(rc).searcherFor(rc); + try (ReadOrderGroup orderGroup = rc.startOrderGroup(); UnfilteredPartitionIterator pi = searcher.search(orderGroup)) { assertTrue(pi.hasNext()); pi.next().close(); @@ -198,7 +200,7 @@ public class SecondaryIndexTest // verify that it's not being indexed under any other value either ReadCommand rc = Util.cmd(cfs).build(); - assertEquals(0, cfs.indexManager.getIndexSearchersFor(rc).size()); + assertNull(cfs.indexManager.getBestIndexFor(rc)); // resurrect w/ a newer timestamp new RowUpdateBuilder(cfs.metadata, 2, "k1").clustering("c").add("birthdate", 1L).build().apply();; @@ -213,14 +215,16 @@ public class SecondaryIndexTest assertIndexedOne(cfs, col, 1L); // delete the entire row (w/ newer timestamp this time) + // todo - checking the # of index searchers for the command is probably not the best thing to test here RowUpdateBuilder.deleteRow(cfs.metadata, 3, "k1", "c").applyUnsafe(); rc = Util.cmd(cfs).build(); - assertEquals(0, cfs.indexManager.getIndexSearchersFor(rc).size()); + assertNull(cfs.indexManager.getBestIndexFor(rc)); // make sure obsolete mutations don't generate an index entry + // todo - checking the # of index searchers for the command is probably not the best thing to test here new RowUpdateBuilder(cfs.metadata, 3, "k1").clustering("c").add("birthdate", 1L).build().apply();; rc = Util.cmd(cfs).build(); - assertEquals(0, cfs.indexManager.getIndexSearchersFor(rc).size()); + assertNull(cfs.indexManager.getBestIndexFor(rc)); } @Test @@ -292,7 +296,9 @@ public class SecondaryIndexTest keyspace.getColumnFamilyStore(WITH_KEYS_INDEX).forceBlockingFlush(); // now apply another update, but force the index update to be skipped - keyspace.apply(new RowUpdateBuilder(cfs.metadata, 2, "k1").noRowMarker().add("birthdate", 2L).build(), true, false); + keyspace.apply(new RowUpdateBuilder(cfs.metadata, 2, "k1").noRowMarker().add("birthdate", 2L).build(), + true, + false); // Now searching the index for either the old or new value should return 0 rows // because the new value was not indexed and the old value should be ignored @@ -303,7 +309,9 @@ public class SecondaryIndexTest // now, reset back to the original value, still skipping the index update, to // make sure the value was expunged from the index when it was discovered to be inconsistent - keyspace.apply(new RowUpdateBuilder(cfs.metadata, 3, "k1").noRowMarker().add("birthdate", 1L).build(), true, false); + keyspace.apply(new RowUpdateBuilder(cfs.metadata, 3, "k1").noRowMarker().add("birthdate", 1L).build(), + true, + false); assertIndexedNone(cfs, col, 1L); } @@ -328,7 +336,9 @@ public class SecondaryIndexTest assertIndexedOne(cfs, col, 10l); // now apply another update, but force the index update to be skipped - keyspace.apply(new RowUpdateBuilder(cfs.metadata, 1, "k1").clustering("c").add("birthdate", 20l).build(), true, false); + keyspace.apply(new RowUpdateBuilder(cfs.metadata, 1, "k1").clustering("c").add("birthdate", 20l).build(), + true, + false); // Now searching the index for either the old or new value should return 0 rows // because the new value was not indexed and the old value should be ignored @@ -422,34 +432,54 @@ public class SecondaryIndexTest ColumnFamilyStore cfs = keyspace.getColumnFamilyStore(COMPOSITE_INDEX_TO_BE_ADDED); // create a row and update the birthdate value, test that the index query fetches the new version - DecoratedKey dk = Util.dk("k1"); new RowUpdateBuilder(cfs.metadata, 0, "k1").clustering("c").add("birthdate", 1L).build().applyUnsafe(); ColumnDefinition old = cfs.metadata.getColumnDefinition(ByteBufferUtil.bytes("birthdate")); - IndexMetadata indexDef = IndexMetadata.legacyIndex(old, - "birthdate_index", - IndexMetadata.IndexType.COMPOSITES, - Collections.EMPTY_MAP); + IndexMetadata indexDef = IndexMetadata.singleColumnIndex(old, + "birthdate_index", + IndexMetadata.IndexType.COMPOSITES, + Collections.EMPTY_MAP); cfs.metadata.indexes(cfs.metadata.getIndexes().with(indexDef)); - Future future = cfs.indexManager.addIndexedColumn(indexDef); + Future future = cfs.indexManager.addIndex(indexDef); future.get(); - // we had a bug (CASSANDRA-2244) where index would get created but not flushed -- check for that - ColumnDefinition cDef = cfs.metadata.getColumnDefinition(ByteBufferUtil.bytes("birthdate")); - assert cfs.indexManager.getIndexForColumn(cDef).getIndexCfs().getLiveSSTables().size() > 0; + // we had a bug (CASSANDRA-2244) where index would get created but not flushed -- check for that + // the way we find the index cfs is a bit convoluted at the moment + ColumnDefinition cDef = cfs.metadata.getColumnDefinition(ByteBufferUtil.bytes("birthdate")); + String indexName = getIndexNameForColumn(cfs, cDef); + assertNotNull(indexName); + boolean flushed = false; + for (ColumnFamilyStore indexCfs : cfs.indexManager.getAllIndexColumnFamilyStores()) + { + if (indexCfs.name.equals(indexName)) + flushed = indexCfs.getLiveSSTables().size() > 0; + } + assertTrue(flushed); assertIndexedOne(cfs, ByteBufferUtil.bytes("birthdate"), 1L); // validate that drop clears it out & rebuild works (CASSANDRA-2320) - SecondaryIndex indexedCfs = cfs.indexManager.getIndexForColumn(cDef); - cfs.indexManager.removeIndexedColumn(ByteBufferUtil.bytes("birthdate")); - assert !indexedCfs.isIndexBuilt(ByteBufferUtil.bytes("birthdate")); + assertTrue(cfs.getBuiltIndexes().contains(indexName)); + cfs.indexManager.removeIndex(indexDef.name); + assertFalse(cfs.getBuiltIndexes().contains(indexName)); // rebuild & re-query - future = cfs.indexManager.addIndexedColumn(indexDef); + future = cfs.indexManager.addIndex(indexDef); future.get(); assertIndexedOne(cfs, ByteBufferUtil.bytes("birthdate"), 1L); } + private String getIndexNameForColumn(ColumnFamilyStore cfs, ColumnDefinition column) + { + // this is mega-ugly because there is a mismatch between the name of an index + // stored in schema metadata & the name used to refer to that index in other + // places (such as system.IndexInfo). Hopefully this is temporary and + // Index.getIndexName() can be made equalivalent to Index.getIndexMetadata().name + // (ideally even removing the former completely) + Collection indexes = cfs.metadata.getIndexes().get(column); + assertEquals(1, indexes.size()); + return cfs.indexManager.getIndex(indexes.iterator().next()).getIndexName(); + } + @Test public void testKeysSearcherSimple() throws Exception { @@ -460,6 +490,7 @@ public class SecondaryIndexTest for (int i = 0; i < 10; i++) new RowUpdateBuilder(cfs.metadata, 0, "k" + i).noRowMarker().add("birthdate", 1l).build().applyUnsafe(); + assertIndexedCount(cfs, ByteBufferUtil.bytes("birthdate"), 1l, 10); cfs.forceBlockingFlush(); assertIndexedCount(cfs, ByteBufferUtil.bytes("birthdate"), 1l, 10); } @@ -477,13 +508,14 @@ public class SecondaryIndexTest ColumnDefinition cdef = cfs.metadata.getColumnDefinition(col); ReadCommand rc = Util.cmd(cfs).filterOn(cdef.name.toString(), Operator.EQ, ((AbstractType) cdef.cellValueType()).decompose(val)).build(); - List searchers = cfs.indexManager.getIndexSearchersFor(rc); + Index.Searcher searcher = cfs.indexManager.getBestIndexFor(rc).searcherFor(rc); if (count != 0) - assertTrue(searchers.size() > 0); + assertNotNull(searcher); try (ReadOrderGroup orderGroup = rc.startOrderGroup(); - PartitionIterator iter = UnfilteredPartitionIterators.filter(searchers.get(0).search(rc, orderGroup), - FBUtilities.nowInSeconds())) { + PartitionIterator iter = UnfilteredPartitionIterators.filter(searcher.search(orderGroup), + FBUtilities.nowInSeconds())) + { assertEquals(count, Util.size(iter)); } } diff --git a/test/unit/org/apache/cassandra/db/index/PerRowSecondaryIndexTest.java b/test/unit/org/apache/cassandra/db/index/PerRowSecondaryIndexTest.java deleted file mode 100644 index 8996dedf0a..0000000000 --- a/test/unit/org/apache/cassandra/db/index/PerRowSecondaryIndexTest.java +++ /dev/null @@ -1,304 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ -package org.apache.cassandra.db.index; - -import java.nio.ByteBuffer; -import java.util.Arrays; -import java.util.Set; - -import org.junit.Before; -import org.junit.BeforeClass; -import org.junit.Test; - -import junit.framework.Assert; -import org.apache.cassandra.SchemaLoader; -import org.apache.cassandra.config.CFMetaData; -import org.apache.cassandra.config.ColumnDefinition; -import org.apache.cassandra.config.Schema; -import org.apache.cassandra.cql3.ColumnIdentifier; -import org.apache.cassandra.cql3.QueryProcessor; -import org.apache.cassandra.cql3.UntypedResultSet; -import org.apache.cassandra.db.*; -import org.apache.cassandra.db.rows.*; -import org.apache.cassandra.db.filter.RowFilter; -import org.apache.cassandra.db.partitions.SingletonUnfilteredPartitionIterator; -import org.apache.cassandra.db.partitions.UnfilteredPartitionIterator; -import org.apache.cassandra.db.partitions.PartitionUpdate; -import org.apache.cassandra.exceptions.ConfigurationException; -import org.apache.cassandra.exceptions.InvalidRequestException; -import org.apache.cassandra.schema.IndexMetadata; -import org.apache.cassandra.schema.KeyspaceParams; -import org.apache.cassandra.utils.ByteBufferUtil; -import org.apache.cassandra.utils.FBUtilities; -import org.apache.cassandra.utils.concurrent.OpOrder; - -import static org.junit.Assert.*; - -public class PerRowSecondaryIndexTest -{ - - // test that when index(key) is called on a PRSI index, - // the data to be indexed can be read using the supplied - // key. TestIndex.index(key) simply reads the data to be - // indexed & stashes it in a static variable for inspection - // in the test. - - private static final String KEYSPACE1 = "PerRowSecondaryIndexTest"; - private static final String CF_INDEXED = "Indexed1"; - - @BeforeClass - public static void defineSchema() throws ConfigurationException - { - SchemaLoader.prepareServer(); - SchemaLoader.createKeyspace(KEYSPACE1, - KeyspaceParams.simple(1), - SchemaLoader.perRowIndexedCFMD(KEYSPACE1, CF_INDEXED)); - } - - @Before - public void clearTestStub() - { - PerRowSecondaryIndexTest.TestIndex.reset(); - } - - @Test - public void testIndexInsertAndUpdate() - { - int nowInSec = FBUtilities.nowInSeconds(); - - // create a row then test that the configured index instance was able to read the row - CFMetaData cfm = Schema.instance.getCFMetaData(KEYSPACE1, CF_INDEXED); - ColumnDefinition cdef = cfm.getColumnDefinition(new ColumnIdentifier("indexed", true)); - - RowUpdateBuilder builder = new RowUpdateBuilder(cfm, FBUtilities.timestampMicros(), "k1"); - builder.add("indexed", ByteBufferUtil.bytes("foo")); - builder.build().apply(); - - - UnfilteredRowIterator indexedRow = PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_PARTITION; - assertNotNull(indexedRow); - assertEquals(ByteBufferUtil.bytes("foo"), UnfilteredRowIterators.filter(indexedRow, nowInSec).next().getCell(cdef).value()); - - // update the row and verify what was indexed - builder = new RowUpdateBuilder(cfm, FBUtilities.timestampMicros() + 1 , "k1"); - builder.add("indexed", ByteBufferUtil.bytes("bar")); - builder.build().apply(); - - indexedRow = PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_PARTITION; - assertNotNull(indexedRow); - assertEquals(ByteBufferUtil.bytes("bar"), UnfilteredRowIterators.filter(indexedRow, nowInSec).next().getCell(cdef).value()); - assertTrue(Arrays.equals("k1".getBytes(), PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_KEY.array())); - } - - @Test - public void testColumnDelete() - { - // issue a column delete and test that the configured index instance was notified to update - CFMetaData cfm = Schema.instance.getCFMetaData(KEYSPACE1, CF_INDEXED); - - new RowUpdateBuilder(cfm, FBUtilities.timestampMicros(), "k2") - .noRowMarker() - .delete("indexed") - .build() - .apply(); - - UnfilteredRowIterator indexedRow = PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_PARTITION; - assertNotNull(indexedRow); - - //We filter tombstones now... - Assert.assertFalse(UnfilteredRowIterators.filter(indexedRow, FBUtilities.nowInSeconds()).hasNext()); - assertTrue(Arrays.equals("k2".getBytes(), PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_KEY.array())); - } - - @Test - public void testRowDelete() - { - // issue a row level delete and test that the configured index instance was notified to update - CFMetaData cfm = Schema.instance.getCFMetaData(KEYSPACE1, CF_INDEXED); - RowUpdateBuilder.deleteRow(cfm, FBUtilities.timestampMicros(), "k3").apply(); - - UnfilteredRowIterator indexedRow = PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_PARTITION; - assertNotNull(indexedRow); - assertNotNull(indexedRow.partitionLevelDeletion()); - Assert.assertFalse(UnfilteredRowIterators.filter(indexedRow, FBUtilities.nowInSeconds()).hasNext()); - assertTrue(Arrays.equals("k3".getBytes(), PerRowSecondaryIndexTest.TestIndex.LAST_INDEXED_KEY.array())); - } - - @Test - public void testInvalidSearch() - { - - CFMetaData cfm = Schema.instance.getCFMetaData(KEYSPACE1, CF_INDEXED); - - RowUpdateBuilder builder = new RowUpdateBuilder(cfm, FBUtilities.timestampMicros(), "k1"); - builder.add("indexed", ByteBufferUtil.bytes("foo")); - builder.build().apply(); - - - // test we can search: - UntypedResultSet result = QueryProcessor.executeInternal(String.format("SELECT * FROM \"%s\".\"Indexed1\" WHERE indexed = 'foo'", KEYSPACE1)); - assertEquals(1, result.size()); - - // test we can't search if the searcher doesn't validate the expression: - try - { - QueryProcessor.executeInternal(String.format("SELECT * FROM \"%s\".\"Indexed1\" WHERE indexed = 'invalid'", KEYSPACE1)); - fail("Query should have been invalid!"); - } - catch (Exception e) - { - assertTrue(e instanceof InvalidRequestException || (e.getCause() != null && (e.getCause() instanceof InvalidRequestException))); - } - } - - public static class TestIndex extends PerRowSecondaryIndex - { - public static UnfilteredRowIterator LAST_INDEXED_PARTITION; - public static ByteBuffer LAST_INDEXED_KEY; - - public static void reset() - { - LAST_INDEXED_KEY = null; - LAST_INDEXED_PARTITION = null; - } - - @Override - public void index(ByteBuffer rowKey, UnfilteredRowIterator cf) - { - LAST_INDEXED_PARTITION = cf; - LAST_INDEXED_KEY = rowKey; - } - - public void index(ByteBuffer rowKey, PartitionUpdate atoms) - { - LAST_INDEXED_PARTITION = atoms.unfilteredIterator(); - LAST_INDEXED_KEY = rowKey; - } - - @Override - public void delete(ByteBuffer key, OpOrder.Group opGroup) - { - } - - @Override - public void init() - { - } - - @Override - public void reload() - { - } - - @Override - public void validateOptions(CFMetaData cfm, IndexMetadata def) throws ConfigurationException{} - { - } - - @Override - public String getIndexName() - { - return null; - } - - @Override - protected SecondaryIndexSearcher createSecondaryIndexSearcher(Set columns) - { - return new SecondaryIndexSearcher(baseCfs.indexManager, columns) - { - @Override - public UnfilteredPartitionIterator search(ReadCommand filter, ReadOrderGroup orderGroup) - { - return new SingletonUnfilteredPartitionIterator(LAST_INDEXED_PARTITION, false); - } - - @Override - public RowFilter.Expression primaryClause(ReadCommand command) - { - RowFilter.Expression expression = command.rowFilter().iterator().next(); - - if (expression.getIndexValue().equals(ByteBufferUtil.bytes("invalid"))) - throw new InvalidRequestException("Invalid search!"); - - return expression; - } - - protected UnfilteredPartitionIterator queryDataFromIndex(AbstractSimplePerColumnSecondaryIndex index, DecoratedKey indexKey, RowIterator indexHits, ReadCommand command, ReadOrderGroup orderGroup) - { - // As we override 'search()' directly for the test, we don't care about this. - throw new UnsupportedOperationException(); - } - }; - } - - @Override - public void forceBlockingFlush() - { - } - - @Override - public ColumnFamilyStore getIndexCfs() - { - return baseCfs; - } - - @Override - public boolean indexes(ColumnDefinition name) - { - return true; - } - - @Override - public void validate(DecoratedKey partitionKey) throws InvalidRequestException - { - - } - - @Override - public void validate(Clustering clustering) throws InvalidRequestException - { - - } - - @Override - public void validate(ByteBuffer cellValue, CellPath path) throws InvalidRequestException - { - - } - - @Override - public void removeIndex(ByteBuffer columnName) - { - } - - @Override - public void invalidate() - { - } - - @Override - public void truncateBlocking(long truncatedAt) - { - } - - @Override - public long estimateResultRows() { - return 0; - } - } -} diff --git a/test/unit/org/apache/cassandra/index/StubIndex.java b/test/unit/org/apache/cassandra/index/StubIndex.java new file mode 100644 index 0000000000..b170c24c6c --- /dev/null +++ b/test/unit/org/apache/cassandra/index/StubIndex.java @@ -0,0 +1,186 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index; + +import java.util.*; +import java.util.concurrent.Callable; +import java.util.function.BiFunction; + +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.cql3.ColumnIdentifier; +import org.apache.cassandra.cql3.Operator; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.filter.RowFilter; +import org.apache.cassandra.db.partitions.PartitionIterator; +import org.apache.cassandra.db.partitions.PartitionUpdate; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.index.transactions.IndexTransaction; +import org.apache.cassandra.schema.IndexMetadata; +import org.apache.cassandra.utils.Pair; +import org.apache.cassandra.utils.concurrent.OpOrder; + +public class StubIndex implements Index +{ + public List rowsInserted = new ArrayList<>(); + public List rowsDeleted = new ArrayList<>(); + public List> rowsUpdated = new ArrayList<>(); + private IndexMetadata indexMetadata; + private ColumnFamilyStore baseCfs; + + public void reset() + { + rowsInserted.clear(); + rowsDeleted.clear(); + } + + public StubIndex(ColumnFamilyStore baseCfs, IndexMetadata metadata) + { + this.baseCfs = baseCfs; + this.indexMetadata = metadata; + } + + public boolean indexes(PartitionColumns columns) + { + for (ColumnDefinition col : columns) + for (ColumnIdentifier indexed : indexMetadata.columns) + if (indexed.equals(col.name)) + return true; + return false; + } + + public boolean supportsExpression(ColumnDefinition column, Operator operator) + { + return operator == Operator.EQ; + } + + public RowFilter getPostIndexQueryFilter(RowFilter filter) + { + return filter; + } + + public Indexer indexerFor(DecoratedKey key, + int nowInSec, + OpOrder.Group opGroup, + IndexTransaction.Type transactionType) + { + return new Indexer() + { + public void begin() + { + } + + public void partitionDelete(DeletionTime deletionTime) + { + } + + public void rangeTombstone(RangeTombstone tombstone) + { + } + + public void insertRow(Row row) + { + rowsInserted.add(row); + } + + public void removeRow(Row row) + { + rowsDeleted.add(row); + } + + public void updateRow(Row oldRowData, Row newRowData) + { + rowsUpdated.add(Pair.create(oldRowData, newRowData)); + } + + public void finish() + { + } + }; + } + + public Callable getInitializationTask() + { + return null; + } + + public IndexMetadata getIndexMetadata() + { + return indexMetadata; + } + + public String getIndexName() + { + return indexMetadata != null ? indexMetadata.name : "default_test_index_name"; + } + + public void register(IndexRegistry registry){ + registry.registerIndex(this); + } + + public Optional getBackingTable() + { + return Optional.empty(); + } + + public Collection getIndexedColumns() + { + return Collections.emptySet(); + } + + public Callable getBlockingFlushTask() + { + return null; + } + + public Callable getTruncateTask(long truncatedAt) + { + return null; + } + + public Callable getInvalidateTask() + { + return null; + } + + public Callable getMetadataReloadTask(IndexMetadata indexMetadata) + { + return null; + } + + public long getEstimatedResultRows() + { + return 0; + } + + public void validate(PartitionUpdate update) throws InvalidRequestException + { + + } + + public Searcher searcherFor(ReadCommand command) + { + return null; + } + + public BiFunction postProcessorFor(ReadCommand readCommand) + { + return null; + } +} diff --git a/test/unit/org/apache/cassandra/index/internal/CassandraIndexTest.java b/test/unit/org/apache/cassandra/index/internal/CassandraIndexTest.java new file mode 100644 index 0000000000..e938737f6f --- /dev/null +++ b/test/unit/org/apache/cassandra/index/internal/CassandraIndexTest.java @@ -0,0 +1,718 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.cassandra.index.internal; + +import java.util.stream.Collectors; +import java.util.stream.Stream; +import java.util.stream.StreamSupport; + +import com.google.common.base.Joiner; +import com.google.common.collect.*; +import org.junit.Test; + +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.cql3.CQLTester; +import org.apache.cassandra.cql3.UntypedResultSet; +import org.apache.cassandra.cql3.restrictions.StatementRestrictions; +import org.apache.cassandra.cql3.statements.SelectStatement; +import org.apache.cassandra.db.*; +import org.apache.cassandra.db.filter.ClusteringIndexFilter; +import org.apache.cassandra.db.filter.ClusteringIndexSliceFilter; +import org.apache.cassandra.db.filter.ColumnFilter; +import org.apache.cassandra.db.rows.Row; +import org.apache.cassandra.db.rows.Unfiltered; +import org.apache.cassandra.db.rows.UnfilteredRowIterator; +import org.apache.cassandra.exceptions.InvalidRequestException; +import org.apache.cassandra.utils.ByteBufferUtil; +import org.apache.cassandra.utils.FBUtilities; + +import static org.apache.cassandra.Util.throwAssert; +import static org.junit.Assert.assertArrayEquals; +import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertFalse; +import static org.junit.Assert.assertNotNull; +import static org.junit.Assert.fail; + +/** + * Smoke tests of built-in secondary index implementations + */ +public class CassandraIndexTest extends CQLTester +{ + + @Test + public void indexOnRegularColumn() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, v int, PRIMARY KEY (k, c));") + .target("v") + .indexName("v_index") + .withFirstRow(row(0, 0, 0)) + .withSecondRow(row(1, 1, 1)) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("v=0") + .secondQueryExpression("v=1") + .updateExpression("SET v=2") + .postUpdateQueryExpression("v=2") + .run(); + } + + @Test + public void indexOnFirstClusteringColumn() throws Throwable + { + // No update allowed on primary key columns, so this script has no update expression + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, v int, PRIMARY KEY (k, c));") + .target("c") + .indexName("c_index") + .withFirstRow(row(0, 0, 0)) + .withSecondRow(row(1, 1, 1)) + .missingIndexMessage(SelectStatement.REQUIRES_ALLOW_FILTERING_MESSAGE) + .firstQueryExpression("c=0") + .secondQueryExpression("c=1") + .run(); + } + + @Test + public void indexOnSecondClusteringColumn() throws Throwable + { + // No update allowed on primary key columns, so this script has no update expression + new TestScript().tableDefinition("CREATE TABLE %s (k int, c1 int, c2 int, v int, PRIMARY KEY (k, c1, c2));") + .target("c2") + .indexName("c2_index") + .withFirstRow(row(0, 0, 0, 0)) + .withSecondRow(row(1, 1, 1, 1)) + .missingIndexMessage(String.format("PRIMARY KEY column \"%s\" cannot be restricted " + + "as preceding column \"%s\" is not restricted", + "c2", "c1")) + .firstQueryExpression("c2=0") + .secondQueryExpression("c2=1") + .run(); + } + + @Test + public void indexOnFirstPartitionKeyColumn() throws Throwable + { + // No update allowed on primary key columns, so this script has no update expression + new TestScript().tableDefinition("CREATE TABLE %s (k1 int, k2 int, c1 int, c2 int, v int, PRIMARY KEY ((k1, k2), c1, c2));") + .target("k1") + .indexName("k1_index") + .withFirstRow(row(0, 0, 0, 0, 0)) + .withSecondRow(row(1, 1, 1, 1, 1)) + .missingIndexMessage("Partition key parts: k2 must be restricted as other parts are") + .firstQueryExpression("k1=0") + .secondQueryExpression("k1=1") + .run(); + } + + @Test + public void indexOnSecondPartitionKeyColumn() throws Throwable + { + // No update allowed on primary key columns, so this script has no update expression + new TestScript().tableDefinition("CREATE TABLE %s (k1 int, k2 int, c1 int, c2 int, v int, PRIMARY KEY ((k1, k2), c1, c2));") + .target("k2") + .indexName("k2_index") + .withFirstRow(row(0, 0, 0, 0, 0)) + .withSecondRow(row(1, 1, 1, 1, 1)) + .missingIndexMessage("Partition key parts: k1 must be restricted as other parts are") + .firstQueryExpression("k2=0") + .secondQueryExpression("k2=1") + .run(); + } + + @Test + public void indexOnNonFrozenListWithReplaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, l list, PRIMARY KEY (k, c));") + .target("l") + .indexName("l_index1") + .withFirstRow(row(0, 0, Lists.newArrayList(10, 20, 30))) + .withSecondRow(row(1, 1, Lists.newArrayList(11, 21, 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("l CONTAINS 10") + .secondQueryExpression("l CONTAINS 11") + .updateExpression("SET l = [40, 50, 60]") + .postUpdateQueryExpression("l CONTAINS 40") + .run(); + } + + @Test + public void indexOnNonFrozenListWithInPlaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, l list, PRIMARY KEY (k, c));") + .target("l") + .indexName("l_index2") + .withFirstRow(row(0, 0, Lists.newArrayList(10, 20, 30))) + .withSecondRow(row(1, 1, Lists.newArrayList(11, 21, 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("l CONTAINS 10") + .secondQueryExpression("l CONTAINS 11") + .updateExpression("SET l = l - [10]") + .postUpdateQueryExpression("l CONTAINS 20") + .run(); + } + + @Test + public void indexOnNonFrozenSetWithReplaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, s set, PRIMARY KEY (k, c));") + .target("s") + .indexName("s_index1") + .withFirstRow(row(0, 0, Sets.newHashSet(10, 20, 30))) + .withSecondRow(row(1, 1, Sets.newHashSet(11, 21, 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("s CONTAINS 10") + .secondQueryExpression("s CONTAINS 11") + .updateExpression("SET s = {40, 50, 60}") + .postUpdateQueryExpression("s CONTAINS 40") + .run(); + } + + @Test + public void indexOnNonFrozenSetWithInPlaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, s set, PRIMARY KEY (k, c));") + .target("s") + .indexName("s_index2") + .withFirstRow(row(0, 0, Sets.newHashSet(10, 20, 30))) + .withSecondRow(row(1, 1, Sets.newHashSet(11, 21, 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("s CONTAINS 10") + .secondQueryExpression("s CONTAINS 11") + .updateExpression("SET s = s - {10}") + .postUpdateQueryExpression("s CONTAINS 20") + .run(); + } + + @Test + public void indexOnNonFrozenMapValuesWithReplaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY (k, c));") + .target("m") + .indexName("m_index1") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m CONTAINS 10") + .secondQueryExpression("m CONTAINS 11") + .updateExpression("SET m = {'x':40, 'y':50, 'z':60}") + .postUpdateQueryExpression("m CONTAINS 40") + .run(); + } + + @Test + public void indexOnNonFrozenMapValuesWithInPlaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY (k, c));") + .target("m") + .indexName("m_index2") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m CONTAINS 10") + .secondQueryExpression("m CONTAINS 11") + .updateExpression("SET m['a'] = 40") + .postUpdateQueryExpression("m CONTAINS 40") + .run(); + } + + @Test + public void indexOnNonFrozenMapKeysWithReplaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY (k, c));") + .target("keys(m)") + .indexName("m_index3") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m CONTAINS KEY 'a'") + .secondQueryExpression("m CONTAINS KEY 'd'") + .updateExpression("SET m = {'x':40, 'y':50, 'z':60}") + .postUpdateQueryExpression("m CONTAINS KEY 'x'") + .run(); + } + + @Test + public void indexOnNonFrozenMapKeysWithInPlaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY (k, c));") + .target("keys(m)") + .indexName("m_index4") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m CONTAINS KEY 'a'") + .secondQueryExpression("m CONTAINS KEY 'd'") + .updateExpression("SET m['a'] = NULL") + .postUpdateQueryExpression("m CONTAINS KEY 'b'") + .run(); + } + + @Test + public void indexOnNonFrozenMapEntriesWithReplaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY (k, c));") + .target("entries(m)") + .indexName("m_index5") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m['a'] = 10") + .secondQueryExpression("m['d'] = 11") + .updateExpression("SET m = {'x':40, 'y':50, 'z':60}") + .postUpdateQueryExpression("m['x'] = 40") + .run(); + } + + @Test + public void indexOnNonFrozenMapEntriesWithInPlaceOperation() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY (k, c));") + .target("entries(m)") + .indexName("m_index6") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m['a'] = 10") + .secondQueryExpression("m['d'] = 11") + .updateExpression("SET m['a'] = 40") + .postUpdateQueryExpression("m['a'] = 40") + .run(); + } + + @Test + public void indexOnFrozenList() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, l frozen>, PRIMARY KEY (k, c));") + .target("full(l)") + .indexName("fl_index") + .withFirstRow(row(0, 0, Lists.newArrayList(10, 20, 30))) + .withSecondRow(row(1, 1, Lists.newArrayList(11, 21, 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("l = [10, 20, 30]") + .secondQueryExpression("l = [11, 21, 31]") + .updateExpression("SET l = [40, 50, 60]") + .postUpdateQueryExpression("l = [40, 50, 60]") + .run(); + } + + @Test + public void indexOnFrozenSet() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, s frozen>, PRIMARY KEY (k, c));") + .target("full(s)") + .indexName("fs_index") + .withFirstRow(row(0, 0, Sets.newHashSet(10, 20, 30))) + .withSecondRow(row(1, 1, Sets.newHashSet(11, 21, 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("s = {10, 20, 30}") + .secondQueryExpression("s = {11, 21, 31}") + .updateExpression("SET s = {40, 50, 60}") + .postUpdateQueryExpression("s = {40, 50, 60}") + .run(); + } + + @Test + public void indexOnFrozenMap() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, c int, m frozen>, PRIMARY KEY (k, c));") + .target("full(m)") + .indexName("fm_index") + .withFirstRow(row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30))) + .withSecondRow(row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 31))) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("m = {'a':10, 'b':20, 'c':30}") + .secondQueryExpression("m = {'d':11, 'e':21, 'f':31}") + .updateExpression("SET m = {'x':40, 'y':50, 'z':60}") + .postUpdateQueryExpression("m = {'x':40, 'y':50, 'z':60}") + .run(); + } + + @Test + public void indexOnRegularColumnWithCompactStorage() throws Throwable + { + new TestScript().tableDefinition("CREATE TABLE %s (k int, v int, PRIMARY KEY (k)) WITH COMPACT STORAGE;") + .target("v") + .indexName("cv_index") + .withFirstRow(row(0, 0)) + .withSecondRow(row(1,1)) + .missingIndexMessage(StatementRestrictions.NO_INDEX_FOUND_MESSAGE) + .firstQueryExpression("v=0") + .secondQueryExpression("v=1") + .updateExpression("SET v=2") + .postUpdateQueryExpression("v=2") + .run(); + } + + @Test + public void createIndexesOnMultipleMapDimensions() throws Throwable + { + Object[] row1 = row(0, 0, ImmutableMap.of("a", 10, "b", 20, "c", 30)); + Object[] row2 = row(1, 1, ImmutableMap.of("d", 11, "e", 21, "f", 32)); + createTable("CREATE TABLE %s (k int, c int, m map, PRIMARY KEY(k, c))"); + createIndex("CREATE INDEX mkey_index on %s(keys(m))"); + createIndex("CREATE INDEX mval_index on %s(m)"); + + execute("INSERT INTO %s (k, c, m) VALUES (?, ?, ?)", row1); + execute("INSERT INTO %s (k, c, m) VALUES (?, ?, ?)", row2); + + assertRows(execute("SELECT * FROM %s WHERE m CONTAINS KEY 'a'"), row1); + assertRows(execute("SELECT * FROM %s WHERE m CONTAINS 20"), row1); + assertRows(execute("SELECT * FROM %s WHERE m CONTAINS KEY 'f'"), row2); + assertRows(execute("SELECT * FROM %s WHERE m CONTAINS 32"), row2); + } + + @Test + public void insertWithTombstoneRemovesEntryFromIndex() throws Throwable + { + int key = 0; + int indexedValue = 99; + createTable("CREATE TABLE %s (k int, v int, PRIMARY KEY(k))"); + createIndex("CREATE INDEX v_index on %s(v)"); + execute("INSERT INTO %s (k, v) VALUES (?, ?)", key, indexedValue); + + assertRows(execute("SELECT * FROM %s WHERE v = ?", indexedValue), row(key, indexedValue)); + execute("DELETE v FROM %s WHERE k=?", key); + assertEmpty(execute("SELECT * FROM %s WHERE v = ?", indexedValue)); + } + + @Test + public void updateTTLOnIndexedClusteringValue() throws Throwable + { + int basePk = 1; + int indexedVal = 2; + int initialTtl = 3600; + createTable("CREATE TABLE %s (k int, c int, v int, PRIMARY KEY(k,c))"); + createIndex("CREATE INDEX c_index on %s(c)"); + execute("INSERT INTO %s (k, c, v) VALUES (?, ?, 0) USING TTL ?", basePk, indexedVal, initialTtl); + ColumnFamilyStore baseCfs = getCurrentColumnFamilyStore(); + ColumnFamilyStore indexCfs = baseCfs.indexManager.listIndexes() + .iterator() + .next() + .getBackingTable() + .orElseThrow(throwAssert("No index found")); + assertIndexRowTtl(indexCfs, indexedVal, initialTtl); + + int updatedTtl = 9999; + execute("INSERT INTO %s (k, c ,v) VALUES (?, ?, 0) USING TTL ?", basePk, indexedVal, updatedTtl); + + assertIndexRowTtl(indexCfs, indexedVal, updatedTtl); + } + + // this is slightly annoying, but we cannot read rows from the methods in Util as + // ReadCommand#executeInternal uses metadata retrieved via the cfId, which the index + // CFS inherits from the base CFS. This has the 'wrong' partitioner (the index table + // uses LocalPartition, the base table a real one, so we cannot read from the index + // table with executeInternal + private void assertIndexRowTtl(ColumnFamilyStore indexCfs, int indexedValue, int ttl) throws Throwable + { + DecoratedKey indexKey = indexCfs.decorateKey(ByteBufferUtil.bytes(indexedValue)); + ClusteringIndexFilter filter = new ClusteringIndexSliceFilter(Slices.with(indexCfs.metadata.comparator, + Slice.ALL), + false); + SinglePartitionReadCommand command = SinglePartitionReadCommand.create(indexCfs.metadata, + FBUtilities.nowInSeconds(), + indexKey, + ColumnFilter.all(indexCfs.metadata), + filter); + try (ReadOrderGroup orderGroup = ReadOrderGroup.forCommand(command); + UnfilteredRowIterator iter = command.queryMemtableAndDisk(indexCfs, orderGroup.indexReadOpOrderGroup())) + { + while( iter.hasNext()) + { + Unfiltered unfiltered = iter.next(); + assert (unfiltered.isRow()); + Row indexRow = (Row) unfiltered; + assertEquals(ttl, indexRow.primaryKeyLivenessInfo().ttl()); + } + } + } + + private class TestScript + { + String tableDefinition; + String indexName; + String indexTarget; + String queryExpression1; + String queryExpression2; + String updateExpression; + String postUpdateQueryExpression; + String missingIndexMessage; + + Object[] firstRow; + Object[] secondRow; + + TestScript indexName(String indexName) + { + this.indexName = indexName; + return this; + } + + TestScript target(String indexTarget) + { + this.indexTarget = indexTarget; + return this; + } + + TestScript tableDefinition(String tableDefinition) + { + this.tableDefinition = tableDefinition; + return this; + } + + TestScript withFirstRow(Object[] row) + { + this.firstRow = row; + return this; + } + + TestScript withSecondRow(Object[] row) + { + this.secondRow = row; + return this; + } + + TestScript firstQueryExpression(String queryExpression) + { + queryExpression1 = queryExpression; + return this; + } + + TestScript secondQueryExpression(String queryExpression) + { + queryExpression2 = queryExpression; + return this; + } + + TestScript updateExpression(String updateExpression) + { + this.updateExpression = updateExpression; + return this; + } + + TestScript postUpdateQueryExpression(String queryExpression) + { + this.postUpdateQueryExpression = queryExpression; + return this; + } + + TestScript missingIndexMessage(String missingIndexMessage) + { + this.missingIndexMessage = missingIndexMessage; + return this; + } + + void run() throws Throwable + { + // check minimum required setup + assertNotNull(indexName); + assertNotNull(indexTarget); + assertNotNull(queryExpression1); + assertNotNull(queryExpression2); + assertNotNull(firstRow); + assertNotNull(secondRow); + assertNotNull(tableDefinition); + if (updateExpression != null) + assertNotNull(postUpdateQueryExpression); + + // first, create the table as we need the CFMetaData to build the other cql statements + createTable(tableDefinition); + + // now setup the cql statements the test will run through. Some are dependent on + // the table definition, others are not. + String createIndexCql = String.format("CREATE INDEX %s ON %%s(%s)", indexName, indexTarget); + String dropIndexCql = String.format("DROP INDEX %s.%s", KEYSPACE, indexName); + + String selectFirstRowCql = String.format("SELECT * FROM %%s WHERE %s", queryExpression1); + String selectSecondRowCql = String.format("SELECT * FROM %%s WHERE %s", queryExpression2); + String insertCql = getInsertCql(); + String deleteRowCql = getDeleteRowCql(); + String deletePartitionCql = getDeletePartitionCql(); + + // everything setup, run through the smoke test + execute(insertCql, firstRow); + // before creating the index, check we cannot query on the indexed column + assertInvalidThrowMessage(missingIndexMessage, InvalidRequestException.class, selectFirstRowCql); + + // create the index, wait for it to be built then validate the indexed value + createIndex(createIndexCql); + waitForIndexBuild(); + assertRows(execute(selectFirstRowCql), firstRow); + assertEmpty(execute(selectSecondRowCql)); + + // flush and check again + flush(); + assertRows(execute(selectFirstRowCql), firstRow); + assertEmpty(execute(selectSecondRowCql)); + + // force major compaction and query again + compact(); + assertRows(execute(selectFirstRowCql), firstRow); + assertEmpty(execute(selectSecondRowCql)); + + // drop the index and assert we can no longer query using it + execute(dropIndexCql); + assertInvalidThrowMessage(missingIndexMessage, InvalidRequestException.class, selectFirstRowCql); + flush(); + compact(); + + // insert second row, re-create the index and query for both indexed values + execute(insertCql, secondRow); + createIndex(createIndexCql); + waitForIndexBuild(); + assertRows(execute(selectFirstRowCql), firstRow); + assertRows(execute(selectSecondRowCql), secondRow); + + // modify the indexed value in the first row, assert we can query by the new value & not the original one + // note: this is not possible if the indexed column is part of the primary key, so we skip it in that case + if (includesUpdate()) + { + execute(getUpdateCql(), getPrimaryKeyValues(firstRow)); + assertEmpty(execute(selectFirstRowCql)); + // update the select statement to query using the updated value + selectFirstRowCql = String.format("SELECT * FROM %%s WHERE %s", postUpdateQueryExpression); + // we can't check the entire row b/c we've modified something. + // so we just check the primary key columns, as they cannot have changed + assertPrimaryKeyColumnsOnly(execute(selectFirstRowCql), firstRow); + } + + // delete row, check that it cannot be found via index query + execute(deleteRowCql, getPrimaryKeyValues(firstRow)); + assertEmpty(execute(selectFirstRowCql)); + + // delete partition, check that its rows cannot be retrieved via index query + execute(deletePartitionCql, getPartitionKeyValues(secondRow)); + assertEmpty(execute(selectSecondRowCql)); + + // flush & compact, then verify that deleted values stay gone + flush(); + compact(); + assertEmpty(execute(selectFirstRowCql)); + assertEmpty(execute(selectSecondRowCql)); + + // add back both rows, reset the select for the first row to query on the original value & verify + execute(insertCql, firstRow); + selectFirstRowCql = String.format("SELECT * FROM %%s WHERE %s", queryExpression1); + assertRows(execute(selectFirstRowCql), firstRow); + execute(insertCql, secondRow); + assertRows(execute(selectSecondRowCql), secondRow); + + // flush and compact, verify again & we're done + flush(); + compact(); + assertRows(execute(selectFirstRowCql), firstRow); + assertRows(execute(selectSecondRowCql), secondRow); + } + + private void assertPrimaryKeyColumnsOnly(UntypedResultSet resultSet, Object[] row) + { + assertFalse(resultSet.isEmpty()); + CFMetaData cfm = getCurrentColumnFamilyStore().metadata; + int columnCount = cfm.partitionKeyColumns().size(); + if (cfm.isCompound()) + columnCount += cfm.clusteringColumns().size(); + Object[] expected = copyValuesFromRow(row, columnCount); + assertArrayEquals(expected, copyValuesFromRow(getRows(resultSet)[0], columnCount)); + } + + private String getInsertCql() + { + CFMetaData cfm = getCurrentColumnFamilyStore().metadata; + String columns = Joiner.on(", ") + .join(Iterators.transform(cfm.allColumnsInSelectOrder(), + (column) -> column.name.toString())); + String markers = Joiner.on(", ").join(Iterators.transform(cfm.allColumnsInSelectOrder(), + (column) -> { + return "?"; + })); + return String.format("INSERT INTO %%s (%s) VALUES (%s)", columns, markers); + } + + private String getUpdateCql() + { + String whereClause = getPrimaryKeyColumns().map(column -> column.name.toString() + "=?") + .collect(Collectors.joining(" AND ")); + return String.format("UPDATE %%s %s WHERE %s", updateExpression, whereClause); + } + + private String getDeleteRowCql() + { + return getPrimaryKeyColumns().map(column -> column.name.toString() + "=?") + .collect(Collectors.joining(" AND ", "DELETE FROM %s WHERE ", "")); + } + + private String getDeletePartitionCql() + { + CFMetaData cfm = getCurrentColumnFamilyStore().metadata; + return StreamSupport.stream(cfm.partitionKeyColumns().spliterator(), false) + .map(column -> column.name.toString() + "=?") + .collect(Collectors.joining(" AND ", "DELETE FROM %s WHERE ", "")); + } + + private Stream getPrimaryKeyColumns() + { + CFMetaData cfm = getCurrentColumnFamilyStore().metadata; + if (cfm.isCompactTable()) + return cfm.partitionKeyColumns().stream(); + else + return Stream.concat(cfm.partitionKeyColumns().stream(), cfm.clusteringColumns().stream()); + } + + private Object[] getPrimaryKeyValues(Object[] row) + { + CFMetaData cfm = getCurrentColumnFamilyStore().metadata; + if (cfm.isCompactTable()) + return getPartitionKeyValues(row); + + return copyValuesFromRow(row, cfm.partitionKeyColumns().size() + cfm.clusteringColumns().size()); + } + + private Object[] getPartitionKeyValues(Object[] row) + { + CFMetaData cfm = getCurrentColumnFamilyStore().metadata; + return copyValuesFromRow(row, cfm.partitionKeyColumns().size()); + } + + private Object[] copyValuesFromRow(Object[] row, int length) + { + Object[] values = new Object[length]; + System.arraycopy(row, 0, values, 0, length); + return values; + } + + private boolean includesUpdate() + { + return updateExpression != null; + } + + // Spin waiting for named index to be built + private void waitForIndexBuild() throws Throwable + { + ColumnFamilyStore cfs = getCurrentColumnFamilyStore(); + String fullIndexName = String.format("%s.%s", currentTable(), indexName); + long maxWaitMillis = 10000; + long startTime = System.currentTimeMillis(); + while (! cfs.indexManager.getBuiltIndexNames().contains(fullIndexName)) + { + Thread.sleep(100); + long wait = System.currentTimeMillis() - startTime; + if (wait > maxWaitMillis) + fail(String.format("Timed out waiting for index %s to build (%s)ms", fullIndexName, wait)); + } + } + } +} diff --git a/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java b/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java index 6f92bd33b1..793348adc9 100644 --- a/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java +++ b/test/unit/org/apache/cassandra/io/sstable/SSTableReaderTest.java @@ -19,15 +19,8 @@ package org.apache.cassandra.io.sstable; import java.io.IOException; import java.nio.ByteBuffer; -import java.util.ArrayList; -import java.util.Arrays; -import java.util.Collection; -import java.util.List; -import java.util.Set; -import java.util.concurrent.ExecutionException; -import java.util.concurrent.Future; -import java.util.concurrent.ScheduledThreadPoolExecutor; -import java.util.concurrent.ThreadPoolExecutor; +import java.util.*; +import java.util.concurrent.*; import com.google.common.collect.Sets; import org.junit.Assert; @@ -40,19 +33,17 @@ import org.apache.cassandra.SchemaLoader; import org.apache.cassandra.Util; import org.apache.cassandra.config.DatabaseDescriptor; import org.apache.cassandra.cql3.Operator; -import org.apache.cassandra.db.BufferDecoratedKey; -import org.apache.cassandra.db.ColumnFamilyStore; -import org.apache.cassandra.db.DecoratedKey; -import org.apache.cassandra.db.Keyspace; -import org.apache.cassandra.db.ReadCommand; -import org.apache.cassandra.db.RowUpdateBuilder; +import org.apache.cassandra.db.*; import org.apache.cassandra.db.compaction.CompactionManager; import org.apache.cassandra.db.compaction.OperationType; import org.apache.cassandra.db.lifecycle.LifecycleTransaction; -import org.apache.cassandra.db.index.SecondaryIndexSearcher; +import org.apache.cassandra.db.partitions.UnfilteredPartitionIterators; import org.apache.cassandra.db.rows.Row; -import org.apache.cassandra.dht.*; +import org.apache.cassandra.dht.IPartitioner; import org.apache.cassandra.dht.LocalPartitioner.LocalToken; +import org.apache.cassandra.dht.Range; +import org.apache.cassandra.dht.Token; +import org.apache.cassandra.index.Index; import org.apache.cassandra.io.sstable.format.SSTableReader; import org.apache.cassandra.io.util.FileDataInput; import org.apache.cassandra.io.util.MmappedSegmentedFile; @@ -64,8 +55,8 @@ import org.apache.cassandra.utils.ByteBufferUtil; import org.apache.cassandra.utils.Pair; import static org.junit.Assert.assertEquals; +import static org.junit.Assert.assertNotNull; import static org.junit.Assert.assertTrue; -import static org.apache.cassandra.utils.ByteBufferUtil.bytes; @RunWith(OrderedJUnit4ClassRunner.class) public class SSTableReaderTest @@ -388,19 +379,23 @@ public class SSTableReaderTest store.forceBlockingFlush(); - ColumnFamilyStore indexCfs = store.indexManager.getIndexForColumn(store.metadata.getColumnDefinition(bytes("birthdate"))).getIndexCfs(); - assert indexCfs.isIndex(); - SSTableReader sstable = indexCfs.getLiveSSTables().iterator().next(); - assert sstable.first.getToken() instanceof LocalToken; - - try(SegmentedFile.Builder ibuilder = SegmentedFile.getBuilder(DatabaseDescriptor.getIndexAccessMode(), false); - SegmentedFile.Builder dbuilder = SegmentedFile.getBuilder(DatabaseDescriptor.getDiskAccessMode(), sstable.compression)) + for(ColumnFamilyStore indexCfs : store.indexManager.getAllIndexColumnFamilyStores()) { - sstable.saveSummary(ibuilder, dbuilder); + assert indexCfs.isIndex(); + SSTableReader sstable = indexCfs.getLiveSSTables().iterator().next(); + assert sstable.first.getToken() instanceof LocalToken; + + try (SegmentedFile.Builder ibuilder = SegmentedFile.getBuilder(DatabaseDescriptor.getIndexAccessMode(), + false); + SegmentedFile.Builder dbuilder = SegmentedFile.getBuilder(DatabaseDescriptor.getDiskAccessMode(), + sstable.compression)) + { + sstable.saveSummary(ibuilder, dbuilder); + } + SSTableReader reopened = SSTableReader.open(sstable.descriptor); + assert reopened.first.getToken() instanceof LocalToken; + reopened.selfRef().release(); } - SSTableReader reopened = SSTableReader.open(sstable.descriptor); - assert reopened.first.getToken() instanceof LocalToken; - reopened.selfRef().release(); } /** see CASSANDRA-5407 */ @@ -545,16 +540,18 @@ public class SSTableReaderTest for (ColumnFamilyStore cfs : indexedCFS.concatWithIndexes()) clearAndLoad(cfs); - ByteBuffer bBB = bytes("birthdate"); // query using index to see if sstable for secondary index opens ReadCommand rc = Util.cmd(indexedCFS).fromKeyIncl("k1").toKeyIncl("k3") .columns("birthdate") .filterOn("birthdate", Operator.EQ, 1L) .build(); - List searchers = indexedCFS.indexManager.getIndexSearchersFor(rc); - assertEquals(searchers.size(), 1); - assertEquals(1, Util.getAll(rc).size()); + Index.Searcher searcher = indexedCFS.indexManager.getBestIndexFor(rc).searcherFor(rc); + assertNotNull(searcher); + try (ReadOrderGroup orderGroup = ReadOrderGroup.forCommand(rc)) + { + assertEquals(1, Util.size(UnfilteredPartitionIterators.filter(searcher.search(orderGroup), rc.nowInSec()))); + } } private List> makeRanges(Token left, Token right) diff --git a/test/unit/org/apache/cassandra/schema/DefsTest.java b/test/unit/org/apache/cassandra/schema/DefsTest.java index c3bcf1a41f..5eed80fabc 100644 --- a/test/unit/org/apache/cassandra/schema/DefsTest.java +++ b/test/unit/org/apache/cassandra/schema/DefsTest.java @@ -40,11 +40,11 @@ import org.apache.cassandra.cql3.UntypedResultSet; import org.apache.cassandra.db.ColumnFamilyStore; import org.apache.cassandra.db.Directories; import org.apache.cassandra.db.Keyspace; -import org.apache.cassandra.db.index.SecondaryIndex; import org.apache.cassandra.db.lifecycle.TransactionLog; import org.apache.cassandra.db.marshal.BytesType; import org.apache.cassandra.db.marshal.UTF8Type; import org.apache.cassandra.exceptions.ConfigurationException; +import org.apache.cassandra.index.Index; import org.apache.cassandra.io.sstable.Component; import org.apache.cassandra.io.sstable.Descriptor; import org.apache.cassandra.locator.OldNetworkTopologyStrategy; @@ -511,9 +511,18 @@ public class DefsTest cfs.forceBlockingFlush(); ColumnDefinition indexedColumn = cfs.metadata.getColumnDefinition(ByteBufferUtil.bytes("birthdate")); - SecondaryIndex index = cfs.indexManager.getIndexForColumn(indexedColumn); - ColumnFamilyStore indexedCfs = index.getIndexCfs(); - Descriptor desc = indexedCfs.getLiveSSTables().iterator().next().descriptor; + IndexMetadata index = cfs.metadata.getIndexes() + .get(indexedColumn) + .iterator() + .next(); + ColumnFamilyStore indexCfs = cfs.indexManager.listIndexes() + .stream() + .filter(i -> i.getIndexMetadata().equals(index)) + .map(Index::getBackingTable) + .findFirst() + .orElseThrow(() -> new AssertionError("Index not found")) + .orElseThrow(() -> new AssertionError("Index has no backing table")); + Descriptor desc = indexCfs.getLiveSSTables().iterator().next().descriptor; // drop the index CFMetaData meta = cfs.metadata.copy(); @@ -536,7 +545,7 @@ public class DefsTest MigrationManager.announceColumnFamilyUpdate(meta, false); // check - assertTrue(cfs.indexManager.getIndexes().isEmpty()); + assertTrue(cfs.indexManager.listIndexes().isEmpty()); TransactionLog.waitForDeletions(); assertFalse(new File(desc.filenameFor(Component.DATA)).exists()); } diff --git a/test/unit/org/apache/cassandra/schema/LegacySchemaMigratorTest.java b/test/unit/org/apache/cassandra/schema/LegacySchemaMigratorTest.java index a1f9570495..73eed06999 100644 --- a/test/unit/org/apache/cassandra/schema/LegacySchemaMigratorTest.java +++ b/test/unit/org/apache/cassandra/schema/LegacySchemaMigratorTest.java @@ -26,7 +26,9 @@ import com.google.common.collect.ImmutableList; import org.junit.Test; import org.apache.cassandra.SchemaLoader; -import org.apache.cassandra.config.*; +import org.apache.cassandra.config.CFMetaData; +import org.apache.cassandra.config.ColumnDefinition; +import org.apache.cassandra.config.Schema; import org.apache.cassandra.cql3.CQLTester; import org.apache.cassandra.cql3.ColumnIdentifier; import org.apache.cassandra.cql3.functions.*;