mirror of https://github.com/apache/cassandra
Improve and clean up documentation and fix typos
This patch includes all the changes from the PRs that introduce small changes related to typos and similar in the documentation. The changes are accumulated from the following PRs: - https://github.com/apache/cassandra/pull/206 - https://github.com/apache/cassandra/pull/359 - https://github.com/apache/cassandra/pull/366 - https://github.com/apache/cassandra/pull/390 - https://github.com/apache/cassandra/pull/450 - https://github.com/apache/cassandra/pull/567 - https://github.com/apache/cassandra/pull/615 - https://github.com/apache/cassandra/pull/618 - https://github.com/apache/cassandra/pull/746 - https://github.com/apache/cassandra/pull/984 - https://github.com/apache/cassandra/pull/1052 - https://github.com/apache/cassandra/pull/1088 - https://github.com/apache/cassandra/pull/1274 - https://github.com/apache/cassandra/pull/1378 - https://github.com/apache/cassandra/pull/1404 - https://github.com/apache/cassandra/pull/1504 - https://github.com/apache/cassandra/pull/1540 - https://github.com/apache/cassandra/pull/1544 - https://github.com/apache/cassandra/pull/1673 - https://github.com/apache/cassandra/pull/1697 - https://github.com/apache/cassandra/pull/1722 - https://github.com/apache/cassandra/pull/1815 - https://github.com/apache/cassandra/pull/1830 - https://github.com/apache/cassandra/pull/1863 - https://github.com/apache/cassandra/pull/1865 - https://github.com/apache/cassandra/pull/1879 - https://github.com/apache/cassandra/pull/2062 patch by Nikita Eshkeev, reviewed by Stefan Miklosovic, Lorina Poland, Michael Semb Wever for CASSANDRA-18185 Co-authored-by: kalmant <kalmant@users.noreply.github.com> Co-authored-by: Dmitry <xotonic@yandex.ru> Co-authored-by: Tibor Répási <rtib@users.noreply.github.com> Co-authored-by: Tzach Livyatan <tzach@scylladb.com> Co-authored-by: Jérôme BAROTIN <jeromebarotin@gmail.com> Co-authored-by: Giorgio Giuffrè <giorgiogiuffre23@gmail.com> Co-authored-by: Siddhartha Tiwari <201851127@iiitvadodara.ac.in> Co-authored-by: Angelo Polo <language.devel@gmail.com> Co-authored-by: Tjeu Kayim <15987676+TjeuKayim@users.noreply.github.com> Co-authored-by: 陳傑夫 <chienfuchen32@gmail.com> Co-authored-by: Bhouse99 <bhouse99@protonmail.com> Co-authored-by: Matthew Hardwick <MatthewRHardwick@gmail.com> Co-authored-by: Paul Wouters <paul.wouters@aiven.io> Co-authored-by: Romain Hardouin <romain_hardouin@yahoo.fr> Co-authored-by: Guilherme Poleto <gpoleto@alunos.utfpr.edu.br> Co-authored-by: 陳傑夫 <chienfuchen32@gmail.com> Co-authored-by: etc-crontab <jujut@free.fr> Co-authored-by: Prashant Bhuruk <prashantbhuruk88@gmail.com> Co-authored-by: Jingchuan Zhu <56401528+codingswag998@users.noreply.github.com> Co-authored-by: Ryan Stewart <ryan.stewart@rackspace.com> Co-authored-by: utkarsh-agrawal-jm <107914361+utkarsh-agrawal-jm@users.noreply.github.com> Co-authored-by: Ben Dalling <b.dalling@locp.co.uk> Co-authored-by: Terry L. Blessing <tlblessing1@gmail.com> Co-authored-by: gruzilkin <gruzmob@gmail.com> Co-authored-by: Kevin <kevin.xgr@gmail.com> Co-authored-by: yziadeh <121903189+yziadeh@users.noreply.github.com> Co-authored-by: Lorina Poland <lorina@datastax.com> Co-authored-by: Stefan Miklosovic <smiklosovic@apache.org>
This commit is contained in:
parent
93cc75ccdf
commit
f27790c969
|
|
@ -364,14 +364,14 @@ dependencies into the constructor is not practical, wrapping accesses to global
|
||||||
|
|
||||||
|
|
||||||
**Example, alternative**
|
**Example, alternative**
|
||||||
```javayy
|
```java
|
||||||
class SomeVerbHandler implements IVerbHandler<SomeMessage>
|
class SomeVerbHandler implements IVerbHandler<SomeMessage>
|
||||||
{
|
{
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
protected boolean isAlive(InetAddress addr) { return FailureDetector.instance.isAlive(msg.payload.otherNode); }
|
protected boolean isAlive(InetAddress addr) { return FailureDetector.instance.isAlive(msg.payload.otherNode); }
|
||||||
|
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
protected void streamSomethind(InetAddress to) { new StreamPlan(to).requestRanges(someRanges).execute(); }
|
protected void streamSomething(InetAddress to) { new StreamPlan(to).requestRanges(someRanges).execute(); }
|
||||||
|
|
||||||
@VisibleForTesting
|
@VisibleForTesting
|
||||||
protected void compactSomething(ColumnFamilyStore cfs ) { CompactionManager.instance.submitBackground(); }
|
protected void compactSomething(ColumnFamilyStore cfs ) { CompactionManager.instance.submitBackground(); }
|
||||||
|
|
@ -404,7 +404,7 @@ class SomeVerbTest
|
||||||
protected boolean isAlive(InetAddress addr) { return alive; }
|
protected boolean isAlive(InetAddress addr) { return alive; }
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
protected void streamSomethind(InetAddress to) { streamCalled = true; }
|
protected void streamSomething(InetAddress to) { streamCalled = true; }
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
protected void compactSomething(ColumnFamilyStore cfs ) { compactCalled = true; }
|
protected void compactSomething(ColumnFamilyStore cfs ) { compactCalled = true; }
|
||||||
|
|
|
||||||
|
|
@ -44,8 +44,8 @@ num_tokens: 16
|
||||||
allocate_tokens_for_local_replication_factor: 3
|
allocate_tokens_for_local_replication_factor: 3
|
||||||
|
|
||||||
# initial_token allows you to specify tokens manually. While you can use it with
|
# initial_token allows you to specify tokens manually. While you can use it with
|
||||||
# vnodes (num_tokens > 1, above) -- in which case you should provide a
|
# vnodes (num_tokens > 1, above) -- in which case you should provide a
|
||||||
# comma-separated list -- it's primarily used when adding nodes to legacy clusters
|
# comma-separated list -- it's primarily used when adding nodes to legacy clusters
|
||||||
# that do not have vnodes enabled.
|
# that do not have vnodes enabled.
|
||||||
# initial_token:
|
# initial_token:
|
||||||
|
|
||||||
|
|
@ -290,7 +290,7 @@ credentials_validity: 2000ms
|
||||||
partitioner: org.apache.cassandra.dht.Murmur3Partitioner
|
partitioner: org.apache.cassandra.dht.Murmur3Partitioner
|
||||||
|
|
||||||
# Directories where Cassandra should store data on disk. If multiple
|
# Directories where Cassandra should store data on disk. If multiple
|
||||||
# directories are specified, Cassandra will spread data evenly across
|
# directories are specified, Cassandra will spread data evenly across
|
||||||
# them by partitioning the token ranges.
|
# them by partitioning the token ranges.
|
||||||
# If not set, the default directory is $CASSANDRA_HOME/data/data.
|
# If not set, the default directory is $CASSANDRA_HOME/data/data.
|
||||||
# data_file_directories:
|
# data_file_directories:
|
||||||
|
|
@ -496,8 +496,8 @@ counter_cache_save_period: 7200s
|
||||||
# Min unit: s
|
# Min unit: s
|
||||||
# cache_load_timeout: 30s
|
# cache_load_timeout: 30s
|
||||||
|
|
||||||
# commitlog_sync may be either "periodic", "group", or "batch."
|
# commitlog_sync may be either "periodic", "group", or "batch."
|
||||||
#
|
#
|
||||||
# When in batch mode, Cassandra won't ack writes until the commit log
|
# When in batch mode, Cassandra won't ack writes until the commit log
|
||||||
# has been flushed to disk. Each incoming write will trigger the flush task.
|
# has been flushed to disk. Each incoming write will trigger the flush task.
|
||||||
# commitlog_sync_batch_window_in_ms is a deprecated value. Previously it had
|
# commitlog_sync_batch_window_in_ms is a deprecated value. Previously it had
|
||||||
|
|
@ -940,7 +940,7 @@ incremental_backups: false
|
||||||
snapshot_before_compaction: false
|
snapshot_before_compaction: false
|
||||||
|
|
||||||
# Whether or not a snapshot is taken of the data before keyspace truncation
|
# Whether or not a snapshot is taken of the data before keyspace truncation
|
||||||
# or dropping of column families. The STRONGLY advised default of true
|
# or dropping of column families. The STRONGLY advised default of true
|
||||||
# should be used to provide data safety. If you set this flag to false, you will
|
# should be used to provide data safety. If you set this flag to false, you will
|
||||||
# lose data on truncation or drop.
|
# lose data on truncation or drop.
|
||||||
auto_snapshot: true
|
auto_snapshot: true
|
||||||
|
|
@ -994,7 +994,7 @@ column_index_cache_size: 2KiB
|
||||||
#
|
#
|
||||||
# concurrent_compactors defaults to the smaller of (number of disks,
|
# concurrent_compactors defaults to the smaller of (number of disks,
|
||||||
# number of cores), with a minimum of 2 and a maximum of 8.
|
# number of cores), with a minimum of 2 and a maximum of 8.
|
||||||
#
|
#
|
||||||
# If your data directories are backed by SSD, you should increase this
|
# If your data directories are backed by SSD, you should increase this
|
||||||
# to the number of cores.
|
# to the number of cores.
|
||||||
# concurrent_compactors: 1
|
# concurrent_compactors: 1
|
||||||
|
|
@ -1022,7 +1022,7 @@ compaction_throughput: 64MiB/s
|
||||||
|
|
||||||
# When compacting, the replacement sstable(s) can be opened before they
|
# When compacting, the replacement sstable(s) can be opened before they
|
||||||
# are completely written, and used in place of the prior sstables for
|
# are completely written, and used in place of the prior sstables for
|
||||||
# any range that has been written. This helps to smoothly transfer reads
|
# any range that has been written. This helps to smoothly transfer reads
|
||||||
# between the sstables, reducing page cache churn and keeping hot rows hot
|
# between the sstables, reducing page cache churn and keeping hot rows hot
|
||||||
# Set sstable_preemptive_open_interval to null for disabled which is equivalent to
|
# Set sstable_preemptive_open_interval to null for disabled which is equivalent to
|
||||||
# sstable_preemptive_open_interval_in_mb being negative
|
# sstable_preemptive_open_interval_in_mb being negative
|
||||||
|
|
@ -1170,10 +1170,10 @@ slow_query_log_timeout: 500ms
|
||||||
# Enable operation timeout information exchange between nodes to accurately
|
# Enable operation timeout information exchange between nodes to accurately
|
||||||
# measure request timeouts. If disabled, replicas will assume that requests
|
# measure request timeouts. If disabled, replicas will assume that requests
|
||||||
# were forwarded to them instantly by the coordinator, which means that
|
# were forwarded to them instantly by the coordinator, which means that
|
||||||
# under overload conditions we will waste that much extra time processing
|
# under overload conditions we will waste that much extra time processing
|
||||||
# already-timed-out requests.
|
# already-timed-out requests.
|
||||||
#
|
#
|
||||||
# Warning: It is generally assumed that users have setup NTP on their clusters, and that clocks are modestly in sync,
|
# Warning: It is generally assumed that users have setup NTP on their clusters, and that clocks are modestly in sync,
|
||||||
# since this is a requirement for general correctness of last write wins.
|
# since this is a requirement for general correctness of last write wins.
|
||||||
# internode_timeout: true
|
# internode_timeout: true
|
||||||
|
|
||||||
|
|
@ -1586,7 +1586,7 @@ compaction_tombstone_warning_threshold: 100000
|
||||||
# max_concurrent_automatic_sstable_upgrades: 1
|
# max_concurrent_automatic_sstable_upgrades: 1
|
||||||
|
|
||||||
# Audit logging - Logs every incoming CQL command request, authentication to a node. See the docs
|
# Audit logging - Logs every incoming CQL command request, authentication to a node. See the docs
|
||||||
# on audit_logging for full details about the various configuration options.
|
# on audit_logging for full details about the various configuration options and production tips.
|
||||||
audit_logging_options:
|
audit_logging_options:
|
||||||
enabled: false
|
enabled: false
|
||||||
logger:
|
logger:
|
||||||
|
|
@ -1602,11 +1602,13 @@ audit_logging_options:
|
||||||
# block: true
|
# block: true
|
||||||
# max_queue_weight: 268435456 # 256 MiB
|
# max_queue_weight: 268435456 # 256 MiB
|
||||||
# max_log_size: 17179869184 # 16 GiB
|
# max_log_size: 17179869184 # 16 GiB
|
||||||
## archive command is "/path/to/script.sh %path" where %path is replaced with the file being rolled:
|
#
|
||||||
|
## If archive_command is empty or unset, Cassandra uses a built-in DeletingArchiver that deletes the oldest files if ``max_log_size`` is reached.
|
||||||
|
## If archive_command is set, Cassandra does not use DeletingArchiver, so it is the responsibility of the script to make any required cleanup.
|
||||||
|
## Example: "/path/to/script.sh %path" where %path is replaced with the file being rolled.
|
||||||
# archive_command:
|
# archive_command:
|
||||||
# max_archive_retries: 10
|
# max_archive_retries: 10
|
||||||
|
|
||||||
|
|
||||||
# default options for full query logging - these can be overridden from command line when executing
|
# default options for full query logging - these can be overridden from command line when executing
|
||||||
# nodetool enablefullquerylog
|
# nodetool enablefullquerylog
|
||||||
# full_query_logging_options:
|
# full_query_logging_options:
|
||||||
|
|
|
||||||
|
|
@ -15,7 +15,7 @@
|
||||||
; specific language governing permissions and limitations
|
; specific language governing permissions and limitations
|
||||||
; under the License.
|
; under the License.
|
||||||
;
|
;
|
||||||
; Sample ~/.cqlshrc file.
|
; Sample ~/.cassandra/cqlshrc file.
|
||||||
|
|
||||||
[authentication]
|
[authentication]
|
||||||
;; If Cassandra has auth enabled, fill out these options
|
;; If Cassandra has auth enabled, fill out these options
|
||||||
|
|
@ -23,7 +23,6 @@
|
||||||
; credentials = ~/.cassandra/credentials
|
; credentials = ~/.cassandra/credentials
|
||||||
; keyspace = ks1
|
; keyspace = ks1
|
||||||
|
|
||||||
|
|
||||||
[auth_provider]
|
[auth_provider]
|
||||||
;; you can specify any auth provider found in your python environment
|
;; you can specify any auth provider found in your python environment
|
||||||
;; module and class will be used to dynamically load the class
|
;; module and class will be used to dynamically load the class
|
||||||
|
|
@ -33,6 +32,10 @@
|
||||||
; classname = PlainTextAuthProvider
|
; classname = PlainTextAuthProvider
|
||||||
; username = user1
|
; username = user1
|
||||||
|
|
||||||
|
[protocol]
|
||||||
|
;; Specify a specific protcol version otherwise the client will default and downgrade as necessary
|
||||||
|
; version = None
|
||||||
|
|
||||||
[ui]
|
[ui]
|
||||||
;; Whether or not to display query results with colors
|
;; Whether or not to display query results with colors
|
||||||
; color = on
|
; color = on
|
||||||
|
|
@ -153,9 +156,8 @@ port = 9042
|
||||||
; boolstyle = True,False
|
; boolstyle = True,False
|
||||||
|
|
||||||
;; The number of child worker processes to create for
|
;; The number of child worker processes to create for
|
||||||
;; COPY tasks. Defaults to a max of 4 for COPY FROM and 16
|
;; COPY tasks. Defaults to 16 for `COPY` tasks.
|
||||||
;; for COPY TO. However, at most (num_cores - 1) processes
|
;; However, at most (num_cores - 1) processes will be created.
|
||||||
;; will be created.
|
|
||||||
; numprocesses =
|
; numprocesses =
|
||||||
|
|
||||||
;; The maximum number of failed attempts to fetch a range of data (when using
|
;; The maximum number of failed attempts to fetch a range of data (when using
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,5 @@
|
||||||
alter_table_statement::= ALTER TABLE [ IF EXISTS ] table_name alter_table_instruction
|
alter_table_statement::= ALTER TABLE [ IF EXISTS ] table_name alter_table_instruction
|
||||||
alter_table_instruction::= ADD [ IF NOT EXISTS ] column_name cql_type ( ',' column_name cql_type )*
|
alter_table_instruction::= ADD [ IF NOT EXISTS ] column_name cql_type ( ',' column_name cql_type )*
|
||||||
| DROP [ IF EXISTS ] column_name ( column_name )*
|
| DROP [ IF EXISTS ] column_name ( ',' column_name )*
|
||||||
| RENAME [ IF EXISTS ] column_name to column_name (AND column_name to column_name)*
|
| RENAME [ IF EXISTS ] column_name to column_name (AND column_name to column_name)*
|
||||||
| WITH options
|
| WITH options
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
= Dynamo
|
= Dynamo
|
||||||
|
|
||||||
Apache Cassandra relies on a number of techniques from Amazon's
|
Apache Cassandra relies on a number of techniques from Amazon's
|
||||||
http://courses.cse.tamu.edu/caverlee/csce438/readings/dynamo-paper.pdf[Dynamo]
|
https://www.cs.cornell.edu/courses/cs5414/2017fa/papers/dynamo.pdf[Dynamo]
|
||||||
distributed storage key-value system. Each node in the Dynamo system has
|
distributed storage key-value system. Each node in the Dynamo system has
|
||||||
three main components:
|
three main components:
|
||||||
|
|
||||||
|
|
@ -22,10 +22,10 @@ protocol
|
||||||
|
|
||||||
Cassandra was designed this way to meet large-scale (PiB+)
|
Cassandra was designed this way to meet large-scale (PiB+)
|
||||||
business-critical storage requirements. In particular, as applications
|
business-critical storage requirements. In particular, as applications
|
||||||
demanded full global replication of petabyte scale datasets along with
|
demanded full global replication of petabyte-scale datasets along with
|
||||||
always available low-latency reads and writes, it became imperative to
|
always available low-latency reads and writes, it became imperative to
|
||||||
design a new kind of database model as the relational database systems
|
design a new kind of database model as the relational database systems
|
||||||
of the time struggled to meet the new requirements of global scale
|
of the time struggled to meet the new requirements of global-scale
|
||||||
applications.
|
applications.
|
||||||
|
|
||||||
== Dataset Partitioning: Consistent Hashing
|
== Dataset Partitioning: Consistent Hashing
|
||||||
|
|
@ -38,11 +38,11 @@ as racks and even datacenters. As every replica can independently accept
|
||||||
mutations to every key that it owns, every key must be versioned. Unlike
|
mutations to every key that it owns, every key must be versioned. Unlike
|
||||||
in the original Dynamo paper where deterministic versions and vector
|
in the original Dynamo paper where deterministic versions and vector
|
||||||
clocks were used to reconcile concurrent updates to a key, Cassandra
|
clocks were used to reconcile concurrent updates to a key, Cassandra
|
||||||
uses a simpler last write wins model where every mutation is timestamped
|
uses a simpler last-write-wins model where every mutation is timestamped
|
||||||
(including deletes) and then the latest version of data is the "winning"
|
(including deletes) and then the latest version of data is the "winning"
|
||||||
value. Formally speaking, Cassandra uses a Last-Write-Wins Element-Set
|
value. Formally speaking, Cassandra uses a Last-Write-Wins Element-Set
|
||||||
conflict-free replicated data type for each CQL row, or
|
conflict-free replicated data type for each CQL row, or
|
||||||
https://en.wikipedia.org/wiki/Conflict-free_replicated_data_type LWW-Element-Set_(Last-Write-Wins-Element-Set)[LWW-Element-Set
|
https://en.wikipedia.org/wiki/Conflict-free_replicated_data_type#LWW-Element-Set_(Last-Write-Wins-Element-Set)[LWW-Element-Set
|
||||||
CRDT], to resolve conflicting mutations on replica sets.
|
CRDT], to resolve conflicting mutations on replica sets.
|
||||||
|
|
||||||
=== Consistent Hashing using a Token Ring
|
=== Consistent Hashing using a Token Ring
|
||||||
|
|
@ -76,14 +76,14 @@ gRF=3 can be visualized as follows:
|
||||||
|
|
||||||
image::ring.svg[image]
|
image::ring.svg[image]
|
||||||
|
|
||||||
You can see that in a Dynamo like system, ranges of keys, also known as
|
You can see that in a Dynamo-like system, ranges of keys, also known as
|
||||||
*token ranges*, map to the same physical set of nodes. In this example,
|
*token ranges*, map to the same physical set of nodes. In this example,
|
||||||
all keys that fall in the token range excluding token 1 and including
|
all keys that fall in the token range excluding token 1 and including
|
||||||
token 2 (grange(t1, t2]) are stored on nodes 2, 3 and 4.
|
token 2 (grange(t1, t2]) are stored on nodes 2, 3 and 4.
|
||||||
|
|
||||||
=== Multiple Tokens per Physical Node (vnodes)
|
=== Multiple Tokens per Physical Node (vnodes)
|
||||||
|
|
||||||
Simple single token consistent hashing works well if you have many
|
Simple single-token consistent hashing works well if you have many
|
||||||
physical nodes to spread data over, but with evenly spaced tokens and a
|
physical nodes to spread data over, but with evenly spaced tokens and a
|
||||||
small number of physical nodes, incremental scaling (adding just a few
|
small number of physical nodes, incremental scaling (adding just a few
|
||||||
nodes of capacity) is difficult because there are no token selections
|
nodes of capacity) is difficult because there are no token selections
|
||||||
|
|
@ -104,8 +104,7 @@ even a single node.
|
||||||
|
|
||||||
Cassandra introduces some nomenclature to handle these concepts:
|
Cassandra introduces some nomenclature to handle these concepts:
|
||||||
|
|
||||||
* *Token*: A single position on the dynamo style hash
|
* *Token*: A single position on the Dynamo-style hash ring.
|
||||||
ring.
|
|
||||||
* *Endpoint*: A single physical IP and port on the network.
|
* *Endpoint*: A single physical IP and port on the network.
|
||||||
* *Host ID*: A unique identifier for a single "physical" node, usually
|
* *Host ID*: A unique identifier for a single "physical" node, usually
|
||||||
present at one gEndpoint and containing one or more
|
present at one gEndpoint and containing one or more
|
||||||
|
|
@ -131,7 +130,7 @@ data across the cluster.
|
||||||
. When a node is decommissioned, it loses data roughly equally to other
|
. When a node is decommissioned, it loses data roughly equally to other
|
||||||
members of the ring, again keeping equal distribution of data across the
|
members of the ring, again keeping equal distribution of data across the
|
||||||
cluster.
|
cluster.
|
||||||
. If a node becomes unavailable, query load (especially token aware
|
. If a node becomes unavailable, query load (especially token-aware
|
||||||
query load), is evenly distributed across many other nodes.
|
query load), is evenly distributed across many other nodes.
|
||||||
|
|
||||||
Multiple tokens, however, can also have disadvantages:
|
Multiple tokens, however, can also have disadvantages:
|
||||||
|
|
@ -152,7 +151,7 @@ Note that in Cassandra `2.x`, the only token allocation algorithm
|
||||||
available was picking random tokens, which meant that to keep balance
|
available was picking random tokens, which meant that to keep balance
|
||||||
the default number of tokens per node had to be quite high, at `256`.
|
the default number of tokens per node had to be quite high, at `256`.
|
||||||
This had the effect of coupling many physical endpoints together,
|
This had the effect of coupling many physical endpoints together,
|
||||||
increasing the risk of unavailability. That is why in `3.x +` the new
|
increasing the risk of unavailability. That is why in `3.x +` a new
|
||||||
deterministic token allocator was added which intelligently picks tokens
|
deterministic token allocator was added which intelligently picks tokens
|
||||||
such that the ring is optimally balanced while requiring a much lower
|
such that the ring is optimally balanced while requiring a much lower
|
||||||
number of tokens per physical node.
|
number of tokens per physical node.
|
||||||
|
|
@ -256,7 +255,7 @@ secondary indices with them.
|
||||||
Transient replication is an experimental feature that is not ready
|
Transient replication is an experimental feature that is not ready
|
||||||
for production use. The expected audience is experienced users of
|
for production use. The expected audience is experienced users of
|
||||||
Cassandra capable of fully validating a deployment of their particular
|
Cassandra capable of fully validating a deployment of their particular
|
||||||
application. That means being able check that operations like reads,
|
application. That means you have the experience to check that operations like reads,
|
||||||
writes, decommission, remove, rebuild, repair, and replace all work with
|
writes, decommission, remove, rebuild, repair, and replace all work with
|
||||||
your queries, data, configuration, operational practices, and
|
your queries, data, configuration, operational practices, and
|
||||||
availability requirements.
|
availability requirements.
|
||||||
|
|
@ -269,18 +268,18 @@ transient replication, as well as LWT, logged batches, and counters.
|
||||||
Cassandra uses mutation timestamp versioning to guarantee eventual
|
Cassandra uses mutation timestamp versioning to guarantee eventual
|
||||||
consistency of data. Specifically all mutations that enter the system do
|
consistency of data. Specifically all mutations that enter the system do
|
||||||
so with a timestamp provided either from a client clock or, absent a
|
so with a timestamp provided either from a client clock or, absent a
|
||||||
client provided timestamp, from the coordinator node's clock. Updates
|
client-provided timestamp, from the coordinator node's clock. Updates
|
||||||
resolve according to the conflict resolution rule of last write wins.
|
resolve according to the conflict resolution rule of last write wins.
|
||||||
Cassandra's correctness does depend on these clocks, so make sure a
|
Cassandra's correctness does depend on these clocks, so make sure a
|
||||||
proper time synchronization process is running such as NTP.
|
proper time synchronization process is running such as NTP.
|
||||||
|
|
||||||
Cassandra applies separate mutation timestamps to every column of every
|
Cassandra applies separate mutation timestamps to every column of every
|
||||||
row within a CQL partition. Rows are guaranteed to be unique by primary
|
row within a CQL partition. Rows are guaranteed to be unique by primary
|
||||||
key, and each column in a row resolve concurrent mutations according to
|
key, and each column in a row resolves concurrent mutations according to
|
||||||
last-write-wins conflict resolution. This means that updates to
|
last-write-wins conflict resolution. This means that updates to
|
||||||
different primary keys within a partition can actually resolve without
|
different primary keys within a partition can actually resolve without
|
||||||
conflict! Furthermore the CQL collection types such as maps and sets use
|
conflict! Furthermore the CQL collection types such as maps and sets use
|
||||||
this same conflict free mechanism, meaning that concurrent updates to
|
this same conflict-free mechanism, meaning that concurrent updates to
|
||||||
maps and sets are guaranteed to resolve as well.
|
maps and sets are guaranteed to resolve as well.
|
||||||
|
|
||||||
==== Replica Synchronization
|
==== Replica Synchronization
|
||||||
|
|
@ -293,7 +292,7 @@ many best-effort techniques to drive convergence of replicas including
|
||||||
|
|
||||||
These techniques are only best-effort, however, and to guarantee
|
These techniques are only best-effort, however, and to guarantee
|
||||||
eventual consistency Cassandra implements `anti-entropy
|
eventual consistency Cassandra implements `anti-entropy
|
||||||
repair <repair>` where replicas calculate hierarchical hash-trees over
|
repair <repair>` where replicas calculate hierarchical hash trees over
|
||||||
their datasets called https://en.wikipedia.org/wiki/Merkle_tree[Merkle
|
their datasets called https://en.wikipedia.org/wiki/Merkle_tree[Merkle
|
||||||
trees] that can then be compared across replicas to identify mismatched
|
trees] that can then be compared across replicas to identify mismatched
|
||||||
data. Like the original Dynamo paper Cassandra supports full repairs
|
data. Like the original Dynamo paper Cassandra supports full repairs
|
||||||
|
|
@ -340,7 +339,7 @@ The following consistency levels are available:
|
||||||
A majority of the replicas in each datacenter must respond.
|
A majority of the replicas in each datacenter must respond.
|
||||||
`LOCAL_ONE`::
|
`LOCAL_ONE`::
|
||||||
Only a single replica must respond. In a multi-datacenter cluster,
|
Only a single replica must respond. In a multi-datacenter cluster,
|
||||||
this also gaurantees that read requests are not sent to replicas in a
|
this also guarantees that read requests are not sent to replicas in a
|
||||||
remote datacenter.
|
remote datacenter.
|
||||||
`ANY`::
|
`ANY`::
|
||||||
A single replica may respond, or the coordinator may store a hint. If
|
A single replica may respond, or the coordinator may store a hint. If
|
||||||
|
|
@ -400,7 +399,7 @@ versions. In Cassandra's gossip system, nodes exchange state information
|
||||||
not only about themselves but also about other nodes they know about.
|
not only about themselves but also about other nodes they know about.
|
||||||
This information is versioned with a vector clock of
|
This information is versioned with a vector clock of
|
||||||
`(generation, version)` tuples, where the generation is a monotonic
|
`(generation, version)` tuples, where the generation is a monotonic
|
||||||
timestamp and version is a logical clock the increments roughly every
|
timestamp and version is a logical clock that increments roughly every
|
||||||
second. These logical clocks allow Cassandra gossip to ignore old
|
second. These logical clocks allow Cassandra gossip to ignore old
|
||||||
versions of cluster state just by inspecting the logical clocks
|
versions of cluster state just by inspecting the logical clocks
|
||||||
presented with gossip messages.
|
presented with gossip messages.
|
||||||
|
|
@ -417,10 +416,10 @@ state with.
|
||||||
one exists)
|
one exists)
|
||||||
. Gossips with a seed node if that didn't happen in step 2.
|
. Gossips with a seed node if that didn't happen in step 2.
|
||||||
|
|
||||||
When an operator first bootstraps a Cassandra cluster they designate
|
When an operator first bootstraps a Cassandra cluster, they designate
|
||||||
certain nodes as seed nodes. Any node can be a seed node and the only
|
certain nodes as seed nodes. Any node can be a seed node, and the only
|
||||||
difference between seed and non-seed nodes is seed nodes are allowed to
|
difference between seed and non-seed nodes is that seed nodes are allowed
|
||||||
bootstrap into the ring without seeing any other seed nodes.
|
to bootstrap into the ring without seeing any other seed nodes.
|
||||||
Furthermore, once a cluster is bootstrapped, seed nodes become
|
Furthermore, once a cluster is bootstrapped, seed nodes become
|
||||||
hotspots for gossip due to step 4 above.
|
hotspots for gossip due to step 4 above.
|
||||||
|
|
||||||
|
|
@ -435,7 +434,7 @@ chosen using existing off-the-shelf service discovery mechanisms.
|
||||||
Nodes do not have to agree on the seed nodes, and indeed once a cluster
|
Nodes do not have to agree on the seed nodes, and indeed once a cluster
|
||||||
is bootstrapped, newly launched nodes can be configured to use any
|
is bootstrapped, newly launched nodes can be configured to use any
|
||||||
existing nodes as seeds. The only advantage to picking the same nodes
|
existing nodes as seeds. The only advantage to picking the same nodes
|
||||||
as seeds is it increases their usefullness as gossip hotspots.
|
as seeds is that it increases their usefulness as gossip hotspots.
|
||||||
====
|
====
|
||||||
|
|
||||||
Currently, gossip also propagates token metadata and schema
|
Currently, gossip also propagates token metadata and schema
|
||||||
|
|
@ -488,7 +487,7 @@ and every additional node brings linear improvements in compute and
|
||||||
storage. In contrast, scaling-up implies adding more capacity to the
|
storage. In contrast, scaling-up implies adding more capacity to the
|
||||||
existing database nodes. Cassandra is also capable of scale-up, and in
|
existing database nodes. Cassandra is also capable of scale-up, and in
|
||||||
certain environments it may be preferable depending on the deployment.
|
certain environments it may be preferable depending on the deployment.
|
||||||
Cassandra gives operators the flexibility to chose either scale-out or
|
Cassandra gives operators the flexibility to choose either scale-out or
|
||||||
scale-up.
|
scale-up.
|
||||||
|
|
||||||
One key aspect of Dynamo that Cassandra follows is to attempt to run on
|
One key aspect of Dynamo that Cassandra follows is to attempt to run on
|
||||||
|
|
@ -507,7 +506,7 @@ API, and allows Cassandra to more easily scale horizontally since
|
||||||
multi-partition transactions spanning multiple nodes are notoriously
|
multi-partition transactions spanning multiple nodes are notoriously
|
||||||
difficult to implement and typically very latent.
|
difficult to implement and typically very latent.
|
||||||
|
|
||||||
Instead, Cassanda chooses to offer fast, consistent, latency at any
|
Instead, Cassandra chooses to offer fast, consistent, latency at any
|
||||||
scale for single partition operations, allowing retrieval of entire
|
scale for single partition operations, allowing retrieval of entire
|
||||||
partitions or only subsets of partitions based on primary key filters.
|
partitions or only subsets of partitions based on primary key filters.
|
||||||
Furthermore, Cassandra does support single partition compare and swap
|
Furthermore, Cassandra does support single partition compare and swap
|
||||||
|
|
@ -516,7 +515,7 @@ functionality via the lightweight transaction CQL API.
|
||||||
=== Simple Interface for Storing Records
|
=== Simple Interface for Storing Records
|
||||||
|
|
||||||
Cassandra, in a slight departure from Dynamo, chooses a storage
|
Cassandra, in a slight departure from Dynamo, chooses a storage
|
||||||
interface that is more sophisticated then "simple key value" stores but
|
interface that is more sophisticated than "simple key-value" stores but
|
||||||
significantly less complex than SQL relational data models. Cassandra
|
significantly less complex than SQL relational data models. Cassandra
|
||||||
presents a wide-column store interface, where partitions of data contain
|
presents a wide-column store interface, where partitions of data contain
|
||||||
multiple rows, each of which contains a flexible set of individually
|
multiple rows, each of which contains a flexible set of individually
|
||||||
|
|
|
||||||
|
|
@ -1,17 +1,17 @@
|
||||||
= Guarantees
|
= Guarantees
|
||||||
|
|
||||||
Apache Cassandra is a highly scalable and reliable database. Cassandra
|
Apache Cassandra is a highly scalable and reliable database. Cassandra
|
||||||
is used in web based applications that serve large number of clients and
|
is used in web-based applications that serve large number of clients and
|
||||||
the quantity of data processed is web-scale (Petabyte) large. Cassandra
|
the quantity of data processed is web-scale (Petabyte) large. Cassandra
|
||||||
makes some guarantees about its scalability, availability and
|
makes some guarantees about its scalability, availability and
|
||||||
reliability. To fully understand the inherent limitations of a storage
|
reliability. To fully understand the inherent limitations of a storage
|
||||||
system in an environment in which a certain level of network partition
|
system in an environment in which a certain level of network partition
|
||||||
failure is to be expected and taken into account when designing the
|
failure is to be expected and taken into account when designing the
|
||||||
system it is important to first briefly introduce the CAP theorem.
|
system, it is important to first briefly introduce the CAP theorem.
|
||||||
|
|
||||||
== What is CAP?
|
== What is CAP?
|
||||||
|
|
||||||
According to the CAP theorem it is not possible for a distributed data
|
According to the CAP theorem, it is not possible for a distributed data
|
||||||
store to provide more than two of the following guarantees
|
store to provide more than two of the following guarantees
|
||||||
simultaneously.
|
simultaneously.
|
||||||
|
|
||||||
|
|
@ -24,7 +24,7 @@ recent write or data.
|
||||||
storage system to failure of a network partition. Even if some of the
|
storage system to failure of a network partition. Even if some of the
|
||||||
messages are dropped or delayed the system continues to operate.
|
messages are dropped or delayed the system continues to operate.
|
||||||
|
|
||||||
CAP theorem implies that when using a network partition, with the
|
The CAP theorem implies that when using a network partition, with the
|
||||||
inherent risk of partition failure, one has to choose between
|
inherent risk of partition failure, one has to choose between
|
||||||
consistency and availability and both cannot be guaranteed at the same
|
consistency and availability and both cannot be guaranteed at the same
|
||||||
time. CAP theorem is illustrated in Figure 1.
|
time. CAP theorem is illustrated in Figure 1.
|
||||||
|
|
@ -33,7 +33,7 @@ image::Figure_1_guarantees.jpg[image]
|
||||||
|
|
||||||
Figure 1. CAP Theorem
|
Figure 1. CAP Theorem
|
||||||
|
|
||||||
High availability is a priority in web based applications and to this
|
High availability is a priority in web-based applications and to this
|
||||||
objective Cassandra chooses Availability and Partition Tolerance from
|
objective Cassandra chooses Availability and Partition Tolerance from
|
||||||
the CAP guarantees, compromising on data Consistency to some extent.
|
the CAP guarantees, compromising on data Consistency to some extent.
|
||||||
|
|
||||||
|
|
@ -47,19 +47,19 @@ Cassandra makes the following guarantees.
|
||||||
* Batched writes across multiple tables are guaranteed to succeed
|
* Batched writes across multiple tables are guaranteed to succeed
|
||||||
completely or not at all
|
completely or not at all
|
||||||
* Secondary indexes are guaranteed to be consistent with their local
|
* Secondary indexes are guaranteed to be consistent with their local
|
||||||
replicas data
|
replicas' data
|
||||||
|
|
||||||
== High Scalability
|
== High Scalability
|
||||||
|
|
||||||
Cassandra is a highly scalable storage system in which nodes may be
|
Cassandra is a highly scalable storage system in which nodes may be
|
||||||
added/removed as needed. Using gossip-based protocol a unified and
|
added/removed as needed. Using gossip-based protocol, a unified and
|
||||||
consistent membership list is kept at each node.
|
consistent membership list is kept at each node.
|
||||||
|
|
||||||
== High Availability
|
== High Availability
|
||||||
|
|
||||||
Cassandra guarantees high availability of data by implementing a
|
Cassandra guarantees high availability of data by implementing a
|
||||||
fault-tolerant storage system. Failure detection in a node is detected
|
fault-tolerant storage system. Failure of a node is detected using
|
||||||
using a gossip-based protocol.
|
a gossip-based protocol.
|
||||||
|
|
||||||
== Durability
|
== Durability
|
||||||
|
|
||||||
|
|
@ -67,26 +67,26 @@ Cassandra guarantees data durability by using replicas. Replicas are
|
||||||
multiple copies of a data stored on different nodes in a cluster. In a
|
multiple copies of a data stored on different nodes in a cluster. In a
|
||||||
multi-datacenter environment the replicas may be stored on different
|
multi-datacenter environment the replicas may be stored on different
|
||||||
datacenters. If one replica is lost due to unrecoverable node/datacenter
|
datacenters. If one replica is lost due to unrecoverable node/datacenter
|
||||||
failure the data is not completely lost as replicas are still available.
|
failure, the data is not completely lost, as replicas are still available.
|
||||||
|
|
||||||
== Eventual Consistency
|
== Eventual Consistency
|
||||||
|
|
||||||
Meeting the requirements of performance, reliability, scalability and
|
Meeting the requirements of performance, reliability, scalability and
|
||||||
high availability in production Cassandra is an eventually consistent
|
high availability in production, Cassandra is an eventually consistent
|
||||||
storage system. Eventually consistent implies that all updates reach all
|
storage system. Eventually consistency implies that all updates reach all
|
||||||
replicas eventually. Divergent versions of the same data may exist
|
replicas eventually. Divergent versions of the same data may exist
|
||||||
temporarily but they are eventually reconciled to a consistent state.
|
temporarily, but they are eventually reconciled to a consistent state.
|
||||||
Eventual consistency is a tradeoff to achieve high availability and it
|
Eventual consistency is a tradeoff to achieve high availability, and it
|
||||||
involves some read and write latencies.
|
involves some read and write latencies.
|
||||||
|
|
||||||
== Lightweight transactions with linearizable consistency
|
== Lightweight transactions with linearizable consistency
|
||||||
|
|
||||||
Data must be read and written in a sequential order. Paxos consensus
|
Data must be read and written in a sequential order. The Paxos consensus
|
||||||
protocol is used to implement lightweight transactions. Paxos protocol
|
protocol is used to implement lightweight transactions. The Paxos protocol
|
||||||
implements lightweight transactions that are able to handle concurrent
|
implements lightweight transactions that are able to handle concurrent
|
||||||
operations using linearizable consistency. Linearizable consistency is
|
operations using linearizable consistency. Linearizable consistency is
|
||||||
sequential consistency with real-time constraints and it ensures
|
sequential consistency with real-time constraints, and it ensures
|
||||||
transaction isolation with compare and set (CAS) transaction. With CAS
|
transaction isolation with compare-and-set (CAS) transactions. With CAS
|
||||||
replica data is compared and data that is found to be out of date is set
|
replica data is compared and data that is found to be out of date is set
|
||||||
to the most consistent value. Reads with linearizable consistency allow
|
to the most consistent value. Reads with linearizable consistency allow
|
||||||
reading the current state of the data, which may possibly be
|
reading the current state of the data, which may possibly be
|
||||||
|
|
@ -97,12 +97,12 @@ uncommitted, without making a new addition or update.
|
||||||
The guarantee for batched writes across multiple tables is that they
|
The guarantee for batched writes across multiple tables is that they
|
||||||
will eventually succeed, or none will. Batch data is first written to
|
will eventually succeed, or none will. Batch data is first written to
|
||||||
batchlog system data, and when the batch data has been successfully
|
batchlog system data, and when the batch data has been successfully
|
||||||
stored in the cluster the batchlog data is removed. The batch is
|
stored in the cluster, the batchlog data is removed. The batch is
|
||||||
replicated to another node to ensure the full batch completes in the
|
replicated to another node to ensure that the full batch completes in
|
||||||
event the coordinator node fails.
|
the event if coordinator node fails.
|
||||||
|
|
||||||
== Secondary Indexes
|
== Secondary Indexes
|
||||||
|
|
||||||
A secondary index is an index on a column and is used to query a table
|
A secondary index is an index on a column, and it's used to query a table
|
||||||
that is normally not queryable. Secondary indexes when built are
|
that is normally not queryable. Secondary indexes, when built, are
|
||||||
guaranteed to be consistent with their local replicas.
|
guaranteed to be consistent with their local replicas.
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
= Overview
|
= Overview
|
||||||
:exper: experimental
|
:exper: experimental
|
||||||
|
|
||||||
Apache Cassandra is an open source, distributed, NoSQL database. It
|
Apache Cassandra is an open-source, distributed, NoSQL database. It
|
||||||
presents a partitioned wide column storage model with eventually
|
presents a partitioned wide column storage model with eventually
|
||||||
consistent semantics.
|
consistent semantics.
|
||||||
|
|
||||||
|
|
@ -10,7 +10,7 @@ https://www.cs.cornell.edu/projects/ladis2009/papers/lakshman-ladis2009.pdf[Face
|
||||||
using a staged event-driven architecture
|
using a staged event-driven architecture
|
||||||
(http://www.sosp.org/2001/papers/welsh.pdf[SEDA]) to implement a
|
(http://www.sosp.org/2001/papers/welsh.pdf[SEDA]) to implement a
|
||||||
combination of Amazon’s
|
combination of Amazon’s
|
||||||
http://courses.cse.tamu.edu/caverlee/csce438/readings/dynamo-paper.pdf[Dynamo]
|
https://www.cs.cornell.edu/courses/cs5414/2017fa/papers/dynamo.pdf[Dynamo]
|
||||||
distributed storage and replication techniques and Google's
|
distributed storage and replication techniques and Google's
|
||||||
https://static.googleusercontent.com/media/research.google.com/en//archive/bigtable-osdi06.pdf[Bigtable]
|
https://static.googleusercontent.com/media/research.google.com/en//archive/bigtable-osdi06.pdf[Bigtable]
|
||||||
data and storage engine model. Dynamo and Bigtable were both developed
|
data and storage engine model. Dynamo and Bigtable were both developed
|
||||||
|
|
@ -23,7 +23,7 @@ storage requirements. As applications began to require full global
|
||||||
replication and always available low-latency reads and writes, it became
|
replication and always available low-latency reads and writes, it became
|
||||||
imperative to design a new kind of database model as the relational
|
imperative to design a new kind of database model as the relational
|
||||||
database systems of the time struggled to meet the new requirements of
|
database systems of the time struggled to meet the new requirements of
|
||||||
global scale applications.
|
global-scale applications.
|
||||||
|
|
||||||
Systems like Cassandra are designed for these challenges and seek the
|
Systems like Cassandra are designed for these challenges and seek the
|
||||||
following design objectives:
|
following design objectives:
|
||||||
|
|
@ -59,19 +59,19 @@ keys.
|
||||||
CQL supports numerous advanced features over a partitioned dataset such
|
CQL supports numerous advanced features over a partitioned dataset such
|
||||||
as:
|
as:
|
||||||
|
|
||||||
* Single partition lightweight transactions with atomic compare and set
|
* Single-partition lightweight transactions with atomic compare and set
|
||||||
semantics.
|
semantics
|
||||||
* User-defined types, functions and aggregates
|
* User-defined types, functions and aggregates
|
||||||
* Collection types including sets, maps, and lists.
|
* Collection types including sets, maps, and lists
|
||||||
* Local secondary indices
|
* Local secondary indices
|
||||||
* (Experimental) materialized views
|
* (Experimental) materialized views
|
||||||
|
|
||||||
Cassandra explicitly chooses not to implement operations that require
|
Cassandra explicitly chooses not to implement operations that require
|
||||||
cross partition coordination as they are typically slow and hard to
|
cross-partition coordination as they are typically slow and hard to
|
||||||
provide highly available global semantics. For example Cassandra does
|
provide highly available global semantics. For example Cassandra does
|
||||||
not support:
|
not support:
|
||||||
|
|
||||||
* Cross partition transactions
|
* Cross-partition transactions
|
||||||
* Distributed joins
|
* Distributed joins
|
||||||
* Foreign keys or referential integrity.
|
* Foreign keys or referential integrity.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -12,7 +12,7 @@ physical location).
|
||||||
|
|
||||||
== Dynamic snitching
|
== Dynamic snitching
|
||||||
|
|
||||||
The dynamic snitch monitor read latencies to avoid reading from hosts
|
The dynamic snitch monitors read latencies to avoid reading from hosts
|
||||||
that have slowed down. The dynamic snitch is configured with the
|
that have slowed down. The dynamic snitch is configured with the
|
||||||
following properties on `cassandra.yaml`:
|
following properties on `cassandra.yaml`:
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -3,17 +3,17 @@
|
||||||
[[commit-log]]
|
[[commit-log]]
|
||||||
== CommitLog
|
== CommitLog
|
||||||
|
|
||||||
Commitlogs are an append only log of all mutations local to a Cassandra
|
Commitlogs are an append-only log of all mutations local to a Cassandra
|
||||||
node. Any data written to Cassandra will first be written to a commit
|
node. Any data written to Cassandra will first be written to a commit
|
||||||
log before being written to a memtable. This provides durability in the
|
log before being written to a memtable. This provides durability in the
|
||||||
case of unexpected shutdown. On startup, any mutations in the commit log
|
case of unexpected shutdown. On startup, any mutations in the commit log
|
||||||
will be applied to memtables.
|
will be applied to memtables.
|
||||||
|
|
||||||
All mutations write optimized by storing in commitlog segments, reducing
|
All mutations are write-optimized by storing in commitlog segments, reducing
|
||||||
the number of seeks needed to write to disk. Commitlog Segments are
|
the number of seeks needed to write to disk. Commitlog segments are
|
||||||
limited by the `commitlog_segment_size` option, once the size is
|
limited by the `commitlog_segment_size` option. Once the size is
|
||||||
reached, a new commitlog segment is created. Commitlog segments can be
|
reached, a new commitlog segment is created. Commitlog segments can be
|
||||||
archived, deleted, or recycled once all its data has been flushed to
|
archived, deleted, or recycled once all the data has been flushed to
|
||||||
SSTables. Commitlog segments are truncated when Cassandra has written
|
SSTables. Commitlog segments are truncated when Cassandra has written
|
||||||
data older than a certain point to the SSTables. Running "nodetool
|
data older than a certain point to the SSTables. Running "nodetool
|
||||||
drain" before stopping Cassandra will write everything in the memtables
|
drain" before stopping Cassandra will write everything in the memtables
|
||||||
|
|
@ -22,19 +22,18 @@ to SSTables and remove the need to sync with the commitlogs on startup.
|
||||||
* `commitlog_segment_size`: The default size is 32MiB, which is
|
* `commitlog_segment_size`: The default size is 32MiB, which is
|
||||||
almost always fine, but if you are archiving commitlog segments (see
|
almost always fine, but if you are archiving commitlog segments (see
|
||||||
commitlog_archiving.properties), then you probably want a finer
|
commitlog_archiving.properties), then you probably want a finer
|
||||||
granularity of archiving; 8 or 16 MiB is reasonable. `commitlog_segment_size`
|
granularity of archiving; 8 or 16 MiB is reasonable.
|
||||||
also determines the default value of `max_mutation_size` in cassandra.yaml.
|
`commitlog_segment_size` also determines the default value of
|
||||||
By default, max_mutation_size is half the size of `commitlog_segment_size`.
|
`max_mutation_size` in `cassandra.yaml`. By default,
|
||||||
|
`max_mutation_size` is a half the size of `commitlog_segment_size`.
|
||||||
|
|
||||||
**NOTE: If `max_mutation_size` is set explicitly then
|
[NOTE]
|
||||||
|
.Note
|
||||||
|
====
|
||||||
|
If `max_mutation_size` is set explicitly then
|
||||||
`commitlog_segment_size` must be set to at least twice the size of
|
`commitlog_segment_size` must be set to at least twice the size of
|
||||||
`max_mutation_size`**.
|
`max_mutation_size`.
|
||||||
|
====
|
||||||
Commitlogs are an append only log of all mutations local to a Cassandra
|
|
||||||
node. Any data written to Cassandra will first be written to a commit
|
|
||||||
log before being written to a memtable. This provides durability in the
|
|
||||||
case of unexpected shutdown. On startup, any mutations in the commit log
|
|
||||||
will be applied.
|
|
||||||
|
|
||||||
* `commitlog_sync`: may be either _periodic_ or _batch_.
|
* `commitlog_sync`: may be either _periodic_ or _batch_.
|
||||||
** `batch`: In batch mode, Cassandra won’t ack writes until the commit
|
** `batch`: In batch mode, Cassandra won’t ack writes until the commit
|
||||||
|
|
@ -55,23 +54,27 @@ _Default Value:_ 10000ms
|
||||||
|
|
||||||
_Default Value:_ batch
|
_Default Value:_ batch
|
||||||
|
|
||||||
** NOTE: In the event of an unexpected shutdown, Cassandra can lose up
|
[NOTE]
|
||||||
|
.Note
|
||||||
|
====
|
||||||
|
In the event of an unexpected shutdown, Cassandra can lose up
|
||||||
to the sync period or more if the sync is delayed. If using "batch"
|
to the sync period or more if the sync is delayed. If using "batch"
|
||||||
mode, it is recommended to store commitlogs in a separate, dedicated
|
mode, it is recommended to store commitlogs in a separate, dedicated
|
||||||
device.*
|
device.
|
||||||
|
====
|
||||||
|
|
||||||
* `commitlog_directory`: This option is commented out by default When
|
* `commitlog_directory`: This option is commented out by default. When
|
||||||
running on magnetic HDD, this should be a separate spindle than the data
|
running on magnetic HDD, this should be a separate spindle than the data
|
||||||
directories. If not set, the default directory is
|
directories. If not set, the default directory is
|
||||||
$CASSANDRA_HOME/data/commitlog.
|
`$CASSANDRA_HOME/data/commitlog`.
|
||||||
|
|
||||||
_Default Value:_ /var/lib/cassandra/commitlog
|
_Default Value:_ `/var/lib/cassandra/commitlog`
|
||||||
|
|
||||||
* `commitlog_compression`: Compression to apply to the commitlog. If
|
* `commitlog_compression`: Compression to apply to the commitlog. If
|
||||||
omitted, the commit log will be written uncompressed. LZ4, Snappy,
|
omitted, the commit log will be written uncompressed. LZ4, Snappy,
|
||||||
Deflate and Zstd compressors are supported.
|
Deflate and Zstd compressors are supported.
|
||||||
|
|
||||||
(Default Value: (complex option):
|
_Default Value:_ (complex option):
|
||||||
|
|
||||||
[source, yaml]
|
[source, yaml]
|
||||||
----
|
----
|
||||||
|
|
@ -86,8 +89,8 @@ If space gets above this value, Cassandra will flush every dirty CF in
|
||||||
the oldest segment and remove it. So a small total commitlog space will
|
the oldest segment and remove it. So a small total commitlog space will
|
||||||
tend to cause more flush activity on less-active columnfamilies.
|
tend to cause more flush activity on less-active columnfamilies.
|
||||||
|
|
||||||
The default value is the smaller of 8192, and 1/4 of the total space of
|
The default value is the smallest between 8192 and 1/4 of the total
|
||||||
the commitlog volume.
|
space of the commitlog volume.
|
||||||
|
|
||||||
_Default Value:_ 8192MiB
|
_Default Value:_ 8192MiB
|
||||||
|
|
||||||
|
|
@ -202,7 +205,7 @@ we should not allow streaming of super columns into this new format)
|
||||||
** index summaries can be downsampled and the sampling level is
|
** index summaries can be downsampled and the sampling level is
|
||||||
persisted
|
persisted
|
||||||
** switch uncompressed checksums to adler32
|
** switch uncompressed checksums to adler32
|
||||||
** tracks presense of legacy (local and remote) counter shards
|
** tracks presence of legacy (local and remote) counter shards
|
||||||
* la (2.2.0): new file name format
|
* la (2.2.0): new file name format
|
||||||
* lb (2.2.7): commit log lower bound included
|
* lb (2.2.7): commit log lower bound included
|
||||||
|
|
||||||
|
|
@ -221,5 +224,5 @@ match the "ib" SSTable version
|
||||||
|
|
||||||
[source,bash]
|
[source,bash]
|
||||||
----
|
----
|
||||||
include:example$find_sstables.sh[]
|
include::example$BASH/find_sstables.sh[]
|
||||||
----
|
----
|
||||||
|
|
|
||||||
|
|
@ -4,9 +4,10 @@
|
||||||
CQL stores data in _tables_, whose schema defines the layout of the
|
CQL stores data in _tables_, whose schema defines the layout of the
|
||||||
data in the table. Tables are located in _keyspaces_.
|
data in the table. Tables are located in _keyspaces_.
|
||||||
A keyspace defines options that apply to all the keyspace's tables.
|
A keyspace defines options that apply to all the keyspace's tables.
|
||||||
The xref:cql/ddl.adoc#replication-strategy[replication strategy] is an important keyspace option, as is the replication factor.
|
The xref:cql/ddl.adoc#replication-strategy[replication strategy]
|
||||||
|
is an important keyspace option, as is the replication factor.
|
||||||
A good general rule is one keyspace per application.
|
A good general rule is one keyspace per application.
|
||||||
It is common for a cluster to define only one keyspace for an actie application.
|
It is common for a cluster to define only one keyspace for an active application.
|
||||||
|
|
||||||
This section describes the statements used to create, modify, and remove
|
This section describes the statements used to create, modify, and remove
|
||||||
those keyspace and tables.
|
those keyspace and tables.
|
||||||
|
|
@ -32,7 +33,7 @@ double-quotes (`"myTable"` is different from `mytable`).
|
||||||
Further, a table is always part of a keyspace and a table name can be
|
Further, a table is always part of a keyspace and a table name can be
|
||||||
provided fully-qualified by the keyspace it is part of. If is is not
|
provided fully-qualified by the keyspace it is part of. If is is not
|
||||||
fully-qualified, the table is assumed to be in the _current_ keyspace
|
fully-qualified, the table is assumed to be in the _current_ keyspace
|
||||||
(see xref:cql/ddl.adoc#use-statement[USE] statement.
|
(see xref:cql/ddl.adoc#use-statement[USE] statement).
|
||||||
|
|
||||||
Further, the valid names for columns are defined as:
|
Further, the valid names for columns are defined as:
|
||||||
|
|
||||||
|
|
@ -502,7 +503,7 @@ A table supports the following options:
|
||||||
| `bloom_filter_fp_chance` |_simple_ |0.00075 |The target probability of
|
| `bloom_filter_fp_chance` |_simple_ |0.00075 |The target probability of
|
||||||
false positive of the sstable bloom filters. Said bloom filters will be
|
false positive of the sstable bloom filters. Said bloom filters will be
|
||||||
sized to provide the provided probability, thus lowering this value
|
sized to provide the provided probability, thus lowering this value
|
||||||
impact the size of bloom filters in-memory and on-disk.
|
impacts the size of bloom filters in-memory and on-disk.
|
||||||
| `default_time_to_live` |_simple_ |0 |Default expiration time (“TTL”) in seconds for a table
|
| `default_time_to_live` |_simple_ |0 |Default expiration time (“TTL”) in seconds for a table
|
||||||
| `compaction` |_map_ |_see below_ | xref:operating/compaction/index.adoc#cql-compaction-options[Compaction options]
|
| `compaction` |_map_ |_see below_ | xref:operating/compaction/index.adoc#cql-compaction-options[Compaction options]
|
||||||
| `compression` |_map_ |_see below_ | xref:operating/compression/index.adoc#cql-compression-options[Compression options]
|
| `compression` |_map_ |_see below_ | xref:operating/compression/index.adoc#cql-compression-options[Compression options]
|
||||||
|
|
|
||||||
|
|
@ -103,7 +103,7 @@ however than float allows the special `NaN` and `Infinity` constants.
|
||||||
* CQL supports
|
* CQL supports
|
||||||
https://en.wikipedia.org/wiki/Universally_unique_identifier[UUID]
|
https://en.wikipedia.org/wiki/Universally_unique_identifier[UUID]
|
||||||
constants.
|
constants.
|
||||||
* Blobs content are provided in hexadecimal and prefixed by `0x`.
|
* The content for blobs is provided in hexadecimal and prefixed by `0x`.
|
||||||
* The special `NULL` constant denotes the absence of value.
|
* The special `NULL` constant denotes the absence of value.
|
||||||
|
|
||||||
For how these constants are typed, see the xref:cql/types.adoc[Data types] section.
|
For how these constants are typed, see the xref:cql/types.adoc[Data types] section.
|
||||||
|
|
|
||||||
|
|
@ -318,7 +318,7 @@ include::example$CQL/update_statement.cql[]
|
||||||
|
|
||||||
The `UPDATE` statement writes one or more columns for a given row in a
|
The `UPDATE` statement writes one or more columns for a given row in a
|
||||||
table.
|
table.
|
||||||
The `WHERE`clause is used to select the row to update and must include all columns of the `PRIMARY KEY`.
|
The `WHERE` clause is used to select the row to update and must include all columns of the `PRIMARY KEY`.
|
||||||
Non-primary key columns are set using the `SET` keyword.
|
Non-primary key columns are set using the `SET` keyword.
|
||||||
In an `UPDATE` statement, all updates within the same partition key are applied atomically and in isolation.
|
In an `UPDATE` statement, all updates within the same partition key are applied atomically and in isolation.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -222,7 +222,7 @@ collections have the following noteworthy characteristics and
|
||||||
limitations:
|
limitations:
|
||||||
|
|
||||||
* Individual collections are not indexed internally. Which means that
|
* Individual collections are not indexed internally. Which means that
|
||||||
even to access a single element of a collection, the while collection
|
even to access a single element of a collection, the whole collection
|
||||||
has to be read (and reading one is not paged internally).
|
has to be read (and reading one is not paged internally).
|
||||||
* While insertion operations on sets and maps never incur a
|
* While insertion operations on sets and maps never incur a
|
||||||
read-before-write internally, some operations on lists do. Further, some
|
read-before-write internally, some operations on lists do. Further, some
|
||||||
|
|
@ -265,7 +265,7 @@ Note that for removing multiple elements in a `map`, you remove from it
|
||||||
a `set` of keys.
|
a `set` of keys.
|
||||||
|
|
||||||
Lastly, TTLs are allowed for both `INSERT` and `UPDATE`, but in both
|
Lastly, TTLs are allowed for both `INSERT` and `UPDATE`, but in both
|
||||||
case the TTL set only apply to the newly inserted/updated elements. In
|
cases the TTL set only apply to the newly inserted/updated elements. In
|
||||||
other words:
|
other words:
|
||||||
|
|
||||||
[source,cql]
|
[source,cql]
|
||||||
|
|
@ -279,7 +279,7 @@ of the map remaining unaffected.
|
||||||
=== Sets
|
=== Sets
|
||||||
|
|
||||||
A `set` is a (sorted) collection of unique values. You can define and
|
A `set` is a (sorted) collection of unique values. You can define and
|
||||||
insert a map with:
|
insert a set with:
|
||||||
|
|
||||||
[source,cql]
|
[source,cql]
|
||||||
----
|
----
|
||||||
|
|
@ -317,7 +317,7 @@ xref:cql/types.adoc#sets[set] instead of list, always prefer a set.
|
||||||
====
|
====
|
||||||
|
|
||||||
A `list` is a (sorted) collection of non-unique values where
|
A `list` is a (sorted) collection of non-unique values where
|
||||||
elements are ordered by there position in the list. You can define and
|
elements are ordered by their position in the list. You can define and
|
||||||
insert a list with:
|
insert a list with:
|
||||||
|
|
||||||
[source,cql]
|
[source,cql]
|
||||||
|
|
@ -338,13 +338,13 @@ include::example$CQL/update_list.cql[]
|
||||||
.Warning
|
.Warning
|
||||||
====
|
====
|
||||||
The append and prepend operations are not idempotent by nature. So in
|
The append and prepend operations are not idempotent by nature. So in
|
||||||
particular, if one of these operation timeout, then retrying the
|
particular, if one of these operations times out, then retrying the
|
||||||
operation is not safe and it may (or may not) lead to
|
operation is not safe and it may (or may not) lead to
|
||||||
appending/prepending the value twice.
|
appending/prepending the value twice.
|
||||||
====
|
====
|
||||||
|
|
||||||
* Setting the value at a particular position in a list that has a pre-existing element for that position. An error
|
* Setting the value at a particular position in a list that has a pre-existing element for that position. An error
|
||||||
will be thrown if the list does not have the position.:
|
will be thrown if the list does not have the position:
|
||||||
+
|
+
|
||||||
[source,cql]
|
[source,cql]
|
||||||
----
|
----
|
||||||
|
|
@ -423,12 +423,11 @@ and can only be used in that keyspace. At creation, if the type name is
|
||||||
prefixed by a keyspace name, it is created in that keyspace. Otherwise,
|
prefixed by a keyspace name, it is created in that keyspace. Otherwise,
|
||||||
it is created in the current keyspace.
|
it is created in the current keyspace.
|
||||||
* As of Cassandra , UDT have to be frozen in most cases, hence the
|
* As of Cassandra , UDT have to be frozen in most cases, hence the
|
||||||
`frozen<address>` in the table definition above. Please see the section
|
`frozen<address>` in the table definition above.
|
||||||
on xref:cql/types.adoc#frozen[frozen] for more details.
|
|
||||||
|
|
||||||
=== UDT literals
|
=== UDT literals
|
||||||
|
|
||||||
Once a used-defined type has been created, value can be input using a
|
Once a user-defined type has been created, value can be input using a
|
||||||
UDT literal:
|
UDT literal:
|
||||||
|
|
||||||
[source,bnf]
|
[source,bnf]
|
||||||
|
|
|
||||||
|
|
@ -17,7 +17,7 @@ image::data_modeling_hotel_relational.png[image]
|
||||||
== Design Differences Between RDBMS and Cassandra
|
== Design Differences Between RDBMS and Cassandra
|
||||||
|
|
||||||
Let’s take a minute to highlight some of the key differences in doing
|
Let’s take a minute to highlight some of the key differences in doing
|
||||||
ata modeling for Cassandra versus a relational database.
|
data modeling for Cassandra versus a relational database.
|
||||||
|
|
||||||
=== No joins
|
=== No joins
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -32,7 +32,7 @@ CREATE TABLE hotel.hotels (
|
||||||
name text,
|
name text,
|
||||||
phone text,
|
phone text,
|
||||||
address frozen<address>,
|
address frozen<address>,
|
||||||
pois set )
|
pois set<text> )
|
||||||
WITH comment = ‘Q2. Find information about a hotel’;
|
WITH comment = ‘Q2. Find information about a hotel’;
|
||||||
|
|
||||||
CREATE TABLE hotel.pois_by_hotel (
|
CREATE TABLE hotel.pois_by_hotel (
|
||||||
|
|
|
||||||
|
|
@ -35,7 +35,7 @@ management and query execution.
|
||||||
Some IDEs and tools that claim to support Cassandra do not actually
|
Some IDEs and tools that claim to support Cassandra do not actually
|
||||||
support CQL natively, but instead access Cassandra using a JDBC/ODBC
|
support CQL natively, but instead access Cassandra using a JDBC/ODBC
|
||||||
driver and interact with Cassandra as if it were a relational database
|
driver and interact with Cassandra as if it were a relational database
|
||||||
with SQL support. Wnen selecting tools for working with Cassandra you’ll
|
with SQL support. When selecting tools for working with Cassandra you’ll
|
||||||
want to make sure they support CQL and reinforce Cassandra best
|
want to make sure they support CQL and reinforce Cassandra best
|
||||||
practices for data modeling as presented in this documentation.
|
practices for data modeling as presented in this documentation.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -79,7 +79,10 @@ intensive process that may result in adverse cluster performance. It's
|
||||||
highly recommended to do rolling repairs, as an attempt to repair the
|
highly recommended to do rolling repairs, as an attempt to repair the
|
||||||
entire cluster at once will most likely swamp it. Note that you will
|
entire cluster at once will most likely swamp it. Note that you will
|
||||||
need to run a full repair (`-full`) to make sure that already repaired
|
need to run a full repair (`-full`) to make sure that already repaired
|
||||||
sstables are not skipped.
|
sstables are not skipped. You should use `ConsistencyLevel.QUORUM` or
|
||||||
|
`ALL` (depending on your existing replication factor) to make sure that
|
||||||
|
a replica that actually has the data is consulted. Otherwise some
|
||||||
|
clients potentially being told no data exists until repair is done.
|
||||||
|
|
||||||
[[can-large-blob]]
|
[[can-large-blob]]
|
||||||
== Can I Store (large) BLOBs in Cassandra?
|
== Can I Store (large) BLOBs in Cassandra?
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ functionality supported by a specific driver.
|
||||||
* https://github.com/Netflix/astyanax/wiki/Getting-Started[Astyanax]
|
* https://github.com/Netflix/astyanax/wiki/Getting-Started[Astyanax]
|
||||||
* https://github.com/noorq/casser[Casser]
|
* https://github.com/noorq/casser[Casser]
|
||||||
* https://github.com/datastax/java-driver[Datastax Java driver]
|
* https://github.com/datastax/java-driver[Datastax Java driver]
|
||||||
* https://github.com/impetus-opensource/Kundera[Kundera]
|
* https://github.com/Impetus/kundera[Kundera]
|
||||||
* https://github.com/deanhiller/playorm[PlayORM]
|
* https://github.com/deanhiller/playorm[PlayORM]
|
||||||
|
|
||||||
== Python
|
== Python
|
||||||
|
|
|
||||||
|
|
@ -51,7 +51,7 @@ appropriate number of replicates, to ensure even token allocation.
|
||||||
Read ahead is an operating system feature that attempts to keep as much
|
Read ahead is an operating system feature that attempts to keep as much
|
||||||
data as possible loaded in the page cache.
|
data as possible loaded in the page cache.
|
||||||
Spinning disks can have long seek times causing high latency, so additional
|
Spinning disks can have long seek times causing high latency, so additional
|
||||||
throughout on reads using page cache can improve performance.
|
throughput on reads using page cache can improve performance.
|
||||||
By leveraging read ahead, the OS can pull additional data into memory without
|
By leveraging read ahead, the OS can pull additional data into memory without
|
||||||
the cost of additional seeks.
|
the cost of additional seeks.
|
||||||
This method works well when the available RAM is greater than the size of the
|
This method works well when the available RAM is greater than the size of the
|
||||||
|
|
@ -80,7 +80,7 @@ The recommended read ahead settings are:
|
||||||
|
|
||||||
Read ahead can be adjusted on Linux systems using the `blockdev` tool.
|
Read ahead can be adjusted on Linux systems using the `blockdev` tool.
|
||||||
|
|
||||||
For example, set the read ahead of the disk `/dev/sda1\` to 4KB:
|
For example, set the read ahead of the disk `/dev/sda1` to 4KB:
|
||||||
|
|
||||||
[source, shell]
|
[source, shell]
|
||||||
----
|
----
|
||||||
|
|
@ -100,7 +100,7 @@ section.
|
||||||
|
|
||||||
== Compression
|
== Compression
|
||||||
|
|
||||||
Compressed data is stored by compressing fixed size byte buffers and writing the
|
Compressed data is stored by compressing fixed-size byte buffers and writing the
|
||||||
data to disk.
|
data to disk.
|
||||||
The buffer size is determined by the `chunk_length_in_kb` element in the compression
|
The buffer size is determined by the `chunk_length_in_kb` element in the compression
|
||||||
map of a table's schema settings for `WITH COMPRESSION`.
|
map of a table's schema settings for `WITH COMPRESSION`.
|
||||||
|
|
@ -158,6 +158,6 @@ of the ability to configure multiple racks and data centers.
|
||||||
**Correctly configuring or changing racks after a cluster has been provisioned is an unsupported process**.
|
**Correctly configuring or changing racks after a cluster has been provisioned is an unsupported process**.
|
||||||
Migrating from a single rack to multiple racks is also unsupported and can
|
Migrating from a single rack to multiple racks is also unsupported and can
|
||||||
result in data loss.
|
result in data loss.
|
||||||
Using `GossipingPropertyFileSnitch` is the most flexible solution for on
|
Using `GossipingPropertyFileSnitch` is the most flexible solution for
|
||||||
premise or mixed cloud environments.
|
on-premise or mixed cloud environments.
|
||||||
`Ec2Snitch` is reliable for AWS EC2 only environments.
|
`Ec2Snitch` is reliable for AWS EC2 only environments.
|
||||||
|
|
|
||||||
|
|
@ -12,7 +12,7 @@ Some of the features of audit logging are:
|
||||||
* Latency of database operations is not affected, so there is no performance impact.
|
* Latency of database operations is not affected, so there is no performance impact.
|
||||||
* Heap memory usage is bounded by a weighted queue, with configurable maximum weight sitting in front of logging thread.
|
* Heap memory usage is bounded by a weighted queue, with configurable maximum weight sitting in front of logging thread.
|
||||||
* Disk utilization is bounded by a configurable size, deleting old log segments once the limit is reached.
|
* Disk utilization is bounded by a configurable size, deleting old log segments once the limit is reached.
|
||||||
* Can be enabled, disabled, or reset (to delete on-disk data) using the JMX tool, ``nodetool``.
|
* Can be enabled or disabled at startup time using `cassandra.yaml` or at runtime using the JMX tool, ``nodetool``.
|
||||||
* Can configure the settings in either the `cassandra.yaml` file or by using ``nodetool``.
|
* Can configure the settings in either the `cassandra.yaml` file or by using ``nodetool``.
|
||||||
|
|
||||||
Audit logging includes all CQL requests, both successful and failed.
|
Audit logging includes all CQL requests, both successful and failed.
|
||||||
|
|
@ -88,10 +88,24 @@ Common audit log entry types are one of the following:
|
||||||
| ERROR | REQUEST_FAILURE
|
| ERROR | REQUEST_FAILURE
|
||||||
|===
|
|===
|
||||||
|
|
||||||
|
== Availability and durability
|
||||||
|
|
||||||
|
NOTE: Unlike data, audit log entries are not replicated
|
||||||
|
|
||||||
|
For a given query, the corresponding audit entry is only stored on the coordinator node.
|
||||||
|
For example, an ``INSERT`` in a keyspace with replication factor of 3 will produce an audit entry on one node, the coordinator who handled the request, and not on the two other nodes.
|
||||||
|
For this reason, and depending on compliance requirements you must meet,
|
||||||
|
make sure that audit logs are stored on a non-ephemeral storage.
|
||||||
|
|
||||||
|
You can achieve custom needs with the <<archive_command>> option.
|
||||||
|
|
||||||
== Configuring audit logging in cassandra.yaml
|
== Configuring audit logging in cassandra.yaml
|
||||||
|
|
||||||
The `cassandra.yaml` file can be used to configure and enable audit logging.
|
The `cassandra.yaml` file can be used to configure and enable audit logging.
|
||||||
Configuration and enablement may be the same or different on each node, depending on the `cassandra.yaml` file settings.
|
Configuration and enablement may be the same or different on each node, depending on the `cassandra.yaml` file settings.
|
||||||
|
|
||||||
|
Audit logging can also be configured using ``nodetool`` when enabling the feature, and will override any values set in the `cassandra.yaml` file, as discussed in <<enabling_audit_with_nodetool, Enabling Audit Logging with nodetool>>.
|
||||||
|
|
||||||
Audit logs are generated on each enabled node, so logs on each node will have that node's queries.
|
Audit logs are generated on each enabled node, so logs on each node will have that node's queries.
|
||||||
All options for audit logging can be set in the `cassandra.yaml` file under the ``audit_logging_options:``.
|
All options for audit logging can be set in the `cassandra.yaml` file under the ``audit_logging_options:``.
|
||||||
|
|
||||||
|
|
@ -123,16 +137,25 @@ audit_logging_options:
|
||||||
|
|
||||||
=== enabled
|
=== enabled
|
||||||
|
|
||||||
Audit logging is enabled by setting the `enabled` option to `true` in
|
Control whether audit logging is enabled or disabled (default).
|
||||||
the `audit_logging_options` setting.
|
|
||||||
|
To enable audit logging set ``enabled: true``.
|
||||||
|
|
||||||
If this option is enabled, audit logging will start when Cassandra is started.
|
If this option is enabled, audit logging will start when Cassandra is started.
|
||||||
For example, ``enabled: true``.
|
It can be disabled afterwards at runtime with <<enabling_audit_with_nodetool, nodetool>>.
|
||||||
|
|
||||||
|
TIP: You can monitor whether audit logging is enabled with ``AuditLogEnabled`` attribute of the JMX MBean ``org.apache.cassandra.db:type=StorageService``.
|
||||||
|
|
||||||
=== logger
|
=== logger
|
||||||
|
|
||||||
The type of audit logger is set with the `logger` option.
|
The type of audit logger is set with the `logger` option.
|
||||||
Supported values are: `BinAuditLogger` (default), `FileAuditLogger` and `NoOpAuditLogger`.
|
Supported values are:
|
||||||
`BinAuditLogger` logs events to a file in binary format.
|
|
||||||
|
- `BinAuditLogger` (default)
|
||||||
|
- `FileAuditLogger`
|
||||||
|
- `NoOpAuditLogger`
|
||||||
|
|
||||||
|
`BinAuditLogger` logs events to a file in binary format.
|
||||||
`FileAuditLogger` uses the standard logging mechanism, `slf4j` to log events to the `audit/audit.log` file. It is a synchronous, file-based audit logger. The roll_cycle will be set in the `logback.xml` file.
|
`FileAuditLogger` uses the standard logging mechanism, `slf4j` to log events to the `audit/audit.log` file. It is a synchronous, file-based audit logger. The roll_cycle will be set in the `logback.xml` file.
|
||||||
`NoOpAuditLogger` is a no-op implementation of the audit logger that shoudl be specified when audit logging is disabled.
|
`NoOpAuditLogger` is a no-op implementation of the audit logger that shoudl be specified when audit logging is disabled.
|
||||||
|
|
||||||
|
|
@ -144,6 +167,8 @@ logger:
|
||||||
- class_name: FileAuditLogger
|
- class_name: FileAuditLogger
|
||||||
----
|
----
|
||||||
|
|
||||||
|
TIP: `BinAuditLogger` make use of open source https://github.com/OpenHFT/Chronicle-Queue[Chronicle Queue] under the hood. If you consider using audit logging for regulatory compliance purpose, it might be wise to be somewhat familiar with this library. See <<archive_command>> and <<roll_cycle>> for an example of the implications.
|
||||||
|
|
||||||
=== audit_logs_dir
|
=== audit_logs_dir
|
||||||
|
|
||||||
To write audit logs, an existing directory must be set in ``audit_logs_dir``.
|
To write audit logs, an existing directory must be set in ``audit_logs_dir``.
|
||||||
|
|
@ -151,7 +176,7 @@ To write audit logs, an existing directory must be set in ``audit_logs_dir``.
|
||||||
The directory must have appropriate permissions set to allow reading, writing, and executing.
|
The directory must have appropriate permissions set to allow reading, writing, and executing.
|
||||||
Logging will recursively delete the directory contents as needed.
|
Logging will recursively delete the directory contents as needed.
|
||||||
Do not place links in this directory to other sections of the filesystem.
|
Do not place links in this directory to other sections of the filesystem.
|
||||||
For example, ``audit_logs_dir: /cassandra/audit/logs/hourly``.
|
For example, ``audit_logs_dir: /non_ephemeral_storage/audit/logs/hourly``.
|
||||||
|
|
||||||
The audit log directory can also be configured using the system property `cassandra.logdir.audit`, which by default is set to `cassandra.logdir + /audit/`.
|
The audit log directory can also be configured using the system property `cassandra.logdir.audit`, which by default is set to `cassandra.logdir + /audit/`.
|
||||||
|
|
||||||
|
|
@ -173,7 +198,7 @@ excluded_keyspaces: system, system_schema, system_virtual_schema
|
||||||
The categories of database operations to include are specified with the `included_categories` option as a comma-separated list.
|
The categories of database operations to include are specified with the `included_categories` option as a comma-separated list.
|
||||||
The categories of database operations to exclude are specified with `excluded_categories` option as a comma-separated list.
|
The categories of database operations to exclude are specified with `excluded_categories` option as a comma-separated list.
|
||||||
The supported categories for audit log are: `AUTH`, `DCL`, `DDL`, `DML`, `ERROR`, `OTHER`, `PREPARE`, and `QUERY`.
|
The supported categories for audit log are: `AUTH`, `DCL`, `DDL`, `DML`, `ERROR`, `OTHER`, `PREPARE`, and `QUERY`.
|
||||||
By default all supported categories are included, and no category is excluded.
|
By default, all supported categories are included, and no category is excluded.
|
||||||
|
|
||||||
[source, yaml]
|
[source, yaml]
|
||||||
----
|
----
|
||||||
|
|
@ -186,7 +211,7 @@ excluded_categories: DDL, DML, QUERY, PREPARE
|
||||||
Users to audit log are set with the `included_users` and `excluded_users` options.
|
Users to audit log are set with the `included_users` and `excluded_users` options.
|
||||||
The `included_users` option specifies a comma-separated list of users to include explicitly.
|
The `included_users` option specifies a comma-separated list of users to include explicitly.
|
||||||
The `excluded_users` option specifies a comma-separated list of users to exclude explicitly.
|
The `excluded_users` option specifies a comma-separated list of users to exclude explicitly.
|
||||||
By default all users are included, and no users are excluded.
|
By default, all users are included, and no users are excluded.
|
||||||
|
|
||||||
[source, yaml]
|
[source, yaml]
|
||||||
----
|
----
|
||||||
|
|
@ -194,20 +219,55 @@ included_users:
|
||||||
excluded_users: john, mary
|
excluded_users: john, mary
|
||||||
----
|
----
|
||||||
|
|
||||||
|
[[roll_cycle]]
|
||||||
=== roll_cycle
|
=== roll_cycle
|
||||||
|
|
||||||
The ``roll_cycle`` defines the frequency with which the audit log segments are rolled.
|
The ``roll_cycle`` defines the frequency with which the audit log segments are rolled.
|
||||||
Supported values are ``HOURLY`` (default), ``MINUTELY``, and ``DAILY``.
|
Supported values are:
|
||||||
|
|
||||||
|
- ``MINUTELY``
|
||||||
|
- ``FIVE_MINUTELY``
|
||||||
|
- ``TEN_MINUTELY``
|
||||||
|
- ``TWENTY_MINUTELY``
|
||||||
|
- ``HALF_HOURLY``
|
||||||
|
- ``HOURLY`` (default)
|
||||||
|
- ``TWO_HOURLY``
|
||||||
|
- ``FOUR_HOURLY``
|
||||||
|
- ``SIX_HOURLY``
|
||||||
|
- ``DAILY``
|
||||||
|
|
||||||
For example: ``roll_cycle: DAILY``
|
For example: ``roll_cycle: DAILY``
|
||||||
|
|
||||||
|
WARNING: Read the following paragraph when changing ``roll_cycle`` on a production node.
|
||||||
|
|
||||||
|
With the `BinLogger` implementation, any attempt to modify the roll cycle on a node where audit logging was previously enabled will fail silentely due to https://github.com/OpenHFT/Chronicle-Queue[Chronicle Queue] roll cycle inference mechanism (even if you delete the ``metadata.cq4t`` file).
|
||||||
|
|
||||||
|
Here is an example of such an override visible in Cassandra logs:
|
||||||
|
----
|
||||||
|
INFO [main] <DATE TIME> BinLog.java:420 - Attempting to configure bin log: Path: /path/to/audit Roll cycle: TWO_HOURLY [...]
|
||||||
|
WARN [main] <DATE TIME> SingleChronicleQueueBuilder.java:477 - Overriding roll cycle from TWO_HOURLY to FIVE_MINUTE
|
||||||
|
----
|
||||||
|
|
||||||
|
In order to change ``roll_cycle`` on a node, you have to:
|
||||||
|
|
||||||
|
1. Stop Cassandra
|
||||||
|
2. Move or offload all audit logs somewhere else (in a safe and durable location)
|
||||||
|
3. Restart Cassandra.
|
||||||
|
4. Check Cassandra logs
|
||||||
|
5. Make sure that audit log filenames under ``audit_logs_dir`` correspond to the new roll cycle.
|
||||||
|
|
||||||
=== block
|
=== block
|
||||||
|
|
||||||
The ``block`` option specifies whether audit logging should block writing or drop log records if the audit logging falls behind. Supported boolean values are ``true`` (default) or ``false``.
|
The ``block`` option specifies whether audit logging should block writing or drop log records if the audit logging falls behind. Supported boolean values are ``true`` (default) or ``false``.
|
||||||
For example: ``block: false`` to drop records
|
|
||||||
|
For example: ``block: false`` to drop records (e.g. if audit is used for troobleshooting)
|
||||||
|
|
||||||
|
For regulatory compliance purposes, it's a good practice to explicitly set ``block: true`` to prevent any regression in case of future default value change.
|
||||||
|
|
||||||
=== max_queue_weight
|
=== max_queue_weight
|
||||||
|
|
||||||
The ``max_queue_weight`` option sets the maximum weight of in-memory queue for records waiting to be written to the file before blocking or dropping. The option must be set to a positive value. The default value is 268435456, or 256 MiB.
|
The ``max_queue_weight`` option sets the maximum weight of in-memory queue for records waiting to be written to the file before blocking or dropping. The option must be set to a positive value. The default value is 268435456, or 256 MiB.
|
||||||
|
|
||||||
For example, to change the default: ``max_queue_weight: 134217728 # 128 MiB``
|
For example, to change the default: ``max_queue_weight: 134217728 # 128 MiB``
|
||||||
|
|
||||||
=== max_log_size
|
=== max_log_size
|
||||||
|
|
@ -215,23 +275,37 @@ For example, to change the default: ``max_queue_weight: 134217728 # 128 MiB``
|
||||||
The ``max_log_size`` option sets the maximum size of the rolled files to retain on disk before deleting the oldest file. The option must be set to a positive value. The default is 17179869184, or 16 GiB.
|
The ``max_log_size`` option sets the maximum size of the rolled files to retain on disk before deleting the oldest file. The option must be set to a positive value. The default is 17179869184, or 16 GiB.
|
||||||
For example, to change the default: ``max_log_size: 34359738368 # 32 GiB``
|
For example, to change the default: ``max_log_size: 34359738368 # 32 GiB``
|
||||||
|
|
||||||
|
WARNING: ``max_log_size`` is ignored if ``archive_command`` option is set.
|
||||||
|
|
||||||
|
[[archive_command]]
|
||||||
=== archive_command
|
=== archive_command
|
||||||
|
|
||||||
|
NOTE: If ``archive_command`` option is empty or unset (default), Cassandra uses a built-in DeletingArchiver that deletes the oldest files if ``max_log_size`` is reached.
|
||||||
|
|
||||||
The ``archive_command`` option sets the user-defined archive script to execute on rolled log files.
|
The ``archive_command`` option sets the user-defined archive script to execute on rolled log files.
|
||||||
For example: ``archive_command: /usr/local/bin/archiveit.sh %path # %path is the file being rolled``
|
For example: ``archive_command: "/usr/local/bin/archiveit.sh %path"``
|
||||||
|
|
||||||
|
``%path`` is replaced with the absolute file path of the file being rolled.
|
||||||
|
|
||||||
|
When using a user-defined script, Cassandra does **not** use the DeletingArchiver, so it's the responsibility of the script to make any required cleanup.
|
||||||
|
|
||||||
|
Cassandra will call the user-defined script as soon as the log file is rolled. It means that Chronicle Queue's QueueFileShrinkManager will not be able to shrink the sparse log file because it's done asynchronously. In other words, all log files will have at least the size of the default block size (80 MiB), even if there are only a few KB of real data. Consequently, some warnings will appear in Cassandra system.log:
|
||||||
|
|
||||||
|
----
|
||||||
|
WARN [main/queue~file~shrink~daemon] <DATE TIME> QueueFileShrinkManager.java:63 - Failed to shrink file as it exists no longer, file=/path/to/xxx.cq4
|
||||||
|
----
|
||||||
|
|
||||||
|
TIP: Because Cassandra does not make use of Pretoucher, you can configure Chronicle Queue to shrink files synchronously -- i.e. as soon as the file is rolled -- with ``chronicle.queue.synchronousFileShrinking`` JVM properties. For instance, you can add the following line at the end of ``cassandra-env.sh``: ``JVM_OPTS="$JVM_OPTS -Dchronicle.queue.synchronousFileShrinking=true"``
|
||||||
|
|
||||||
=== max_archive_retries
|
=== max_archive_retries
|
||||||
|
|
||||||
The ``max_archive_retries`` option sets the max number of retries of failed archive commands. The default is 10.
|
The ``max_archive_retries`` option sets the max number of retries of failed archive commands. The default is 10.
|
||||||
|
|
||||||
For example: ``max_archive_retries: 10``
|
For example: ``max_archive_retries: 10``
|
||||||
|
|
||||||
|
Interval between each retry is hard coded to 5 minutes.
|
||||||
|
|
||||||
An audit log file could get rolled for other reasons as well such as a
|
[[enabling_audit_with_nodetool]]
|
||||||
log file reaches the configured size threshold.
|
|
||||||
|
|
||||||
Audit logging can also be configured using ``nodetool` when enabling the feature, and will override any values set in the `cassandra.yaml` file, as discussed in the next section.
|
|
||||||
|
|
||||||
|
|
||||||
== Enabling Audit Logging with ``nodetool``
|
== Enabling Audit Logging with ``nodetool``
|
||||||
|
|
||||||
Audit logging is enabled on a per-node basis using the ``nodetool enableauditlog`` command. The logging directory must be defined with ``audit_logs_dir`` in the `cassandra.yaml` file or uses the default value ``cassandra.logdir.audit``.
|
Audit logging is enabled on a per-node basis using the ``nodetool enableauditlog`` command. The logging directory must be defined with ``audit_logs_dir`` in the `cassandra.yaml` file or uses the default value ``cassandra.logdir.audit``.
|
||||||
|
|
|
||||||
|
|
@ -508,7 +508,7 @@ The two main tools/commands for restoring a table after it has been
|
||||||
dropped are:
|
dropped are:
|
||||||
|
|
||||||
* sstableloader
|
* sstableloader
|
||||||
* nodetool import
|
* nodetool refresh
|
||||||
|
|
||||||
A snapshot contains essentially the same set of SSTable files as an
|
A snapshot contains essentially the same set of SSTable files as an
|
||||||
incremental backup does with a few additional files. A snapshot includes
|
incremental backup does with a few additional files. A snapshot includes
|
||||||
|
|
|
||||||
|
|
@ -38,7 +38,7 @@ modules that are central to the performance of `COPY`.
|
||||||
== cqlshrc
|
== cqlshrc
|
||||||
|
|
||||||
The `cqlshrc` file holds configuration options for `cqlsh`.
|
The `cqlshrc` file holds configuration options for `cqlsh`.
|
||||||
By default, the file is locagted the user's home directory at `~/.cassandra/cqlsh`, but a
|
By default, the file is located the user's home directory at `~/.cassandra/cqlshrc`, but a
|
||||||
custom location can be specified with the `--cqlshrc` option.
|
custom location can be specified with the `--cqlshrc` option.
|
||||||
|
|
||||||
Example config values and documentation can be found in the
|
Example config values and documentation can be found in the
|
||||||
|
|
@ -452,7 +452,7 @@ representing a path to the source file. This can also the special value
|
||||||
See `shared-copy-options` for options that apply to both `COPY TO` and
|
See `shared-copy-options` for options that apply to both `COPY TO` and
|
||||||
`COPY FROM`.
|
`COPY FROM`.
|
||||||
|
|
||||||
==== Options for `COPY TO`
|
==== Options for `COPY FROM`
|
||||||
|
|
||||||
`INGESTRATE`::
|
`INGESTRATE`::
|
||||||
The maximum number of rows to process per second. Defaults to 100000.
|
The maximum number of rows to process per second. Defaults to 100000.
|
||||||
|
|
@ -477,10 +477,10 @@ See `shared-copy-options` for options that apply to both `COPY TO` and
|
||||||
`MAXBATCHSIZE`::
|
`MAXBATCHSIZE`::
|
||||||
The max number of rows inserted in a single batch. Defaults to 20.
|
The max number of rows inserted in a single batch. Defaults to 20.
|
||||||
`MINBATCHSIZE`::
|
`MINBATCHSIZE`::
|
||||||
The min number of rows inserted in a single batch. Defaults to 2.
|
The min number of rows inserted in a single batch. Defaults to 10.
|
||||||
`CHUNKSIZE`::
|
`CHUNKSIZE`::
|
||||||
The number of rows that are passed to child worker processes from the
|
The number of rows that are passed to child worker processes from the
|
||||||
main process at a time. Defaults to 1000.
|
main process at a time. Defaults to 5000.
|
||||||
|
|
||||||
==== Shared COPY Options
|
==== Shared COPY Options
|
||||||
|
|
||||||
|
|
@ -504,8 +504,8 @@ Options that are common to both `COPY TO` and `COPY FROM`.
|
||||||
`True,False`.
|
`True,False`.
|
||||||
`NUMPROCESSES`::
|
`NUMPROCESSES`::
|
||||||
The number of child worker processes to create for `COPY` tasks.
|
The number of child worker processes to create for `COPY` tasks.
|
||||||
Defaults to a max of 4 for `COPY FROM` and 16 for `COPY TO`. However,
|
Defaults to 16 for `COPY` tasks. However, at most (num_cores - 1)
|
||||||
at most (num_cores - 1) processes will be created.
|
processes will be created.
|
||||||
`MAXATTEMPTS`::
|
`MAXATTEMPTS`::
|
||||||
The maximum number of failed attempts to fetch a range of data (when
|
The maximum number of failed attempts to fetch a range of data (when
|
||||||
using `COPY TO`) or insert a chunk of data (when using `COPY FROM`)
|
using `COPY TO`) or insert a chunk of data (when using `COPY FROM`)
|
||||||
|
|
@ -515,3 +515,28 @@ Options that are common to both `COPY TO` and `COPY FROM`.
|
||||||
`RATEFILE`::
|
`RATEFILE`::
|
||||||
An optional file to output rate statistics to. By default, statistics
|
An optional file to output rate statistics to. By default, statistics
|
||||||
are not output to a file.
|
are not output to a file.
|
||||||
|
|
||||||
|
== Escaping Quotes
|
||||||
|
|
||||||
|
Dates, IP addresses, and strings need to be enclosed in single quotation marks. To use a single quotation mark itself in a string literal, escape it using a single quotation mark.
|
||||||
|
|
||||||
|
When fetching simple text data, `cqlsh` will return an unquoted string. However, when fetching text data from complex types (collections, user-defined types, etc.) `cqlsh` will return a quoted string containing the escaped characters. For example:
|
||||||
|
|
||||||
|
Simple data
|
||||||
|
[source,none]
|
||||||
|
----
|
||||||
|
cqlsh> CREATE TABLE test.simple_data (id int, data text, PRIMARY KEY (id));
|
||||||
|
cqlsh> INSERT INTO test.simple_data (id, data) values(1, 'I''m fine');
|
||||||
|
cqlsh> SELECT data from test.simple_data; data
|
||||||
|
----------
|
||||||
|
I'm fine
|
||||||
|
----
|
||||||
|
Complex data
|
||||||
|
[source,none]
|
||||||
|
----
|
||||||
|
cqlsh> CREATE TABLE test.complex_data (id int, data map<int, text>, PRIMARY KEY (id));
|
||||||
|
cqlsh> INSERT INTO test.complex_data (id, data) values(1, {1:'I''m fine'});
|
||||||
|
cqlsh> SELECT data from test.complex_data; data
|
||||||
|
------------------
|
||||||
|
{1: 'I''m fine'}
|
||||||
|
----
|
||||||
|
|
|
||||||
|
|
@ -6,7 +6,7 @@ for example, change the minimum sstable size, and therefore restart the
|
||||||
compaction process using this new configuration.
|
compaction process using this new configuration.
|
||||||
|
|
||||||
See
|
See
|
||||||
http://cassandra.apache.org/doc/latest/operating/compaction.html#leveled-compaction-strategy
|
https://cassandra.apache.org/doc/latest/operating/compaction/lcs.html#lcs
|
||||||
for information on how levels are used in this compaction strategy.
|
for information on how levels are used in this compaction strategy.
|
||||||
|
|
||||||
Cassandra must be stopped before this tool is executed, or unexpected
|
Cassandra must be stopped before this tool is executed, or unexpected
|
||||||
|
|
|
||||||
|
|
@ -85,7 +85,7 @@ Example: -1 is a leaf cell with content `contentArray[0]`.
|
||||||
|
|
||||||
Chain nodes are one-child nodes. Multiple chain nodes, forming a chain of transitions to one target, can reside in a
|
Chain nodes are one-child nodes. Multiple chain nodes, forming a chain of transitions to one target, can reside in a
|
||||||
single cell. Chain nodes are identified by the lowest 5 bits of a pointer being between `0x00` and `0x1B`. In addition
|
single cell. Chain nodes are identified by the lowest 5 bits of a pointer being between `0x00` and `0x1B`. In addition
|
||||||
to the the type of node, in this case the bits also define the length of the chain — the difference between
|
to the type of node, in this case the bits also define the length of the chain — the difference between
|
||||||
`0x1C` and the pointer offset specifies the number of characters in the chain.
|
`0x1C` and the pointer offset specifies the number of characters in the chain.
|
||||||
|
|
||||||
The simplest chain node has one transition leading to one child and is laid out like this:
|
The simplest chain node has one transition leading to one child and is laid out like this:
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue