Merge branch 'main' into vv

Fix Conflicts:
	flow/error_definitions.h
This commit is contained in:
Jingyu Zhou 2022-04-01 21:49:24 -07:00
commit 64d4658034
49 changed files with 2081 additions and 382 deletions

View File

@ -949,12 +949,10 @@ std::map<std::string, std::string> fillInRecords(int n) {
return data;
}
GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transaction& tr) {
GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transaction& tr, std::string mapper) {
std::string indexEntryKeyBegin = indexEntryKey(beginId);
std::string indexEntryKeyEnd = indexEntryKey(endId);
std::string mapper = Tuple().append(prefix).append(RECORD).append("{K[3]}"_sr).append("{...}"_sr).pack().toString();
return get_mapped_range(
tr,
FDB_KEYSEL_FIRST_GREATER_OR_EQUAL((const uint8_t*)indexEntryKeyBegin.c_str(), indexEntryKeyBegin.size()),
@ -969,6 +967,11 @@ GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transact
/* reverse */ 0);
}
GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transaction& tr) {
std::string mapper = Tuple().append(prefix).append(RECORD).append("{K[3]}"_sr).append("{...}"_sr).pack().toString();
return getMappedIndexEntries(beginId, endId, tr, mapper);
}
TEST_CASE("fdb_transaction_get_mapped_range") {
const int TOTAL_RECORDS = 20;
fillInRecords(TOTAL_RECORDS);
@ -1009,7 +1012,6 @@ TEST_CASE("fdb_transaction_get_mapped_range") {
TEST_CASE("fdb_transaction_get_mapped_range_restricted_to_serializable") {
std::string mapper = Tuple().append(prefix).append(RECORD).append("{K[3]}"_sr).pack().toString();
fdb::Transaction tr(db);
fdb_check(tr.set_option(FDB_TR_OPTION_READ_YOUR_WRITES_DISABLE, nullptr, 0));
auto result = get_mapped_range(
tr,
FDB_KEYSEL_FIRST_GREATER_OR_EQUAL((const uint8_t*)indexEntryKey(0).c_str(), indexEntryKey(0).size()),
@ -1039,11 +1041,36 @@ TEST_CASE("fdb_transaction_get_mapped_range_restricted_to_ryw_enable") {
/* target_bytes */ 0,
/* FDBStreamingMode */ FDB_STREAMING_MODE_WANT_ALL,
/* iteration */ 0,
/* snapshot */ true,
/* snapshot */ false,
/* reverse */ 0);
ASSERT(result.err == error_code_unsupported_operation);
}
void assertNotTuple(std::string str) {
try {
Tuple::unpack(str);
} catch (Error& e) {
return;
}
UNREACHABLE();
}
TEST_CASE("fdb_transaction_get_mapped_range_fail_on_mapper_not_tuple") {
// A string that cannot be parsed as tuple.
// "\x15:\x152\x15E\x15\x09\x15\x02\x02MySimpleRecord$repeater-version\x00\x15\x013\x00\x00\x00\x00\x1aU\x90\xba\x00\x00\x00\x02\x15\x04"
std::string mapper = {
'\x15', ':', '\x15', '2', '\x15', 'E', '\x15', '\t', '\x15', '\x02', '\x02', 'M',
'y', 'S', 'i', 'm', 'p', 'l', 'e', 'R', 'e', 'c', 'o', 'r',
'd', '$', 'r', 'e', 'p', 'e', 'a', 't', 'e', 'r', '-', 'v',
'e', 'r', 's', 'i', 'o', 'n', '\x00', '\x15', '\x01', '3', '\x00', '\x00',
'\x00', '\x00', '\x1a', 'U', '\x90', '\xba', '\x00', '\x00', '\x00', '\x02', '\x15', '\x04'
};
assertNotTuple(mapper);
fdb::Transaction tr(db);
auto result = getMappedIndexEntries(1, 3, tr, mapper);
ASSERT(result.err == error_code_mapper_not_tuple);
}
TEST_CASE("fdb_transaction_get_range reverse") {
std::map<std::string, std::string> data = create_data({ { "a", "1" }, { "b", "2" }, { "c", "3" }, { "d", "4" } });
insert_data(db, data);

View File

@ -42,7 +42,10 @@ import com.apple.foundationdb.tuple.Tuple;
*/
public interface Database extends AutoCloseable, TransactionContext {
/**
* Opens an existing tenant to be used for running transactions.
* Opens an existing tenant to be used for running transactions.<br>
* <br>
* <b>Note:</b> opening a tenant does not check its existence in the cluster. If the tenant does not exist,
* attempts to read or write data with it will fail.
*
* @param tenantName The name of the tenant to open.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
@ -53,7 +56,10 @@ public interface Database extends AutoCloseable, TransactionContext {
/**
* Opens an existing tenant to be used for running transactions. This is a convenience method that generates the
* tenant name by packing a {@code Tuple}.
* tenant name by packing a {@code Tuple}.<br>
* <br>
* <b>Note:</b> opening a tenant does not check its existence in the cluster. If the tenant does not exist,
* attempts to read or write data with it will fail.
*
* @param tenantName The name of the tenant to open, as a Tuple.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.

View File

@ -233,7 +233,8 @@ def suspend(logger):
port = address.split(':')[1]
logger.debug("Port: {}".format(port))
# use the port number to find the exact fdb process we are connecting to
pinfo = list(filter(lambda x: port in x, pinfos))
# child process like fdbserver -r flowprocess does not provide `datadir` in the command line
pinfo = list(filter(lambda x: port in x and 'datadir' in x, pinfos))
assert len(pinfo) == 1
pid = pinfo[0].split(' ')[0]
logger.debug("Pid: {}".format(pid))

View File

@ -322,6 +322,8 @@ A |database-blurb1| |database-blurb2|
The tenant name can be either a byte string or a tuple. If a tuple is provided, the tuple will be packed using the tuple layer to generate the byte string tenant name.
.. note :: Opening a tenant does not check its existence in the cluster. If the tenant does not exist, attempts to read or write data with it will fail.
.. |sync-read| replace:: This read is fully synchronous.
.. |sync-write| replace:: This change will be committed immediately, and is fully synchronous.

View File

@ -26,6 +26,8 @@ FoundationDB supports language bindings for application development using the or
* :doc:`known-limitations` describes both long-term design limitations of FoundationDB and short-term limitations applicable to the current version.
* :doc:`tenants` describes the use of the tenants feature to define named transaction domains.
.. toctree::
:maxdepth: 1
:titlesonly:
@ -42,3 +44,4 @@ FoundationDB supports language bindings for application development using the or
known-limitations
transaction-profiler-analyzer
api-version-upgrade-guide
tenants

View File

@ -273,6 +273,16 @@ Directory partitions have the following drawbacks, and in general they should no
* Directories in a partition have longer prefixes than their counterparts outside of partitions, which reduces performance. Nesting partitions inside of other partitions results in even longer prefixes.
* The root directory of a partition cannot be used to pack/unpack keys and therefore cannot be used to create subspaces. You must create at least one subdirectory of a partition in order to store content in it.
Tenants
-------
:doc:`tenants` in FoundationDB provide a way to divide the cluster key-space into named transaction domains. Each tenant has a byte-string name that can be used to open transactions on the tenant's data, and tenant transactions are not permitted to access data outside of the tenant. Tenants can be useful for enforcing separation between unrelated use-cases.
Tenants and directories
~~~~~~~~~~~~~~~~~~~~~~~
Because tenants enforce that transactions operate within the tenant boundaries, it is not recommended to use a global directory layer shared between tenants. It is possible, however, to use the directory layer within each tenant. To do so, simply use the directory layer as normal with tenant transactions.
Working with the APIs
=====================

View File

@ -0,0 +1,60 @@
#######
Tenants
#######
.. warning :: Tenants are currently experimental and are not recommended for use in production.
FoundationDB provides a feature called tenants that allow you to configure one or more named transaction domains in your cluster. A transaction domain is a key-space in which a transaction is allowed to operate, and no tenant operations are allowed to use keys outside the tenant key-space. Tenants can be useful for managing separate, unrelated use-cases and preventing them from interfering with each other. They can also be helpful for defining safe boundaries when moving a subset of data between clusters.
By default, FoundationDB has a single transaction domain that contains both the normal key-space (``['', '\xff')``) as well as the system keys (``['\xff', '\xff\xff')``) and the :doc:`special-keys` (``['\xff\xff', '\xff\xff\xff')``).
Overview
========
A tenant in a FoundationDB cluster maps a byte-string name to a key-space that can be used to store data associated with that tenant. This key-space is stored in the clusters global key-space under a prefix assigned to that tenant, with each tenant being assigned a separate non-intersecting prefix.
In addition to being each assigned a separate tenant prefix, tenants can be configured to have a common shared prefix. By default, the shared prefix is empty and tenants are allocated prefixes throughout the normal key-space. To configure an alternate shared prefix, set the ``\xff/tenantDataPrefix`` key to have the desired prefix as the value.
Tenant operations are implicitly confined to the key-space associated with the tenant. It is not necessary for client applications to use or be aware of the prefix assigned to the tenant.
Enabling tenants
================
In order to use tenants, the cluster must be configured with an appropriate tenant mode using ``fdbcli``::
fdb> configure tenant_mode=<MODE>
FoundationDB clusters support the following tenant modes:
* ``disabled`` - Tenants cannot be created or used. Disabled is the default tenant mode.
* ``optional_experimental`` - Tenants can be created. Each transaction can choose whether or not to use a tenant. This mode is primarily intended for migration and testing purposes, and care should be taken to avoid conflicts between tenant and non-tenant data.
* ``required_experimental`` - Tenants can be created. Each normal transaction must use a tenant. To support special access needs, transactions will be permitted to access the raw key-space using the ``RAW_ACCESS`` transaction option.
Creating and deleting tenants
=============================
Tenants can be created and deleted using the ``\xff\xff/management/tenant_map/<tenant_name>`` :doc:`special key <special-keys>` range as well as by using APIs provided in some language bindings.
Tenants can be created with any byte-string name that does not begin with the ``\xff`` character. Once created, a tenant will be assigned an ID and a prefix where its data will reside.
In order to delete a tenant, it must first be empty. If a tenant contains any keys, they must be cleared prior to deleting the tenant.
Using tenants
=============
In order to use the key-space associated with an existing tenant, you must open the tenant using the ``Database`` object provided by your language binding. The resulting ``Tenant`` object can be used to create transactions much like with a ``Database``, and the resulting transactions will be restricted to the tenant's key-space.
All operations performed within a tenant transaction will occur within the tenant key-space. It is not necessary to use or even be aware of the prefix assigned to a tenant in the global key-space. Operations that could resolve outside of the tenant key-space (e.g. resolving key selectors) will be clamped to the tenant.
.. note :: Tenant transactions are not permitted to access system keys.
Raw access
----------
When operating in the tenant mode ``required_experimental``, transactions are not ordinarily permitted to run without using a tenant. In order to access the system keys or perform maintenance operations that span multiple tenants, it is required to use the ``RAW_ACCESS`` transaction option to access the global key-space. It is an error to specify ``RAW_ACCESS`` on a transaction that is configured to use a tenant.
.. note :: Setting the ``READ_SYSTEM_KEYS`` or ``ACCESS_SYSTEM_KEYS`` options implies ``RAW_ACCESS`` for your transaction.
.. note :: Many :doc:`special keys <special-keys>` operations access parts of the system keys and will implictly enable raw access on the transactions in which they are used.
.. warning :: Care should be taken when using raw access to run transactions spanning multiple tenants if the tenant feature is being utilized to aid in moving data between clusters. In such scenarios, it may not be guaranteed that all of the data you intend to access is on a single cluster.

View File

@ -667,6 +667,7 @@ struct GetRangeLimits {
};
struct RangeResultRef : VectorRef<KeyValueRef> {
constexpr static FileIdentifier file_identifier = 3985192;
bool more; // True if (but not necessarily only if) values remain in the *key* range requested (possibly beyond the
// limits requested) False implies that no such values remain
Optional<KeyRef> readThrough; // Only present when 'more' is true. When present, this value represent the end (or
@ -973,6 +974,7 @@ struct TLogSpillType {
// Contains the amount of free and total space for a storage server, in bytes
struct StorageBytes {
constexpr static FileIdentifier file_identifier = 3928581;
// Free space on the filesystem
int64_t free;
// Total space on the filesystem
@ -1386,7 +1388,7 @@ struct StorageMetadataType {
StorageMetadataType() : createdTime(0) {}
StorageMetadataType(uint64_t t) : createdTime(t) {}
static uint64_t currentTime() { return g_network->timer() * 1e9; }
static uint64_t currentTime() { return g_network->timer_int(); }
// To change this serialization, ProtocolVersion::StorageMetadata must be updated, and downgrades need
// to be considered

View File

@ -785,7 +785,7 @@ void MultiVersionTransaction::updateTransaction() {
TransactionInfo newTr;
if (tenant.present()) {
ASSERT(tenant.get());
auto currentTenant = tenant.get()->tenantVar->get();
auto currentTenant = tenant.get()->tenantState->tenantVar->get();
if (currentTenant.value) {
newTr.transaction = currentTenant.value->createTransaction();
}
@ -1104,7 +1104,7 @@ ThreadFuture<Void> MultiVersionTransaction::onError(Error const& e) {
Optional<TenantName> MultiVersionTransaction::getTenant() {
if (tenant.present()) {
return tenant.get()->tenantName;
return tenant.get()->tenantState->tenantName;
} else {
return Optional<TenantName>();
}
@ -1238,20 +1238,27 @@ bool MultiVersionTransaction::isValid() {
// MultiVersionTenant
MultiVersionTenant::MultiVersionTenant(Reference<MultiVersionDatabase> db, StringRef tenantName)
: tenantVar(new ThreadSafeAsyncVar<Reference<ITenant>>(Reference<ITenant>(nullptr))), tenantName(tenantName), db(db) {
updateTenant();
: tenantState(makeReference<TenantState>(db, tenantName)) {}
MultiVersionTenant::~MultiVersionTenant() {
tenantState->close();
}
MultiVersionTenant::~MultiVersionTenant() {}
Reference<ITransaction> MultiVersionTenant::createTransaction() {
return Reference<ITransaction>(new MultiVersionTransaction(
db, Reference<MultiVersionTenant>::addRef(this), db->dbState->transactionDefaultOptions));
return Reference<ITransaction>(new MultiVersionTransaction(tenantState->db,
Reference<MultiVersionTenant>::addRef(this),
tenantState->db->dbState->transactionDefaultOptions));
}
MultiVersionTenant::TenantState::TenantState(Reference<MultiVersionDatabase> db, StringRef tenantName)
: tenantVar(new ThreadSafeAsyncVar<Reference<ITenant>>(Reference<ITenant>(nullptr))), tenantName(tenantName), db(db),
closed(false) {
updateTenant();
}
// Creates a new underlying tenant object whenever the database connection changes. This change is signaled
// to open transactions via an AsyncVar.
void MultiVersionTenant::updateTenant() {
void MultiVersionTenant::TenantState::updateTenant() {
Reference<ITenant> tenant;
auto currentDb = db->dbState->dbVar->get();
if (currentDb.value) {
@ -1262,13 +1269,27 @@ void MultiVersionTenant::updateTenant() {
tenantVar->set(tenant);
Reference<TenantState> self = Reference<TenantState>::addRef(this);
MutexHolder holder(tenantLock);
tenantUpdater = mapThreadFuture<Void, Void>(currentDb.onChange, [this](ErrorOr<Void> result) {
updateTenant();
if (closed) {
return;
}
tenantUpdater = mapThreadFuture<Void, Void>(currentDb.onChange, [self](ErrorOr<Void> result) {
self->updateTenant();
return Void();
});
}
void MultiVersionTenant::TenantState::close() {
MutexHolder holder(tenantLock);
closed = true;
if (tenantUpdater.isValid()) {
tenantUpdater.cancel();
}
}
// MultiVersionDatabase
MultiVersionDatabase::MultiVersionDatabase(MultiVersionApi* api,
int threadIdx,

View File

@ -650,18 +650,30 @@ public:
void addref() override { ThreadSafeReferenceCounted<MultiVersionTenant>::addref(); }
void delref() override { ThreadSafeReferenceCounted<MultiVersionTenant>::delref(); }
Reference<ThreadSafeAsyncVar<Reference<ITenant>>> tenantVar;
const Standalone<StringRef> tenantName;
// A struct that manages the current connection state of the MultiVersionDatabase. This wraps the underlying
// IDatabase object that is currently interacting with the cluster.
struct TenantState : ThreadSafeReferenceCounted<TenantState> {
TenantState(Reference<MultiVersionDatabase> db, StringRef tenantName);
private:
Reference<MultiVersionDatabase> db;
// Creates a new underlying tenant object whenever the database connection changes. This change is signaled
// to open transactions via an AsyncVar.
void updateTenant();
Mutex tenantLock;
ThreadFuture<Void> tenantUpdater;
// Cleans up local state to break reference cycles
void close();
// Creates a new underlying tenant object whenever the database connection changes. This change is signaled
// to open transactions via an AsyncVar.
void updateTenant();
Reference<ThreadSafeAsyncVar<Reference<ITenant>>> tenantVar;
const Standalone<StringRef> tenantName;
Reference<MultiVersionDatabase> db;
Mutex tenantLock;
ThreadFuture<Void> tenantUpdater;
bool closed;
};
Reference<TenantState> tenantState;
};
// An implementation of IDatabase that wraps a database created either locally or through a dynamically loaded

View File

@ -255,6 +255,9 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( DEBOUNCE_RECRUITING_DELAY, 5.0 );
init( DD_FAILURE_TIME, 1.0 ); if( randomize && BUGGIFY ) DD_FAILURE_TIME = 10.0;
init( DD_ZERO_HEALTHY_TEAM_DELAY, 1.0 );
init( REMOTE_KV_STORE, false ); if( randomize && BUGGIFY ) REMOTE_KV_STORE = true;
init( REMOTE_KV_STORE_INIT_DELAY, 0.1 );
init( REMOTE_KV_STORE_MAX_INIT_DURATION, 10.0 );
init( REBALANCE_MAX_RETRIES, 100 );
init( DD_OVERLAP_PENALTY, 10000 );
init( DD_EXCLUDE_MIN_REPLICAS, 1 );
@ -561,6 +564,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( MIN_REBOOT_TIME, 4.0 ); if( longReboots ) MIN_REBOOT_TIME = 10.0;
init( MAX_REBOOT_TIME, 5.0 ); if( longReboots ) MAX_REBOOT_TIME = 20.0;
init( LOG_DIRECTORY, "."); // Will be set to the command line flag.
init( CONN_FILE, ""); // Will be set to the command line flag.
init( SERVER_MEM_LIMIT, 8LL << 30 );
init( SYSTEM_MONITOR_FREQUENCY, 5.0 );
@ -659,6 +663,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( FETCH_KEYS_LOWER_PRIORITY, 0 );
init( FETCH_CHANGEFEED_PARALLELISM, 2 );
init( BUGGIFY_BLOCK_BYTES, 10000 );
init( STORAGE_RECOVERY_VERSION_LAG_LIMIT, 2 * MAX_READ_TRANSACTION_LIFE_VERSIONS );
init( STORAGE_COMMIT_BYTES, 10000000 ); if( randomize && BUGGIFY ) STORAGE_COMMIT_BYTES = 2000000;
init( STORAGE_FETCH_BYTES, 2500000 ); if( randomize && BUGGIFY ) STORAGE_FETCH_BYTES = 500000;
init( STORAGE_DURABILITY_LAG_REJECT_THRESHOLD, 0.25 );

View File

@ -236,6 +236,14 @@ public:
double DD_FAILURE_TIME;
double DD_ZERO_HEALTHY_TEAM_DELAY;
// Run storage enginee on a child process on the same machine with storage process
bool REMOTE_KV_STORE;
// A delay to avoid race on file resources if the new kv store process started immediately after the previous kv
// store process died
double REMOTE_KV_STORE_INIT_DELAY;
// max waiting time for the remote kv store to initialize
double REMOTE_KV_STORE_MAX_INIT_DURATION;
// KeyValueStore SQLITE
int CLEAR_BUFFER_SIZE;
double READ_VALUE_TIME_ESTIMATE;
@ -492,6 +500,7 @@ public:
double MIN_REBOOT_TIME;
double MAX_REBOOT_TIME;
std::string LOG_DIRECTORY;
std::string CONN_FILE;
int64_t SERVER_MEM_LIMIT;
double SYSTEM_MONITOR_FREQUENCY;
@ -593,6 +602,7 @@ public:
int FETCH_KEYS_LOWER_PRIORITY;
int FETCH_CHANGEFEED_PARALLELISM;
int BUGGIFY_BLOCK_BYTES;
int64_t STORAGE_RECOVERY_VERSION_LAG_LIMIT;
double STORAGE_DURABILITY_LAG_REJECT_THRESHOLD;
double STORAGE_DURABILITY_LAG_MIN_RATE;
int STORAGE_COMMIT_BYTES;

View File

@ -90,12 +90,19 @@ struct StorageServerInterface {
RequestStream<struct GetCheckpointRequest> checkpoint;
RequestStream<struct FetchCheckpointRequest> fetchCheckpoint;
explicit StorageServerInterface(UID uid) : uniqueID(uid) {}
StorageServerInterface() : uniqueID(deterministicRandom()->randomUniqueID()) {}
private:
bool acceptingRequests;
public:
explicit StorageServerInterface(UID uid) : uniqueID(uid) { acceptingRequests = false; }
StorageServerInterface() : uniqueID(deterministicRandom()->randomUniqueID()) { acceptingRequests = false; }
NetworkAddress address() const { return getValue.getEndpoint().getPrimaryAddress(); }
NetworkAddress stableAddress() const { return getValue.getEndpoint().getStableAddress(); }
Optional<NetworkAddress> secondaryAddress() const { return getValue.getEndpoint().addresses.secondaryAddress; }
UID id() const { return uniqueID; }
bool isAcceptingRequests() const { return acceptingRequests; }
void startAcceptingRequests() { acceptingRequests = true; }
void stopAcceptingRequests() { acceptingRequests = false; }
bool isTss() const { return tssPairID.present(); }
std::string toString() const { return id().shortString(); }
template <class Ar>
@ -106,7 +113,11 @@ struct StorageServerInterface {
if (ar.protocolVersion().hasSmallEndpoints()) {
if (ar.protocolVersion().hasTSS()) {
serializer(ar, uniqueID, locality, getValue, tssPairID);
if (ar.protocolVersion().hasStorageInterfaceReadiness()) {
serializer(ar, uniqueID, locality, getValue, tssPairID, acceptingRequests);
} else {
serializer(ar, uniqueID, locality, getValue, tssPairID);
}
} else {
serializer(ar, uniqueID, locality, getValue);
}

View File

@ -587,28 +587,18 @@ const Key serverListKeyFor(UID serverID) {
return wr.toValue();
}
// TODO use flatbuffers depending on version
const Value serverListValue(StorageServerInterface const& server) {
BinaryWriter wr(IncludeVersion(ProtocolVersion::withServerListValue()));
wr << server;
return wr.toValue();
auto protocolVersion = currentProtocolVersion;
protocolVersion.addObjectSerializerFlag();
return ObjectWriter::toValue(server, IncludeVersion(protocolVersion));
}
UID decodeServerListKey(KeyRef const& key) {
UID serverID;
BinaryReader rd(key.removePrefix(serverListKeys.begin), Unversioned());
rd >> serverID;
return serverID;
}
StorageServerInterface decodeServerListValue(ValueRef const& value) {
StorageServerInterface s;
BinaryReader reader(value, IncludeVersion());
reader >> s;
return s;
}
const Value serverListValueFB(StorageServerInterface const& server) {
return ObjectWriter::toValue(server, IncludeVersion());
}
StorageServerInterface decodeServerListValueFB(ValueRef const& value) {
StorageServerInterface s;
@ -617,6 +607,18 @@ StorageServerInterface decodeServerListValueFB(ValueRef const& value) {
return s;
}
StorageServerInterface decodeServerListValue(ValueRef const& value) {
StorageServerInterface s;
BinaryReader reader(value, IncludeVersion());
if (!reader.protocolVersion().hasStorageInterfaceReadiness()) {
reader >> s;
return s;
}
return decodeServerListValueFB(value);
}
// processClassKeys.contains(k) iff k.startsWith( processClassKeys.begin ) because '/'+1 == '0'
const KeyRangeRef processClassKeys(LiteralStringRef("\xff/processClass/"), LiteralStringRef("\xff/processClass0"));
const KeyRef processClassPrefix = processClassKeys.begin;
@ -1393,29 +1395,31 @@ const KeyRef tenantLastIdKey = "\xff/tenantLastId/"_sr;
const KeyRef tenantDataPrefixKey = "\xff/tenantDataPrefix"_sr;
// for tests
void testSSISerdes(StorageServerInterface const& ssi, bool useFB) {
printf("ssi=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\naddress=%s\ngetValue=%s\n\n\n",
void testSSISerdes(StorageServerInterface const& ssi) {
printf("ssi=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\nacceptingRequests=%s\naddress=%s\ngetValue=%s\n\n\n",
ssi.id().toString().c_str(),
ssi.locality.toString().c_str(),
ssi.isTss() ? "true" : "false",
ssi.isTss() ? ssi.tssPairID.get().toString().c_str() : "",
ssi.isAcceptingRequests() ? "true" : "false",
ssi.address().toString().c_str(),
ssi.getValue.getEndpoint().token.toString().c_str());
StorageServerInterface ssi2 =
(useFB) ? decodeServerListValueFB(serverListValueFB(ssi)) : decodeServerListValue(serverListValue(ssi));
StorageServerInterface ssi2 = decodeServerListValue(serverListValue(ssi));
printf("ssi2=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\naddress=%s\ngetValue=%s\n\n\n",
printf("ssi2=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\nacceptingRequests=%s\naddress=%s\ngetValue=%s\n\n\n",
ssi2.id().toString().c_str(),
ssi2.locality.toString().c_str(),
ssi2.isTss() ? "true" : "false",
ssi2.isTss() ? ssi2.tssPairID.get().toString().c_str() : "",
ssi2.isAcceptingRequests() ? "true" : "false",
ssi2.address().toString().c_str(),
ssi2.getValue.getEndpoint().token.toString().c_str());
ASSERT(ssi.id() == ssi2.id());
ASSERT(ssi.locality == ssi2.locality);
ASSERT(ssi.isTss() == ssi2.isTss());
ASSERT(ssi.isAcceptingRequests() == ssi2.isAcceptingRequests());
if (ssi.isTss()) {
ASSERT(ssi2.tssPairID.get() == ssi2.tssPairID.get());
}
@ -1437,13 +1441,11 @@ TEST_CASE("/SystemData/SerDes/SSI") {
ssi.locality = localityData;
ssi.initEndpoints();
testSSISerdes(ssi, false);
testSSISerdes(ssi, true);
testSSISerdes(ssi);
ssi.tssPairID = UID(0x2345234523452345, 0x1238123812381238);
testSSISerdes(ssi, false);
testSSISerdes(ssi, true);
testSSISerdes(ssi);
printf("ssi serdes test complete\n");
return Void();

View File

@ -10,6 +10,7 @@ set(FDBRPC_SRCS
AsyncFileNonDurable.actor.cpp
AsyncFileWriteChecker.cpp
FailureMonitor.actor.cpp
FlowProcess.actor.h
FlowTransport.actor.cpp
genericactors.actor.h
genericactors.actor.cpp

View File

@ -0,0 +1,94 @@
/*
* FlowProcess.actor.h
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#if defined(NO_INTELLISENSE) && !defined(FDBRPC_FLOW_PROCESS_ACTOR_G_H)
#define FDBRPC_FLOW_PROCESS_ACTOR_G_H
#include "fdbrpc/FlowProcess.actor.g.h"
#elif !defined(FDBRPC_FLOW_PROCESS_ACTOR_H)
#define FDBRPC_FLOW_PROCESS_ACTOR_H
#include "fdbrpc/fdbrpc.h"
#include <string>
#include <map>
#include <flow/actorcompiler.h> // has to be last include
struct FlowProcessInterface {
constexpr static FileIdentifier file_identifier = 3491839;
RequestStream<struct FlowProcessRegistrationRequest> registerProcess;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, registerProcess);
}
};
struct FlowProcessRegistrationRequest {
constexpr static FileIdentifier file_identifier = 3411838;
Standalone<StringRef> flowProcessInterface;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, flowProcessInterface);
}
};
class FlowProcess {
public:
virtual ~FlowProcess() {}
virtual StringRef name() const = 0;
virtual StringRef serializedInterface() const = 0;
virtual Future<Void> run() = 0;
virtual void registerEndpoint(Endpoint p) = 0;
};
struct IProcessFactory {
static FlowProcess* create(std::string const& name) {
auto it = factories().find(name);
if (it == factories().end())
return nullptr; // or throw?
return it->second->create();
}
static std::map<std::string, IProcessFactory*>& factories() {
static std::map<std::string, IProcessFactory*> theFactories;
return theFactories;
}
virtual FlowProcess* create() = 0;
virtual const char* getName() = 0;
};
template <class ProcessType>
struct ProcessFactory : IProcessFactory {
ProcessFactory(const char* name) : name(name) { factories()[name] = this; }
FlowProcess* create() override { return new ProcessType(); }
const char* getName() override { return this->name; }
private:
const char* name;
};
#include <flow/unactorcompiler.h>
#endif

View File

@ -991,7 +991,8 @@ static void scanPackets(TransportData* transport,
Arena& arena,
NetworkAddress const& peerAddress,
ProtocolVersion peerProtocolVersion,
Future<Void> disconnect) {
Future<Void> disconnect,
bool isStableConnection) {
// Find each complete packet in the given byte range and queue a ready task to deliver it.
// Remove the complete packets from the range by increasing unprocessed_begin.
// There won't be more than 64K of data plus one packet, so this shouldn't take a long time.
@ -1030,7 +1031,7 @@ static void scanPackets(TransportData* transport,
if (checksumEnabled) {
bool isBuggifyEnabled = false;
if (g_network->isSimulated() &&
if (g_network->isSimulated() && !isStableConnection &&
g_network->now() - g_simulator.lastConnectionFailure > g_simulator.connectionFailuresDisableDuration &&
BUGGIFY_WITH_PROB(0.0001)) {
g_simulator.lastConnectionFailure = g_network->now();
@ -1057,7 +1058,8 @@ static void scanPackets(TransportData* transport,
if (isBuggifyEnabled) {
TraceEvent(SevInfo, "ChecksumMismatchExp")
.detail("PacketChecksum", packetChecksum)
.detail("CalculatedChecksum", calculatedChecksum);
.detail("CalculatedChecksum", calculatedChecksum)
.detail("PeerAddress", peerAddress.toString());
} else {
TraceEvent(SevWarnAlways, "ChecksumMismatchUnexp")
.detail("PacketChecksum", packetChecksum)
@ -1305,7 +1307,8 @@ ACTOR static Future<Void> connectionReader(TransportData* transport,
arena,
peerAddress,
peerProtocolVersion,
peer->disconnect.getFuture());
peer->disconnect.getFuture(),
g_network->isSimulated() && conn->isStableConnection());
} else {
unprocessed_begin = unprocessed_end;
peer->resetPing.trigger();
@ -1364,6 +1367,11 @@ ACTOR static Future<Void> listen(TransportData* self, NetworkAddress listenAddr)
state ActorCollectionNoErrors
incoming; // Actors monitoring incoming connections that haven't yet been associated with a peer
state Reference<IListener> listener = INetworkConnections::net()->listen(listenAddr);
if (!g_network->isSimulated() && self->localAddresses.address.port == 0) {
TraceEvent(SevInfo, "UpdatingListenAddress")
.detail("AssignedListenAddress", listener->getListenAddress().toString());
self->localAddresses.address = listener->getListenAddress();
}
state uint64_t connectionCount = 0;
try {
loop {

View File

@ -20,6 +20,7 @@
#include <cinttypes>
#include <memory>
#include <string>
#include "contrib/fmt-8.1.1/include/fmt/format.h"
#include "fdbrpc/simulator.h"
@ -121,20 +122,24 @@ void ISimulator::displayWorkers() const {
int openCount = 0;
struct SimClogging {
double getSendDelay(NetworkAddress from, NetworkAddress to) const { return halfLatency(); }
double getSendDelay(NetworkAddress from, NetworkAddress to, bool stableConnection = false) const {
// stable connection here means it's a local connection between processes on the same machine
// we expect it to have much lower latency
return (stableConnection ? 0.1 : 1.0) * halfLatency();
}
double getRecvDelay(NetworkAddress from, NetworkAddress to) {
double getRecvDelay(NetworkAddress from, NetworkAddress to, bool stableConnection = false) {
auto pair = std::make_pair(from.ip, to.ip);
double tnow = now();
double t = tnow + halfLatency();
if (!g_simulator.speedUpSimulation)
double t = tnow + (stableConnection ? 0.1 : 1.0) * halfLatency();
if (!g_simulator.speedUpSimulation && !stableConnection)
t += clogPairLatency[pair];
if (!g_simulator.speedUpSimulation && clogPairUntil.count(pair))
if (!g_simulator.speedUpSimulation && !stableConnection && clogPairUntil.count(pair))
t = std::max(t, clogPairUntil[pair]);
if (!g_simulator.speedUpSimulation && clogRecvUntil.count(to.ip))
if (!g_simulator.speedUpSimulation && !stableConnection && clogRecvUntil.count(to.ip))
t = std::max(t, clogRecvUntil[to.ip]);
return t - tnow;
@ -182,8 +187,8 @@ SimClogging g_clogging;
struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
Sim2Conn(ISimulator::ProcessInfo* process)
: opened(false), closedByCaller(false), process(process), dbgid(deterministicRandom()->randomUniqueID()),
stopReceive(Never()) {
: opened(false), closedByCaller(false), stableConnection(false), process(process),
dbgid(deterministicRandom()->randomUniqueID()), stopReceive(Never()) {
pipes = sender(this) && receiver(this);
}
@ -202,7 +207,18 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
process->address.ip,
FLOW_KNOBS->MAX_CLOGGING_LATENCY * deterministicRandom()->random01());
sendBufSize = std::max<double>(deterministicRandom()->randomInt(0, 5000000), 25e6 * (latency + .002));
TraceEvent("Sim2Connection").detail("SendBufSize", sendBufSize).detail("Latency", latency);
// options like clogging or bitsflip are disabled for stable connections
stableConnection = std::any_of(process->childs.begin(),
process->childs.end(),
[&](ISimulator::ProcessInfo* child) { return child && child == peerProcess; }) ||
std::any_of(peerProcess->childs.begin(),
peerProcess->childs.end(),
[&](ISimulator::ProcessInfo* child) { return child && child == process; });
TraceEvent("Sim2Connection")
.detail("SendBufSize", sendBufSize)
.detail("Latency", latency)
.detail("StableConnection", stableConnection);
}
~Sim2Conn() { ASSERT_ABORT(!opened || closedByCaller); }
@ -222,6 +238,8 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
bool isPeerGone() const { return !peer || peerProcess->failed; }
bool isStableConnection() const override { return stableConnection; }
void peerClosed() {
leakedConnectionTracker = trackLeakedConnection(this);
stopReceive = delay(1.0);
@ -249,7 +267,7 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
ASSERT(limit > 0);
int toSend = 0;
if (BUGGIFY) {
if (BUGGIFY && !stableConnection) {
toSend = std::min(limit, buffer->bytes_written - buffer->bytes_sent);
} else {
for (auto p = buffer; p; p = p->next) {
@ -262,7 +280,7 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
}
}
ASSERT(toSend);
if (BUGGIFY)
if (BUGGIFY && !stableConnection)
toSend = std::min(toSend, deterministicRandom()->randomInt(0, 1000));
if (!peer)
@ -286,7 +304,7 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
NetworkAddress getPeerAddress() const override { return peerEndpoint; }
UID getDebugID() const override { return dbgid; }
bool opened, closedByCaller;
bool opened, closedByCaller, stableConnection;
private:
ISimulator::ProcessInfo *process, *peerProcess;
@ -336,10 +354,12 @@ private:
deterministicRandom()->random01() < .5
? self->sentBytes.get()
: deterministicRandom()->randomInt64(self->receivedBytes.get(), self->sentBytes.get() + 1);
wait(delay(g_clogging.getSendDelay(self->process->address, self->peerProcess->address)));
wait(delay(g_clogging.getSendDelay(
self->process->address, self->peerProcess->address, self->isStableConnection())));
wait(g_simulator.onProcess(self->process));
ASSERT(g_simulator.getCurrentProcess() == self->process);
wait(delay(g_clogging.getRecvDelay(self->process->address, self->peerProcess->address)));
wait(delay(g_clogging.getRecvDelay(
self->process->address, self->peerProcess->address, self->isStableConnection())));
ASSERT(g_simulator.getCurrentProcess() == self->process);
if (self->stopReceive.isReady()) {
wait(Future<Void>(Never()));
@ -389,7 +409,9 @@ private:
}
void rollRandomClose() {
if (now() - g_simulator.lastConnectionFailure > g_simulator.connectionFailuresDisableDuration &&
// make sure connections between parenta and their childs are not closed
if (!stableConnection &&
now() - g_simulator.lastConnectionFailure > g_simulator.connectionFailuresDisableDuration &&
deterministicRandom()->random01() < .00001) {
g_simulator.lastConnectionFailure = now();
double a = deterministicRandom()->random01(), b = deterministicRandom()->random01();
@ -1101,6 +1123,10 @@ public:
if (mustBeDurable || deterministicRandom()->random01() < 0.5) {
state ISimulator::ProcessInfo* currentProcess = g_simulator.getCurrentProcess();
state TaskPriority currentTaskID = g_network->getCurrentTask();
TraceEvent(SevDebug, "Sim2DeleteFileImpl")
.detail("CurrentProcess", currentProcess->toString())
.detail("Filename", filename)
.detail("Durable", mustBeDurable);
wait(g_simulator.onMachine(currentProcess));
try {
wait(::delay(0.05 * deterministicRandom()->random01()));
@ -1118,6 +1144,9 @@ public:
throw err;
}
} else {
TraceEvent(SevDebug, "Sim2DeleteFileImplNonDurable")
.detail("Filename", filename)
.detail("Durable", mustBeDurable);
TEST(true); // Simulated non-durable delete
return Void();
}
@ -1163,6 +1192,9 @@ public:
MachineInfo& machine = machines[locality.machineId().get()];
if (!machine.machineId.present())
machine.machineId = locality.machineId();
if (port == 0 && std::string(name) == "remote flow process") {
port = machine.getRandomPort();
}
for (int i = 0; i < machine.processes.size(); i++) {
if (machine.processes[i]->locality.machineId() !=
locality.machineId()) { // SOMEDAY: compute ip from locality to avoid this check
@ -1220,6 +1252,11 @@ public:
.detail("Excluded", m->excluded)
.detail("Cleared", m->cleared);
if (std::string(name) == "remote flow process") {
protectedAddresses.insert(m->address);
TraceEvent(SevDebug, "NewFlowProcessProtected").detail("Address", m->address);
}
// FIXME: Sometimes, connections to/from this process will explicitly close
return m;
@ -1497,6 +1534,7 @@ public:
.detail("MachineId", p->locality.machineId());
currentlyRebootingProcesses.insert(std::pair<NetworkAddress, ProcessInfo*>(p->address, p));
std::vector<ProcessInfo*>& processes = machines[p->locality.machineId().get()].processes;
machines[p->locality.machineId().get()].removeRemotePort(p->address.port);
if (p != processes.back()) {
auto it = std::find(processes.begin(), processes.end(), p);
std::swap(*it, processes.back());
@ -1520,7 +1558,8 @@ public:
.detail("Protected", protectedAddresses.count(machine->address))
.backtrace();
// This will remove all the "tracked" messages that came from the machine being killed
latestEventCache.clear();
if (std::string(machine->name) != "remote flow process")
latestEventCache.clear();
machine->failed = true;
} else if (kt == InjectFaults) {
TraceEvent(SevWarn, "FaultMachine")
@ -1548,7 +1587,8 @@ public:
} else {
ASSERT(false);
}
ASSERT(!protectedAddresses.count(machine->address) || machine->rebooting);
ASSERT(!protectedAddresses.count(machine->address) || machine->rebooting ||
std::string(machine->name) == "remote flow process");
}
void rebootProcess(ProcessInfo* process, KillType kt) override {
if (kt == RebootProcessAndDelete && protectedAddresses.count(process->address)) {
@ -2390,8 +2430,19 @@ ACTOR void doReboot(ISimulator::ProcessInfo* p, ISimulator::KillType kt) {
kt ==
ISimulator::RebootProcessAndDelete); // Simulated process rebooted with data and coordination state deletion
if (p->rebooting || !p->isReliable())
if (p->rebooting || !p->isReliable()) {
TraceEvent(SevDebug, "DoRebootFailed")
.detail("Rebooting", p->rebooting)
.detail("Reliable", p->isReliable());
return;
} else if (std::string(p->name) == "remote flow process") {
TraceEvent(SevDebug, "DoRebootFailed").detail("Name", p->name).detail("Address", p->address);
return;
} else if (p->getChilds().size()) {
TraceEvent(SevDebug, "DoRebootFailedOnParentProcess").detail("Address", p->address);
return;
}
TraceEvent("RebootingProcess")
.detail("KillType", kt)
.detail("Address", p->address)

View File

@ -21,6 +21,7 @@
#ifndef FLOW_SIMULATOR_H
#define FLOW_SIMULATOR_H
#include "flow/ProtocolVersion.h"
#include <algorithm>
#include <string>
#pragma once
@ -87,6 +88,8 @@ public:
ProtocolVersion protocolVersion;
std::vector<ProcessInfo*> childs;
ProcessInfo(const char* name,
LocalityData locality,
ProcessClass startingClass,
@ -117,6 +120,7 @@ public:
<< " fault_injection_p2:" << fault_injection_p2;
return ss.str();
}
std::vector<ProcessInfo*> const& getChilds() const { return childs; }
// Return true if the class type is suitable for stateful roles, such as tLog and StorageServer.
bool isAvailableClass() const {
@ -202,7 +206,30 @@ public:
std::set<std::string> closingFiles;
Optional<Standalone<StringRef>> machineId;
MachineInfo() : machineProcess(nullptr) {}
const uint16_t remotePortStart;
std::vector<uint16_t> usedRemotePorts;
MachineInfo() : machineProcess(nullptr), remotePortStart(1000) {}
short getRandomPort() {
for (uint16_t i = remotePortStart; i < 60000; i++) {
if (std::find(usedRemotePorts.begin(), usedRemotePorts.end(), i) == usedRemotePorts.end()) {
TraceEvent(SevDebug, "RandomPortOpened").detail("PortNum", i);
usedRemotePorts.push_back(i);
return i;
}
}
UNREACHABLE();
}
void removeRemotePort(uint16_t port) {
if (port < remotePortStart)
return;
auto pos = std::find(usedRemotePorts.begin(), usedRemotePorts.end(), port);
if (pos != usedRemotePorts.end()) {
usedRemotePorts.erase(pos);
}
}
};
ProcessInfo* getProcess(Endpoint const& endpoint) { return getProcessByAddress(endpoint.getPrimaryAddress()); }

View File

@ -273,7 +273,8 @@ struct BlobManagerData : NonCopyable, ReferenceCounted<BlobManagerData> {
ACTOR Future<Standalone<VectorRef<KeyRef>>> splitRange(Reference<BlobManagerData> bmData,
KeyRange range,
bool writeHot) {
bool writeHot,
bool initialSplit) {
try {
if (BM_DEBUG) {
fmt::print("Splitting new range [{0} - {1}): {2}\n",
@ -290,8 +291,24 @@ ACTOR Future<Standalone<VectorRef<KeyRef>>> splitRange(Reference<BlobManagerData
estimated.bytes);
}
int64_t splitThreshold = SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES;
if (!initialSplit) {
// If we have X MB target granule size, we want to do the initial split to split up into X MB chunks.
// However, if we already have a granule that we are evaluating for split, if we split it as soon as it is
// larger than X MB, we will end up with 2 X/2 MB granules.
// To ensure an average size of X MB, we split granules at 4/3*X, so that they range between 2/3*X and
// 4/3*X, averaging X
splitThreshold = (splitThreshold * 4) / 3;
}
// if write-hot, we want to be able to split smaller, but not infinitely. Allow write-hot granules to be 3x
// smaller
// TODO knob?
// TODO: re-evaluate after we have granule merging?
if (writeHot) {
splitThreshold /= 3;
}
TEST(writeHot); // Change feed write hot split
if (estimated.bytes > SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES || writeHot) {
if (estimated.bytes > splitThreshold) {
// only split on bytes and write rate
state StorageMetrics splitMetrics;
splitMetrics.bytes = SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES;
@ -325,6 +342,7 @@ ACTOR Future<Standalone<VectorRef<KeyRef>>> splitRange(Reference<BlobManagerData
ASSERT(keys.back() == range.end);
return keys;
} else {
TEST(writeHot); // Not splitting write-hot because granules would be too small
if (BM_DEBUG) {
printf("Not splitting range\n");
}
@ -791,7 +809,7 @@ ACTOR Future<Void> monitorClientRanges(Reference<BlobManagerData> bmData) {
// Divide new ranges up into equal chunks by using SS byte sample
for (KeyRangeRef range : rangesToAdd) {
TraceEvent("ClientBlobRangeAdded", bmData->id).detail("Range", range);
splitFutures.push_back(splitRange(bmData, range, false));
splitFutures.push_back(splitRange(bmData, range, false, true));
}
for (auto f : splitFutures) {
@ -892,7 +910,7 @@ ACTOR Future<Void> maybeSplitRange(Reference<BlobManagerData> bmData,
state Standalone<VectorRef<KeyRef>> newRanges;
// first get ranges to split
Standalone<VectorRef<KeyRef>> _newRanges = wait(splitRange(bmData, granuleRange, writeHot));
Standalone<VectorRef<KeyRef>> _newRanges = wait(splitRange(bmData, granuleRange, writeHot, false));
newRanges = _newRanges;
ASSERT(newRanges.size() >= 2);

View File

@ -98,6 +98,8 @@ set(FDBSERVER_SRCS
Ratekeeper.h
RatekeeperInterface.h
RecoveryState.h
RemoteIKeyValueStore.actor.h
RemoteIKeyValueStore.actor.cpp
ResolutionBalancer.actor.cpp
ResolutionBalancer.actor.h
Resolver.actor.cpp

View File

@ -296,7 +296,7 @@ Future<Void> StorageWiggler::restoreStats() {
return map(readFuture, assignFunc);
}
Future<Void> StorageWiggler::startWiggle() {
metrics.last_wiggle_start = timer_int();
metrics.last_wiggle_start = g_network->timer_int();
if (shouldStartNewRound()) {
metrics.last_round_start = metrics.last_wiggle_start;
}
@ -304,7 +304,7 @@ Future<Void> StorageWiggler::startWiggle() {
}
Future<Void> StorageWiggler::finishWiggle() {
metrics.last_wiggle_finish = timer_int();
metrics.last_wiggle_finish = g_network->timer_int();
metrics.finished_wiggle += 1;
auto duration = metrics.last_wiggle_finish - metrics.last_wiggle_start;
metrics.smoothed_wiggle_duration.setTotal((double)duration);

View File

@ -1378,6 +1378,7 @@ ACTOR Future<Void> dataDistributionRelocator(DDQueueData* self, RelocateData rd,
} else {
TEST(true); // move to removed server
healthyDestinations.addDataInFlightToTeam(-metrics.bytes);
rd.completeDests.clear();
wait(delay(SERVER_KNOBS->RETRY_RELOCATESHARD_DELAY, TaskPriority::DataDistributionLaunch));
}
}

View File

@ -1077,7 +1077,7 @@ public:
Node* node(DeltaTree2* tree) const { return tree->nodeAt(nodeOffset); }
std::string toString() {
std::string toString() const {
return format("DecodedNode{nodeOffset=%d leftChildIndex=%d rightChildIndex=%d leftParentIndex=%d "
"rightParentIndex=%d}",
(int)nodeOffset,
@ -1155,6 +1155,19 @@ public:
arena = a;
updateUsedMemory();
}
std::string toString() const {
std::string s = format("DecodeCache{%p\n", this);
s += format("upperBound %s\n", upperBound.toString().c_str());
s += format("lowerBound %s\n", lowerBound.toString().c_str());
s += format("arenaSize %d\n", arena.getSize());
s += format("decodedNodes %d {\n", decodedNodes.size());
for (auto const& n : decodedNodes) {
s += format(" %s\n", n.toString().c_str());
}
s += format("}}\n");
return s;
}
};
// Cursor provides a way to seek into a DeltaTree and iterate over its contents
@ -1686,7 +1699,7 @@ public:
int count = end - begin;
numItems = count;
nodeBytesDeleted = 0;
initialHeight = (uint8_t)log2(count) + 1;
initialHeight = count == 0 ? 0 : (uint8_t)log2(count) + 1;
maxHeight = 0;
// The boundary leading to the new page acts as the last time we branched right

View File

@ -18,17 +18,30 @@
* limitations under the License.
*/
#include "flow/TLSConfig.actor.h"
#include "flow/Trace.h"
#include "flow/Platform.h"
#include "flow/flow.h"
#include "flow/genericactors.actor.h"
#include "flow/network.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/Net2FileSystem.h"
#include "fdbrpc/simulator.h"
#include "fdbclient/WellKnownEndpoints.h"
#include "fdbclient/versions.h"
#include "fdbserver/CoroFlow.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/Knobs.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#if !defined(_WIN32) && !defined(__APPLE__) && !defined(__INTEL_COMPILER)
#define BOOST_SYSTEM_NO_LIB
#define BOOST_DATE_TIME_NO_LIB
#define BOOST_REGEX_NO_LIB
#include <boost/process.hpp>
#endif
#include "fdbserver/FDBExecHelper.actor.h"
#include "flow/Trace.h"
#include "flow/flow.h"
#include "fdbclient/versions.h"
#include "fdbserver/Knobs.h"
#include <boost/algorithm/string.hpp>
#include "flow/actorcompiler.h" // This must be the last #include.
ExecCmdValueString::ExecCmdValueString(StringRef pCmdValueString) {
@ -90,12 +103,138 @@ void ExecCmdValueString::dbgPrint() const {
return;
}
ACTOR void destoryChildProcess(Future<Void> parentSSClosed, ISimulator::ProcessInfo* childInfo, std::string message) {
// This code path should be bug free
wait(parentSSClosed);
TraceEvent(SevDebug, message.c_str()).log();
// This one is root cause for most failures, make sure it's okay to destory
g_pSimulator->destroyProcess(childInfo);
// Explicitly reset the connection with the child process in case re-spawn very quickly
FlowTransport::transport().resetConnection(childInfo->address);
}
ACTOR Future<int> spawnSimulated(std::vector<std::string> paramList,
double maxWaitTime,
bool isSync,
double maxSimDelayTime,
IClosable* parent) {
state ISimulator::ProcessInfo* self = g_pSimulator->getCurrentProcess();
state ISimulator::ProcessInfo* child;
state std::string role;
state std::string addr;
state std::string flowProcessName;
state Endpoint parentProcessEndpoint;
state int i = 0;
// fdbserver -r flowprocess --process-name ikvs --process-endpoint ip:port,token,id
for (; i < paramList.size(); i++) {
if (paramList.size() > i + 1) {
// temporary args parser that only supports the flowprocess role
if (paramList[i] == "-r") {
role = paramList[i + 1];
} else if (paramList[i] == "-p" || paramList[i] == "--public_address") {
addr = paramList[i + 1];
} else if (paramList[i] == "--process-name") {
flowProcessName = paramList[i + 1];
} else if (paramList[i] == "--process-endpoint") {
state std::vector<std::string> addressArray;
boost::split(addressArray, paramList[i + 1], [](char c) { return c == ','; });
if (addressArray.size() != 3) {
std::cerr << "Invalid argument, expected 3 elements in --process-endpoint got "
<< addressArray.size() << std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
try {
auto addr = NetworkAddress::parse(addressArray[0]);
uint64_t fst = std::stoul(addressArray[1]);
uint64_t snd = std::stoul(addressArray[2]);
UID token(fst, snd);
NetworkAddressList l;
l.address = addr;
parentProcessEndpoint = Endpoint(l, token);
} catch (Error& e) {
std::cerr << "Could not parse network address " << addressArray[0] << std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
}
}
}
state int result = 0;
child = g_pSimulator->newProcess("remote flow process",
self->address.ip,
0,
self->address.isTLS(),
self->addresses.secondaryAddress.present() ? 2 : 1,
self->locality,
ProcessClass(ProcessClass::UnsetClass, ProcessClass::AutoSource),
self->dataFolder,
self->coordinationFolder, // do we need to customize this coordination folder path?
self->protocolVersion);
wait(g_pSimulator->onProcess(child));
state Future<ISimulator::KillType> onShutdown = child->onShutdown();
state Future<ISimulator::KillType> parentShutdown = self->onShutdown();
state Future<Void> flowProcessF;
try {
TraceEvent(SevDebug, "SpawnedChildProcess")
.detail("Child", child->toString())
.detail("Parent", self->toString());
std::string role = "";
std::string addr = "";
for (int i = 0; i < paramList.size(); i++) {
if (paramList.size() > i + 1 && paramList[i] == "-r") {
role = paramList[i + 1];
}
}
if (role == "flowprocess" && !parentShutdown.isReady()) {
self->childs.push_back(child);
state Future<Void> parentSSClosed = parent->onClosed();
FlowTransport::createInstance(false, 1, WLTOKEN_RESERVED_COUNT);
FlowTransport::transport().bind(child->address, child->address);
Sim2FileSystem::newFileSystem();
ProcessFactory<KeyValueStoreProcess>(flowProcessName.c_str());
flowProcessF = runFlowProcess(flowProcessName, parentProcessEndpoint);
choose {
when(wait(flowProcessF)) {
TraceEvent(SevDebug, "ChildProcessKilled").log();
wait(g_pSimulator->onProcess(self));
TraceEvent(SevDebug, "BackOnParentProcess").detail("Result", std::to_string(result));
destoryChildProcess(parentSSClosed, child, "StorageServerReceivedClosedMessage");
}
when(wait(success(onShutdown))) {
ASSERT(false);
// In prod, we use prctl to bind parent and child processes to die together
// In simulation, we simply disable killing parent or child processes as we cannot use the same
// mechanism here
}
when(wait(success(parentShutdown))) {
ASSERT(false);
// Parent process is not killed, see above
}
}
} else {
ASSERT(false);
}
} catch (Error& e) {
TraceEvent(SevError, "RemoteIKVSDied").errorUnsuppressed(e);
result = -1;
}
return result;
}
#if defined(_WIN32) || defined(__APPLE__) || defined(__INTEL_COMPILER)
ACTOR Future<int> spawnProcess(std::string binPath,
std::vector<std::string> paramList,
double maxWaitTime,
bool isSync,
double maxSimDelayTime) {
double maxSimDelayTime,
IClosable* parent) {
if (g_network->isSimulated() && getExecPath() == binPath) {
int res = wait(spawnSimulated(paramList, maxWaitTime, isSync, maxSimDelayTime, parent));
return res;
}
wait(delay(0.0));
return 0;
}
@ -125,6 +264,9 @@ static auto fork_child(const std::string& path, std::vector<char*>& paramList) {
}
static void setupTraceWithOutput(TraceEvent& event, size_t bytesRead, char* outputBuffer) {
// get some errors printed for spawned process
std::cout << "Output bytesRead: " << bytesRead << std::endl;
std::cout << "output buffer: " << std::string(outputBuffer) << std::endl;
if (bytesRead == 0)
return;
ASSERT(bytesRead <= SERVER_KNOBS->MAX_FORKED_PROCESS_OUTPUT);
@ -139,7 +281,12 @@ ACTOR Future<int> spawnProcess(std::string path,
std::vector<std::string> args,
double maxWaitTime,
bool isSync,
double maxSimDelayTime) {
double maxSimDelayTime,
IClosable* parent) {
if (g_network->isSimulated() && getExecPath() == path) {
int res = wait(spawnSimulated(args, maxWaitTime, isSync, maxSimDelayTime, parent));
return res;
}
// for async calls in simulator, always delay by a deterministic amount of time and then
// do the call synchronously, otherwise the predictability of the simulator breaks
if (!isSync && g_network->isSimulated()) {
@ -182,7 +329,7 @@ ACTOR Future<int> spawnProcess(std::string path,
int flags = fcntl(readFD.get(), F_GETFL, 0);
fcntl(readFD.get(), F_SETFL, flags | O_NONBLOCK);
while (true) {
if (runTime > maxWaitTime) {
if (maxWaitTime >= 0 && runTime > maxWaitTime) {
// timing out
TraceEvent(SevWarnAlways, "SpawnProcessFailure")
@ -203,7 +350,6 @@ ACTOR Future<int> spawnProcess(std::string path,
break;
bytesRead += bytes;
}
if (err < 0) {
TraceEvent event(SevWarnAlways, "SpawnProcessFailure");
setupTraceWithOutput(event, bytesRead, outputBuffer);

View File

@ -63,16 +63,19 @@ private: // data
StringRef binaryPath;
};
class IClosable; // Forward declaration
// FIXME: move this function to a common location
// spawns a process pointed by `binPath` and the arguments provided at `paramList`,
// if the process spawned takes more than `maxWaitTime` then it will be killed
// if isSync is set to true then the process will be synchronously executed
// if async and in simulator then delay spawning the process to max of maxSimDelayTime
// if the process spawned takes more than `maxWaitTime` then it will be killed, if `maxWaitTime` < 0, then there won't
// be timeout if isSync is set to true then the process will be synchronously executed if async and in simulator then
// delay spawning the process to max of maxSimDelayTime
ACTOR Future<int> spawnProcess(std::string binPath,
std::vector<std::string> paramList,
double maxWaitTime,
bool isSync,
double maxSimDelayTime);
double maxSimDelayTime,
IClosable* parent = nullptr);
// helper to run all the work related to running the exec command
ACTOR Future<int> execHelper(ExecCmdValueString* execArg, UID snapUID, std::string folder, std::string role);

View File

@ -159,12 +159,23 @@ extern IKeyValueStore* keyValueStoreLogSystem(class IDiskQueue* queue,
bool replaceContent,
bool exactRecovery);
extern IKeyValueStore* openRemoteKVStore(KeyValueStoreType storeType,
std::string const& filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums = false,
bool checkIntegrity = false);
inline IKeyValueStore* openKVStore(KeyValueStoreType storeType,
std::string const& filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums = false,
bool checkIntegrity = false) {
bool checkIntegrity = false,
bool openRemotely = false) {
if (openRemotely) {
return openRemoteKVStore(storeType, filename, logID, memoryLimit, checkChecksums, checkIntegrity);
}
switch (storeType) {
case KeyValueStoreType::SSD_BTREE_V1:
return keyValueStoreSQLite(filename, logID, KeyValueStoreType::SSD_BTREE_V1, false, checkIntegrity);

View File

@ -147,6 +147,7 @@ private:
};
using DB = rocksdb::DB*;
using CF = rocksdb::ColumnFamilyHandle*;
std::shared_ptr<rocksdb::Cache> rocksdb_block_cache = nullptr;
#define PERSIST_PREFIX "\xff\xff"
const KeyRef persistVersion = LiteralStringRef(PERSIST_PREFIX "Version");
@ -288,7 +289,10 @@ rocksdb::ColumnFamilyOptions getCFOptions() {
}
if (SERVER_KNOBS->ROCKSDB_BLOCK_CACHE_SIZE > 0) {
bbOpts.block_cache = rocksdb::NewLRUCache(SERVER_KNOBS->ROCKSDB_BLOCK_CACHE_SIZE);
if (rocksdb_block_cache == nullptr) {
rocksdb_block_cache = rocksdb::NewLRUCache(SERVER_KNOBS->ROCKSDB_BLOCK_CACHE_SIZE);
}
bbOpts.block_cache = rocksdb_block_cache;
}
options.table_factory.reset(rocksdb::NewBlockBasedTableFactory(bbOpts));

View File

@ -121,7 +121,8 @@ public:
newServers[serverId] = ssi;
if (oldServers.count(serverId)) {
if (ssi.getValue.getEndpoint() != oldServers[serverId].getValue.getEndpoint()) {
if (ssi.getValue.getEndpoint() != oldServers[serverId].getValue.getEndpoint() ||
ssi.isAcceptingRequests() != oldServers[serverId].isAcceptingRequests()) {
serverChanges.send(std::make_pair(serverId, Optional<StorageServerInterface>(ssi)));
}
oldServers.erase(serverId);
@ -158,6 +159,7 @@ public:
StorageQueuingMetricsRequest(), 0, 0)); // SOMEDAY: or tryGetReply?
if (reply.present()) {
myQueueInfo->value.update(reply.get(), self->smoothTotalDurableBytes);
myQueueInfo->value.acceptingRequests = ssi.isAcceptingRequests();
} else {
if (myQueueInfo->value.valid) {
TraceEvent("RkStorageServerDidNotRespond", self->id).detail("StorageServer", ssi.id());
@ -487,7 +489,7 @@ void Ratekeeper::updateRate(RatekeeperLimits* limits) {
// Look at each storage server's write queue and local rate, compute and store the desired rate ratio
for (auto i = storageQueueInfo.begin(); i != storageQueueInfo.end(); ++i) {
auto const& ss = i->value;
if (!ss.valid || (remoteDC.present() && ss.locality.dcId() == remoteDC))
if (!ss.valid || !ss.acceptingRequests || (remoteDC.present() && ss.locality.dcId() == remoteDC))
continue;
++sscount;
@ -941,7 +943,7 @@ ACTOR Future<Void> ratekeeper(RatekeeperInterface rkInterf, Reference<AsyncVar<S
StorageQueueInfo::StorageQueueInfo(UID id, LocalityData locality)
: busiestWriteTagEventHolder(makeReference<EventCacheHolder>(id.toString() + "/BusiestWriteTag")), valid(false),
id(id), locality(locality), smoothDurableBytes(SERVER_KNOBS->SMOOTHING_AMOUNT),
id(id), locality(locality), acceptingRequests(false), smoothDurableBytes(SERVER_KNOBS->SMOOTHING_AMOUNT),
smoothInputBytes(SERVER_KNOBS->SMOOTHING_AMOUNT), verySmoothDurableBytes(SERVER_KNOBS->SLOW_SMOOTHING_AMOUNT),
smoothDurableVersion(SERVER_KNOBS->SMOOTHING_AMOUNT), smoothLatestVersion(SERVER_KNOBS->SMOOTHING_AMOUNT),
smoothFreeSpace(SERVER_KNOBS->SMOOTHING_AMOUNT), smoothTotalSpace(SERVER_KNOBS->SMOOTHING_AMOUNT),

View File

@ -59,6 +59,7 @@ public:
UID id;
LocalityData locality;
StorageQueuingMetricsReply lastReply;
bool acceptingRequests;
Smoother smoothDurableBytes, smoothInputBytes, verySmoothDurableBytes;
Smoother smoothDurableVersion, smoothLatestVersion;
Smoother smoothFreeSpace;

View File

@ -0,0 +1,246 @@
/*
* RemoteIKeyValueStore.actor.cpp
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "flow/ActorCollection.h"
#include "flow/Error.h"
#include "flow/Platform.h"
#include "flow/Trace.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/fdbrpc.h"
#include "fdbclient/FDBTypes.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/Knobs.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#include "flow/actorcompiler.h" // This must be the last #include.
StringRef KeyValueStoreProcess::_name = "KeyValueStoreProcess"_sr;
// A guard for guaranteed killing of machine after runIKVS returns
struct AfterReturn {
IKeyValueStore* kvStore;
UID id;
AfterReturn() : kvStore(nullptr) {}
AfterReturn(IKeyValueStore* store, UID& uid) : kvStore(store), id(uid) {}
~AfterReturn() {
TraceEvent(SevDebug, "RemoteKVStoreAfterReturn")
.detail("Valid", kvStore != nullptr ? "True" : "False")
.detail("UID", id)
.log();
if (kvStore != nullptr) {
kvStore->close();
}
}
// called when we already explicitly closed the kv store
void invalidate() { kvStore = nullptr; }
};
ACTOR void sendCommitReply(IKVSCommitRequest commitReq, IKeyValueStore* kvStore, Future<Void> onClosed) {
try {
choose {
when(wait(onClosed)) { commitReq.reply.sendError(remote_kvs_cancelled()); }
when(wait(kvStore->commit(commitReq.sequential))) {
StorageBytes storageBytes = kvStore->getStorageBytes();
commitReq.reply.send(IKVSCommitReply(storageBytes));
}
}
} catch (Error& e) {
TraceEvent(SevDebug, "RemoteKVSCommitReplyError").errorUnsuppressed(e);
commitReq.reply.sendError(e.code() == error_code_actor_cancelled ? remote_kvs_cancelled() : e);
}
}
ACTOR template <class T>
Future<Void> cancellableForwardPromise(ReplyPromise<T> output, Future<T> input) {
try {
T value = wait(input);
output.send(value);
} catch (Error& e) {
TraceEvent(SevDebug, "CancellableForwardPromiseError").errorUnsuppressed(e).backtrace();
output.sendError(e.code() == error_code_actor_cancelled ? remote_kvs_cancelled() : e);
}
return Void();
}
ACTOR Future<Void> runIKVS(OpenKVStoreRequest openReq, IKVSInterface ikvsInterface) {
state IKeyValueStore* kvStore = openKVStore(openReq.storeType,
openReq.filename,
openReq.logID,
openReq.memoryLimit,
openReq.checkChecksums,
openReq.checkIntegrity);
state UID kvsId(ikvsInterface.id());
state ActorCollection actors(false);
state AfterReturn guard(kvStore, kvsId);
state Promise<Void> onClosed;
TraceEvent(SevDebug, "RemoteKVStoreInitializing").detail("UID", kvsId);
wait(kvStore->init());
openReq.reply.send(ikvsInterface);
TraceEvent(SevInfo, "RemoteKVStoreInitialized").detail("IKVSInterfaceUID", kvsId);
loop {
try {
choose {
when(IKVSGetValueRequest getReq = waitNext(ikvsInterface.getValue.getFuture())) {
actors.add(cancellableForwardPromise(getReq.reply,
kvStore->readValue(getReq.key, getReq.type, getReq.debugID)));
}
when(IKVSSetRequest req = waitNext(ikvsInterface.set.getFuture())) { kvStore->set(req.keyValue); }
when(IKVSClearRequest req = waitNext(ikvsInterface.clear.getFuture())) { kvStore->clear(req.range); }
when(IKVSCommitRequest commitReq = waitNext(ikvsInterface.commit.getFuture())) {
sendCommitReply(commitReq, kvStore, onClosed.getFuture());
}
when(IKVSReadValuePrefixRequest readPrefixReq = waitNext(ikvsInterface.readValuePrefix.getFuture())) {
actors.add(cancellableForwardPromise(
readPrefixReq.reply,
kvStore->readValuePrefix(
readPrefixReq.key, readPrefixReq.maxLength, readPrefixReq.type, readPrefixReq.debugID)));
}
when(IKVSReadRangeRequest readRangeReq = waitNext(ikvsInterface.readRange.getFuture())) {
actors.add(cancellableForwardPromise(
readRangeReq.reply,
fmap(
[](const RangeResult& result) { return IKVSReadRangeReply(result); },
kvStore->readRange(
readRangeReq.keys, readRangeReq.rowLimit, readRangeReq.byteLimit, readRangeReq.type))));
}
when(IKVSGetStorageByteRequest req = waitNext(ikvsInterface.getStorageBytes.getFuture())) {
StorageBytes storageBytes = kvStore->getStorageBytes();
req.reply.send(storageBytes);
}
when(IKVSGetErrorRequest getFutureReq = waitNext(ikvsInterface.getError.getFuture())) {
actors.add(cancellableForwardPromise(getFutureReq.reply, kvStore->getError()));
}
when(IKVSOnClosedRequest onClosedReq = waitNext(ikvsInterface.onClosed.getFuture())) {
// onClosed request is not cancelled even this actor is cancelled
forwardPromise(onClosedReq.reply, kvStore->onClosed());
}
when(IKVSDisposeRequest disposeReq = waitNext(ikvsInterface.dispose.getFuture())) {
TraceEvent(SevDebug, "RemoteIKVSDisposeReceivedRequest").detail("UID", kvsId);
kvStore->dispose();
guard.invalidate();
onClosed.send(Void());
return Void();
}
when(IKVSCloseRequest closeReq = waitNext(ikvsInterface.close.getFuture())) {
TraceEvent(SevDebug, "RemoteIKVSCloseReceivedRequest").detail("UID", kvsId);
kvStore->close();
guard.invalidate();
onClosed.send(Void());
return Void();
}
}
} catch (Error& e) {
if (e.code() == error_code_actor_cancelled) {
TraceEvent(SevInfo, "RemoteKVStoreCancelled").detail("UID", kvsId).backtrace();
onClosed.send(Void());
return Void();
} else {
TraceEvent(SevError, "RemoteKVStoreError").error(e).detail("UID", kvsId).backtrace();
throw;
}
}
}
}
ACTOR static Future<int> flowProcessRunner(RemoteIKeyValueStore* self, Promise<Void> ready) {
state FlowProcessInterface processInterface;
state Future<int> process;
auto path = abspath(getExecPath());
auto endpoint = processInterface.registerProcess.getEndpoint();
auto address = endpoint.addresses.address.toString();
auto token = endpoint.token;
// port 0 means we will find a random available port number for it
std::string flowProcessAddr = g_network->getLocalAddress().ip.toString().append(":0");
std::vector<std::string> args = { "bin/fdbserver",
"-r",
"flowprocess",
"-C",
SERVER_KNOBS->CONN_FILE,
"--logdir",
SERVER_KNOBS->LOG_DIRECTORY,
"-p",
flowProcessAddr,
"--process-name",
KeyValueStoreProcess::_name.toString(),
"--process-endpoint",
format("%s,%lu,%lu", address.c_str(), token.first(), token.second()) };
// For remote IKV store, we need to make sure the shutdown signal is sent back until we can destroy it in the
// simulation
process = spawnProcess(path, args, -1.0, false, 0.01 /*not used*/, self);
choose {
when(FlowProcessRegistrationRequest req = waitNext(processInterface.registerProcess.getFuture())) {
self->consumeInterface(req.flowProcessInterface);
ready.send(Void());
}
when(int res = wait(process)) {
// 0 means process normally shut down; non-zero means errors
// process should not shut down normally before not ready
ASSERT(res);
return res;
}
}
int res = wait(process);
return res;
}
ACTOR static Future<Void> initializeRemoteKVStore(RemoteIKeyValueStore* self, OpenKVStoreRequest openKVSReq) {
TraceEvent(SevInfo, "WaitingOnFlowProcess").detail("StoreType", openKVSReq.storeType).log();
Promise<Void> ready;
self->returnCode = flowProcessRunner(self, ready);
wait(ready.getFuture());
IKVSInterface ikvsInterface = wait(self->kvsProcess.openKVStore.getReply(openKVSReq));
TraceEvent(SevInfo, "IKVSInterfaceReceived").detail("UID", ikvsInterface.id());
self->interf = ikvsInterface;
self->interf.storeType = openKVSReq.storeType;
return Void();
}
IKeyValueStore* openRemoteKVStore(KeyValueStoreType storeType,
std::string const& filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums,
bool checkIntegrity) {
RemoteIKeyValueStore* self = new RemoteIKeyValueStore();
self->initialized = initializeRemoteKVStore(
self, OpenKVStoreRequest(storeType, filename, logID, memoryLimit, checkChecksums, checkIntegrity));
return self;
}
ACTOR static Future<Void> delayFlowProcessRunAction(FlowProcess* self, double time) {
wait(delay(time));
wait(self->run());
return Void();
}
Future<Void> runFlowProcess(std::string const& name, Endpoint endpoint) {
TraceEvent(SevInfo, "RunFlowProcessStart").log();
FlowProcess* self = IProcessFactory::create(name.c_str());
self->registerEndpoint(endpoint);
RequestStream<FlowProcessRegistrationRequest> registerProcess(endpoint);
FlowProcessRegistrationRequest req;
req.flowProcessInterface = self->serializedInterface();
registerProcess.send(req);
TraceEvent(SevDebug, "FlowProcessInitFinished").log();
return delayFlowProcessRunAction(self, g_network->isSimulated() ? 0 : SERVER_KNOBS->REMOTE_KV_STORE_INIT_DELAY);
}

View File

@ -0,0 +1,504 @@
/*
* RemoteIKeyValueStore.actor.h
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#if defined(NO_INTELLISENSE) && !defined(FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_G_H)
#define FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_G_H
#include "fdbserver/RemoteIKeyValueStore.actor.g.h"
#elif !defined(FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_H)
#define FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_H
#include "flow/ActorCollection.h"
#include "flow/IRandom.h"
#include "flow/Knobs.h"
#include "flow/Trace.h"
#include "flow/flow.h"
#include "flow/network.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/FlowTransport.h"
#include "fdbrpc/fdbrpc.h"
#include "fdbclient/FDBTypes.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/IKeyValueStore.h"
#include "fdbserver/Knobs.h"
#include "flow/actorcompiler.h" // This must be the last #include.
struct IKVSCommitReply {
constexpr static FileIdentifier file_identifier = 3958189;
StorageBytes storeBytes;
IKVSCommitReply() : storeBytes(0, 0, 0, 0) {}
IKVSCommitReply(const StorageBytes& sb) : storeBytes(sb) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, storeBytes);
}
};
struct RemoteKVSProcessInterface {
constexpr static FileIdentifier file_identifier = 3491838;
RequestStream<struct GetRemoteKVSProcessInterfaceRequest> getProcessInterface;
RequestStream<struct OpenKVStoreRequest> openKVStore;
UID uniqueID = deterministicRandom()->randomUniqueID();
UID id() const { return uniqueID; }
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, getProcessInterface, openKVStore);
}
};
struct IKVSInterface {
constexpr static FileIdentifier file_identifier = 4929113;
RequestStream<struct IKVSGetValueRequest> getValue;
RequestStream<struct IKVSSetRequest> set;
RequestStream<struct IKVSClearRequest> clear;
RequestStream<struct IKVSCommitRequest> commit;
RequestStream<struct IKVSReadValuePrefixRequest> readValuePrefix;
RequestStream<struct IKVSReadRangeRequest> readRange;
RequestStream<struct IKVSGetStorageByteRequest> getStorageBytes;
RequestStream<struct IKVSGetErrorRequest> getError;
RequestStream<struct IKVSOnClosedRequest> onClosed;
RequestStream<struct IKVSDisposeRequest> dispose;
RequestStream<struct IKVSCloseRequest> close;
UID uniqueID;
UID id() const { return uniqueID; }
KeyValueStoreType storeType;
KeyValueStoreType type() const { return storeType; }
IKVSInterface() {}
IKVSInterface(KeyValueStoreType type) : uniqueID(deterministicRandom()->randomUniqueID()), storeType(type) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar,
getValue,
set,
clear,
commit,
readValuePrefix,
readRange,
getStorageBytes,
getError,
onClosed,
dispose,
close,
uniqueID);
}
};
struct GetRemoteKVSProcessInterfaceRequest {
constexpr static FileIdentifier file_identifier = 8382983;
ReplyPromise<struct RemoteKVSProcessInterface> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct OpenKVStoreRequest {
constexpr static FileIdentifier file_identifier = 5918682;
KeyValueStoreType storeType;
std::string filename;
UID logID;
int64_t memoryLimit;
bool checkChecksums;
bool checkIntegrity;
ReplyPromise<struct IKVSInterface> reply;
OpenKVStoreRequest(){};
OpenKVStoreRequest(KeyValueStoreType storeType,
std::string filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums = false,
bool checkIntegrity = false)
: storeType(storeType), filename(filename), logID(logID), memoryLimit(memoryLimit),
checkChecksums(checkChecksums), checkIntegrity(checkIntegrity) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, storeType, filename, logID, memoryLimit, checkChecksums, checkIntegrity, reply);
}
};
struct IKVSGetValueRequest {
constexpr static FileIdentifier file_identifier = 1029439;
KeyRef key;
IKeyValueStore::ReadType type;
Optional<UID> debugID = Optional<UID>();
ReplyPromise<Optional<Value>> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, key, type, debugID, reply);
}
};
struct IKVSSetRequest {
constexpr static FileIdentifier file_identifier = 7283948;
KeyValueRef keyValue;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, keyValue, reply);
}
};
struct IKVSClearRequest {
constexpr static FileIdentifier file_identifier = 2838575;
KeyRangeRef range;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, range, reply);
}
};
struct IKVSCommitRequest {
constexpr static FileIdentifier file_identifier = 2985129;
bool sequential;
ReplyPromise<IKVSCommitReply> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, sequential, reply);
}
};
struct IKVSReadValuePrefixRequest {
constexpr static FileIdentifier file_identifier = 1928374;
KeyRef key;
int maxLength;
IKeyValueStore::ReadType type;
Optional<UID> debugID = Optional<UID>();
ReplyPromise<Optional<Value>> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, key, maxLength, type, debugID, reply);
}
};
// Use this instead of RangeResult as reply for better serialization performance
struct IKVSReadRangeReply {
constexpr static FileIdentifier file_identifier = 6682449;
Arena arena;
VectorRef<KeyValueRef, VecSerStrategy::String> data;
bool more;
Optional<KeyRef> readThrough;
bool readToBegin;
bool readThroughEnd;
IKVSReadRangeReply() = default;
explicit IKVSReadRangeReply(const RangeResult& res)
: arena(res.arena()), data(static_cast<const VectorRef<KeyValueRef>&>(res)), more(res.more),
readThrough(res.readThrough), readToBegin(res.readToBegin), readThroughEnd(res.readThroughEnd) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, data, more, readThrough, readToBegin, readThroughEnd, arena);
}
RangeResult toRangeResult() const {
RangeResult r(RangeResultRef(data, more, readThrough), arena);
r.readToBegin = readToBegin;
r.readThroughEnd = readThroughEnd;
return r;
}
};
struct IKVSReadRangeRequest {
constexpr static FileIdentifier file_identifier = 5918394;
KeyRangeRef keys;
int rowLimit;
int byteLimit;
IKeyValueStore::ReadType type;
ReplyPromise<IKVSReadRangeReply> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, keys, rowLimit, byteLimit, type, reply);
}
};
struct IKVSGetStorageByteRequest {
constexpr static FileIdentifier file_identifier = 3512344;
ReplyPromise<StorageBytes> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct IKVSGetErrorRequest {
constexpr static FileIdentifier file_identifier = 3942891;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct IKVSOnClosedRequest {
constexpr static FileIdentifier file_identifier = 1923894;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct IKVSDisposeRequest {
constexpr static FileIdentifier file_identifier = 1235952;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar);
}
};
struct IKVSCloseRequest {
constexpr static FileIdentifier file_identifier = 13859172;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar);
}
};
ACTOR Future<Void> runIKVS(OpenKVStoreRequest openReq, IKVSInterface ikvsInterface);
struct KeyValueStoreProcess : FlowProcess {
RemoteKVSProcessInterface kvsIf;
Standalone<StringRef> serializedIf;
Endpoint ssProcess; // endpoint for the storage process
RequestStream<FlowProcessRegistrationRequest> ssRequestStream;
KeyValueStoreProcess() {
TraceEvent(SevDebug, "InitKeyValueStoreProcess").log();
ObjectWriter writer(IncludeVersion());
writer.serialize(kvsIf);
serializedIf = writer.toString();
}
void registerEndpoint(Endpoint p) override {
ssProcess = p;
ssRequestStream = RequestStream<FlowProcessRegistrationRequest>(p);
}
StringRef name() const override { return _name; }
StringRef serializedInterface() const override { return serializedIf; }
ACTOR static Future<Void> _run(KeyValueStoreProcess* self) {
state ActorCollection actors(true);
TraceEvent("WaitingForOpenKVStoreRequest").log();
loop {
choose {
when(OpenKVStoreRequest req = waitNext(self->kvsIf.openKVStore.getFuture())) {
TraceEvent("OpenKVStoreRequestReceived").log();
IKVSInterface interf;
actors.add(runIKVS(req, interf));
}
when(ErrorOr<Void> e = wait(errorOr(actors.getResult()))) {
if (e.isError()) {
TraceEvent("KeyValueStoreProcessRunActorError").errorUnsuppressed(e.getError());
throw e.getError();
} else {
TraceEvent("KeyValueStoreProcessFinished").log();
return e.get();
}
}
}
}
}
Future<Void> run() override { return _run(this); }
static StringRef _name;
};
struct RemoteIKeyValueStore : public IKeyValueStore {
RemoteKVSProcessInterface kvsProcess;
IKVSInterface interf;
Future<Void> initialized;
Future<int> returnCode;
StorageBytes storageBytes;
RemoteIKeyValueStore() : storageBytes(0, 0, 0, 0) {}
Future<Void> init() override {
TraceEvent(SevInfo, "RemoteIKeyValueStoreInit").log();
return initialized;
}
Future<Void> getError() const override { return getErrorImpl(this, returnCode); }
Future<Void> onClosed() const override { return onCloseImpl(this); }
void dispose() override {
TraceEvent(SevDebug, "RemoteIKVSDisposeRequest").backtrace();
interf.dispose.send(IKVSDisposeRequest{});
// hold the future to not cancel the spawned process
uncancellable(returnCode);
delete this;
}
void close() override {
TraceEvent(SevDebug, "RemoteIKVSCloseRequest").backtrace();
interf.close.send(IKVSCloseRequest{});
// hold the future to not cancel the spawned process
uncancellable(returnCode);
delete this;
}
KeyValueStoreType getType() const override { return interf.type(); }
void set(KeyValueRef keyValue, const Arena* arena = nullptr) override {
interf.set.send(IKVSSetRequest{ keyValue, ReplyPromise<Void>() });
}
void clear(KeyRangeRef range, const Arena* arena = nullptr) override {
interf.clear.send(IKVSClearRequest{ range, ReplyPromise<Void>() });
}
Future<Void> commit(bool sequential = false) override {
Future<IKVSCommitReply> commitReply =
interf.commit.getReply(IKVSCommitRequest{ sequential, ReplyPromise<IKVSCommitReply>() });
return commitAndGetStorageBytes(this, commitReply);
}
Future<Optional<Value>> readValue(KeyRef key,
ReadType type = ReadType::NORMAL,
Optional<UID> debugID = Optional<UID>()) override {
return readValueImpl(this, IKVSGetValueRequest{ key, type, debugID, ReplyPromise<Optional<Value>>() });
}
Future<Optional<Value>> readValuePrefix(KeyRef key,
int maxLength,
ReadType type = ReadType::NORMAL,
Optional<UID> debugID = Optional<UID>()) override {
return interf.readValuePrefix.getReply(
IKVSReadValuePrefixRequest{ key, maxLength, type, debugID, ReplyPromise<Optional<Value>>() });
}
Future<RangeResult> readRange(KeyRangeRef keys,
int rowLimit = 1 << 30,
int byteLimit = 1 << 30,
ReadType type = ReadType::NORMAL) override {
IKVSReadRangeRequest req{ keys, rowLimit, byteLimit, type, ReplyPromise<IKVSReadRangeReply>() };
return fmap([](const IKVSReadRangeReply& reply) { return reply.toRangeResult(); },
interf.readRange.getReply(req));
}
StorageBytes getStorageBytes() const override { return storageBytes; }
void consumeInterface(StringRef intf) {
kvsProcess = ObjectReader::fromStringRef<RemoteKVSProcessInterface>(intf, IncludeVersion());
}
ACTOR static Future<Void> commitAndGetStorageBytes(RemoteIKeyValueStore* self,
Future<IKVSCommitReply> commitReplyFuture) {
IKVSCommitReply commitReply = wait(commitReplyFuture);
self->storageBytes = commitReply.storeBytes;
return Void();
}
ACTOR static Future<Optional<Value>> readValueImpl(RemoteIKeyValueStore* self, IKVSGetValueRequest req) {
Optional<Value> val = wait(self->interf.getValue.getReply(req));
return val;
}
ACTOR static Future<Void> getErrorImpl(const RemoteIKeyValueStore* self, Future<int> returnCode) {
choose {
when(wait(self->initialized)) {}
when(wait(delay(SERVER_KNOBS->REMOTE_KV_STORE_MAX_INIT_DURATION))) {
TraceEvent(SevError, "RemoteIKVSInitTooLong")
.detail("TimeLimit", SERVER_KNOBS->REMOTE_KV_STORE_MAX_INIT_DURATION);
throw please_reboot_remote_kv_store();
}
}
state Future<Void> connectionCheckingDelay = delay(FLOW_KNOBS->FAILURE_DETECTION_DELAY);
state Future<ErrorOr<Void>> storeError = errorOr(self->interf.getError.getReply(IKVSGetErrorRequest{}));
loop choose {
when(ErrorOr<Void> e = wait(storeError)) {
TraceEvent(SevDebug, "RemoteIKVSGetError")
.errorUnsuppressed(e.isError() ? e.getError() : success())
.backtrace();
if (e.isError())
throw e.getError();
else
return e.get();
}
when(int res = wait(returnCode)) {
TraceEvent(res != 0 ? SevError : SevInfo, "SpawnedProcessDied").detail("Res", res);
if (res)
throw please_reboot_remote_kv_store(); // this will reboot the worker
else
return Void();
}
when(wait(connectionCheckingDelay)) {
// for the corner case where the child process stuck and waitpid also does not give update on it
// In this scenario, we need to manually reboot the storage engine process
if (IFailureMonitor::failureMonitor()
.getState(self->interf.getError.getEndpoint().getPrimaryAddress())
.isFailed()) {
TraceEvent(SevError, "RemoteKVStoreConnectionStuck").log();
throw please_reboot_remote_kv_store(); // this will reboot the worker
}
connectionCheckingDelay = delay(FLOW_KNOBS->FAILURE_DETECTION_DELAY);
}
}
}
ACTOR static Future<Void> onCloseImpl(const RemoteIKeyValueStore* self) {
try {
wait(self->initialized);
wait(self->interf.onClosed.getReply(IKVSOnClosedRequest{}));
TraceEvent(SevDebug, "RemoteIKVSOnCloseImplOnClosedFinished");
} catch (Error& e) {
TraceEvent(SevInfo, "RemoteIKVSOnCloseImplError").errorUnsuppressed(e).backtrace();
throw;
}
return Void();
}
};
Future<Void> runFlowProcess(std::string const& name, Endpoint endpoint);
#include "flow/unactorcompiler.h"
#endif

View File

@ -263,6 +263,9 @@ class TestConfig {
if (attrib == "disableHostname") {
disableHostname = strcmp(value.c_str(), "true") == 0;
}
if (attrib == "disableRemoteKVS") {
disableRemoteKVS = strcmp(value.c_str(), "true") == 0;
}
if (attrib == "restartInfoLocation") {
isFirstTestInRestart = true;
}
@ -298,6 +301,8 @@ public:
bool disableTss = false;
// 7.1 cannot be downgraded to 7.0 and below after enabling hostname, so disable hostname for 7.0 downgrade tests
bool disableHostname = false;
// remote key value store is a child process spawned by the SS process to run the storage engine
bool disableRemoteKVS = false;
// Storage Engine Types: Verify match with SimulationConfig::generateNormalConfig
// 0 = "ssd"
// 1 = "memory"
@ -357,6 +362,7 @@ public:
.add("maxTLogVersion", &maxTLogVersion)
.add("disableTss", &disableTss)
.add("disableHostname", &disableHostname)
.add("disableRemoteKVS", &disableRemoteKVS)
.add("simpleConfig", &simpleConfig)
.add("generateFearless", &generateFearless)
.add("datacenters", &datacenters)
@ -1084,6 +1090,11 @@ ACTOR Future<Void> restartSimulatedSystem(std::vector<Future<Void>>* systemActor
INetworkConnections::net()->parseMockDNSFromString(mockDNSStr);
}
}
if (testConfig.disableRemoteKVS) {
IKnobCollection::getMutableGlobalKnobCollection().setKnob("remote_kv_store",
KnobValueRef::create(bool{ false }));
TraceEvent(SevDebug, "DisaableRemoteKVS").log();
}
*pConnString = conn;
*pTesterCount = testerCount;
bool usingSSL = conn.toString().find(":tls") != std::string::npos || listenersPerProcess > 1;
@ -1836,6 +1847,11 @@ void setupSimulatedSystem(std::vector<Future<Void>>* systemActors,
if (testConfig.configureLocked) {
startingConfigString += " locked";
}
if (testConfig.disableRemoteKVS) {
IKnobCollection::getMutableGlobalKnobCollection().setKnob("remote_kv_store",
KnobValueRef::create(bool{ false }));
TraceEvent(SevDebug, "DisaableRemoteKVS").log();
}
auto configDBType = testConfig.getConfigDBType();
for (auto kv : startingConfigJSON) {
if ("tss_storage_engine" == kv.first) {

View File

@ -62,7 +62,7 @@
{ \
std::string prefix = format("%s %f %04d ", g_network->getLocalAddress().toString().c_str(), now(), __LINE__); \
std::string msg = format(__VA_ARGS__); \
writePrefixedLines(debug_printf_stream, prefix, msg); \
fputs(addPrefix(prefix, msg).c_str(), debug_printf_stream); \
fflush(debug_printf_stream); \
}
@ -73,11 +73,13 @@
std::string prefix = \
format("%s %f %04d ", g_network->getLocalAddress().toString().c_str(), now(), __LINE__); \
std::string msg = format(__VA_ARGS__); \
writePrefixedLines(debug_printf_stream, prefix, msg); \
fputs(addPrefix(prefix, msg).c_str(), debug_printf_stream); \
fflush(debug_printf_stream); \
} \
}
#define debug_print(str) debug_printf("%s\n", str.c_str())
#define debug_print_always(str) debug_printf_always("%s\n", str.c_str())
#define debug_printf_noop(...)
#if defined(NO_INTELLISENSE)
@ -97,13 +99,18 @@
#define TRACE \
debug_printf_always("%s: %s line %d %s\n", __FUNCTION__, __FILE__, __LINE__, platform::get_backtrace().c_str());
// Writes prefix:line for each line in msg to fout
void writePrefixedLines(FILE* fout, std::string prefix, std::string msg) {
StringRef m = msg;
// Returns a string where every line in lines is prefixed with prefix
std::string addPrefix(std::string prefix, std::string lines) {
StringRef m = lines;
std::string s;
while (m.size() != 0) {
StringRef line = m.eat("\n");
fprintf(fout, "%s %s\n", prefix.c_str(), line.toString().c_str());
s += prefix;
s += ' ';
s += line.toString();
s += '\n';
}
return s;
}
#define PRIORITYMULTILOCK_DEBUG 0
@ -917,12 +924,15 @@ public:
}
}
// If readNext() cannot complete immediately, it will route to here
// The mutex will be taken if locked is false
// The next page will be waited for if load is true
// If readNext() cannot complete immediately because it must wait for IO, it will route to here.
// The purpose of this function is to serialize simultaneous readers on self while letting the
// common case (>99.8% of the time) be handled with low overhead by the non-actor readNext() function.
//
// The mutex will be taken if locked is false.
// The next page will be waited for if load is true.
// Only mutex holders will wait on the page read.
ACTOR static Future<Optional<T>> waitThenReadNext(Cursor* self,
Optional<T> upperBound,
Optional<T> inclusiveMaximum,
FlowMutex::Lock* lock,
bool load) {
state FlowMutex::Lock localLock;
@ -940,7 +950,7 @@ public:
wait(success(self->nextPageReader));
}
state Optional<T> result = wait(self->readNext(upperBound, &localLock));
state Optional<T> result = wait(self->readNext(inclusiveMaximum, &localLock));
// If a lock was not passed in, so this actor locked the mutex above, then unlock it
if (lock == nullptr) {
@ -959,10 +969,12 @@ public:
return result;
}
// Read the next item at the cursor (if < upperBound), moving to a new page first if the current page is
// exhausted If locked is true, this call owns the mutex, which would have been locked by readNext() before a
// recursive call
Future<Optional<T>> readNext(const Optional<T>& upperBound = {}, FlowMutex::Lock* lock = nullptr) {
// Read the next item from the cursor, possibly moving to and waiting for a new page if the prior page was
// exhausted. If the item is <= inclusiveMaximum, then return it after advancing the cursor to the next item.
// Otherwise, return nothing and do not advance the cursor.
// If locked is true, this call owns the mutex, which would have been locked by readNext() before a recursive
// call. See waitThenReadNext() for more detail.
Future<Optional<T>> readNext(const Optional<T>& inclusiveMaximum = {}, FlowMutex::Lock* lock = nullptr) {
if ((mode != POP && mode != READONLY) || pageID == invalidLogicalPageID || pageID == endPageID) {
debug_printf("FIFOQueue::Cursor(%s) readNext returning nothing\n", toString().c_str());
return Optional<T>();
@ -970,7 +982,7 @@ public:
// If we don't have a lock and the mutex isn't available then acquire it
if (lock == nullptr && isBusy()) {
return waitThenReadNext(this, upperBound, lock, false);
return waitThenReadNext(this, inclusiveMaximum, lock, false);
}
// We now know pageID is valid and should be used, but page might not point to it yet
@ -986,7 +998,7 @@ public:
}
if (!nextPageReader.isReady()) {
return waitThenReadNext(this, upperBound, lock, true);
return waitThenReadNext(this, inclusiveMaximum, lock, true);
}
page = nextPageReader.get();
@ -1007,11 +1019,11 @@ public:
int bytesRead;
const T result = Codec::readFromBytes(p->begin() + offset, bytesRead);
if (upperBound.present() && upperBound.get() < result) {
if (inclusiveMaximum.present() && inclusiveMaximum.get() < result) {
debug_printf("FIFOQueue::Cursor(%s) not popping %s, exceeds upper bound %s\n",
toString().c_str(),
::toString(result).c_str(),
::toString(upperBound.get()).c_str());
::toString(inclusiveMaximum.get()).c_str());
return Optional<T>();
}
@ -1059,10 +1071,10 @@ public:
}
}
debug_printf("FIFOQueue(%s) %s(upperBound=%s) -> %s\n",
debug_printf("FIFOQueue(%s) %s(inclusiveMaximum=%s) -> %s\n",
queue->name.c_str(),
(mode == POP ? "pop" : "peek"),
::toString(upperBound).c_str(),
::toString(inclusiveMaximum).c_str(),
::toString(result).c_str());
return Optional<T>(result);
}
@ -1290,8 +1302,8 @@ public:
Future<Optional<T>> peek() { return peek_impl(this); }
// Pop the next item on front of queue if it is <= upperBound or if upperBound is not present
Future<Optional<T>> pop(Optional<T> upperBound = {}) { return headReader.readNext(upperBound); }
// Pop the next item on front of queue if it is <= inclusiveMaximum or if inclusiveMaximum is not present
Future<Optional<T>> pop(Optional<T> inclusiveMaximum = {}) { return headReader.readNext(inclusiveMaximum); }
QueueState getState() const {
QueueState s;
@ -1484,8 +1496,8 @@ public:
int64_t numEntries;
int dataBytesPerPage;
int pagesPerExtent;
bool usesExtents;
bool tailPageNewExtent;
bool usesExtents = false;
bool tailPageNewExtent = false;
LogicalPageID prevExtentEndPageID;
Cursor headReader;
@ -2758,6 +2770,8 @@ public:
return f;
}
// Free pageID as of version v. This means that once the oldest readable pager snapshot is at version v, pageID is
// not longer in use by any structure so it can be used to write new data.
void freeUnmappedPage(PhysicalPageID pageID, Version v) {
// If v is older than the oldest version still readable then mark pageID as free as of the next commit
if (v < effectiveOldestVersion()) {
@ -2823,7 +2837,7 @@ public:
void freePage(LogicalPageID pageID, Version v) override {
// If pageID has been remapped, then it can't be freed until all existing remaps for that page have been undone,
// so queue it for later deletion
// so queue it for later deletion during remap cleanup
auto i = remappedPages.find(pageID);
if (i != remappedPages.end()) {
debug_printf("DWALPager(%s) op=freeRemapped %s @%" PRId64 " oldestVersion=%" PRId64 "\n",
@ -3331,7 +3345,11 @@ public:
// Since the next item can be arbitrarily ahead in the queue, secondType is determined by
// looking at the remappedPages structure.
//
// R == Remap F == Free D == Detach | == oldestRetaineedVersion
// R == Remap F == Free D == Detach | == oldestRetainedVersion
//
// oldestRetainedVersion is the oldest version being maintained as readable, either because it is explicitly the
// oldest readable version set or because there is an active snapshot for the version even though it is older
// than the explicitly set oldest readable version.
//
// R R | free new ID
// R F | free new ID if R and D are at different versions
@ -3411,13 +3429,32 @@ public:
}
if (freeNewID) {
debug_printf("DWALPager(%s) remapCleanup freeNew %s\n", self->filename.c_str(), p.toString().c_str());
self->freeUnmappedPage(p.newPageID, 0);
debug_printf("DWALPager(%s) remapCleanup freeNew %s %s\n",
self->filename.c_str(),
p.toString().c_str(),
toString(self->getLastCommittedVersion()).c_str());
// newID must be freed at the latest committed version to avoid a read race between caching and non-caching
// readers. It is possible that there are readers of newID in flight right now that either
// - Did not read through the page cache
// - Did read through the page cache but there was no entry for the page at the time, so one was created
// and the read future is still pending
// In either case the physical read of newID from disk can happen at some time after right now and after the
// current commit is finished.
//
// If newID is freed immediately, meaning as of the end of the current commit, then it could be reused in
// the next commit which could be before any reads fitting the above description have completed, causing
// those reads to the new write which is incorrect. Since such readers could be using pager snapshots at
// versions up to and including the latest committed version, newID must be freed *after* that version is no
// longer readable.
self->freeUnmappedPage(p.newPageID, self->getLastCommittedVersion() + 1);
++g_redwoodMetrics.metric.pagerRemapFree;
}
if (freeOriginalID) {
debug_printf("DWALPager(%s) remapCleanup freeOriginal %s\n", self->filename.c_str(), p.toString().c_str());
// originalID can be freed immediately because it is already the case that there are no readers at a version
// prior to oldestRetainedVersion so no reader will need originalID.
self->freeUnmappedPage(p.originalPageID, 0);
++g_redwoodMetrics.metric.pagerRemapFree;
}
@ -3654,6 +3691,7 @@ public:
self->operations.clear();
debug_printf("DWALPager(%s) shutdown destroy page cache\n", self->filename.c_str());
wait(self->extentCache.clear());
wait(self->pageCache.clear());
wait(delay(0));
@ -4575,7 +4613,7 @@ struct BTreePage {
ValueTree* valueTree() const { return (ValueTree*)(this + 1); }
std::string toString(bool write,
std::string toString(const char* context,
BTreePageIDRef id,
Version ver,
const RedwoodRecordRef& lowerBound,
@ -4583,7 +4621,7 @@ struct BTreePage {
std::string r;
r += format("BTreePage op=%s %s @%" PRId64
" ptr=%p height=%d count=%d kvBytes=%d\n lowerBound: %s\n upperBound: %s\n",
write ? "write" : "read",
context,
::toString(id).c_str(),
ver,
this,
@ -4684,24 +4722,43 @@ struct DecodeBoundaryVerifier {
typedef std::map<Version, DecodeBoundaries> BoundariesByVersion;
std::unordered_map<LogicalPageID, BoundariesByVersion> boundariesByPageID;
std::vector<Key> boundarySamples;
int boundarySampleSize = 1000;
int boundaryPopulation = 0;
static DecodeBoundaryVerifier* getVerifier(std::string name) {
static std::map<std::string, DecodeBoundaryVerifier> verifiers;
// Verifier disabled due to not being finished
//
// Only use verifier in a non-restarted simulation so that all page writes are captured
// if (g_network->isSimulated() && !g_simulator.restarted) {
// return &verifiers[name];
// }
if (g_network->isSimulated() && !g_simulator.restarted) {
return &verifiers[name];
}
return nullptr;
}
void sampleBoundary(Key b) {
if (boundaryPopulation <= boundarySampleSize) {
boundarySamples.push_back(b);
} else if (deterministicRandom()->random01() < ((double)boundarySampleSize / boundaryPopulation)) {
boundarySamples[deterministicRandom()->randomInt(0, boundarySampleSize)] = b;
}
++boundaryPopulation;
}
Key getSample() const {
if (boundarySamples.empty()) {
return Key();
}
return boundarySamples[deterministicRandom()->randomInt(0, boundarySamples.size())];
}
void update(BTreePageIDRef id, Version v, Key lowerBound, Key upperBound) {
sampleBoundary(lowerBound);
sampleBoundary(upperBound);
debug_printf("decodeBoundariesUpdate %s %s '%s' to '%s'\n",
::toString(id).c_str(),
::toString(v).c_str(),
lowerBound.toString().c_str(),
upperBound.toString().c_str());
lowerBound.printable().c_str(),
upperBound.printable().c_str());
auto& b = boundariesByPageID[id.front()][v];
ASSERT(b.empty());
@ -4717,28 +4774,53 @@ struct DecodeBoundaryVerifier {
--b;
if (b->second.lower != lowerBound || b->second.upper != upperBound) {
fprintf(stderr,
"Boundary mismatch on %s %s\nFound :%s %s\nExpected:%s %s\n",
"Boundary mismatch on %s %s\nUsing:\n\t'%s'\n\t'%s'\nWritten %s:\n\t'%s'\n\t'%s'\n",
::toString(id).c_str(),
::toString(v).c_str(),
lowerBound.toString().c_str(),
upperBound.toString().c_str(),
b->second.lower.toString().c_str(),
b->second.upper.toString().c_str());
lowerBound.printable().c_str(),
upperBound.printable().c_str(),
::toString(b->first).c_str(),
b->second.lower.printable().c_str(),
b->second.upper.printable().c_str());
return false;
}
return true;
}
void update(Version v, LogicalPageID oldID, LogicalPageID newID) {
debug_printf("decodeBoundariesUpdate copy %s %s to %s\n",
::toString(v).c_str(),
::toString(oldID).c_str(),
::toString(newID).c_str());
auto& old = boundariesByPageID[oldID];
ASSERT(!old.empty());
auto i = old.end();
--i;
boundariesByPageID[newID][v] = i->second;
debug_printf("decodeBoundariesUpdate copy %s %s to %s '%s' to '%s'\n",
::toString(v).c_str(),
::toString(oldID).c_str(),
::toString(newID).c_str(),
i->second.lower.printable().c_str(),
i->second.upper.printable().c_str());
}
void removeAfterVersion(Version version) {
auto i = boundariesByPageID.begin();
while (i != boundariesByPageID.end()) {
auto v = i->second.upper_bound(version);
while (v != i->second.end()) {
debug_printf("decodeBoundariesUpdate remove %s %s '%s' to '%s'\n",
::toString(v->first).c_str(),
::toString(i->first).c_str(),
v->second.lower.printable().c_str(),
v->second.upper.printable().c_str());
v = i->second.erase(v);
}
if (i->second.empty()) {
debug_printf("decodeBoundariesUpdate remove empty map for %s\n", ::toString(i->first).c_str());
i = boundariesByPageID.erase(i);
} else {
++i;
}
}
}
};
@ -5024,8 +5106,14 @@ public:
self->m_newOldestVersion = self->m_pager->getOldestReadableVersion();
debug_printf("Recovered pager to version %" PRId64 ", oldest version is %" PRId64 "\n",
self->getLastCommittedVersion(),
self->m_newOldestVersion);
// Clear any changes that occurred after the latest committed version
if (self->m_pBoundaryVerifier != nullptr) {
self->m_pBoundaryVerifier->removeAfterVersion(self->getLastCommittedVersion());
}
state Key meta = self->m_pager->getMetaKey();
if (meta.size() == 0) {
// Create new BTree
@ -5825,10 +5913,17 @@ private:
const RedwoodRecordRef& lowerBound,
const RedwoodRecordRef& upperBound) {
if (page->userData == nullptr) {
debug_printf("Creating DecodeCache for ptr=%p lower=%s upper=%s\n",
debug_printf("Creating DecodeCache for ptr=%p lower=%s upper=%s %s\n",
page->begin(),
lowerBound.toString(false).c_str(),
upperBound.toString(false).c_str());
upperBound.toString(false).c_str(),
((BTreePage*)page->begin())
->toString("cursor",
lowerBound.value.present() ? lowerBound.getChildPage() : BTreePageIDRef(),
-1,
lowerBound,
upperBound)
.c_str());
BTreePage::BinaryTree::DecodeCache* cache =
new BTreePage::BinaryTree::DecodeCache(lowerBound, upperBound, m_pDecodeCacheMemory);
@ -5890,12 +5985,13 @@ private:
BTreePage* btPage = (BTreePage*)page->begin();
BTreePage::BinaryTree::DecodeCache* cache = (BTreePage::BinaryTree::DecodeCache*)page->userData;
debug_printf_always(
"updateBTreePage(%s, %s) %s\n",
"updateBTreePage(%s, %s) start, page:\n%s\n",
::toString(oldID).c_str(),
::toString(writeVersion).c_str(),
cache == nullptr
? "<noDecodeCache>"
: btPage->toString(true, oldID, writeVersion, cache->lowerBound, cache->upperBound).c_str());
: btPage->toString("updateBTreePage", oldID, writeVersion, cache->lowerBound, cache->upperBound)
.c_str());
}
state unsigned int height = (unsigned int)((BTreePage*)page->begin())->height;
@ -5912,7 +6008,11 @@ private:
LogicalPageID id = wait(self->m_pager->newPageID());
emptyPages[i] = id;
}
debug_printf("updateBTreePage: newPages %s", toString(emptyPages).c_str());
debug_printf("updateBTreePage(%s, %s): newPages %s",
::toString(oldID).c_str(),
::toString(writeVersion).c_str(),
toString(emptyPages).c_str());
self->m_pager->updatePage(PagerEventReasons::Commit, height, emptyPages, page);
i = 0;
for (const LogicalPageID id : emptyPages) {
@ -5956,13 +6056,15 @@ private:
RedwoodRecordRef decodeLowerBound;
RedwoodRecordRef decodeUpperBound;
// Returns true of BTree logical boundaries and DeltaTree decoding boundaries are the same.
bool boundariesNormal() const {
// If the decode upper boundary is the subtree upper boundary the pointers will be the same
// For the lower boundary, if the pointers are not the same there is still a possibility
// that the keys are the same. This happens for the first remaining subtree of an internal page
// after the prior subtree(s) were cleared.
return (decodeUpperBound == subtreeUpperBound) &&
(decodeLowerBound == subtreeLowerBound || decodeLowerBound.sameExceptValue(subtreeLowerBound));
// Often these strings will refer to the same memory so same() is used as a faster way of determining
// equality in thec common case, but if it does not match a string comparison is needed as they can
// still be the same. This can happen for the first remaining subtree of an internal page
// after all prior subtree(s) were cleared.
return (
(decodeUpperBound.key.same(subtreeUpperBound.key) || decodeUpperBound.key == subtreeUpperBound.key) &&
(decodeLowerBound.key.same(subtreeLowerBound.key) || decodeLowerBound.key == subtreeLowerBound.key));
}
// The record range of the subtree slice is cBegin to cEnd
@ -6026,6 +6128,7 @@ private:
// Set the child page ID, which has already been allocated in result.arena()
newLinks.back().setChildPage(maybeNewID);
childrenChanged = true;
expectedUpperBound = decodeUpperBound;
} else {
childrenChanged = false;
}
@ -6070,6 +6173,7 @@ private:
s += format("SubtreeUpper: %s\n", subtreeUpperBound.toString(false).c_str());
s += format("expectedUpperBound: %s\n",
expectedUpperBound.present() ? expectedUpperBound.get().toString(false).c_str() : "(null)");
s += format("newLinks:\n");
for (int i = 0; i < newLinks.size(); ++i) {
s += format(" %i: %s\n", i, newLinks[i].toString(false).c_str());
}
@ -6178,10 +6282,10 @@ private:
// This must be called for each of the InternalPageSliceUpdates in sorted order.
void applyUpdate(InternalPageSliceUpdate& u, const RedwoodRecordRef* nextBoundary) {
debug_printf("applyUpdate nextBoundary=(%p) %s %s\n",
debug_printf("applyUpdate nextBoundary=(%p) %s\n",
nextBoundary,
(nextBoundary != nullptr) ? nextBoundary->toString(false).c_str() : "",
u.toString().c_str());
(nextBoundary != nullptr) ? nextBoundary->toString(false).c_str() : "");
debug_print(addPrefix("applyUpdate", u.toString()));
// If the children changed, replace [cBegin, cEnd) with newLinks
if (u.childrenChanged) {
@ -6195,7 +6299,7 @@ private:
}
while (c != u.cEnd) {
debug_printf("internal page (updating) erasing: %s\n", c.get().toString(false).c_str());
debug_printf("applyUpdate (updating) erasing: %s\n", c.get().toString(false).c_str());
btPage()->kvBytes -= c.get().kvBytes();
c.erase();
}
@ -6226,7 +6330,7 @@ private:
keep(u.cBegin, u.cEnd);
}
// If there is an expected upper boundary for the next range after u
// If there is an expected upper boundary for the next range start after u
if (u.expectedUpperBound.present()) {
// Then if it does not match the next boundary then insert a dummy record
if (nextBoundary == nullptr || (nextBoundary != &u.expectedUpperBound.get() &&
@ -6253,23 +6357,29 @@ private:
state std::string context;
if (REDWOOD_DEBUG) {
context = format("CommitSubtree(root=%s): ", toString(rootID).c_str());
context = format("CommitSubtree(root=%s+%d %s): ",
toString(rootID.front()).c_str(),
rootID.size() - 1,
::toString(batch->writeVersion).c_str());
}
debug_printf("%s %s\n", context.c_str(), update->toString().c_str());
debug_printf("%s rootID=%s\n", context.c_str(), toString(rootID).c_str());
debug_print(addPrefix(context, update->toString()));
if (REDWOOD_DEBUG) {
debug_printf("%s ---------MUTATION BUFFER SLICE ---------------------\n", context.c_str());
auto begin = mBegin;
int c = 0;
auto i = mBegin;
while (1) {
debug_printf("%s Mutation: '%s': %s\n",
debug_printf("%s Mutation %4d '%s': %s\n",
context.c_str(),
printable(begin.key()).c_str(),
begin.mutation().toString().c_str());
if (begin == mEnd) {
c,
printable(i.key()).c_str(),
i.mutation().toString().c_str());
if (i == mEnd) {
break;
}
++begin;
++c;
++i;
}
debug_printf("%s -------------------------------------\n", context.c_str());
}
state Reference<const ArenaPage> page =
@ -6291,13 +6401,13 @@ private:
// TryToUpdate indicates insert and erase operations should be tried on the existing page first
state bool tryToUpdate = btPage->tree()->numItems > 0 && update->boundariesNormal();
debug_printf(
"%s commitSubtree(): %s\n",
context.c_str(),
btPage
->toString(
false, rootID, batch->snapshot->getVersion(), update->decodeLowerBound, update->decodeUpperBound)
.c_str());
debug_printf("%s tryToUpdate=%d\n", context.c_str(), tryToUpdate);
debug_print(addPrefix(context,
btPage->toString("commitSubtreeStart",
rootID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)));
state BTreePage::BinaryTree::Cursor cursor = update->cBegin.valid()
? self->getCursor(page.getPtr(), update->cBegin)
@ -6312,22 +6422,6 @@ private:
}
}
if (REDWOOD_DEBUG) {
debug_printf("%s ---------MUTATION BUFFER SLICE ---------------------\n", context.c_str());
auto begin = mBegin;
while (1) {
debug_printf("%s Mutation: '%s': %s\n",
context.c_str(),
printable(begin.key()).c_str(),
begin.mutation().toString().c_str());
if (begin == mEnd) {
break;
}
++begin;
}
debug_printf("%s -------------------------------------\n", context.c_str());
}
// Leaf Page
if (btPage->isLeaf()) {
// When true, we are modifying the existing DeltaTree
@ -6566,9 +6660,8 @@ private:
// No changes were actually made. This could happen if the only mutations are clear ranges which do not
// match any records.
if (!changesMade) {
debug_printf("%s No changes were made during mutation merge, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s No changes were made during mutation merge, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
} else {
debug_printf(
@ -6581,17 +6674,26 @@ private:
if (cursor.tree->numItems == 0) {
update->cleared();
self->freeBTreePage(height, rootID, batch->writeVersion);
debug_printf("%s Page updates cleared all entries, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s Page updates cleared all entries, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
} else {
// Otherwise update it.
BTreePageIDRef newID = wait(self->updateBTreePage(
self, rootID, &update->newLinks.arena(), pageCopy.castTo<ArenaPage>(), batch->writeVersion));
debug_printf("%s Leaf node updated in-place at version %s, new contents:\n",
context.c_str(),
toString(batch->writeVersion).c_str());
debug_print(addPrefix(context,
btPage->toString("updateLeafNode",
newID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)));
update->updatedInPlace(newID, btPage, newID.size() * self->m_blockSize);
debug_printf(
"%s Page updated in-place, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Leaf node updated in-place, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
}
return Void();
}
@ -6601,9 +6703,8 @@ private:
update->cleared();
self->freeBTreePage(height, rootID, batch->writeVersion);
debug_printf("%s All leaf page contents were cleared, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s All leaf page contents were cleared, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
}
@ -6619,7 +6720,8 @@ private:
// Put new links into update and tell update that pages were rebuilt
update->rebuilt(entries);
debug_printf("%s Merge complete, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Merge complete, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
} else {
// Internal Page
@ -6668,8 +6770,8 @@ private:
if (!cursor.get().value.present()) {
// If the upper bound is provided by a dummy record in [cBegin, cEnd) then there is no
// requirement on the next subtree range or the parent page to have a specific upper boundary
// for decoding the subtree.
u.expectedUpperBound.reset();
// for decoding the subtree. The expected upper bound has not yet been set so it can remain
// empty.
cursor.moveNext();
// If there is another record after the null child record, it must have a child page value
ASSERT(!cursor.valid() || cursor.get().value.present());
@ -6756,12 +6858,12 @@ private:
RedwoodRecordRef rec = c.get();
if (rec.value.present()) {
if (height == 2) {
debug_printf("%s: freeing child page in cleared subtree range: %s\n",
debug_printf("%s freeing child page in cleared subtree range: %s\n",
context.c_str(),
::toString(rec.getChildPage()).c_str());
self->freeBTreePage(height, rec.getChildPage(), batch->writeVersion);
} else {
debug_printf("%s: queuing subtree deletion cleared subtree range: %s\n",
debug_printf("%s queuing subtree deletion cleared subtree range: %s\n",
context.c_str(),
::toString(rec.getChildPage()).c_str());
self->m_lazyClearQueue.pushBack(LazyClearQueueEntry{
@ -6774,9 +6876,8 @@ private:
// Subtree range unchanged
}
debug_printf("%s: MutationBuffer covers this range in a single mutation, not recursing: %s\n",
context.c_str(),
u.toString().c_str());
debug_printf("%s Not recursing, one mutation range covers this slice:\n", context.c_str());
debug_print(addPrefix(context, u.toString()));
// u has already been initialized with the correct result, no recursion needed, so restart the
// loop.
@ -6785,6 +6886,9 @@ private:
}
// If this page has height of 2 then its children are leaf nodes
debug_printf("%s Recursing for %s\n", context.c_str(), toString(pageID).c_str());
debug_print(addPrefix(context, u.toString()));
recursions.push_back(self->commitSubtree(self, batch, pageID, height - 1, mBegin, mEnd, &u));
}
@ -6823,10 +6927,11 @@ private:
// passed, so in the event a different upper boundary is needed it will be added to the already-modified
// page. Otherwise, the decode boundary is used which will prevent this page from being modified for the
// sole purpose of adding a dummy upper bound record.
debug_printf("%s Applying final child range update. changesMade=%d Parent update is: %s\n",
debug_printf("%s Applying final child range update. changesMade=%d\nSubtree Root Update:\n",
context.c_str(),
modifier.changesMade,
update->toString().c_str());
modifier.changesMade);
debug_print(addPrefix(context, update->toString()));
modifier.applyUpdate(*slices.back(),
modifier.changesMade ? &update->subtreeUpperBound : &update->decodeUpperBound);
@ -6859,9 +6964,11 @@ private:
if (modifier.changesMade || forceUpdate) {
if (modifier.empty()) {
update->cleared();
debug_printf("%s All internal page children were deleted so deleting this page too, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf(
"%s All internal page children were deleted so deleting this page too. Returning slice:\n",
context.c_str());
debug_print(addPrefix(context, update->toString()));
self->freeBTreePage(height, rootID, batch->writeVersion);
} else {
if (modifier.updating) {
@ -6899,9 +7006,10 @@ private:
}
parentInfo->clear();
if (forceUpdate && detached == 0) {
debug_printf("%s No children detached during forced update, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s No children detached during forced update, returning slice:\n",
context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
}
}
@ -6912,21 +7020,19 @@ private:
pageCopy.castTo<ArenaPage>(),
batch->writeVersion));
debug_printf(
"%s commitSubtree(): Internal page updated in-place at version %s, new contents: %s\n",
"%s commitSubtree(): Internal node updated in-place at version %s, new contents:\n",
context.c_str(),
toString(batch->writeVersion).c_str(),
btPage
->toString(false,
newID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)
.c_str());
toString(batch->writeVersion).c_str());
debug_print(addPrefix(context,
btPage->toString("updateInternalNode",
newID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)));
update->updatedInPlace(newID, btPage, newID.size() * self->m_blockSize);
debug_printf("%s Internal page updated in-place, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s Internal node updated in-place, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
} else {
// Page was rebuilt, possibly split.
debug_printf("%s Internal page could not be modified, rebuilding replacement(s).\n",
@ -6973,12 +7079,13 @@ private:
rootID));
update->rebuilt(newChildEntries);
debug_printf(
"%s Internal page rebuilt, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Internal page rebuilt, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
}
}
} else {
debug_printf("%s Page has no changes, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Page has no changes, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
}
return Void();
}
@ -9472,9 +9579,11 @@ TEST_CASE("Lredwood/correctness/btree") {
state double clearProbability =
params.getDouble("clearProbability").orDefault(deterministicRandom()->random01() * .1);
state double clearExistingBoundaryProbability =
params.getDouble("clearProbability").orDefault(deterministicRandom()->random01() * .5);
params.getDouble("clearExistingBoundaryProbability").orDefault(deterministicRandom()->random01() * .5);
state double clearSingleKeyProbability =
params.getDouble("clearSingleKeyProbability").orDefault(deterministicRandom()->random01());
params.getDouble("clearSingleKeyProbability").orDefault(deterministicRandom()->random01() * .1);
state double clearKnownNodeBoundaryProbability =
params.getDouble("clearKnownNodeBoundaryProbability").orDefault(deterministicRandom()->random01() * .1);
state double clearPostSetProbability =
params.getDouble("clearPostSetProbability").orDefault(deterministicRandom()->random01() * .1);
state double coldStartProbability =
@ -9495,10 +9604,11 @@ TEST_CASE("Lredwood/correctness/btree") {
// These settings are an attempt to keep the test execution real reasonably short
state int64_t maxPageOps = params.getInt("maxPageOps").orDefault((shortTest || serialTest) ? 50e3 : 1e6);
state int maxVerificationMapEntries =
params.getInt("maxVerificationMapEntries").orDefault((1.0 - coldStartProbability) * 300e3);
state int maxVerificationMapEntries = params.getInt("maxVerificationMapEntries").orDefault(300e3);
state int maxColdStarts = params.getInt("maxColdStarts").orDefault(300);
// Max number of records in the BTree or the versioned written map to visit
state int64_t maxRecordsRead = 300e6;
state int64_t maxRecordsRead = params.getInt("maxRecordsRead").orDefault(300e6);
printf("\n");
printf("file: %s\n", file.c_str());
@ -9516,9 +9626,11 @@ TEST_CASE("Lredwood/correctness/btree") {
printf("setExistingKeyProbability: %f\n", setExistingKeyProbability);
printf("clearProbability: %f\n", clearProbability);
printf("clearExistingBoundaryProbability: %f\n", clearExistingBoundaryProbability);
printf("clearKnownNodeBoundaryProbability: %f\n", clearKnownNodeBoundaryProbability);
printf("clearSingleKeyProbability: %f\n", clearSingleKeyProbability);
printf("clearPostSetProbability: %f\n", clearPostSetProbability);
printf("coldStartProbability: %f\n", coldStartProbability);
printf("maxColdStarts: %d\n", maxColdStarts);
printf("advanceOldVersionProbability: %f\n", advanceOldVersionProbability);
printf("pageCacheBytes: %s\n", pageCacheBytes == 0 ? "default" : format("%" PRId64, pageCacheBytes).c_str());
printf("versionIncrement: %" PRId64 "\n", versionIncrement);
@ -9534,9 +9646,11 @@ TEST_CASE("Lredwood/correctness/btree") {
state VersionedBTree* btree = new VersionedBTree(pager, file);
wait(btree->init());
state DecodeBoundaryVerifier* pBoundaries = DecodeBoundaryVerifier::getVerifier(file);
state std::map<std::pair<std::string, Version>, Optional<std::string>> written;
state int64_t totalRecordsRead = 0;
state std::set<Key> keys;
state int coldStarts = 0;
state Version lastVer = btree->getLastCommittedVersion();
printf("Starting from version: %" PRId64 "\n", lastVer);
@ -9595,6 +9709,21 @@ TEST_CASE("Lredwood/correctness/btree") {
end = *i;
}
if (!pBoundaries->boundarySamples.empty() &&
deterministicRandom()->random01() < clearKnownNodeBoundaryProbability) {
start = pBoundaries->getSample();
// Can't allow the end boundary to be a start, so just convert to empty string.
if (start == VersionedBTree::dbEnd.key) {
start = Key();
}
}
if (!pBoundaries->boundarySamples.empty() &&
deterministicRandom()->random01() < clearKnownNodeBoundaryProbability) {
end = pBoundaries->getSample();
}
// Do a single key clear based on probability or end being randomly chosen to be the same as begin
// (unlikely)
if (deterministicRandom()->random01() < clearSingleKeyProbability || end == start) {
@ -9730,7 +9859,9 @@ TEST_CASE("Lredwood/correctness/btree") {
mutationBytesTargetThisCommit = randomSize(maxCommitSize);
// Recover from disk at random
if (!pagerMemoryOnly && deterministicRandom()->random01() < coldStartProbability) {
if (!pagerMemoryOnly && coldStarts < maxColdStarts &&
deterministicRandom()->random01() < coldStartProbability) {
++coldStarts;
printf("Recovering from disk after next commit.\n");
// Wait for outstanding commit
@ -10239,7 +10370,7 @@ TEST_CASE(":/redwood/performance/set") {
state Future<Void> stats =
traceMetrics ? Void()
: repeatEvery(1.0, [&]() { printf("Stats:\n%s\n", g_redwoodMetrics.toString(true).c_str()); });
: recurring([&]() { printf("Stats:\n%s\n", g_redwoodMetrics.toString(true).c_str()); }, 1.0);
if (scans > 0) {
printf("Parallel scans, concurrency=%d, scans=%d, scanWidth=%d, scanPreftchBytes=%d ...\n",

View File

@ -45,16 +45,20 @@
#include "fdbclient/WellKnownEndpoints.h"
#include "fdbclient/SimpleIni.h"
#include "fdbrpc/AsyncFileCached.actor.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/Net2FileSystem.h"
#include "fdbrpc/PerfMetric.h"
#include "fdbrpc/fdbrpc.h"
#include "fdbrpc/simulator.h"
#include "fdbserver/ConflictSet.h"
#include "fdbserver/CoordinationInterface.h"
#include "fdbserver/CoroFlow.h"
#include "fdbserver/DataDistribution.actor.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/IKeyValueStore.h"
#include "fdbserver/MoveKeys.actor.h"
#include "fdbserver/NetworkTest.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#include "fdbserver/RestoreWorkerInterface.actor.h"
#include "fdbserver/ServerDBInfo.h"
#include "fdbserver/SimulatedCluster.h"
@ -74,10 +78,13 @@
#include "flow/WriteOnlySet.h"
#include "flow/UnitTest.h"
#include "flow/FaultInjection.h"
#include "flow/flow.h"
#include "flow/network.h"
#if defined(__linux__) || defined(__FreeBSD__)
#include <execinfo.h>
#include <signal.h>
#include <sys/prctl.h>
#ifdef ALLOC_INSTRUMENTATION
#include <cxxabi.h>
#endif
@ -100,7 +107,7 @@ enum {
OPT_DCID, OPT_MACHINE_CLASS, OPT_BUGGIFY, OPT_VERSION, OPT_BUILD_FLAGS, OPT_CRASHONERROR, OPT_HELP, OPT_NETWORKIMPL, OPT_NOBUFSTDOUT, OPT_BUFSTDOUTERR,
OPT_TRACECLOCK, OPT_NUMTESTERS, OPT_DEVHELP, OPT_ROLLSIZE, OPT_MAXLOGS, OPT_MAXLOGSSIZE, OPT_KNOB, OPT_UNITTESTPARAM, OPT_TESTSERVERS, OPT_TEST_ON_SERVERS, OPT_METRICSCONNFILE,
OPT_METRICSPREFIX, OPT_LOGGROUP, OPT_LOCALITY, OPT_IO_TRUST_SECONDS, OPT_IO_TRUST_WARN_ONLY, OPT_FILESYSTEM, OPT_PROFILER_RSS_SIZE, OPT_KVFILE,
OPT_TRACE_FORMAT, OPT_WHITELIST_BINPATH, OPT_BLOB_CREDENTIAL_FILE, OPT_CONFIG_PATH, OPT_USE_TEST_CONFIG_DB, OPT_FAULT_INJECTION, OPT_PROFILER, OPT_PRINT_SIMTIME,
OPT_TRACE_FORMAT, OPT_WHITELIST_BINPATH, OPT_BLOB_CREDENTIAL_FILE, OPT_CONFIG_PATH, OPT_USE_TEST_CONFIG_DB, OPT_FAULT_INJECTION, OPT_PROFILER, OPT_PRINT_SIMTIME, OPT_FLOW_PROCESS_NAME, OPT_FLOW_PROCESS_ENDPOINT
};
CSimpleOpt::SOption g_rgOptions[] = {
@ -187,8 +194,10 @@ CSimpleOpt::SOption g_rgOptions[] = {
{ OPT_USE_TEST_CONFIG_DB, "--use-test-config-db", SO_NONE },
{ OPT_FAULT_INJECTION, "-fi", SO_REQ_SEP },
{ OPT_FAULT_INJECTION, "--fault-injection", SO_REQ_SEP },
{ OPT_PROFILER, "--profiler-", SO_REQ_SEP},
{ OPT_PROFILER, "--profiler-", SO_REQ_SEP },
{ OPT_PRINT_SIMTIME, "--print-sim-time", SO_NONE },
{ OPT_FLOW_PROCESS_NAME, "--process-name", SO_REQ_SEP },
{ OPT_FLOW_PROCESS_ENDPOINT, "--process-endpoint", SO_REQ_SEP },
#ifndef TLS_DISABLED
TLS_OPTION_FLAGS
@ -285,6 +294,13 @@ private:
};
UID getSharedMemoryMachineId() {
// new UID to use if an existing one is not found
UID newUID = deterministicRandom()->randomUniqueID();
#if DEBUG_DETERMINISM
// Don't use shared memory if DEBUG_DETERMINISM is set
return newUID;
#else
UID* machineId = nullptr;
int numTries = 0;
@ -297,7 +313,7 @@ UID getSharedMemoryMachineId() {
// "0" is the default parameter "addr"
boost::interprocess::managed_shared_memory segment(
boost::interprocess::open_or_create, sharedMemoryIdentifier.c_str(), 1000, 0, p.permission);
machineId = segment.find_or_construct<UID>("machineId")(deterministicRandom()->randomUniqueID());
machineId = segment.find_or_construct<UID>("machineId")(newUID);
if (!machineId)
criticalError(
FDB_EXIT_ERROR, "SharedMemoryError", "Could not locate or create shared memory - 'machineId'");
@ -321,6 +337,7 @@ UID getSharedMemoryMachineId() {
}
}
}
#endif
}
ACTOR void failAfter(Future<Void> trigger, ISimulator::ProcessInfo* m = g_simulator.getCurrentProcess()) {
@ -959,7 +976,8 @@ enum class ServerRole {
SkipListTest,
Test,
VersionedMapTest,
UnitTests
UnitTests,
FlowProcess
};
struct CLIOptions {
std::string commandLine;
@ -1015,6 +1033,8 @@ struct CLIOptions {
UnitTestParameters testParams;
std::map<std::string, std::string> profilerConfig;
std::string flowProcessName;
Endpoint flowProcessEndpoint;
bool printSimTime = false;
static CLIOptions parseArgs(int argc, char* argv[]) {
@ -1193,6 +1213,8 @@ private:
role = ServerRole::ConsistencyCheck;
else if (!strcmp(sRole, "unittests"))
role = ServerRole::UnitTests;
else if (!strcmp(sRole, "flowprocess"))
role = ServerRole::FlowProcess;
else {
fprintf(stderr, "ERROR: Unknown role `%s'\n", sRole);
printHelpTeaser(argv[0]);
@ -1517,6 +1539,42 @@ private:
case OPT_USE_TEST_CONFIG_DB:
configDBType = ConfigDBType::SIMPLE;
break;
case OPT_FLOW_PROCESS_NAME:
flowProcessName = args.OptionArg();
std::cout << flowProcessName << std::endl;
break;
case OPT_FLOW_PROCESS_ENDPOINT: {
std::vector<std::string> strings;
std::cout << args.OptionArg() << std::endl;
boost::split(strings, args.OptionArg(), [](char c) { return c == ','; });
for (auto& str : strings) {
std::cout << str << " ";
}
std::cout << "\n";
if (strings.size() != 3) {
std::cerr << "Invalid argument, expected 3 elements in --process-endpoint got " << strings.size()
<< std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
try {
auto addr = NetworkAddress::parse(strings[0]);
uint64_t fst = std::stoul(strings[1]);
uint64_t snd = std::stoul(strings[2]);
UID token(fst, snd);
NetworkAddressList l;
l.address = addr;
flowProcessEndpoint = Endpoint(l, token);
std::cout << "flowProcessEndpoint: " << flowProcessEndpoint.getPrimaryAddress().toString()
<< ", token: " << flowProcessEndpoint.token.toString() << "\n";
} catch (Error& e) {
std::cerr << "Could not parse network address " << strings[0] << std::endl;
flushAndExit(FDB_EXIT_ERROR);
} catch (std::exception& e) {
std::cerr << "Could not parse token " << strings[1] << "," << strings[2] << std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
break;
}
case OPT_PRINT_SIMTIME:
printSimTime = true;
break;
@ -1723,6 +1781,7 @@ int main(int argc, char* argv[]) {
role == ServerRole::Simulation ? IsSimulated::True
: IsSimulated::False);
IKnobCollection::getMutableGlobalKnobCollection().setKnob("log_directory", KnobValue::create(opts.logFolder));
IKnobCollection::getMutableGlobalKnobCollection().setKnob("conn_file", KnobValue::create(opts.connFile));
if (role != ServerRole::Simulation) {
IKnobCollection::getMutableGlobalKnobCollection().setKnob("commit_batches_mem_bytes_hard_limit",
KnobValue::create(int64_t{ opts.memLimit }));
@ -1802,8 +1861,8 @@ int main(int argc, char* argv[]) {
FlowTransport::createInstance(false, 1, WLTOKEN_RESERVED_COUNT);
opts.buildNetwork(argv[0]);
const bool expectsPublicAddress =
(role == ServerRole::FDBD || role == ServerRole::NetworkTestServer || role == ServerRole::Restore);
const bool expectsPublicAddress = (role == ServerRole::FDBD || role == ServerRole::NetworkTestServer ||
role == ServerRole::Restore || role == ServerRole::FlowProcess);
if (opts.publicAddressStrs.empty()) {
if (expectsPublicAddress) {
fprintf(stderr, "ERROR: The -p or --public-address option is required\n");
@ -2139,6 +2198,19 @@ int main(int argc, char* argv[]) {
}
f = result;
} else if (role == ServerRole::FlowProcess) {
TraceEvent(SevDebug, "StartingFlowProcess").detail("From", "fdbserver");
#if defined(__linux__) || defined(__FreeBSD__)
prctl(PR_SET_PDEATHSIG, SIGTERM);
if (getppid() == 1) /* parent already died before prctl */
flushAndExit(FDB_EXIT_SUCCESS);
#endif
if (opts.flowProcessName == "KeyValueStoreProcess") {
ProcessFactory<KeyValueStoreProcess>(opts.flowProcessName.c_str());
}
f = stopAfter(runFlowProcess(opts.flowProcessName, opts.flowProcessEndpoint));
g_network->run();
} else if (role == ServerRole::KVFileDump) {
f = stopAfter(KVFileDump(opts.kvFile));
g_network->run();

View File

@ -28,11 +28,13 @@
#include "fdbrpc/LoadBalance.h"
#include "flow/ActorCollection.h"
#include "flow/Arena.h"
#include "flow/Error.h"
#include "flow/Hash3.h"
#include "flow/Histogram.h"
#include "flow/IRandom.h"
#include "flow/IndexedSet.h"
#include "flow/SystemMonitor.h"
#include "flow/Trace.h"
#include "flow/Tracing.h"
#include "flow/Util.h"
#include "fdbclient/Atomic.h"
@ -100,6 +102,9 @@ bool canReplyWith(Error e) {
case error_code_quick_get_value_miss:
case error_code_quick_get_key_values_miss:
case error_code_get_mapped_key_values_has_more:
case error_code_key_not_tuple:
case error_code_value_not_tuple:
case error_code_mapper_not_tuple:
// case error_code_all_alternatives_failed:
return true;
default:
@ -834,6 +839,9 @@ public:
Promise<Void> coreStarted;
bool shuttingDown;
Promise<Void> registerInterfaceAcceptingRequests;
Future<Void> interfaceRegistered;
bool behind;
bool versionBehind;
@ -3490,14 +3498,24 @@ Key constructMappedKey(KeyValueRef* keyValue, Tuple& mappedKeyFormatTuple, bool&
// Use keyTuple as reference.
if (!keyTuple.present()) {
// May throw exception if the key is not parsable as a tuple.
keyTuple = Tuple::unpack(keyValue->key);
try {
keyTuple = Tuple::unpack(keyValue->key);
} catch (Error& e) {
TraceEvent("KeyNotTuple").error(e).detail("Key", keyValue->key.printable());
throw key_not_tuple();
}
}
referenceTuple = &keyTuple.get();
} else if (s[1] == 'V') {
// Use valueTuple as reference.
if (!valueTuple.present()) {
// May throw exception if the value is not parsable as a tuple.
valueTuple = Tuple::unpack(keyValue->value);
try {
valueTuple = Tuple::unpack(keyValue->value);
} catch (Error& e) {
TraceEvent("ValueNotTuple").error(e).detail("Value", keyValue->value.printable());
throw value_not_tuple();
}
}
referenceTuple = &valueTuple.get();
} else {
@ -3631,7 +3649,13 @@ ACTOR Future<GetMappedKeyValuesReply> mapKeyValues(StorageServer* data,
result.data.reserve(result.arena, input.data.size());
state Tuple mappedKeyFormatTuple = Tuple::unpack(mapper);
state Tuple mappedKeyFormatTuple;
try {
mappedKeyFormatTuple = Tuple::unpack(mapper);
} catch (Error& e) {
TraceEvent("MapperNotTuple").error(e).detail("Mapper", mapper.printable());
throw mapper_not_tuple();
}
state KeyValueRef* it = input.data.begin();
for (; it != input.data.end(); it++) {
state MappedKeyValueRef kvm;
@ -6459,6 +6483,7 @@ ACTOR Future<Void> tssDelayForever() {
ACTOR Future<Void> update(StorageServer* data, bool* pReceivedUpdate) {
state double start;
try {
// If we are disk bound and durableVersion is very old, we need to block updates or we could run out of
// memory. This is often referred to as the storage server e-brake (emergency brake)
@ -6857,6 +6882,16 @@ ACTOR Future<Void> update(StorageServer* data, bool* pReceivedUpdate) {
validate(data);
if ((data->lastTLogVersion - data->version.get()) < SERVER_KNOBS->STORAGE_RECOVERY_VERSION_LAG_LIMIT) {
if (data->registerInterfaceAcceptingRequests.canBeSet()) {
data->registerInterfaceAcceptingRequests.send(Void());
ErrorOr<Void> e = wait(errorOr(data->interfaceRegistered));
if (e.isError()) {
TraceEvent(SevWarn, "StorageInterfaceRegistrationFailed", data->thisServerID).error(e.getError());
}
}
}
data->logCursor->advanceTo(cloneCursor2->version());
if (cursor->version().version >= data->lastTLogVersion) {
if (data->behind) {
@ -8465,7 +8500,8 @@ bool storageServerTerminated(StorageServer& self, IKeyValueStore* persistentData
}
if (e.code() == error_code_worker_removed || e.code() == error_code_recruitment_failed ||
e.code() == error_code_file_not_found || e.code() == error_code_actor_cancelled) {
e.code() == error_code_file_not_found || e.code() == error_code_actor_cancelled ||
e.code() == error_code_remote_kvs_cancelled) {
TraceEvent("StorageServerTerminated", self.thisServerID).errorUnsuppressed(e);
return true;
} else
@ -8535,88 +8571,6 @@ ACTOR Future<Void> initTenantMap(StorageServer* self) {
return Void();
}
// for creating a new storage server
ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
StorageServerInterface ssi,
Tag seedTag,
UID clusterId,
Version tssSeedVersion,
ReplyPromise<InitializeStorageReply> recruitReply,
Reference<AsyncVar<ServerDBInfo> const> db,
std::string folder) {
state StorageServer self(persistentData, db, ssi);
state Future<Void> ssCore;
self.clusterId.send(clusterId);
if (ssi.isTss()) {
self.setTssPair(ssi.tssPairID.get());
ASSERT(self.isTss());
}
self.sk = serverKeysPrefixFor(self.tssPairID.present() ? self.tssPairID.get() : self.thisServerID)
.withPrefix(systemKeys.begin); // FFFF/serverKeys/[this server]/
self.folder = folder;
try {
wait(self.storage.init());
wait(self.storage.commit());
++self.counters.kvCommits;
if (seedTag == invalidTag) {
// Might throw recruitment_failed in case of simultaneous master failure
std::pair<Version, Tag> verAndTag = wait(addStorageServer(self.cx, ssi));
self.tag = verAndTag.second;
if (ssi.isTss()) {
self.setInitialVersion(tssSeedVersion);
} else {
self.setInitialVersion(verAndTag.first - 1);
}
wait(initTenantMap(&self));
} else {
self.tag = seedTag;
}
self.storage.makeNewStorageServerDurable();
wait(self.storage.commit());
++self.counters.kvCommits;
TraceEvent("StorageServerInit", ssi.id())
.detail("Version", self.version.get())
.detail("SeedTag", seedTag.toString())
.detail("TssPair", ssi.isTss() ? ssi.tssPairID.get().toString() : "");
InitializeStorageReply rep;
rep.interf = ssi;
rep.addedVersion = self.version.get();
recruitReply.send(rep);
self.byteSampleRecovery = Void();
ssCore = storageServerCore(&self, ssi);
wait(ssCore);
throw internal_error();
} catch (Error& e) {
// If we die with an error before replying to the recruitment request, send the error to the recruiter
// (ClusterController, and from there to the DataDistributionTeamCollection)
if (!recruitReply.isSet())
recruitReply.sendError(recruitment_failed());
// If the storage server dies while something that uses self is still on the stack,
// we want that actor to complete before we terminate and that memory goes out of scope
state Error err = e;
if (storageServerTerminated(self, persistentData, err)) {
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
return Void();
}
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
throw err;
}
}
ACTOR Future<Void> replaceInterface(StorageServer* self, StorageServerInterface ssi) {
ASSERT(!ssi.isTss());
state Transaction tr(self->cx);
@ -8752,6 +8706,119 @@ ACTOR Future<Void> replaceTSSInterface(StorageServer* self, StorageServerInterfa
return Void();
}
ACTOR Future<Void> storageInterfaceRegistration(StorageServer* self,
StorageServerInterface ssi,
Optional<Future<Void>> readyToAcceptRequests) {
if (readyToAcceptRequests.present()) {
wait(readyToAcceptRequests.get());
ssi.startAcceptingRequests();
} else {
ssi.stopAcceptingRequests();
}
try {
if (self->isTss()) {
wait(replaceTSSInterface(self, ssi));
} else {
wait(replaceInterface(self, ssi));
}
} catch (Error& e) {
throw;
}
return Void();
}
// for creating a new storage server
ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
StorageServerInterface ssi,
Tag seedTag,
UID clusterId,
Version tssSeedVersion,
ReplyPromise<InitializeStorageReply> recruitReply,
Reference<AsyncVar<ServerDBInfo> const> db,
std::string folder) {
state StorageServer self(persistentData, db, ssi);
state Future<Void> ssCore;
self.clusterId.send(clusterId);
if (ssi.isTss()) {
self.setTssPair(ssi.tssPairID.get());
ASSERT(self.isTss());
}
self.sk = serverKeysPrefixFor(self.tssPairID.present() ? self.tssPairID.get() : self.thisServerID)
.withPrefix(systemKeys.begin); // FFFF/serverKeys/[this server]/
self.folder = folder;
try {
wait(self.storage.init());
wait(self.storage.commit());
++self.counters.kvCommits;
if (seedTag == invalidTag) {
ssi.startAcceptingRequests();
self.registerInterfaceAcceptingRequests.send(Void());
// Might throw recruitment_failed in case of simultaneous master failure
std::pair<Version, Tag> verAndTag = wait(addStorageServer(self.cx, ssi));
self.tag = verAndTag.second;
if (ssi.isTss()) {
self.setInitialVersion(tssSeedVersion);
} else {
self.setInitialVersion(verAndTag.first - 1);
}
wait(initTenantMap(&self));
} else {
self.tag = seedTag;
}
self.storage.makeNewStorageServerDurable();
wait(self.storage.commit());
++self.counters.kvCommits;
self.interfaceRegistered =
storageInterfaceRegistration(&self, ssi, self.registerInterfaceAcceptingRequests.getFuture());
wait(delay(0));
TraceEvent("StorageServerInit", ssi.id())
.detail("Version", self.version.get())
.detail("SeedTag", seedTag.toString())
.detail("TssPair", ssi.isTss() ? ssi.tssPairID.get().toString() : "");
InitializeStorageReply rep;
rep.interf = ssi;
rep.addedVersion = self.version.get();
recruitReply.send(rep);
self.byteSampleRecovery = Void();
ssCore = storageServerCore(&self, ssi);
wait(ssCore);
throw internal_error();
} catch (Error& e) {
// If we die with an error before replying to the recruitment request, send the error to the recruiter
// (ClusterController, and from there to the DataDistributionTeamCollection)
if (!recruitReply.isSet())
recruitReply.sendError(recruitment_failed());
// If the storage server dies while something that uses self is still on the stack,
// we want that actor to complete before we terminate and that memory goes out of scope
state Error err = e;
if (storageServerTerminated(self, persistentData, err)) {
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
return Void();
}
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
throw err;
}
}
// for recovering an existing storage server
ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
StorageServerInterface ssi,
@ -8807,15 +8874,14 @@ ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
if (recovered.canBeSet())
recovered.send(Void());
try {
if (self.isTss()) {
wait(replaceTSSInterface(&self, ssi));
} else {
wait(replaceInterface(&self, ssi));
}
} catch (Error& e) {
state Future<Void> f = storageInterfaceRegistration(&self, ssi, {});
wait(delay(0));
ErrorOr<Void> e = wait(errorOr(f));
if (e.isError()) {
Error e = f.getError();
if (e.code() != error_code_worker_removed) {
throw;
throw e;
}
state UID clusterId = wait(getClusterId(&self));
ASSERT(self.clusterId.isValid());
@ -8829,15 +8895,19 @@ ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
// We want to avoid this and force a manual removal of the storage
// servers' old data when being assigned to a new cluster to avoid
// accidental data loss.
TraceEvent(SevError, "StorageServerBelongsToExistingCluster")
TraceEvent(SevWarn, "StorageServerBelongsToExistingCluster")
.detail("ServerID", ssi.id())
.detail("ClusterID", durableClusterId)
.detail("NewClusterID", clusterId);
wait(Future<Void>(Never()));
}
self.interfaceRegistered =
storageInterfaceRegistration(&self, ssi, self.registerInterfaceAcceptingRequests.getFuture());
wait(delay(0));
TraceEvent("StorageServerStartingCore", self.thisServerID).detail("TimeTaken", now() - start);
// wait( delay(0) ); // To make sure self->zkMasterInfo.onChanged is available to wait on
ssCore = storageServerCore(&self, ssi);
wait(ssCore);

View File

@ -1092,7 +1092,8 @@ std::map<std::string, std::function<void(const std::string&)>> testSpecGlobalKey
[](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedMaxTLogVersion", ""); } },
{ "disableTss", [](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedDisableTSS", ""); } },
{ "disableHostname",
[](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedDisableHostname", ""); } }
[](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedDisableHostname", ""); } },
{ "disableRemoteKVS", [](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedRemoteKVS", ""); } }
};
std::map<std::string, std::function<void(const std::string& value, TestSpec* spec)>> testSpecTestKeys = {

View File

@ -18,6 +18,7 @@
* limitations under the License.
*/
#include <cstdlib>
#include <tuple>
#include <boost/lexical_cast.hpp>
@ -49,6 +50,7 @@
#include "fdbserver/CoordinationInterface.h"
#include "fdbserver/ConfigNode.h"
#include "fdbserver/LocalConfiguration.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#include "fdbclient/MonitorLeader.h"
#include "fdbclient/ClientWorkerInterface.h"
#include "flow/Profiler.h"
@ -208,31 +210,44 @@ ACTOR Future<Void> handleIOErrors(Future<Void> actor, IClosable* store, UID id,
state Future<ErrorOr<Void>> storeError = actor.isReady() ? Never() : errorOr(store->getError());
choose {
when(state ErrorOr<Void> e = wait(errorOr(actor))) {
TraceEvent(SevDebug, "HandleIOErrorsActorIsReady")
.detail("Error", e.isError() ? e.getError().code() : -1)
.detail("UID", id);
if (e.isError() && e.getError().code() == error_code_please_reboot) {
// no need to wait.
} else {
TraceEvent(SevDebug, "HandleIOErrorsActorBeforeOnClosed").detail("IsClosed", onClosed.isReady());
wait(onClosed);
TraceEvent(SevDebug, "HandleIOErrorsActorOnClosedFinished")
.detail("StoreError",
storeError.isReady() ? (storeError.get().isError() ? storeError.get().getError().code() : 0)
: -1);
}
if (e.isError() && e.getError().code() == error_code_broken_promise && !storeError.isReady()) {
wait(delay(0.00001 + FLOW_KNOBS->MAX_BUGGIFIED_DELAY));
}
if (storeError.isReady())
throw storeError.get().getError();
if (e.isError())
if (storeError.isReady() &&
!((storeError.get().isError() && storeError.get().getError().code() == error_code_file_not_found))) {
throw storeError.get().isError() ? storeError.get().getError() : actor_cancelled();
}
if (e.isError()) {
throw e.getError();
else
} else
return e.get();
}
when(ErrorOr<Void> e = wait(storeError)) {
TraceEvent("WorkerTerminatingByIOError", id).errorUnsuppressed(e.getError());
// for remote kv store, worker can terminate without an error, so throws actor_cancelled
// (there's probably a better way tho)
TraceEvent("WorkerTerminatingByIOError", id)
.errorUnsuppressed(e.isError() ? e.getError() : actor_cancelled());
actor.cancel();
// file_not_found can occur due to attempting to open a partially deleted DiskQueue, which should not be
// reported SevError.
if (e.getError().code() == error_code_file_not_found) {
if (e.isError() && e.getError().code() == error_code_file_not_found) {
TEST(true); // Worker terminated with file_not_found error
return Void();
}
throw e.getError();
throw e.isError() ? e.getError() : actor_cancelled();
}
}
}
@ -243,6 +258,7 @@ ACTOR Future<Void> workerHandleErrors(FutureStream<ErrorInfo> errors) {
ErrorInfo err = _err;
bool ok = err.error.code() == error_code_success || err.error.code() == error_code_please_reboot ||
err.error.code() == error_code_actor_cancelled ||
err.error.code() == error_code_remote_kvs_cancelled ||
err.error.code() == error_code_coordinators_changed || // The worker server was cancelled
err.error.code() == error_code_shutdown_in_progress;
@ -253,6 +269,7 @@ ACTOR Future<Void> workerHandleErrors(FutureStream<ErrorInfo> errors) {
endRole(err.role, err.id, "Error", ok, err.error);
if (err.error.code() == error_code_please_reboot ||
err.error.code() == error_code_please_reboot_remote_kv_store ||
(err.role == Role::SHARED_TRANSACTION_LOG &&
(err.error.code() == error_code_io_error || err.error.code() == error_code_io_timeout)))
throw err.error;
@ -1090,9 +1107,13 @@ struct TrackRunningStorage {
KeyValueStoreType storeType,
std::set<std::pair<UID, KeyValueStoreType>>* runningStorages)
: self(self), storeType(storeType), runningStorages(runningStorages) {
TraceEvent(SevDebug, "TrackingRunningStorageConstruction").detail("StorageID", self);
runningStorages->emplace(self, storeType);
}
~TrackRunningStorage() { runningStorages->erase(std::make_pair(self, storeType)); };
~TrackRunningStorage() {
runningStorages->erase(std::make_pair(self, storeType));
TraceEvent(SevDebug, "TrackingRunningStorageDesctruction").detail("StorageID", self);
};
};
ACTOR Future<Void> storageServerRollbackRebooter(std::set<std::pair<UID, KeyValueStoreType>>* runningStorages,
@ -1523,8 +1544,15 @@ ACTOR Future<Void> workerServer(Reference<IClusterConnectionRecord> connRecord,
if (s.storedComponent == DiskStore::Storage) {
LocalLineage _;
getCurrentLineage()->modify(&RoleLineage::role) = ProcessClass::ClusterRole::Storage;
IKeyValueStore* kv =
openKVStore(s.storeType, s.filename, s.storeID, memoryLimit, false, validateDataFiles);
IKeyValueStore* kv = openKVStore(
s.storeType,
s.filename,
s.storeID,
memoryLimit,
false,
validateDataFiles,
SERVER_KNOBS->REMOTE_KV_STORE && /* testing mixed mode in simulation if remote kvs enabled*/
(g_network->isSimulated() ? deterministicRandom()->coinflip() : true));
Future<Void> kvClosed = kv->onClosed();
filesClosed.add(kvClosed);
@ -1598,6 +1626,7 @@ ACTOR Future<Void> workerServer(Reference<IClusterConnectionRecord> connRecord,
logQueueBasename = fileLogQueuePrefix.toString() + optionsString.toString() + "-";
}
ASSERT_WE_THINK(abspath(parentDirectory(s.filename)) == folder);
// TraceEvent(SevDebug, "openRemoteKVStore").detail("storeType", "TlogData");
IKeyValueStore* kv = openKVStore(s.storeType, s.filename, s.storeID, memoryLimit, validateDataFiles);
const DiskQueueVersion dqv = s.tLogOptions.getDiskQueueVersion();
const int64_t diskQueueWarnSize =
@ -2002,6 +2031,7 @@ ACTOR Future<Void> workerServer(Reference<IClusterConnectionRecord> connRecord,
req.logVersion > TLogVersion::V2 ? fileVersionedLogDataPrefix : fileLogDataPrefix;
std::string filename =
filenameFromId(req.storeType, folder, prefix.toString() + tLogOptions.toPrefix(), logId);
// TraceEvent(SevDebug, "openRemoteKVStore").detail("storeType", "3");
IKeyValueStore* data = openKVStore(req.storeType, filename, logId, memoryLimit);
const DiskQueueVersion dqv = tLogOptions.getDiskQueueVersion();
IDiskQueue* queue = openDiskQueue(
@ -2086,7 +2116,17 @@ ACTOR Future<Void> workerServer(Reference<IClusterConnectionRecord> connRecord,
folder,
isTss ? testingStoragePrefix.toString() : fileStoragePrefix.toString(),
recruited.id());
IKeyValueStore* data = openKVStore(req.storeType, filename, recruited.id(), memoryLimit);
IKeyValueStore* data = openKVStore(
req.storeType,
filename,
recruited.id(),
memoryLimit,
false,
false,
SERVER_KNOBS->REMOTE_KV_STORE && /* testing mixed mode in simulation if remote kvs enabled*/
(g_network->isSimulated() ? deterministicRandom()->coinflip() : true));
Future<Void> kvClosed = data->onClosed();
filesClosed.add(kvClosed);
ReplyPromise<InitializeStorageReply> storageReady = req.reply;
@ -2333,20 +2373,26 @@ ACTOR Future<Void> workerServer(Reference<IClusterConnectionRecord> connRecord,
when(wait(handleErrors)) {}
}
} catch (Error& err) {
TraceEvent(SevDebug, "WorkerServer").detail("Error", err.code()).backtrace();
// Make sure actors are cancelled before "recovery" promises are destructed.
for (auto f : recoveries)
f.cancel();
state Error e = err;
bool ok = e.code() == error_code_please_reboot || e.code() == error_code_actor_cancelled ||
e.code() == error_code_please_reboot_delete;
e.code() == error_code_please_reboot_delete || e.code() == error_code_please_reboot_remote_kv_store;
endRole(Role::WORKER, interf.id(), "WorkerError", ok, e);
errorForwarders.clear(false);
sharedLogs.clear();
if (e.code() !=
error_code_actor_cancelled) { // We get cancelled e.g. when an entire simulation times out, but in that case
// we won't be restarted and don't need to wait for shutdown
if (e.code() != error_code_actor_cancelled && e.code() != error_code_please_reboot_remote_kv_store) {
// actor_cancelled:
// We get cancelled e.g. when an entire simulation times out, but in that case
// we won't be restarted and don't need to wait for shutdown
// reboot_remote_kv_store:
// The child process running the storage engine died abnormally,
// the current solution is to reboot the worker.
// Some refactoring work in the future can make it only reboot the storage server
stopping.send(Void());
wait(filesClosed.getResult()); // Wait for complete shutdown of KV stores
wait(delay(0.0)); // Unwind the callstack to make sure that IAsyncFile references are all gone

View File

@ -22,6 +22,7 @@
#include "fdbserver/TesterInterface.actor.h"
#include "fdbserver/workloads/workloads.actor.h"
#include "fdbrpc/simulator.h"
#include "boost/algorithm/string/predicate.hpp"
#undef state
#include "fdbclient/SimpleIni.h"
@ -70,12 +71,14 @@ struct SaveAndKillWorkload : TestWorkload {
std::map<NetworkAddress, ISimulator::ProcessInfo*> rebootingProcesses = g_simulator.currentlyRebootingProcesses;
std::map<std::string, ISimulator::ProcessInfo*> allProcessesMap;
for (const auto& [_, process] : rebootingProcesses) {
if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end()) {
if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end() &&
std::string(process->name) != "remote flow process") {
allProcessesMap[process->dataFolder] = process;
}
}
for (const auto& process : processes) {
if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end()) {
if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end() &&
std::string(process->name) != "remote flow process") {
allProcessesMap[process->dataFolder] = process;
}
}

View File

@ -638,6 +638,9 @@ public:
return tokens;
}
// True if both StringRefs reference exactly the same memory
bool same(const StringRef& s) const { return data == s.data && length == s.length; }
private:
// Unimplemented; blocks conversion through std::string
StringRef(char*);

View File

@ -743,6 +743,13 @@ class Listener final : public IListener, ReferenceCounted<Listener> {
public:
Listener(boost::asio::io_context& io_service, NetworkAddress listenAddress)
: io_service(io_service), listenAddress(listenAddress), acceptor(io_service, tcpEndpoint(listenAddress)) {
// when port 0 is passed in, a random port will be opened
// set listenAddress as the address with the actual port opened instead of port 0
if (listenAddress.port == 0) {
this->listenAddress =
NetworkAddress::parse(acceptor.local_endpoint().address().to_string().append(":").append(
std::to_string(acceptor.local_endpoint().port())));
}
platform::setCloseOnExec(acceptor.native_handle());
}

View File

@ -3755,6 +3755,40 @@ void fdb_probe_actor_exit(const char* name, unsigned long id, int index) {
}
#endif
void throwExecPathError(Error e, char path[]) {
Severity sev = e.code() == error_code_io_error ? SevError : SevWarnAlways;
TraceEvent(sev, "GetPathError").error(e).detail("Path", path);
throw e;
}
std::string getExecPath() {
char path[1024];
uint32_t size = sizeof(path);
#if defined(__APPLE__)
if (_NSGetExecutablePath(path, &size) == 0) {
return std::string(path);
} else {
throwExecPathError(platform_error(), path);
}
#elif defined(__linux__)
ssize_t len = ::readlink("/proc/self/exe", path, size);
if (len != -1) {
path[len] = '\0';
return std::string(path);
} else {
throwExecPathError(platform_error(), path);
}
#elif defined(_WIN32)
auto len = GetModuleFileName(nullptr, path, size);
if (len != 0) {
return std::string(path);
} else {
throwExecPathError(platform_error(), path);
}
#endif
return "unsupported OS";
}
void setupRunLoopProfiler() {
#ifdef __linux__
if (!profileThread && FLOW_KNOBS->RUN_LOOP_PROFILING_INTERVAL > 0) {

View File

@ -275,7 +275,7 @@ double
timer(); // Returns the system real time clock with high precision. May jump around when system time is adjusted!
double timer_monotonic(); // Returns a high precision monotonic clock which is adjusted to be kind of similar to timer()
// at startup, but might not be a globally accurate time.
uint64_t timer_int(); // Return timer as uint64_t
uint64_t timer_int(); // Return timer as uint64_t representing epoch nanoseconds
void getLocalTime(const time_t* timep, struct tm* result);
@ -703,6 +703,9 @@ void* loadFunction(void* lib, const char* func_name);
std::string exePath();
// get the absolute path
std::string getExecPath();
#ifdef _WIN32
inline static int ctzll(uint64_t value) {
unsigned long count = 0;

View File

@ -162,6 +162,7 @@ public: // introduced features
PROTOCOL_VERSION_FEATURE(0x0FDB00B071010000LL, StorageMetadata);
PROTOCOL_VERSION_FEATURE(0x0FDB00B071010000LL, PerpetualWiggleMetadata);
PROTOCOL_VERSION_FEATURE(0x0FDB00B071010000LL, Tenants);
PROTOCOL_VERSION_FEATURE(0x0FDB00B071010000LL, StorageInterfaceReadiness);
};
template <>

View File

@ -87,7 +87,8 @@ ERROR( blob_granule_file_load_error, 1063, "Error loading a blob file during gra
ERROR( blob_granule_transaction_too_old, 1064, "Read version is older than blob granule history supports" )
ERROR( blob_manager_replaced, 1065, "This blob manager has been replaced." )
ERROR( change_feed_popped, 1066, "Tried to read a version older than what has been popped from the change feed" )
ERROR( stale_version_vector, 1067, "Client version vector is stale" )
ERROR( remote_kvs_cancelled, 1067, "The remote key-value store is cancelled" )
ERROR( stale_version_vector, 1068, "Client version vector is stale" )
ERROR( broken_promise, 1100, "Broken promise" )
ERROR( operation_cancelled, 1101, "Asynchronous operation cancelled" )
@ -114,6 +115,7 @@ ERROR( dd_tracker_cancelled, 1215, "The data distribution tracker has been cance
ERROR( failed_to_progress, 1216, "Process has failed to make sufficient progress" )
ERROR( invalid_cluster_id, 1217, "Attempted to join cluster with a different cluster ID" )
ERROR( restart_cluster_controller, 1218, "Restart cluster controller process" )
ERROR( please_reboot_remote_kv_store, 1219, "Need to reboot the storage engine process as it died abnormally")
// 15xx Platform errors
ERROR( platform_error, 1500, "Platform error" )
@ -179,6 +181,10 @@ ERROR( blob_granule_not_materialized, 2037, "Blob Granule Read was not materiali
ERROR( get_mapped_key_values_has_more, 2038, "getMappedRange does not support continuation for now" )
ERROR( get_mapped_range_reads_your_writes, 2039, "getMappedRange tries to read data that were previously written in the transaction" )
ERROR( checkpoint_not_found, 2040, "Checkpoint not found" )
ERROR( key_not_tuple, 2041, "The key cannot be parsed as a tuple" );
ERROR( value_not_tuple, 2042, "The value cannot be parsed as a tuple" );
ERROR( mapper_not_tuple, 2043, "The mapper cannot be parsed as a tuple" );
ERROR( incompatible_protocol_version, 2100, "Incompatible protocol version" )
ERROR( transaction_too_large, 2101, "Transaction exceeds byte limit" )

View File

@ -80,7 +80,7 @@ Future<Optional<T>> stopAfter(Future<T> what) {
ret = Optional<T>(_);
} catch (Error& e) {
bool ok = e.code() == error_code_please_reboot || e.code() == error_code_please_reboot_delete ||
e.code() == error_code_actor_cancelled;
e.code() == error_code_actor_cancelled || e.code() == error_code_please_reboot_remote_kv_store;
TraceEvent(ok ? SevInfo : SevError, "StopAfterError").error(e);
if (!ok) {
fprintf(stderr, "Fatal Error: %s\n", e.what());
@ -221,6 +221,7 @@ Future<T> delayed(Future<T> what, double time = 0.0, TaskPriority taskID = TaskP
}
}
// wait <interval> then call what() in a loop forever
ACTOR template <class Func>
Future<Void> recurring(Func what, double interval, TaskPriority taskID = TaskPriority::DefaultDelay) {
loop choose {
@ -2048,15 +2049,6 @@ private:
Reference<UnsafeWeakFutureReferenceData> data;
};
// Call a lambda every <interval> seconds
ACTOR template <typename Fn>
Future<Void> repeatEvery(double interval, Fn fn) {
loop {
wait(delay(interval));
fn();
}
}
#include "flow/unactorcompiler.h"
#endif

View File

@ -508,6 +508,10 @@ public:
virtual NetworkAddress getPeerAddress() const = 0;
virtual UID getDebugID() const = 0;
// At present, implemented by Sim2Conn where we want to disable bits flip for connections between parent process and
// child process, also reduce latency for this kind of connection
virtual bool isStableConnection() const { throw unsupported_operation(); }
};
class IListener {
@ -568,6 +572,10 @@ public:
// A wrapper for directly getting the system time. The time returned by now() only updates in the run loop,
// so it cannot be used to measure times of functions that do not have wait statements.
// Simulation version of timer_int for convenience, based on timer()
// Returns epoch nanoseconds
uint64_t timer_int() { return (uint64_t)(g_network->timer() * 1e9); }
virtual double timer_monotonic() = 0;
// Similar to timer, but monotonic

View File

@ -4,6 +4,7 @@ storageEngineType = 4
processesPerMachine = 1
coordinators = 3
machineCount = 15
disableRemoteKVS = true
[[test]]
testTitle = 'PhysicalShardMove'

View File

@ -4,6 +4,7 @@ minimumReplication = 3
minimumRegions = 3
logAntiQuorum = 0
storageEngineExcludeTypes = [4]
disableRemoteKVS = true
[[test]]
testTitle = 'DiskFailureCycle'