merge main

This commit is contained in:
Xiaoxi Wang 2022-04-06 09:57:52 -07:00
commit aba9d85560
128 changed files with 4596 additions and 1041 deletions

View File

@ -35,7 +35,7 @@ The official docker image for building is [`foundationdb/build`](https://hub.doc
To build outside the official docker image you'll need at least these dependencies:
1. Install cmake Version 3.13 or higher [CMake](https://cmake.org/)
1. Install [Mono](http://www.mono-project.com/download/stable/)
1. Install [Mono](https://www.mono-project.com/download/stable/)
1. Install [Ninja](https://ninja-build.org/) (optional, but recommended)
If compiling for local development, please set `-DUSE_WERROR=ON` in
@ -177,7 +177,7 @@ Under Windows, only Visual Studio with ClangCl is supported
1. Install [Python](https://www.python.org/downloads/) if is not already installed by Visual Studio
1. (Optional) Install [OpenJDK 11](https://developers.redhat.com/products/openjdk/download) to build Java bindings
1. (Optional) Install [OpenSSL 3.x](https://slproweb.com/products/Win32OpenSSL.html) to build with TLS support
1. (Optional) Install [WIX Toolset](http://wixtoolset.org/) to build Windows installer
1. (Optional) Install [WIX Toolset](https://wixtoolset.org/) to build Windows installer
1. `mkdir build && cd build`
1. `cmake -G "Visual Studio 16 2019" -A x64 -T ClangCl <PATH_TO_FOUNDATIONDB_SOURCE>`
1. `msbuild /p:Configuration=Release foundationdb.sln`

View File

@ -202,7 +202,7 @@ class TestRunner(object):
self.args.types = list(reduce(lambda t1, t2: filter(t1.__contains__, t2), map(lambda tester: tester.types, self.testers)))
self.args.no_directory_snapshot_ops = self.args.no_directory_snapshot_ops or any([not tester.directory_snapshot_ops_enabled for tester in self.testers])
self.args.no_tenants = self.args.no_tenants or any([not tester.tenants_enabled for tester in self.testers])
self.args.no_tenants = self.args.no_tenants or any([not tester.tenants_enabled for tester in self.testers]) or self.args.api_version < 710
def print_test(self):
test_instructions = self._generate_test()

View File

@ -61,8 +61,8 @@ testers = {
'python': Tester('python', 'python ' + _absolute_path('python/tests/tester.py'), 2040, 23, MAX_API_VERSION, types=ALL_TYPES, tenants_enabled=True),
'python3': Tester('python3', 'python3 ' + _absolute_path('python/tests/tester.py'), 2040, 23, MAX_API_VERSION, types=ALL_TYPES, tenants_enabled=True),
'ruby': Tester('ruby', _absolute_path('ruby/tests/tester.rb'), 2040, 23, MAX_API_VERSION),
'java': Tester('java', _java_cmd + 'StackTester', 2040, 510, MAX_API_VERSION, types=ALL_TYPES),
'java_async': Tester('java', _java_cmd + 'AsyncStackTester', 2040, 510, MAX_API_VERSION, types=ALL_TYPES),
'java': Tester('java', _java_cmd + 'StackTester', 2040, 510, MAX_API_VERSION, types=ALL_TYPES, tenants_enabled=True),
'java_async': Tester('java', _java_cmd + 'AsyncStackTester', 2040, 510, MAX_API_VERSION, types=ALL_TYPES, tenants_enabled=True),
'go': Tester('go', _absolute_path('go/build/bin/_stacktester'), 2040, 200, MAX_API_VERSION, types=ALL_TYPES),
'flow': Tester('flow', _absolute_path('flow/bin/fdb_flow_tester'), 63, 500, MAX_API_VERSION, directory_snapshot_ops_enabled=False),
}

View File

@ -124,6 +124,7 @@ if(NOT WIN32)
add_library(fdb_c_performance_test OBJECT test/performance_test.c test/test.h)
add_library(fdb_c_ryw_benchmark OBJECT test/ryw_benchmark.c test/test.h)
add_library(fdb_c_txn_size_test OBJECT test/txn_size_test.c test/test.h)
add_library(fdb_c_client_memory_test OBJECT test/client_memory_test.cpp test/unit/fdb_api.cpp test/unit/fdb_api.hpp)
add_library(mako OBJECT ${MAKO_SRCS})
add_library(fdb_c_setup_tests OBJECT test/unit/setup_tests.cpp)
add_library(fdb_c_unit_tests OBJECT ${UNIT_TEST_SRCS})

View File

@ -138,6 +138,12 @@ Tenant::Tenant(FDBDatabase* db, const uint8_t* name, int name_length) {
}
}
Tenant::~Tenant() {
if (tenant != nullptr) {
fdb_tenant_destroy(tenant);
}
}
// Transaction
Transaction::Transaction(FDBDatabase* db) {
if (fdb_error_t err = fdb_database_create_transaction(db, &tr_)) {
@ -146,7 +152,7 @@ Transaction::Transaction(FDBDatabase* db) {
}
}
Transaction::Transaction(Tenant tenant) {
Transaction::Transaction(Tenant& tenant) {
if (fdb_error_t err = fdb_tenant_create_transaction(tenant.tenant, &tr_)) {
std::cerr << fdb_get_error(err) << std::endl;
std::abort();

View File

@ -206,6 +206,11 @@ public:
class Tenant final {
public:
Tenant(FDBDatabase* db, const uint8_t* name, int name_length);
~Tenant();
Tenant(const Tenant&) = delete;
Tenant& operator=(const Tenant&) = delete;
Tenant(Tenant&&) = delete;
Tenant& operator=(Tenant&&) = delete;
private:
friend class Transaction;
@ -219,7 +224,7 @@ class Transaction final {
public:
// Given an FDBDatabase, initializes a new transaction.
Transaction(FDBDatabase* db);
Transaction(Tenant tenant);
Transaction(Tenant& tenant);
~Transaction();
// Wrapper around fdb_transaction_reset.

View File

@ -949,12 +949,10 @@ std::map<std::string, std::string> fillInRecords(int n) {
return data;
}
GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transaction& tr) {
GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transaction& tr, std::string mapper) {
std::string indexEntryKeyBegin = indexEntryKey(beginId);
std::string indexEntryKeyEnd = indexEntryKey(endId);
std::string mapper = Tuple().append(prefix).append(RECORD).append("{K[3]}"_sr).append("{...}"_sr).pack().toString();
return get_mapped_range(
tr,
FDB_KEYSEL_FIRST_GREATER_OR_EQUAL((const uint8_t*)indexEntryKeyBegin.c_str(), indexEntryKeyBegin.size()),
@ -969,6 +967,11 @@ GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transact
/* reverse */ 0);
}
GetMappedRangeResult getMappedIndexEntries(int beginId, int endId, fdb::Transaction& tr) {
std::string mapper = Tuple().append(prefix).append(RECORD).append("{K[3]}"_sr).append("{...}"_sr).pack().toString();
return getMappedIndexEntries(beginId, endId, tr, mapper);
}
TEST_CASE("fdb_transaction_get_mapped_range") {
const int TOTAL_RECORDS = 20;
fillInRecords(TOTAL_RECORDS);
@ -1009,7 +1012,6 @@ TEST_CASE("fdb_transaction_get_mapped_range") {
TEST_CASE("fdb_transaction_get_mapped_range_restricted_to_serializable") {
std::string mapper = Tuple().append(prefix).append(RECORD).append("{K[3]}"_sr).pack().toString();
fdb::Transaction tr(db);
fdb_check(tr.set_option(FDB_TR_OPTION_READ_YOUR_WRITES_DISABLE, nullptr, 0));
auto result = get_mapped_range(
tr,
FDB_KEYSEL_FIRST_GREATER_OR_EQUAL((const uint8_t*)indexEntryKey(0).c_str(), indexEntryKey(0).size()),
@ -1039,11 +1041,36 @@ TEST_CASE("fdb_transaction_get_mapped_range_restricted_to_ryw_enable") {
/* target_bytes */ 0,
/* FDBStreamingMode */ FDB_STREAMING_MODE_WANT_ALL,
/* iteration */ 0,
/* snapshot */ true,
/* snapshot */ false,
/* reverse */ 0);
ASSERT(result.err == error_code_unsupported_operation);
}
void assertNotTuple(std::string str) {
try {
Tuple::unpack(str);
} catch (Error& e) {
return;
}
UNREACHABLE();
}
TEST_CASE("fdb_transaction_get_mapped_range_fail_on_mapper_not_tuple") {
// A string that cannot be parsed as tuple.
// "\x15:\x152\x15E\x15\x09\x15\x02\x02MySimpleRecord$repeater-version\x00\x15\x013\x00\x00\x00\x00\x1aU\x90\xba\x00\x00\x00\x02\x15\x04"
std::string mapper = {
'\x15', ':', '\x15', '2', '\x15', 'E', '\x15', '\t', '\x15', '\x02', '\x02', 'M',
'y', 'S', 'i', 'm', 'p', 'l', 'e', 'R', 'e', 'c', 'o', 'r',
'd', '$', 'r', 'e', 'p', 'e', 'a', 't', 'e', 'r', '-', 'v',
'e', 'r', 's', 'i', 'o', 'n', '\x00', '\x15', '\x01', '3', '\x00', '\x00',
'\x00', '\x00', '\x1a', 'U', '\x90', '\xba', '\x00', '\x00', '\x00', '\x02', '\x15', '\x04'
};
assertNotTuple(mapper);
fdb::Transaction tr(db);
auto result = getMappedIndexEntries(1, 3, tr, mapper);
ASSERT(result.err == error_code_mapper_not_tuple);
}
TEST_CASE("fdb_transaction_get_range reverse") {
std::map<std::string, std::string> data = create_data({ { "a", "1" }, { "b", "2" }, { "c", "3" }, { "d", "4" } });
insert_data(db, data);

View File

@ -32,6 +32,7 @@ set(JAVA_BINDING_SRCS
src/main/com/apple/foundationdb/DirectBufferPool.java
src/main/com/apple/foundationdb/FDB.java
src/main/com/apple/foundationdb/FDBDatabase.java
src/main/com/apple/foundationdb/FDBTenant.java
src/main/com/apple/foundationdb/FDBTransaction.java
src/main/com/apple/foundationdb/FutureInt64.java
src/main/com/apple/foundationdb/FutureKey.java
@ -64,6 +65,8 @@ set(JAVA_BINDING_SRCS
src/main/com/apple/foundationdb/ReadTransactionContext.java
src/main/com/apple/foundationdb/subspace/package-info.java
src/main/com/apple/foundationdb/subspace/Subspace.java
src/main/com/apple/foundationdb/Tenant.java
src/main/com/apple/foundationdb/TenantManagement.java
src/main/com/apple/foundationdb/Transaction.java
src/main/com/apple/foundationdb/TransactionContext.java
src/main/com/apple/foundationdb/EventKeeper.java

View File

@ -663,6 +663,34 @@ JNIEXPORT jbyteArray JNICALL Java_com_apple_foundationdb_FutureKey_FutureKey_1ge
return result;
}
JNIEXPORT jlong JNICALL Java_com_apple_foundationdb_FDBDatabase_Database_1openTenant(JNIEnv* jenv,
jobject,
jlong dbPtr,
jbyteArray tenantNameBytes) {
if (!dbPtr || !tenantNameBytes) {
throwParamNotNull(jenv);
return 0;
}
FDBDatabase* database = (FDBDatabase*)dbPtr;
FDBTenant* tenant;
uint8_t* barr = (uint8_t*)jenv->GetByteArrayElements(tenantNameBytes, JNI_NULL);
if (!barr) {
if (!jenv->ExceptionOccurred())
throwRuntimeEx(jenv, "Error getting handle to native resources");
return 0;
}
fdb_error_t err = fdb_database_open_tenant(database, barr, jenv->GetArrayLength(tenantNameBytes), &tenant);
if (err) {
safeThrow(jenv, getThrowable(jenv, err));
return 0;
}
jenv->ReleaseByteArrayElements(tenantNameBytes, (jbyte*)barr, JNI_ABORT);
return (jlong)tenant;
}
JNIEXPORT jlong JNICALL Java_com_apple_foundationdb_FDBDatabase_Database_1createTransaction(JNIEnv* jenv,
jobject,
jlong dbPtr) {
@ -764,6 +792,31 @@ JNIEXPORT jlong JNICALL Java_com_apple_foundationdb_FDB_Database_1create(JNIEnv*
return (jlong)db;
}
JNIEXPORT jlong JNICALL Java_com_apple_foundationdb_FDBTenant_Tenant_1createTransaction(JNIEnv* jenv,
jobject,
jlong tPtr) {
if (!tPtr) {
throwParamNotNull(jenv);
return 0;
}
FDBTenant* tenant = (FDBTenant*)tPtr;
FDBTransaction* tr;
fdb_error_t err = fdb_tenant_create_transaction(tenant, &tr);
if (err) {
safeThrow(jenv, getThrowable(jenv, err));
return 0;
}
return (jlong)tr;
}
JNIEXPORT void JNICALL Java_com_apple_foundationdb_FDBTenant_Tenant_1dispose(JNIEnv* jenv, jobject, jlong tPtr) {
if (!tPtr) {
throwParamNotNull(jenv);
return;
}
fdb_tenant_destroy((FDBTenant*)tPtr);
}
JNIEXPORT void JNICALL Java_com_apple_foundationdb_FDBTransaction_Transaction_1setVersion(JNIEnv* jenv,
jobject,
jlong tPtr,

View File

@ -23,6 +23,7 @@ package com.apple.foundationdb;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.Executor;
import java.util.function.Function;
import com.apple.foundationdb.tuple.Tuple;
/**
* A mutable, lexicographically ordered mapping from binary keys to binary values.
@ -41,11 +42,82 @@ import java.util.function.Function;
*/
public interface Database extends AutoCloseable, TransactionContext {
/**
* Creates a {@link Transaction} that operates on this {@code Database}.<br>
* Opens an existing tenant to be used for running transactions.<br>
* <br>
* <b>Note:</b> opening a tenant does not check its existence in the cluster. If the tenant does not exist,
* attempts to read or write data with it will fail.
*
* @param tenantName The name of the tenant to open.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
*/
default Tenant openTenant(byte[] tenantName) {
return openTenant(tenantName, getExecutor());
}
/**
* Opens an existing tenant to be used for running transactions. This is a convenience method that generates the
* tenant name by packing a {@code Tuple}.<br>
* <br>
* <b>Note:</b> opening a tenant does not check its existence in the cluster. If the tenant does not exist,
* attempts to read or write data with it will fail.
*
* @param tenantName The name of the tenant to open, as a Tuple.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
*/
Tenant openTenant(Tuple tenantName);
/**
* Opens an existing tenant to be used for running transactions.
*
* @param tenantName The name of the tenant to open.
* @param e the {@link Executor} to use when executing asynchronous callbacks.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
*/
Tenant openTenant(byte[] tenantName, Executor e);
/**
* Opens an existing tenant to be used for running transactions. This is a convenience method that generates the
* tenant name by packing a {@code Tuple}.
*
* @param tenantName The name of the tenant to open, as a Tuple.
* @param e the {@link Executor} to use when executing asynchronous callbacks.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
*/
Tenant openTenant(Tuple tenantName, Executor e);
/**
* Opens an existing tenant to be used for running transactions.
*
* @param tenantName The name of the tenant to open.
* @param e the {@link Executor} to use when executing asynchronous callbacks.
* @param eventKeeper the {@link EventKeeper} to use when tracking instrumented calls for the tenant's transactions.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
*/
Tenant openTenant(byte[] tenantName, Executor e, EventKeeper eventKeeper);
/**
* Opens an existing tenant to be used for running transactions. This is a convenience method that generates the
* tenant name by packing a {@code Tuple}.
*
* @param tenantName The name of the tenant to open, as a Tuple.
* @param e the {@link Executor} to use when executing asynchronous callbacks.
* @param eventKeeper the {@link EventKeeper} to use when tracking instrumented calls for the tenant's transactions.
* @return a {@link Tenant} that can be used to create transactions that will operate in the tenant's key-space.
*/
Tenant openTenant(Tuple tenantName, Executor e, EventKeeper eventKeeper);
/**
* Creates a {@link Transaction} that operates on this {@code Database}. Creating a transaction
* in this way does not associate it with a {@code Tenant}, and as a result the transaction will
* operate on the entire key-space for the database.<br>
* <br>
* <b>Note:</b> Java transactions automatically set the {@link TransactionOptions#setUsedDuringCommitProtectionDisable}
* option. This is because the Java bindings disallow use of {@code Transaction} objects after
* {@link Transaction#onError} is called.
* {@link Transaction#onError} is called.<br>
* <br>
* <b>Note:</b> Transactions created directly on a {@code Database} object cannot be used in a cluster
* that requires tenant-based access. To run transactions in those clusters, you must first open a tenant
* with {@link #openTenant(byte[])}.
*
* @return a newly created {@code Transaction} that reads from and writes to this {@code Database}.
*/

View File

@ -27,6 +27,8 @@ import java.util.concurrent.atomic.AtomicReference;
import java.util.function.Function;
import com.apple.foundationdb.async.AsyncUtil;
import com.apple.foundationdb.tuple.ByteArrayUtil;
import com.apple.foundationdb.tuple.Tuple;
class FDBDatabase extends NativeObjectWrapper implements Database, OptionConsumer {
private DatabaseOptions options;
@ -116,6 +118,44 @@ class FDBDatabase extends NativeObjectWrapper implements Database, OptionConsume
}
}
@Override
public Tenant openTenant(byte[] tenantName, Executor e) {
return openTenant(tenantName, e, eventKeeper);
}
@Override
public Tenant openTenant(Tuple tenantName) {
return openTenant(tenantName.pack());
}
@Override
public Tenant openTenant(Tuple tenantName, Executor e) {
return openTenant(tenantName.pack(), e);
}
@Override
public Tenant openTenant(byte[] tenantName, Executor e, EventKeeper eventKeeper) {
pointerReadLock.lock();
Tenant tenant = null;
try {
tenant = new FDBTenant(Database_openTenant(getPtr(), tenantName), this, tenantName, e, eventKeeper);
return tenant;
} catch (RuntimeException err) {
if (tenant != null) {
tenant.close();
}
throw err;
} finally {
pointerReadLock.unlock();
}
}
@Override
public Tenant openTenant(Tuple tenantName, Executor e, EventKeeper eventKeeper) {
return openTenant(tenantName.pack(), e, eventKeeper);
}
@Override
public Transaction createTransaction(Executor e) {
return createTransaction(e, eventKeeper);
@ -170,6 +210,7 @@ class FDBDatabase extends NativeObjectWrapper implements Database, OptionConsume
Database_dispose(cPtr);
}
private native long Database_openTenant(long cPtr, byte[] tenantName);
private native long Database_createTransaction(long cPtr);
private native void Database_dispose(long cPtr);
private native void Database_setOption(long cPtr, int code, byte[] value) throws FDBException;

View File

@ -0,0 +1,158 @@
/*
* FDBTenant.java
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package com.apple.foundationdb;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.CompletionException;
import java.util.concurrent.Executor;
import java.util.concurrent.atomic.AtomicReference;
import java.util.function.Function;
import com.apple.foundationdb.async.AsyncUtil;
import com.apple.foundationdb.tuple.ByteArrayUtil;
class FDBTenant extends NativeObjectWrapper implements Tenant {
private final Database database;
private final byte[] name;
private final Executor executor;
private final EventKeeper eventKeeper;
protected FDBTenant(long cPtr, Database database, byte[] name, Executor executor) {
this(cPtr, database, name, executor, null);
}
protected FDBTenant(long cPtr, Database database, byte[] name, Executor executor, EventKeeper eventKeeper) {
super(cPtr);
this.database = database;
this.name = name;
this.executor = executor;
this.eventKeeper = eventKeeper;
}
@Override
public <T> T run(Function<? super Transaction, T> retryable, Executor e) {
Transaction t = this.createTransaction(e);
try {
while (true) {
try {
T returnVal = retryable.apply(t);
t.commit().join();
return returnVal;
} catch (RuntimeException err) {
t = t.onError(err).join();
}
}
} finally {
t.close();
}
}
@Override
public <T> T read(Function<? super ReadTransaction, T> retryable, Executor e) {
return this.run(retryable, e);
}
@Override
public <T> CompletableFuture<T> runAsync(final Function<? super Transaction, ? extends CompletableFuture<T>> retryable, Executor e) {
final AtomicReference<Transaction> trRef = new AtomicReference<>(createTransaction(e));
final AtomicReference<T> returnValue = new AtomicReference<>();
return AsyncUtil.whileTrue(() -> {
CompletableFuture<T> process = AsyncUtil.applySafely(retryable, trRef.get());
return AsyncUtil.composeHandleAsync(process.thenComposeAsync(returnVal ->
trRef.get().commit().thenApply(o -> {
returnValue.set(returnVal);
return false;
}), e),
(value, t) -> {
if(t == null)
return CompletableFuture.completedFuture(value);
if(!(t instanceof RuntimeException))
throw new CompletionException(t);
return trRef.get().onError(t).thenApply(newTr -> {
trRef.set(newTr);
return true;
});
}, e);
}, e)
.thenApply(o -> returnValue.get())
.whenComplete((v, t) -> trRef.get().close());
}
@Override
public <T> CompletableFuture<T> readAsync(
Function<? super ReadTransaction, ? extends CompletableFuture<T>> retryable, Executor e) {
return this.runAsync(retryable, e);
}
@Override
protected void finalize() throws Throwable {
try {
checkUnclosed("Tenant");
close();
}
finally {
super.finalize();
}
}
@Override
public Transaction createTransaction(Executor e) {
return createTransaction(e, eventKeeper);
}
@Override
public Transaction createTransaction(Executor e, EventKeeper eventKeeper) {
pointerReadLock.lock();
Transaction tr = null;
try {
tr = new FDBTransaction(Tenant_createTransaction(getPtr()), database, e, eventKeeper);
tr.options().setUsedDuringCommitProtectionDisable();
return tr;
} catch (RuntimeException err) {
if (tr != null) {
tr.close();
}
throw err;
} finally {
pointerReadLock.unlock();
}
}
@Override
public byte[] getName() {
return name;
}
@Override
public Executor getExecutor() {
return executor;
}
@Override
protected void closeInternal(long cPtr) {
Tenant_dispose(cPtr);
}
private native long Tenant_createTransaction(long cPtr);
private native void Tenant_dispose(long cPtr);
}

View File

@ -0,0 +1,257 @@
/*
* Tenant.java
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package com.apple.foundationdb;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.Executor;
import java.util.function.Function;
/**
* A tenant represents a named key-space within a database that can be interacted with
* transactionally.<br>
* <br>
* The simplest correct programs using tenants will make use of the methods defined
* in the {@link TransactionContext} interface. When used on a {@code Tenant} these
* methods will call {@code Transaction#commit()} after user code has been
* executed. These methods will not return successfully until {@code commit()} has
* returned successfully.<br>
* <br>
* <b>Note:</b> {@code Tenant} objects must be {@link #close closed} when no longer
* in use in order to free any associated resources.
*/
public interface Tenant extends AutoCloseable, TransactionContext {
/**
* Creates a {@link Transaction} that operates on this {@code Tenant}.<br>
* <br>
* <b>Note:</b> Java transactions automatically set the {@link TransactionOptions#setUsedDuringCommitProtectionDisable}
* option. This is because the Java bindings disallow use of {@code Transaction} objects after
* {@link Transaction#onError} is called.
*
* @return a newly created {@code Transaction} that reads from and writes to this {@code Tenant}.
*/
default Transaction createTransaction() {
return createTransaction(getExecutor());
}
/**
* Creates a {@link Transaction} that operates on this {@code Tenant} with the given {@link Executor}
* for asynchronous callbacks.
*
* @param e the {@link Executor} to use when executing asynchronous callbacks.
* @return a newly created {@code Transaction} that reads from and writes to this {@code Tenant}.
*/
Transaction createTransaction(Executor e);
/**
* Creates a {@link Transaction} that operates on this {@code Tenant} with the given {@link Executor}
* for asynchronous callbacks.
*
* @param e the {@link Executor} to use when executing asynchronous callbacks.
* @param eventKeeper the {@link EventKeeper} to use when tracking instrumented calls for the transaction.
*
* @return a newly created {@code Transaction} that reads from and writes to this {@code Tenant}.
*/
Transaction createTransaction(Executor e, EventKeeper eventKeeper);
/**
* Returns the name of this {@code Tenant}.
*
* @return the name of this {@code Tenant} as a byte string.
*/
byte[] getName();
/**
* Runs a read-only transactional function against this {@code Tenant} with retry logic.
* {@link Function#apply(Object) apply(ReadTransaction)} will be called on the
* supplied {@link Function} until a non-retryable
* FDBException (or any {@code Throwable} other than an {@code FDBException})
* is thrown. This call is blocking -- this
* method will not return until the {@code Function} has been called and completed without error.<br>
*
* @param retryable the block of logic to execute in a {@link Transaction} against
* this tenant
* @param <T> the return type of {@code retryable}
*
* @return the result of the last run of {@code retryable}
*/
@Override
default <T> T read(Function<? super ReadTransaction, T> retryable) {
return read(retryable, getExecutor());
}
/**
* Runs a read-only transactional function against this {@code Tenant} with retry logic. Use
* this formulation of {@link #read(Function)} if one wants to set a custom {@link Executor}
* for the transaction when run.
*
* @param retryable the block of logic to execute in a {@link Transaction} against
* this tenant
* @param e the {@link Executor} to use for asynchronous callbacks
* @param <T> the return type of {@code retryable}
* @return the result of the last run of {@code retryable}
*
* @see #read(Function)
*/
<T> T read(Function<? super ReadTransaction, T> retryable, Executor e);
/**
* Runs a read-only transactional function against this {@code Tenant} with retry logic.
* {@link Function#apply(Object) apply(ReadTransaction)} will be called on the
* supplied {@link Function} until a non-retryable
* FDBException (or any {@code Throwable} other than an {@code FDBException})
* is thrown. This call is non-blocking -- this
* method will return immediately and with a {@link CompletableFuture} that will be
* set when the {@code Function} has been called and completed without error.<br>
* <br>
* Any errors encountered executing {@code retryable}, or received from the
* database, will be set on the returned {@code CompletableFuture}.
*
* @param retryable the block of logic to execute in a {@link ReadTransaction} against
* this tenant
* @param <T> the return type of {@code retryable}
*
* @return a {@code CompletableFuture} that will be set to the value returned by the last call
* to {@code retryable}
*/
@Override
default <T> CompletableFuture<T> readAsync(
Function<? super ReadTransaction, ? extends CompletableFuture<T>> retryable) {
return readAsync(retryable, getExecutor());
}
/**
* Runs a read-only transactional function against this {@code Tenant} with retry logic.
* Use this version of {@link #readAsync(Function)} if one wants to set a custom
* {@link Executor} for the transaction when run.
*
* @param retryable the block of logic to execute in a {@link ReadTransaction} against
* this tenant
* @param e the {@link Executor} to use for asynchronous callbacks
* @param <T> the return type of {@code retryable}
*
* @return a {@code CompletableFuture} that will be set to the value returned by the last call
* to {@code retryable}
*
* @see #readAsync(Function)
*/
<T> CompletableFuture<T> readAsync(
Function<? super ReadTransaction, ? extends CompletableFuture<T>> retryable, Executor e);
/**
* Runs a transactional function against this {@code Tenant} with retry logic.
* {@link Function#apply(Object) apply(Transaction)} will be called on the
* supplied {@link Function} until a non-retryable
* FDBException (or any {@code Throwable} other than an {@code FDBException})
* is thrown or {@link Transaction#commit() commit()},
* when called after {@code apply()}, returns success. This call is blocking -- this
* method will not return until {@code commit()} has been called and returned success.<br>
* <br>
* As with other client/server databases, in some failure scenarios a client may
* be unable to determine whether a transaction succeeded. In these cases, your
* transaction may be executed twice. For more information about how to reason
* about these situations see
* <a href="/foundationdb/developer-guide.html#transactions-with-unknown-results"
* target="_blank">the FounationDB Developer Guide</a>
*
* @param retryable the block of logic to execute in a {@link Transaction} against
* this tenant
* @param <T> the return type of {@code retryable}
*
* @return the result of the last run of {@code retryable}
*/
@Override
default <T> T run(Function<? super Transaction, T> retryable) {
return run(retryable, getExecutor());
}
/**
* Runs a transactional function against this {@code Tenant} with retry logic.
* Use this formulation of {@link #run(Function)} if one would like to set a
* custom {@link Executor} for the transaction when run.
*
* @param retryable the block of logic to execute in a {@link Transaction} against
* this tenant
* @param e the {@link Executor} to use for asynchronous callbacks
* @param <T> the return type of {@code retryable}
*
* @return the result of the last run of {@code retryable}
*/
<T> T run(Function<? super Transaction, T> retryable, Executor e);
/**
* Runs a transactional function against this {@code Tenant} with retry logic.
* {@link Function#apply(Object) apply(Transaction)} will be called on the
* supplied {@link Function} until a non-retryable
* FDBException (or any {@code Throwable} other than an {@code FDBException})
* is thrown or {@link Transaction#commit() commit()},
* when called after {@code apply()}, returns success. This call is non-blocking -- this
* method will return immediately and with a {@link CompletableFuture} that will be
* set when {@code commit()} has been called and returned success.<br>
* <br>
* As with other client/server databases, in some failure scenarios a client may
* be unable to determine whether a transaction succeeded. In these cases, your
* transaction may be executed twice. For more information about how to reason
* about these situations see
* <a href="/foundationdb/developer-guide.html#transactions-with-unknown-results"
* target="_blank">the FounationDB Developer Guide</a><br>
* <br>
* Any errors encountered executing {@code retryable}, or received from the
* database, will be set on the returned {@code CompletableFuture}.
*
* @param retryable the block of logic to execute in a {@link Transaction} against
* this tenant
* @param <T> the return type of {@code retryable}
*
* @return a {@code CompletableFuture} that will be set to the value returned by the last call
* to {@code retryable}
*/
@Override
default <T> CompletableFuture<T> runAsync(
Function<? super Transaction, ? extends CompletableFuture<T>> retryable) {
return runAsync(retryable, getExecutor());
}
/**
* Runs a transactional function against this {@code Tenant} with retry logic. Use
* this formulation of the non-blocking {@link #runAsync(Function)} if one wants
* to set a custom {@link Executor} for the transaction when run.
*
* @param retryable the block of logic to execute in a {@link Transaction} against
* this tenant
* @param e the {@link Executor} to use for asynchronous callbacks
* @param <T> the return type of {@code retryable}
*
* @return a {@code CompletableFuture} that will be set to the value returned by the last call
* to {@code retryable}
*
* @see #run(Function)
*/
<T> CompletableFuture<T> runAsync(
Function<? super Transaction, ? extends CompletableFuture<T>> retryable, Executor e);
/**
* Close the {@code Tenant} object and release any associated resources. This must be called at
* least once after the {@code Tenant} object is no longer in use. This can be called multiple
* times, but care should be taken that it is not in use in another thread at the time of the call.
*/
@Override
void close();
}

View File

@ -0,0 +1,214 @@
/*
* TenantManagement.java
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package com.apple.foundationdb;
import java.nio.charset.Charset;
import java.util.Arrays;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.CompletionException;
import java.util.concurrent.Executor;
import java.util.concurrent.atomic.AtomicBoolean;
import java.util.function.BiFunction;
import com.apple.foundationdb.async.AsyncIterable;
import com.apple.foundationdb.async.AsyncIterator;
import com.apple.foundationdb.async.AsyncUtil;
import com.apple.foundationdb.async.CloseableAsyncIterator;
import com.apple.foundationdb.tuple.ByteArrayUtil;
import com.apple.foundationdb.tuple.Tuple;
/**
* The FoundationDB API includes function to manage the set of tenants in a cluster.
*/
public class TenantManagement {
static final byte[] TENANT_MAP_PREFIX = ByteArrayUtil.join(new byte[] { (byte)255, (byte)255 },
"/management/tenant_map/".getBytes());
/**
* Creates a new tenant in the cluster. If the tenant already exists, this operation will complete
* successfully without changing anything. The transaction must be committed for the creation to take
* effect or to observe any errors.
*
* @param tr The transaction used to create the tenant.
* @param tenantName The name of the tenant. Can be any byte string that does not begin a 0xFF byte.
*/
public static void createTenant(Transaction tr, byte[] tenantName) {
tr.options().setSpecialKeySpaceEnableWrites();
tr.set(ByteArrayUtil.join(TENANT_MAP_PREFIX, tenantName), new byte[0]);
}
/**
* Creates a new tenant in the cluster. If the tenant already exists, this operation will complete
* successfully without changing anything. The transaction must be committed for the creation to take
* effect or to observe any errors.<br>
* <br>
* This is a convenience method that generates the tenant name by packing a {@code Tuple}.
*
* @param tr The transaction used to create the tenant.
* @param tenantName The name of the tenant, as a Tuple.
*/
public static void createTenant(Transaction tr, Tuple tenantName) {
createTenant(tr, tenantName.pack());
}
/**
* Creates a new tenant in the cluster using a transaction created on the specified {@code Database}.
* This operation will first check whether the tenant exists, and if it does it will set the
* {@code CompletableFuture} to a tenant_already_exists error. Otherwise, it will attempt to create
* the tenant in a retry loop. If the tenant is created concurrently by another transaction, this
* function may still return successfully.
*
* @param db The database used to create a transaction for creating the tenant.
* @param tenantName The name of the tenant. Can be any byte string that does not begin a 0xFF byte.
* @return a {@code CompletableFuture} that when set without error will indicate that the tenant has
* been created.
*/
public static CompletableFuture<Void> createTenant(Database db, byte[] tenantName) {
final AtomicBoolean checkedExistence = new AtomicBoolean(false);
final byte[] key = ByteArrayUtil.join(TENANT_MAP_PREFIX, tenantName);
return db.runAsync(tr -> {
tr.options().setSpecialKeySpaceEnableWrites();
if(checkedExistence.get()) {
tr.set(key, new byte[0]);
return CompletableFuture.completedFuture(null);
}
else {
return tr.get(key).thenAcceptAsync(result -> {
checkedExistence.set(true);
if(result != null) {
throw new FDBException("A tenant with the given name already exists", 2132);
}
tr.set(key, new byte[0]);
});
}
});
}
/**
* Creates a new tenant in the cluster using a transaction created on the specified {@code Database}.
* This operation will first check whether the tenant exists, and if it does it will set the
* {@code CompletableFuture} to a tenant_already_exists error. Otherwise, it will attempt to create
* the tenant in a retry loop. If the tenant is created concurrently by another transaction, this
* function may still return successfully.<br>
* <br>
* This is a convenience method that generates the tenant name by packing a {@code Tuple}.
*
* @param db The database used to create a transaction for creating the tenant.
* @param tenantName The name of the tenant, as a Tuple.
* @return a {@code CompletableFuture} that when set without error will indicate that the tenant has
* been created.
*/
public static CompletableFuture<Void> createTenant(Database db, Tuple tenantName) {
return createTenant(db, tenantName.pack());
}
/**
* Deletes a tenant from the cluster. If the tenant does not exists, this operation will complete
* successfully without changing anything. The transaction must be committed for the deletion to take
* effect or to observe any errors.<br>
* <br>
* <b>Note:</b> A tenant cannot be deleted if it has any data in it. To delete a non-empty tenant, you must
* first use a clear operation to delete all of its keys.
*
* @param tr The transaction used to delete the tenant.
* @param tenantName The name of the tenant being deleted.
*/
public static void deleteTenant(Transaction tr, byte[] tenantName) {
tr.options().setSpecialKeySpaceEnableWrites();
tr.clear(ByteArrayUtil.join(TENANT_MAP_PREFIX, tenantName));
}
/**
* Deletes a tenant from the cluster. If the tenant does not exists, this operation will complete
* successfully without changing anything. The transaction must be committed for the deletion to take
* effect or to observe any errors.<br>
* <br>
* <b>Note:</b> A tenant cannot be deleted if it has any data in it. To delete a non-empty tenant, you must
* first use a clear operation to delete all of its keys.<br>
* <br>
* This is a convenience method that generates the tenant name by packing a {@code Tuple}.
*
* @param tr The transaction used to delete the tenant.
* @param tenantName The name of the tenant being deleted, as a Tuple.
*/
public static void deleteTenant(Transaction tr, Tuple tenantName) {
deleteTenant(tr, tenantName.pack());
}
/**
* Deletes a tenant from the cluster using a transaction created on the specified {@code Database}. This
* operation will first check whether the tenant exists, and if it does not it will set the
* {@code CompletableFuture} to a tenant_not_found error. Otherwise, it will attempt to delete the
* tenant in a retry loop. If the tenant is deleted concurrently by another transaction, this function may
* still return successfully.<br>
* <br>
* <b>Note:</b> A tenant cannot be deleted if it has any data in it. To delete a non-empty tenant, you must
* first use a clear operation to delete all of its keys.
*
* @param db The database used to create a transaction for deleting the tenant.
* @param tenantName The name of the tenant being deleted.
* @return a {@code CompletableFuture} that when set without error will indicate that the tenant has
* been deleted.
*/
public static CompletableFuture<Void> deleteTenant(Database db, byte[] tenantName) {
final AtomicBoolean checkedExistence = new AtomicBoolean(false);
final byte[] key = ByteArrayUtil.join(TENANT_MAP_PREFIX, tenantName);
return db.runAsync(tr -> {
tr.options().setSpecialKeySpaceEnableWrites();
if(checkedExistence.get()) {
tr.clear(key);
return CompletableFuture.completedFuture(null);
}
else {
return tr.get(key).thenAcceptAsync(result -> {
checkedExistence.set(true);
if(result == null) {
throw new FDBException("Tenant does not exist", 2131);
}
tr.clear(key);
});
}
});
}
/**
* Deletes a tenant from the cluster using a transaction created on the specified {@code Database}. This
* operation will first check whether the tenant exists, and if it does not it will set the
* {@code CompletableFuture} to a tenant_not_found error. Otherwise, it will attempt to delete the
* tenant in a retry loop. If the tenant is deleted concurrently by another transaction, this function may
* still return successfully.<br>
* <br>
* <b>Note:</b> A tenant cannot be deleted if it has any data in it. To delete a non-empty tenant, you must
* first use a clear operation to delete all of its keys.<br>
* <br>
* This is a convenience method that generates the tenant name by packing a {@code Tuple}.
*
* @param db The database used to create a transaction for deleting the tenant.
* @param tenantName The name of the tenant being deleted.
* @return a {@code CompletableFuture} that when set without error will indicate that the tenant has
* been deleted.
*/
public static CompletableFuture<Void> deleteTenant(Database db, Tuple tenantName) {
return deleteTenant(db, tenantName.pack());
}
private TenantManagement() {}
}

View File

@ -30,6 +30,7 @@ import java.util.HashMap;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.Optional;
import java.util.concurrent.CompletableFuture;
import java.util.function.Function;
@ -42,6 +43,7 @@ import com.apple.foundationdb.KeyArrayResult;
import com.apple.foundationdb.MutationType;
import com.apple.foundationdb.Range;
import com.apple.foundationdb.StreamingMode;
import com.apple.foundationdb.TenantManagement;
import com.apple.foundationdb.Transaction;
import com.apple.foundationdb.async.AsyncUtil;
import com.apple.foundationdb.tuple.ByteArrayUtil;
@ -184,7 +186,7 @@ public class AsyncStackTester {
return AsyncUtil.DONE;
}
else if(op == StackOperation.RESET) {
inst.context.newTransaction();
inst.context.resetTransaction();
return AsyncUtil.DONE;
}
else if(op == StackOperation.CANCEL) {
@ -332,9 +334,9 @@ public class AsyncStackTester {
final Transaction oldTr = inst.tr;
CompletableFuture<Void> f = oldTr.onError(err).whenComplete((tr, t) -> {
if(t != null) {
inst.context.newTransaction(oldTr); // Other bindings allow reuse of non-retryable transactions, so we need to emulate that behavior.
inst.context.resetTransaction(oldTr); // Other bindings allow reuse of non-retryable transactions, so we need to emulate that behavior.
}
else if(!inst.setTransaction(oldTr, tr)) {
else if(!inst.replaceTransaction(oldTr, tr)) {
tr.close();
}
}).thenApply(v -> null);
@ -469,6 +471,28 @@ public class AsyncStackTester {
inst.push(ByteBuffer.allocate(8).order(ByteOrder.BIG_ENDIAN).putDouble(value).array());
}, FDB.DEFAULT_EXECUTOR);
}
else if (op == StackOperation.TENANT_CREATE) {
return inst.popParam().thenAcceptAsync(param -> {
byte[] tenantName = (byte[])param;
inst.push(TenantManagement.createTenant(inst.context.db, tenantName));
}, FDB.DEFAULT_EXECUTOR);
}
else if (op == StackOperation.TENANT_DELETE) {
return inst.popParam().thenAcceptAsync(param -> {
byte[] tenantName = (byte[])param;
inst.push(TenantManagement.deleteTenant(inst.context.db, tenantName));
}, FDB.DEFAULT_EXECUTOR);
}
else if (op == StackOperation.TENANT_SET_ACTIVE) {
return inst.popParam().thenAcceptAsync(param -> {
byte[] tenantName = (byte[])param;
inst.context.setTenant(Optional.of(tenantName));
}, FDB.DEFAULT_EXECUTOR);
}
else if (op == StackOperation.TENANT_CLEAR_ACTIVE) {
inst.context.setTenant(Optional.empty());
return AsyncUtil.DONE;
}
else if(op == StackOperation.UNIT_TESTS) {
inst.context.db.options().setLocationCacheSize(100001);
return inst.context.db.runAsync(tr -> {
@ -554,7 +578,7 @@ public class AsyncStackTester {
private static CompletableFuture<Void> executeMutation(final Instruction inst, Function<Transaction, CompletableFuture<Void>> r) {
// run this with a retry loop
return inst.tcx.runAsync(r).thenRunAsync(() -> {
if(inst.isDatabase)
if(inst.isDatabase || inst.isTenant)
inst.push("RESULT_NOT_PRESENT".getBytes());
}, FDB.DEFAULT_EXECUTOR);
}

View File

@ -25,6 +25,7 @@ import java.util.HashMap;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.Optional;
import java.util.concurrent.atomic.AtomicInteger;
import java.util.concurrent.CompletableFuture;
@ -35,6 +36,7 @@ import com.apple.foundationdb.FDBException;
import com.apple.foundationdb.KeySelector;
import com.apple.foundationdb.Range;
import com.apple.foundationdb.StreamingMode;
import com.apple.foundationdb.Tenant;
import com.apple.foundationdb.Transaction;
import com.apple.foundationdb.tuple.ByteArrayUtil;
import com.apple.foundationdb.tuple.Tuple;
@ -42,15 +44,27 @@ import com.apple.foundationdb.tuple.Tuple;
abstract class Context implements Runnable, AutoCloseable {
final Stack stack = new Stack();
final Database db;
Optional<Tenant> tenant = Optional.empty();
final String preStr;
int instructionIndex = 0;
KeySelector nextKey, endKey;
Long lastVersion = null;
private static class TransactionState {
public Transaction transaction;
public Optional<Tenant> tenant;
public TransactionState(Transaction transaction, Optional<Tenant> tenant) {
this.transaction = transaction;
this.tenant = tenant;
}
}
private String trName;
private List<Thread> children = new LinkedList<>();
private static Map<String, Transaction> transactionMap = new HashMap<>();
private static Map<String, TransactionState> transactionMap = new HashMap<>();
private static Map<Transaction, AtomicInteger> transactionRefCounts = new HashMap<>();
private static Map<byte[], Tenant> tenantMap = new HashMap<>();
Context(Database db, byte[] prefix) {
this.db = db;
@ -86,15 +100,24 @@ abstract class Context implements Runnable, AutoCloseable {
}
}
public synchronized void setTenant(Optional<byte[]> tenantName) {
if (tenantName.isPresent()) {
tenant = Optional.of(tenantMap.computeIfAbsent(tenantName.get(), tn -> db.openTenant(tenantName.get())));
}
else {
tenant = Optional.empty();
}
}
public static synchronized void addTransactionReference(Transaction tr) {
transactionRefCounts.computeIfAbsent(tr, x -> new AtomicInteger(0)).incrementAndGet();
}
private static synchronized Transaction getTransaction(String trName) {
Transaction tr = transactionMap.get(trName);
assert tr != null : "Null transaction";
addTransactionReference(tr);
return tr;
TransactionState state = transactionMap.get(trName);
assert state != null : "Null transaction";
addTransactionReference(state.transaction);
return state.transaction;
}
public Transaction getCurrentTransaction() {
@ -105,59 +128,78 @@ abstract class Context implements Runnable, AutoCloseable {
if(tr != null) {
AtomicInteger count = transactionRefCounts.get(tr);
if(count.decrementAndGet() == 0) {
assert !transactionMap.containsValue(tr);
transactionRefCounts.remove(tr);
tr.close();
}
}
}
private static synchronized void updateTransaction(String trName, Transaction tr) {
releaseTransaction(transactionMap.put(trName, tr));
addTransactionReference(tr);
}
private static synchronized boolean updateTransaction(String trName, Transaction oldTr, Transaction newTr) {
boolean added;
if(oldTr == null) {
added = (transactionMap.putIfAbsent(trName, newTr) == null);
private static Transaction createTransaction(Database db, Optional<Tenant> creatingTenant) {
if (creatingTenant.isPresent()) {
return creatingTenant.get().createTransaction();
}
else {
added = transactionMap.replace(trName, oldTr, newTr);
return db.createTransaction();
}
}
private static synchronized boolean newTransaction(Database db, Optional<Tenant> tenant, String trName, boolean allowReplace) {
TransactionState oldState = transactionMap.get(trName);
if (oldState != null) {
releaseTransaction(oldState.transaction);
}
else if (!allowReplace) {
return false;
}
if(added) {
TransactionState newState = new TransactionState(createTransaction(db, tenant), tenant);
transactionMap.put(trName, newState);
addTransactionReference(newState.transaction);
return true;
}
private static synchronized boolean replaceTransaction(Database db, String trName, Transaction oldTr, Transaction newTr) {
TransactionState trState = transactionMap.get(trName);
assert trState != null : "Null transaction";
if(oldTr == null || trState.transaction == oldTr) {
if(newTr == null) {
newTr = createTransaction(db, trState.tenant);
}
releaseTransaction(trState.transaction);
addTransactionReference(newTr);
releaseTransaction(oldTr);
trState.transaction = newTr;
return true;
}
return false;
}
public void updateCurrentTransaction(Transaction tr) {
updateTransaction(trName, tr);
}
public boolean updateCurrentTransaction(Transaction oldTr, Transaction newTr) {
return updateTransaction(trName, oldTr, newTr);
}
public void newTransaction() {
Transaction tr = db.createTransaction();
updateCurrentTransaction(tr);
newTransaction(db, tenant, trName, true);
}
public void newTransaction(Transaction oldTr) {
Transaction newTr = db.createTransaction();
if(!updateCurrentTransaction(oldTr, newTr)) {
newTr.close();
}
public void replaceTransaction(Transaction tr) {
replaceTransaction(db, trName, null, tr);
}
public boolean replaceTransaction(Transaction oldTr, Transaction newTr) {
return replaceTransaction(db, trName, oldTr, newTr);
}
public void resetTransaction() {
replaceTransaction(db, trName, null, null);
}
public boolean resetTransaction(Transaction oldTr) {
return replaceTransaction(db, trName, oldTr, null);
}
public void switchTransaction(byte[] rawTrName) {
trName = ByteArrayUtil.printable(rawTrName);
newTransaction(null);
newTransaction(db, tenant, trName, false);
}
abstract void executeOperations() throws Throwable;
@ -224,8 +266,12 @@ abstract class Context implements Runnable, AutoCloseable {
@Override
public void close() {
for(Transaction tr : transactionMap.values()) {
tr.close();
for(TransactionState tr : transactionMap.values()) {
tr.transaction.close();
}
for(Tenant tenant : tenantMap.values()) {
tenant.close();
}
}
}

View File

@ -33,11 +33,13 @@ import com.apple.foundationdb.tuple.Tuple;
class Instruction extends Stack {
private static final String SUFFIX_SNAPSHOT = "_SNAPSHOT";
private static final String SUFFIX_DATABASE = "_DATABASE";
private static final String SUFFIX_TENANT = "_TENANT";
final String op;
final Tuple tokens;
final Context context;
final boolean isDatabase;
final boolean isTenant;
final boolean isSnapshot;
final Transaction tr;
final ReadTransaction readTr;
@ -49,14 +51,23 @@ class Instruction extends Stack {
this.tokens = tokens;
String fullOp = tokens.getString(0);
isDatabase = fullOp.endsWith(SUFFIX_DATABASE);
boolean isDatabaseLocal = fullOp.endsWith(SUFFIX_DATABASE);
isTenant = fullOp.endsWith(SUFFIX_TENANT);
isSnapshot = fullOp.endsWith(SUFFIX_SNAPSHOT);
if(isDatabase) {
if(isDatabaseLocal) {
tr = null;
readTr = null;
op = fullOp.substring(0, fullOp.length() - SUFFIX_DATABASE.length());
}
else if(isTenant) {
tr = null;
readTr = null;
op = fullOp.substring(0, fullOp.length() - SUFFIX_TENANT.length());
if (!context.tenant.isPresent()) {
isDatabaseLocal = true;
}
}
else if(isSnapshot) {
tr = context.getCurrentTransaction();
readTr = tr.snapshot();
@ -68,22 +79,24 @@ class Instruction extends Stack {
op = fullOp;
}
tcx = isDatabase ? context.db : tr;
readTcx = isDatabase ? context.db : readTr;
isDatabase = isDatabaseLocal;
tcx = isDatabase ? context.db : isTenant ? context.tenant.get() : tr;
readTcx = isDatabase ? context.db : isTenant ? context.tenant.get() : readTr;
}
boolean setTransaction(Transaction newTr) {
if(!isDatabase) {
context.updateCurrentTransaction(newTr);
boolean replaceTransaction(Transaction newTr) {
if(!isDatabase && !isTenant) {
context.replaceTransaction(newTr);
return true;
}
return false;
}
boolean setTransaction(Transaction oldTr, Transaction newTr) {
if(!isDatabase) {
return context.updateCurrentTransaction(oldTr, newTr);
boolean replaceTransaction(Transaction oldTr, Transaction newTr) {
if(!isDatabase && !isTenant) {
return context.replaceTransaction(oldTr, newTr);
}
return false;

View File

@ -73,5 +73,11 @@ enum StackOperation {
DECODE_DOUBLE,
UNIT_TESTS, /* Possibly unimplemented */
// Tenants
TENANT_CREATE,
TENANT_DELETE,
TENANT_SET_ACTIVE,
TENANT_CLEAR_ACTIVE,
LOG_STACK
}

View File

@ -30,6 +30,7 @@ import java.util.HashMap;
import java.util.LinkedList;
import java.util.List;
import java.util.Map;
import java.util.Optional;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.CompletionException;
import java.util.function.Function;
@ -44,11 +45,13 @@ import com.apple.foundationdb.LocalityUtil;
import com.apple.foundationdb.MutationType;
import com.apple.foundationdb.Range;
import com.apple.foundationdb.StreamingMode;
import com.apple.foundationdb.TenantManagement;
import com.apple.foundationdb.Transaction;
import com.apple.foundationdb.async.AsyncIterable;
import com.apple.foundationdb.async.AsyncUtil;
import com.apple.foundationdb.async.CloseableAsyncIterator;
import com.apple.foundationdb.tuple.ByteArrayUtil;
import com.apple.foundationdb.Tenant;
import com.apple.foundationdb.tuple.Tuple;
/**
@ -197,7 +200,7 @@ public class StackTester {
inst.tr.options().setNextWriteNoWriteConflictRange();
}
else if(op == StackOperation.RESET) {
inst.context.newTransaction();
inst.context.resetTransaction();
}
else if(op == StackOperation.CANCEL) {
inst.tr.cancel();
@ -300,12 +303,12 @@ public class StackTester {
try {
Transaction tr = inst.tr.onError(err).join();
if(!inst.setTransaction(tr)) {
if(!inst.replaceTransaction(tr)) {
tr.close();
}
}
catch(Throwable t) {
inst.context.newTransaction(); // Other bindings allow reuse of non-retryable transactions, so we need to emulate that behavior.
inst.context.resetTransaction(); // Other bindings allow reuse of non-retryable transactions, so we need to emulate that behavior.
throw t;
}
@ -418,6 +421,21 @@ public class StackTester {
double value = ((Number)param).doubleValue();
inst.push(ByteBuffer.allocate(8).order(ByteOrder.BIG_ENDIAN).putDouble(value).array());
}
else if (op == StackOperation.TENANT_CREATE) {
byte[] tenantName = (byte[])inst.popParam().join();
inst.push(TenantManagement.createTenant(inst.context.db, tenantName));
}
else if (op == StackOperation.TENANT_DELETE) {
byte[] tenantName = (byte[])inst.popParam().join();
inst.push(TenantManagement.deleteTenant(inst.context.db, tenantName));
}
else if (op == StackOperation.TENANT_SET_ACTIVE) {
byte[] tenantName = (byte[])inst.popParam().join();
inst.context.setTenant(Optional.of(tenantName));
}
else if (op == StackOperation.TENANT_CLEAR_ACTIVE) {
inst.context.setTenant(Optional.empty());
}
else if(op == StackOperation.UNIT_TESTS) {
try {
inst.context.db.options().setLocationCacheSize(100001);
@ -490,6 +508,7 @@ public class StackTester {
testWatches(inst.context.db);
testLocality(inst.context.db);
testTenantTupleNames(inst.context.db);
}
catch(Exception e) {
throw new RuntimeException("Unit tests failed: " + e.getMessage());
@ -579,7 +598,7 @@ public class StackTester {
private static void executeMutation(Instruction inst, Function<Transaction, Void> r) {
// run this with a retry loop (and commit)
inst.tcx.run(r);
if(inst.isDatabase)
if(inst.isDatabase || inst.isTenant)
inst.push("RESULT_NOT_PRESENT".getBytes());
}
@ -741,6 +760,35 @@ public class StackTester {
});
}
private static void testTenantTupleNames(Database db) {
try {
TenantManagement.createTenant(db, Tuple.from("tenant")).join();
Tenant tenant = db.openTenant(Tuple.from("tenant"));
tenant.run(tr -> {
tr.set(Tuple.from("hello").pack(), Tuple.from("world").pack());
return null;
});
String output = tenant.read(tr -> {
byte[] result = tr.get(Tuple.from("hello").pack()).join();
return Tuple.fromBytes(result).getString(0);
});
assert output.equals("world");
tenant.run(tr -> {
tr.clear(Tuple.from("hello").pack());
return null;
});
TenantManagement.deleteTenant(db, Tuple.from("tenant")).join();
}
catch(Exception e) {
e.printStackTrace();
}
}
/**
* Run a stack-machine based test.
*

View File

@ -5,6 +5,7 @@ set(SRCS
fdb/locality.py
fdb/six.py
fdb/subspace_impl.py
fdb/tenant_management.py
fdb/tuple.py
README.rst
MANIFEST.in)

View File

@ -100,6 +100,9 @@ def api_version(ver):
_add_symbols(fdb.impl, list)
if ver >= 710:
import fdb.tenant_management
if ver < 610:
globals()["init"] = getattr(fdb.impl, "init")
globals()["open"] = getattr(fdb.impl, "open_v609")

View File

@ -1178,52 +1178,6 @@ class Database(_TransactionCreator):
self.capi.fdb_database_create_transaction(self.dpointer, ctypes.byref(pointer))
return Transaction(pointer.value, self)
def allocate_tenant(self, name):
Database.__database_allocate_tenant(self, process_tenant_name(name), [])
def delete_tenant(self, name):
Database.__database_delete_tenant(self, process_tenant_name(name), [])
# Attempt to allocate a tenant in the cluster. If the tenant already exists,
# this function will return a tenant_already_exists error. If the tenant is created
# concurrently, then this function may return success even if another caller creates
# it.
#
# The existence_check_marker is expected to be an empty list. This function will
# modify the list after completing the existence check to avoid checking for existence
# on retries. This allows the operation to be idempotent.
@staticmethod
@transactional
def __database_allocate_tenant(tr, name, existence_check_marker):
tr.options.set_special_key_space_enable_writes()
key = b'\xff\xff/management/tenant_map/%s' % name
if not existence_check_marker:
existing_tenant = tr[key].wait()
existence_check_marker.append(None)
if existing_tenant != None:
raise fdb.FDBError(2132) # tenant_already_exists
tr[key] = b''
# Attempt to remove a tenant in the cluster. If the tenant doesn't exist, this
# function will return a tenant_not_found error. If the tenant is deleted
# concurrently, then this function may return success even if another caller deletes
# it.
#
# The existence_check_marker is expected to be an empty list. This function will
# modify the list after completing the existence check to avoid checking for existence
# on retries. This allows the operation to be idempotent.
@staticmethod
@transactional
def __database_delete_tenant(tr, name, existence_check_marker):
tr.options.set_special_key_space_enable_writes()
key = b'\xff\xff/management/tenant_map/%s' % name
if not existence_check_marker:
existing_tenant = tr[key].wait()
existence_check_marker.append(None)
if existing_tenant == None:
raise fdb.FDBError(2131) # tenant_not_found
del tr[key]
class Tenant(_TransactionCreator):
def __init__(self, tpointer):

View File

@ -0,0 +1,95 @@
#
# tenant_management.py
#
# This source file is part of the FoundationDB open source project
#
# Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# FoundationDB Python API
"""Documentation for this API can be found at
https://apple.github.io/foundationdb/api-python.html"""
from fdb import impl as _impl
_tenant_map_prefix = b'\xff\xff/management/tenant_map/'
# If the existence_check_marker is an empty list, then check whether the tenant exists.
# After the check, append an item to the existence_check_marker list so that subsequent
# calls to this function will not perform the existence check.
#
# If the existence_check_marker is a non-empty list, return None.
def _check_tenant_existence(tr, key, existence_check_marker, force_maybe_commited):
if not existence_check_marker:
existing_tenant = tr[key].wait()
existence_check_marker.append(None)
if force_maybe_commited:
raise _impl.FDBError(1021) # maybe_committed
return existing_tenant != None
return None
# Attempt to create a tenant in the cluster. If existence_check_marker is an empty
# list, then this function will check if the tenant already exists and fail if it does.
# Once the existence check is completed, it will not be done again if this function
# retries. As a result, this function may return successfully if the tenant is created
# by someone else concurrently. This behavior allows the operation to be idempotent with
# respect to retries.
#
# If the existence_check_marker is a non-empty list, then the existence check is skipped.
@_impl.transactional
def _create_tenant_impl(tr, tenant_name, existence_check_marker, force_existence_check_maybe_committed=False):
tr.options.set_special_key_space_enable_writes()
key = b'%s%s' % (_tenant_map_prefix, tenant_name)
if _check_tenant_existence(tr, key, existence_check_marker, force_existence_check_maybe_committed) is True:
raise _impl.FDBError(2132) # tenant_already_exists
tr[key] = b''
# Attempt to delete a tenant from the cluster. If existence_check_marker is an empty
# list, then this function will check if the tenant already exists and fail if it does
# not. Once the existence check is completed, it will not be done again if this function
# retries. As a result, this function may return successfully if the tenant is deleted
# by someone else concurrently. This behavior allows the operation to be idempotent with
# respect to retries.
#
# If the existence_check_marker is a non-empty list, then the existence check is skipped.
@_impl.transactional
def _delete_tenant_impl(tr, tenant_name, existence_check_marker, force_existence_check_maybe_committed=False):
tr.options.set_special_key_space_enable_writes()
key = b'%s%s' % (_tenant_map_prefix, tenant_name)
if _check_tenant_existence(tr, key, existence_check_marker, force_existence_check_maybe_committed) is False:
raise _impl.FDBError(2131) # tenant_not_found
del tr[key]
def create_tenant(db_or_tr, tenant_name):
tenant_name = _impl.process_tenant_name(tenant_name)
# Only perform the existence check when run using a database
# Callers using a transaction are expected to check existence themselves if required
existence_check_marker = [] if not isinstance(db_or_tr, _impl.TransactionRead) else [None]
_create_tenant_impl(db_or_tr, tenant_name, existence_check_marker)
def delete_tenant(db_or_tr, tenant_name):
tenant_name = _impl.process_tenant_name(tenant_name)
# Only perform the existence check when run using a database
# Callers using a transaction are expected to check existence themselves if required
existence_check_marker = [] if not isinstance(db_or_tr, _impl.TransactionRead) else [None]
_delete_tenant_impl(db_or_tr, tenant_name, existence_check_marker)

View File

@ -233,7 +233,8 @@ def suspend(logger):
port = address.split(':')[1]
logger.debug("Port: {}".format(port))
# use the port number to find the exact fdb process we are connecting to
pinfo = list(filter(lambda x: port in x, pinfos))
# child process like fdbserver -r flowprocess does not provide `datadir` in the command line
pinfo = list(filter(lambda x: port in x and 'datadir' in x, pinfos))
assert len(pinfo) == 1
pid = pinfo[0].split(' ')[0]
logger.debug("Pid: {}".format(pid))

View File

@ -26,9 +26,22 @@ from fdb.tuple import pack
if __name__ == '__main__':
fdb.api_version(710)
def cleanup_tenant(db, tenant_name):
try:
tenant = db.open_tenant(tenant_name)
del tenant[:]
fdb.tenant_management.delete_tenant(db, tenant_name)
except fdb.FDBError as e:
if e.code == 2131: # tenant not found
pass
else:
raise
def test_tenant_tuple_name(db):
tuplename=(b'test', b'level', b'hierarchy', 3, 1.24, 'str')
db.allocate_tenant(tuplename)
cleanup_tenant(db, tuplename)
fdb.tenant_management.create_tenant(db, tuplename)
tenant=db.open_tenant(tuplename)
tenant[b'foo'] = b'bar'
@ -36,25 +49,15 @@ def test_tenant_tuple_name(db):
assert tenant[b'foo'] == b'bar'
del tenant[b'foo']
db.delete_tenant(tuplename)
fdb.tenant_management.delete_tenant(db, tuplename)
def cleanup_tenant(db, tenant_name):
try:
tenant = db.open_tenant(tenant_name)
del tenant[:]
db.delete_tenant(tenant_name)
except fdb.FDBError as e:
if e.code == 2131: # tenant not found
pass
else:
raise
def test_tenant_operations(db):
cleanup_tenant(db, b'tenant1')
cleanup_tenant(db, b'tenant2')
db.allocate_tenant(b'tenant1')
db.allocate_tenant(b'tenant2')
fdb.tenant_management.create_tenant(db, b'tenant1')
fdb.tenant_management.create_tenant(db, b'tenant2')
tenant1 = db.open_tenant(b'tenant1')
tenant2 = db.open_tenant(b'tenant2')
@ -90,7 +93,7 @@ def test_tenant_operations(db):
assert db[prefix2 + b'tenant_test_key'] == b'tenant2'
assert db[b'tenant_test_key'] == b'no_tenant'
db.delete_tenant(b'tenant1')
fdb.tenant_management.delete_tenant(db, b'tenant1')
try:
tenant1[b'tenant_test_key']
assert False
@ -98,7 +101,7 @@ def test_tenant_operations(db):
assert e.code == 2131 # tenant not found
del tenant2[:]
db.delete_tenant(b'tenant2')
fdb.tenant_management.delete_tenant(db, b'tenant2')
assert db[prefix1 + b'tenant_test_key'] == None
assert db[prefix2 + b'tenant_test_key'] == None
@ -108,9 +111,70 @@ def test_tenant_operations(db):
assert db[b'tenant_test_key'] == None
def test_tenant_operation_retries(db):
cleanup_tenant(db, b'tenant1')
cleanup_tenant(db, b'tenant2')
# Test that the tenant creation only performs the existence check once
fdb.tenant_management._create_tenant_impl(db, b'tenant1', [], force_existence_check_maybe_committed=True)
# An attempt to create the tenant again should fail
try:
fdb.tenant_management.create_tenant(db, b'tenant1')
assert False
except fdb.FDBError as e:
assert e.code == 2132 # tenant already exists
# Using a transaction skips the existence check
tr = db.create_transaction()
fdb.tenant_management.create_tenant(tr, b'tenant1')
# Test that a concurrent tenant creation doesn't interfere with the existence check logic
tr = db.create_transaction()
existence_check_marker = []
fdb.tenant_management._create_tenant_impl(tr, b'tenant2', existence_check_marker)
fdb.tenant_management.create_tenant(db, b'tenant2')
tr = db.create_transaction()
try:
fdb.tenant_management._create_tenant_impl(tr, b'tenant2', existence_check_marker)
tr.commit().wait()
except fdb.FDBError as e:
tr.on_error(e).wait()
# Test that tenant deletion only performs the existence check once
fdb.tenant_management._delete_tenant_impl(db, b'tenant1', [], force_existence_check_maybe_committed=True)
# An attempt to delete the tenant again should fail
try:
fdb.tenant_management.delete_tenant(db, b'tenant1')
assert False
except fdb.FDBError as e:
assert e.code == 2131 # tenant not found
# Using a transaction skips the existence check
tr = db.create_transaction()
fdb.tenant_management.delete_tenant(tr, b'tenant1')
# Test that a concurrent tenant deletion doesn't interfere with the existence check logic
tr = db.create_transaction()
existence_check_marker = []
fdb.tenant_management._delete_tenant_impl(tr, b'tenant2', existence_check_marker)
fdb.tenant_management.delete_tenant(db, b'tenant2')
tr = db.create_transaction()
try:
fdb.tenant_management._delete_tenant_impl(tr, b'tenant2', existence_check_marker)
tr.commit().wait()
except fdb.FDBError as e:
tr.on_error(e).wait()
def test_tenants(db):
test_tenant_tuple_name(db)
test_tenant_operations(db)
test_tenant_operation_retries(db)
# Expect a cluster file as input. This test will write to the FDB cluster, so
# be aware of potential side effects.

View File

@ -593,11 +593,11 @@ class Tester:
inst.push(b"WAITED_FOR_EMPTY")
elif inst.op == six.u("TENANT_CREATE"):
name = inst.pop()
self.db.allocate_tenant(name)
fdb.tenant_management.create_tenant(self.db, name)
inst.push(b"RESULT_NOT_PRESENT")
elif inst.op == six.u("TENANT_DELETE"):
name = inst.pop()
self.db.delete_tenant(name)
fdb.tenant_management.delete_tenant(self.db, name)
inst.push(b"RESULT_NOT_PRESENT")
elif inst.op == six.u("TENANT_SET_ACTIVE"):
name = inst.pop()
@ -621,7 +621,8 @@ class Tester:
test_size_limit_option(db)
test_get_approximate_size(db)
test_tenants(db)
if fdb.get_api_version() >= 710:
test_tenants(db)
except fdb.FDBError as e:
print("Unit tests failed: %s" % e.description)

View File

@ -0,0 +1,237 @@
# FDB Encryption **data at-rest**
## Threat Model
The proposed solution is `able to handle` the following attacks:
* An attacker, if able to get access to any FDB cluster host or attached disk, would not be able to read the persisted data. Further, for cloud deployments, returning a cloud instance back to the cloud provider will prevent the cloud provider from reading the contents of data stored on the disk.
* Data stored on a lost or stolen FDB host persistent disk storage device cant be recovered.
The proposed solution `will not be able` to handle the following attacks:
* Encryption is enabled for data at-rest only, generating a memory dump of FDB processes could enable an attacker to read in-memory data contents.
* An FDB cluster host access, if compromised, would allow an attacker to read/write data managed by the FDB cluster.
## Goals
FoundationDB being a multi-model, easily scalable and fault-tolerant, with an ability to provide great performance even with commodity hardware, plays a critical role enabling enterprises to deploy, manage and run mission critical applications.
Data encryption support is a table-stake feature for modern day enterprise service offerings in the cloud. Customers expect, and at times warrant, that their data and metadata be fully encrypted using the latest security standards. The goal of this document includes:
* Discuss detailed design to support data at-rest encryption support for data stored in FDB clusters. Encrypting data in-transit and/or in-memory caches at various layers in the query execution pipeline (inside and external to FDB) is out of the scope of this feature.
* Isolation guarantees: the encryption domain matches with `tenant` partition semantics supported by FDB clusters. Tenants are discrete namespaces in FDB that serve as transaction domains. A tenant is a `identifier` that maps to a `prefix` within the data-FDB cluster, and all operations within a tenant are implicitly bound within a `tenant-prefix`. Refer to `Multi-Tenant FoundationDB API` documentation more details. However, it is possible to use a single encryption key for the whole cluster, in case `tenant partitioning` isnt available.
* Ease of integration with external Key Management Services enabling persisting, caching, and lookup of encryption keys.
## Config Knobs
* `ServerKnob::ENABLE_ENCRYPION` allows enable/disable encryption feature.
* `ServerKnob::ENCRYPTION_MODE` controls the encryption mode supported. The current scheme supports `AES-256-CTR` encryption mode.
## Encryption Mode
The proposal is to use strong AES-256 CTR encryption mode. Salient properties are:
* HMAC_SHA256 key hashing technique is used to derive encryption keys using a base encryption key and locally generated random number. The formula used is as follows:
```
DEK = HMAC SHA256(BEK || UID)
Where
DEK = Derived Encryption Key
BEK = Base Encryption key
UID = Host local random generated number
```
UID is an 8 byte host-local random number. Another option would have been a simple host-local incrementing counter, however, the scheme runs the risk of repeated encryption-key generation on cluster/process restarts.
* An encryption key derived using the above formula will be cached (in-memory) for a short time interval (10 mins, for instance). The encryption-key is immutable, but, the TTL approach allows refreshing encryption key by reaching out to External Encryption KeyManagement solutions, hence, supporting “restricting lifetime of an encryption” feature if implemented by Encryption Key Management solution.
* Initialization Vector (IV) selection would be random.
## Architecture
The encryption responsibilities are split across multiple modules to ensure data and metadata stored in the cluster is never persisted in plain text on any durable storages (temporary and/or long-term durable storage).
## Encryption Request Workflow
### **Write Request**
* An FDB client initiates a write transaction providing {key, value} in plaintext format.
* An FDB cluster host as part of processing a write transaction would do the following:
1. Obtain required encryption key based on the transaction request tenant information.
2. Encrypt mutations before persisting them on Transaction Logs (TLogs). As a background process, the mutations are moved to a long-term durable storage by the Storage Server processes.
Refer to the sections below for more details.
### **Read Request**
* An FDB client initiates a read transaction request.
* An FDB cluster host as part of processing request would do the following:
1. StorageServer would read desired data blocks from the persistent storage.
2. Regenerate the encryption key required to decrypt the data.
3. Decrypt data and pass results as plaintext to the caller.
Below diagram depicts the end-to-end encryption workflow detailing various modules involved and their interactions. The following section discusses detailed design for involved components.
```
_______________________________________________________
| FDB CLUSER HOST |
| |
_____________________ | ________________________ _________________ |
| | (proprietary) | | | | |
| |<---------- |--| KMS CONNECTOR | | COMMIT PROXIES | |
| ENCRYPTION KEY | | | | | | |
| MANAGEMENT SOLUTION | | |(non FDB - proprietary) | | | |
| | | |________________________| |_________________| |
| | | ^ | |
|_____________________| | | (REST API) | (Encrypt |
| | V Mutation) |
| _________________________________________ | __________________
| | | | | |
| | ENCRYPT KEYPROXY SERVER |<------|-----------| |
| |_________________________________________| | | |
| | | | BACKUP FILES |
| | (Encrypt Node) | | |
| V | | |
| _________________________________________ | | (Encrypt file) |
| | |<------|-----------| |
| | REDWOOD STORAGE SERVER | | |__________________|
| |_________________________________________| |
|_______________________________________________________|
```
## FDB Encryption
An FDB client would insert data i.e. plaintext {key, value} in a FDB cluster for persistence.
### KMS-Connector
A non-FDB process running on FDB cluster hosts enables an FDB cluster to interact with external Encryption Key Managements services. Salient features includes:
* An external (non-FDB) standalone process implementing a REST server.
* Abstracts organization specific KeyManagementService integration details. The proposed design ensures ease of integration given limited infrastructure needed to implement a local/remote REST server.
* Ensure organization specific code is implemented outside the FDB codebase.
* The KMS-Connector process is launched and maintained by the FDBMonitor. The process needs to handle the following REST endpoint:
1. GET - http://localhost/getEncryptionKey
Define a single interface returning “encryption key string in plaintext” and accepting an
JSON input which can be customized as needed:
```json
json_input_payload
{
“Version” : int // version
“KeyId” : keyId // string
}
```
Few benefits of the above proposed schemes are:
* JSON input format is extensible (adding new fields is backward compatible).
* Popular Cloud KMS “getPublicKey” API accepts “keyId” as a string, hence, API should be easy to integrate.
1. AWS: https://docs.aws.amazon.com/cli/latest/reference/kms/get-public-key.html
2. GCP: https://cloud.google.com/kms/docs/retrieve-public-key
`Future improvements`: FDBMonitor at present will launch one KMS-Connector process per FDB cluster host. Though multiple KMS-Connector processes are launched, only one process (collocated with EncryptKeyServer) would consume cluster resources. In future, possible enhancements could be:
* Enable FDBMonitor to launch “N” (configurable) processes per cluster.
* Enable the FDB cluster to manage external processes as well.
### Encrypt KeyServer
Salient features include:
* New FDB role/process to allow fetching of encryption keys from external KeyManagementService interfaces. The process connects to the KMS-Connector REST interface to fetch desired encryption keys.
* On an encryption-key fetch from KMS-Connector, it applies HMAC derivative function to generate a new encryption key and cache it in-memory. The in-memory cache is used to serve encryption key fetch requests from other FDB processes.
Given encryption keys will be needed as part of cluster-recovery, this process/role needs to be recruited at the start of the cluster-recovery process (just after the “master/sequencer” process/role recruitment). All other FDB processes will interact with this process to obtain encryption keys needed to encrypt and/or decrypt the data payload.
`Note`: An alternative would be to incorporate the functionality into the ClusterController process itself, however, having clear responsibility separation would make design more flexible and extensible in future if needed.
### Commit Proxies (CPs)
When a FDB client initiates a write transaction to insert/update data stored in a FDB cluster, the transaction is received by a CP, which then resolves the transaction by checking if the transaction is allowed. If allowed, it commits the transaction to TLogs. The proposal is to extend CP responsibilities by encrypting mutations using the desired encryption key before mutations get persisted into TLogs (durable storage). The encryption key derivation is achieved using the following formula:
```
DEK = HMAC SHA256(BEK || UID)
Where:
DEK = Derived Encryption Key
BEK = Base Encryption Key
UID = Host local random generated number
```
The Transaction State Store (commonly referred as TxnStateStore) is a Key-Value datastore used by FDB to store metadata about the database itself for bootstrap purposes. The data stored in this store plays a critical role in: guiding the transaction system to persist writes (storage tags to mutations at CPs), and managing FDB internal data movement. The TxnStateStore data gets encrypted with the desired encryption key before getting persisted on the disk queues.
As part of encryption, every Mutation would be appended by a plaintext `BlobCipherEncryptHeader` to assist decrypting the information for reads.
CPs would cache (in-memory) recently used encryption-keys to optimize network traffic due to encryption related operations. Further, the caching would improve overall performance, avoiding frequent RPC calls to EncryptKeyServer which may eventually become a scalability bottleneck. Each encryption-key in the cache has a short Time-To-Live (10 mins) and on expiry the process will interact with the EncryptKeyServer to fetch the required encryption-keys. The same caching policy is followed by the Redwood Storage Server and the Backup File processes too.
### **Caveats**
The encryption is done inline in the transaction path, which will increase the total commit latencies. Few possible ways to minimize this impact are:
* Overlap encryption operations with the CP::resolution phase, which would minimize the latency penalty per transaction at the cost of spending more CPU cycles. If needed, for production deployments, we may need to increase the number of CPs per FDB cluster.
* Implement an external process to offload encryption. If done, encryption would appear no different than the CP::resolution phase, where the process would invoke RPC calls to encrypt the buffer and wait for operation completion.
### Storage Servers
The encryption design only supports Redwood Storage Server integration, support for other storage engines is yet to be planned.
### Redwood Storage Nodes
Redwood at heart is a B+ tree and stores data in two types of nodes:
* `Non-leaf` nodes: Nodes will only store keys and not values(prefix compression is applied).
* `Leaf` Nodes: Will store `{key, value}` tuples for a given key-range.
Both above-mentioned nodes will be converted into one or more fixed size pages (likely 4K or 8K) before being persisted on a durable storage. The encryption will be performed at the node level instead of “page level”, i.e. all pages constituting a given Redwood node will be encrypted using the same encryption key generated using the following formula:
```
DEK = HMAC SHA256(BEK || UID)
Where:
DEK = Derived Encryption Key
BEK = Base Encryption Key
UID = Host local random generated number
```
### Backup Files
Backup Files are designed to pull committed mutations from StorageServers and persist them as “files” stored on cloud backed BlobStorage such as Amazon S3. Each persisted file stores mutations for a given key-range and will be encrypted by generating an encryption key using below formula:
```
DEK = HMAC SHA256(BEK || FID)
Where:
DEK = Derived Encryption Key
BEK = Base Encryption Key
FID = File Identifier (unique)
```
## Decryption on Reads
To assist reads, FDB processes (StorageServers, Backup Files workers) will be modified to read/parse the encryption header. The data decryption will be done as follows:
* The FDB process will interact with Encrypt KeyServer to fetch the desired base encryption key corresponding to the key-id persisted in the encryption header.
* Reconstruct the encryption key and decrypt the data block.
## Future Work
* Extend the TLog API to allow clients to read “plaintext mutations” directly from a TLogServer. In current implementations there are two consumers of TLogs:
1. Storage Server: At present the plan is for StorageServer to decrypt the mutations.
2. BackupWorker (Apple implementation) which is currently not used in the code.

View File

@ -322,23 +322,11 @@ A |database-blurb1| |database-blurb2|
The tenant name can be either a byte string or a tuple. If a tuple is provided, the tuple will be packed using the tuple layer to generate the byte string tenant name.
.. note :: Opening a tenant does not check its existence in the cluster. If the tenant does not exist, attempts to read or write data with it will fail.
.. |sync-read| replace:: This read is fully synchronous.
.. |sync-write| replace:: This change will be committed immediately, and is fully synchronous.
.. method:: Database.allocate_tenant(tenant_name):
Creates a new tenant in the cluster. |sync-write|
The tenant name can be either a byte string or a tuple and cannot start with the ``\xff`` byte. If a tuple is provided, the tuple will be packed using the tuple layer to generate the byte string tenant name.
.. method:: Database.delete_tenant(tenant_name):
Delete a tenant from the cluster. |sync-write|
The tenant name can be either a byte string or a tuple. If a tuple is provided, the tuple will be packed using the tuple layer to generate the byte string tenant name.
It is an error to delete a tenant that still has data. To delete a non-empty tenant, first clear all of the keys in the tenant.
.. method:: Database.get(key)
Returns the value associated with the specified key in the database (or ``None`` if the key does not exist). |sync-read|
@ -1590,3 +1578,32 @@ Locality information
.. method:: fdb.locality.get_addresses_for_key(tr, key)
Returns a :class:`fdb.FutureStringArray`. You must call the :meth:`fdb.Future.wait()` method on this object to retrieve a list of public network addresses as strings, one for each of the storage servers responsible for storing ``key`` and its associated value.
Tenant management
=================
.. module:: fdb.tenant_management
The FoundationDB API includes functions to manage the set of tenants in a cluster.
.. method:: fdb.tenant_management.create_tenant(db_or_tr, tenant_name)
Creates a new tenant in the cluster.
The tenant name can be either a byte string or a tuple and cannot start with the ``\xff`` byte. If a tuple is provided, the tuple will be packed using the tuple layer to generate the byte string tenant name.
If a database is provided to this function for the ``db_or_tr`` parameter, then this function will first check if the tenant already exists. If it does, it will fail with a ``tenant_already_exists`` error. Otherwise, it will create a transaction and attempt to create the tenant in a retry loop. If the tenant is created concurrently by another transaction, this function may still return successfully.
If a transaction is provided to this function for the ``db_or_tr`` parameter, then this function will not check if the tenant already exists. It is up to the user to perform that check if required. The user must also successfully commit the transaction in order for the creation to take effect.
.. method:: fdb.tenant_management.delete_tenant(db_or_tr, tenant_name)
Delete a tenant from the cluster.
The tenant name can be either a byte string or a tuple. If a tuple is provided, the tuple will be packed using the tuple layer to generate the byte string tenant name.
It is an error to delete a tenant that still has data. To delete a non-empty tenant, first clear all of the keys in the tenant.
If a database is provided to this function for the ``db_or_tr`` parameter, then this function will first check if the tenant already exists. If it does not, it will fail with a ``tenant_not_found`` error. Otherwise, it will create a transaction and attempt to delete the tenant in a retry loop. If the tenant is deleted concurrently by another transaction, this function may still return successfully.
If a transaction is provided to this function for the ``db_or_tr`` parameter, then this function will not check if the tenant already exists. It is up to the user to perform that check if required. The user must also successfully commit the transaction in order for the deletion to take effect.

View File

@ -26,6 +26,8 @@ FoundationDB supports language bindings for application development using the or
* :doc:`known-limitations` describes both long-term design limitations of FoundationDB and short-term limitations applicable to the current version.
* :doc:`tenants` describes the use of the tenants feature to define named transaction domains.
.. toctree::
:maxdepth: 1
:titlesonly:
@ -42,3 +44,4 @@ FoundationDB supports language bindings for application development using the or
known-limitations
transaction-profiler-analyzer
api-version-upgrade-guide
tenants

View File

@ -64,7 +64,7 @@ The ``commit`` command commits the current transaction. Any sets or clears execu
configure
---------
The ``configure`` command changes the database configuration. Its syntax is ``configure [new|tss] [single|double|triple|three_data_hall|three_datacenter] [ssd|memory] [grv_proxies=<N>] [commit_proxies=<N>] [resolvers=<N>] [logs=<N>] [count=<TSS_COUNT>] [perpetual_storage_wiggle=<WIGGLE_SPEED>] [perpetual_storage_wiggle_locality=<<LOCALITY_KEY>:<LOCALITY_VALUE>|0>] [storage_migration_type={disabled|aggressive|gradual}]``.
The ``configure`` command changes the database configuration. Its syntax is ``configure [new|tss] [single|double|triple|three_data_hall|three_datacenter] [ssd|memory] [grv_proxies=<N>] [commit_proxies=<N>] [resolvers=<N>] [logs=<N>] [count=<TSS_COUNT>] [perpetual_storage_wiggle=<WIGGLE_SPEED>] [perpetual_storage_wiggle_locality=<<LOCALITY_KEY>:<LOCALITY_VALUE>|0>] [storage_migration_type={disabled|aggressive|gradual}] [tenant_mode={disabled|optional_experimental|required_experimental}]``.
The ``new`` option, if present, initializes a new database with the given configuration rather than changing the configuration of an existing one. When ``new`` is used, both a redundancy mode and a storage engine must be specified.

View File

@ -273,6 +273,16 @@ Directory partitions have the following drawbacks, and in general they should no
* Directories in a partition have longer prefixes than their counterparts outside of partitions, which reduces performance. Nesting partitions inside of other partitions results in even longer prefixes.
* The root directory of a partition cannot be used to pack/unpack keys and therefore cannot be used to create subspaces. You must create at least one subdirectory of a partition in order to store content in it.
Tenants
-------
:doc:`tenants` in FoundationDB provide a way to divide the cluster key-space into named transaction domains. Each tenant has a byte-string name that can be used to open transactions on the tenant's data, and tenant transactions are not permitted to access data outside of the tenant. Tenants can be useful for enforcing separation between unrelated use-cases.
Tenants and directories
~~~~~~~~~~~~~~~~~~~~~~~
Because tenants enforce that transactions operate within the tenant boundaries, it is not recommended to use a global directory layer shared between tenants. It is possible, however, to use the directory layer within each tenant. To do so, simply use the directory layer as normal with tenant transactions.
Working with the APIs
=====================

View File

@ -701,7 +701,7 @@
"ssd-1",
"ssd-2",
"ssd-redwood-1-experimental",
"ssd-rocksdb-experimental",
"ssd-rocksdb-v1",
"memory",
"memory-1",
"memory-2",
@ -714,7 +714,7 @@
"ssd-1",
"ssd-2",
"ssd-redwood-1-experimental",
"ssd-rocksdb-experimental",
"ssd-rocksdb-v1",
"memory",
"memory-1",
"memory-2",

View File

@ -0,0 +1,60 @@
#######
Tenants
#######
.. warning :: Tenants are currently experimental and are not recommended for use in production.
FoundationDB provides a feature called tenants that allow you to configure one or more named transaction domains in your cluster. A transaction domain is a key-space in which a transaction is allowed to operate, and no tenant operations are allowed to use keys outside the tenant key-space. Tenants can be useful for managing separate, unrelated use-cases and preventing them from interfering with each other. They can also be helpful for defining safe boundaries when moving a subset of data between clusters.
By default, FoundationDB has a single transaction domain that contains both the normal key-space (``['', '\xff')``) as well as the system keys (``['\xff', '\xff\xff')``) and the :doc:`special-keys` (``['\xff\xff', '\xff\xff\xff')``).
Overview
========
A tenant in a FoundationDB cluster maps a byte-string name to a key-space that can be used to store data associated with that tenant. This key-space is stored in the clusters global key-space under a prefix assigned to that tenant, with each tenant being assigned a separate non-intersecting prefix.
In addition to being each assigned a separate tenant prefix, tenants can be configured to have a common shared prefix. By default, the shared prefix is empty and tenants are allocated prefixes throughout the normal key-space. To configure an alternate shared prefix, set the ``\xff/tenantDataPrefix`` key to have the desired prefix as the value.
Tenant operations are implicitly confined to the key-space associated with the tenant. It is not necessary for client applications to use or be aware of the prefix assigned to the tenant.
Enabling tenants
================
In order to use tenants, the cluster must be configured with an appropriate tenant mode using ``fdbcli``::
fdb> configure tenant_mode=<MODE>
FoundationDB clusters support the following tenant modes:
* ``disabled`` - Tenants cannot be created or used. Disabled is the default tenant mode.
* ``optional_experimental`` - Tenants can be created. Each transaction can choose whether or not to use a tenant. This mode is primarily intended for migration and testing purposes, and care should be taken to avoid conflicts between tenant and non-tenant data.
* ``required_experimental`` - Tenants can be created. Each normal transaction must use a tenant. To support special access needs, transactions will be permitted to access the raw key-space using the ``RAW_ACCESS`` transaction option.
Creating and deleting tenants
=============================
Tenants can be created and deleted using the ``\xff\xff/management/tenant_map/<tenant_name>`` :doc:`special key <special-keys>` range as well as by using APIs provided in some language bindings.
Tenants can be created with any byte-string name that does not begin with the ``\xff`` character. Once created, a tenant will be assigned an ID and a prefix where its data will reside.
In order to delete a tenant, it must first be empty. If a tenant contains any keys, they must be cleared prior to deleting the tenant.
Using tenants
=============
In order to use the key-space associated with an existing tenant, you must open the tenant using the ``Database`` object provided by your language binding. The resulting ``Tenant`` object can be used to create transactions much like with a ``Database``, and the resulting transactions will be restricted to the tenant's key-space.
All operations performed within a tenant transaction will occur within the tenant key-space. It is not necessary to use or even be aware of the prefix assigned to a tenant in the global key-space. Operations that could resolve outside of the tenant key-space (e.g. resolving key selectors) will be clamped to the tenant.
.. note :: Tenant transactions are not permitted to access system keys.
Raw access
----------
When operating in the tenant mode ``required_experimental``, transactions are not ordinarily permitted to run without using a tenant. In order to access the system keys or perform maintenance operations that span multiple tenants, it is required to use the ``RAW_ACCESS`` transaction option to access the global key-space. It is an error to specify ``RAW_ACCESS`` on a transaction that is configured to use a tenant.
.. note :: Setting the ``READ_SYSTEM_KEYS`` or ``ACCESS_SYSTEM_KEYS`` options implies ``RAW_ACCESS`` for your transaction.
.. note :: Many :doc:`special keys <special-keys>` operations access parts of the system keys and will implictly enable raw access on the transactions in which they are used.
.. warning :: Care should be taken when using raw access to run transactions spanning multiple tenants if the tenant feature is being utilized to aid in moving data between clusters. In such scenarios, it may not be guaranteed that all of the data you intend to access is on a single cluster.

View File

@ -101,6 +101,7 @@ std::vector<LogFile> getRelevantLogFiles(const std::vector<LogFile>& files, Vers
struct ConvertParams {
std::string container_url;
Optional<std::string> proxy;
Version begin = invalidVersion;
Version end = invalidVersion;
bool log_enabled = false;
@ -112,6 +113,10 @@ struct ConvertParams {
std::string s;
s.append("ContainerURL:");
s.append(container_url);
if (proxy.present()) {
s.append(" Proxy:");
s.append(proxy.get());
}
s.append(" Begin:");
s.append(format("%" PRId64, begin));
s.append(" End:");
@ -448,7 +453,8 @@ private:
};
ACTOR Future<Void> convert(ConvertParams params) {
state Reference<IBackupContainer> container = IBackupContainer::openContainer(params.container_url);
state Reference<IBackupContainer> container =
IBackupContainer::openContainer(params.container_url, params.proxy, {});
state BackupFileList listing = wait(container->dumpFileList());
std::sort(listing.logs.begin(), listing.logs.end());
TraceEvent("Container").detail("URL", params.container_url).detail("Logs", listing.logs.size());

View File

@ -94,6 +94,7 @@ void printBuildInformation() {
struct DecodeParams {
std::string container_url;
Optional<std::string> proxy;
std::string fileFilter; // only files match the filter will be decoded
bool log_enabled = true;
std::string log_dir, trace_format, trace_log_group;
@ -115,6 +116,10 @@ struct DecodeParams {
std::string s;
s.append("ContainerURL: ");
s.append(container_url);
if (proxy.present()) {
s.append(", Proxy: ");
s.append(proxy.get());
}
s.append(", FileFilter: ");
s.append(fileFilter);
if (log_enabled) {
@ -526,7 +531,8 @@ ACTOR Future<Void> process_file(Reference<IBackupContainer> container, LogFile f
}
ACTOR Future<Void> decode_logs(DecodeParams params) {
state Reference<IBackupContainer> container = IBackupContainer::openContainer(params.container_url);
state Reference<IBackupContainer> container =
IBackupContainer::openContainer(params.container_url, params.proxy, {});
state UID uid = deterministicRandom()->randomUniqueID();
state BackupFileList listing = wait(container->dumpFileList());
// remove partitioned logs

View File

@ -130,6 +130,7 @@ enum {
OPT_USE_PARTITIONED_LOG,
// Backup and Restore constants
OPT_PROXY,
OPT_TAGNAME,
OPT_BACKUPKEYS,
OPT_WAITFORDONE,
@ -234,6 +235,7 @@ CSimpleOpt::SOption g_rgBackupStartOptions[] = {
{ OPT_NOSTOPWHENDONE, "--no-stop-when-done", SO_NONE },
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
// Enable "-p" option after GA
// { OPT_USE_PARTITIONED_LOG, "-p", SO_NONE },
{ OPT_USE_PARTITIONED_LOG, "--partitioned-log-experimental", SO_NONE },
@ -294,6 +296,7 @@ CSimpleOpt::SOption g_rgBackupModifyOptions[] = {
{ OPT_MOD_VERIFY_UID, "--verify-uid", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_SNAPSHOTINTERVAL, "-s", SO_REQ_SEP },
{ OPT_SNAPSHOTINTERVAL, "--snapshot-interval", SO_REQ_SEP },
{ OPT_MOD_ACTIVE_INTERVAL, "--active-snapshot-interval", SO_REQ_SEP },
@ -482,6 +485,7 @@ CSimpleOpt::SOption g_rgBackupExpireOptions[] = {
{ OPT_CLUSTERFILE, "--cluster-file", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_TRACE, "--log", SO_NONE },
{ OPT_TRACE_DIR, "--logdir", SO_REQ_SEP },
{ OPT_TRACE_FORMAT, "--trace-format", SO_REQ_SEP },
@ -517,6 +521,7 @@ CSimpleOpt::SOption g_rgBackupDeleteOptions[] = {
#endif
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_TRACE, "--log", SO_NONE },
{ OPT_TRACE_DIR, "--logdir", SO_REQ_SEP },
{ OPT_TRACE_FORMAT, "--trace-format", SO_REQ_SEP },
@ -546,6 +551,7 @@ CSimpleOpt::SOption g_rgBackupDescribeOptions[] = {
{ OPT_CLUSTERFILE, "--cluster-file", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_TRACE, "--log", SO_NONE },
{ OPT_TRACE_DIR, "--logdir", SO_REQ_SEP },
{ OPT_TRACE_FORMAT, "--trace-format", SO_REQ_SEP },
@ -578,6 +584,7 @@ CSimpleOpt::SOption g_rgBackupDumpOptions[] = {
{ OPT_CLUSTERFILE, "--cluster-file", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_TRACE, "--log", SO_NONE },
{ OPT_TRACE_DIR, "--logdir", SO_REQ_SEP },
{ OPT_TRACE_LOG_GROUP, "--loggroup", SO_REQ_SEP },
@ -652,6 +659,7 @@ CSimpleOpt::SOption g_rgBackupQueryOptions[] = {
{ OPT_RESTORE_TIMESTAMP, "--query-restore-timestamp", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "-d", SO_REQ_SEP },
{ OPT_DESTCONTAINER, "--destcontainer", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_RESTORE_VERSION, "-qrv", SO_REQ_SEP },
{ OPT_RESTORE_VERSION, "--query-restore-version", SO_REQ_SEP },
{ OPT_BACKUPKEYS_FILTER, "-k", SO_REQ_SEP },
@ -689,6 +697,7 @@ CSimpleOpt::SOption g_rgRestoreOptions[] = {
{ OPT_RESTORE_TIMESTAMP, "--timestamp", SO_REQ_SEP },
{ OPT_KNOB, "--knob-", SO_REQ_SEP },
{ OPT_RESTORECONTAINER, "-r", SO_REQ_SEP },
{ OPT_PROXY, "--proxy", SO_REQ_SEP },
{ OPT_PREFIX_ADD, "--add-prefix", SO_REQ_SEP },
{ OPT_PREFIX_REMOVE, "--remove-prefix", SO_REQ_SEP },
{ OPT_TAGNAME, "-t", SO_REQ_SEP },
@ -1920,6 +1929,7 @@ ACTOR Future<Void> submitDBBackup(Database src,
ACTOR Future<Void> submitBackup(Database db,
std::string url,
Optional<std::string> proxy,
int initialSnapshotIntervalSeconds,
int snapshotIntervalSeconds,
Standalone<VectorRef<KeyRangeRef>> backupRanges,
@ -1977,6 +1987,7 @@ ACTOR Future<Void> submitBackup(Database db,
else {
wait(backupAgent.submitBackup(db,
KeyRef(url),
proxy,
initialSnapshotIntervalSeconds,
snapshotIntervalSeconds,
tagName,
@ -2260,8 +2271,9 @@ ACTOR Future<Void> changeDBBackupResumed(Database src, Database dest, bool pause
}
Reference<IBackupContainer> openBackupContainer(const char* name,
std::string destinationContainer,
Optional<std::string> const& encryptionKeyFile = {}) {
const std::string& destinationContainer,
const Optional<std::string>& proxy,
const Optional<std::string>& encryptionKeyFile) {
// Error, if no dest container was specified
if (destinationContainer.empty()) {
fprintf(stderr, "ERROR: No backup destination was specified.\n");
@ -2271,7 +2283,7 @@ Reference<IBackupContainer> openBackupContainer(const char* name,
Reference<IBackupContainer> c;
try {
c = IBackupContainer::openContainer(destinationContainer, encryptionKeyFile);
c = IBackupContainer::openContainer(destinationContainer, proxy, encryptionKeyFile);
} catch (Error& e) {
std::string msg = format("ERROR: '%s' on URL '%s'", e.what(), destinationContainer.c_str());
if (e.code() == error_code_backup_invalid_url && !IBackupContainer::lastOpenError.empty()) {
@ -2291,6 +2303,7 @@ ACTOR Future<Void> runRestore(Database db,
std::string originalClusterFile,
std::string tagName,
std::string container,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> ranges,
Version beginVersion,
Version targetVersion,
@ -2339,7 +2352,7 @@ ACTOR Future<Void> runRestore(Database db,
state FileBackupAgent backupAgent;
state Reference<IBackupContainer> bc =
openBackupContainer(exeRestore.toString().c_str(), container, encryptionKeyFile);
openBackupContainer(exeRestore.toString().c_str(), container, proxy, encryptionKeyFile);
// If targetVersion is unset then use the maximum restorable version from the backup description
if (targetVersion == invalidVersion) {
@ -2368,6 +2381,7 @@ ACTOR Future<Void> runRestore(Database db,
origDb,
KeyRef(tagName),
KeyRef(container),
proxy,
ranges,
waitForDone,
targetVersion,
@ -2411,6 +2425,7 @@ ACTOR Future<Void> runRestore(Database db,
ACTOR Future<Void> runFastRestoreTool(Database db,
std::string tagName,
std::string container,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> ranges,
Version dbVersion,
bool performRestore,
@ -2440,7 +2455,7 @@ ACTOR Future<Void> runFastRestoreTool(Database db,
if (performRestore) {
if (dbVersion == invalidVersion) {
TraceEvent("FastRestoreTool").detail("TargetRestoreVersion", "Largest restorable version");
BackupDescription desc = wait(IBackupContainer::openContainer(container)->describeBackup());
BackupDescription desc = wait(IBackupContainer::openContainer(container, proxy, {})->describeBackup());
if (!desc.maxRestorableVersion.present()) {
fprintf(stderr, "The specified backup is not restorable to any version.\n");
throw restore_error();
@ -2457,6 +2472,7 @@ ACTOR Future<Void> runFastRestoreTool(Database db,
KeyRef(tagName),
ranges,
KeyRef(container),
proxy,
dbVersion,
LockDB::True,
randomUID,
@ -2478,7 +2494,7 @@ ACTOR Future<Void> runFastRestoreTool(Database db,
restoreVersion = dbVersion;
} else {
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(container);
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(container, proxy, {});
state BackupDescription description = wait(bc->describeBackup());
if (dbVersion <= 0) {
@ -2522,9 +2538,10 @@ ACTOR Future<Void> runFastRestoreTool(Database db,
ACTOR Future<Void> dumpBackupData(const char* name,
std::string destinationContainer,
Optional<std::string> proxy,
Version beginVersion,
Version endVersion) {
state Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer);
state Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer, proxy, {});
if (beginVersion < 0 || endVersion < 0) {
BackupDescription desc = wait(c->describeBackup());
@ -2552,6 +2569,7 @@ ACTOR Future<Void> dumpBackupData(const char* name,
ACTOR Future<Void> expireBackupData(const char* name,
std::string destinationContainer,
Optional<std::string> proxy,
Version endVersion,
std::string endDatetime,
Database db,
@ -2577,7 +2595,7 @@ ACTOR Future<Void> expireBackupData(const char* name,
}
try {
Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer, encryptionKeyFile);
Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer, proxy, encryptionKeyFile);
state IBackupContainer::ExpireProgress progress;
state std::string lastProgress;
@ -2623,9 +2641,11 @@ ACTOR Future<Void> expireBackupData(const char* name,
return Void();
}
ACTOR Future<Void> deleteBackupContainer(const char* name, std::string destinationContainer) {
ACTOR Future<Void> deleteBackupContainer(const char* name,
std::string destinationContainer,
Optional<std::string> proxy) {
try {
state Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer);
state Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer, proxy, {});
state int numDeleted = 0;
state Future<Void> done = c->deleteContainer(&numDeleted);
@ -2657,12 +2677,13 @@ ACTOR Future<Void> deleteBackupContainer(const char* name, std::string destinati
ACTOR Future<Void> describeBackup(const char* name,
std::string destinationContainer,
Optional<std::string> proxy,
bool deep,
Optional<Database> cx,
bool json,
Optional<std::string> encryptionKeyFile) {
try {
Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer, encryptionKeyFile);
Reference<IBackupContainer> c = openBackupContainer(name, destinationContainer, proxy, encryptionKeyFile);
state BackupDescription desc = wait(c->describeBackup(deep));
if (cx.present())
wait(desc.resolveVersionTimes(cx.get()));
@ -2688,6 +2709,7 @@ static void reportBackupQueryError(UID operationId, JsonBuilderObject& result, s
// resolved to that timestamp.
ACTOR Future<Void> queryBackup(const char* name,
std::string destinationContainer,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> keyRangesFilter,
Version restoreVersion,
std::string originalClusterFile,
@ -2734,7 +2756,7 @@ ACTOR Future<Void> queryBackup(const char* name,
}
try {
state Reference<IBackupContainer> bc = openBackupContainer(name, destinationContainer);
state Reference<IBackupContainer> bc = openBackupContainer(name, destinationContainer, proxy, {});
if (restoreVersion == invalidVersion) {
BackupDescription desc = wait(bc->describeBackup());
if (desc.maxRestorableVersion.present()) {
@ -2814,9 +2836,9 @@ ACTOR Future<Void> queryBackup(const char* name,
return Void();
}
ACTOR Future<Void> listBackup(std::string baseUrl) {
ACTOR Future<Void> listBackup(std::string baseUrl, Optional<std::string> proxy) {
try {
std::vector<std::string> containers = wait(IBackupContainer::listContainers(baseUrl));
std::vector<std::string> containers = wait(IBackupContainer::listContainers(baseUrl, proxy));
for (std::string container : containers) {
printf("%s\n", container.c_str());
}
@ -2852,6 +2874,7 @@ ACTOR Future<Void> listBackupTags(Database cx) {
struct BackupModifyOptions {
Optional<std::string> verifyUID;
Optional<std::string> destURL;
Optional<std::string> proxy;
Optional<int> snapshotIntervalSeconds;
Optional<int> activeSnapshotIntervalSeconds;
bool hasChanges() const {
@ -2869,7 +2892,7 @@ ACTOR Future<Void> modifyBackup(Database db, std::string tagName, BackupModifyOp
state Reference<IBackupContainer> bc;
if (options.destURL.present()) {
bc = openBackupContainer(exeBackup.toString().c_str(), options.destURL.get());
bc = openBackupContainer(exeBackup.toString().c_str(), options.destURL.get(), options.proxy, {});
try {
wait(timeoutError(bc->create(), 30));
} catch (Error& e) {
@ -3342,6 +3365,7 @@ int main(int argc, char* argv[]) {
break;
}
Optional<std::string> proxy;
std::string destinationContainer;
bool describeDeep = false;
bool describeTimestamps = false;
@ -3595,6 +3619,14 @@ int main(int argc, char* argv[]) {
return FDB_EXIT_ERROR;
}
break;
case OPT_PROXY:
proxy = args->OptionArg();
if (!Hostname::isHostname(proxy.get()) && !NetworkAddress::parseOptional(proxy.get()).present()) {
fprintf(stderr, "ERROR: Proxy format should be either IP:port or host:port\n");
return FDB_EXIT_ERROR;
}
modifyOptions.proxy = proxy;
break;
case OPT_DESTCONTAINER:
destinationContainer = args->OptionArg();
// If the url starts with '/' then prepend "file://" for backwards compatibility
@ -3962,9 +3994,10 @@ int main(int argc, char* argv[]) {
if (!initCluster())
return FDB_EXIT_ERROR;
// Test out the backup url to make sure it parses. Doesn't test to make sure it's actually writeable.
openBackupContainer(argv[0], destinationContainer, encryptionKeyFile);
openBackupContainer(argv[0], destinationContainer, proxy, encryptionKeyFile);
f = stopAfter(submitBackup(db,
destinationContainer,
proxy,
initialSnapshotIntervalSeconds,
snapshotIntervalSeconds,
backupKeys,
@ -4036,6 +4069,7 @@ int main(int argc, char* argv[]) {
}
f = stopAfter(expireBackupData(argv[0],
destinationContainer,
proxy,
expireVersion,
expireDatetime,
db,
@ -4047,7 +4081,7 @@ int main(int argc, char* argv[]) {
case BackupType::DELETE_BACKUP:
initTraceFile();
f = stopAfter(deleteBackupContainer(argv[0], destinationContainer));
f = stopAfter(deleteBackupContainer(argv[0], destinationContainer, proxy));
break;
case BackupType::DESCRIBE:
@ -4060,6 +4094,7 @@ int main(int argc, char* argv[]) {
// given, but quietly skip them if not.
f = stopAfter(describeBackup(argv[0],
destinationContainer,
proxy,
describeDeep,
describeTimestamps ? Optional<Database>(db) : Optional<Database>(),
jsonOutput,
@ -4068,7 +4103,7 @@ int main(int argc, char* argv[]) {
case BackupType::LIST:
initTraceFile();
f = stopAfter(listBackup(baseUrl));
f = stopAfter(listBackup(baseUrl, proxy));
break;
case BackupType::TAGS:
@ -4081,6 +4116,7 @@ int main(int argc, char* argv[]) {
initTraceFile();
f = stopAfter(queryBackup(argv[0],
destinationContainer,
proxy,
backupKeysFilter,
restoreVersion,
restoreClusterFileOrig,
@ -4090,7 +4126,7 @@ int main(int argc, char* argv[]) {
case BackupType::DUMP:
initTraceFile();
f = stopAfter(dumpBackupData(argv[0], destinationContainer, dumpBegin, dumpEnd));
f = stopAfter(dumpBackupData(argv[0], destinationContainer, proxy, dumpBegin, dumpEnd));
break;
case BackupType::UNDEFINED:
@ -4141,6 +4177,7 @@ int main(int argc, char* argv[]) {
restoreClusterFileOrig,
tagName,
restoreContainer,
proxy,
backupKeys,
beginVersion,
restoreVersion,
@ -4218,6 +4255,7 @@ int main(int argc, char* argv[]) {
f = stopAfter(runFastRestoreTool(db,
tagName,
restoreContainer,
proxy,
backupKeys,
restoreVersion,
!dryRun,

View File

@ -51,7 +51,9 @@ ACTOR Future<bool> createTenantCommandActor(Reference<IDatabase> db, std::vector
tr->setOption(FDBTransactionOptions::SPECIAL_KEY_SPACE_ENABLE_WRITES);
try {
if (!doneExistenceCheck) {
Optional<Value> existingTenant = wait(safeThreadFutureToFuture(tr->get(tenantNameKey)));
// Hold the reference to the standalone's memory
state ThreadFuture<Optional<Value>> existingTenantFuture = tr->get(tenantNameKey);
Optional<Value> existingTenant = wait(safeThreadFutureToFuture(existingTenantFuture));
if (existingTenant.present()) {
throw tenant_already_exists();
}
@ -96,7 +98,9 @@ ACTOR Future<bool> deleteTenantCommandActor(Reference<IDatabase> db, std::vector
tr->setOption(FDBTransactionOptions::SPECIAL_KEY_SPACE_ENABLE_WRITES);
try {
if (!doneExistenceCheck) {
Optional<Value> existingTenant = wait(safeThreadFutureToFuture(tr->get(tenantNameKey)));
// Hold the reference to the standalone's memory
state ThreadFuture<Optional<Value>> existingTenantFuture = tr->get(tenantNameKey);
Optional<Value> existingTenant = wait(safeThreadFutureToFuture(existingTenantFuture));
if (!existingTenant.present()) {
throw tenant_not_found();
}
@ -163,8 +167,10 @@ ACTOR Future<bool> listTenantsCommandActor(Reference<IDatabase> db, std::vector<
loop {
try {
RangeResult tenants = wait(safeThreadFutureToFuture(
tr->getRange(firstGreaterOrEqual(beginTenantKey), firstGreaterOrEqual(endTenantKey), limit)));
// Hold the reference to the standalone's memory
state ThreadFuture<RangeResult> kvsFuture =
tr->getRange(firstGreaterOrEqual(beginTenantKey), firstGreaterOrEqual(endTenantKey), limit);
RangeResult tenants = wait(safeThreadFutureToFuture(kvsFuture));
if (tenants.empty()) {
if (tokens.size() == 1) {
@ -213,7 +219,9 @@ ACTOR Future<bool> getTenantCommandActor(Reference<IDatabase> db, std::vector<St
loop {
try {
Optional<Value> tenant = wait(safeThreadFutureToFuture(tr->get(tenantNameKey)));
// Hold the reference to the standalone's memory
state ThreadFuture<Optional<Value>> tenantFuture = tr->get(tenantNameKey);
Optional<Value> tenant = wait(safeThreadFutureToFuture(tenantFuture));
if (!tenant.present()) {
throw tenant_not_found();
}

View File

@ -354,10 +354,13 @@ static std::vector<std::vector<StringRef>> parseLine(std::string& line, bool& er
forcetoken = true;
break;
case ' ':
case '\n':
case '\t':
case '\r':
if (!quoted) {
if (i > offset || (forcetoken && i == offset))
buf.push_back(StringRef((uint8_t*)(line.data() + offset), i - offset));
offset = i = line.find_first_not_of(' ', i);
offset = i = line.find_first_not_of(" \n\t\r", i);
forcetoken = false;
} else
i++;
@ -788,7 +791,7 @@ void configureGenerator(const char* text, const char* line, std::vector<std::str
"resolvers=",
"perpetual_storage_wiggle=",
"perpetual_storage_wiggle_locality=",
"storage_migration_type="
"storage_migration_type=",
"tenant_mode=",
"blob_granules_enabled=",
nullptr };

View File

@ -165,6 +165,7 @@ public:
Key backupTag,
Standalone<VectorRef<KeyRangeRef>> backupRanges,
Key bcUrl,
Optional<std::string> proxy,
Version targetVersion,
LockDB lockDB,
UID randomUID,
@ -187,6 +188,7 @@ public:
Optional<Database> cxOrig,
Key tagName,
Key url,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> ranges,
WaitForComplete = WaitForComplete::True,
Version targetVersion = ::invalidVersion,
@ -202,6 +204,7 @@ public:
Optional<Database> cxOrig,
Key tagName,
Key url,
Optional<std::string> proxy,
WaitForComplete waitForComplete = WaitForComplete::True,
Version targetVersion = ::invalidVersion,
Verbose verbose = Verbose::True,
@ -219,6 +222,7 @@ public:
cxOrig,
tagName,
url,
proxy,
rangeRef,
waitForComplete,
targetVersion,
@ -263,6 +267,7 @@ public:
Future<Void> submitBackup(Reference<ReadYourWritesTransaction> tr,
Key outContainer,
Optional<std::string> proxy,
int initialSnapshotIntervalSeconds,
int snapshotIntervalSeconds,
std::string const& tagName,
@ -273,6 +278,7 @@ public:
Optional<std::string> const& encryptionKeyFileName = {});
Future<Void> submitBackup(Database cx,
Key outContainer,
Optional<std::string> proxy,
int initialSnapshotIntervalSeconds,
int snapshotIntervalSeconds,
std::string const& tagName,
@ -284,6 +290,7 @@ public:
return runRYWTransactionFailIfLocked(cx, [=](Reference<ReadYourWritesTransaction> tr) {
return submitBackup(tr,
outContainer,
proxy,
initialSnapshotIntervalSeconds,
snapshotIntervalSeconds,
tagName,
@ -720,20 +727,37 @@ template <>
inline Tuple Codec<Reference<IBackupContainer>>::pack(Reference<IBackupContainer> const& bc) {
Tuple tuple;
tuple.append(StringRef(bc->getURL()));
if (bc->getEncryptionKeyFileName().present()) {
tuple.append(bc->getEncryptionKeyFileName().get());
} else {
tuple.append(StringRef());
}
if (bc->getProxy().present()) {
tuple.append(StringRef(bc->getProxy().get()));
} else {
tuple.append(StringRef());
}
return tuple;
}
template <>
inline Reference<IBackupContainer> Codec<Reference<IBackupContainer>>::unpack(Tuple const& val) {
ASSERT(val.size() == 1 || val.size() == 2);
ASSERT(val.size() >= 1);
auto url = val.getString(0).toString();
Optional<std::string> encryptionKeyFileName;
if (val.size() == 2) {
if (val.size() > 1 && !val.getString(1).empty()) {
encryptionKeyFileName = val.getString(1).toString();
}
return IBackupContainer::openContainer(url, encryptionKeyFileName);
Optional<std::string> proxy;
if (val.size() > 2 && !val.getString(2).empty()) {
proxy = val.getString(2).toString();
}
return IBackupContainer::openContainer(url, proxy, encryptionKeyFileName);
}
class BackupConfig : public KeyBackedConfig {

View File

@ -256,7 +256,8 @@ std::vector<std::string> IBackupContainer::getURLFormats() {
// Get an IBackupContainer based on a container URL string
Reference<IBackupContainer> IBackupContainer::openContainer(const std::string& url,
Optional<std::string> const& encryptionKeyFileName) {
const Optional<std::string>& proxy,
const Optional<std::string>& encryptionKeyFileName) {
static std::map<std::string, Reference<IBackupContainer>> m_cache;
Reference<IBackupContainer>& r = m_cache[url];
@ -273,7 +274,7 @@ Reference<IBackupContainer> IBackupContainer::openContainer(const std::string& u
// The URL parameters contain blobstore endpoint tunables as well as possible backup-specific options.
S3BlobStoreEndpoint::ParametersT backupParams;
Reference<S3BlobStoreEndpoint> bstore =
S3BlobStoreEndpoint::fromString(url, &resource, &lastOpenError, &backupParams);
S3BlobStoreEndpoint::fromString(url, proxy, &resource, &lastOpenError, &backupParams);
if (resource.empty())
throw backup_invalid_url();
@ -317,7 +318,7 @@ Reference<IBackupContainer> IBackupContainer::openContainer(const std::string& u
// Get a list of URLS to backup containers based on some a shorter URL. This function knows about some set of supported
// URL types which support this sort of backup discovery.
ACTOR Future<std::vector<std::string>> listContainers_impl(std::string baseURL) {
ACTOR Future<std::vector<std::string>> listContainers_impl(std::string baseURL, Optional<std::string> proxy) {
try {
StringRef u(baseURL);
if (u.startsWith("file://"_sr)) {
@ -327,8 +328,8 @@ ACTOR Future<std::vector<std::string>> listContainers_impl(std::string baseURL)
std::string resource;
S3BlobStoreEndpoint::ParametersT backupParams;
Reference<S3BlobStoreEndpoint> bstore =
S3BlobStoreEndpoint::fromString(baseURL, &resource, &IBackupContainer::lastOpenError, &backupParams);
Reference<S3BlobStoreEndpoint> bstore = S3BlobStoreEndpoint::fromString(
baseURL, proxy, &resource, &IBackupContainer::lastOpenError, &backupParams);
if (!resource.empty()) {
TraceEvent(SevWarn, "BackupContainer")
@ -370,8 +371,9 @@ ACTOR Future<std::vector<std::string>> listContainers_impl(std::string baseURL)
}
}
Future<std::vector<std::string>> IBackupContainer::listContainers(const std::string& baseURL) {
return listContainers_impl(baseURL);
Future<std::vector<std::string>> IBackupContainer::listContainers(const std::string& baseURL,
const Optional<std::string>& proxy) {
return listContainers_impl(baseURL, proxy);
}
ACTOR Future<Version> timeKeeperVersionFromDatetime(std::string datetime, Database db) {

View File

@ -156,6 +156,7 @@ struct BackupFileList {
struct BackupDescription {
BackupDescription() : snapshotBytes(0) {}
std::string url;
Optional<std::string> proxy;
std::vector<KeyspaceSnapshotFile> snapshots;
int64_t snapshotBytes;
// The version before which everything has been deleted by an expire
@ -294,11 +295,14 @@ public:
// Get an IBackupContainer based on a container spec string
static Reference<IBackupContainer> openContainer(const std::string& url,
const Optional<std::string>& encryptionKeyFileName = {});
const Optional<std::string>& proxy,
const Optional<std::string>& encryptionKeyFileName);
static std::vector<std::string> getURLFormats();
static Future<std::vector<std::string>> listContainers(const std::string& baseURL);
static Future<std::vector<std::string>> listContainers(const std::string& baseURL,
const Optional<std::string>& proxy);
std::string const& getURL() const { return URL; }
Optional<std::string> const& getProxy() const { return proxy; }
Optional<std::string> const& getEncryptionKeyFileName() const { return encryptionKeyFileName; }
static std::string lastOpenError;
@ -306,6 +310,7 @@ public:
// TODO: change the following back to `private` once blob obj access is refactored
protected:
std::string URL;
Optional<std::string> proxy;
Optional<std::string> encryptionKeyFileName;
};

View File

@ -409,6 +409,7 @@ public:
Version logStartVersionOverride) {
state BackupDescription desc;
desc.url = bc->getURL();
desc.proxy = bc->getProxy();
TraceEvent("BackupContainerDescribe1")
.detail("URL", bc->getURL())
@ -1500,7 +1501,8 @@ Future<Void> BackupContainerFileSystem::createTestEncryptionKeyFile(std::string
// code but returning a different template type because you can't cast between them
Reference<BackupContainerFileSystem> BackupContainerFileSystem::openContainerFS(
const std::string& url,
Optional<std::string> const& encryptionKeyFileName) {
const Optional<std::string>& proxy,
const Optional<std::string>& encryptionKeyFileName) {
static std::map<std::string, Reference<BackupContainerFileSystem>> m_cache;
Reference<BackupContainerFileSystem>& r = m_cache[url];
@ -1517,7 +1519,7 @@ Reference<BackupContainerFileSystem> BackupContainerFileSystem::openContainerFS(
// The URL parameters contain blobstore endpoint tunables as well as possible backup-specific options.
S3BlobStoreEndpoint::ParametersT backupParams;
Reference<S3BlobStoreEndpoint> bstore =
S3BlobStoreEndpoint::fromString(url, &resource, &lastOpenError, &backupParams);
S3BlobStoreEndpoint::fromString(url, proxy, &resource, &lastOpenError, &backupParams);
if (resource.empty())
throw backup_invalid_url();
@ -1635,7 +1637,9 @@ ACTOR static Future<Void> testWriteSnapshotFile(Reference<IBackupFile> file, Key
return Void();
}
ACTOR Future<Void> testBackupContainer(std::string url, Optional<std::string> encryptionKeyFileName) {
ACTOR Future<Void> testBackupContainer(std::string url,
Optional<std::string> proxy,
Optional<std::string> encryptionKeyFileName) {
state FlowLock lock(100e6);
if (encryptionKeyFileName.present()) {
@ -1644,7 +1648,7 @@ ACTOR Future<Void> testBackupContainer(std::string url, Optional<std::string> en
printf("BackupContainerTest URL %s\n", url.c_str());
state Reference<IBackupContainer> c = IBackupContainer::openContainer(url, encryptionKeyFileName);
state Reference<IBackupContainer> c = IBackupContainer::openContainer(url, proxy, encryptionKeyFileName);
// Make sure container doesn't exist, then create it.
try {
@ -1789,12 +1793,13 @@ ACTOR Future<Void> testBackupContainer(std::string url, Optional<std::string> en
}
TEST_CASE("/backup/containers/localdir/unencrypted") {
wait(testBackupContainer(format("file://%s/fdb_backups/%llx", params.getDataDir().c_str(), timer_int()), {}));
wait(testBackupContainer(format("file://%s/fdb_backups/%llx", params.getDataDir().c_str(), timer_int()), {}, {}));
return Void();
}
TEST_CASE("/backup/containers/localdir/encrypted") {
wait(testBackupContainer(format("file://%s/fdb_backups/%llx", params.getDataDir().c_str(), timer_int()),
{},
format("%s/test_encryption_key", params.getDataDir().c_str())));
return Void();
}
@ -1803,7 +1808,7 @@ TEST_CASE("/backup/containers/url") {
if (!g_network->isSimulated()) {
const char* url = getenv("FDB_TEST_BACKUP_URL");
ASSERT(url != nullptr);
wait(testBackupContainer(url, {}));
wait(testBackupContainer(url, {}, {}));
}
return Void();
}
@ -1813,7 +1818,7 @@ TEST_CASE("/backup/containers_list") {
state const char* url = getenv("FDB_TEST_BACKUP_URL");
ASSERT(url != nullptr);
printf("Listing %s\n", url);
std::vector<std::string> urls = wait(IBackupContainer::listContainers(url));
std::vector<std::string> urls = wait(IBackupContainer::listContainers(url, {}));
for (auto& u : urls) {
printf("%s\n", u.c_str());
}

View File

@ -81,9 +81,9 @@ public:
Future<bool> exists() override = 0;
// TODO: refactor this to separate out the "deal with blob store" stuff from the backup business logic
static Reference<BackupContainerFileSystem> openContainerFS(
const std::string& url,
const Optional<std::string>& encryptionKeyFileName = {});
static Reference<BackupContainerFileSystem> openContainerFS(const std::string& url,
const Optional<std::string>& proxy,
const Optional<std::string>& encryptionKeyFileName);
// Get a list of fileNames and their sizes in the container under the given path
// Although not required, an implementation can avoid traversing unwanted subfolders

View File

@ -52,19 +52,20 @@ struct BlobFilePointerRef {
StringRef filename;
int64_t offset;
int64_t length;
int64_t fullFileLength;
BlobFilePointerRef() {}
BlobFilePointerRef(Arena& to, const std::string& filename, int64_t offset, int64_t length)
: filename(to, filename), offset(offset), length(length) {}
BlobFilePointerRef(Arena& to, const std::string& filename, int64_t offset, int64_t length, int64_t fullFileLength)
: filename(to, filename), offset(offset), length(length), fullFileLength(fullFileLength) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, filename, offset, length);
serializer(ar, filename, offset, length, fullFileLength);
}
std::string toString() const {
std::stringstream ss;
ss << filename.toString() << ":" << offset << ":" << length;
ss << filename.toString() << ":" << offset << ":" << length << ":" << fullFileLength;
return std::move(ss).str();
}
};

View File

@ -240,22 +240,27 @@ static void startLoad(const ReadBlobGranuleContext granuleContext,
// Start load process for all files in chunk
if (chunk.snapshotFile.present()) {
std::string snapshotFname = chunk.snapshotFile.get().filename.toString();
// FIXME: full file length won't always be length of read
// FIXME: remove when we implement file multiplexing
ASSERT(chunk.snapshotFile.get().offset == 0);
ASSERT(chunk.snapshotFile.get().length == chunk.snapshotFile.get().fullFileLength);
loadIds.snapshotId = granuleContext.start_load_f(snapshotFname.c_str(),
snapshotFname.size(),
chunk.snapshotFile.get().offset,
chunk.snapshotFile.get().length,
chunk.snapshotFile.get().length,
chunk.snapshotFile.get().fullFileLength,
granuleContext.userContext);
}
loadIds.deltaIds.reserve(chunk.deltaFiles.size());
for (int deltaFileIdx = 0; deltaFileIdx < chunk.deltaFiles.size(); deltaFileIdx++) {
std::string deltaFName = chunk.deltaFiles[deltaFileIdx].filename.toString();
// FIXME: remove when we implement file multiplexing
ASSERT(chunk.deltaFiles[deltaFileIdx].offset == 0);
ASSERT(chunk.deltaFiles[deltaFileIdx].length == chunk.deltaFiles[deltaFileIdx].fullFileLength);
int64_t deltaLoadId = granuleContext.start_load_f(deltaFName.c_str(),
deltaFName.size(),
chunk.deltaFiles[deltaFileIdx].offset,
chunk.deltaFiles[deltaFileIdx].length,
chunk.deltaFiles[deltaFileIdx].length,
chunk.deltaFiles[deltaFileIdx].fullFileLength,
granuleContext.userContext);
loadIds.deltaIds.push_back(deltaLoadId);
}

View File

@ -70,7 +70,7 @@ void ClientKnobs::initialize(Randomize randomize) {
init( RESOURCE_CONSTRAINED_MAX_BACKOFF, 30.0 );
init( PROXY_COMMIT_OVERHEAD_BYTES, 23 ); //The size of serializing 7 tags (3 primary, 3 remote, 1 log router) + 2 for the tag length
init( SHARD_STAT_SMOOTH_AMOUNT, 5.0 );
init( INIT_MID_SHARD_BYTES, 50000000 ); if( randomize && BUGGIFY ) INIT_MID_SHARD_BYTES = 40000; else if(randomize && !BUGGIFY) INIT_MID_SHARD_BYTES = 200000; // The same value as SERVER_KNOBS->MIN_SHARD_BYTES
init( INIT_MID_SHARD_BYTES, 50000000 ); if( randomize && BUGGIFY ) INIT_MID_SHARD_BYTES = 40000; else if(randomize && BUGGIFY_WITH_PROB(0.75)) INIT_MID_SHARD_BYTES = 200000; // The same value as SERVER_KNOBS->MIN_SHARD_BYTES
init( TRANSACTION_SIZE_LIMIT, 1e7 );
init( KEY_SIZE_LIMIT, 1e4 );

View File

@ -302,7 +302,7 @@ StatusObject DatabaseConfiguration::toJSON(bool noPolicies) const {
result["storage_engine"] = "ssd-redwood-1-experimental";
} else if (tLogDataStoreType == KeyValueStoreType::SSD_BTREE_V2 &&
storageServerStoreType == KeyValueStoreType::SSD_ROCKSDB_V1) {
result["storage_engine"] = "ssd-rocksdb-experimental";
result["storage_engine"] = "ssd-rocksdb-v1";
} else if (tLogDataStoreType == KeyValueStoreType::MEMORY && storageServerStoreType == KeyValueStoreType::MEMORY) {
result["storage_engine"] = "memory-1";
} else if (tLogDataStoreType == KeyValueStoreType::SSD_BTREE_V2 &&
@ -324,7 +324,7 @@ StatusObject DatabaseConfiguration::toJSON(bool noPolicies) const {
} else if (testingStorageServerStoreType == KeyValueStoreType::SSD_REDWOOD_V1) {
result["tss_storage_engine"] = "ssd-redwood-1-experimental";
} else if (testingStorageServerStoreType == KeyValueStoreType::SSD_ROCKSDB_V1) {
result["tss_storage_engine"] = "ssd-rocksdb-experimental";
result["tss_storage_engine"] = "ssd-rocksdb-v1";
} else if (testingStorageServerStoreType == KeyValueStoreType::MEMORY_RADIXTREE) {
result["tss_storage_engine"] = "memory-radixtree-beta";
} else if (testingStorageServerStoreType == KeyValueStoreType::MEMORY) {

View File

@ -652,6 +652,7 @@ struct GetRangeLimits {
};
struct RangeResultRef : VectorRef<KeyValueRef> {
constexpr static FileIdentifier file_identifier = 3985192;
bool more; // True if (but not necessarily only if) values remain in the *key* range requested (possibly beyond the
// limits requested) False implies that no such values remain
Optional<KeyRef> readThrough; // Only present when 'more' is true. When present, this value represent the end (or
@ -831,7 +832,7 @@ struct KeyValueStoreType {
case SSD_REDWOOD_V1:
return "ssd-redwood-1-experimental";
case SSD_ROCKSDB_V1:
return "ssd-rocksdb-experimental";
return "ssd-rocksdb-v1";
case MEMORY:
return "memory";
case MEMORY_RADIXTREE:
@ -958,6 +959,7 @@ struct TLogSpillType {
// Contains the amount of free and total space for a storage server, in bytes
struct StorageBytes {
constexpr static FileIdentifier file_identifier = 3928581;
// Free space on the filesystem
int64_t free;
// Total space on the filesystem
@ -1366,12 +1368,12 @@ struct ReadBlobGranuleContext {
// Store metadata associated with each storage server. Now it only contains data be used in perpetual storage wiggle.
struct StorageMetadataType {
constexpr static FileIdentifier file_identifier = 732123;
// when the SS is initialized
uint64_t createdTime; // comes from currentTime()
// when the SS is initialized, in epoch seconds, comes from currentTime()
double createdTime;
StorageMetadataType() : createdTime(0) {}
StorageMetadataType(uint64_t t) : createdTime(t) {}
static uint64_t currentTime() { return g_network->timer() * 1e9; }
static double currentTime() { return g_network->timer(); }
// To change this serialization, ProtocolVersion::StorageMetadata must be updated, and downgrades need
// to be considered

View File

@ -4363,13 +4363,14 @@ public:
Key backupTag,
Standalone<VectorRef<KeyRangeRef>> backupRanges,
Key bcUrl,
Optional<std::string> proxy,
Version targetVersion,
LockDB lockDB,
UID randomUID,
Key addPrefix,
Key removePrefix) {
// Sanity check backup is valid
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(bcUrl.toString());
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(bcUrl.toString(), proxy, {});
state BackupDescription desc = wait(bc->describeBackup());
wait(desc.resolveVersionTimes(cx));
@ -4430,6 +4431,7 @@ public:
struct RestoreRequest restoreRequest(restoreIndex,
restoreTag,
bcUrl,
proxy,
targetVersion,
range,
deterministicRandom()->randomUniqueID(),
@ -4510,6 +4512,7 @@ public:
ACTOR static Future<Void> submitBackup(FileBackupAgent* backupAgent,
Reference<ReadYourWritesTransaction> tr,
Key outContainer,
Optional<std::string> proxy,
int initialSnapshotIntervalSeconds,
int snapshotIntervalSeconds,
std::string tagName,
@ -4555,7 +4558,8 @@ public:
backupContainer = joinPath(backupContainer, std::string("backup-") + nowStr.toString());
}
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(backupContainer, encryptionKeyFileName);
state Reference<IBackupContainer> bc =
IBackupContainer::openContainer(backupContainer, proxy, encryptionKeyFileName);
try {
wait(timeoutError(bc->create(), 30));
} catch (Error& e) {
@ -4642,6 +4646,7 @@ public:
Reference<ReadYourWritesTransaction> tr,
Key tagName,
Key backupURL,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> ranges,
Version restoreVersion,
Key addPrefix,
@ -4710,7 +4715,7 @@ public:
// Point the tag to the new uid
tag.set(tr, { uid, false });
Reference<IBackupContainer> bc = IBackupContainer::openContainer(backupURL.toString());
Reference<IBackupContainer> bc = IBackupContainer::openContainer(backupURL.toString(), proxy, {});
// Configure the new restore
restore.tag().set(tr, tagName.toString());
@ -5303,6 +5308,7 @@ public:
Optional<Database> cxOrig,
Key tagName,
Key url,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> ranges,
WaitForComplete waitForComplete,
Version targetVersion,
@ -5320,7 +5326,7 @@ public:
throw restore_error();
}
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(url.toString());
state Reference<IBackupContainer> bc = IBackupContainer::openContainer(url.toString(), proxy, {});
state BackupDescription desc = wait(bc->describeBackup(true));
if (cxOrig.present()) {
@ -5360,6 +5366,7 @@ public:
tr,
tagName,
url,
proxy,
ranges,
targetVersion,
addPrefix,
@ -5499,6 +5506,7 @@ public:
tagName,
ranges,
KeyRef(bc->getURL()),
bc->getProxy(),
targetVersion,
LockDB::True,
randomUid,
@ -5520,6 +5528,7 @@ public:
cx,
tagName,
KeyRef(bc->getURL()),
bc->getProxy(),
ranges,
WaitForComplete::True,
::invalidVersion,
@ -5561,13 +5570,14 @@ Future<Void> FileBackupAgent::submitParallelRestore(Database cx,
Key backupTag,
Standalone<VectorRef<KeyRangeRef>> backupRanges,
Key bcUrl,
Optional<std::string> proxy,
Version targetVersion,
LockDB lockDB,
UID randomUID,
Key addPrefix,
Key removePrefix) {
return FileBackupAgentImpl::submitParallelRestore(
cx, backupTag, backupRanges, bcUrl, targetVersion, lockDB, randomUID, addPrefix, removePrefix);
cx, backupTag, backupRanges, bcUrl, proxy, targetVersion, lockDB, randomUID, addPrefix, removePrefix);
}
Future<Void> FileBackupAgent::atomicParallelRestore(Database cx,
@ -5582,6 +5592,7 @@ Future<Version> FileBackupAgent::restore(Database cx,
Optional<Database> cxOrig,
Key tagName,
Key url,
Optional<std::string> proxy,
Standalone<VectorRef<KeyRangeRef>> ranges,
WaitForComplete waitForComplete,
Version targetVersion,
@ -5598,6 +5609,7 @@ Future<Version> FileBackupAgent::restore(Database cx,
cxOrig,
tagName,
url,
proxy,
ranges,
waitForComplete,
targetVersion,
@ -5639,6 +5651,7 @@ Future<ERestoreState> FileBackupAgent::waitRestore(Database cx, Key tagName, Ver
Future<Void> FileBackupAgent::submitBackup(Reference<ReadYourWritesTransaction> tr,
Key outContainer,
Optional<std::string> proxy,
int initialSnapshotIntervalSeconds,
int snapshotIntervalSeconds,
std::string const& tagName,
@ -5650,6 +5663,7 @@ Future<Void> FileBackupAgent::submitBackup(Reference<ReadYourWritesTransaction>
return FileBackupAgentImpl::submitBackup(this,
tr,
outContainer,
proxy,
initialSnapshotIntervalSeconds,
snapshotIntervalSeconds,
tagName,

View File

@ -214,7 +214,7 @@ std::map<std::string, std::string> configForToken(std::string const& mode) {
} else if (mode == "ssd-redwood-1-experimental") {
logType = KeyValueStoreType::SSD_BTREE_V2;
storeType = KeyValueStoreType::SSD_REDWOOD_V1;
} else if (mode == "ssd-rocksdb-experimental") {
} else if (mode == "ssd-rocksdb-v1") {
logType = KeyValueStoreType::SSD_BTREE_V2;
storeType = KeyValueStoreType::SSD_ROCKSDB_V1;
} else if (mode == "memory" || mode == "memory-2") {

View File

@ -780,7 +780,7 @@ void MultiVersionTransaction::updateTransaction() {
TransactionInfo newTr;
if (tenant.present()) {
ASSERT(tenant.get());
auto currentTenant = tenant.get()->tenantVar->get();
auto currentTenant = tenant.get()->tenantState->tenantVar->get();
if (currentTenant.value) {
newTr.transaction = currentTenant.value->createTransaction();
}
@ -1080,7 +1080,7 @@ ThreadFuture<Void> MultiVersionTransaction::onError(Error const& e) {
Optional<TenantName> MultiVersionTransaction::getTenant() {
if (tenant.present()) {
return tenant.get()->tenantName;
return tenant.get()->tenantState->tenantName;
} else {
return Optional<TenantName>();
}
@ -1214,20 +1214,27 @@ bool MultiVersionTransaction::isValid() {
// MultiVersionTenant
MultiVersionTenant::MultiVersionTenant(Reference<MultiVersionDatabase> db, StringRef tenantName)
: tenantVar(new ThreadSafeAsyncVar<Reference<ITenant>>(Reference<ITenant>(nullptr))), tenantName(tenantName), db(db) {
updateTenant();
: tenantState(makeReference<TenantState>(db, tenantName)) {}
MultiVersionTenant::~MultiVersionTenant() {
tenantState->close();
}
MultiVersionTenant::~MultiVersionTenant() {}
Reference<ITransaction> MultiVersionTenant::createTransaction() {
return Reference<ITransaction>(new MultiVersionTransaction(
db, Reference<MultiVersionTenant>::addRef(this), db->dbState->transactionDefaultOptions));
return Reference<ITransaction>(new MultiVersionTransaction(tenantState->db,
Reference<MultiVersionTenant>::addRef(this),
tenantState->db->dbState->transactionDefaultOptions));
}
MultiVersionTenant::TenantState::TenantState(Reference<MultiVersionDatabase> db, StringRef tenantName)
: tenantVar(new ThreadSafeAsyncVar<Reference<ITenant>>(Reference<ITenant>(nullptr))), tenantName(tenantName), db(db),
closed(false) {
updateTenant();
}
// Creates a new underlying tenant object whenever the database connection changes. This change is signaled
// to open transactions via an AsyncVar.
void MultiVersionTenant::updateTenant() {
void MultiVersionTenant::TenantState::updateTenant() {
Reference<ITenant> tenant;
auto currentDb = db->dbState->dbVar->get();
if (currentDb.value) {
@ -1238,13 +1245,27 @@ void MultiVersionTenant::updateTenant() {
tenantVar->set(tenant);
Reference<TenantState> self = Reference<TenantState>::addRef(this);
MutexHolder holder(tenantLock);
tenantUpdater = mapThreadFuture<Void, Void>(currentDb.onChange, [this](ErrorOr<Void> result) {
updateTenant();
if (closed) {
return;
}
tenantUpdater = mapThreadFuture<Void, Void>(currentDb.onChange, [self](ErrorOr<Void> result) {
self->updateTenant();
return Void();
});
}
void MultiVersionTenant::TenantState::close() {
MutexHolder holder(tenantLock);
closed = true;
if (tenantUpdater.isValid()) {
tenantUpdater.cancel();
}
}
// MultiVersionDatabase
MultiVersionDatabase::MultiVersionDatabase(MultiVersionApi* api,
int threadIdx,

View File

@ -646,18 +646,30 @@ public:
void addref() override { ThreadSafeReferenceCounted<MultiVersionTenant>::addref(); }
void delref() override { ThreadSafeReferenceCounted<MultiVersionTenant>::delref(); }
Reference<ThreadSafeAsyncVar<Reference<ITenant>>> tenantVar;
const Standalone<StringRef> tenantName;
// A struct that manages the current connection state of the MultiVersionDatabase. This wraps the underlying
// IDatabase object that is currently interacting with the cluster.
struct TenantState : ThreadSafeReferenceCounted<TenantState> {
TenantState(Reference<MultiVersionDatabase> db, StringRef tenantName);
private:
Reference<MultiVersionDatabase> db;
// Creates a new underlying tenant object whenever the database connection changes. This change is signaled
// to open transactions via an AsyncVar.
void updateTenant();
Mutex tenantLock;
ThreadFuture<Void> tenantUpdater;
// Cleans up local state to break reference cycles
void close();
// Creates a new underlying tenant object whenever the database connection changes. This change is signaled
// to open transactions via an AsyncVar.
void updateTenant();
Reference<ThreadSafeAsyncVar<Reference<ITenant>>> tenantVar;
const Standalone<StringRef> tenantName;
Reference<MultiVersionDatabase> db;
Mutex tenantLock;
ThreadFuture<Void> tenantUpdater;
bool closed;
};
Reference<TenantState> tenantState;
};
// An implementation of IDatabase that wraps a database created either locally or through a dynamically loaded

View File

@ -7372,6 +7372,7 @@ ACTOR Future<Standalone<VectorRef<BlobGranuleChunkRef>>> readBlobGranulesActor(
fmt::print(
"BG Mapping for [{0} - %{1}) too large!\n", keyRange.begin.printable(), keyRange.end.printable());
}
TraceEvent(SevWarn, "BGMappingTooLarge").detail("Range", range).detail("Max", 1000);
throw unsupported_operation();
}
ASSERT(!blobGranuleMapping.more && blobGranuleMapping.size() < CLIENT_KNOBS->TOO_MANY);

View File

@ -49,6 +49,7 @@ struct RestoreRequest {
int index;
Key tagName;
Key url;
Optional<std::string> proxy;
Version targetVersion;
KeyRange range;
UID randomUid;
@ -64,27 +65,29 @@ struct RestoreRequest {
explicit RestoreRequest(const int index,
const Key& tagName,
const Key& url,
const Optional<std::string>& proxy,
Version targetVersion,
const KeyRange& range,
const UID& randomUid,
Key& addPrefix,
Key removePrefix)
: index(index), tagName(tagName), url(url), targetVersion(targetVersion), range(range), randomUid(randomUid),
addPrefix(addPrefix), removePrefix(removePrefix) {}
: index(index), tagName(tagName), url(url), proxy(proxy), targetVersion(targetVersion), range(range),
randomUid(randomUid), addPrefix(addPrefix), removePrefix(removePrefix) {}
// To change this serialization, ProtocolVersion::RestoreRequestValue must be updated, and downgrades need to be
// considered
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, index, tagName, url, targetVersion, range, randomUid, addPrefix, removePrefix, reply);
serializer(ar, index, tagName, url, proxy, targetVersion, range, randomUid, addPrefix, removePrefix, reply);
}
std::string toString() const {
std::stringstream ss;
ss << "index:" << std::to_string(index) << " tagName:" << tagName.contents().toString()
<< " url:" << url.contents().toString() << " targetVersion:" << std::to_string(targetVersion)
<< " range:" << range.toString() << " randomUid:" << randomUid.toString()
<< " addPrefix:" << addPrefix.toString() << " removePrefix:" << removePrefix.toString();
<< " url:" << url.contents().toString() << " proxy:" << (proxy.present() ? proxy.get() : "")
<< " targetVersion:" << std::to_string(targetVersion) << " range:" << range.toString()
<< " randomUid:" << randomUid.toString() << " addPrefix:" << addPrefix.toString()
<< " removePrefix:" << removePrefix.toString();
return ss.str();
}
};

View File

@ -162,7 +162,8 @@ std::string S3BlobStoreEndpoint::BlobKnobs::getURLParameters() const {
return r;
}
Reference<S3BlobStoreEndpoint> S3BlobStoreEndpoint::fromString(std::string const& url,
Reference<S3BlobStoreEndpoint> S3BlobStoreEndpoint::fromString(const std::string& url,
const Optional<std::string>& proxy,
std::string* resourceFromURL,
std::string* error,
ParametersT* ignored_parameters) {
@ -175,6 +176,17 @@ Reference<S3BlobStoreEndpoint> S3BlobStoreEndpoint::fromString(std::string const
if (prefix != LiteralStringRef("blobstore"))
throw format("Invalid blobstore URL prefix '%s'", prefix.toString().c_str());
Optional<std::string> proxyHost, proxyPort;
if (proxy.present()) {
if (!Hostname::isHostname(proxy.get()) && !NetworkAddress::parseOptional(proxy.get()).present()) {
throw format("'%s' is not a valid value for proxy. Format should be either IP:port or host:port.",
proxy.get().c_str());
}
StringRef p(proxy.get());
proxyHost = p.eat(":").toString();
proxyPort = p.eat().toString();
}
Optional<StringRef> cred;
if (url.find("@") != std::string::npos) {
cred = t.eat("@");
@ -261,7 +273,8 @@ Reference<S3BlobStoreEndpoint> S3BlobStoreEndpoint::fromString(std::string const
creds = S3BlobStoreEndpoint::Credentials{ key.toString(), secret.toString(), securityToken.toString() };
}
return makeReference<S3BlobStoreEndpoint>(host.toString(), service.toString(), creds, knobs, extraHeaders);
return makeReference<S3BlobStoreEndpoint>(
host.toString(), service.toString(), proxyHost, proxyPort, creds, knobs, extraHeaders);
} catch (std::string& err) {
if (error != nullptr)
@ -624,11 +637,11 @@ ACTOR Future<S3BlobStoreEndpoint::ReusableConnection> connect_impl(Reference<S3B
return rconn;
}
}
std::string service = b->service;
std::string host = b->host, service = b->service;
if (service.empty())
service = b->knobs.secure_connection ? "https" : "http";
state Reference<IConnection> conn =
wait(INetworkConnections::net()->connect(b->host, service, b->knobs.secure_connection ? true : false));
wait(INetworkConnections::net()->connect(host, service, b->knobs.secure_connection ? true : false));
wait(conn->connectHandshake());
TraceEvent("S3BlobStoreEndpointNewConnection")
@ -1609,7 +1622,7 @@ TEST_CASE("/backup/s3/v4headers") {
S3BlobStoreEndpoint::Credentials creds{ "AKIAIOSFODNN7EXAMPLE", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", "" }
// GET without query parameters
{
S3BlobStoreEndpoint s3("s3.amazonaws.com", "s3", creds);
S3BlobStoreEndpoint s3("s3.amazonaws.com", "s3", "proxy", "port", creds);
std::string verb("GET");
std::string resource("/test.txt");
HTTP::Headers headers;
@ -1624,7 +1637,7 @@ TEST_CASE("/backup/s3/v4headers") {
// GET with query parameters
{
S3BlobStoreEndpoint s3("s3.amazonaws.com", "s3", creds);
S3BlobStoreEndpoint s3("s3.amazonaws.com", "s3", "proxy", "port", creds);
std::string verb("GET");
std::string resource("/test/examplebucket?Action=DescribeRegions&Version=2013-10-15");
HTTP::Headers headers;
@ -1639,7 +1652,7 @@ TEST_CASE("/backup/s3/v4headers") {
// POST
{
S3BlobStoreEndpoint s3("s3.us-west-2.amazonaws.com", "s3", creds);
S3BlobStoreEndpoint s3("s3.us-west-2.amazonaws.com", "s3", "proxy", "port", creds);
std::string verb("POST");
std::string resource("/simple.json");
HTTP::Headers headers;

View File

@ -99,11 +99,15 @@ public:
};
S3BlobStoreEndpoint(std::string const& host,
std::string service,
std::string const& service,
Optional<std::string> const& proxyHost,
Optional<std::string> const& proxyPort,
Optional<Credentials> const& creds,
BlobKnobs const& knobs = BlobKnobs(),
HTTP::Headers extraHeaders = HTTP::Headers())
: host(host), service(service), credentials(creds), lookupKey(creds.present() && creds.get().key.empty()),
: host(host), service(service), proxyHost(proxyHost), proxyPort(proxyPort),
useProxy(proxyHost.present() && proxyPort.present()), credentials(creds),
lookupKey(creds.present() && creds.get().key.empty()),
lookupSecret(creds.present() && creds.get().secret.empty()), knobs(knobs), extraHeaders(extraHeaders),
requestRate(new SpeedLimit(knobs.requests_per_second, 1)),
requestRateList(new SpeedLimit(knobs.list_requests_per_second, 1)),
@ -114,7 +118,7 @@ public:
recvRate(new SpeedLimit(knobs.max_recv_bytes_per_second, 1)), concurrentRequests(knobs.concurrent_requests),
concurrentUploads(knobs.concurrent_uploads), concurrentLists(knobs.concurrent_lists) {
if (host.empty())
if (host.empty() || (proxyHost.present() != proxyPort.present()))
throw connection_string_invalid();
}
@ -132,10 +136,11 @@ public:
// Parse url and return a S3BlobStoreEndpoint
// If the url has parameters that S3BlobStoreEndpoint can't consume then an error will be thrown unless
// ignored_parameters is given in which case the unconsumed parameters will be added to it.
static Reference<S3BlobStoreEndpoint> fromString(std::string const& url,
std::string* resourceFromURL = nullptr,
std::string* error = nullptr,
ParametersT* ignored_parameters = nullptr);
static Reference<S3BlobStoreEndpoint> fromString(const std::string& url,
const Optional<std::string>& proxy,
std::string* resourceFromURL,
std::string* error,
ParametersT* ignored_parameters);
// Get a normalized version of this URL with the given resource and any non-default BlobKnob values as URL
// parameters in addition to the passed params string
@ -151,6 +156,10 @@ public:
std::string host;
std::string service;
Optional<std::string> proxyHost;
Optional<std::string> proxyPort;
bool useProxy;
Optional<Credentials> credentials;
bool lookupKey;
bool lookupSecret;

View File

@ -24,37 +24,37 @@
const KeyRef JSONSchemas::statusSchema = LiteralStringRef(R"statusSchema(
{
"cluster":{
"storage_wiggler": {
"wiggle_server_ids":["0ccb4e0feddb55"],
"wiggle_server_addresses": ["127.0.0.1"],
"storage_wiggler": {
"wiggle_server_ids":["0ccb4e0feddb55"],
"wiggle_server_addresses": ["127.0.0.1"],
"primary": {
"last_round_start_datetime": "Wed Feb 4 09:36:37 2022 +0000",
"last_round_start_timestamp": 63811229797,
"last_round_finish_datetime": "Thu Jan 1 00:00:00 1970 +0000",
"last_round_finish_timestamp": 0,
"smoothed_round_seconds": 1,
"finished_round": 1,
"last_wiggle_start_datetime": "Wed Feb 4 09:36:37 2022 +0000",
"last_wiggle_start_timestamp": 63811229797,
"last_wiggle_finish_datetime": "Thu Jan 1 00:00:00 1970 +0000",
"last_wiggle_finish_timestamp": 0,
"smoothed_wiggle_seconds": 1,
"finished_wiggle": 1
},
"remote": {
"last_round_start_datetime": "Wed Feb 4 09:36:37 2022 +0000",
"last_round_start_timestamp": 63811229797,
"last_round_finish_datetime": "Thu Jan 1 00:00:00 1970 +0000",
"last_round_finish_timestamp": 0,
"smoothed_round_seconds": 1,
"finished_round": 1,
"last_wiggle_start_datetime": "Wed Feb 4 09:36:37 2022 +0000",
"last_wiggle_start_timestamp": 63811229797,
"last_wiggle_finish_datetime": "Thu Jan 1 00:00:00 1970 +0000",
"last_wiggle_finish_timestamp": 0,
"smoothed_wiggle_seconds": 1,
"finished_wiggle": 1
}
"last_round_start_datetime": "2022-04-02 00:05:05.123 +0000",
"last_round_start_timestamp": 1648857905.123,
"last_round_finish_datetime": "1970-01-01 00:00:00.000 +0000",
"last_round_finish_timestamp": 0,
"smoothed_round_seconds": 1,
"finished_round": 1,
"last_wiggle_start_datetime": "2022-04-02 00:05:05.123 +0000",
"last_wiggle_start_timestamp": 1648857905.123,
"last_wiggle_finish_datetime": "1970-01-01 00:00:00.000 +0000",
"last_wiggle_finish_timestamp": 0,
"smoothed_wiggle_seconds": 1,
"finished_wiggle": 1
},
"remote": {
"last_round_start_datetime": "2022-04-02 00:05:05.123 +0000",
"last_round_start_timestamp": 1648857905.123,
"last_round_finish_datetime": "1970-01-01 00:00:00.000 +0000",
"last_round_finish_timestamp": 0,
"smoothed_round_seconds": 1,
"finished_round": 1,
"last_wiggle_start_datetime": "2022-04-02 00:05:05.123 +0000",
"last_wiggle_start_timestamp": 1648857905.123,
"last_wiggle_finish_datetime": "1970-01-01 00:00:00.000 +0000",
"last_wiggle_finish_timestamp": 0,
"smoothed_wiggle_seconds": 1,
"finished_wiggle": 1
}
},
"layers":{
"_valid":true,
@ -136,7 +136,7 @@ const KeyRef JSONSchemas::statusSchema = LiteralStringRef(R"statusSchema(
]
},
"storage_metadata":{
"created_time_datetime":"Thu Jan 1 00:00:00 1970 +0000",
"created_time_datetime":"1970-01-01 00:00:00.000 +0000",
"created_time_timestamp": 0
},
"data_version":12341234,
@ -769,7 +769,7 @@ const KeyRef JSONSchemas::statusSchema = LiteralStringRef(R"statusSchema(
"ssd-1",
"ssd-2",
"ssd-redwood-1-experimental",
"ssd-rocksdb-experimental",
"ssd-rocksdb-v1",
"memory",
"memory-1",
"memory-2",
@ -782,7 +782,7 @@ const KeyRef JSONSchemas::statusSchema = LiteralStringRef(R"statusSchema(
"ssd-1",
"ssd-2",
"ssd-redwood-1-experimental",
"ssd-rocksdb-experimental",
"ssd-rocksdb-v1",
"memory",
"memory-1",
"memory-2",

View File

@ -152,7 +152,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( RETRY_RELOCATESHARD_DELAY, 0.1 );
init( DATA_DISTRIBUTION_FAILURE_REACTION_TIME, 60.0 ); if( randomize && BUGGIFY ) DATA_DISTRIBUTION_FAILURE_REACTION_TIME = 1.0;
bool buggifySmallShards = randomize && BUGGIFY;
bool simulationMediumShards = !buggifySmallShards && randomize && !BUGGIFY; // prefer smaller shards in simulation
bool simulationMediumShards = !buggifySmallShards && isSimulated && randomize && !BUGGIFY; // prefer smaller shards in simulation
init( MIN_SHARD_BYTES, 50000000 ); if( buggifySmallShards ) MIN_SHARD_BYTES = 40000; if (simulationMediumShards) MIN_SHARD_BYTES = 200000; //FIXME: data distribution tracker (specifically StorageMetrics) relies on this number being larger than the maximum size of a key value pair
init( SHARD_BYTES_RATIO, 4 );
init( SHARD_BYTES_PER_SQRT_BYTES, 45 ); if( buggifySmallShards ) SHARD_BYTES_PER_SQRT_BYTES = 0;//Approximately 10000 bytes per shard
@ -250,6 +250,9 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( DEBOUNCE_RECRUITING_DELAY, 5.0 );
init( DD_FAILURE_TIME, 1.0 ); if( randomize && BUGGIFY ) DD_FAILURE_TIME = 10.0;
init( DD_ZERO_HEALTHY_TEAM_DELAY, 1.0 );
init( REMOTE_KV_STORE, false ); if( randomize && BUGGIFY ) REMOTE_KV_STORE = true;
init( REMOTE_KV_STORE_INIT_DELAY, 0.1 );
init( REMOTE_KV_STORE_MAX_INIT_DURATION, 10.0 );
init( REBALANCE_MAX_RETRIES, 100 );
init( DD_OVERLAP_PENALTY, 10000 );
init( DD_EXCLUDE_MIN_REPLICAS, 1 );
@ -555,6 +558,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( MIN_REBOOT_TIME, 4.0 ); if( longReboots ) MIN_REBOOT_TIME = 10.0;
init( MAX_REBOOT_TIME, 5.0 ); if( longReboots ) MAX_REBOOT_TIME = 20.0;
init( LOG_DIRECTORY, "."); // Will be set to the command line flag.
init( CONN_FILE, ""); // Will be set to the command line flag.
init( SERVER_MEM_LIMIT, 8LL << 30 );
init( SYSTEM_MONITOR_FREQUENCY, 5.0 );
@ -653,6 +657,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( FETCH_KEYS_LOWER_PRIORITY, 0 );
init( FETCH_CHANGEFEED_PARALLELISM, 2 );
init( BUGGIFY_BLOCK_BYTES, 10000 );
init( STORAGE_RECOVERY_VERSION_LAG_LIMIT, 2 * MAX_READ_TRANSACTION_LIFE_VERSIONS );
init( STORAGE_COMMIT_BYTES, 10000000 ); if( randomize && BUGGIFY ) STORAGE_COMMIT_BYTES = 2000000;
init( STORAGE_FETCH_BYTES, 2500000 ); if( randomize && BUGGIFY ) STORAGE_FETCH_BYTES = 500000;
init( STORAGE_DURABILITY_LAG_REJECT_THRESHOLD, 0.25 );
@ -838,6 +843,9 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi
init( BG_MAX_SPLIT_FANOUT, 10 ); if( randomize && BUGGIFY ) BG_MAX_SPLIT_FANOUT = deterministicRandom()->randomInt(5, 15);
init( BG_HOT_SNAPSHOT_VERSIONS, 5000000 );
init( BG_CONSISTENCY_CHECK_ENABLED, true ); if (randomize && BUGGIFY) BG_CONSISTENCY_CHECK_ENABLED = false;
init( BG_CONSISTENCY_CHECK_TARGET_SPEED_KB, 1000 ); if (randomize && BUGGIFY) BG_CONSISTENCY_CHECK_TARGET_SPEED_KB *= (deterministicRandom()->randomInt(2, 50) / 10);
init( BLOB_WORKER_INITIAL_SNAPSHOT_PARALLELISM, 8 ); if( randomize && BUGGIFY ) BLOB_WORKER_INITIAL_SNAPSHOT_PARALLELISM = 1;
init( BLOB_WORKER_TIMEOUT, 10.0 ); if( randomize && BUGGIFY ) BLOB_WORKER_TIMEOUT = 1.0;
init( BLOB_WORKER_REQUEST_TIMEOUT, 5.0 ); if( randomize && BUGGIFY ) BLOB_WORKER_REQUEST_TIMEOUT = 1.0;

View File

@ -233,6 +233,14 @@ public:
double DD_FAILURE_TIME;
double DD_ZERO_HEALTHY_TEAM_DELAY;
// Run storage enginee on a child process on the same machine with storage process
bool REMOTE_KV_STORE;
// A delay to avoid race on file resources if the new kv store process started immediately after the previous kv
// store process died
double REMOTE_KV_STORE_INIT_DELAY;
// max waiting time for the remote kv store to initialize
double REMOTE_KV_STORE_MAX_INIT_DURATION;
// KeyValueStore SQLITE
int CLEAR_BUFFER_SIZE;
double READ_VALUE_TIME_ESTIMATE;
@ -488,6 +496,7 @@ public:
double MIN_REBOOT_TIME;
double MAX_REBOOT_TIME;
std::string LOG_DIRECTORY;
std::string CONN_FILE;
int64_t SERVER_MEM_LIMIT;
double SYSTEM_MONITOR_FREQUENCY;
@ -589,6 +598,7 @@ public:
int FETCH_KEYS_LOWER_PRIORITY;
int FETCH_CHANGEFEED_PARALLELISM;
int BUGGIFY_BLOCK_BYTES;
int64_t STORAGE_RECOVERY_VERSION_LAG_LIMIT;
double STORAGE_DURABILITY_LAG_REJECT_THRESHOLD;
double STORAGE_DURABILITY_LAG_MIN_RATE;
int STORAGE_COMMIT_BYTES;
@ -788,6 +798,8 @@ public:
int BG_DELTA_BYTES_BEFORE_COMPACT;
int BG_MAX_SPLIT_FANOUT;
int BG_HOT_SNAPSHOT_VERSIONS;
int BG_CONSISTENCY_CHECK_ENABLED;
int BG_CONSISTENCY_CHECK_TARGET_SPEED_KB;
int BLOB_WORKER_INITIAL_SNAPSHOT_PARALLELISM;
double BLOB_WORKER_TIMEOUT; // Blob Manager's reaction time to a blob worker failure

View File

@ -89,12 +89,19 @@ struct StorageServerInterface {
RequestStream<struct GetCheckpointRequest> checkpoint;
RequestStream<struct FetchCheckpointRequest> fetchCheckpoint;
explicit StorageServerInterface(UID uid) : uniqueID(uid) {}
StorageServerInterface() : uniqueID(deterministicRandom()->randomUniqueID()) {}
private:
bool acceptingRequests;
public:
explicit StorageServerInterface(UID uid) : uniqueID(uid) { acceptingRequests = false; }
StorageServerInterface() : uniqueID(deterministicRandom()->randomUniqueID()) { acceptingRequests = false; }
NetworkAddress address() const { return getValue.getEndpoint().getPrimaryAddress(); }
NetworkAddress stableAddress() const { return getValue.getEndpoint().getStableAddress(); }
Optional<NetworkAddress> secondaryAddress() const { return getValue.getEndpoint().addresses.secondaryAddress; }
UID id() const { return uniqueID; }
bool isAcceptingRequests() const { return acceptingRequests; }
void startAcceptingRequests() { acceptingRequests = true; }
void stopAcceptingRequests() { acceptingRequests = false; }
bool isTss() const { return tssPairID.present(); }
std::string toString() const { return id().shortString(); }
template <class Ar>
@ -105,7 +112,11 @@ struct StorageServerInterface {
if (ar.protocolVersion().hasSmallEndpoints()) {
if (ar.protocolVersion().hasTSS()) {
serializer(ar, uniqueID, locality, getValue, tssPairID);
if (ar.protocolVersion().hasStorageInterfaceReadiness()) {
serializer(ar, uniqueID, locality, getValue, tssPairID, acceptingRequests);
} else {
serializer(ar, uniqueID, locality, getValue, tssPairID);
}
} else {
serializer(ar, uniqueID, locality, getValue);
}
@ -925,7 +936,7 @@ struct OverlappingChangeFeedsReply {
};
struct OverlappingChangeFeedsRequest {
constexpr static FileIdentifier file_identifier = 10726174;
constexpr static FileIdentifier file_identifier = 7228462;
KeyRange range;
Version minVersion;
ReplyPromise<OverlappingChangeFeedsReply> reply;
@ -940,7 +951,7 @@ struct OverlappingChangeFeedsRequest {
};
struct ChangeFeedVersionUpdateReply {
constexpr static FileIdentifier file_identifier = 11815134;
constexpr static FileIdentifier file_identifier = 4246160;
Version version = 0;
ChangeFeedVersionUpdateReply() {}

View File

@ -602,28 +602,18 @@ const Key serverListKeyFor(UID serverID) {
return wr.toValue();
}
// TODO use flatbuffers depending on version
const Value serverListValue(StorageServerInterface const& server) {
BinaryWriter wr(IncludeVersion(ProtocolVersion::withServerListValue()));
wr << server;
return wr.toValue();
auto protocolVersion = currentProtocolVersion;
protocolVersion.addObjectSerializerFlag();
return ObjectWriter::toValue(server, IncludeVersion(protocolVersion));
}
UID decodeServerListKey(KeyRef const& key) {
UID serverID;
BinaryReader rd(key.removePrefix(serverListKeys.begin), Unversioned());
rd >> serverID;
return serverID;
}
StorageServerInterface decodeServerListValue(ValueRef const& value) {
StorageServerInterface s;
BinaryReader reader(value, IncludeVersion());
reader >> s;
return s;
}
const Value serverListValueFB(StorageServerInterface const& server) {
return ObjectWriter::toValue(server, IncludeVersion());
}
StorageServerInterface decodeServerListValueFB(ValueRef const& value) {
StorageServerInterface s;
@ -632,6 +622,18 @@ StorageServerInterface decodeServerListValueFB(ValueRef const& value) {
return s;
}
StorageServerInterface decodeServerListValue(ValueRef const& value) {
StorageServerInterface s;
BinaryReader reader(value, IncludeVersion());
if (!reader.protocolVersion().hasStorageInterfaceReadiness()) {
reader >> s;
return s;
}
return decodeServerListValueFB(value);
}
// processClassKeys.contains(k) iff k.startsWith( processClassKeys.begin ) because '/'+1 == '0'
const KeyRangeRef processClassKeys(LiteralStringRef("\xff/processClass/"), LiteralStringRef("\xff/processClass0"));
const KeyRef processClassPrefix = processClassKeys.begin;
@ -1205,23 +1207,26 @@ const KeyRange blobGranuleFileKeyRangeFor(UID granuleID) {
return KeyRangeRef(startKey, strinc(startKey));
}
const Value blobGranuleFileValueFor(StringRef const& filename, int64_t offset, int64_t length) {
const Value blobGranuleFileValueFor(StringRef const& filename, int64_t offset, int64_t length, int64_t fullFileLength) {
BinaryWriter wr(IncludeVersion(ProtocolVersion::withBlobGranule()));
wr << filename;
wr << offset;
wr << length;
wr << fullFileLength;
return wr.toValue();
}
std::tuple<Standalone<StringRef>, int64_t, int64_t> decodeBlobGranuleFileValue(ValueRef const& value) {
std::tuple<Standalone<StringRef>, int64_t, int64_t, int64_t> decodeBlobGranuleFileValue(ValueRef const& value) {
StringRef filename;
int64_t offset;
int64_t length;
int64_t fullFileLength;
BinaryReader reader(value, IncludeVersion());
reader >> filename;
reader >> offset;
reader >> length;
return std::tuple(filename, offset, length);
reader >> fullFileLength;
return std::tuple(filename, offset, length, fullFileLength);
}
const Value blobGranulePruneValueFor(Version version, KeyRange range, bool force) {
@ -1405,29 +1410,31 @@ const KeyRef tenantLastIdKey = "\xff/tenantLastId/"_sr;
const KeyRef tenantDataPrefixKey = "\xff/tenantDataPrefix"_sr;
// for tests
void testSSISerdes(StorageServerInterface const& ssi, bool useFB) {
printf("ssi=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\naddress=%s\ngetValue=%s\n\n\n",
void testSSISerdes(StorageServerInterface const& ssi) {
printf("ssi=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\nacceptingRequests=%s\naddress=%s\ngetValue=%s\n\n\n",
ssi.id().toString().c_str(),
ssi.locality.toString().c_str(),
ssi.isTss() ? "true" : "false",
ssi.isTss() ? ssi.tssPairID.get().toString().c_str() : "",
ssi.isAcceptingRequests() ? "true" : "false",
ssi.address().toString().c_str(),
ssi.getValue.getEndpoint().token.toString().c_str());
StorageServerInterface ssi2 =
(useFB) ? decodeServerListValueFB(serverListValueFB(ssi)) : decodeServerListValue(serverListValue(ssi));
StorageServerInterface ssi2 = decodeServerListValue(serverListValue(ssi));
printf("ssi2=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\naddress=%s\ngetValue=%s\n\n\n",
printf("ssi2=\nid=%s\nlocality=%s\nisTss=%s\ntssId=%s\nacceptingRequests=%s\naddress=%s\ngetValue=%s\n\n\n",
ssi2.id().toString().c_str(),
ssi2.locality.toString().c_str(),
ssi2.isTss() ? "true" : "false",
ssi2.isTss() ? ssi2.tssPairID.get().toString().c_str() : "",
ssi2.isAcceptingRequests() ? "true" : "false",
ssi2.address().toString().c_str(),
ssi2.getValue.getEndpoint().token.toString().c_str());
ASSERT(ssi.id() == ssi2.id());
ASSERT(ssi.locality == ssi2.locality);
ASSERT(ssi.isTss() == ssi2.isTss());
ASSERT(ssi.isAcceptingRequests() == ssi2.isAcceptingRequests());
if (ssi.isTss()) {
ASSERT(ssi2.tssPairID.get() == ssi2.tssPairID.get());
}
@ -1449,13 +1456,11 @@ TEST_CASE("/SystemData/SerDes/SSI") {
ssi.locality = localityData;
ssi.initEndpoints();
testSSISerdes(ssi, false);
testSSISerdes(ssi, true);
testSSISerdes(ssi);
ssi.tssPairID = UID(0x2345234523452345, 0x1238123812381238);
testSSISerdes(ssi, false);
testSSISerdes(ssi, true);
testSSISerdes(ssi);
printf("ssi serdes test complete\n");
return Void();

View File

@ -575,8 +575,8 @@ const Key blobGranuleFileKeyFor(UID granuleID, Version fileVersion, uint8_t file
std::tuple<UID, Version, uint8_t> decodeBlobGranuleFileKey(KeyRef const& key);
const KeyRange blobGranuleFileKeyRangeFor(UID granuleID);
const Value blobGranuleFileValueFor(StringRef const& filename, int64_t offset, int64_t length);
std::tuple<Standalone<StringRef>, int64_t, int64_t> decodeBlobGranuleFileValue(ValueRef const& value);
const Value blobGranuleFileValueFor(StringRef const& filename, int64_t offset, int64_t length, int64_t fullFileLength);
std::tuple<Standalone<StringRef>, int64_t, int64_t, int64_t> decodeBlobGranuleFileValue(ValueRef const& value);
const Value blobGranulePruneValueFor(Version version, KeyRange range, bool force);
std::tuple<Version, KeyRange, bool> decodeBlobGranulePruneValue(ValueRef const& value);

View File

@ -6,7 +6,7 @@ assert_no_version_h(fdbmonitor)
if(UNIX AND NOT APPLE)
target_link_libraries(fdbmonitor PRIVATE rt)
endif()
# FIXME: This include directory is an ugly hack. We probably want to fix this
# FIXME: This include directory is an ugly hack. We probably want to fix this.
# as soon as we get rid of the old build system
target_link_libraries(fdbmonitor PUBLIC Threads::Threads)
@ -14,11 +14,17 @@ target_link_libraries(fdbmonitor PUBLIC Threads::Threads)
# appears to change its behavior (it no longer seems to restart killed
# processes). fdbmonitor is single-threaded anyway.
get_target_property(fdbmonitor_options fdbmonitor COMPILE_OPTIONS)
list(REMOVE_ITEM fdbmonitor_options "-fsanitize=thread")
set_property(TARGET fdbmonitor PROPERTY COMPILE_OPTIONS ${target_options})
if (NOT "${fdbmonitor_options}" STREQUAL "fdbmonitor_options-NOTFOUND")
list(REMOVE_ITEM fdbmonitor_options "-fsanitize=thread")
set_property(TARGET fdbmonitor PROPERTY COMPILE_OPTIONS ${fdbmonitor_options})
endif ()
get_target_property(fdbmonitor_options fdbmonitor LINK_OPTIONS)
list(REMOVE_ITEM fdbmonitor_options "-fsanitize=thread")
set_property(TARGET fdbmonitor PROPERTY LINK_OPTIONS ${target_options})
if (NOT "${fdbmonitor_options}" STREQUAL "fdbmonitor_options-NOTFOUND")
list(REMOVE_ITEM fdbmonitor_options "-fsanitize=thread")
set_property(TARGET fdbmonitor PROPERTY LINK_OPTIONS ${fdbmonitor_options})
endif ()
if(GENERATE_DEBUG_PACKAGES)
fdb_install(TARGETS fdbmonitor DESTINATION fdbmonitor COMPONENT server)

View File

@ -30,14 +30,17 @@
#define FLOW_ASYNCFILEKAIO_ACTOR_H
#include "fdbrpc/IAsyncFile.h"
#include <stdio.h>
#include <fcntl.h>
#include <sys/stat.h>
#include <sys/eventfd.h>
#include <sys/syscall.h>
#include "fdbrpc/linux_kaio.h"
#include "fdbserver/Knobs.h"
#include "flow/Knobs.h"
#include "flow/Histogram.h"
#include "flow/UnitTest.h"
#include <stdio.h>
#include "flow/crc32c.h"
#include "flow/genericactors.actor.h"
#include "flow/actorcompiler.h" // This must be the last #include.
@ -46,6 +49,14 @@
// /data/v7/fdb/
#define KAIO_LOGGING 0
struct AsyncFileKAIOMetrics {
Reference<Histogram> readLatencyDist;
Reference<Histogram> writeLatencyDist;
Reference<Histogram> syncLatencyDist;
} g_asyncFileKAIOMetrics;
Future<Void> g_asyncFileKAIOHistogramLogger;
DESCR struct SlowAioSubmit {
int64_t submitDuration; // ns
int64_t truncateDuration; // ns
@ -343,6 +354,7 @@ public:
#endif
KAIOLogEvent(logFile, id, OpLogEntry::SYNC, OpLogEntry::START);
double start_time = now();
Future<Void> fsync = throwErrorIfFailed(
Reference<AsyncFileKAIO>::addRef(this),
@ -352,12 +364,11 @@ public:
submit(io, "write");
fsync=success(io->result.getFuture());*/
#if KAIO_LOGGING
fsync = map(fsync, [=](Void r) mutable {
KAIOLogEvent(logFile, id, OpLogEntry::SYNC, OpLogEntry::COMPLETE);
g_asyncFileKAIOMetrics.syncLatencyDist->sampleSeconds(now() - start_time);
return r;
});
#endif
if (flags & OPEN_ATOMIC_WRITE_AND_CREATE) {
flags &= ~OPEN_ATOMIC_WRITE_AND_CREATE;
@ -630,6 +641,16 @@ private:
countFileLogicalReads.init(LiteralStringRef("AsyncFile.CountFileLogicalReads"), filename);
countLogicalWrites.init(LiteralStringRef("AsyncFile.CountLogicalWrites"));
countLogicalReads.init(LiteralStringRef("AsyncFile.CountLogicalReads"));
if (!g_asyncFileKAIOHistogramLogger.isValid()) {
auto& metrics = g_asyncFileKAIOMetrics;
metrics.readLatencyDist = Reference<Histogram>(new Histogram(
Reference<HistogramRegistry>(), "AsyncFileKAIO", "ReadLatency", Histogram::Unit::microseconds));
metrics.writeLatencyDist = Reference<Histogram>(new Histogram(
Reference<HistogramRegistry>(), "AsyncFileKAIO", "WriteLatency", Histogram::Unit::microseconds));
metrics.syncLatencyDist = Reference<Histogram>(new Histogram(
Reference<HistogramRegistry>(), "AsyncFileKAIO", "SyncLatency", Histogram::Unit::microseconds));
g_asyncFileKAIOHistogramLogger = histogramLogger(SERVER_KNOBS->DISK_METRIC_LOGGING_INTERVAL);
}
}
#if KAIO_LOGGING
@ -749,10 +770,33 @@ private:
ctx.removeFromRequestList(iob);
}
auto& metrics = g_asyncFileKAIOMetrics;
switch (iob->aio_lio_opcode) {
case IO_CMD_PREAD:
metrics.readLatencyDist->sampleSeconds(now() - iob->startTime);
break;
case IO_CMD_PWRITE:
metrics.writeLatencyDist->sampleSeconds(now() - iob->startTime);
break;
}
iob->setResult(ev[i].result);
}
}
}
ACTOR static Future<Void> histogramLogger(double interval) {
state double currentTime;
loop {
currentTime = now();
wait(delay(interval));
double elapsed = now() - currentTime;
auto& metrics = g_asyncFileKAIOMetrics;
metrics.readLatencyDist->writeToLog(elapsed);
metrics.writeLatencyDist->writeToLog(elapsed);
metrics.syncLatencyDist->writeToLog(elapsed);
}
}
};
#if KAIO_LOGGING

View File

@ -10,6 +10,7 @@ set(FDBRPC_SRCS
AsyncFileNonDurable.actor.cpp
AsyncFileWriteChecker.cpp
FailureMonitor.actor.cpp
FlowProcess.actor.h
FlowTransport.actor.cpp
genericactors.actor.h
genericactors.actor.cpp

View File

@ -0,0 +1,94 @@
/*
* FlowProcess.actor.h
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#if defined(NO_INTELLISENSE) && !defined(FDBRPC_FLOW_PROCESS_ACTOR_G_H)
#define FDBRPC_FLOW_PROCESS_ACTOR_G_H
#include "fdbrpc/FlowProcess.actor.g.h"
#elif !defined(FDBRPC_FLOW_PROCESS_ACTOR_H)
#define FDBRPC_FLOW_PROCESS_ACTOR_H
#include "fdbrpc/fdbrpc.h"
#include <string>
#include <map>
#include <flow/actorcompiler.h> // has to be last include
struct FlowProcessInterface {
constexpr static FileIdentifier file_identifier = 3491839;
RequestStream<struct FlowProcessRegistrationRequest> registerProcess;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, registerProcess);
}
};
struct FlowProcessRegistrationRequest {
constexpr static FileIdentifier file_identifier = 3411838;
Standalone<StringRef> flowProcessInterface;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, flowProcessInterface);
}
};
class FlowProcess {
public:
virtual ~FlowProcess() {}
virtual StringRef name() const = 0;
virtual StringRef serializedInterface() const = 0;
virtual Future<Void> run() = 0;
virtual void registerEndpoint(Endpoint p) = 0;
};
struct IProcessFactory {
static FlowProcess* create(std::string const& name) {
auto it = factories().find(name);
if (it == factories().end())
return nullptr; // or throw?
return it->second->create();
}
static std::map<std::string, IProcessFactory*>& factories() {
static std::map<std::string, IProcessFactory*> theFactories;
return theFactories;
}
virtual FlowProcess* create() = 0;
virtual const char* getName() = 0;
};
template <class ProcessType>
struct ProcessFactory : IProcessFactory {
ProcessFactory(const char* name) : name(name) { factories()[name] = this; }
FlowProcess* create() override { return new ProcessType(); }
const char* getName() override { return this->name; }
private:
const char* name;
};
#include <flow/unactorcompiler.h>
#endif

View File

@ -991,7 +991,8 @@ static void scanPackets(TransportData* transport,
Arena& arena,
NetworkAddress const& peerAddress,
ProtocolVersion peerProtocolVersion,
Future<Void> disconnect) {
Future<Void> disconnect,
bool isStableConnection) {
// Find each complete packet in the given byte range and queue a ready task to deliver it.
// Remove the complete packets from the range by increasing unprocessed_begin.
// There won't be more than 64K of data plus one packet, so this shouldn't take a long time.
@ -1030,7 +1031,7 @@ static void scanPackets(TransportData* transport,
if (checksumEnabled) {
bool isBuggifyEnabled = false;
if (g_network->isSimulated() &&
if (g_network->isSimulated() && !isStableConnection &&
g_network->now() - g_simulator.lastConnectionFailure > g_simulator.connectionFailuresDisableDuration &&
BUGGIFY_WITH_PROB(0.0001)) {
g_simulator.lastConnectionFailure = g_network->now();
@ -1057,7 +1058,8 @@ static void scanPackets(TransportData* transport,
if (isBuggifyEnabled) {
TraceEvent(SevInfo, "ChecksumMismatchExp")
.detail("PacketChecksum", packetChecksum)
.detail("CalculatedChecksum", calculatedChecksum);
.detail("CalculatedChecksum", calculatedChecksum)
.detail("PeerAddress", peerAddress.toString());
} else {
TraceEvent(SevWarnAlways, "ChecksumMismatchUnexp")
.detail("PacketChecksum", packetChecksum)
@ -1305,7 +1307,8 @@ ACTOR static Future<Void> connectionReader(TransportData* transport,
arena,
peerAddress,
peerProtocolVersion,
peer->disconnect.getFuture());
peer->disconnect.getFuture(),
g_network->isSimulated() && conn->isStableConnection());
} else {
unprocessed_begin = unprocessed_end;
peer->resetPing.trigger();
@ -1364,6 +1367,11 @@ ACTOR static Future<Void> listen(TransportData* self, NetworkAddress listenAddr)
state ActorCollectionNoErrors
incoming; // Actors monitoring incoming connections that haven't yet been associated with a peer
state Reference<IListener> listener = INetworkConnections::net()->listen(listenAddr);
if (!g_network->isSimulated() && self->localAddresses.address.port == 0) {
TraceEvent(SevInfo, "UpdatingListenAddress")
.detail("AssignedListenAddress", listener->getListenAddress().toString());
self->localAddresses.address = listener->getListenAddress();
}
state uint64_t connectionCount = 0;
try {
loop {

View File

@ -20,6 +20,7 @@
#include <cinttypes>
#include <memory>
#include <string>
#include "contrib/fmt-8.1.1/include/fmt/format.h"
#include "fdbrpc/simulator.h"
@ -121,20 +122,24 @@ void ISimulator::displayWorkers() const {
int openCount = 0;
struct SimClogging {
double getSendDelay(NetworkAddress from, NetworkAddress to) const { return halfLatency(); }
double getSendDelay(NetworkAddress from, NetworkAddress to, bool stableConnection = false) const {
// stable connection here means it's a local connection between processes on the same machine
// we expect it to have much lower latency
return (stableConnection ? 0.1 : 1.0) * halfLatency();
}
double getRecvDelay(NetworkAddress from, NetworkAddress to) {
double getRecvDelay(NetworkAddress from, NetworkAddress to, bool stableConnection = false) {
auto pair = std::make_pair(from.ip, to.ip);
double tnow = now();
double t = tnow + halfLatency();
if (!g_simulator.speedUpSimulation)
double t = tnow + (stableConnection ? 0.1 : 1.0) * halfLatency();
if (!g_simulator.speedUpSimulation && !stableConnection)
t += clogPairLatency[pair];
if (!g_simulator.speedUpSimulation && clogPairUntil.count(pair))
if (!g_simulator.speedUpSimulation && !stableConnection && clogPairUntil.count(pair))
t = std::max(t, clogPairUntil[pair]);
if (!g_simulator.speedUpSimulation && clogRecvUntil.count(to.ip))
if (!g_simulator.speedUpSimulation && !stableConnection && clogRecvUntil.count(to.ip))
t = std::max(t, clogRecvUntil[to.ip]);
return t - tnow;
@ -182,8 +187,8 @@ SimClogging g_clogging;
struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
Sim2Conn(ISimulator::ProcessInfo* process)
: opened(false), closedByCaller(false), process(process), dbgid(deterministicRandom()->randomUniqueID()),
stopReceive(Never()) {
: opened(false), closedByCaller(false), stableConnection(false), process(process),
dbgid(deterministicRandom()->randomUniqueID()), stopReceive(Never()) {
pipes = sender(this) && receiver(this);
}
@ -202,7 +207,18 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
process->address.ip,
FLOW_KNOBS->MAX_CLOGGING_LATENCY * deterministicRandom()->random01());
sendBufSize = std::max<double>(deterministicRandom()->randomInt(0, 5000000), 25e6 * (latency + .002));
TraceEvent("Sim2Connection").detail("SendBufSize", sendBufSize).detail("Latency", latency);
// options like clogging or bitsflip are disabled for stable connections
stableConnection = std::any_of(process->childs.begin(),
process->childs.end(),
[&](ISimulator::ProcessInfo* child) { return child && child == peerProcess; }) ||
std::any_of(peerProcess->childs.begin(),
peerProcess->childs.end(),
[&](ISimulator::ProcessInfo* child) { return child && child == process; });
TraceEvent("Sim2Connection")
.detail("SendBufSize", sendBufSize)
.detail("Latency", latency)
.detail("StableConnection", stableConnection);
}
~Sim2Conn() { ASSERT_ABORT(!opened || closedByCaller); }
@ -222,6 +238,8 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
bool isPeerGone() const { return !peer || peerProcess->failed; }
bool isStableConnection() const override { return stableConnection; }
void peerClosed() {
leakedConnectionTracker = trackLeakedConnection(this);
stopReceive = delay(1.0);
@ -249,7 +267,7 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
ASSERT(limit > 0);
int toSend = 0;
if (BUGGIFY) {
if (BUGGIFY && !stableConnection) {
toSend = std::min(limit, buffer->bytes_written - buffer->bytes_sent);
} else {
for (auto p = buffer; p; p = p->next) {
@ -262,7 +280,7 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
}
}
ASSERT(toSend);
if (BUGGIFY)
if (BUGGIFY && !stableConnection)
toSend = std::min(toSend, deterministicRandom()->randomInt(0, 1000));
if (!peer)
@ -286,7 +304,7 @@ struct Sim2Conn final : IConnection, ReferenceCounted<Sim2Conn> {
NetworkAddress getPeerAddress() const override { return peerEndpoint; }
UID getDebugID() const override { return dbgid; }
bool opened, closedByCaller;
bool opened, closedByCaller, stableConnection;
private:
ISimulator::ProcessInfo *process, *peerProcess;
@ -336,10 +354,12 @@ private:
deterministicRandom()->random01() < .5
? self->sentBytes.get()
: deterministicRandom()->randomInt64(self->receivedBytes.get(), self->sentBytes.get() + 1);
wait(delay(g_clogging.getSendDelay(self->process->address, self->peerProcess->address)));
wait(delay(g_clogging.getSendDelay(
self->process->address, self->peerProcess->address, self->isStableConnection())));
wait(g_simulator.onProcess(self->process));
ASSERT(g_simulator.getCurrentProcess() == self->process);
wait(delay(g_clogging.getRecvDelay(self->process->address, self->peerProcess->address)));
wait(delay(g_clogging.getRecvDelay(
self->process->address, self->peerProcess->address, self->isStableConnection())));
ASSERT(g_simulator.getCurrentProcess() == self->process);
if (self->stopReceive.isReady()) {
wait(Future<Void>(Never()));
@ -389,7 +409,9 @@ private:
}
void rollRandomClose() {
if (now() - g_simulator.lastConnectionFailure > g_simulator.connectionFailuresDisableDuration &&
// make sure connections between parenta and their childs are not closed
if (!stableConnection &&
now() - g_simulator.lastConnectionFailure > g_simulator.connectionFailuresDisableDuration &&
deterministicRandom()->random01() < .00001) {
g_simulator.lastConnectionFailure = now();
double a = deterministicRandom()->random01(), b = deterministicRandom()->random01();
@ -1101,6 +1123,10 @@ public:
if (mustBeDurable || deterministicRandom()->random01() < 0.5) {
state ISimulator::ProcessInfo* currentProcess = g_simulator.getCurrentProcess();
state TaskPriority currentTaskID = g_network->getCurrentTask();
TraceEvent(SevDebug, "Sim2DeleteFileImpl")
.detail("CurrentProcess", currentProcess->toString())
.detail("Filename", filename)
.detail("Durable", mustBeDurable);
wait(g_simulator.onMachine(currentProcess));
try {
wait(::delay(0.05 * deterministicRandom()->random01()));
@ -1118,6 +1144,9 @@ public:
throw err;
}
} else {
TraceEvent(SevDebug, "Sim2DeleteFileImplNonDurable")
.detail("Filename", filename)
.detail("Durable", mustBeDurable);
TEST(true); // Simulated non-durable delete
return Void();
}
@ -1163,6 +1192,9 @@ public:
MachineInfo& machine = machines[locality.machineId().get()];
if (!machine.machineId.present())
machine.machineId = locality.machineId();
if (port == 0 && std::string(name) == "remote flow process") {
port = machine.getRandomPort();
}
for (int i = 0; i < machine.processes.size(); i++) {
if (machine.processes[i]->locality.machineId() !=
locality.machineId()) { // SOMEDAY: compute ip from locality to avoid this check
@ -1220,6 +1252,11 @@ public:
.detail("Excluded", m->excluded)
.detail("Cleared", m->cleared);
if (std::string(name) == "remote flow process") {
protectedAddresses.insert(m->address);
TraceEvent(SevDebug, "NewFlowProcessProtected").detail("Address", m->address);
}
// FIXME: Sometimes, connections to/from this process will explicitly close
return m;
@ -1497,6 +1534,7 @@ public:
.detail("MachineId", p->locality.machineId());
currentlyRebootingProcesses.insert(std::pair<NetworkAddress, ProcessInfo*>(p->address, p));
std::vector<ProcessInfo*>& processes = machines[p->locality.machineId().get()].processes;
machines[p->locality.machineId().get()].removeRemotePort(p->address.port);
if (p != processes.back()) {
auto it = std::find(processes.begin(), processes.end(), p);
std::swap(*it, processes.back());
@ -1520,7 +1558,8 @@ public:
.detail("Protected", protectedAddresses.count(machine->address))
.backtrace();
// This will remove all the "tracked" messages that came from the machine being killed
latestEventCache.clear();
if (std::string(machine->name) != "remote flow process")
latestEventCache.clear();
machine->failed = true;
} else if (kt == InjectFaults) {
TraceEvent(SevWarn, "FaultMachine")
@ -1548,7 +1587,8 @@ public:
} else {
ASSERT(false);
}
ASSERT(!protectedAddresses.count(machine->address) || machine->rebooting);
ASSERT(!protectedAddresses.count(machine->address) || machine->rebooting ||
std::string(machine->name) == "remote flow process");
}
void rebootProcess(ProcessInfo* process, KillType kt) override {
if (kt == RebootProcessAndDelete && protectedAddresses.count(process->address)) {
@ -2390,8 +2430,19 @@ ACTOR void doReboot(ISimulator::ProcessInfo* p, ISimulator::KillType kt) {
kt ==
ISimulator::RebootProcessAndDelete); // Simulated process rebooted with data and coordination state deletion
if (p->rebooting || !p->isReliable())
if (p->rebooting || !p->isReliable()) {
TraceEvent(SevDebug, "DoRebootFailed")
.detail("Rebooting", p->rebooting)
.detail("Reliable", p->isReliable());
return;
} else if (std::string(p->name) == "remote flow process") {
TraceEvent(SevDebug, "DoRebootFailed").detail("Name", p->name).detail("Address", p->address);
return;
} else if (p->getChilds().size()) {
TraceEvent(SevDebug, "DoRebootFailedOnParentProcess").detail("Address", p->address);
return;
}
TraceEvent("RebootingProcess")
.detail("KillType", kt)
.detail("Address", p->address)

View File

@ -21,6 +21,7 @@
#ifndef FLOW_SIMULATOR_H
#define FLOW_SIMULATOR_H
#include "flow/ProtocolVersion.h"
#include <algorithm>
#include <string>
#pragma once
@ -37,12 +38,6 @@ enum ClogMode { ClogDefault, ClogAll, ClogSend, ClogReceive };
class ISimulator : public INetwork {
public:
ISimulator()
: desiredCoordinators(1), physicalDatacenters(1), processesPerMachine(0), listenersPerProcess(1),
extraDB(nullptr), usableRegions(1), allowLogSetKills(true), tssMode(TSSMode::Disabled), isStopped(false),
lastConnectionFailure(0), connectionFailuresDisableDuration(0), speedUpSimulation(false),
backupAgents(BackupAgentType::WaitForType), drAgents(BackupAgentType::WaitForType), allSwapsDisabled(false) {}
// Order matters!
enum KillType {
KillInstantly,
@ -93,6 +88,8 @@ public:
ProtocolVersion protocolVersion;
std::vector<ProcessInfo*> childs;
ProcessInfo(const char* name,
LocalityData locality,
ProcessClass startingClass,
@ -123,6 +120,7 @@ public:
<< " fault_injection_p2:" << fault_injection_p2;
return ss.str();
}
std::vector<ProcessInfo*> const& getChilds() const { return childs; }
// Return true if the class type is suitable for stateful roles, such as tLog and StorageServer.
bool isAvailableClass() const {
@ -208,7 +206,30 @@ public:
std::set<std::string> closingFiles;
Optional<Standalone<StringRef>> machineId;
MachineInfo() : machineProcess(nullptr) {}
const uint16_t remotePortStart;
std::vector<uint16_t> usedRemotePorts;
MachineInfo() : machineProcess(nullptr), remotePortStart(1000) {}
short getRandomPort() {
for (uint16_t i = remotePortStart; i < 60000; i++) {
if (std::find(usedRemotePorts.begin(), usedRemotePorts.end(), i) == usedRemotePorts.end()) {
TraceEvent(SevDebug, "RandomPortOpened").detail("PortNum", i);
usedRemotePorts.push_back(i);
return i;
}
}
UNREACHABLE();
}
void removeRemotePort(uint16_t port) {
if (port < remotePortStart)
return;
auto pos = std::find(usedRemotePorts.begin(), usedRemotePorts.end(), port);
if (pos != usedRemotePorts.end()) {
usedRemotePorts.erase(pos);
}
}
};
ProcessInfo* getProcess(Endpoint const& endpoint) { return getProcessByAddress(endpoint.getPrimaryAddress()); }
@ -393,7 +414,7 @@ public:
int listenersPerProcess;
std::set<NetworkAddress> protectedAddresses;
std::map<NetworkAddress, ProcessInfo*> currentlyRebootingProcesses;
class ClusterConnectionString* extraDB;
std::unique_ptr<class ClusterConnectionString> extraDB;
Reference<IReplicationPolicy> storagePolicy;
Reference<IReplicationPolicy> tLogPolicy;
int32_t tLogWriteAntiQuorum;
@ -456,6 +477,9 @@ public:
return false;
}
ISimulator();
virtual ~ISimulator();
protected:
Mutex mutex;

View File

@ -60,13 +60,14 @@ ACTOR Future<Void> readGranuleFiles(Transaction* tr, Key* startKey, Key endKey,
Standalone<StringRef> filename;
int64_t offset;
int64_t length;
int64_t fullFileLength;
std::tie(gid, version, fileType) = decodeBlobGranuleFileKey(it.key);
ASSERT(gid == granuleID);
std::tie(filename, offset, length) = decodeBlobGranuleFileValue(it.value);
std::tie(filename, offset, length, fullFileLength) = decodeBlobGranuleFileValue(it.value);
BlobFileIndex idx(version, filename.toString(), offset, length);
BlobFileIndex idx(version, filename.toString(), offset, length, fullFileLength);
if (fileType == 'S') {
ASSERT(files->snapshotFiles.empty() || files->snapshotFiles.back().version < idx.version);
files->snapshotFiles.push_back(idx);
@ -168,14 +169,16 @@ void GranuleFiles::getFiles(Version beginVersion,
Version lastIncluded = invalidVersion;
if (snapshotF != snapshotFiles.end()) {
chunk.snapshotVersion = snapshotF->version;
chunk.snapshotFile = BlobFilePointerRef(replyArena, snapshotF->filename, snapshotF->offset, snapshotF->length);
chunk.snapshotFile = BlobFilePointerRef(
replyArena, snapshotF->filename, snapshotF->offset, snapshotF->length, snapshotF->fullFileLength);
lastIncluded = chunk.snapshotVersion;
} else {
chunk.snapshotVersion = invalidVersion;
}
while (deltaF != deltaFiles.end() && deltaF->version < readVersion) {
chunk.deltaFiles.emplace_back_deep(replyArena, deltaF->filename, deltaF->offset, deltaF->length);
chunk.deltaFiles.emplace_back_deep(
replyArena, deltaF->filename, deltaF->offset, deltaF->length, deltaF->fullFileLength);
deltaBytesCounter += deltaF->length;
ASSERT(lastIncluded < deltaF->version);
lastIncluded = deltaF->version;
@ -183,7 +186,8 @@ void GranuleFiles::getFiles(Version beginVersion,
}
// include last delta file that passes readVersion, if it exists
if (deltaF != deltaFiles.end() && lastIncluded < readVersion) {
chunk.deltaFiles.emplace_back_deep(replyArena, deltaF->filename, deltaF->offset, deltaF->length);
chunk.deltaFiles.emplace_back_deep(
replyArena, deltaF->filename, deltaF->offset, deltaF->length, deltaF->fullFileLength);
deltaBytesCounter += deltaF->length;
lastIncluded = deltaF->version;
}
@ -194,7 +198,7 @@ static std::string makeTestFileName(Version v) {
}
static BlobFileIndex makeTestFile(Version v, int64_t len) {
return BlobFileIndex(v, makeTestFileName(v), 0, len);
return BlobFileIndex(v, makeTestFileName(v), 0, len, len);
}
static void checkFile(int expectedVersion, const BlobFilePointerRef& actualFile) {

View File

@ -49,11 +49,12 @@ struct BlobFileIndex {
std::string filename;
int64_t offset;
int64_t length;
int64_t fullFileLength;
BlobFileIndex() {}
BlobFileIndex(Version version, std::string filename, int64_t offset, int64_t length)
: version(version), filename(filename), offset(offset), length(length) {}
BlobFileIndex(Version version, std::string filename, int64_t offset, int64_t length, int64_t fullFileLength)
: version(version), filename(filename), offset(offset), length(length), fullFileLength(fullFileLength) {}
// compare on version
bool operator<(const BlobFileIndex& r) const { return version < r.version; }

View File

@ -0,0 +1,165 @@
/*
* BlobGranuleValidation.actor.cpp
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "fdbserver/BlobGranuleValidation.actor.h"
#include "flow/actorcompiler.h" // has to be last include
ACTOR Future<std::pair<RangeResult, Version>> readFromFDB(Database cx, KeyRange range) {
state bool first = true;
state Version v;
state RangeResult out;
state Transaction tr(cx);
state KeyRange currentRange = range;
loop {
try {
state RangeResult r = wait(tr.getRange(currentRange, CLIENT_KNOBS->TOO_MANY));
Version grv = wait(tr.getReadVersion());
// need consistent version snapshot of range
if (first) {
v = grv;
first = false;
} else if (v != grv) {
// reset the range and restart the read at a higher version
first = true;
out = RangeResult();
currentRange = range;
tr.reset();
continue;
}
out.arena().dependsOn(r.arena());
out.append(out.arena(), r.begin(), r.size());
if (r.more) {
currentRange = KeyRangeRef(keyAfter(r.back().key), currentRange.end);
} else {
break;
}
} catch (Error& e) {
wait(tr.onError(e));
}
}
return std::pair(out, v);
}
// FIXME: typedef this pair type and/or chunk list
ACTOR Future<std::pair<RangeResult, Standalone<VectorRef<BlobGranuleChunkRef>>>> readFromBlob(
Database cx,
Reference<BackupContainerFileSystem> bstore,
KeyRange range,
Version beginVersion,
Version readVersion) {
state RangeResult out;
state Standalone<VectorRef<BlobGranuleChunkRef>> chunks;
state Transaction tr(cx);
loop {
try {
Standalone<VectorRef<BlobGranuleChunkRef>> chunks_ =
wait(tr.readBlobGranules(range, beginVersion, readVersion));
chunks = chunks_;
break;
} catch (Error& e) {
wait(tr.onError(e));
}
}
for (const BlobGranuleChunkRef& chunk : chunks) {
RangeResult chunkRows = wait(readBlobGranule(chunk, range, beginVersion, readVersion, bstore));
out.arena().dependsOn(chunkRows.arena());
out.append(out.arena(), chunkRows.begin(), chunkRows.size());
}
return std::pair(out, chunks);
}
bool compareFDBAndBlob(RangeResult fdb,
std::pair<RangeResult, Standalone<VectorRef<BlobGranuleChunkRef>>> blob,
KeyRange range,
Version v,
bool debug) {
bool correct = fdb == blob.first;
if (!correct) {
TraceEvent ev(SevError, "GranuleMismatch");
ev.detail("RangeStart", range.begin)
.detail("RangeEnd", range.end)
.detail("Version", v)
.detail("FDBSize", fdb.size())
.detail("BlobSize", blob.first.size());
if (debug) {
fmt::print("\nMismatch for [{0} - {1}) @ {2} ({3}). F({4}) B({5}):\n",
range.begin.printable(),
range.end.printable(),
v,
fdb.size(),
blob.first.size());
Optional<KeyValueRef> lastCorrect;
for (int i = 0; i < std::max(fdb.size(), blob.first.size()); i++) {
if (i >= fdb.size() || i >= blob.first.size() || fdb[i] != blob.first[i]) {
printf(" Found mismatch at %d.\n", i);
if (lastCorrect.present()) {
printf(" last correct: %s=%s\n",
lastCorrect.get().key.printable().c_str(),
lastCorrect.get().value.printable().c_str());
}
if (i < fdb.size()) {
printf(" FDB: %s=%s\n", fdb[i].key.printable().c_str(), fdb[i].value.printable().c_str());
} else {
printf(" FDB: <missing>\n");
}
if (i < blob.first.size()) {
printf(" BLB: %s=%s\n",
blob.first[i].key.printable().c_str(),
blob.first[i].value.printable().c_str());
} else {
printf(" BLB: <missing>\n");
}
printf("\n");
break;
}
if (i < fdb.size()) {
lastCorrect = fdb[i];
} else {
lastCorrect = blob.first[i];
}
}
printf("Chunks:\n");
for (auto& chunk : blob.second) {
printf("[%s - %s)\n", chunk.keyRange.begin.printable().c_str(), chunk.keyRange.end.printable().c_str());
printf(" SnapshotFile:\n %s\n",
chunk.snapshotFile.present() ? chunk.snapshotFile.get().toString().c_str() : "<none>");
printf(" DeltaFiles:\n");
for (auto& df : chunk.deltaFiles) {
printf(" %s\n", df.toString().c_str());
}
printf(" Deltas: (%d)", chunk.newDeltas.size());
if (chunk.newDeltas.size() > 0) {
fmt::print(" with version [{0} - {1}]",
chunk.newDeltas[0].version,
chunk.newDeltas[chunk.newDeltas.size() - 1].version);
}
fmt::print(" IncludedVersion: {}\n", chunk.includedVersion);
}
printf("\n");
}
}
return correct;
}

View File

@ -0,0 +1,53 @@
/*
* BlobGranuleValidation.actor.h
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#if defined(NO_INTELLISENSE) && !defined(FDBSERVER_BLOBGRANULEVALIDATION_ACTOR_G_H)
#define FDBSERVER_BLOBGRANULEVALIDATION_ACTOR_G_H
#include "fdbserver/BlobGranuleValidation.actor.g.h"
#elif !defined(FDBSERVER_BLOBGRANULEVALIDATION_ACTOR_H)
#define FDBSERVER_BLOBGRANULEVALIDATION_ACTOR_H
#pragma once
#include "flow/flow.h"
#include "fdbclient/BlobGranuleReader.actor.h"
#include "fdbclient/CommitTransaction.h"
#include "fdbclient/FDBTypes.h"
#include "fdbclient/BlobGranuleCommon.h"
#include "flow/actorcompiler.h" // has to be last include
/* Contains utility functions for validating blob granule data */
ACTOR Future<std::pair<RangeResult, Standalone<VectorRef<BlobGranuleChunkRef>>>> readFromBlob(
Database cx,
Reference<BackupContainerFileSystem> bstore,
KeyRange range,
Version beginVersion,
Version readVersion);
ACTOR Future<std::pair<RangeResult, Version>> readFromFDB(Database cx, KeyRange range);
bool compareFDBAndBlob(RangeResult fdb,
std::pair<RangeResult, Standalone<VectorRef<BlobGranuleChunkRef>>> blob,
KeyRange range,
Version v,
bool debug);
#endif

View File

@ -34,6 +34,7 @@
#include "fdbclient/SystemData.h"
#include "fdbserver/BlobManagerInterface.h"
#include "fdbserver/Knobs.h"
#include "fdbserver/BlobGranuleValidation.actor.h"
#include "fdbserver/BlobGranuleServerCommon.actor.h"
#include "fdbserver/QuietDatabase.h"
#include "fdbserver/WaitFailure.h"
@ -195,10 +196,11 @@ struct RangeAssignment {
};
// SOMEDAY: track worker's reads/writes eventually
struct BlobWorkerStats {
// FIXME: namespace?
struct BlobWorkerInfo {
int numGranulesAssigned;
BlobWorkerStats(int numGranulesAssigned = 0) : numGranulesAssigned(numGranulesAssigned) {}
BlobWorkerInfo(int numGranulesAssigned = 0) : numGranulesAssigned(numGranulesAssigned) {}
};
struct SplitEvaluation {
@ -211,6 +213,29 @@ struct SplitEvaluation {
: epoch(epoch), seqno(seqno), inProgress(inProgress) {}
};
struct BlobManagerStats {
CounterCollection cc;
// FIXME: pruning stats
Counter granuleSplits;
Counter granuleWriteHotSplits;
Counter ccGranulesChecked;
Counter ccRowsChecked;
Counter ccBytesChecked;
Counter ccMismatches;
Future<Void> logger;
// Current stats maintained for a given blob worker process
explicit BlobManagerStats(UID id, double interval, std::unordered_map<UID, BlobWorkerInterface>* workers)
: cc("BlobManagerStats", id.toString()), granuleSplits("GranuleSplits", cc),
granuleWriteHotSplits("GranuleWriteHotSplits", cc), ccGranulesChecked("CCGranulesChecked", cc),
ccRowsChecked("CCRowsChecked", cc), ccBytesChecked("CCBytesChecked", cc), ccMismatches("CCMismatches", cc) {
specialCounter(cc, "WorkerCount", [workers]() { return workers->size(); });
logger = traceCounters("BlobManagerMetrics", id, interval, &cc, "BlobManagerMetrics");
}
};
struct BlobManagerData : NonCopyable, ReferenceCounted<BlobManagerData> {
UID id;
Database db;
@ -218,10 +243,12 @@ struct BlobManagerData : NonCopyable, ReferenceCounted<BlobManagerData> {
PromiseStream<Future<Void>> addActor;
Promise<Void> doLockCheck;
BlobManagerStats stats;
Reference<BackupContainerFileSystem> bstore;
std::unordered_map<UID, BlobWorkerInterface> workersById;
std::unordered_map<UID, BlobWorkerStats> workerStats; // mapping between workerID -> workerStats
std::unordered_map<UID, BlobWorkerInfo> workerStats; // mapping between workerID -> workerStats
std::unordered_set<NetworkAddress> workerAddresses;
std::unordered_set<UID> deadWorkers;
KeyRangeMap<UID> workerAssignments;
@ -246,13 +273,28 @@ struct BlobManagerData : NonCopyable, ReferenceCounted<BlobManagerData> {
PromiseStream<RangeAssignment> rangesToAssign;
BlobManagerData(UID id, Database db, Optional<Key> dcId)
: id(id), db(db), dcId(dcId), knownBlobRanges(false, normalKeys.end),
restartRecruiting(SERVER_KNOBS->DEBOUNCE_RECRUITING_DELAY), recruitingStream(0) {}
: id(id), db(db), dcId(dcId), stats(id, SERVER_KNOBS->WORKER_LOGGING_INTERVAL, &workersById),
knownBlobRanges(false, normalKeys.end), restartRecruiting(SERVER_KNOBS->DEBOUNCE_RECRUITING_DELAY),
recruitingStream(0) {}
// only initialize blob store if actually needed
void initBStore() {
if (!bstore.isValid()) {
if (BM_DEBUG) {
fmt::print("BM {} constructing backup container from {}\n", epoch, SERVER_KNOBS->BG_URL.c_str());
}
bstore = BackupContainerFileSystem::openContainerFS(SERVER_KNOBS->BG_URL, {}, {});
if (BM_DEBUG) {
fmt::print("BM {} constructed backup container\n", epoch);
}
}
}
};
ACTOR Future<Standalone<VectorRef<KeyRef>>> splitRange(Reference<BlobManagerData> bmData,
KeyRange range,
bool writeHot) {
bool writeHot,
bool initialSplit) {
try {
if (BM_DEBUG) {
fmt::print("Splitting new range [{0} - {1}): {2}\n",
@ -269,8 +311,24 @@ ACTOR Future<Standalone<VectorRef<KeyRef>>> splitRange(Reference<BlobManagerData
estimated.bytes);
}
int64_t splitThreshold = SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES;
if (!initialSplit) {
// If we have X MB target granule size, we want to do the initial split to split up into X MB chunks.
// However, if we already have a granule that we are evaluating for split, if we split it as soon as it is
// larger than X MB, we will end up with 2 X/2 MB granules.
// To ensure an average size of X MB, we split granules at 4/3*X, so that they range between 2/3*X and
// 4/3*X, averaging X
splitThreshold = (splitThreshold * 4) / 3;
}
// if write-hot, we want to be able to split smaller, but not infinitely. Allow write-hot granules to be 3x
// smaller
// TODO knob?
// TODO: re-evaluate after we have granule merging?
if (writeHot) {
splitThreshold /= 3;
}
TEST(writeHot); // Change feed write hot split
if (estimated.bytes > SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES || writeHot) {
if (estimated.bytes > splitThreshold) {
// only split on bytes and write rate
state StorageMetrics splitMetrics;
splitMetrics.bytes = SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES;
@ -304,6 +362,7 @@ ACTOR Future<Standalone<VectorRef<KeyRef>>> splitRange(Reference<BlobManagerData
ASSERT(keys.back() == range.end);
return keys;
} else {
TEST(writeHot); // Not splitting write-hot because granules would be too small
if (BM_DEBUG) {
printf("Not splitting range\n");
}
@ -753,6 +812,7 @@ ACTOR Future<Void> monitorClientRanges(Reference<BlobManagerData> bmData) {
}
for (KeyRangeRef range : rangesToRemove) {
TraceEvent("ClientBlobRangeRemoved", bmData->id).detail("Range", range);
if (BM_DEBUG) {
fmt::print(
"BM Got range to revoke [{0} - {1})\n", range.begin.printable(), range.end.printable());
@ -768,7 +828,8 @@ ACTOR Future<Void> monitorClientRanges(Reference<BlobManagerData> bmData) {
state std::vector<Future<Standalone<VectorRef<KeyRef>>>> splitFutures;
// Divide new ranges up into equal chunks by using SS byte sample
for (KeyRangeRef range : rangesToAdd) {
splitFutures.push_back(splitRange(bmData, range, false));
TraceEvent("ClientBlobRangeAdded", bmData->id).detail("Range", range);
splitFutures.push_back(splitRange(bmData, range, false, true));
}
for (auto f : splitFutures) {
@ -869,7 +930,7 @@ ACTOR Future<Void> maybeSplitRange(Reference<BlobManagerData> bmData,
state Standalone<VectorRef<KeyRef>> newRanges;
// first get ranges to split
Standalone<VectorRef<KeyRef>> _newRanges = wait(splitRange(bmData, granuleRange, writeHot));
Standalone<VectorRef<KeyRef>> _newRanges = wait(splitRange(bmData, granuleRange, writeHot, false));
newRanges = _newRanges;
ASSERT(newRanges.size() >= 2);
@ -1096,6 +1157,11 @@ ACTOR Future<Void> maybeSplitRange(Reference<BlobManagerData> bmData,
splitVersion);
}
++bmData->stats.granuleSplits;
if (writeHot) {
++bmData->stats.granuleWriteHotSplits;
}
// transaction committed, send range assignments
// range could have been moved since split eval started, so just revoke from whoever has it
RangeAssignment raRevoke;
@ -1182,6 +1248,8 @@ ACTOR Future<Void> killBlobWorker(Reference<BlobManagerData> bmData, BlobWorkerI
// Remove it from workersById also since otherwise that worker addr will remain excluded
// when we try to recruit new blob workers.
TraceEvent("KillBlobWorker", bmData->id).detail("WorkerId", bwId);
if (registered) {
bmData->deadWorkers.insert(bwId);
bmData->workerStats.erase(bwId);
@ -1471,7 +1539,7 @@ ACTOR Future<Void> checkBlobWorkerList(Reference<BlobManagerData> bmData, Promis
worker.locality.dcId() == bmData->dcId) {
bmData->workerAddresses.insert(worker.stableAddress());
bmData->workersById[worker.id()] = worker;
bmData->workerStats[worker.id()] = BlobWorkerStats();
bmData->workerStats[worker.id()] = BlobWorkerInfo();
bmData->addActor.send(monitorBlobWorker(bmData, worker));
foundAnyNew = true;
} else if (!bmData->workersById.count(worker.id())) {
@ -1581,6 +1649,7 @@ static void addAssignment(KeyRangeMap<std::tuple<UID, int64_t, int64_t>>& map,
}
ACTOR Future<Void> recoverBlobManager(Reference<BlobManagerData> bmData) {
state double recoveryStartTime = now();
state Promise<Void> workerListReady;
bmData->addActor.send(checkBlobWorkerList(bmData, workerListReady));
wait(workerListReady.getFuture());
@ -1836,7 +1905,8 @@ ACTOR Future<Void> recoverBlobManager(Reference<BlobManagerData> bmData) {
TraceEvent("BlobManagerRecovered", bmData->id)
.detail("Epoch", bmData->epoch)
.detail("Granules", bmData->workerAssignments.size())
.detail("Duration", now() - recoveryStartTime)
.detail("Granules", bmData->workerAssignments.size()) // TODO this includes un-set ranges, so it is inaccurate
.detail("Assigned", explicitAssignments)
.detail("Revoked", outOfDateAssignments.size());
@ -1972,7 +2042,7 @@ ACTOR Future<Void> initializeBlobWorker(Reference<BlobManagerData> self, Recruit
if (!self->workerAddresses.count(bwi.stableAddress()) && bwi.locality.dcId() == self->dcId) {
self->workerAddresses.insert(bwi.stableAddress());
self->workersById[bwi.id()] = bwi;
self->workerStats[bwi.id()] = BlobWorkerStats();
self->workerStats[bwi.id()] = BlobWorkerInfo();
self->addActor.send(monitorBlobWorker(self, bwi));
} else if (!self->workersById.count(bwi.id())) {
self->addActor.send(killBlobWorker(self, bwi, false));
@ -2087,6 +2157,8 @@ ACTOR Future<GranuleFiles> loadHistoryFiles(Reference<BlobManagerData> bmData, U
}
}
// FIXME: trace events for pruning
/*
* Deletes all files pertaining to the granule with id granuleId and
* also removes the history entry for this granule from the system keyspace
@ -2502,14 +2574,7 @@ ACTOR Future<Void> pruneRange(Reference<BlobManagerData> self, KeyRangeRef range
* case that the timer is up before any new prune intents arrive).
*/
ACTOR Future<Void> monitorPruneKeys(Reference<BlobManagerData> self) {
// setup bstore
if (BM_DEBUG) {
fmt::print("BM constructing backup container from {}\n", SERVER_KNOBS->BG_URL.c_str());
}
self->bstore = BackupContainerFileSystem::openContainerFS(SERVER_KNOBS->BG_URL);
if (BM_DEBUG) {
printf("BM constructed backup container\n");
}
self->initBStore();
loop {
state Reference<ReadYourWritesTransaction> tr = makeReference<ReadYourWritesTransaction>(self->db);
@ -2678,6 +2743,73 @@ static void blobManagerExclusionSafetyCheck(Reference<BlobManagerData> self,
req.reply.send(reply);
}
// FIXME: could eventually make this more thorough by storing some state in the DB or something
// FIXME: simpler solution could be to shuffle ranges
ACTOR Future<Void> bgConsistencyCheck(Reference<BlobManagerData> bmData) {
state Reference<IRateControl> rateLimiter =
Reference<IRateControl>(new SpeedLimit(SERVER_KNOBS->BG_CONSISTENCY_CHECK_TARGET_SPEED_KB * 1024, 1));
bmData->initBStore();
if (BM_DEBUG) {
fmt::print("BGCC starting\n");
}
loop {
if (g_network->isSimulated() && g_simulator.speedUpSimulation) {
if (BM_DEBUG) {
printf("BGCC stopping\n");
}
return Void();
}
if (bmData->workersById.size() >= 1) {
int tries = 10;
state KeyRange range;
while (tries > 0) {
auto randomRange = bmData->workerAssignments.randomRange();
if (randomRange.value() != UID()) {
range = randomRange.range();
break;
}
tries--;
}
if (tries == 0) {
if (BM_DEBUG) {
printf("BGCC couldn't find random range to check, skipping\n");
}
wait(rateLimiter->getAllowance(SERVER_KNOBS->BG_SNAPSHOT_FILE_TARGET_BYTES));
} else {
state std::pair<RangeResult, Version> fdbResult = wait(readFromFDB(bmData->db, range));
std::pair<RangeResult, Standalone<VectorRef<BlobGranuleChunkRef>>> blobResult =
wait(readFromBlob(bmData->db, bmData->bstore, range, 0, fdbResult.second));
if (!compareFDBAndBlob(fdbResult.first, blobResult, range, fdbResult.second, BM_DEBUG)) {
++bmData->stats.ccMismatches;
}
int64_t bytesRead = fdbResult.first.expectedSize();
++bmData->stats.ccGranulesChecked;
bmData->stats.ccRowsChecked += fdbResult.first.size();
bmData->stats.ccBytesChecked += bytesRead;
// clear fdb result to release memory since it is a state variable
fdbResult = std::pair(RangeResult(), 0);
wait(rateLimiter->getAllowance(bytesRead));
}
} else {
if (BM_DEBUG) {
fmt::print("BGCC found no workers, skipping\n", bmData->workerAssignments.size());
}
wait(delay(60.0));
}
}
}
// Simulation validation that multiple blob managers aren't started with the same epoch
static std::map<int64_t, UID> managerEpochsSeen;
@ -2724,6 +2856,9 @@ ACTOR Future<Void> blobManager(BlobManagerInterface bmInterf,
self->addActor.send(doLockChecks(self));
self->addActor.send(monitorClientRanges(self));
self->addActor.send(monitorPruneKeys(self));
if (SERVER_KNOBS->BG_CONSISTENCY_CHECK_ENABLED) {
self->addActor.send(bgConsistencyCheck(self));
}
if (BUGGIFY) {
self->addActor.send(chaosRangeMover(self));

View File

@ -206,6 +206,7 @@ struct BlobWorkerData : NonCopyable, ReferenceCounted<BlobWorkerData> {
if (BW_DEBUG) {
fmt::print("BW {0} found new manager epoch {1}\n", id.toString(), currentManagerEpoch);
}
TraceEvent(SevDebug, "BlobWorkerFoundNewManager", id).detail("Epoch", epoch);
}
return true;
@ -511,7 +512,8 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(Reference<BlobWorkerData> bwData,
numIterations++;
Key dfKey = blobGranuleFileKeyFor(granuleID, currentDeltaVersion, 'D');
Value dfValue = blobGranuleFileValueFor(fname, 0, serializedSize);
// TODO change once we support file multiplexing
Value dfValue = blobGranuleFileValueFor(fname, 0, serializedSize, serializedSize);
tr->set(dfKey, dfValue);
if (oldGranuleComplete.present()) {
@ -538,7 +540,8 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(Reference<BlobWorkerData> bwData,
if (BUGGIFY_WITH_PROB(0.01)) {
wait(delay(deterministicRandom()->random01()));
}
return BlobFileIndex(currentDeltaVersion, fname, 0, serializedSize);
// FIXME: change when we implement multiplexing
return BlobFileIndex(currentDeltaVersion, fname, 0, serializedSize, serializedSize);
} catch (Error& e) {
wait(tr->onError(e));
}
@ -648,7 +651,8 @@ ACTOR Future<BlobFileIndex> writeSnapshot(Reference<BlobWorkerData> bwData,
wait(readAndCheckGranuleLock(tr, keyRange, epoch, seqno));
numIterations++;
Key snapshotFileKey = blobGranuleFileKeyFor(granuleID, version, 'S');
Key snapshotFileValue = blobGranuleFileValueFor(fname, 0, serializedSize);
// TODO change once we support file multiplexing
Key snapshotFileValue = blobGranuleFileValueFor(fname, 0, serializedSize, serializedSize);
tr->set(snapshotFileKey, snapshotFileValue);
// create granule history at version if this is a new granule with the initial dump from FDB
if (createGranuleHistory) {
@ -692,7 +696,8 @@ ACTOR Future<BlobFileIndex> writeSnapshot(Reference<BlobWorkerData> bwData,
wait(delay(deterministicRandom()->random01()));
}
return BlobFileIndex(version, fname, 0, serializedSize);
// FIXME: change when we implement multiplexing
return BlobFileIndex(version, fname, 0, serializedSize, serializedSize);
}
ACTOR Future<BlobFileIndex> dumpInitialSnapshotFromFDB(Reference<BlobWorkerData> bwData,
@ -731,7 +736,7 @@ ACTOR Future<BlobFileIndex> dumpInitialSnapshotFromFDB(Reference<BlobWorkerData>
Future<Void> streamFuture =
tr->getTransaction().getRangeStream(rowsStream, metadata->keyRange, GetRangeLimits(), Snapshot::True);
wait(streamFuture && success(snapshotWriter));
TraceEvent("BlobGranuleSnapshotFile", bwData->id)
TraceEvent(SevDebug, "BlobGranuleSnapshotFile", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("Version", readVersion);
DEBUG_KEY_RANGE("BlobWorkerFDBSnapshot", readVersion, metadata->keyRange, bwData->id);
@ -755,7 +760,8 @@ ACTOR Future<BlobFileIndex> dumpInitialSnapshotFromFDB(Reference<BlobWorkerData>
wait(tr->onError(e));
retries++;
TEST(true); // Granule initial snapshot failed
TraceEvent(SevWarn, "BlobGranuleInitialSnapshotRetry", bwData->id)
// FIXME: why can't we supress error event?
TraceEvent(retries < 10 ? SevDebug : SevWarn, "BlobGranuleInitialSnapshotRetry", bwData->id)
.error(err)
.detail("Granule", metadata->keyRange)
.detail("Count", retries);
@ -797,7 +803,8 @@ ACTOR Future<BlobFileIndex> compactFromBlob(Reference<BlobWorkerData> bwData,
ASSERT(snapshotVersion < version);
chunk.snapshotFile = BlobFilePointerRef(filenameArena, snapshotF.filename, snapshotF.offset, snapshotF.length);
chunk.snapshotFile = BlobFilePointerRef(
filenameArena, snapshotF.filename, snapshotF.offset, snapshotF.length, snapshotF.fullFileLength);
compactBytesRead += snapshotF.length;
int deltaIdx = files.deltaFiles.size() - 1;
while (deltaIdx >= 0 && files.deltaFiles[deltaIdx].version > snapshotVersion) {
@ -807,7 +814,8 @@ ACTOR Future<BlobFileIndex> compactFromBlob(Reference<BlobWorkerData> bwData,
Version lastDeltaVersion = invalidVersion;
while (deltaIdx < files.deltaFiles.size() && files.deltaFiles[deltaIdx].version <= version) {
BlobFileIndex deltaF = files.deltaFiles[deltaIdx];
chunk.deltaFiles.emplace_back_deep(filenameArena, deltaF.filename, deltaF.offset, deltaF.length);
chunk.deltaFiles.emplace_back_deep(
filenameArena, deltaF.filename, deltaF.offset, deltaF.length, deltaF.fullFileLength);
compactBytesRead += deltaF.length;
lastDeltaVersion = files.deltaFiles[deltaIdx].version;
deltaIdx++;
@ -877,7 +885,7 @@ ACTOR Future<BlobFileIndex> checkSplitAndReSnapshot(Reference<BlobWorkerData> bw
metadata->bytesInNewDeltaFiles);
}
TraceEvent("BlobGranuleSnapshotCheck", bwData->id)
TraceEvent(SevDebug, "BlobGranuleSnapshotCheck", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("Version", reSnapshotVersion);
@ -954,7 +962,7 @@ ACTOR Future<BlobFileIndex> checkSplitAndReSnapshot(Reference<BlobWorkerData> bw
metadata->keyRange.end.printable(),
bytesInNewDeltaFiles);
}
TraceEvent("BlobGranuleSnapshotFile", bwData->id)
TraceEvent(SevDebug, "BlobGranuleSnapshotFile", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("Version", metadata->durableDeltaVersion.get());
@ -1534,7 +1542,7 @@ ACTOR Future<Void> blobGranuleUpdateFiles(Reference<BlobWorkerData> bwData,
bwData->id.toString().substr(0, 5).c_str(),
deltas.version,
rollbackVersion);
TraceEvent(SevWarn, "GranuleRollback", bwData->id)
TraceEvent(SevDebug, "GranuleRollback", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("Version", deltas.version)
.detail("RollbackVersion", rollbackVersion);
@ -1648,7 +1656,7 @@ ACTOR Future<Void> blobGranuleUpdateFiles(Reference<BlobWorkerData> bwData,
lastDeltaVersion,
oldChangeFeedDataComplete.present() ? ". Finalizing " : "");
}
TraceEvent("BlobGranuleDeltaFile", bwData->id)
TraceEvent(SevDebug, "BlobGranuleDeltaFile", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("Version", lastDeltaVersion);
@ -1825,13 +1833,13 @@ ACTOR Future<Void> blobGranuleUpdateFiles(Reference<BlobWorkerData> bwData,
}
if (e.code() == error_code_granule_assignment_conflict) {
TraceEvent(SevInfo, "GranuleAssignmentConflict", bwData->id)
TraceEvent("GranuleAssignmentConflict", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("GranuleID", startState.granuleID);
return Void();
}
if (e.code() == error_code_change_feed_popped) {
TraceEvent(SevInfo, "GranuleGotChangeFeedPopped", bwData->id)
TraceEvent("GranuleChangeFeedPopped", bwData->id)
.detail("Granule", metadata->keyRange)
.detail("GranuleID", startState.granuleID);
return Void();
@ -2573,7 +2581,16 @@ ACTOR Future<GranuleStartState> openGranule(Reference<BlobWorkerData> bwData, As
info.changeFeedStartVersion = tr.getCommittedVersion();
}
TraceEvent("GranuleOpen", bwData->id).detail("Granule", req.keyRange);
TraceEvent openEv("GranuleOpen", bwData->id);
openEv.detail("GranuleID", info.granuleID)
.detail("Granule", req.keyRange)
.detail("Epoch", req.managerEpoch)
.detail("Seqno", req.managerSeqno)
.detail("CFStartVersion", info.changeFeedStartVersion)
.detail("PreviousDurableVersion", info.previousDurableVersion);
if (info.parentGranule.present()) {
openEv.detail("ParentGranuleID", info.parentGranule.get().second);
}
return info;
} catch (Error& e) {
@ -2894,6 +2911,7 @@ ACTOR Future<Void> handleRangeRevoke(Reference<BlobWorkerData> bwData, RevokeBlo
ACTOR Future<Void> registerBlobWorker(Reference<BlobWorkerData> bwData, BlobWorkerInterface interf) {
state Reference<ReadYourWritesTransaction> tr = makeReference<ReadYourWritesTransaction>(bwData->db);
TraceEvent("BlobWorkerRegister", bwData->id);
loop {
tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS);
tr->setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE);
@ -2914,6 +2932,7 @@ ACTOR Future<Void> registerBlobWorker(Reference<BlobWorkerData> bwData, BlobWork
if (BW_DEBUG) {
fmt::print("Registered blob worker {}\n", interf.id().toString());
}
TraceEvent("BlobWorkerRegistered", bwData->id);
return Void();
} catch (Error& e) {
if (BW_DEBUG) {
@ -3021,7 +3040,7 @@ ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
if (BW_DEBUG) {
fmt::print("BW constructing backup container from {0}\n", SERVER_KNOBS->BG_URL);
}
self->bstore = BackupContainerFileSystem::openContainerFS(SERVER_KNOBS->BG_URL);
self->bstore = BackupContainerFileSystem::openContainerFS(SERVER_KNOBS->BG_URL, {}, {});
if (BW_DEBUG) {
printf("BW constructed backup container\n");
}

View File

@ -7,6 +7,8 @@ set(FDBSERVER_SRCS
BackupWorker.actor.cpp
BlobGranuleServerCommon.actor.cpp
BlobGranuleServerCommon.actor.h
BlobGranuleValidation.actor.cpp
BlobGranuleValidation.actor.h
BlobManager.actor.cpp
BlobManagerInterface.h
BlobWorker.actor.cpp
@ -98,6 +100,8 @@ set(FDBSERVER_SRCS
Ratekeeper.h
RatekeeperInterface.h
RecoveryState.h
RemoteIKeyValueStore.actor.h
RemoteIKeyValueStore.actor.cpp
ResolutionBalancer.actor.cpp
ResolutionBalancer.actor.h
Resolver.actor.cpp

View File

@ -1383,7 +1383,10 @@ public:
bool foundSSToRemove = false;
for (auto& server : self->server_info) {
if (!server.second->isCorrectStoreType(self->configuration.storageServerStoreType)) {
// If this server isn't the right storage type and its wrong-type trigger has not yet been set
// then set it if we're in aggressive mode and log its presence either way.
if (!server.second->isCorrectStoreType(self->configuration.storageServerStoreType) &&
!server.second->wrongStoreTypeToRemove.get()) {
// Server may be removed due to failure while the wrongStoreTypeToRemove is sent to the
// storageServerTracker. This race may cause the server to be removed before react to
// wrongStoreTypeToRemove
@ -1396,12 +1399,16 @@ public:
TraceEvent("WrongStoreTypeRemover", self->distributorId)
.detail("Server", server.first)
.detail("StoreType", server.second->getStoreType())
.detail("ConfiguredStoreType", self->configuration.storageServerStoreType);
break;
.detail("ConfiguredStoreType", self->configuration.storageServerStoreType)
.detail("RemovingNow",
self->configuration.storageMigrationType == StorageMigrationType::AGGRESSIVE);
}
}
if (!foundSSToRemove) {
// Stop if no incorrect storage types were found, or if we're not in aggressive mode and can't act on any
// found. Aggressive mode is checked at this location so that in non-aggressive mode the loop will execute
// once and log any incorrect storage types found.
if (!foundSSToRemove || self->configuration.storageMigrationType != StorageMigrationType::AGGRESSIVE) {
break;
}
}

View File

@ -296,7 +296,7 @@ Future<Void> StorageWiggler::restoreStats() {
return map(readFuture, assignFunc);
}
Future<Void> StorageWiggler::startWiggle() {
metrics.last_wiggle_start = timer_int();
metrics.last_wiggle_start = StorageMetadataType::currentTime();
if (shouldStartNewRound()) {
metrics.last_round_start = metrics.last_wiggle_start;
}
@ -304,7 +304,7 @@ Future<Void> StorageWiggler::startWiggle() {
}
Future<Void> StorageWiggler::finishWiggle() {
metrics.last_wiggle_finish = timer_int();
metrics.last_wiggle_finish = StorageMetadataType::currentTime();
metrics.finished_wiggle += 1;
auto duration = metrics.last_wiggle_finish - metrics.last_wiggle_start;
metrics.smoothed_wiggle_duration.setTotal((double)duration);

View File

@ -340,15 +340,17 @@ struct StorageWiggleMetrics {
// round statistics
// One StorageServer wiggle round is considered 'complete', when all StorageServers with creationTime < T are
// wiggled
uint64_t last_round_start = 0; // wall timer: timer_int()
uint64_t last_round_finish = 0;
// Start and finish are in epoch seconds
double last_round_start = 0;
double last_round_finish = 0;
TimerSmoother smoothed_round_duration;
int finished_round = 0; // finished round since storage wiggle is open
// step statistics
// 1 wiggle step as 1 storage server is wiggled in the current round
uint64_t last_wiggle_start = 0; // wall timer: timer_int()
uint64_t last_wiggle_finish = 0;
// Start and finish are in epoch seconds
double last_wiggle_start = 0;
double last_wiggle_finish = 0;
TimerSmoother smoothed_wiggle_duration;
int finished_wiggle = 0; // finished step since storage wiggle is open
@ -406,15 +408,15 @@ struct StorageWiggleMetrics {
StatusObject toJSON() const {
StatusObject result;
result["last_round_start_datetime"] = timerIntToGmt(last_round_start);
result["last_round_finish_datetime"] = timerIntToGmt(last_round_finish);
result["last_round_start_datetime"] = epochsToGMTString(last_round_start);
result["last_round_finish_datetime"] = epochsToGMTString(last_round_finish);
result["last_round_start_timestamp"] = last_round_start;
result["last_round_finish_timestamp"] = last_round_finish;
result["smoothed_round_seconds"] = smoothed_round_duration.smoothTotal();
result["finished_round"] = finished_round;
result["last_wiggle_start_datetime"] = timerIntToGmt(last_wiggle_start);
result["last_wiggle_finish_datetime"] = timerIntToGmt(last_wiggle_finish);
result["last_wiggle_start_datetime"] = epochsToGMTString(last_wiggle_start);
result["last_wiggle_finish_datetime"] = epochsToGMTString(last_wiggle_finish);
result["last_wiggle_start_timestamp"] = last_wiggle_start;
result["last_wiggle_finish_timestamp"] = last_wiggle_finish;
result["smoothed_wiggle_seconds"] = smoothed_wiggle_duration.smoothTotal();

View File

@ -1426,6 +1426,7 @@ ACTOR Future<Void> dataDistributionRelocator(DDQueueData* self, RelocateData rd,
self->noErrorActors.add(
trigger([destinationRef, readLoad]() mutable { destinationRef.addDataInFlightToTeam(-readLoad); },
delay(SERVER_KNOBS->STORAGE_METRICS_AVERAGE_INTERVAL, TaskPriority::DataDistributionLow)));
rd.completeDests.clear();
wait(delay(SERVER_KNOBS->RETRY_RELOCATESHARD_DELAY, TaskPriority::DataDistributionLaunch));
}
}

View File

@ -1077,7 +1077,7 @@ public:
Node* node(DeltaTree2* tree) const { return tree->nodeAt(nodeOffset); }
std::string toString() {
std::string toString() const {
return format("DecodedNode{nodeOffset=%d leftChildIndex=%d rightChildIndex=%d leftParentIndex=%d "
"rightParentIndex=%d}",
(int)nodeOffset,
@ -1155,6 +1155,19 @@ public:
arena = a;
updateUsedMemory();
}
std::string toString() const {
std::string s = format("DecodeCache{%p\n", this);
s += format("upperBound %s\n", upperBound.toString().c_str());
s += format("lowerBound %s\n", lowerBound.toString().c_str());
s += format("arenaSize %d\n", arena.getSize());
s += format("decodedNodes %d {\n", decodedNodes.size());
for (auto const& n : decodedNodes) {
s += format(" %s\n", n.toString().c_str());
}
s += format("}}\n");
return s;
}
};
// Cursor provides a way to seek into a DeltaTree and iterate over its contents
@ -1686,7 +1699,7 @@ public:
int count = end - begin;
numItems = count;
nodeBytesDeleted = 0;
initialHeight = (uint8_t)log2(count) + 1;
initialHeight = count == 0 ? 0 : (uint8_t)log2(count) + 1;
maxHeight = 0;
// The boundary leading to the new page acts as the last time we branched right

View File

@ -18,17 +18,30 @@
* limitations under the License.
*/
#include "flow/TLSConfig.actor.h"
#include "flow/Trace.h"
#include "flow/Platform.h"
#include "flow/flow.h"
#include "flow/genericactors.actor.h"
#include "flow/network.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/Net2FileSystem.h"
#include "fdbrpc/simulator.h"
#include "fdbclient/WellKnownEndpoints.h"
#include "fdbclient/versions.h"
#include "fdbserver/CoroFlow.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/Knobs.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#if !defined(_WIN32) && !defined(__APPLE__) && !defined(__INTEL_COMPILER)
#define BOOST_SYSTEM_NO_LIB
#define BOOST_DATE_TIME_NO_LIB
#define BOOST_REGEX_NO_LIB
#include <boost/process.hpp>
#endif
#include "fdbserver/FDBExecHelper.actor.h"
#include "flow/Trace.h"
#include "flow/flow.h"
#include "fdbclient/versions.h"
#include "fdbserver/Knobs.h"
#include <boost/algorithm/string.hpp>
#include "flow/actorcompiler.h" // This must be the last #include.
ExecCmdValueString::ExecCmdValueString(StringRef pCmdValueString) {
@ -90,12 +103,138 @@ void ExecCmdValueString::dbgPrint() const {
return;
}
ACTOR void destoryChildProcess(Future<Void> parentSSClosed, ISimulator::ProcessInfo* childInfo, std::string message) {
// This code path should be bug free
wait(parentSSClosed);
TraceEvent(SevDebug, message.c_str()).log();
// This one is root cause for most failures, make sure it's okay to destory
g_pSimulator->destroyProcess(childInfo);
// Explicitly reset the connection with the child process in case re-spawn very quickly
FlowTransport::transport().resetConnection(childInfo->address);
}
ACTOR Future<int> spawnSimulated(std::vector<std::string> paramList,
double maxWaitTime,
bool isSync,
double maxSimDelayTime,
IClosable* parent) {
state ISimulator::ProcessInfo* self = g_pSimulator->getCurrentProcess();
state ISimulator::ProcessInfo* child;
state std::string role;
state std::string addr;
state std::string flowProcessName;
state Endpoint parentProcessEndpoint;
state int i = 0;
// fdbserver -r flowprocess --process-name ikvs --process-endpoint ip:port,token,id
for (; i < paramList.size(); i++) {
if (paramList.size() > i + 1) {
// temporary args parser that only supports the flowprocess role
if (paramList[i] == "-r") {
role = paramList[i + 1];
} else if (paramList[i] == "-p" || paramList[i] == "--public_address") {
addr = paramList[i + 1];
} else if (paramList[i] == "--process-name") {
flowProcessName = paramList[i + 1];
} else if (paramList[i] == "--process-endpoint") {
state std::vector<std::string> addressArray;
boost::split(addressArray, paramList[i + 1], [](char c) { return c == ','; });
if (addressArray.size() != 3) {
std::cerr << "Invalid argument, expected 3 elements in --process-endpoint got "
<< addressArray.size() << std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
try {
auto addr = NetworkAddress::parse(addressArray[0]);
uint64_t fst = std::stoul(addressArray[1]);
uint64_t snd = std::stoul(addressArray[2]);
UID token(fst, snd);
NetworkAddressList l;
l.address = addr;
parentProcessEndpoint = Endpoint(l, token);
} catch (Error& e) {
std::cerr << "Could not parse network address " << addressArray[0] << std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
}
}
}
state int result = 0;
child = g_pSimulator->newProcess("remote flow process",
self->address.ip,
0,
self->address.isTLS(),
self->addresses.secondaryAddress.present() ? 2 : 1,
self->locality,
ProcessClass(ProcessClass::UnsetClass, ProcessClass::AutoSource),
self->dataFolder,
self->coordinationFolder, // do we need to customize this coordination folder path?
self->protocolVersion);
wait(g_pSimulator->onProcess(child));
state Future<ISimulator::KillType> onShutdown = child->onShutdown();
state Future<ISimulator::KillType> parentShutdown = self->onShutdown();
state Future<Void> flowProcessF;
try {
TraceEvent(SevDebug, "SpawnedChildProcess")
.detail("Child", child->toString())
.detail("Parent", self->toString());
std::string role = "";
std::string addr = "";
for (int i = 0; i < paramList.size(); i++) {
if (paramList.size() > i + 1 && paramList[i] == "-r") {
role = paramList[i + 1];
}
}
if (role == "flowprocess" && !parentShutdown.isReady()) {
self->childs.push_back(child);
state Future<Void> parentSSClosed = parent->onClosed();
FlowTransport::createInstance(false, 1, WLTOKEN_RESERVED_COUNT);
FlowTransport::transport().bind(child->address, child->address);
Sim2FileSystem::newFileSystem();
ProcessFactory<KeyValueStoreProcess>(flowProcessName.c_str());
flowProcessF = runFlowProcess(flowProcessName, parentProcessEndpoint);
choose {
when(wait(flowProcessF)) {
TraceEvent(SevDebug, "ChildProcessKilled").log();
wait(g_pSimulator->onProcess(self));
TraceEvent(SevDebug, "BackOnParentProcess").detail("Result", std::to_string(result));
destoryChildProcess(parentSSClosed, child, "StorageServerReceivedClosedMessage");
}
when(wait(success(onShutdown))) {
ASSERT(false);
// In prod, we use prctl to bind parent and child processes to die together
// In simulation, we simply disable killing parent or child processes as we cannot use the same
// mechanism here
}
when(wait(success(parentShutdown))) {
ASSERT(false);
// Parent process is not killed, see above
}
}
} else {
ASSERT(false);
}
} catch (Error& e) {
TraceEvent(SevError, "RemoteIKVSDied").errorUnsuppressed(e);
result = -1;
}
return result;
}
#if defined(_WIN32) || defined(__APPLE__) || defined(__INTEL_COMPILER)
ACTOR Future<int> spawnProcess(std::string binPath,
std::vector<std::string> paramList,
double maxWaitTime,
bool isSync,
double maxSimDelayTime) {
double maxSimDelayTime,
IClosable* parent) {
if (g_network->isSimulated() && getExecPath() == binPath) {
int res = wait(spawnSimulated(paramList, maxWaitTime, isSync, maxSimDelayTime, parent));
return res;
}
wait(delay(0.0));
return 0;
}
@ -125,6 +264,9 @@ static auto fork_child(const std::string& path, std::vector<char*>& paramList) {
}
static void setupTraceWithOutput(TraceEvent& event, size_t bytesRead, char* outputBuffer) {
// get some errors printed for spawned process
std::cout << "Output bytesRead: " << bytesRead << std::endl;
std::cout << "output buffer: " << std::string(outputBuffer) << std::endl;
if (bytesRead == 0)
return;
ASSERT(bytesRead <= SERVER_KNOBS->MAX_FORKED_PROCESS_OUTPUT);
@ -139,7 +281,12 @@ ACTOR Future<int> spawnProcess(std::string path,
std::vector<std::string> args,
double maxWaitTime,
bool isSync,
double maxSimDelayTime) {
double maxSimDelayTime,
IClosable* parent) {
if (g_network->isSimulated() && getExecPath() == path) {
int res = wait(spawnSimulated(args, maxWaitTime, isSync, maxSimDelayTime, parent));
return res;
}
// for async calls in simulator, always delay by a deterministic amount of time and then
// do the call synchronously, otherwise the predictability of the simulator breaks
if (!isSync && g_network->isSimulated()) {
@ -182,7 +329,7 @@ ACTOR Future<int> spawnProcess(std::string path,
int flags = fcntl(readFD.get(), F_GETFL, 0);
fcntl(readFD.get(), F_SETFL, flags | O_NONBLOCK);
while (true) {
if (runTime > maxWaitTime) {
if (maxWaitTime >= 0 && runTime > maxWaitTime) {
// timing out
TraceEvent(SevWarnAlways, "SpawnProcessFailure")
@ -203,7 +350,6 @@ ACTOR Future<int> spawnProcess(std::string path,
break;
bytesRead += bytes;
}
if (err < 0) {
TraceEvent event(SevWarnAlways, "SpawnProcessFailure");
setupTraceWithOutput(event, bytesRead, outputBuffer);

View File

@ -63,16 +63,19 @@ private: // data
StringRef binaryPath;
};
class IClosable; // Forward declaration
// FIXME: move this function to a common location
// spawns a process pointed by `binPath` and the arguments provided at `paramList`,
// if the process spawned takes more than `maxWaitTime` then it will be killed
// if isSync is set to true then the process will be synchronously executed
// if async and in simulator then delay spawning the process to max of maxSimDelayTime
// if the process spawned takes more than `maxWaitTime` then it will be killed, if `maxWaitTime` < 0, then there won't
// be timeout if isSync is set to true then the process will be synchronously executed if async and in simulator then
// delay spawning the process to max of maxSimDelayTime
ACTOR Future<int> spawnProcess(std::string binPath,
std::vector<std::string> paramList,
double maxWaitTime,
bool isSync,
double maxSimDelayTime);
double maxSimDelayTime,
IClosable* parent = nullptr);
// helper to run all the work related to running the exec command
ACTOR Future<int> execHelper(ExecCmdValueString* execArg, UID snapUID, std::string folder, std::string role);

View File

@ -159,12 +159,23 @@ extern IKeyValueStore* keyValueStoreLogSystem(class IDiskQueue* queue,
bool replaceContent,
bool exactRecovery);
extern IKeyValueStore* openRemoteKVStore(KeyValueStoreType storeType,
std::string const& filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums = false,
bool checkIntegrity = false);
inline IKeyValueStore* openKVStore(KeyValueStoreType storeType,
std::string const& filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums = false,
bool checkIntegrity = false) {
bool checkIntegrity = false,
bool openRemotely = false) {
if (openRemotely) {
return openRemoteKVStore(storeType, filename, logID, memoryLimit, checkChecksums, checkIntegrity);
}
switch (storeType) {
case KeyValueStoreType::SSD_BTREE_V1:
return keyValueStoreSQLite(filename, logID, KeyValueStoreType::SSD_BTREE_V1, false, checkIntegrity);

View File

@ -147,6 +147,7 @@ private:
};
using DB = rocksdb::DB*;
using CF = rocksdb::ColumnFamilyHandle*;
std::shared_ptr<rocksdb::Cache> rocksdb_block_cache = nullptr;
#define PERSIST_PREFIX "\xff\xff"
const KeyRef persistVersion = LiteralStringRef(PERSIST_PREFIX "Version");
@ -288,7 +289,10 @@ rocksdb::ColumnFamilyOptions getCFOptions() {
}
if (SERVER_KNOBS->ROCKSDB_BLOCK_CACHE_SIZE > 0) {
bbOpts.block_cache = rocksdb::NewLRUCache(SERVER_KNOBS->ROCKSDB_BLOCK_CACHE_SIZE);
if (rocksdb_block_cache == nullptr) {
rocksdb_block_cache = rocksdb::NewLRUCache(SERVER_KNOBS->ROCKSDB_BLOCK_CACHE_SIZE);
}
bbOpts.block_cache = rocksdb_block_cache;
}
options.table_factory.reset(rocksdb::NewBlockBasedTableFactory(bbOpts));

View File

@ -121,7 +121,8 @@ public:
newServers[serverId] = ssi;
if (oldServers.count(serverId)) {
if (ssi.getValue.getEndpoint() != oldServers[serverId].getValue.getEndpoint()) {
if (ssi.getValue.getEndpoint() != oldServers[serverId].getValue.getEndpoint() ||
ssi.isAcceptingRequests() != oldServers[serverId].isAcceptingRequests()) {
serverChanges.send(std::make_pair(serverId, Optional<StorageServerInterface>(ssi)));
}
oldServers.erase(serverId);
@ -158,6 +159,7 @@ public:
StorageQueuingMetricsRequest(), 0, 0)); // SOMEDAY: or tryGetReply?
if (reply.present()) {
myQueueInfo->value.update(reply.get(), self->smoothTotalDurableBytes);
myQueueInfo->value.acceptingRequests = ssi.isAcceptingRequests();
} else {
if (myQueueInfo->value.valid) {
TraceEvent("RkStorageServerDidNotRespond", self->id).detail("StorageServer", ssi.id());
@ -487,7 +489,7 @@ void Ratekeeper::updateRate(RatekeeperLimits* limits) {
// Look at each storage server's write queue and local rate, compute and store the desired rate ratio
for (auto i = storageQueueInfo.begin(); i != storageQueueInfo.end(); ++i) {
auto const& ss = i->value;
if (!ss.valid || (remoteDC.present() && ss.locality.dcId() == remoteDC))
if (!ss.valid || !ss.acceptingRequests || (remoteDC.present() && ss.locality.dcId() == remoteDC))
continue;
++sscount;
@ -941,7 +943,7 @@ ACTOR Future<Void> ratekeeper(RatekeeperInterface rkInterf, Reference<AsyncVar<S
StorageQueueInfo::StorageQueueInfo(UID id, LocalityData locality)
: busiestWriteTagEventHolder(makeReference<EventCacheHolder>(id.toString() + "/BusiestWriteTag")), valid(false),
id(id), locality(locality), smoothDurableBytes(SERVER_KNOBS->SMOOTHING_AMOUNT),
id(id), locality(locality), acceptingRequests(false), smoothDurableBytes(SERVER_KNOBS->SMOOTHING_AMOUNT),
smoothInputBytes(SERVER_KNOBS->SMOOTHING_AMOUNT), verySmoothDurableBytes(SERVER_KNOBS->SLOW_SMOOTHING_AMOUNT),
smoothDurableVersion(SERVER_KNOBS->SMOOTHING_AMOUNT), smoothLatestVersion(SERVER_KNOBS->SMOOTHING_AMOUNT),
smoothFreeSpace(SERVER_KNOBS->SMOOTHING_AMOUNT), smoothTotalSpace(SERVER_KNOBS->SMOOTHING_AMOUNT),

View File

@ -59,6 +59,7 @@ public:
UID id;
LocalityData locality;
StorageQueuingMetricsReply lastReply;
bool acceptingRequests;
Smoother smoothDurableBytes, smoothInputBytes, verySmoothDurableBytes;
Smoother smoothDurableVersion, smoothLatestVersion;
Smoother smoothFreeSpace;

View File

@ -0,0 +1,246 @@
/*
* RemoteIKeyValueStore.actor.cpp
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "flow/ActorCollection.h"
#include "flow/Error.h"
#include "flow/Platform.h"
#include "flow/Trace.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/fdbrpc.h"
#include "fdbclient/FDBTypes.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/Knobs.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#include "flow/actorcompiler.h" // This must be the last #include.
StringRef KeyValueStoreProcess::_name = "KeyValueStoreProcess"_sr;
// A guard for guaranteed killing of machine after runIKVS returns
struct AfterReturn {
IKeyValueStore* kvStore;
UID id;
AfterReturn() : kvStore(nullptr) {}
AfterReturn(IKeyValueStore* store, UID& uid) : kvStore(store), id(uid) {}
~AfterReturn() {
TraceEvent(SevDebug, "RemoteKVStoreAfterReturn")
.detail("Valid", kvStore != nullptr ? "True" : "False")
.detail("UID", id)
.log();
if (kvStore != nullptr) {
kvStore->close();
}
}
// called when we already explicitly closed the kv store
void invalidate() { kvStore = nullptr; }
};
ACTOR void sendCommitReply(IKVSCommitRequest commitReq, IKeyValueStore* kvStore, Future<Void> onClosed) {
try {
choose {
when(wait(onClosed)) { commitReq.reply.sendError(remote_kvs_cancelled()); }
when(wait(kvStore->commit(commitReq.sequential))) {
StorageBytes storageBytes = kvStore->getStorageBytes();
commitReq.reply.send(IKVSCommitReply(storageBytes));
}
}
} catch (Error& e) {
TraceEvent(SevDebug, "RemoteKVSCommitReplyError").errorUnsuppressed(e);
commitReq.reply.sendError(e.code() == error_code_actor_cancelled ? remote_kvs_cancelled() : e);
}
}
ACTOR template <class T>
Future<Void> cancellableForwardPromise(ReplyPromise<T> output, Future<T> input) {
try {
T value = wait(input);
output.send(value);
} catch (Error& e) {
TraceEvent(SevDebug, "CancellableForwardPromiseError").errorUnsuppressed(e).backtrace();
output.sendError(e.code() == error_code_actor_cancelled ? remote_kvs_cancelled() : e);
}
return Void();
}
ACTOR Future<Void> runIKVS(OpenKVStoreRequest openReq, IKVSInterface ikvsInterface) {
state IKeyValueStore* kvStore = openKVStore(openReq.storeType,
openReq.filename,
openReq.logID,
openReq.memoryLimit,
openReq.checkChecksums,
openReq.checkIntegrity);
state UID kvsId(ikvsInterface.id());
state ActorCollection actors(false);
state AfterReturn guard(kvStore, kvsId);
state Promise<Void> onClosed;
TraceEvent(SevDebug, "RemoteKVStoreInitializing").detail("UID", kvsId);
wait(kvStore->init());
openReq.reply.send(ikvsInterface);
TraceEvent(SevInfo, "RemoteKVStoreInitialized").detail("IKVSInterfaceUID", kvsId);
loop {
try {
choose {
when(IKVSGetValueRequest getReq = waitNext(ikvsInterface.getValue.getFuture())) {
actors.add(cancellableForwardPromise(getReq.reply,
kvStore->readValue(getReq.key, getReq.type, getReq.debugID)));
}
when(IKVSSetRequest req = waitNext(ikvsInterface.set.getFuture())) { kvStore->set(req.keyValue); }
when(IKVSClearRequest req = waitNext(ikvsInterface.clear.getFuture())) { kvStore->clear(req.range); }
when(IKVSCommitRequest commitReq = waitNext(ikvsInterface.commit.getFuture())) {
sendCommitReply(commitReq, kvStore, onClosed.getFuture());
}
when(IKVSReadValuePrefixRequest readPrefixReq = waitNext(ikvsInterface.readValuePrefix.getFuture())) {
actors.add(cancellableForwardPromise(
readPrefixReq.reply,
kvStore->readValuePrefix(
readPrefixReq.key, readPrefixReq.maxLength, readPrefixReq.type, readPrefixReq.debugID)));
}
when(IKVSReadRangeRequest readRangeReq = waitNext(ikvsInterface.readRange.getFuture())) {
actors.add(cancellableForwardPromise(
readRangeReq.reply,
fmap(
[](const RangeResult& result) { return IKVSReadRangeReply(result); },
kvStore->readRange(
readRangeReq.keys, readRangeReq.rowLimit, readRangeReq.byteLimit, readRangeReq.type))));
}
when(IKVSGetStorageByteRequest req = waitNext(ikvsInterface.getStorageBytes.getFuture())) {
StorageBytes storageBytes = kvStore->getStorageBytes();
req.reply.send(storageBytes);
}
when(IKVSGetErrorRequest getFutureReq = waitNext(ikvsInterface.getError.getFuture())) {
actors.add(cancellableForwardPromise(getFutureReq.reply, kvStore->getError()));
}
when(IKVSOnClosedRequest onClosedReq = waitNext(ikvsInterface.onClosed.getFuture())) {
// onClosed request is not cancelled even this actor is cancelled
forwardPromise(onClosedReq.reply, kvStore->onClosed());
}
when(IKVSDisposeRequest disposeReq = waitNext(ikvsInterface.dispose.getFuture())) {
TraceEvent(SevDebug, "RemoteIKVSDisposeReceivedRequest").detail("UID", kvsId);
kvStore->dispose();
guard.invalidate();
onClosed.send(Void());
return Void();
}
when(IKVSCloseRequest closeReq = waitNext(ikvsInterface.close.getFuture())) {
TraceEvent(SevDebug, "RemoteIKVSCloseReceivedRequest").detail("UID", kvsId);
kvStore->close();
guard.invalidate();
onClosed.send(Void());
return Void();
}
}
} catch (Error& e) {
if (e.code() == error_code_actor_cancelled) {
TraceEvent(SevInfo, "RemoteKVStoreCancelled").detail("UID", kvsId).backtrace();
onClosed.send(Void());
return Void();
} else {
TraceEvent(SevError, "RemoteKVStoreError").error(e).detail("UID", kvsId).backtrace();
throw;
}
}
}
}
ACTOR static Future<int> flowProcessRunner(RemoteIKeyValueStore* self, Promise<Void> ready) {
state FlowProcessInterface processInterface;
state Future<int> process;
auto path = abspath(getExecPath());
auto endpoint = processInterface.registerProcess.getEndpoint();
auto address = endpoint.addresses.address.toString();
auto token = endpoint.token;
// port 0 means we will find a random available port number for it
std::string flowProcessAddr = g_network->getLocalAddress().ip.toString().append(":0");
std::vector<std::string> args = { "bin/fdbserver",
"-r",
"flowprocess",
"-C",
SERVER_KNOBS->CONN_FILE,
"--logdir",
SERVER_KNOBS->LOG_DIRECTORY,
"-p",
flowProcessAddr,
"--process-name",
KeyValueStoreProcess::_name.toString(),
"--process-endpoint",
format("%s,%lu,%lu", address.c_str(), token.first(), token.second()) };
// For remote IKV store, we need to make sure the shutdown signal is sent back until we can destroy it in the
// simulation
process = spawnProcess(path, args, -1.0, false, 0.01 /*not used*/, self);
choose {
when(FlowProcessRegistrationRequest req = waitNext(processInterface.registerProcess.getFuture())) {
self->consumeInterface(req.flowProcessInterface);
ready.send(Void());
}
when(int res = wait(process)) {
// 0 means process normally shut down; non-zero means errors
// process should not shut down normally before not ready
ASSERT(res);
return res;
}
}
int res = wait(process);
return res;
}
ACTOR static Future<Void> initializeRemoteKVStore(RemoteIKeyValueStore* self, OpenKVStoreRequest openKVSReq) {
TraceEvent(SevInfo, "WaitingOnFlowProcess").detail("StoreType", openKVSReq.storeType).log();
Promise<Void> ready;
self->returnCode = flowProcessRunner(self, ready);
wait(ready.getFuture());
IKVSInterface ikvsInterface = wait(self->kvsProcess.openKVStore.getReply(openKVSReq));
TraceEvent(SevInfo, "IKVSInterfaceReceived").detail("UID", ikvsInterface.id());
self->interf = ikvsInterface;
self->interf.storeType = openKVSReq.storeType;
return Void();
}
IKeyValueStore* openRemoteKVStore(KeyValueStoreType storeType,
std::string const& filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums,
bool checkIntegrity) {
RemoteIKeyValueStore* self = new RemoteIKeyValueStore();
self->initialized = initializeRemoteKVStore(
self, OpenKVStoreRequest(storeType, filename, logID, memoryLimit, checkChecksums, checkIntegrity));
return self;
}
ACTOR static Future<Void> delayFlowProcessRunAction(FlowProcess* self, double time) {
wait(delay(time));
wait(self->run());
return Void();
}
Future<Void> runFlowProcess(std::string const& name, Endpoint endpoint) {
TraceEvent(SevInfo, "RunFlowProcessStart").log();
FlowProcess* self = IProcessFactory::create(name.c_str());
self->registerEndpoint(endpoint);
RequestStream<FlowProcessRegistrationRequest> registerProcess(endpoint);
FlowProcessRegistrationRequest req;
req.flowProcessInterface = self->serializedInterface();
registerProcess.send(req);
TraceEvent(SevDebug, "FlowProcessInitFinished").log();
return delayFlowProcessRunAction(self, g_network->isSimulated() ? 0 : SERVER_KNOBS->REMOTE_KV_STORE_INIT_DELAY);
}

View File

@ -0,0 +1,504 @@
/*
* RemoteIKeyValueStore.actor.h
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2022 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#if defined(NO_INTELLISENSE) && !defined(FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_G_H)
#define FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_G_H
#include "fdbserver/RemoteIKeyValueStore.actor.g.h"
#elif !defined(FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_H)
#define FDBSERVER_REMOTE_IKEYVALUESTORE_ACTOR_H
#include "flow/ActorCollection.h"
#include "flow/IRandom.h"
#include "flow/Knobs.h"
#include "flow/Trace.h"
#include "flow/flow.h"
#include "flow/network.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/FlowTransport.h"
#include "fdbrpc/fdbrpc.h"
#include "fdbclient/FDBTypes.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/IKeyValueStore.h"
#include "fdbserver/Knobs.h"
#include "flow/actorcompiler.h" // This must be the last #include.
struct IKVSCommitReply {
constexpr static FileIdentifier file_identifier = 3958189;
StorageBytes storeBytes;
IKVSCommitReply() : storeBytes(0, 0, 0, 0) {}
IKVSCommitReply(const StorageBytes& sb) : storeBytes(sb) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, storeBytes);
}
};
struct RemoteKVSProcessInterface {
constexpr static FileIdentifier file_identifier = 3491838;
RequestStream<struct GetRemoteKVSProcessInterfaceRequest> getProcessInterface;
RequestStream<struct OpenKVStoreRequest> openKVStore;
UID uniqueID = deterministicRandom()->randomUniqueID();
UID id() const { return uniqueID; }
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, getProcessInterface, openKVStore);
}
};
struct IKVSInterface {
constexpr static FileIdentifier file_identifier = 4929113;
RequestStream<struct IKVSGetValueRequest> getValue;
RequestStream<struct IKVSSetRequest> set;
RequestStream<struct IKVSClearRequest> clear;
RequestStream<struct IKVSCommitRequest> commit;
RequestStream<struct IKVSReadValuePrefixRequest> readValuePrefix;
RequestStream<struct IKVSReadRangeRequest> readRange;
RequestStream<struct IKVSGetStorageByteRequest> getStorageBytes;
RequestStream<struct IKVSGetErrorRequest> getError;
RequestStream<struct IKVSOnClosedRequest> onClosed;
RequestStream<struct IKVSDisposeRequest> dispose;
RequestStream<struct IKVSCloseRequest> close;
UID uniqueID;
UID id() const { return uniqueID; }
KeyValueStoreType storeType;
KeyValueStoreType type() const { return storeType; }
IKVSInterface() {}
IKVSInterface(KeyValueStoreType type) : uniqueID(deterministicRandom()->randomUniqueID()), storeType(type) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar,
getValue,
set,
clear,
commit,
readValuePrefix,
readRange,
getStorageBytes,
getError,
onClosed,
dispose,
close,
uniqueID);
}
};
struct GetRemoteKVSProcessInterfaceRequest {
constexpr static FileIdentifier file_identifier = 8382983;
ReplyPromise<struct RemoteKVSProcessInterface> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct OpenKVStoreRequest {
constexpr static FileIdentifier file_identifier = 5918682;
KeyValueStoreType storeType;
std::string filename;
UID logID;
int64_t memoryLimit;
bool checkChecksums;
bool checkIntegrity;
ReplyPromise<struct IKVSInterface> reply;
OpenKVStoreRequest(){};
OpenKVStoreRequest(KeyValueStoreType storeType,
std::string filename,
UID logID,
int64_t memoryLimit,
bool checkChecksums = false,
bool checkIntegrity = false)
: storeType(storeType), filename(filename), logID(logID), memoryLimit(memoryLimit),
checkChecksums(checkChecksums), checkIntegrity(checkIntegrity) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, storeType, filename, logID, memoryLimit, checkChecksums, checkIntegrity, reply);
}
};
struct IKVSGetValueRequest {
constexpr static FileIdentifier file_identifier = 1029439;
KeyRef key;
IKeyValueStore::ReadType type;
Optional<UID> debugID = Optional<UID>();
ReplyPromise<Optional<Value>> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, key, type, debugID, reply);
}
};
struct IKVSSetRequest {
constexpr static FileIdentifier file_identifier = 7283948;
KeyValueRef keyValue;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, keyValue, reply);
}
};
struct IKVSClearRequest {
constexpr static FileIdentifier file_identifier = 2838575;
KeyRangeRef range;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, range, reply);
}
};
struct IKVSCommitRequest {
constexpr static FileIdentifier file_identifier = 2985129;
bool sequential;
ReplyPromise<IKVSCommitReply> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, sequential, reply);
}
};
struct IKVSReadValuePrefixRequest {
constexpr static FileIdentifier file_identifier = 1928374;
KeyRef key;
int maxLength;
IKeyValueStore::ReadType type;
Optional<UID> debugID = Optional<UID>();
ReplyPromise<Optional<Value>> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, key, maxLength, type, debugID, reply);
}
};
// Use this instead of RangeResult as reply for better serialization performance
struct IKVSReadRangeReply {
constexpr static FileIdentifier file_identifier = 6682449;
Arena arena;
VectorRef<KeyValueRef, VecSerStrategy::String> data;
bool more;
Optional<KeyRef> readThrough;
bool readToBegin;
bool readThroughEnd;
IKVSReadRangeReply() = default;
explicit IKVSReadRangeReply(const RangeResult& res)
: arena(res.arena()), data(static_cast<const VectorRef<KeyValueRef>&>(res)), more(res.more),
readThrough(res.readThrough), readToBegin(res.readToBegin), readThroughEnd(res.readThroughEnd) {}
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, data, more, readThrough, readToBegin, readThroughEnd, arena);
}
RangeResult toRangeResult() const {
RangeResult r(RangeResultRef(data, more, readThrough), arena);
r.readToBegin = readToBegin;
r.readThroughEnd = readThroughEnd;
return r;
}
};
struct IKVSReadRangeRequest {
constexpr static FileIdentifier file_identifier = 5918394;
KeyRangeRef keys;
int rowLimit;
int byteLimit;
IKeyValueStore::ReadType type;
ReplyPromise<IKVSReadRangeReply> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, keys, rowLimit, byteLimit, type, reply);
}
};
struct IKVSGetStorageByteRequest {
constexpr static FileIdentifier file_identifier = 3512344;
ReplyPromise<StorageBytes> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct IKVSGetErrorRequest {
constexpr static FileIdentifier file_identifier = 3942891;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct IKVSOnClosedRequest {
constexpr static FileIdentifier file_identifier = 1923894;
ReplyPromise<Void> reply;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, reply);
}
};
struct IKVSDisposeRequest {
constexpr static FileIdentifier file_identifier = 1235952;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar);
}
};
struct IKVSCloseRequest {
constexpr static FileIdentifier file_identifier = 13859172;
template <class Ar>
void serialize(Ar& ar) {
serializer(ar);
}
};
ACTOR Future<Void> runIKVS(OpenKVStoreRequest openReq, IKVSInterface ikvsInterface);
struct KeyValueStoreProcess : FlowProcess {
RemoteKVSProcessInterface kvsIf;
Standalone<StringRef> serializedIf;
Endpoint ssProcess; // endpoint for the storage process
RequestStream<FlowProcessRegistrationRequest> ssRequestStream;
KeyValueStoreProcess() {
TraceEvent(SevDebug, "InitKeyValueStoreProcess").log();
ObjectWriter writer(IncludeVersion());
writer.serialize(kvsIf);
serializedIf = writer.toString();
}
void registerEndpoint(Endpoint p) override {
ssProcess = p;
ssRequestStream = RequestStream<FlowProcessRegistrationRequest>(p);
}
StringRef name() const override { return _name; }
StringRef serializedInterface() const override { return serializedIf; }
ACTOR static Future<Void> _run(KeyValueStoreProcess* self) {
state ActorCollection actors(true);
TraceEvent("WaitingForOpenKVStoreRequest").log();
loop {
choose {
when(OpenKVStoreRequest req = waitNext(self->kvsIf.openKVStore.getFuture())) {
TraceEvent("OpenKVStoreRequestReceived").log();
IKVSInterface interf;
actors.add(runIKVS(req, interf));
}
when(ErrorOr<Void> e = wait(errorOr(actors.getResult()))) {
if (e.isError()) {
TraceEvent("KeyValueStoreProcessRunActorError").errorUnsuppressed(e.getError());
throw e.getError();
} else {
TraceEvent("KeyValueStoreProcessFinished").log();
return e.get();
}
}
}
}
}
Future<Void> run() override { return _run(this); }
static StringRef _name;
};
struct RemoteIKeyValueStore : public IKeyValueStore {
RemoteKVSProcessInterface kvsProcess;
IKVSInterface interf;
Future<Void> initialized;
Future<int> returnCode;
StorageBytes storageBytes;
RemoteIKeyValueStore() : storageBytes(0, 0, 0, 0) {}
Future<Void> init() override {
TraceEvent(SevInfo, "RemoteIKeyValueStoreInit").log();
return initialized;
}
Future<Void> getError() const override { return getErrorImpl(this, returnCode); }
Future<Void> onClosed() const override { return onCloseImpl(this); }
void dispose() override {
TraceEvent(SevDebug, "RemoteIKVSDisposeRequest").backtrace();
interf.dispose.send(IKVSDisposeRequest{});
// hold the future to not cancel the spawned process
uncancellable(returnCode);
delete this;
}
void close() override {
TraceEvent(SevDebug, "RemoteIKVSCloseRequest").backtrace();
interf.close.send(IKVSCloseRequest{});
// hold the future to not cancel the spawned process
uncancellable(returnCode);
delete this;
}
KeyValueStoreType getType() const override { return interf.type(); }
void set(KeyValueRef keyValue, const Arena* arena = nullptr) override {
interf.set.send(IKVSSetRequest{ keyValue, ReplyPromise<Void>() });
}
void clear(KeyRangeRef range, const Arena* arena = nullptr) override {
interf.clear.send(IKVSClearRequest{ range, ReplyPromise<Void>() });
}
Future<Void> commit(bool sequential = false) override {
Future<IKVSCommitReply> commitReply =
interf.commit.getReply(IKVSCommitRequest{ sequential, ReplyPromise<IKVSCommitReply>() });
return commitAndGetStorageBytes(this, commitReply);
}
Future<Optional<Value>> readValue(KeyRef key,
ReadType type = ReadType::NORMAL,
Optional<UID> debugID = Optional<UID>()) override {
return readValueImpl(this, IKVSGetValueRequest{ key, type, debugID, ReplyPromise<Optional<Value>>() });
}
Future<Optional<Value>> readValuePrefix(KeyRef key,
int maxLength,
ReadType type = ReadType::NORMAL,
Optional<UID> debugID = Optional<UID>()) override {
return interf.readValuePrefix.getReply(
IKVSReadValuePrefixRequest{ key, maxLength, type, debugID, ReplyPromise<Optional<Value>>() });
}
Future<RangeResult> readRange(KeyRangeRef keys,
int rowLimit = 1 << 30,
int byteLimit = 1 << 30,
ReadType type = ReadType::NORMAL) override {
IKVSReadRangeRequest req{ keys, rowLimit, byteLimit, type, ReplyPromise<IKVSReadRangeReply>() };
return fmap([](const IKVSReadRangeReply& reply) { return reply.toRangeResult(); },
interf.readRange.getReply(req));
}
StorageBytes getStorageBytes() const override { return storageBytes; }
void consumeInterface(StringRef intf) {
kvsProcess = ObjectReader::fromStringRef<RemoteKVSProcessInterface>(intf, IncludeVersion());
}
ACTOR static Future<Void> commitAndGetStorageBytes(RemoteIKeyValueStore* self,
Future<IKVSCommitReply> commitReplyFuture) {
IKVSCommitReply commitReply = wait(commitReplyFuture);
self->storageBytes = commitReply.storeBytes;
return Void();
}
ACTOR static Future<Optional<Value>> readValueImpl(RemoteIKeyValueStore* self, IKVSGetValueRequest req) {
Optional<Value> val = wait(self->interf.getValue.getReply(req));
return val;
}
ACTOR static Future<Void> getErrorImpl(const RemoteIKeyValueStore* self, Future<int> returnCode) {
choose {
when(wait(self->initialized)) {}
when(wait(delay(SERVER_KNOBS->REMOTE_KV_STORE_MAX_INIT_DURATION))) {
TraceEvent(SevError, "RemoteIKVSInitTooLong")
.detail("TimeLimit", SERVER_KNOBS->REMOTE_KV_STORE_MAX_INIT_DURATION);
throw please_reboot_remote_kv_store();
}
}
state Future<Void> connectionCheckingDelay = delay(FLOW_KNOBS->FAILURE_DETECTION_DELAY);
state Future<ErrorOr<Void>> storeError = errorOr(self->interf.getError.getReply(IKVSGetErrorRequest{}));
loop choose {
when(ErrorOr<Void> e = wait(storeError)) {
TraceEvent(SevDebug, "RemoteIKVSGetError")
.errorUnsuppressed(e.isError() ? e.getError() : success())
.backtrace();
if (e.isError())
throw e.getError();
else
return e.get();
}
when(int res = wait(returnCode)) {
TraceEvent(res != 0 ? SevError : SevInfo, "SpawnedProcessDied").detail("Res", res);
if (res)
throw please_reboot_remote_kv_store(); // this will reboot the worker
else
return Void();
}
when(wait(connectionCheckingDelay)) {
// for the corner case where the child process stuck and waitpid also does not give update on it
// In this scenario, we need to manually reboot the storage engine process
if (IFailureMonitor::failureMonitor()
.getState(self->interf.getError.getEndpoint().getPrimaryAddress())
.isFailed()) {
TraceEvent(SevError, "RemoteKVStoreConnectionStuck").log();
throw please_reboot_remote_kv_store(); // this will reboot the worker
}
connectionCheckingDelay = delay(FLOW_KNOBS->FAILURE_DETECTION_DELAY);
}
}
}
ACTOR static Future<Void> onCloseImpl(const RemoteIKeyValueStore* self) {
try {
wait(self->initialized);
wait(self->interf.onClosed.getReply(IKVSOnClosedRequest{}));
TraceEvent(SevDebug, "RemoteIKVSOnCloseImplOnClosedFinished");
} catch (Error& e) {
TraceEvent(SevInfo, "RemoteIKVSOnCloseImplError").errorUnsuppressed(e).backtrace();
throw;
}
return Void();
}
};
Future<Void> runFlowProcess(std::string const& name, Endpoint endpoint);
#include "flow/unactorcompiler.h"
#endif

View File

@ -47,7 +47,8 @@ ACTOR static Future<Version> collectBackupFiles(Reference<IBackupContainer> bc,
RestoreRequest request);
ACTOR static Future<Void> buildRangeVersions(KeyRangeMap<Version>* pRangeVersions,
std::vector<RestoreFileFR>* pRangeFiles,
Key url);
Key url,
Optional<std::string> proxy);
ACTOR static Future<Version> processRestoreRequest(Reference<RestoreControllerData> self,
Database cx,
@ -317,7 +318,7 @@ ACTOR static Future<Version> processRestoreRequest(Reference<RestoreControllerDa
state std::vector<RestoreFileFR> allFiles;
state Version minRangeVersion = MAX_VERSION;
self->initBackupContainer(request.url);
self->initBackupContainer(request.url, request.proxy);
// Get all backup files' description and save them to files
state Version targetVersion =
@ -334,7 +335,7 @@ ACTOR static Future<Version> processRestoreRequest(Reference<RestoreControllerDa
// Build range versions: version of key ranges in range file
state KeyRangeMap<Version> rangeVersions(minRangeVersion, allKeys.end);
if (SERVER_KNOBS->FASTRESTORE_GET_RANGE_VERSIONS_EXPENSIVE) {
wait(buildRangeVersions(&rangeVersions, &rangeFiles, request.url));
wait(buildRangeVersions(&rangeVersions, &rangeFiles, request.url, request.proxy));
} else {
// Debug purpose, dump range versions
auto ranges = rangeVersions.ranges();
@ -881,13 +882,14 @@ ACTOR static Future<Void> insertRangeVersion(KeyRangeMap<Version>* pRangeVersion
// Expensive and slow operation that should not run in real prod.
ACTOR static Future<Void> buildRangeVersions(KeyRangeMap<Version>* pRangeVersions,
std::vector<RestoreFileFR>* pRangeFiles,
Key url) {
Key url,
Optional<std::string> proxy) {
if (!g_network->isSimulated()) {
TraceEvent(SevError, "ExpensiveBuildRangeVersions")
.detail("Reason", "Parsing all range files is slow and memory intensive");
return Void();
}
Reference<IBackupContainer> bc = IBackupContainer::openContainer(url.toString());
Reference<IBackupContainer> bc = IBackupContainer::openContainer(url.toString(), proxy, {});
// Key ranges not in range files are empty;
// Assign highest version to avoid applying any mutation in these ranges

View File

@ -446,13 +446,15 @@ struct RestoreControllerData : RestoreRoleData, public ReferenceCounted<RestoreC
}
}
void initBackupContainer(Key url) {
void initBackupContainer(Key url, Optional<std::string> proxy) {
if (bcUrl == url && bc.isValid()) {
return;
}
TraceEvent("FastRestoreControllerInitBackupContainer").detail("URL", url);
TraceEvent("FastRestoreControllerInitBackupContainer")
.detail("URL", url)
.detail("Proxy", proxy.present() ? proxy.get() : "");
bcUrl = url;
bc = IBackupContainer::openContainer(url.toString());
bc = IBackupContainer::openContainer(url.toString(), proxy, {});
}
};

View File

@ -262,7 +262,7 @@ ACTOR Future<Void> restoreLoaderCore(RestoreLoaderInterface loaderInterf,
when(RestoreLoadFileRequest req = waitNext(loaderInterf.loadFile.getFuture())) {
requestTypeStr = "loadFile";
hasQueuedRequests = !self->loadingQueue.empty() || !self->sendingQueue.empty();
self->initBackupContainer(req.param.url);
self->initBackupContainer(req.param.url, req.param.proxy);
self->loadingQueue.push(req);
if (!hasQueuedRequests) {
self->hasPendingRequests->set(true);

View File

@ -226,12 +226,12 @@ struct RestoreLoaderData : RestoreRoleData, public ReferenceCounted<RestoreLoade
finishedBatch = NotifiedVersion(0);
}
void initBackupContainer(Key url) {
void initBackupContainer(Key url, Optional<std::string> proxy) {
if (bcUrl == url && bc.isValid()) {
return;
}
bcUrl = url;
bc = IBackupContainer::openContainer(url.toString());
bc = IBackupContainer::openContainer(url.toString(), proxy, {});
}
};

View File

@ -368,6 +368,7 @@ struct LoadingParam {
bool isRangeFile;
Key url;
Optional<std::string> proxy;
Optional<Version> rangeVersion; // range file's version
int64_t blockSize;
@ -386,12 +387,13 @@ struct LoadingParam {
template <class Ar>
void serialize(Ar& ar) {
serializer(ar, isRangeFile, url, rangeVersion, blockSize, asset);
serializer(ar, isRangeFile, url, proxy, rangeVersion, blockSize, asset);
}
std::string toString() const {
std::stringstream str;
str << "isRangeFile:" << isRangeFile << " url:" << url.toString()
<< " proxy:" << (proxy.present() ? proxy.get() : "")
<< " rangeVersion:" << (rangeVersion.present() ? rangeVersion.get() : -1) << " blockSize:" << blockSize
<< " RestoreAsset:" << asset.toString();
return str.str();

View File

@ -53,6 +53,13 @@
extern "C" int g_expect_full_pointermap;
extern const char* getSourceVersion();
ISimulator::ISimulator()
: desiredCoordinators(1), physicalDatacenters(1), processesPerMachine(0), listenersPerProcess(1), usableRegions(1),
allowLogSetKills(true), tssMode(TSSMode::Disabled), isStopped(false), lastConnectionFailure(0),
connectionFailuresDisableDuration(0), speedUpSimulation(false), backupAgents(BackupAgentType::WaitForType),
drAgents(BackupAgentType::WaitForType), allSwapsDisabled(false) {}
ISimulator::~ISimulator() = default;
using namespace std::literals;
// TODO: Defining these here is just asking for ODR violations.
@ -256,6 +263,9 @@ class TestConfig {
if (attrib == "disableHostname") {
disableHostname = strcmp(value.c_str(), "true") == 0;
}
if (attrib == "disableRemoteKVS") {
disableRemoteKVS = strcmp(value.c_str(), "true") == 0;
}
if (attrib == "restartInfoLocation") {
isFirstTestInRestart = true;
}
@ -266,6 +276,9 @@ class TestConfig {
configDBType = configDBTypeFromString(value);
}
}
if (attrib == "randomlyRenameZoneId") {
randomlyRenameZoneId = strcmp(value.c_str(), "true") == 0;
}
if (attrib == "blobGranulesEnabled") {
blobGranulesEnabled = strcmp(value.c_str(), "true") == 0;
}
@ -288,12 +301,14 @@ public:
bool disableTss = false;
// 7.1 cannot be downgraded to 7.0 and below after enabling hostname, so disable hostname for 7.0 downgrade tests
bool disableHostname = false;
// remote key value store is a child process spawned by the SS process to run the storage engine
bool disableRemoteKVS = false;
// Storage Engine Types: Verify match with SimulationConfig::generateNormalConfig
// 0 = "ssd"
// 1 = "memory"
// 2 = "memory-radixtree-beta"
// 3 = "ssd-redwood-1-experimental"
// 4 = "ssd-rocksdb-experimental"
// 4 = "ssd-rocksdb-v1"
// Requires a comma-separated list of numbers WITHOUT whitespaces
std::vector<int> storageEngineExcludeTypes;
// Set the maximum TLog version that can be selected for a test
@ -307,6 +322,7 @@ public:
stderrSeverity, machineCount, processesPerMachine, coordinators;
bool blobGranulesEnabled = false;
Optional<std::string> config;
bool randomlyRenameZoneId = false;
bool allowDefaultTenant = true;
bool allowDisablingTenants = true;
@ -346,6 +362,7 @@ public:
.add("maxTLogVersion", &maxTLogVersion)
.add("disableTss", &disableTss)
.add("disableHostname", &disableHostname)
.add("disableRemoteKVS", &disableRemoteKVS)
.add("simpleConfig", &simpleConfig)
.add("generateFearless", &generateFearless)
.add("datacenters", &datacenters)
@ -364,7 +381,8 @@ public:
.add("extraMachineCountDC", &extraMachineCountDC)
.add("blobGranulesEnabled", &blobGranulesEnabled)
.add("allowDefaultTenant", &allowDefaultTenant)
.add("allowDisablingTenants", &allowDisablingTenants);
.add("allowDisablingTenants", &allowDisablingTenants)
.add("randomlyRenameZoneId", &randomlyRenameZoneId);
try {
auto file = toml::parse(testFile);
if (file.contains("configuration") && toml::find(file, "configuration").is_table()) {
@ -1039,6 +1057,11 @@ ACTOR Future<Void> restartSimulatedSystem(std::vector<Future<Void>>* systemActor
auto configDBType = testConfig.getConfigDBType();
// Randomly change data center id names to test that localities
// can be modified on cluster restart
bool renameZoneIds = testConfig.randomlyRenameZoneId ? deterministicRandom()->random01() < 0.1 : false;
TEST(renameZoneIds); // Zone ID names altered in restart test
// allows multiple ipAddr entries
ini.SetMultiKey();
@ -1059,7 +1082,7 @@ ACTOR Future<Void> restartSimulatedSystem(std::vector<Future<Void>>* systemActor
bool enableExtraDB = (testConfig.extraDB == 3);
ClusterConnectionString conn(ini.GetValue("META", "connectionString"));
if (enableExtraDB) {
g_simulator.extraDB = new ClusterConnectionString(ini.GetValue("META", "connectionString"));
g_simulator.extraDB = std::make_unique<ClusterConnectionString>(ini.GetValue("META", "connectionString"));
}
if (!testConfig.disableHostname) {
auto mockDNSStr = ini.GetValue("META", "mockDNS");
@ -1067,6 +1090,11 @@ ACTOR Future<Void> restartSimulatedSystem(std::vector<Future<Void>>* systemActor
INetworkConnections::net()->parseMockDNSFromString(mockDNSStr);
}
}
if (testConfig.disableRemoteKVS) {
IKnobCollection::getMutableGlobalKnobCollection().setKnob("remote_kv_store",
KnobValueRef::create(bool{ false }));
TraceEvent(SevDebug, "DisaableRemoteKVS").log();
}
*pConnString = conn;
*pTesterCount = testerCount;
bool usingSSL = conn.toString().find(":tls") != std::string::npos || listenersPerProcess > 1;
@ -1087,7 +1115,11 @@ ACTOR Future<Void> restartSimulatedSystem(std::vector<Future<Void>>* systemActor
if (zoneIDini == nullptr) {
zoneId = machineId;
} else {
zoneId = StringRef(zoneIDini);
auto zoneIdStr = std::string(zoneIDini);
if (renameZoneIds) {
zoneIdStr = "modified/" + zoneIdStr;
}
zoneId = Standalone<StringRef>(zoneIdStr);
}
ProcessClass::ClassType cType =
@ -1142,7 +1174,7 @@ ACTOR Future<Void> restartSimulatedSystem(std::vector<Future<Void>>* systemActor
}
LocalityData localities(Optional<Standalone<StringRef>>(), zoneId, machineId, dcUID);
localities.set(LiteralStringRef("data_hall"), dcUID);
localities.set("data_hall"_sr, dcUID);
// SOMEDAY: parse backup agent from test file
systemActors->push_back(reportErrors(
@ -1374,7 +1406,7 @@ void SimulationConfig::setStorageEngine(const TestConfig& testConfig) {
}
case 4: {
TEST(true); // Simulated cluster using RocksDB storage engine
set_config("ssd-rocksdb-experimental");
set_config("ssd-rocksdb-v1");
// Tests using the RocksDB engine are necessarily non-deterministic because of RocksDB
// background threads.
TraceEvent(SevWarnAlways, "RocksDBNonDeterminism")
@ -1815,6 +1847,11 @@ void setupSimulatedSystem(std::vector<Future<Void>>* systemActors,
if (testConfig.configureLocked) {
startingConfigString += " locked";
}
if (testConfig.disableRemoteKVS) {
IKnobCollection::getMutableGlobalKnobCollection().setKnob("remote_kv_store",
KnobValueRef::create(bool{ false }));
TraceEvent(SevDebug, "DisaableRemoteKVS").log();
}
auto configDBType = testConfig.getConfigDBType();
for (auto kv : startingConfigJSON) {
if ("tss_storage_engine" == kv.first) {
@ -2040,9 +2077,9 @@ void setupSimulatedSystem(std::vector<Future<Void>>* systemActors,
deterministicRandom()->randomShuffle(coordinatorAddresses);
ASSERT_EQ(coordinatorAddresses.size(), coordinatorCount);
ClusterConnectionString conn(coordinatorAddresses, LiteralStringRef("TestCluster:0"));
ClusterConnectionString conn(coordinatorAddresses, "TestCluster:0"_sr);
if (useHostname) {
conn = ClusterConnectionString(coordinatorHostnames, LiteralStringRef("TestCluster:0"));
conn = ClusterConnectionString(coordinatorHostnames, "TestCluster:0"_sr);
}
// If extraDB==0, leave g_simulator.extraDB as null because the test does not use DR.
@ -2050,21 +2087,21 @@ void setupSimulatedSystem(std::vector<Future<Void>>* systemActors,
// The DR database can be either a new database or itself
g_simulator.extraDB =
BUGGIFY
? (useHostname ? new ClusterConnectionString(coordinatorHostnames, LiteralStringRef("TestCluster:0"))
: new ClusterConnectionString(coordinatorAddresses, LiteralStringRef("TestCluster:0")))
? (useHostname ? std::make_unique<ClusterConnectionString>(coordinatorHostnames, "TestCluster:0"_sr)
: std::make_unique<ClusterConnectionString>(coordinatorAddresses, "TestCluster:0"_sr))
: (useHostname
? new ClusterConnectionString(extraCoordinatorHostnames, LiteralStringRef("ExtraCluster:0"))
: new ClusterConnectionString(extraCoordinatorAddresses, LiteralStringRef("ExtraCluster:0")));
? std::make_unique<ClusterConnectionString>(extraCoordinatorHostnames, "ExtraCluster:0"_sr)
: std::make_unique<ClusterConnectionString>(extraCoordinatorAddresses, "ExtraCluster:0"_sr));
} else if (testConfig.extraDB == 2) {
// The DR database is a new database
g_simulator.extraDB =
useHostname ? new ClusterConnectionString(extraCoordinatorHostnames, LiteralStringRef("ExtraCluster:0"))
: new ClusterConnectionString(extraCoordinatorAddresses, LiteralStringRef("ExtraCluster:0"));
useHostname ? std::make_unique<ClusterConnectionString>(extraCoordinatorHostnames, "ExtraCluster:0"_sr)
: std::make_unique<ClusterConnectionString>(extraCoordinatorAddresses, "ExtraCluster:0"_sr);
} else if (testConfig.extraDB == 3) {
// The DR database is the same database
g_simulator.extraDB =
useHostname ? new ClusterConnectionString(coordinatorHostnames, LiteralStringRef("TestCluster:0"))
: new ClusterConnectionString(coordinatorAddresses, LiteralStringRef("TestCluster:0"));
g_simulator.extraDB = useHostname
? std::make_unique<ClusterConnectionString>(coordinatorHostnames, "TestCluster:0"_sr)
: std::make_unique<ClusterConnectionString>(coordinatorAddresses, "TestCluster:0"_sr);
}
*pConnString = conn;
@ -2164,7 +2201,7 @@ void setupSimulatedSystem(std::vector<Future<Void>>* systemActors,
// check the sslEnablementMap using only one ip
LocalityData localities(Optional<Standalone<StringRef>>(), zoneId, machineId, dcUID);
localities.set(LiteralStringRef("data_hall"), dcUID);
localities.set("data_hall"_sr, dcUID);
systemActors->push_back(reportErrors(simulatedMachine(conn,
ips,
sslEnabled,
@ -2191,7 +2228,7 @@ void setupSimulatedSystem(std::vector<Future<Void>>* systemActors,
Standalone<StringRef> newMachineId(deterministicRandom()->randomUniqueID().toString());
LocalityData localities(Optional<Standalone<StringRef>>(), newZoneId, newMachineId, dcUID);
localities.set(LiteralStringRef("data_hall"), dcUID);
localities.set("data_hall"_sr, dcUID);
systemActors->push_back(reportErrors(simulatedMachine(*g_simulator.extraDB,
extraIps,
sslEnabled,
@ -2395,7 +2432,7 @@ ACTOR void setupAndRun(std::string dataFolder,
100.0));
// FIXME: snapshot restore does not support multi-region restore, hence restore it as single region always
if (restoring) {
startingConfiguration = LiteralStringRef("usable_regions=1");
startingConfiguration = "usable_regions=1"_sr;
}
} else {
g_expect_full_pointermap = 1;

View File

@ -1939,7 +1939,7 @@ ACTOR static Future<std::vector<std::pair<StorageServerInterface, EventMap>>> ge
if (metadata[i].present()) {
TraceEventFields metadataField;
metadataField.addField("CreatedTimeTimestamp", std::to_string(metadata[i].get().createdTime));
metadataField.addField("CreatedTimeDatetime", timerIntToGmt(metadata[i].get().createdTime));
metadataField.addField("CreatedTimeDatetime", epochsToGMTString(metadata[i].get().createdTime));
results[i].second.emplace("Metadata", metadataField);
} else if (!servers[i].isTss()) {
TraceEventFields metadataField;

View File

@ -62,7 +62,7 @@
{ \
std::string prefix = format("%s %f %04d ", g_network->getLocalAddress().toString().c_str(), now(), __LINE__); \
std::string msg = format(__VA_ARGS__); \
writePrefixedLines(debug_printf_stream, prefix, msg); \
fputs(addPrefix(prefix, msg).c_str(), debug_printf_stream); \
fflush(debug_printf_stream); \
}
@ -73,11 +73,13 @@
std::string prefix = \
format("%s %f %04d ", g_network->getLocalAddress().toString().c_str(), now(), __LINE__); \
std::string msg = format(__VA_ARGS__); \
writePrefixedLines(debug_printf_stream, prefix, msg); \
fputs(addPrefix(prefix, msg).c_str(), debug_printf_stream); \
fflush(debug_printf_stream); \
} \
}
#define debug_print(str) debug_printf("%s\n", str.c_str())
#define debug_print_always(str) debug_printf_always("%s\n", str.c_str())
#define debug_printf_noop(...)
#if defined(NO_INTELLISENSE)
@ -97,13 +99,18 @@
#define TRACE \
debug_printf_always("%s: %s line %d %s\n", __FUNCTION__, __FILE__, __LINE__, platform::get_backtrace().c_str());
// Writes prefix:line for each line in msg to fout
void writePrefixedLines(FILE* fout, std::string prefix, std::string msg) {
StringRef m = msg;
// Returns a string where every line in lines is prefixed with prefix
std::string addPrefix(std::string prefix, std::string lines) {
StringRef m = lines;
std::string s;
while (m.size() != 0) {
StringRef line = m.eat("\n");
fprintf(fout, "%s %s\n", prefix.c_str(), line.toString().c_str());
s += prefix;
s += ' ';
s += line.toString();
s += '\n';
}
return s;
}
#define PRIORITYMULTILOCK_DEBUG 0
@ -917,12 +924,15 @@ public:
}
}
// If readNext() cannot complete immediately, it will route to here
// The mutex will be taken if locked is false
// The next page will be waited for if load is true
// If readNext() cannot complete immediately because it must wait for IO, it will route to here.
// The purpose of this function is to serialize simultaneous readers on self while letting the
// common case (>99.8% of the time) be handled with low overhead by the non-actor readNext() function.
//
// The mutex will be taken if locked is false.
// The next page will be waited for if load is true.
// Only mutex holders will wait on the page read.
ACTOR static Future<Optional<T>> waitThenReadNext(Cursor* self,
Optional<T> upperBound,
Optional<T> inclusiveMaximum,
FlowMutex::Lock* lock,
bool load) {
state FlowMutex::Lock localLock;
@ -940,7 +950,7 @@ public:
wait(success(self->nextPageReader));
}
state Optional<T> result = wait(self->readNext(upperBound, &localLock));
state Optional<T> result = wait(self->readNext(inclusiveMaximum, &localLock));
// If a lock was not passed in, so this actor locked the mutex above, then unlock it
if (lock == nullptr) {
@ -959,10 +969,12 @@ public:
return result;
}
// Read the next item at the cursor (if < upperBound), moving to a new page first if the current page is
// exhausted If locked is true, this call owns the mutex, which would have been locked by readNext() before a
// recursive call
Future<Optional<T>> readNext(const Optional<T>& upperBound = {}, FlowMutex::Lock* lock = nullptr) {
// Read the next item from the cursor, possibly moving to and waiting for a new page if the prior page was
// exhausted. If the item is <= inclusiveMaximum, then return it after advancing the cursor to the next item.
// Otherwise, return nothing and do not advance the cursor.
// If locked is true, this call owns the mutex, which would have been locked by readNext() before a recursive
// call. See waitThenReadNext() for more detail.
Future<Optional<T>> readNext(const Optional<T>& inclusiveMaximum = {}, FlowMutex::Lock* lock = nullptr) {
if ((mode != POP && mode != READONLY) || pageID == invalidLogicalPageID || pageID == endPageID) {
debug_printf("FIFOQueue::Cursor(%s) readNext returning nothing\n", toString().c_str());
return Optional<T>();
@ -970,7 +982,7 @@ public:
// If we don't have a lock and the mutex isn't available then acquire it
if (lock == nullptr && isBusy()) {
return waitThenReadNext(this, upperBound, lock, false);
return waitThenReadNext(this, inclusiveMaximum, lock, false);
}
// We now know pageID is valid and should be used, but page might not point to it yet
@ -986,7 +998,7 @@ public:
}
if (!nextPageReader.isReady()) {
return waitThenReadNext(this, upperBound, lock, true);
return waitThenReadNext(this, inclusiveMaximum, lock, true);
}
page = nextPageReader.get();
@ -1007,11 +1019,11 @@ public:
int bytesRead;
const T result = Codec::readFromBytes(p->begin() + offset, bytesRead);
if (upperBound.present() && upperBound.get() < result) {
if (inclusiveMaximum.present() && inclusiveMaximum.get() < result) {
debug_printf("FIFOQueue::Cursor(%s) not popping %s, exceeds upper bound %s\n",
toString().c_str(),
::toString(result).c_str(),
::toString(upperBound.get()).c_str());
::toString(inclusiveMaximum.get()).c_str());
return Optional<T>();
}
@ -1059,10 +1071,10 @@ public:
}
}
debug_printf("FIFOQueue(%s) %s(upperBound=%s) -> %s\n",
debug_printf("FIFOQueue(%s) %s(inclusiveMaximum=%s) -> %s\n",
queue->name.c_str(),
(mode == POP ? "pop" : "peek"),
::toString(upperBound).c_str(),
::toString(inclusiveMaximum).c_str(),
::toString(result).c_str());
return Optional<T>(result);
}
@ -1290,8 +1302,8 @@ public:
Future<Optional<T>> peek() { return peek_impl(this); }
// Pop the next item on front of queue if it is <= upperBound or if upperBound is not present
Future<Optional<T>> pop(Optional<T> upperBound = {}) { return headReader.readNext(upperBound); }
// Pop the next item on front of queue if it is <= inclusiveMaximum or if inclusiveMaximum is not present
Future<Optional<T>> pop(Optional<T> inclusiveMaximum = {}) { return headReader.readNext(inclusiveMaximum); }
QueueState getState() const {
QueueState s;
@ -1484,8 +1496,8 @@ public:
int64_t numEntries;
int dataBytesPerPage;
int pagesPerExtent;
bool usesExtents;
bool tailPageNewExtent;
bool usesExtents = false;
bool tailPageNewExtent = false;
LogicalPageID prevExtentEndPageID;
Cursor headReader;
@ -2758,6 +2770,8 @@ public:
return f;
}
// Free pageID as of version v. This means that once the oldest readable pager snapshot is at version v, pageID is
// not longer in use by any structure so it can be used to write new data.
void freeUnmappedPage(PhysicalPageID pageID, Version v) {
// If v is older than the oldest version still readable then mark pageID as free as of the next commit
if (v < effectiveOldestVersion()) {
@ -2823,7 +2837,7 @@ public:
void freePage(LogicalPageID pageID, Version v) override {
// If pageID has been remapped, then it can't be freed until all existing remaps for that page have been undone,
// so queue it for later deletion
// so queue it for later deletion during remap cleanup
auto i = remappedPages.find(pageID);
if (i != remappedPages.end()) {
debug_printf("DWALPager(%s) op=freeRemapped %s @%" PRId64 " oldestVersion=%" PRId64 "\n",
@ -3331,7 +3345,11 @@ public:
// Since the next item can be arbitrarily ahead in the queue, secondType is determined by
// looking at the remappedPages structure.
//
// R == Remap F == Free D == Detach | == oldestRetaineedVersion
// R == Remap F == Free D == Detach | == oldestRetainedVersion
//
// oldestRetainedVersion is the oldest version being maintained as readable, either because it is explicitly the
// oldest readable version set or because there is an active snapshot for the version even though it is older
// than the explicitly set oldest readable version.
//
// R R | free new ID
// R F | free new ID if R and D are at different versions
@ -3411,13 +3429,32 @@ public:
}
if (freeNewID) {
debug_printf("DWALPager(%s) remapCleanup freeNew %s\n", self->filename.c_str(), p.toString().c_str());
self->freeUnmappedPage(p.newPageID, 0);
debug_printf("DWALPager(%s) remapCleanup freeNew %s %s\n",
self->filename.c_str(),
p.toString().c_str(),
toString(self->getLastCommittedVersion()).c_str());
// newID must be freed at the latest committed version to avoid a read race between caching and non-caching
// readers. It is possible that there are readers of newID in flight right now that either
// - Did not read through the page cache
// - Did read through the page cache but there was no entry for the page at the time, so one was created
// and the read future is still pending
// In either case the physical read of newID from disk can happen at some time after right now and after the
// current commit is finished.
//
// If newID is freed immediately, meaning as of the end of the current commit, then it could be reused in
// the next commit which could be before any reads fitting the above description have completed, causing
// those reads to the new write which is incorrect. Since such readers could be using pager snapshots at
// versions up to and including the latest committed version, newID must be freed *after* that version is no
// longer readable.
self->freeUnmappedPage(p.newPageID, self->getLastCommittedVersion() + 1);
++g_redwoodMetrics.metric.pagerRemapFree;
}
if (freeOriginalID) {
debug_printf("DWALPager(%s) remapCleanup freeOriginal %s\n", self->filename.c_str(), p.toString().c_str());
// originalID can be freed immediately because it is already the case that there are no readers at a version
// prior to oldestRetainedVersion so no reader will need originalID.
self->freeUnmappedPage(p.originalPageID, 0);
++g_redwoodMetrics.metric.pagerRemapFree;
}
@ -3654,6 +3691,7 @@ public:
self->operations.clear();
debug_printf("DWALPager(%s) shutdown destroy page cache\n", self->filename.c_str());
wait(self->extentCache.clear());
wait(self->pageCache.clear());
wait(delay(0));
@ -4575,7 +4613,7 @@ struct BTreePage {
ValueTree* valueTree() const { return (ValueTree*)(this + 1); }
std::string toString(bool write,
std::string toString(const char* context,
BTreePageIDRef id,
Version ver,
const RedwoodRecordRef& lowerBound,
@ -4583,7 +4621,7 @@ struct BTreePage {
std::string r;
r += format("BTreePage op=%s %s @%" PRId64
" ptr=%p height=%d count=%d kvBytes=%d\n lowerBound: %s\n upperBound: %s\n",
write ? "write" : "read",
context,
::toString(id).c_str(),
ver,
this,
@ -4684,24 +4722,43 @@ struct DecodeBoundaryVerifier {
typedef std::map<Version, DecodeBoundaries> BoundariesByVersion;
std::unordered_map<LogicalPageID, BoundariesByVersion> boundariesByPageID;
std::vector<Key> boundarySamples;
int boundarySampleSize = 1000;
int boundaryPopulation = 0;
static DecodeBoundaryVerifier* getVerifier(std::string name) {
static std::map<std::string, DecodeBoundaryVerifier> verifiers;
// Verifier disabled due to not being finished
//
// Only use verifier in a non-restarted simulation so that all page writes are captured
// if (g_network->isSimulated() && !g_simulator.restarted) {
// return &verifiers[name];
// }
if (g_network->isSimulated() && !g_simulator.restarted) {
return &verifiers[name];
}
return nullptr;
}
void sampleBoundary(Key b) {
if (boundaryPopulation <= boundarySampleSize) {
boundarySamples.push_back(b);
} else if (deterministicRandom()->random01() < ((double)boundarySampleSize / boundaryPopulation)) {
boundarySamples[deterministicRandom()->randomInt(0, boundarySampleSize)] = b;
}
++boundaryPopulation;
}
Key getSample() const {
if (boundarySamples.empty()) {
return Key();
}
return boundarySamples[deterministicRandom()->randomInt(0, boundarySamples.size())];
}
void update(BTreePageIDRef id, Version v, Key lowerBound, Key upperBound) {
sampleBoundary(lowerBound);
sampleBoundary(upperBound);
debug_printf("decodeBoundariesUpdate %s %s '%s' to '%s'\n",
::toString(id).c_str(),
::toString(v).c_str(),
lowerBound.toString().c_str(),
upperBound.toString().c_str());
lowerBound.printable().c_str(),
upperBound.printable().c_str());
auto& b = boundariesByPageID[id.front()][v];
ASSERT(b.empty());
@ -4717,28 +4774,53 @@ struct DecodeBoundaryVerifier {
--b;
if (b->second.lower != lowerBound || b->second.upper != upperBound) {
fprintf(stderr,
"Boundary mismatch on %s %s\nFound :%s %s\nExpected:%s %s\n",
"Boundary mismatch on %s %s\nUsing:\n\t'%s'\n\t'%s'\nWritten %s:\n\t'%s'\n\t'%s'\n",
::toString(id).c_str(),
::toString(v).c_str(),
lowerBound.toString().c_str(),
upperBound.toString().c_str(),
b->second.lower.toString().c_str(),
b->second.upper.toString().c_str());
lowerBound.printable().c_str(),
upperBound.printable().c_str(),
::toString(b->first).c_str(),
b->second.lower.printable().c_str(),
b->second.upper.printable().c_str());
return false;
}
return true;
}
void update(Version v, LogicalPageID oldID, LogicalPageID newID) {
debug_printf("decodeBoundariesUpdate copy %s %s to %s\n",
::toString(v).c_str(),
::toString(oldID).c_str(),
::toString(newID).c_str());
auto& old = boundariesByPageID[oldID];
ASSERT(!old.empty());
auto i = old.end();
--i;
boundariesByPageID[newID][v] = i->second;
debug_printf("decodeBoundariesUpdate copy %s %s to %s '%s' to '%s'\n",
::toString(v).c_str(),
::toString(oldID).c_str(),
::toString(newID).c_str(),
i->second.lower.printable().c_str(),
i->second.upper.printable().c_str());
}
void removeAfterVersion(Version version) {
auto i = boundariesByPageID.begin();
while (i != boundariesByPageID.end()) {
auto v = i->second.upper_bound(version);
while (v != i->second.end()) {
debug_printf("decodeBoundariesUpdate remove %s %s '%s' to '%s'\n",
::toString(v->first).c_str(),
::toString(i->first).c_str(),
v->second.lower.printable().c_str(),
v->second.upper.printable().c_str());
v = i->second.erase(v);
}
if (i->second.empty()) {
debug_printf("decodeBoundariesUpdate remove empty map for %s\n", ::toString(i->first).c_str());
i = boundariesByPageID.erase(i);
} else {
++i;
}
}
}
};
@ -5024,8 +5106,14 @@ public:
self->m_newOldestVersion = self->m_pager->getOldestReadableVersion();
debug_printf("Recovered pager to version %" PRId64 ", oldest version is %" PRId64 "\n",
self->getLastCommittedVersion(),
self->m_newOldestVersion);
// Clear any changes that occurred after the latest committed version
if (self->m_pBoundaryVerifier != nullptr) {
self->m_pBoundaryVerifier->removeAfterVersion(self->getLastCommittedVersion());
}
state Key meta = self->m_pager->getMetaKey();
if (meta.size() == 0) {
// Create new BTree
@ -5825,10 +5913,17 @@ private:
const RedwoodRecordRef& lowerBound,
const RedwoodRecordRef& upperBound) {
if (page->userData == nullptr) {
debug_printf("Creating DecodeCache for ptr=%p lower=%s upper=%s\n",
debug_printf("Creating DecodeCache for ptr=%p lower=%s upper=%s %s\n",
page->begin(),
lowerBound.toString(false).c_str(),
upperBound.toString(false).c_str());
upperBound.toString(false).c_str(),
((BTreePage*)page->begin())
->toString("cursor",
lowerBound.value.present() ? lowerBound.getChildPage() : BTreePageIDRef(),
-1,
lowerBound,
upperBound)
.c_str());
BTreePage::BinaryTree::DecodeCache* cache =
new BTreePage::BinaryTree::DecodeCache(lowerBound, upperBound, m_pDecodeCacheMemory);
@ -5890,12 +5985,13 @@ private:
BTreePage* btPage = (BTreePage*)page->begin();
BTreePage::BinaryTree::DecodeCache* cache = (BTreePage::BinaryTree::DecodeCache*)page->userData;
debug_printf_always(
"updateBTreePage(%s, %s) %s\n",
"updateBTreePage(%s, %s) start, page:\n%s\n",
::toString(oldID).c_str(),
::toString(writeVersion).c_str(),
cache == nullptr
? "<noDecodeCache>"
: btPage->toString(true, oldID, writeVersion, cache->lowerBound, cache->upperBound).c_str());
: btPage->toString("updateBTreePage", oldID, writeVersion, cache->lowerBound, cache->upperBound)
.c_str());
}
state unsigned int height = (unsigned int)((BTreePage*)page->begin())->height;
@ -5912,7 +6008,11 @@ private:
LogicalPageID id = wait(self->m_pager->newPageID());
emptyPages[i] = id;
}
debug_printf("updateBTreePage: newPages %s", toString(emptyPages).c_str());
debug_printf("updateBTreePage(%s, %s): newPages %s",
::toString(oldID).c_str(),
::toString(writeVersion).c_str(),
toString(emptyPages).c_str());
self->m_pager->updatePage(PagerEventReasons::Commit, height, emptyPages, page);
i = 0;
for (const LogicalPageID id : emptyPages) {
@ -5956,13 +6056,15 @@ private:
RedwoodRecordRef decodeLowerBound;
RedwoodRecordRef decodeUpperBound;
// Returns true of BTree logical boundaries and DeltaTree decoding boundaries are the same.
bool boundariesNormal() const {
// If the decode upper boundary is the subtree upper boundary the pointers will be the same
// For the lower boundary, if the pointers are not the same there is still a possibility
// that the keys are the same. This happens for the first remaining subtree of an internal page
// after the prior subtree(s) were cleared.
return (decodeUpperBound == subtreeUpperBound) &&
(decodeLowerBound == subtreeLowerBound || decodeLowerBound.sameExceptValue(subtreeLowerBound));
// Often these strings will refer to the same memory so same() is used as a faster way of determining
// equality in thec common case, but if it does not match a string comparison is needed as they can
// still be the same. This can happen for the first remaining subtree of an internal page
// after all prior subtree(s) were cleared.
return (
(decodeUpperBound.key.same(subtreeUpperBound.key) || decodeUpperBound.key == subtreeUpperBound.key) &&
(decodeLowerBound.key.same(subtreeLowerBound.key) || decodeLowerBound.key == subtreeLowerBound.key));
}
// The record range of the subtree slice is cBegin to cEnd
@ -6026,6 +6128,7 @@ private:
// Set the child page ID, which has already been allocated in result.arena()
newLinks.back().setChildPage(maybeNewID);
childrenChanged = true;
expectedUpperBound = decodeUpperBound;
} else {
childrenChanged = false;
}
@ -6070,6 +6173,7 @@ private:
s += format("SubtreeUpper: %s\n", subtreeUpperBound.toString(false).c_str());
s += format("expectedUpperBound: %s\n",
expectedUpperBound.present() ? expectedUpperBound.get().toString(false).c_str() : "(null)");
s += format("newLinks:\n");
for (int i = 0; i < newLinks.size(); ++i) {
s += format(" %i: %s\n", i, newLinks[i].toString(false).c_str());
}
@ -6178,10 +6282,10 @@ private:
// This must be called for each of the InternalPageSliceUpdates in sorted order.
void applyUpdate(InternalPageSliceUpdate& u, const RedwoodRecordRef* nextBoundary) {
debug_printf("applyUpdate nextBoundary=(%p) %s %s\n",
debug_printf("applyUpdate nextBoundary=(%p) %s\n",
nextBoundary,
(nextBoundary != nullptr) ? nextBoundary->toString(false).c_str() : "",
u.toString().c_str());
(nextBoundary != nullptr) ? nextBoundary->toString(false).c_str() : "");
debug_print(addPrefix("applyUpdate", u.toString()));
// If the children changed, replace [cBegin, cEnd) with newLinks
if (u.childrenChanged) {
@ -6195,7 +6299,7 @@ private:
}
while (c != u.cEnd) {
debug_printf("internal page (updating) erasing: %s\n", c.get().toString(false).c_str());
debug_printf("applyUpdate (updating) erasing: %s\n", c.get().toString(false).c_str());
btPage()->kvBytes -= c.get().kvBytes();
c.erase();
}
@ -6226,7 +6330,7 @@ private:
keep(u.cBegin, u.cEnd);
}
// If there is an expected upper boundary for the next range after u
// If there is an expected upper boundary for the next range start after u
if (u.expectedUpperBound.present()) {
// Then if it does not match the next boundary then insert a dummy record
if (nextBoundary == nullptr || (nextBoundary != &u.expectedUpperBound.get() &&
@ -6253,23 +6357,29 @@ private:
state std::string context;
if (REDWOOD_DEBUG) {
context = format("CommitSubtree(root=%s): ", toString(rootID).c_str());
context = format("CommitSubtree(root=%s+%d %s): ",
toString(rootID.front()).c_str(),
rootID.size() - 1,
::toString(batch->writeVersion).c_str());
}
debug_printf("%s %s\n", context.c_str(), update->toString().c_str());
debug_printf("%s rootID=%s\n", context.c_str(), toString(rootID).c_str());
debug_print(addPrefix(context, update->toString()));
if (REDWOOD_DEBUG) {
debug_printf("%s ---------MUTATION BUFFER SLICE ---------------------\n", context.c_str());
auto begin = mBegin;
int c = 0;
auto i = mBegin;
while (1) {
debug_printf("%s Mutation: '%s': %s\n",
debug_printf("%s Mutation %4d '%s': %s\n",
context.c_str(),
printable(begin.key()).c_str(),
begin.mutation().toString().c_str());
if (begin == mEnd) {
c,
printable(i.key()).c_str(),
i.mutation().toString().c_str());
if (i == mEnd) {
break;
}
++begin;
++c;
++i;
}
debug_printf("%s -------------------------------------\n", context.c_str());
}
state Reference<const ArenaPage> page =
@ -6291,13 +6401,13 @@ private:
// TryToUpdate indicates insert and erase operations should be tried on the existing page first
state bool tryToUpdate = btPage->tree()->numItems > 0 && update->boundariesNormal();
debug_printf(
"%s commitSubtree(): %s\n",
context.c_str(),
btPage
->toString(
false, rootID, batch->snapshot->getVersion(), update->decodeLowerBound, update->decodeUpperBound)
.c_str());
debug_printf("%s tryToUpdate=%d\n", context.c_str(), tryToUpdate);
debug_print(addPrefix(context,
btPage->toString("commitSubtreeStart",
rootID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)));
state BTreePage::BinaryTree::Cursor cursor = update->cBegin.valid()
? self->getCursor(page.getPtr(), update->cBegin)
@ -6312,22 +6422,6 @@ private:
}
}
if (REDWOOD_DEBUG) {
debug_printf("%s ---------MUTATION BUFFER SLICE ---------------------\n", context.c_str());
auto begin = mBegin;
while (1) {
debug_printf("%s Mutation: '%s': %s\n",
context.c_str(),
printable(begin.key()).c_str(),
begin.mutation().toString().c_str());
if (begin == mEnd) {
break;
}
++begin;
}
debug_printf("%s -------------------------------------\n", context.c_str());
}
// Leaf Page
if (btPage->isLeaf()) {
// When true, we are modifying the existing DeltaTree
@ -6566,9 +6660,8 @@ private:
// No changes were actually made. This could happen if the only mutations are clear ranges which do not
// match any records.
if (!changesMade) {
debug_printf("%s No changes were made during mutation merge, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s No changes were made during mutation merge, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
} else {
debug_printf(
@ -6581,17 +6674,26 @@ private:
if (cursor.tree->numItems == 0) {
update->cleared();
self->freeBTreePage(height, rootID, batch->writeVersion);
debug_printf("%s Page updates cleared all entries, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s Page updates cleared all entries, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
} else {
// Otherwise update it.
BTreePageIDRef newID = wait(self->updateBTreePage(
self, rootID, &update->newLinks.arena(), pageCopy.castTo<ArenaPage>(), batch->writeVersion));
debug_printf("%s Leaf node updated in-place at version %s, new contents:\n",
context.c_str(),
toString(batch->writeVersion).c_str());
debug_print(addPrefix(context,
btPage->toString("updateLeafNode",
newID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)));
update->updatedInPlace(newID, btPage, newID.size() * self->m_blockSize);
debug_printf(
"%s Page updated in-place, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Leaf node updated in-place, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
}
return Void();
}
@ -6601,9 +6703,8 @@ private:
update->cleared();
self->freeBTreePage(height, rootID, batch->writeVersion);
debug_printf("%s All leaf page contents were cleared, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s All leaf page contents were cleared, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
}
@ -6619,7 +6720,8 @@ private:
// Put new links into update and tell update that pages were rebuilt
update->rebuilt(entries);
debug_printf("%s Merge complete, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Merge complete, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
} else {
// Internal Page
@ -6668,8 +6770,8 @@ private:
if (!cursor.get().value.present()) {
// If the upper bound is provided by a dummy record in [cBegin, cEnd) then there is no
// requirement on the next subtree range or the parent page to have a specific upper boundary
// for decoding the subtree.
u.expectedUpperBound.reset();
// for decoding the subtree. The expected upper bound has not yet been set so it can remain
// empty.
cursor.moveNext();
// If there is another record after the null child record, it must have a child page value
ASSERT(!cursor.valid() || cursor.get().value.present());
@ -6756,12 +6858,12 @@ private:
RedwoodRecordRef rec = c.get();
if (rec.value.present()) {
if (height == 2) {
debug_printf("%s: freeing child page in cleared subtree range: %s\n",
debug_printf("%s freeing child page in cleared subtree range: %s\n",
context.c_str(),
::toString(rec.getChildPage()).c_str());
self->freeBTreePage(height, rec.getChildPage(), batch->writeVersion);
} else {
debug_printf("%s: queuing subtree deletion cleared subtree range: %s\n",
debug_printf("%s queuing subtree deletion cleared subtree range: %s\n",
context.c_str(),
::toString(rec.getChildPage()).c_str());
self->m_lazyClearQueue.pushBack(LazyClearQueueEntry{
@ -6774,9 +6876,8 @@ private:
// Subtree range unchanged
}
debug_printf("%s: MutationBuffer covers this range in a single mutation, not recursing: %s\n",
context.c_str(),
u.toString().c_str());
debug_printf("%s Not recursing, one mutation range covers this slice:\n", context.c_str());
debug_print(addPrefix(context, u.toString()));
// u has already been initialized with the correct result, no recursion needed, so restart the
// loop.
@ -6785,6 +6886,9 @@ private:
}
// If this page has height of 2 then its children are leaf nodes
debug_printf("%s Recursing for %s\n", context.c_str(), toString(pageID).c_str());
debug_print(addPrefix(context, u.toString()));
recursions.push_back(self->commitSubtree(self, batch, pageID, height - 1, mBegin, mEnd, &u));
}
@ -6823,10 +6927,11 @@ private:
// passed, so in the event a different upper boundary is needed it will be added to the already-modified
// page. Otherwise, the decode boundary is used which will prevent this page from being modified for the
// sole purpose of adding a dummy upper bound record.
debug_printf("%s Applying final child range update. changesMade=%d Parent update is: %s\n",
debug_printf("%s Applying final child range update. changesMade=%d\nSubtree Root Update:\n",
context.c_str(),
modifier.changesMade,
update->toString().c_str());
modifier.changesMade);
debug_print(addPrefix(context, update->toString()));
modifier.applyUpdate(*slices.back(),
modifier.changesMade ? &update->subtreeUpperBound : &update->decodeUpperBound);
@ -6859,9 +6964,11 @@ private:
if (modifier.changesMade || forceUpdate) {
if (modifier.empty()) {
update->cleared();
debug_printf("%s All internal page children were deleted so deleting this page too, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf(
"%s All internal page children were deleted so deleting this page too. Returning slice:\n",
context.c_str());
debug_print(addPrefix(context, update->toString()));
self->freeBTreePage(height, rootID, batch->writeVersion);
} else {
if (modifier.updating) {
@ -6899,9 +7006,10 @@ private:
}
parentInfo->clear();
if (forceUpdate && detached == 0) {
debug_printf("%s No children detached during forced update, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s No children detached during forced update, returning slice:\n",
context.c_str());
debug_print(addPrefix(context, update->toString()));
return Void();
}
}
@ -6912,21 +7020,19 @@ private:
pageCopy.castTo<ArenaPage>(),
batch->writeVersion));
debug_printf(
"%s commitSubtree(): Internal page updated in-place at version %s, new contents: %s\n",
"%s commitSubtree(): Internal node updated in-place at version %s, new contents:\n",
context.c_str(),
toString(batch->writeVersion).c_str(),
btPage
->toString(false,
newID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)
.c_str());
toString(batch->writeVersion).c_str());
debug_print(addPrefix(context,
btPage->toString("updateInternalNode",
newID,
batch->snapshot->getVersion(),
update->decodeLowerBound,
update->decodeUpperBound)));
update->updatedInPlace(newID, btPage, newID.size() * self->m_blockSize);
debug_printf("%s Internal page updated in-place, returning %s\n",
context.c_str(),
toString(*update).c_str());
debug_printf("%s Internal node updated in-place, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
} else {
// Page was rebuilt, possibly split.
debug_printf("%s Internal page could not be modified, rebuilding replacement(s).\n",
@ -6973,12 +7079,13 @@ private:
rootID));
update->rebuilt(newChildEntries);
debug_printf(
"%s Internal page rebuilt, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Internal page rebuilt, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
}
}
} else {
debug_printf("%s Page has no changes, returning %s\n", context.c_str(), toString(*update).c_str());
debug_printf("%s Page has no changes, returning slice:\n", context.c_str());
debug_print(addPrefix(context, update->toString()));
}
return Void();
}
@ -9472,9 +9579,11 @@ TEST_CASE("Lredwood/correctness/btree") {
state double clearProbability =
params.getDouble("clearProbability").orDefault(deterministicRandom()->random01() * .1);
state double clearExistingBoundaryProbability =
params.getDouble("clearProbability").orDefault(deterministicRandom()->random01() * .5);
params.getDouble("clearExistingBoundaryProbability").orDefault(deterministicRandom()->random01() * .5);
state double clearSingleKeyProbability =
params.getDouble("clearSingleKeyProbability").orDefault(deterministicRandom()->random01());
params.getDouble("clearSingleKeyProbability").orDefault(deterministicRandom()->random01() * .1);
state double clearKnownNodeBoundaryProbability =
params.getDouble("clearKnownNodeBoundaryProbability").orDefault(deterministicRandom()->random01() * .1);
state double clearPostSetProbability =
params.getDouble("clearPostSetProbability").orDefault(deterministicRandom()->random01() * .1);
state double coldStartProbability =
@ -9495,10 +9604,11 @@ TEST_CASE("Lredwood/correctness/btree") {
// These settings are an attempt to keep the test execution real reasonably short
state int64_t maxPageOps = params.getInt("maxPageOps").orDefault((shortTest || serialTest) ? 50e3 : 1e6);
state int maxVerificationMapEntries =
params.getInt("maxVerificationMapEntries").orDefault((1.0 - coldStartProbability) * 300e3);
state int maxVerificationMapEntries = params.getInt("maxVerificationMapEntries").orDefault(300e3);
state int maxColdStarts = params.getInt("maxColdStarts").orDefault(300);
// Max number of records in the BTree or the versioned written map to visit
state int64_t maxRecordsRead = 300e6;
state int64_t maxRecordsRead = params.getInt("maxRecordsRead").orDefault(300e6);
printf("\n");
printf("file: %s\n", file.c_str());
@ -9516,9 +9626,11 @@ TEST_CASE("Lredwood/correctness/btree") {
printf("setExistingKeyProbability: %f\n", setExistingKeyProbability);
printf("clearProbability: %f\n", clearProbability);
printf("clearExistingBoundaryProbability: %f\n", clearExistingBoundaryProbability);
printf("clearKnownNodeBoundaryProbability: %f\n", clearKnownNodeBoundaryProbability);
printf("clearSingleKeyProbability: %f\n", clearSingleKeyProbability);
printf("clearPostSetProbability: %f\n", clearPostSetProbability);
printf("coldStartProbability: %f\n", coldStartProbability);
printf("maxColdStarts: %d\n", maxColdStarts);
printf("advanceOldVersionProbability: %f\n", advanceOldVersionProbability);
printf("pageCacheBytes: %s\n", pageCacheBytes == 0 ? "default" : format("%" PRId64, pageCacheBytes).c_str());
printf("versionIncrement: %" PRId64 "\n", versionIncrement);
@ -9534,9 +9646,11 @@ TEST_CASE("Lredwood/correctness/btree") {
state VersionedBTree* btree = new VersionedBTree(pager, file);
wait(btree->init());
state DecodeBoundaryVerifier* pBoundaries = DecodeBoundaryVerifier::getVerifier(file);
state std::map<std::pair<std::string, Version>, Optional<std::string>> written;
state int64_t totalRecordsRead = 0;
state std::set<Key> keys;
state int coldStarts = 0;
state Version lastVer = btree->getLastCommittedVersion();
printf("Starting from version: %" PRId64 "\n", lastVer);
@ -9595,6 +9709,21 @@ TEST_CASE("Lredwood/correctness/btree") {
end = *i;
}
if (!pBoundaries->boundarySamples.empty() &&
deterministicRandom()->random01() < clearKnownNodeBoundaryProbability) {
start = pBoundaries->getSample();
// Can't allow the end boundary to be a start, so just convert to empty string.
if (start == VersionedBTree::dbEnd.key) {
start = Key();
}
}
if (!pBoundaries->boundarySamples.empty() &&
deterministicRandom()->random01() < clearKnownNodeBoundaryProbability) {
end = pBoundaries->getSample();
}
// Do a single key clear based on probability or end being randomly chosen to be the same as begin
// (unlikely)
if (deterministicRandom()->random01() < clearSingleKeyProbability || end == start) {
@ -9730,7 +9859,9 @@ TEST_CASE("Lredwood/correctness/btree") {
mutationBytesTargetThisCommit = randomSize(maxCommitSize);
// Recover from disk at random
if (!pagerMemoryOnly && deterministicRandom()->random01() < coldStartProbability) {
if (!pagerMemoryOnly && coldStarts < maxColdStarts &&
deterministicRandom()->random01() < coldStartProbability) {
++coldStarts;
printf("Recovering from disk after next commit.\n");
// Wait for outstanding commit
@ -10239,7 +10370,7 @@ TEST_CASE(":/redwood/performance/set") {
state Future<Void> stats =
traceMetrics ? Void()
: repeatEvery(1.0, [&]() { printf("Stats:\n%s\n", g_redwoodMetrics.toString(true).c_str()); });
: recurring([&]() { printf("Stats:\n%s\n", g_redwoodMetrics.toString(true).c_str()); }, 1.0);
if (scans > 0) {
printf("Parallel scans, concurrency=%d, scans=%d, scanWidth=%d, scanPreftchBytes=%d ...\n",

View File

@ -45,16 +45,20 @@
#include "fdbclient/WellKnownEndpoints.h"
#include "fdbclient/SimpleIni.h"
#include "fdbrpc/AsyncFileCached.actor.h"
#include "fdbrpc/FlowProcess.actor.h"
#include "fdbrpc/Net2FileSystem.h"
#include "fdbrpc/PerfMetric.h"
#include "fdbrpc/fdbrpc.h"
#include "fdbrpc/simulator.h"
#include "fdbserver/ConflictSet.h"
#include "fdbserver/CoordinationInterface.h"
#include "fdbserver/CoroFlow.h"
#include "fdbserver/DataDistribution.actor.h"
#include "fdbserver/FDBExecHelper.actor.h"
#include "fdbserver/IKeyValueStore.h"
#include "fdbserver/MoveKeys.actor.h"
#include "fdbserver/NetworkTest.h"
#include "fdbserver/RemoteIKeyValueStore.actor.h"
#include "fdbserver/RestoreWorkerInterface.actor.h"
#include "fdbserver/ServerDBInfo.h"
#include "fdbserver/SimulatedCluster.h"
@ -74,10 +78,13 @@
#include "flow/WriteOnlySet.h"
#include "flow/UnitTest.h"
#include "flow/FaultInjection.h"
#include "flow/flow.h"
#include "flow/network.h"
#if defined(__linux__) || defined(__FreeBSD__)
#include <execinfo.h>
#include <signal.h>
#include <sys/prctl.h>
#ifdef ALLOC_INSTRUMENTATION
#include <cxxabi.h>
#endif
@ -100,7 +107,7 @@ enum {
OPT_DCID, OPT_MACHINE_CLASS, OPT_BUGGIFY, OPT_VERSION, OPT_BUILD_FLAGS, OPT_CRASHONERROR, OPT_HELP, OPT_NETWORKIMPL, OPT_NOBUFSTDOUT, OPT_BUFSTDOUTERR,
OPT_TRACECLOCK, OPT_NUMTESTERS, OPT_DEVHELP, OPT_ROLLSIZE, OPT_MAXLOGS, OPT_MAXLOGSSIZE, OPT_KNOB, OPT_UNITTESTPARAM, OPT_TESTSERVERS, OPT_TEST_ON_SERVERS, OPT_METRICSCONNFILE,
OPT_METRICSPREFIX, OPT_LOGGROUP, OPT_LOCALITY, OPT_IO_TRUST_SECONDS, OPT_IO_TRUST_WARN_ONLY, OPT_FILESYSTEM, OPT_PROFILER_RSS_SIZE, OPT_KVFILE,
OPT_TRACE_FORMAT, OPT_WHITELIST_BINPATH, OPT_BLOB_CREDENTIAL_FILE, OPT_CONFIG_PATH, OPT_USE_TEST_CONFIG_DB, OPT_FAULT_INJECTION, OPT_PROFILER, OPT_PRINT_SIMTIME,
OPT_TRACE_FORMAT, OPT_WHITELIST_BINPATH, OPT_BLOB_CREDENTIAL_FILE, OPT_CONFIG_PATH, OPT_USE_TEST_CONFIG_DB, OPT_FAULT_INJECTION, OPT_PROFILER, OPT_PRINT_SIMTIME, OPT_FLOW_PROCESS_NAME, OPT_FLOW_PROCESS_ENDPOINT
};
CSimpleOpt::SOption g_rgOptions[] = {
@ -187,8 +194,10 @@ CSimpleOpt::SOption g_rgOptions[] = {
{ OPT_USE_TEST_CONFIG_DB, "--use-test-config-db", SO_NONE },
{ OPT_FAULT_INJECTION, "-fi", SO_REQ_SEP },
{ OPT_FAULT_INJECTION, "--fault-injection", SO_REQ_SEP },
{ OPT_PROFILER, "--profiler-", SO_REQ_SEP},
{ OPT_PROFILER, "--profiler-", SO_REQ_SEP },
{ OPT_PRINT_SIMTIME, "--print-sim-time", SO_NONE },
{ OPT_FLOW_PROCESS_NAME, "--process-name", SO_REQ_SEP },
{ OPT_FLOW_PROCESS_ENDPOINT, "--process-endpoint", SO_REQ_SEP },
#ifndef TLS_DISABLED
TLS_OPTION_FLAGS
@ -285,6 +294,13 @@ private:
};
UID getSharedMemoryMachineId() {
// new UID to use if an existing one is not found
UID newUID = deterministicRandom()->randomUniqueID();
#if DEBUG_DETERMINISM
// Don't use shared memory if DEBUG_DETERMINISM is set
return newUID;
#else
UID* machineId = nullptr;
int numTries = 0;
@ -297,7 +313,7 @@ UID getSharedMemoryMachineId() {
// "0" is the default parameter "addr"
boost::interprocess::managed_shared_memory segment(
boost::interprocess::open_or_create, sharedMemoryIdentifier.c_str(), 1000, 0, p.permission);
machineId = segment.find_or_construct<UID>("machineId")(deterministicRandom()->randomUniqueID());
machineId = segment.find_or_construct<UID>("machineId")(newUID);
if (!machineId)
criticalError(
FDB_EXIT_ERROR, "SharedMemoryError", "Could not locate or create shared memory - 'machineId'");
@ -321,6 +337,7 @@ UID getSharedMemoryMachineId() {
}
}
}
#endif
}
ACTOR void failAfter(Future<Void> trigger, ISimulator::ProcessInfo* m = g_simulator.getCurrentProcess()) {
@ -959,7 +976,8 @@ enum class ServerRole {
SkipListTest,
Test,
VersionedMapTest,
UnitTests
UnitTests,
FlowProcess
};
struct CLIOptions {
std::string commandLine;
@ -1015,6 +1033,8 @@ struct CLIOptions {
UnitTestParameters testParams;
std::map<std::string, std::string> profilerConfig;
std::string flowProcessName;
Endpoint flowProcessEndpoint;
bool printSimTime = false;
static CLIOptions parseArgs(int argc, char* argv[]) {
@ -1193,6 +1213,8 @@ private:
role = ServerRole::ConsistencyCheck;
else if (!strcmp(sRole, "unittests"))
role = ServerRole::UnitTests;
else if (!strcmp(sRole, "flowprocess"))
role = ServerRole::FlowProcess;
else {
fprintf(stderr, "ERROR: Unknown role `%s'\n", sRole);
printHelpTeaser(argv[0]);
@ -1517,6 +1539,42 @@ private:
case OPT_USE_TEST_CONFIG_DB:
configDBType = ConfigDBType::SIMPLE;
break;
case OPT_FLOW_PROCESS_NAME:
flowProcessName = args.OptionArg();
std::cout << flowProcessName << std::endl;
break;
case OPT_FLOW_PROCESS_ENDPOINT: {
std::vector<std::string> strings;
std::cout << args.OptionArg() << std::endl;
boost::split(strings, args.OptionArg(), [](char c) { return c == ','; });
for (auto& str : strings) {
std::cout << str << " ";
}
std::cout << "\n";
if (strings.size() != 3) {
std::cerr << "Invalid argument, expected 3 elements in --process-endpoint got " << strings.size()
<< std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
try {
auto addr = NetworkAddress::parse(strings[0]);
uint64_t fst = std::stoul(strings[1]);
uint64_t snd = std::stoul(strings[2]);
UID token(fst, snd);
NetworkAddressList l;
l.address = addr;
flowProcessEndpoint = Endpoint(l, token);
std::cout << "flowProcessEndpoint: " << flowProcessEndpoint.getPrimaryAddress().toString()
<< ", token: " << flowProcessEndpoint.token.toString() << "\n";
} catch (Error& e) {
std::cerr << "Could not parse network address " << strings[0] << std::endl;
flushAndExit(FDB_EXIT_ERROR);
} catch (std::exception& e) {
std::cerr << "Could not parse token " << strings[1] << "," << strings[2] << std::endl;
flushAndExit(FDB_EXIT_ERROR);
}
break;
}
case OPT_PRINT_SIMTIME:
printSimTime = true;
break;
@ -1723,6 +1781,7 @@ int main(int argc, char* argv[]) {
role == ServerRole::Simulation ? IsSimulated::True
: IsSimulated::False);
IKnobCollection::getMutableGlobalKnobCollection().setKnob("log_directory", KnobValue::create(opts.logFolder));
IKnobCollection::getMutableGlobalKnobCollection().setKnob("conn_file", KnobValue::create(opts.connFile));
if (role != ServerRole::Simulation) {
IKnobCollection::getMutableGlobalKnobCollection().setKnob("commit_batches_mem_bytes_hard_limit",
KnobValue::create(int64_t{ opts.memLimit }));
@ -1802,8 +1861,8 @@ int main(int argc, char* argv[]) {
FlowTransport::createInstance(false, 1, WLTOKEN_RESERVED_COUNT);
opts.buildNetwork(argv[0]);
const bool expectsPublicAddress =
(role == ServerRole::FDBD || role == ServerRole::NetworkTestServer || role == ServerRole::Restore);
const bool expectsPublicAddress = (role == ServerRole::FDBD || role == ServerRole::NetworkTestServer ||
role == ServerRole::Restore || role == ServerRole::FlowProcess);
if (opts.publicAddressStrs.empty()) {
if (expectsPublicAddress) {
fprintf(stderr, "ERROR: The -p or --public-address option is required\n");
@ -2139,6 +2198,19 @@ int main(int argc, char* argv[]) {
}
f = result;
} else if (role == ServerRole::FlowProcess) {
TraceEvent(SevDebug, "StartingFlowProcess").detail("From", "fdbserver");
#if defined(__linux__) || defined(__FreeBSD__)
prctl(PR_SET_PDEATHSIG, SIGTERM);
if (getppid() == 1) /* parent already died before prctl */
flushAndExit(FDB_EXIT_SUCCESS);
#endif
if (opts.flowProcessName == "KeyValueStoreProcess") {
ProcessFactory<KeyValueStoreProcess>(opts.flowProcessName.c_str());
}
f = stopAfter(runFlowProcess(opts.flowProcessName, opts.flowProcessEndpoint));
g_network->run();
} else if (role == ServerRole::KVFileDump) {
f = stopAfter(KVFileDump(opts.kvFile));
g_network->run();

View File

@ -28,11 +28,13 @@
#include "fdbrpc/LoadBalance.h"
#include "flow/ActorCollection.h"
#include "flow/Arena.h"
#include "flow/Error.h"
#include "flow/Hash3.h"
#include "flow/Histogram.h"
#include "flow/IRandom.h"
#include "flow/IndexedSet.h"
#include "flow/SystemMonitor.h"
#include "flow/Trace.h"
#include "flow/Tracing.h"
#include "flow/Util.h"
#include "fdbclient/Atomic.h"
@ -100,6 +102,9 @@ bool canReplyWith(Error e) {
case error_code_quick_get_value_miss:
case error_code_quick_get_key_values_miss:
case error_code_get_mapped_key_values_has_more:
case error_code_key_not_tuple:
case error_code_value_not_tuple:
case error_code_mapper_not_tuple:
// case error_code_all_alternatives_failed:
return true;
default:
@ -834,6 +839,9 @@ public:
Promise<Void> coreStarted;
bool shuttingDown;
Promise<Void> registerInterfaceAcceptingRequests;
Future<Void> interfaceRegistered;
bool behind;
bool versionBehind;
@ -858,7 +866,7 @@ public:
CounterCollection cc;
Counter allQueries, getKeyQueries, getValueQueries, getRangeQueries, getMappedRangeQueries,
getRangeStreamQueries, finishedQueries, lowPriorityQueries, rowsQueried, bytesQueried, watchQueries,
emptyQueries, feedRowsQueried, feedBytesQueried;
emptyQueries, feedRowsQueried, feedBytesQueried, feedStreamQueries, feedVersionQueries;
// Bytes of the mutations that have been added to the memory of the storage server. When the data is durable
// and cleared from the memory, we do not subtract it but add it to bytesDurable.
@ -930,6 +938,7 @@ public:
lowPriorityQueries("LowPriorityQueries", cc), rowsQueried("RowsQueried", cc),
bytesQueried("BytesQueried", cc), watchQueries("WatchQueries", cc), emptyQueries("EmptyQueries", cc),
feedRowsQueried("FeedRowsQueried", cc), feedBytesQueried("FeedBytesQueried", cc),
feedStreamQueries("FeedStreamQueries", cc), feedVersionQueries("FeedVersionQueries", cc),
bytesInput("BytesInput", cc), logicalBytesInput("LogicalBytesInput", cc),
logicalBytesMoveInOverhead("LogicalBytesMoveInOverhead", cc),
kvCommitLogicalBytes("KVCommitLogicalBytes", cc), kvClearRanges("KVClearRanges", cc),
@ -2436,6 +2445,8 @@ ACTOR Future<Void> changeFeedStreamQ(StorageServer* data, ChangeFeedStreamReques
req.reply.setByteLimit(std::min((int64_t)req.replyBufferSize, SERVER_KNOBS->CHANGEFEEDSTREAM_LIMIT_BYTES));
}
++data->counters.feedStreamQueries;
wait(delay(0, TaskPriority::DefaultEndpoint));
try {
@ -2587,6 +2598,7 @@ ACTOR Future<Void> changeFeedStreamQ(StorageServer* data, ChangeFeedStreamReques
}
ACTOR Future<Void> changeFeedVersionUpdateQ(StorageServer* data, ChangeFeedVersionUpdateRequest req) {
++data->counters.feedVersionQueries;
wait(data->version.whenAtLeast(req.minVersion));
wait(delay(0));
Version minVersion = data->minFeedVersionForAddress(req.reply.getEndpoint().getPrimaryAddress());
@ -3433,14 +3445,24 @@ Key constructMappedKey(KeyValueRef* keyValue, Tuple& mappedKeyFormatTuple, bool&
// Use keyTuple as reference.
if (!keyTuple.present()) {
// May throw exception if the key is not parsable as a tuple.
keyTuple = Tuple::unpack(keyValue->key);
try {
keyTuple = Tuple::unpack(keyValue->key);
} catch (Error& e) {
TraceEvent("KeyNotTuple").error(e).detail("Key", keyValue->key.printable());
throw key_not_tuple();
}
}
referenceTuple = &keyTuple.get();
} else if (s[1] == 'V') {
// Use valueTuple as reference.
if (!valueTuple.present()) {
// May throw exception if the value is not parsable as a tuple.
valueTuple = Tuple::unpack(keyValue->value);
try {
valueTuple = Tuple::unpack(keyValue->value);
} catch (Error& e) {
TraceEvent("ValueNotTuple").error(e).detail("Value", keyValue->value.printable());
throw value_not_tuple();
}
}
referenceTuple = &valueTuple.get();
} else {
@ -3574,7 +3596,13 @@ ACTOR Future<GetMappedKeyValuesReply> mapKeyValues(StorageServer* data,
result.data.reserve(result.arena, input.data.size());
state Tuple mappedKeyFormatTuple = Tuple::unpack(mapper);
state Tuple mappedKeyFormatTuple;
try {
mappedKeyFormatTuple = Tuple::unpack(mapper);
} catch (Error& e) {
TraceEvent("MapperNotTuple").error(e).detail("Mapper", mapper.printable());
throw mapper_not_tuple();
}
state KeyValueRef* it = input.data.begin();
for (; it != input.data.end(); it++) {
state MappedKeyValueRef kvm;
@ -6397,6 +6425,7 @@ ACTOR Future<Void> tssDelayForever() {
ACTOR Future<Void> update(StorageServer* data, bool* pReceivedUpdate) {
state double start;
try {
// If we are disk bound and durableVersion is very old, we need to block updates or we could run out of
// memory. This is often referred to as the storage server e-brake (emergency brake)
@ -6795,6 +6824,16 @@ ACTOR Future<Void> update(StorageServer* data, bool* pReceivedUpdate) {
validate(data);
if ((data->lastTLogVersion - data->version.get()) < SERVER_KNOBS->STORAGE_RECOVERY_VERSION_LAG_LIMIT) {
if (data->registerInterfaceAcceptingRequests.canBeSet()) {
data->registerInterfaceAcceptingRequests.send(Void());
ErrorOr<Void> e = wait(errorOr(data->interfaceRegistered));
if (e.isError()) {
TraceEvent(SevWarn, "StorageInterfaceRegistrationFailed", data->thisServerID).error(e.getError());
}
}
}
data->logCursor->advanceTo(cloneCursor2->version());
if (cursor->version().version >= data->lastTLogVersion) {
if (data->behind) {
@ -8398,7 +8437,8 @@ bool storageServerTerminated(StorageServer& self, IKeyValueStore* persistentData
}
if (e.code() == error_code_worker_removed || e.code() == error_code_recruitment_failed ||
e.code() == error_code_file_not_found || e.code() == error_code_actor_cancelled) {
e.code() == error_code_file_not_found || e.code() == error_code_actor_cancelled ||
e.code() == error_code_remote_kvs_cancelled) {
TraceEvent("StorageServerTerminated", self.thisServerID).errorUnsuppressed(e);
return true;
} else
@ -8468,88 +8508,6 @@ ACTOR Future<Void> initTenantMap(StorageServer* self) {
return Void();
}
// for creating a new storage server
ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
StorageServerInterface ssi,
Tag seedTag,
UID clusterId,
Version tssSeedVersion,
ReplyPromise<InitializeStorageReply> recruitReply,
Reference<AsyncVar<ServerDBInfo> const> db,
std::string folder) {
state StorageServer self(persistentData, db, ssi);
state Future<Void> ssCore;
self.clusterId.send(clusterId);
if (ssi.isTss()) {
self.setTssPair(ssi.tssPairID.get());
ASSERT(self.isTss());
}
self.sk = serverKeysPrefixFor(self.tssPairID.present() ? self.tssPairID.get() : self.thisServerID)
.withPrefix(systemKeys.begin); // FFFF/serverKeys/[this server]/
self.folder = folder;
try {
wait(self.storage.init());
wait(self.storage.commit());
++self.counters.kvCommits;
if (seedTag == invalidTag) {
// Might throw recruitment_failed in case of simultaneous master failure
std::pair<Version, Tag> verAndTag = wait(addStorageServer(self.cx, ssi));
self.tag = verAndTag.second;
if (ssi.isTss()) {
self.setInitialVersion(tssSeedVersion);
} else {
self.setInitialVersion(verAndTag.first - 1);
}
wait(initTenantMap(&self));
} else {
self.tag = seedTag;
}
self.storage.makeNewStorageServerDurable();
wait(self.storage.commit());
++self.counters.kvCommits;
TraceEvent("StorageServerInit", ssi.id())
.detail("Version", self.version.get())
.detail("SeedTag", seedTag.toString())
.detail("TssPair", ssi.isTss() ? ssi.tssPairID.get().toString() : "");
InitializeStorageReply rep;
rep.interf = ssi;
rep.addedVersion = self.version.get();
recruitReply.send(rep);
self.byteSampleRecovery = Void();
ssCore = storageServerCore(&self, ssi);
wait(ssCore);
throw internal_error();
} catch (Error& e) {
// If we die with an error before replying to the recruitment request, send the error to the recruiter
// (ClusterController, and from there to the DataDistributionTeamCollection)
if (!recruitReply.isSet())
recruitReply.sendError(recruitment_failed());
// If the storage server dies while something that uses self is still on the stack,
// we want that actor to complete before we terminate and that memory goes out of scope
state Error err = e;
if (storageServerTerminated(self, persistentData, err)) {
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
return Void();
}
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
throw err;
}
}
ACTOR Future<Void> replaceInterface(StorageServer* self, StorageServerInterface ssi) {
ASSERT(!ssi.isTss());
state Transaction tr(self->cx);
@ -8685,6 +8643,119 @@ ACTOR Future<Void> replaceTSSInterface(StorageServer* self, StorageServerInterfa
return Void();
}
ACTOR Future<Void> storageInterfaceRegistration(StorageServer* self,
StorageServerInterface ssi,
Optional<Future<Void>> readyToAcceptRequests) {
if (readyToAcceptRequests.present()) {
wait(readyToAcceptRequests.get());
ssi.startAcceptingRequests();
} else {
ssi.stopAcceptingRequests();
}
try {
if (self->isTss()) {
wait(replaceTSSInterface(self, ssi));
} else {
wait(replaceInterface(self, ssi));
}
} catch (Error& e) {
throw;
}
return Void();
}
// for creating a new storage server
ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
StorageServerInterface ssi,
Tag seedTag,
UID clusterId,
Version tssSeedVersion,
ReplyPromise<InitializeStorageReply> recruitReply,
Reference<AsyncVar<ServerDBInfo> const> db,
std::string folder) {
state StorageServer self(persistentData, db, ssi);
state Future<Void> ssCore;
self.clusterId.send(clusterId);
if (ssi.isTss()) {
self.setTssPair(ssi.tssPairID.get());
ASSERT(self.isTss());
}
self.sk = serverKeysPrefixFor(self.tssPairID.present() ? self.tssPairID.get() : self.thisServerID)
.withPrefix(systemKeys.begin); // FFFF/serverKeys/[this server]/
self.folder = folder;
try {
wait(self.storage.init());
wait(self.storage.commit());
++self.counters.kvCommits;
if (seedTag == invalidTag) {
ssi.startAcceptingRequests();
self.registerInterfaceAcceptingRequests.send(Void());
// Might throw recruitment_failed in case of simultaneous master failure
std::pair<Version, Tag> verAndTag = wait(addStorageServer(self.cx, ssi));
self.tag = verAndTag.second;
if (ssi.isTss()) {
self.setInitialVersion(tssSeedVersion);
} else {
self.setInitialVersion(verAndTag.first - 1);
}
wait(initTenantMap(&self));
} else {
self.tag = seedTag;
}
self.storage.makeNewStorageServerDurable();
wait(self.storage.commit());
++self.counters.kvCommits;
self.interfaceRegistered =
storageInterfaceRegistration(&self, ssi, self.registerInterfaceAcceptingRequests.getFuture());
wait(delay(0));
TraceEvent("StorageServerInit", ssi.id())
.detail("Version", self.version.get())
.detail("SeedTag", seedTag.toString())
.detail("TssPair", ssi.isTss() ? ssi.tssPairID.get().toString() : "");
InitializeStorageReply rep;
rep.interf = ssi;
rep.addedVersion = self.version.get();
recruitReply.send(rep);
self.byteSampleRecovery = Void();
ssCore = storageServerCore(&self, ssi);
wait(ssCore);
throw internal_error();
} catch (Error& e) {
// If we die with an error before replying to the recruitment request, send the error to the recruiter
// (ClusterController, and from there to the DataDistributionTeamCollection)
if (!recruitReply.isSet())
recruitReply.sendError(recruitment_failed());
// If the storage server dies while something that uses self is still on the stack,
// we want that actor to complete before we terminate and that memory goes out of scope
state Error err = e;
if (storageServerTerminated(self, persistentData, err)) {
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
return Void();
}
ssCore.cancel();
self.actors.clear(true);
wait(delay(0));
throw err;
}
}
// for recovering an existing storage server
ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
StorageServerInterface ssi,
@ -8740,15 +8811,14 @@ ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
if (recovered.canBeSet())
recovered.send(Void());
try {
if (self.isTss()) {
wait(replaceTSSInterface(&self, ssi));
} else {
wait(replaceInterface(&self, ssi));
}
} catch (Error& e) {
state Future<Void> f = storageInterfaceRegistration(&self, ssi, {});
wait(delay(0));
ErrorOr<Void> e = wait(errorOr(f));
if (e.isError()) {
Error e = f.getError();
if (e.code() != error_code_worker_removed) {
throw;
throw e;
}
state UID clusterId = wait(getClusterId(&self));
ASSERT(self.clusterId.isValid());
@ -8762,15 +8832,19 @@ ACTOR Future<Void> storageServer(IKeyValueStore* persistentData,
// We want to avoid this and force a manual removal of the storage
// servers' old data when being assigned to a new cluster to avoid
// accidental data loss.
TraceEvent(SevError, "StorageServerBelongsToExistingCluster")
TraceEvent(SevWarn, "StorageServerBelongsToExistingCluster")
.detail("ServerID", ssi.id())
.detail("ClusterID", durableClusterId)
.detail("NewClusterID", clusterId);
wait(Future<Void>(Never()));
}
self.interfaceRegistered =
storageInterfaceRegistration(&self, ssi, self.registerInterfaceAcceptingRequests.getFuture());
wait(delay(0));
TraceEvent("StorageServerStartingCore", self.thisServerID).detail("TimeTaken", now() - start);
// wait( delay(0) ); // To make sure self->zkMasterInfo.onChanged is available to wait on
ssCore = storageServerCore(&self, ssi);
wait(ssCore);

View File

@ -1106,7 +1106,8 @@ std::map<std::string, std::function<void(const std::string&)>> testSpecGlobalKey
[](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedMaxTLogVersion", ""); } },
{ "disableTss", [](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedDisableTSS", ""); } },
{ "disableHostname",
[](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedDisableHostname", ""); } }
[](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedDisableHostname", ""); } },
{ "disableRemoteKVS", [](const std::string& value) { TraceEvent("TestParserTest").detail("ParsedRemoteKVS", ""); } }
};
std::map<std::string, std::function<void(const std::string& value, TestSpec* spec)>> testSpecTestKeys = {

Some files were not shown because too many files have changed in this diff Show More