foundationdb/fdbserver/workloads/BackupAndParallelRestoreCor...

804 lines
34 KiB
C++

/*
* BackupAndParallelRestoreCorrectness.cpp
*
* This source file is part of the FoundationDB open source project
*
* Copyright 2013-2026 Apple Inc. and the FoundationDB project authors
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "fdbrpc/simulator.h"
#include "fdbclient/BackupAgent.h"
#include "fdbclient/BackupContainer.h"
#include "fdbclient/BackupContainerFileSystem.h"
#include "fdbclient/ManagementAPI.h"
#include "fdbserver/restoreworker/RestoreWorkerInterface.h"
#include "fdbclient/RunRYWTransaction.h"
#include "fdbserver/restoreworker/RestoreCommon.h"
#include "fdbserver/tester/workloads.h"
#include "fdbserver/tester/TestEncryptionUtils.h"
#include "BulkSetup.h"
#define TEST_ABORT_FASTRESTORE 0
struct BackupAndParallelRestoreCorrectnessWorkload : TestWorkload {
static constexpr auto NAME = "BackupAndParallelRestoreCorrectness";
double backupAfter, restoreAfter, abortAndRestartAfter;
double backupStartAt, restoreStartAfterBackupFinished, stopDifferentialAfter;
Key backupTag;
int backupRangesCount, backupRangeLengthMax;
bool differentialBackup, performRestore, agentRequest;
Standalone<VectorRef<KeyRangeRef>> backupRanges;
static int backupAgentRequests;
LockDB locked{ false };
bool allowPauses;
bool shareLogRange;
UsePartitionedLog usePartitionedLogs{ false };
Key addPrefix, removePrefix; // Original key will be first applied removePrefix and then applied addPrefix
// CAVEAT: When removePrefix is used, we must ensure every key in backup have the removePrefix
Optional<std::string> encryptionKeyFileName;
std::map<Standalone<KeyRef>, Standalone<ValueRef>> dbKVs;
// This workload is not compatible with RandomRangeLock workload because they will race in locked range
void disableFailureInjectionWorkloads(std::set<std::string>& out) const override {
out.insert({ "RandomRangeLock" });
}
BackupAndParallelRestoreCorrectnessWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) {
locked.set(sharedRandomNumber % 2);
backupAfter = getOption(options, "backupAfter"_sr, 10.0);
restoreAfter = getOption(options, "restoreAfter"_sr, 35.0);
performRestore = getOption(options, "performRestore"_sr, true);
backupTag = getOption(options, "backupTag"_sr, BackupAgentBase::getDefaultTag());
backupRangesCount = getOption(options, "backupRangesCount"_sr, 5);
backupRangeLengthMax = getOption(options, "backupRangeLengthMax"_sr, 1);
abortAndRestartAfter =
getOption(options,
"abortAndRestartAfter"_sr,
deterministicRandom()->random01() < 0.5
? deterministicRandom()->random01() * (restoreAfter - backupAfter) + backupAfter
: 0.0);
differentialBackup =
getOption(options, "differentialBackup"_sr, deterministicRandom()->random01() < 0.5 ? true : false);
stopDifferentialAfter =
getOption(options,
"stopDifferentialAfter"_sr,
differentialBackup ? deterministicRandom()->random01() *
(restoreAfter - std::max(abortAndRestartAfter, backupAfter)) +
std::max(abortAndRestartAfter, backupAfter)
: 0.0);
agentRequest = getOption(options, "simBackupAgents"_sr, true);
allowPauses = getOption(options, "allowPauses"_sr, true);
shareLogRange = getOption(options, "shareLogRange"_sr, false);
usePartitionedLogs.set(getOption(options, "usePartitionedLogs"_sr, deterministicRandom()->coinflip()));
addPrefix = getOption(options, "addPrefix"_sr, ""_sr);
removePrefix = getOption(options, "removePrefix"_sr, ""_sr);
if (getOption(options, "encrypted"_sr, deterministicRandom()->random01() < 0.5)) {
encryptionKeyFileName = "simfdb/" + getTestEncryptionFileName();
}
KeyRef beginRange;
KeyRef endRange;
UID randomID = nondeterministicRandom()->randomUniqueID();
// Correctness is not clean for addPrefix feature yet. Uncomment below to enable the test
// Generate addPrefix
// if (addPrefix.size() == 0 && removePrefix.size() == 0) {
// if (deterministicRandom()->random01() < 0.5) { // Generate random addPrefix
// int len = deterministicRandom()->randomInt(1, 100);
// std::string randomStr = deterministicRandom()->randomAlphaNumeric(len);
// TraceEvent("BackupAndParallelRestoreCorrectness")
// .detail("GenerateAddPrefix", randomStr)
// .detail("Length", len)
// .detail("StrLen", randomStr.size());
// addPrefix = Key(randomStr);
// }
// }
TraceEvent("BackupAndParallelRestoreCorrectness")
.detail("AddPrefix", addPrefix)
.detail("RemovePrefix", removePrefix);
ASSERT(addPrefix.size() == 0 && removePrefix.size() == 0);
// Do not support removePrefix right now because we must ensure all backup keys have the removePrefix
// otherwise, test will fail because fast restore will simply add the removePrefix to every key in the end.
ASSERT(removePrefix.size() == 0);
if (shareLogRange) {
bool beforePrefix = sharedRandomNumber & 1;
if (beforePrefix)
backupRanges.push_back_deep(backupRanges.arena(), KeyRangeRef(normalKeys.begin, "\xfe\xff\xfe"_sr));
else
backupRanges.push_back_deep(backupRanges.arena(),
KeyRangeRef(strinc("\x00\x00\x01"_sr), normalKeys.end));
} else if (backupRangesCount <= 0) {
backupRanges.push_back_deep(backupRanges.arena(), normalKeys);
} else {
// Add backup ranges
std::set<std::string> rangeEndpoints;
while (rangeEndpoints.size() < backupRangesCount * 2) {
rangeEndpoints.insert(deterministicRandom()->randomAlphaNumeric(
deterministicRandom()->randomInt(1, backupRangeLengthMax + 1)));
}
// Create ranges from the keys, in order, to prevent overlaps
std::vector<std::string> sortedEndpoints(rangeEndpoints.begin(), rangeEndpoints.end());
sort(sortedEndpoints.begin(), sortedEndpoints.end());
for (auto i = sortedEndpoints.begin(); i != sortedEndpoints.end(); ++i) {
const std::string& start = *i++;
backupRanges.push_back_deep(backupRanges.arena(), KeyRangeRef(start, *i));
// Track the added range
TraceEvent("BARW_BackupCorrectnessRange", randomID)
.detail("RangeBegin", (beginRange < endRange) ? printable(beginRange) : printable(endRange))
.detail("RangeEnd", (beginRange < endRange) ? printable(endRange) : printable(beginRange));
}
}
}
Future<Void> setup(Database const& cx) override { return Void(); }
Future<Void> start(Database const& cx) override {
if (clientId != 0)
return Void();
TraceEvent(SevInfo, "BARW_Param").detail("Locked", locked);
TraceEvent(SevInfo, "BARW_Param").detail("BackupAfter", backupAfter);
TraceEvent(SevInfo, "BARW_Param").detail("RestoreAfter", restoreAfter);
TraceEvent(SevInfo, "BARW_Param").detail("PerformRestore", performRestore);
TraceEvent(SevInfo, "BARW_Param").detail("BackupTag", printable(backupTag).c_str());
TraceEvent(SevInfo, "BARW_Param").detail("BackupRangesCount", backupRangesCount);
TraceEvent(SevInfo, "BARW_Param").detail("BackupRangeLengthMax", backupRangeLengthMax);
TraceEvent(SevInfo, "BARW_Param").detail("AbortAndRestartAfter", abortAndRestartAfter);
TraceEvent(SevInfo, "BARW_Param").detail("DifferentialBackup", differentialBackup);
TraceEvent(SevInfo, "BARW_Param").detail("StopDifferentialAfter", stopDifferentialAfter);
TraceEvent(SevInfo, "BARW_Param").detail("AgentRequest", agentRequest);
TraceEvent(SevInfo, "BARW_Param").detail("Encrypted", encryptionKeyFileName.present());
return _start(cx);
}
bool hasPrefix() const { return addPrefix != ""_sr || removePrefix != ""_sr; }
Future<bool> check(Database const& cx) override { return true; }
void getMetrics(std::vector<PerfMetric>& m) override {}
static Future<Void> changePaused(Database cx, FileBackupAgent* backupAgent) {
while (true) {
co_await backupAgent->changePause(cx, true);
co_await delay(30 * deterministicRandom()->random01());
co_await backupAgent->changePause(cx, false);
co_await delay(120 * deterministicRandom()->random01());
}
}
static Future<Void> statusLoop(Database cx, std::string tag) {
FileBackupAgent agent;
while (true) {
std::string status = co_await agent.getStatus(cx, ShowErrors::True, tag);
puts(status.c_str());
co_await delay(2.0);
}
}
Future<Void> doBackup(double startDelay,
FileBackupAgent* backupAgent,
Database cx,
Key tag,
Standalone<VectorRef<KeyRangeRef>> backupRanges,
double stopDifferentialDelay,
Promise<Void> submitted) {
UID randomID = nondeterministicRandom()->randomUniqueID();
Future<Void> stopDifferentialFuture = delay(stopDifferentialDelay);
co_await delay(startDelay);
if (startDelay || BUGGIFY) {
TraceEvent("BARW_DoBackupAbortBackup1", randomID)
.detail("Tag", printable(tag))
.detail("StartDelay", startDelay);
try {
co_await backupAgent->abortBackup(cx, tag.toString());
} catch (Error& e) {
TraceEvent("BARW_DoBackupAbortBackupException", randomID).error(e).detail("Tag", printable(tag));
if (e.code() != error_code_backup_unneeded)
throw;
}
}
TraceEvent("BARW_DoBackupSubmitBackup", randomID)
.detail("Tag", printable(tag))
.detail("StopWhenDone", stopDifferentialDelay ? "False" : "True");
std::string backupContainer = "file://simfdb/backups/";
Future<Void> status = statusLoop(cx, tag.toString());
try {
co_await backupAgent->submitBackup(cx,
StringRef(backupContainer),
{},
deterministicRandom()->randomInt(0, 60),
deterministicRandom()->randomInt(0, 100),
tag.toString(),
backupRanges,
StopWhenDone{ !stopDifferentialDelay },
usePartitionedLogs,
IncrementalBackupOnly::False,
encryptionKeyFileName);
} catch (Error& e) {
TraceEvent("BARW_DoBackupSubmitBackupException", randomID).error(e).detail("Tag", printable(tag));
if (e.code() != error_code_backup_unneeded && e.code() != error_code_backup_duplicate)
throw;
}
submitted.send(Void());
// Stop the differential backup, if enabled
if (stopDifferentialDelay) {
CODE_PROBE(!stopDifferentialFuture.isReady(),
"Restore starts at specified time - stopDifferential not ready");
co_await stopDifferentialFuture;
TraceEvent("BARW_DoBackupWaitToDiscontinue", randomID)
.detail("Tag", printable(tag))
.detail("DifferentialAfter", stopDifferentialDelay);
try {
if (BUGGIFY) {
KeyBackedTag backupTag = makeBackupTag(tag.toString());
TraceEvent("BARW_DoBackupWaitForRestorable", randomID).detail("Tag", backupTag.tagName);
// Wait until the backup is in a restorable state and get the status, URL, and UID atomically
Reference<IBackupContainer> lastBackupContainer;
UID lastBackupUID;
EBackupState resultWait = co_await backupAgent->waitBackup(
cx, backupTag.tagName, StopWhenDone::False, &lastBackupContainer, &lastBackupUID);
TraceEvent("BARW_DoBackupWaitForRestorable", randomID)
.detail("Tag", backupTag.tagName)
.detail("Result", BackupAgentBase::getStateText(resultWait));
bool restorable = false;
if (lastBackupContainer) {
Future<BackupDescription> fdesc = lastBackupContainer->describeBackup();
co_await ready(fdesc);
if (!fdesc.isError()) {
BackupDescription desc = fdesc.get();
co_await desc.resolveVersionTimes(cx);
printf("BackupDescription:\n%s\n", desc.toString().c_str());
restorable = desc.maxRestorableVersion.present();
}
}
TraceEvent("BARW_LastBackupContainer", randomID)
.detail("BackupTag", printable(tag))
.detail("LastBackupContainer", lastBackupContainer ? lastBackupContainer->getURL() : "")
.detail("LastBackupUID", lastBackupUID)
.detail("WaitStatus", BackupAgentBase::getStateText(resultWait))
.detail("Restorable", restorable);
// Do not check the backup, if aborted
if (resultWait == EBackupState::STATE_ABORTED) {
}
// Ensure that a backup container was found
else if (!lastBackupContainer) {
TraceEvent(SevError, "BARW_MissingBackupContainer", randomID)
.detail("LastBackupUID", lastBackupUID)
.detail("BackupTag", printable(tag))
.detail("WaitStatus", resultWait);
printf("BackupCorrectnessMissingBackupContainer tag: %s status: %s\n",
printable(tag).c_str(),
BackupAgentBase::getStateText(resultWait));
}
// Check that backup is restorable
else if (!restorable) {
TraceEvent(SevError, "BARW_NotRestorable", randomID)
.detail("LastBackupUID", lastBackupUID)
.detail("BackupTag", printable(tag))
.detail("BackupFolder", lastBackupContainer->getURL())
.detail("WaitStatus", BackupAgentBase::getStateText(resultWait));
printf("BackupCorrectnessNotRestorable: tag: %s\n", printable(tag).c_str());
}
// Abort the backup, if not the first backup because the second backup may have aborted the backup
// by now
if (startDelay) {
TraceEvent("BARW_DoBackupAbortBackup2", randomID)
.detail("Tag", printable(tag))
.detail("WaitStatus", BackupAgentBase::getStateText(resultWait))
.detail("LastBackupContainer", lastBackupContainer ? lastBackupContainer->getURL() : "")
.detail("Restorable", restorable);
co_await backupAgent->abortBackup(cx, tag.toString());
} else {
TraceEvent("BARW_DoBackupDiscontinueBackup", randomID)
.detail("Tag", printable(tag))
.detail("DifferentialAfter", stopDifferentialDelay);
co_await backupAgent->discontinueBackup(cx, tag);
}
}
else {
TraceEvent("BARW_DoBackupDiscontinueBackup", randomID)
.detail("Tag", printable(tag))
.detail("DifferentialAfter", stopDifferentialDelay);
co_await backupAgent->discontinueBackup(cx, tag);
}
} catch (Error& e) {
TraceEvent("BARW_DoBackupDiscontinueBackupException", randomID).error(e).detail("Tag", printable(tag));
if (e.code() != error_code_backup_unneeded && e.code() != error_code_backup_duplicate)
throw;
}
}
// Wait for the backup to complete
TraceEvent("BARW_DoBackupWaitBackup", randomID).detail("Tag", printable(tag));
EBackupState statusValue = co_await backupAgent->waitBackup(cx, tag.toString(), StopWhenDone::True);
std::string statusText;
std::string _statusText = co_await backupAgent->getStatus(cx, ShowErrors::True, tag.toString());
statusText = _statusText;
// Can we validate anything about status?
TraceEvent("BARW_DoBackupComplete", randomID)
.detail("Tag", printable(tag))
.detail("Status", statusText)
.detail("StatusValue", BackupAgentBase::getStateText(statusValue));
}
// This actor attempts to restore the database without clearing the keyspace.
// TODO: Enable this function in correctness test
Future<Void> attemptDirtyRestore(Database cx,
FileBackupAgent* backupAgent,
Standalone<StringRef> lastBackupContainer,
UID randomID) {
Transaction tr(cx);
int rowCount = 0;
while (true) {
Error err;
try {
RangeResult existingRows = co_await tr.getRange(normalKeys, 1);
rowCount = existingRows.size();
break;
} catch (Error& e) {
err = e;
}
co_await tr.onError(err);
}
// Try doing a restore without clearing the keys
if (rowCount > 0) {
try {
// TODO: Change to my restore agent code
TraceEvent(SevError, "MXFastRestore").detail("RestoreFunction", "ShouldChangeToMyOwnRestoreLogic");
co_await backupAgent->restore(cx,
cx,
backupTag,
KeyRef(lastBackupContainer),
{},
WaitForComplete::True,
::invalidVersion,
Verbose::True,
normalKeys,
Key(),
Key(),
locked,
OnlyApplyMutationLogs::False,
InconsistentSnapshotOnly::False,
::invalidVersion,
encryptionKeyFileName);
TraceEvent(SevError, "BARW_RestoreAllowedOverwrittingDatabase", randomID).log();
ASSERT(false);
} catch (Error& e) {
if (e.code() != error_code_restore_destination_not_empty) {
throw;
}
}
}
}
Future<Void> _start(Database cx) {
FileBackupAgent backupAgent;
Future<Void> extraBackup;
UID randomID = nondeterministicRandom()->randomUniqueID();
int restoreIndex = 0;
ReadYourWritesTransaction tr2(cx);
TraceEvent("BARW_Arguments")
.detail("BackupTag", printable(backupTag))
.detail("PerformRestore", performRestore)
.detail("BackupAfter", backupAfter)
.detail("RestoreAfter", restoreAfter)
.detail("AbortAndRestartAfter", abortAndRestartAfter)
.detail("DifferentialAfter", stopDifferentialAfter);
if (allowPauses && BUGGIFY) {
Future<Void> cp = changePaused(cx, &backupAgent);
}
// Increment the backup agent requests
if (agentRequest) {
BackupAndParallelRestoreCorrectnessWorkload::backupAgentRequests++;
}
if (encryptionKeyFileName.present()) {
co_await BackupContainerFileSystem::createTestEncryptionKeyFile(encryptionKeyFileName.get());
}
try {
Future<Void> startRestore = delay(restoreAfter);
// backup
co_await delay(backupAfter);
TraceEvent("BARW_DoBackup1", randomID).detail("Tag", printable(backupTag));
Promise<Void> submitted;
Future<Void> b = doBackup(0, &backupAgent, cx, backupTag, backupRanges, stopDifferentialAfter, submitted);
if (abortAndRestartAfter) {
TraceEvent("BARW_DoBackup2", randomID)
.detail("Tag", printable(backupTag))
.detail("AbortWait", abortAndRestartAfter);
co_await submitted.getFuture();
b = b && doBackup(abortAndRestartAfter,
&backupAgent,
cx,
backupTag,
backupRanges,
stopDifferentialAfter,
Promise<Void>());
}
TraceEvent("BARW_DoBackupWait", randomID)
.detail("BackupTag", printable(backupTag))
.detail("AbortAndRestartAfter", abortAndRestartAfter);
try {
co_await b;
} catch (Error& e) {
if (e.code() != error_code_database_locked)
throw;
if (performRestore)
throw;
co_return;
}
TraceEvent("BARW_DoBackupDone", randomID)
.detail("BackupTag", printable(backupTag))
.detail("AbortAndRestartAfter", abortAndRestartAfter);
KeyBackedTag keyBackedTag = makeBackupTag(backupTag.toString());
UidAndAbortedFlagT uidFlag = co_await keyBackedTag.getOrThrow(cx.getReference());
UID logUid = uidFlag.first;
Key destUidValue = co_await BackupConfig(logUid).destUidValue().getD(cx.getReference());
Reference<IBackupContainer> lastBackupContainer =
co_await BackupConfig(logUid).backupContainer().getD(cx.getReference());
// Occasionally start yet another backup that might still be running when we restore
if (!locked && BUGGIFY) {
TraceEvent("BARW_SubmitBackup2", randomID).detail("Tag", printable(backupTag));
try {
// Note the "partitionedLog" must be false, because we change
// the configuration to disable backup workers before restore.
extraBackup = backupAgent.submitBackup(cx,
"file://simfdb/backups/"_sr,
{},
deterministicRandom()->randomInt(0, 60),
deterministicRandom()->randomInt(0, 100),
backupTag.toString(),
backupRanges,
StopWhenDone::True,
UsePartitionedLog::False,
IncrementalBackupOnly::False,
encryptionKeyFileName);
} catch (Error& e) {
TraceEvent("BARW_SubmitBackup2Exception", randomID)
.error(e)
.detail("BackupTag", printable(backupTag));
if (e.code() != error_code_backup_unneeded && e.code() != error_code_backup_duplicate)
throw;
}
}
CODE_PROBE(!startRestore.isReady(), "Restore starts at specified time");
co_await startRestore;
if (lastBackupContainer && performRestore) {
if (deterministicRandom()->random01() < 0.5) {
printf("TODO: Check if restore can succeed if dirty restore is performed first\n");
// TODO: To support restore even after we attempt dirty restore. Not implemented in the 1st version
// fast restore
// wait(attemptDirtyRestore(cx, &backupAgent, StringRef(lastBackupContainer->getURL()),
// randomID));
}
// We must ensure no backup workers are running, otherwise the clear DB
// below can be picked up by backup workers and applied during restore.
co_await ManagementAPI::changeConfig(cx.getReference(), "backup_worker_enabled:=0", true);
// Clear DB before restore
co_await runRYWTransaction(cx, [=](Reference<ReadYourWritesTransaction> tr) -> Future<Void> {
for (auto& kvrange : backupRanges)
tr->clear(kvrange);
return Void();
});
// restore database
TraceEvent("BAFRW_Restore", randomID)
.detail("LastBackupContainer", lastBackupContainer->getURL())
.detail("RestoreAfter", restoreAfter)
.detail("BackupTag", printable(backupTag));
// start restoring
auto container = IBackupContainer::openContainer(lastBackupContainer->getURL(),
lastBackupContainer->getProxy(),
lastBackupContainer->getEncryptionKeyFileName());
BackupDescription desc = co_await container->describeBackup();
ASSERT(usePartitionedLogs == desc.partitioned);
ASSERT(desc.minRestorableVersion.present()); // We must have a valid backup now.
Version targetVersion = -1;
if (desc.maxRestorableVersion.present()) {
if (deterministicRandom()->random01() < 0.1) {
targetVersion = desc.minRestorableVersion.get();
} else if (deterministicRandom()->random01() < 0.1) {
targetVersion = desc.maxRestorableVersion.get();
} else if (deterministicRandom()->random01() < 0.5) {
targetVersion = (desc.minRestorableVersion.get() != desc.maxRestorableVersion.get())
? deterministicRandom()->randomInt64(desc.minRestorableVersion.get(),
desc.maxRestorableVersion.get())
: desc.maxRestorableVersion.get();
}
}
TraceEvent("BAFRW_Restore", randomID)
.detail("LastBackupContainer", lastBackupContainer->getURL())
.detail("MinRestorableVersion", desc.minRestorableVersion.get())
.detail("MaxRestorableVersion", desc.maxRestorableVersion.get())
.detail("ContiguousLogEnd", desc.contiguousLogEnd.get())
.detail("TargetVersion", targetVersion);
std::vector<Future<Version>> restores;
std::vector<Standalone<StringRef>> restoreTags;
// Submit parallel restore requests
TraceEvent("BackupAndParallelRestoreWorkload")
.detail("PrepareRestores", backupRanges.size())
.detail("AddPrefix", addPrefix)
.detail("RemovePrefix", removePrefix);
co_await backupAgent.submitParallelRestore(cx,
backupTag,
backupRanges,
KeyRef(lastBackupContainer->getURL()),
lastBackupContainer->getProxy(),
targetVersion,
locked,
randomID,
addPrefix,
removePrefix);
TraceEvent("BackupAndParallelRestoreWorkload")
.detail("TriggerRestore", "Setting up restoreRequestTriggerKey");
// Sometimes kill and restart the restore
// In real cluster, aborting a restore needs:
// (1) kill restore cluster; (2) clear dest. DB restore system keyspace.
// TODO: Consider gracefully abort a restore and restart.
if (BUGGIFY && TEST_ABORT_FASTRESTORE) {
TraceEvent(SevError, "FastRestore").detail("Buggify", "NotImplementedYet");
co_await delay(deterministicRandom()->randomInt(0, 10));
for (restoreIndex = 0; restoreIndex < restores.size(); restoreIndex++) {
FileBackupAgent::ERestoreState rs =
co_await backupAgent.abortRestore(cx, restoreTags[restoreIndex]);
// The restore may have already completed, or the abort may have been done before the restore
// was even able to start. Only run a new restore if the previous one was actually aborted.
if (rs == FileBackupAgent::ERestoreState::ABORTED) {
co_await runRYWTransaction(cx,
[=](Reference<ReadYourWritesTransaction> tr) -> Future<Void> {
tr->clear(backupRanges[restoreIndex]);
return Void();
});
// TODO: Not Implemented yet
// restores[restoreIndex] = backupAgent.restore(cx, restoreTags[restoreIndex],
// KeyRef(lastBackupContainer->getURL()), true, -1, true, backupRanges[restoreIndex],
// Key(), Key(), locked);
}
}
}
// Wait for parallel restore to finish before we can proceed
TraceEvent("FastRestoreWorkload").detail("WaitForRestoreToFinish", randomID);
// Do not unlock DB when restore finish because we need to transformDatabaseContents
co_await backupAgent.parallelRestoreFinish(cx, randomID, UnlockDB{ !hasPrefix() });
TraceEvent("FastRestoreWorkload").detail("RestoreFinished", randomID);
for (auto& restore : restores) {
ASSERT(!restore.isError());
}
// If addPrefix or removePrefix set, we want to transform the effect by copying data
if (hasPrefix()) {
co_await transformRestoredDatabase(cx, backupRanges, addPrefix, removePrefix);
co_await unlockDatabase(cx, randomID);
}
}
// Q: What is the extra backup and why do we need to care about it?
if (extraBackup.isValid()) { // SOMEDAY: Handle this case
TraceEvent("BARW_WaitExtraBackup", randomID).detail("BackupTag", printable(backupTag));
try {
co_await extraBackup;
} catch (Error& e) {
TraceEvent("BARW_ExtraBackupException", randomID)
.error(e)
.detail("BackupTag", printable(backupTag));
if (e.code() != error_code_backup_unneeded && e.code() != error_code_backup_duplicate)
throw;
}
TraceEvent("BARW_AbortBackupExtra", randomID).detail("BackupTag", printable(backupTag));
try {
co_await backupAgent.abortBackup(cx, backupTag.toString());
} catch (Error& e) {
TraceEvent("BARW_AbortBackupExtraException", randomID).error(e);
if (e.code() != error_code_backup_unneeded)
throw;
}
}
Key backupAgentKey = uidPrefixKey(logRangesRange.begin, logUid);
Key backupLogValuesKey = destUidValue.withPrefix(backupLogKeys.begin);
Key backupLatestVersionsPath = destUidValue.withPrefix(backupLatestVersionsPrefix);
Key backupLatestVersionsKey = uidPrefixKey(backupLatestVersionsPath, logUid);
int displaySystemKeys = 0;
// Ensure that there is no left over key within the backup subspace
while (true) {
Reference<ReadYourWritesTransaction> tr(new ReadYourWritesTransaction(cx));
TraceEvent("BARW_CheckLeftoverKeys", randomID).detail("BackupTag", printable(backupTag));
Error err;
try {
tr->reset();
tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS);
// Check the left over tasks
// We have to wait for the list to empty since an abort and get status
// can leave extra tasks in the queue
TraceEvent("BARW_CheckLeftoverTasks", randomID).detail("BackupTag", printable(backupTag));
int64_t taskCount = co_await backupAgent.getTaskCount(tr);
int waitCycles = 0;
while (true) {
waitCycles++;
TraceEvent("BARW_NonzeroTaskWait", randomID)
.detail("BackupTag", printable(backupTag))
.detail("TaskCount", taskCount)
.detail("WaitCycles", waitCycles);
printf("%.6f %-10s Wait #%4d for %lld tasks to end\n",
now(),
randomID.toString().c_str(),
waitCycles,
(long long)taskCount);
co_await delay(5.0);
co_await tr->commit();
tr = makeReference<ReadYourWritesTransaction>(cx);
int64_t _taskCount = co_await backupAgent.getTaskCount(tr);
taskCount = _taskCount;
if (!taskCount) {
break;
}
}
if (taskCount) {
displaySystemKeys++;
TraceEvent(SevError, "BARW_NonzeroTaskCount", randomID)
.detail("BackupTag", printable(backupTag))
.detail("TaskCount", taskCount)
.detail("WaitCycles", waitCycles);
printf("BackupCorrectnessLeftOverLogTasks: %ld\n", (long)taskCount);
}
RangeResult agentValues =
co_await tr->getRange(KeyRange(KeyRangeRef(backupAgentKey, strinc(backupAgentKey))), 100);
// Error if the system keyspace for the backup tag is not empty
if (agentValues.size() > 0) {
displaySystemKeys++;
printf("BackupCorrectnessLeftOverMutationKeys: (%d) %s\n",
agentValues.size(),
printable(backupAgentKey).c_str());
TraceEvent(SevError, "BackupCorrectnessLeftOverMutationKeys", randomID)
.detail("BackupTag", printable(backupTag))
.detail("LeftOverKeys", agentValues.size())
.detail("KeySpace", printable(backupAgentKey));
for (auto& s : agentValues) {
TraceEvent("BARW_LeftOverKey", randomID)
.detail("Key", printable(StringRef(s.key.toString())))
.detail("Value", printable(StringRef(s.value.toString())));
printf(" Key: %-50s Value: %s\n",
printable(StringRef(s.key.toString())).c_str(),
printable(StringRef(s.value.toString())).c_str());
}
} else {
printf("No left over backup agent configuration keys\n");
}
Optional<Value> latestVersion = co_await tr->get(backupLatestVersionsKey);
if (latestVersion.present()) {
TraceEvent(SevError, "BackupCorrectnessLeftOverVersionKey", randomID)
.detail("BackupTag", printable(backupTag))
.detail("BackupLatestVersionsKey", backupLatestVersionsKey.printable())
.detail("DestUidValue", destUidValue.printable());
} else {
printf("No left over backup version key\n");
}
RangeResult versions = co_await tr->getRange(
KeyRange(KeyRangeRef(backupLatestVersionsPath, strinc(backupLatestVersionsPath))), 1);
if (!shareLogRange || !versions.size()) {
RangeResult logValues = co_await tr->getRange(
KeyRange(KeyRangeRef(backupLogValuesKey, strinc(backupLogValuesKey))), 100);
// Error if the log/mutation keyspace for the backup tag is not empty
if (logValues.size() > 0) {
displaySystemKeys++;
printf("BackupCorrectnessLeftOverLogKeys: (%d) %s\n",
logValues.size(),
printable(backupLogValuesKey).c_str());
TraceEvent(SevError, "BackupCorrectnessLeftOverLogKeys", randomID)
.detail("BackupTag", printable(backupTag))
.detail("LeftOverKeys", logValues.size())
.detail("KeySpace", printable(backupLogValuesKey));
} else {
printf("No left over backup log keys\n");
}
}
break;
} catch (Error& e) {
err = e;
}
TraceEvent("BARW_CheckException", randomID).error(err);
co_await tr->onError(err);
}
if (displaySystemKeys) {
co_await TaskBucket::debugPrintRange(cx, "\xff"_sr, StringRef());
}
TraceEvent("BARW_Complete", randomID).detail("BackupTag", printable(backupTag));
// Decrement the backup agent requests
if (agentRequest) {
BackupAndParallelRestoreCorrectnessWorkload::backupAgentRequests--;
}
// SOMEDAY: Remove after backup agents can exist quiescently
if ((g_simulator->backupAgents == ISimulator::BackupAgentType::BackupToFile) &&
(!BackupAndParallelRestoreCorrectnessWorkload::backupAgentRequests)) {
g_simulator->backupAgents = ISimulator::BackupAgentType::NoBackupAgents;
}
} catch (Error& e) {
TraceEvent(SevError, "BackupAndParallelRestoreCorrectness").error(e).GetLastError();
throw;
}
}
};
int BackupAndParallelRestoreCorrectnessWorkload::backupAgentRequests = 0;
WorkloadFactory<BackupAndParallelRestoreCorrectnessWorkload> BackupAndParallelRestoreCorrectnessWorkloadFactory;