|
|
|
|
@ -34,6 +34,7 @@
|
|
|
|
|
#include "fdbserver/MutationTracking.h"
|
|
|
|
|
#include "fdbserver/WaitFailure.h"
|
|
|
|
|
#include "flow/Arena.h"
|
|
|
|
|
#include "flow/Error.h"
|
|
|
|
|
#include "flow/IRandom.h"
|
|
|
|
|
#include "flow/actorcompiler.h" // has to be last include
|
|
|
|
|
#include "flow/flow.h"
|
|
|
|
|
@ -41,8 +42,6 @@
|
|
|
|
|
#define BW_DEBUG true
|
|
|
|
|
#define BW_REQUEST_DEBUG false
|
|
|
|
|
|
|
|
|
|
// FIXME: change all BlobWorkerData* to Reference<BlobWorkerData> to avoid segfaults if core loop gets error
|
|
|
|
|
|
|
|
|
|
// TODO add comments + documentation
|
|
|
|
|
struct BlobFileIndex {
|
|
|
|
|
Version version;
|
|
|
|
|
@ -73,28 +72,22 @@ struct GranuleChangeFeedInfo {
|
|
|
|
|
Optional<GranuleFiles> existingFiles;
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
// FIXME: the circular dependencies here are getting kind of gross
|
|
|
|
|
struct GranuleMetadata;
|
|
|
|
|
struct BlobWorkerData;
|
|
|
|
|
ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(BlobWorkerData* bwData, AssignBlobRangeRequest req);
|
|
|
|
|
ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<GranuleMetadata> metadata);
|
|
|
|
|
|
|
|
|
|
// for a range that is active
|
|
|
|
|
struct GranuleMetadata : NonCopyable, ReferenceCounted<GranuleMetadata> {
|
|
|
|
|
KeyRange keyRange;
|
|
|
|
|
|
|
|
|
|
GranuleFiles files;
|
|
|
|
|
GranuleDeltas currentDeltas;
|
|
|
|
|
GranuleDeltas currentDeltas; // only contain deltas in pendingDeltaVersion + 1, bufferedDeltaVersion
|
|
|
|
|
// TODO get rid of this and do Reference<Standalone<GranuleDeltas>>?
|
|
|
|
|
Arena deltaArena;
|
|
|
|
|
|
|
|
|
|
uint64_t bytesInNewDeltaFiles = 0;
|
|
|
|
|
uint64_t bufferedDeltaBytes = 0;
|
|
|
|
|
|
|
|
|
|
NotifiedVersion bufferedDeltaVersion;
|
|
|
|
|
Version pendingDeltaVersion = 0;
|
|
|
|
|
NotifiedVersion durableDeltaVersion;
|
|
|
|
|
NotifiedVersion durableSnapshotVersion;
|
|
|
|
|
// for client to know when it is safe to read a certain version and from where (check waitForVersion)
|
|
|
|
|
NotifiedVersion bufferedDeltaVersion; // largest delta version in currentDeltas (including empty versions)
|
|
|
|
|
Version pendingDeltaVersion = 0; // largest version in progress writing to s3/fdb
|
|
|
|
|
NotifiedVersion durableDeltaVersion; // largest version persisted in s3/fdb
|
|
|
|
|
NotifiedVersion durableSnapshotVersion; // same as delta vars, except for snapshots
|
|
|
|
|
Version pendingSnapshotVersion = 0;
|
|
|
|
|
|
|
|
|
|
AsyncVar<int> rollbackCount;
|
|
|
|
|
@ -104,68 +97,50 @@ struct GranuleMetadata : NonCopyable, ReferenceCounted<GranuleMetadata> {
|
|
|
|
|
int64_t continueEpoch;
|
|
|
|
|
int64_t continueSeqno;
|
|
|
|
|
|
|
|
|
|
Future<GranuleChangeFeedInfo> assignFuture;
|
|
|
|
|
Future<Void> fileUpdaterFuture;
|
|
|
|
|
Promise<Void> resumeSnapshot;
|
|
|
|
|
|
|
|
|
|
// used to coordinate granule file updater is done
|
|
|
|
|
Promise<Void> cancelled;
|
|
|
|
|
Promise<Void> readable;
|
|
|
|
|
|
|
|
|
|
AssignBlobRangeRequest originalReq;
|
|
|
|
|
|
|
|
|
|
Future<Void> start(BlobWorkerData* bwData, AssignBlobRangeRequest req) {
|
|
|
|
|
originalReq = req;
|
|
|
|
|
assignFuture = persistAssignWorkerRange(bwData, req);
|
|
|
|
|
fileUpdaterFuture = blobGranuleUpdateFiles(bwData, Reference<GranuleMetadata>::addRef(this));
|
|
|
|
|
|
|
|
|
|
return success(assignFuture);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
void resume() {
|
|
|
|
|
ASSERT(resumeSnapshot.canBeSet());
|
|
|
|
|
resumeSnapshot.send(Void());
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// FIXME: right now there is a dependency because this contains both the actual file/delta data as well as the
|
|
|
|
|
// metadata (worker futures), so removing this reference from the map doesn't actually cancel the workers. It'd be
|
|
|
|
|
// better to have this in 2 separate objects, where the granule metadata map has the futures, but the read
|
|
|
|
|
// queries/file updater/range feed only copy the reference to the file/delta data.
|
|
|
|
|
Future<Void> cancel(bool dispose) {
|
|
|
|
|
if (cancelled.canBeSet()) {
|
|
|
|
|
// Could have been cancelled already by rollback or error in BGUpdateFiles
|
|
|
|
|
cancelled.send(Void());
|
|
|
|
|
}
|
|
|
|
|
assignFuture.cancel();
|
|
|
|
|
fileUpdaterFuture.cancel();
|
|
|
|
|
|
|
|
|
|
if (dispose) {
|
|
|
|
|
// FIXME: implement dispose!
|
|
|
|
|
return delay(0.1);
|
|
|
|
|
}
|
|
|
|
|
return Future<Void>(Void());
|
|
|
|
|
}
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
// for a range that may or may not be set
|
|
|
|
|
// TODO: rename this struct
|
|
|
|
|
struct GranuleRangeMetadata {
|
|
|
|
|
int64_t lastEpoch;
|
|
|
|
|
int64_t lastSeqno;
|
|
|
|
|
Reference<GranuleMetadata> activeMetadata;
|
|
|
|
|
|
|
|
|
|
Future<GranuleChangeFeedInfo> assignFuture;
|
|
|
|
|
Future<Void> fileUpdaterFuture;
|
|
|
|
|
|
|
|
|
|
GranuleRangeMetadata() : lastEpoch(0), lastSeqno(0) {}
|
|
|
|
|
GranuleRangeMetadata(int64_t epoch, int64_t seqno, Reference<GranuleMetadata> activeMetadata)
|
|
|
|
|
: lastEpoch(epoch), lastSeqno(seqno), activeMetadata(activeMetadata) {}
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
struct BlobWorkerData {
|
|
|
|
|
// FIXME: there is a reference cycle here. BWData has GranuleRangeMetadata objects in a map,
|
|
|
|
|
// but each of those has a future to a forever-running actor which has a reference to BWData.
|
|
|
|
|
// To fix this, we should only pass the necessary, specfic fields of BWData to those actors
|
|
|
|
|
// rather than the reference to BWData itself.
|
|
|
|
|
struct BlobWorkerData : NonCopyable, ReferenceCounted<BlobWorkerData> {
|
|
|
|
|
UID id;
|
|
|
|
|
Database db;
|
|
|
|
|
|
|
|
|
|
BlobWorkerStats stats;
|
|
|
|
|
|
|
|
|
|
PromiseStream<Future<Void>> addActor;
|
|
|
|
|
|
|
|
|
|
LocalityData locality;
|
|
|
|
|
int64_t currentManagerEpoch = -1;
|
|
|
|
|
|
|
|
|
|
ReplyPromiseStream<GranuleStatusReply> currentManagerStatusStream;
|
|
|
|
|
AsyncVar<ReplyPromiseStream<GranuleStatusReply>> currentManagerStatusStream;
|
|
|
|
|
|
|
|
|
|
// FIXME: refactor out the parts of this that are just for interacting with blob stores from the backup business
|
|
|
|
|
// logic
|
|
|
|
|
@ -221,7 +196,8 @@ static void checkGranuleLock(int64_t epoch, int64_t seqno, int64_t ownerEpoch, i
|
|
|
|
|
// sanity check - lock value should never go backwards because of acquireGranuleLock
|
|
|
|
|
/*
|
|
|
|
|
printf(
|
|
|
|
|
"Checking granule lock: \n mine: (%lld, %lld)\n owner: (%lld, %lld)\n", epoch, seqno, ownerEpoch, ownerSeqno);
|
|
|
|
|
"Checking granule lock: \n mine: (%lld, %lld)\n owner: (%lld, %lld)\n", epoch, seqno, ownerEpoch,
|
|
|
|
|
ownerSeqno);
|
|
|
|
|
*/
|
|
|
|
|
ASSERT(epoch <= ownerEpoch);
|
|
|
|
|
ASSERT(epoch < ownerEpoch || (epoch == ownerEpoch && seqno <= ownerSeqno));
|
|
|
|
|
@ -322,21 +298,23 @@ ACTOR Future<GranuleFiles> loadPreviousFiles(Transaction* tr, KeyRange keyRange)
|
|
|
|
|
// update shared state to coordinate when it is safe to clean up the old change feed.
|
|
|
|
|
// his goes through 3 phases for each new sub-granule:
|
|
|
|
|
// 1. Starting - the blob manager writes all sub-granules with this state as a durable intent to split the range
|
|
|
|
|
// 2. Assigned - a worker that is assigned a sub-granule updates that granule's state here. This means that the worker
|
|
|
|
|
// 2. Assigned - a worker that is assigned a sub-granule updates that granule's state here. This means that the
|
|
|
|
|
// worker
|
|
|
|
|
// has started a new change feed for the new sub-granule, but still needs to consume from the old change feed.
|
|
|
|
|
// 3. Done - the worker that is assigned this sub-granule has persisted all of the data from its part of the old change
|
|
|
|
|
// 3. Done - the worker that is assigned this sub-granule has persisted all of the data from its part of the old
|
|
|
|
|
// change
|
|
|
|
|
// feed in delta files. From this granule's perspective, it is safe to clean up the old change feed.
|
|
|
|
|
|
|
|
|
|
// Once all sub-granules have reached step 2 (Assigned), the change feed can be safely "stopped" - it needs to continue
|
|
|
|
|
// to serve the mutations it has seen so far, but will not need any new mutations after this version.
|
|
|
|
|
// The last sub-granule to reach this step is responsible for commiting the change feed stop as part of its
|
|
|
|
|
// transaction. Because this change feed stops commits in the same transaction as the worker's new change feed start,
|
|
|
|
|
// it is guaranteed that no versions are missed between the old and new change feed.
|
|
|
|
|
// Once all sub-granules have reached step 2 (Assigned), the change feed can be safely "stopped" - it needs to
|
|
|
|
|
// continue to serve the mutations it has seen so far, but will not need any new mutations after this version. The
|
|
|
|
|
// last sub-granule to reach this step is responsible for commiting the change feed stop as part of its transaction.
|
|
|
|
|
// Because this change feed stops commits in the same transaction as the worker's new change feed start, it is
|
|
|
|
|
// guaranteed that no versions are missed between the old and new change feed.
|
|
|
|
|
//
|
|
|
|
|
// Once all sub-granules have reached step 3 (Done), the change feed can be safely destroyed, as all of the mutations in
|
|
|
|
|
// the old change feed are guaranteed to be persisted in delta files. The last sub-granule to reach this step is
|
|
|
|
|
// responsible for committing the change feed destroy, and for cleaning up the split state for all sub-granules as part
|
|
|
|
|
// of its transaction.
|
|
|
|
|
// Once all sub-granules have reached step 3 (Done), the change feed can be safely destroyed, as all of the
|
|
|
|
|
// mutations in the old change feed are guaranteed to be persisted in delta files. The last sub-granule to reach
|
|
|
|
|
// this step is responsible for committing the change feed destroy, and for cleaning up the split state for all
|
|
|
|
|
// sub-granules as part of its transaction.
|
|
|
|
|
|
|
|
|
|
ACTOR Future<Void> updateGranuleSplitState(Transaction* tr,
|
|
|
|
|
KeyRange previousGranule,
|
|
|
|
|
@ -411,8 +389,8 @@ ACTOR Future<Void> updateGranuleSplitState(Transaction* tr,
|
|
|
|
|
Key myStateKey = myStateTuple.getDataAsStandalone().withPrefix(blobGranuleSplitKeys.begin);
|
|
|
|
|
if (newState == BlobGranuleSplitState::Done && currentState == BlobGranuleSplitState::Assigned &&
|
|
|
|
|
totalDone == total - 1) {
|
|
|
|
|
// we are the last one to change from Assigned -> Done, so everything can be cleaned up for the old change
|
|
|
|
|
// feed and splitting state
|
|
|
|
|
// we are the last one to change from Assigned -> Done, so everything can be cleaned up for the old
|
|
|
|
|
// change feed and splitting state
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("[%s - %s) destroying old change feed %s and granule lock + split state for [%s - %s)\n",
|
|
|
|
|
currentGranule.begin.printable().c_str(),
|
|
|
|
|
@ -477,11 +455,12 @@ static Value getFileValue(std::string fname, int64_t offset, int64_t length) {
|
|
|
|
|
return fileValue.getDataAsStandalone();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// writeDelta file writes speculatively in the common case to optimize throughput. It creates the s3 object even though
|
|
|
|
|
// the data in it may not yet be committed, and even though previous delta fiels with lower versioned data may still be
|
|
|
|
|
// in flight. The synchronization happens after the s3 file is written, but before we update the FDB index of what files
|
|
|
|
|
// exist. Before updating FDB, we ensure the version is committed and all previous delta files have updated FDB.
|
|
|
|
|
ACTOR Future<BlobFileIndex> writeDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
// writeDelta file writes speculatively in the common case to optimize throughput. It creates the s3 object even
|
|
|
|
|
// though the data in it may not yet be committed, and even though previous delta fiels with lower versioned data
|
|
|
|
|
// may still be in flight. The synchronization happens after the s3 file is written, but before we update the FDB
|
|
|
|
|
// index of what files exist. Before updating FDB, we ensure the version is committed and all previous delta files
|
|
|
|
|
// have updated FDB.
|
|
|
|
|
ACTOR Future<BlobFileIndex> writeDeltaFile(Reference<BlobWorkerData> bwData,
|
|
|
|
|
KeyRange keyRange,
|
|
|
|
|
int64_t epoch,
|
|
|
|
|
int64_t seqno,
|
|
|
|
|
@ -517,6 +496,7 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
wait(objectFile->append(serialized.begin(), serialized.size()));
|
|
|
|
|
wait(objectFile->finish());
|
|
|
|
|
|
|
|
|
|
state int numIterations = 0;
|
|
|
|
|
try {
|
|
|
|
|
// before updating FDB, wait for the delta file version to be committed and previous delta files to finish
|
|
|
|
|
while (bwData->knownCommittedVersion.get() < currentDeltaVersion) {
|
|
|
|
|
@ -543,8 +523,8 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
Key dfKey = deltaFileKey.getDataAsStandalone().withPrefix(blobGranuleFileKeys.begin);
|
|
|
|
|
tr->set(dfKey, getFileValue(fname, 0, serialized.size()));
|
|
|
|
|
|
|
|
|
|
// FIXME: if previous granule present and delta file version >= previous change feed version, update the
|
|
|
|
|
// state here
|
|
|
|
|
// FIXME: if previous granule present and delta file version >= previous change feed version, update
|
|
|
|
|
// the state here
|
|
|
|
|
if (oldChangeFeedDataComplete.present()) {
|
|
|
|
|
ASSERT(oldChangeFeedId.present());
|
|
|
|
|
wait(updateGranuleSplitState(&tr->getTransaction(),
|
|
|
|
|
@ -570,6 +550,7 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
}
|
|
|
|
|
return BlobFileIndex(currentDeltaVersion, fname, 0, serialized.size());
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
numIterations++;
|
|
|
|
|
wait(tr->onError(e));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
@ -578,6 +559,13 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
throw e;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// if commit failed the first time due to granule assignment conflict (which is non-retryable),
|
|
|
|
|
// then the file key was persisted and we should delete it. Otherwise, the commit failed
|
|
|
|
|
// for some other reason and the key wasn't persisted, so we should just propogate the error
|
|
|
|
|
if (numIterations != 1 || e.code() != error_code_granule_assignment_conflict) {
|
|
|
|
|
throw e;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// FIXME: only delete if key doesn't exist
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("deleting s3 delta file %s after error %s\n", fname.c_str(), e.name());
|
|
|
|
|
@ -589,7 +577,7 @@ ACTOR Future<BlobFileIndex> writeDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<BlobFileIndex> writeSnapshot(BlobWorkerData* bwData,
|
|
|
|
|
ACTOR Future<BlobFileIndex> writeSnapshot(Reference<BlobWorkerData> bwData,
|
|
|
|
|
KeyRange keyRange,
|
|
|
|
|
int64_t epoch,
|
|
|
|
|
int64_t seqno,
|
|
|
|
|
@ -663,6 +651,7 @@ ACTOR Future<BlobFileIndex> writeSnapshot(BlobWorkerData* bwData,
|
|
|
|
|
snapshotFileKey.append(LiteralStringRef("S")).append(version);
|
|
|
|
|
|
|
|
|
|
state Reference<ReadYourWritesTransaction> tr = makeReference<ReadYourWritesTransaction>(bwData->db);
|
|
|
|
|
state int numIterations = 0;
|
|
|
|
|
|
|
|
|
|
try {
|
|
|
|
|
loop {
|
|
|
|
|
@ -674,6 +663,7 @@ ACTOR Future<BlobFileIndex> writeSnapshot(BlobWorkerData* bwData,
|
|
|
|
|
wait(tr->commit());
|
|
|
|
|
break;
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
numIterations++;
|
|
|
|
|
wait(tr->onError(e));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
@ -682,6 +672,13 @@ ACTOR Future<BlobFileIndex> writeSnapshot(BlobWorkerData* bwData,
|
|
|
|
|
throw e;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// if commit failed the first time due to granule assignment conflict (which is non-retryable),
|
|
|
|
|
// then the file key was persisted and we should delete it. Otherwise, the commit failed
|
|
|
|
|
// for some other reason and the key wasn't persisted, so we should just propogate the error
|
|
|
|
|
if (numIterations != 1 || e.code() != error_code_granule_assignment_conflict) {
|
|
|
|
|
throw e;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// FIXME: only delete if key doesn't exist
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("deleting s3 snapshot file %s after error %s\n", fname.c_str(), e.name());
|
|
|
|
|
@ -707,7 +704,8 @@ ACTOR Future<BlobFileIndex> writeSnapshot(BlobWorkerData* bwData,
|
|
|
|
|
return BlobFileIndex(version, fname, 0, serialized.size());
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<BlobFileIndex> dumpInitialSnapshotFromFDB(BlobWorkerData* bwData, Reference<GranuleMetadata> metadata) {
|
|
|
|
|
ACTOR Future<BlobFileIndex> dumpInitialSnapshotFromFDB(Reference<BlobWorkerData> bwData,
|
|
|
|
|
Reference<GranuleMetadata> metadata) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Dumping snapshot from FDB for [%s - %s)\n",
|
|
|
|
|
metadata->keyRange.begin.printable().c_str(),
|
|
|
|
|
@ -755,7 +753,7 @@ ACTOR Future<BlobFileIndex> dumpInitialSnapshotFromFDB(BlobWorkerData* bwData, R
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// files might not be the current set of files in metadata, in the case of doing the initial snapshot of a granule.
|
|
|
|
|
ACTOR Future<BlobFileIndex> compactFromBlob(BlobWorkerData* bwData,
|
|
|
|
|
ACTOR Future<BlobFileIndex> compactFromBlob(Reference<BlobWorkerData> bwData,
|
|
|
|
|
Reference<GranuleMetadata> metadata,
|
|
|
|
|
GranuleFiles files) {
|
|
|
|
|
wait(delay(0, TaskPriority::BlobWorkerUpdateStorage));
|
|
|
|
|
@ -821,8 +819,8 @@ ACTOR Future<BlobFileIndex> compactFromBlob(BlobWorkerData* bwData,
|
|
|
|
|
DEBUG_KEY_RANGE("BlobWorkerBlobSnapshot", version, metadata->keyRange, bwData->id);
|
|
|
|
|
return f;
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
// TODO better error handling eventually - should retry unless the error is because another worker took over
|
|
|
|
|
// the range
|
|
|
|
|
// TODO better error handling eventually - should retry unless the error is because another worker took
|
|
|
|
|
// over the range
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Compacting snapshot from blob for [%s - %s) got error %s\n",
|
|
|
|
|
metadata->keyRange.begin.printable().c_str(),
|
|
|
|
|
@ -874,7 +872,7 @@ static bool filterOldMutations(const KeyRange& range,
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<Void> handleCompletedDeltaFile(BlobWorkerData* bwData,
|
|
|
|
|
ACTOR Future<Void> handleCompletedDeltaFile(Reference<BlobWorkerData> bwData,
|
|
|
|
|
Reference<GranuleMetadata> metadata,
|
|
|
|
|
BlobFileIndex completedDeltaFile,
|
|
|
|
|
Key cfKey,
|
|
|
|
|
@ -998,8 +996,8 @@ static Version doGranuleRollback(Reference<GranuleMetadata> metadata,
|
|
|
|
|
|
|
|
|
|
metadata->currentDeltas.resize(metadata->deltaArena, mIdx);
|
|
|
|
|
|
|
|
|
|
// delete all deltas in rollback range, but we can optimize here to just skip the uncommitted mutations directly
|
|
|
|
|
// and immediately pop the rollback out of inProgress
|
|
|
|
|
// delete all deltas in rollback range, but we can optimize here to just skip the uncommitted mutations
|
|
|
|
|
// directly and immediately pop the rollback out of inProgress
|
|
|
|
|
metadata->bufferedDeltaVersion.set(rollbackVersion);
|
|
|
|
|
cfRollbackVersion = mutationVersion;
|
|
|
|
|
}
|
|
|
|
|
@ -1026,7 +1024,9 @@ static Version doGranuleRollback(Reference<GranuleMetadata> metadata,
|
|
|
|
|
// updater for a single granule
|
|
|
|
|
// TODO: this is getting kind of large. Should try to split out this actor if it continues to grow?
|
|
|
|
|
// FIXME: handle errors here (forward errors)
|
|
|
|
|
ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<GranuleMetadata> metadata) {
|
|
|
|
|
ACTOR Future<Void> blobGranuleUpdateFiles(Reference<BlobWorkerData> bwData,
|
|
|
|
|
Reference<GranuleMetadata> metadata,
|
|
|
|
|
Future<GranuleChangeFeedInfo> assignFuture) {
|
|
|
|
|
state PromiseStream<Standalone<VectorRef<MutationsAndVersionRef>>> oldChangeFeedStream;
|
|
|
|
|
state PromiseStream<Standalone<VectorRef<MutationsAndVersionRef>>> changeFeedStream;
|
|
|
|
|
state Future<BlobFileIndex> inFlightBlobSnapshot;
|
|
|
|
|
@ -1051,7 +1051,7 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
metadata->resumeSnapshot.send(Void());
|
|
|
|
|
|
|
|
|
|
// before starting, make sure worker persists range assignment and acquires the granule lock
|
|
|
|
|
GranuleChangeFeedInfo _info = wait(metadata->assignFuture);
|
|
|
|
|
GranuleChangeFeedInfo _info = wait(assignFuture);
|
|
|
|
|
changeFeedInfo = _info;
|
|
|
|
|
|
|
|
|
|
wait(delay(0, TaskPriority::BlobWorkerUpdateStorage));
|
|
|
|
|
@ -1131,9 +1131,9 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
metadata->readable.send(Void());
|
|
|
|
|
|
|
|
|
|
if (changeFeedInfo.prevChangeFeedId.present()) {
|
|
|
|
|
// FIXME: once we have empty versions, only include up to changeFeedInfo.changeFeedStartVersion in the read
|
|
|
|
|
// stream. Then we can just stop the old stream when we get end_of_stream from this and not handle the
|
|
|
|
|
// mutation version truncation stuff
|
|
|
|
|
// FIXME: once we have empty versions, only include up to changeFeedInfo.changeFeedStartVersion in the
|
|
|
|
|
// read stream. Then we can just stop the old stream when we get end_of_stream from this and not handle
|
|
|
|
|
// the mutation version truncation stuff
|
|
|
|
|
|
|
|
|
|
// FIXME: filtering on key range != change feed range doesn't work
|
|
|
|
|
ASSERT(changeFeedInfo.granuleSplitFrom.present());
|
|
|
|
|
@ -1197,8 +1197,8 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
// TODO filter old mutations won't be necessary, SS does it already
|
|
|
|
|
if (filterOldMutations(
|
|
|
|
|
metadata->keyRange, &oldMutations, &mutations, changeFeedInfo.changeFeedStartVersion)) {
|
|
|
|
|
// if old change feed has caught up with where new one would start, finish last one and start new
|
|
|
|
|
// one
|
|
|
|
|
// if old change feed has caught up with where new one would start, finish last one and start
|
|
|
|
|
// new one
|
|
|
|
|
|
|
|
|
|
Key cfKey = StringRef(changeFeedInfo.changeFeedId.toString());
|
|
|
|
|
changeFeedFuture = bwData->db->getChangeFeedStream(changeFeedStream,
|
|
|
|
|
@ -1223,8 +1223,8 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
state MutationsAndVersionRef deltas = d;
|
|
|
|
|
ASSERT(deltas.version >= metadata->bufferedDeltaVersion.get());
|
|
|
|
|
// Write a new delta file IF we have enough bytes, and we have all of the previous version's stuff
|
|
|
|
|
// there to ensure no versions span multiple delta files. Check this by ensuring the version of this new
|
|
|
|
|
// delta is larger than the previous largest seen version
|
|
|
|
|
// there to ensure no versions span multiple delta files. Check this by ensuring the version of this
|
|
|
|
|
// new delta is larger than the previous largest seen version
|
|
|
|
|
if (metadata->bufferedDeltaBytes >= SERVER_KNOBS->BG_DELTA_FILE_TARGET_BYTES &&
|
|
|
|
|
deltas.version > metadata->bufferedDeltaVersion.get()) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
@ -1283,8 +1283,8 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
metadata->bufferedDeltaBytes = 0;
|
|
|
|
|
|
|
|
|
|
// if we just wrote a delta file, check if we need to compact here.
|
|
|
|
|
// exhaust old change feed before compacting - otherwise we could end up with an endlessly growing
|
|
|
|
|
// list of previous change feeds in the worst case.
|
|
|
|
|
// exhaust old change feed before compacting - otherwise we could end up with an endlessly
|
|
|
|
|
// growing list of previous change feeds in the worst case.
|
|
|
|
|
snapshotEligible = true;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
@ -1292,7 +1292,6 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
// bunch of extra delta files at some point, even if we don't consider it for a split yet
|
|
|
|
|
if (snapshotEligible && metadata->bytesInNewDeltaFiles >= SERVER_KNOBS->BG_DELTA_BYTES_BEFORE_COMPACT &&
|
|
|
|
|
!readOldChangeFeed) {
|
|
|
|
|
|
|
|
|
|
if (BW_DEBUG && (inFlightBlobSnapshot.isValid() || !inFlightDeltaFiles.empty())) {
|
|
|
|
|
printf("Granule [%s - %s) ready to re-snapshot, waiting for outstanding %d snapshot and %d "
|
|
|
|
|
"deltas to "
|
|
|
|
|
@ -1341,18 +1340,29 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
state int64_t statusEpoch = metadata->continueEpoch;
|
|
|
|
|
state int64_t statusSeqno = metadata->continueSeqno;
|
|
|
|
|
loop {
|
|
|
|
|
bwData->currentManagerStatusStream.send(
|
|
|
|
|
GranuleStatusReply(metadata->keyRange, true, statusEpoch, statusSeqno));
|
|
|
|
|
|
|
|
|
|
Optional<Void> result = wait(timeout(metadata->resumeSnapshot.getFuture(), 1.0));
|
|
|
|
|
if (result.present()) {
|
|
|
|
|
break;
|
|
|
|
|
loop {
|
|
|
|
|
try {
|
|
|
|
|
wait(bwData->currentManagerStatusStream.get().onReady());
|
|
|
|
|
bwData->currentManagerStatusStream.get().send(
|
|
|
|
|
GranuleStatusReply(metadata->keyRange, true, statusEpoch, statusSeqno));
|
|
|
|
|
break;
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
printf("manager stream was changed\n");
|
|
|
|
|
wait(bwData->currentManagerStatusStream.onChange());
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
// FIXME: re-trigger this loop if blob manager status stream changes
|
|
|
|
|
|
|
|
|
|
choose {
|
|
|
|
|
when(wait(metadata->resumeSnapshot.getFuture())) { break; }
|
|
|
|
|
when(wait(delay(1.0))) {}
|
|
|
|
|
when(wait(bwData->currentManagerStatusStream.onChange())) {}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Granule [%s - %s)\n, hasn't heard back from BM, re-sending status\n",
|
|
|
|
|
printf("Granule [%s - %s)\n, hasn't heard back from BM in BW %s, re-sending status\n",
|
|
|
|
|
metadata->keyRange.begin.printable().c_str(),
|
|
|
|
|
metadata->keyRange.end.printable().c_str());
|
|
|
|
|
metadata->keyRange.end.printable().c_str(),
|
|
|
|
|
bwData->id.toString().c_str());
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
@ -1379,8 +1389,8 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
metadata->bytesInNewDeltaFiles = 0;
|
|
|
|
|
} else if (snapshotEligible &&
|
|
|
|
|
metadata->bytesInNewDeltaFiles >= SERVER_KNOBS->BG_DELTA_BYTES_BEFORE_COMPACT) {
|
|
|
|
|
// if we're in the old change feed case and can't snapshot but we have enough data to, don't queue
|
|
|
|
|
// too many delta files in parallel
|
|
|
|
|
// if we're in the old change feed case and can't snapshot but we have enough data to, don't
|
|
|
|
|
// queue too many delta files in parallel
|
|
|
|
|
while (inFlightDeltaFiles.size() > 10) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("[%s - %s) Waiting on delta file b/c old change feed\n",
|
|
|
|
|
@ -1526,7 +1536,6 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
}
|
|
|
|
|
justDidRollback = false;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
if (e.code() == error_code_operation_cancelled) {
|
|
|
|
|
throw;
|
|
|
|
|
@ -1549,8 +1558,8 @@ ACTOR Future<Void> blobGranuleUpdateFiles(BlobWorkerData* bwData, Reference<Gran
|
|
|
|
|
TraceEvent(SevWarn, "GranuleFileUpdaterError", bwData->id).detail("Granule", metadata->keyRange).error(e);
|
|
|
|
|
|
|
|
|
|
if (granuleCanRetry(e)) {
|
|
|
|
|
// explicitly cancel all outstanding write futures BEFORE updating promise stream, to ensure they can't
|
|
|
|
|
// update files after the re-assigned granule acquires the lock
|
|
|
|
|
// explicitly cancel all outstanding write futures BEFORE updating promise stream, to ensure they
|
|
|
|
|
// can't update files after the re-assigned granule acquires the lock
|
|
|
|
|
inFlightBlobSnapshot.cancel();
|
|
|
|
|
for (auto& f : inFlightDeltaFiles) {
|
|
|
|
|
f.future.cancel();
|
|
|
|
|
@ -1619,7 +1628,8 @@ static Future<Void> waitForVersion(Reference<GranuleMetadata> metadata, Version
|
|
|
|
|
// if we don't have to wait for change feed version to catch up or wait for any pending file writes to complete,
|
|
|
|
|
// nothing to do
|
|
|
|
|
|
|
|
|
|
/*printf(" [%s - %s) waiting for %lld\n readable:%s\n bufferedDelta=%lld\n pendingDelta=%lld\n "
|
|
|
|
|
/*
|
|
|
|
|
printf(" [%s - %s) waiting for %lld\n readable:%s\n bufferedDelta=%lld\n pendingDelta=%lld\n "
|
|
|
|
|
"durableDelta=%lld\n pendingSnapshot=%lld\n durableSnapshot=%lld\n",
|
|
|
|
|
metadata->keyRange.begin.printable().c_str(),
|
|
|
|
|
metadata->keyRange.end.printable().c_str(),
|
|
|
|
|
@ -1629,7 +1639,8 @@ static Future<Void> waitForVersion(Reference<GranuleMetadata> metadata, Version
|
|
|
|
|
metadata->pendingDeltaVersion,
|
|
|
|
|
metadata->durableDeltaVersion.get(),
|
|
|
|
|
metadata->pendingSnapshotVersion,
|
|
|
|
|
metadata->durableSnapshotVersion.get());*/
|
|
|
|
|
metadata->durableSnapshotVersion.get());
|
|
|
|
|
*/
|
|
|
|
|
|
|
|
|
|
if (metadata->readable.isSet() && v <= metadata->bufferedDeltaVersion.get() &&
|
|
|
|
|
(v <= metadata->durableDeltaVersion.get() ||
|
|
|
|
|
@ -1642,7 +1653,7 @@ static Future<Void> waitForVersion(Reference<GranuleMetadata> metadata, Version
|
|
|
|
|
return waitForVersionActor(metadata, v);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<Void> handleBlobGranuleFileRequest(BlobWorkerData* bwData, BlobGranuleFileRequest req) {
|
|
|
|
|
ACTOR Future<Void> handleBlobGranuleFileRequest(Reference<BlobWorkerData> bwData, BlobGranuleFileRequest req) {
|
|
|
|
|
try {
|
|
|
|
|
// TODO REMOVE in api V2
|
|
|
|
|
ASSERT(req.beginVersion == 0);
|
|
|
|
|
@ -1664,6 +1675,7 @@ ACTOR Future<Void> handleBlobGranuleFileRequest(BlobWorkerData* bwData, BlobGran
|
|
|
|
|
req.keyRange.begin.printable().c_str(),
|
|
|
|
|
req.keyRange.end.printable().c_str());
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
throw wrong_shard_server();
|
|
|
|
|
}
|
|
|
|
|
granules.push_back(r.value().activeMetadata);
|
|
|
|
|
@ -1821,7 +1833,8 @@ ACTOR Future<Void> handleBlobGranuleFileRequest(BlobWorkerData* bwData, BlobGran
|
|
|
|
|
return Void();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(BlobWorkerData* bwData, AssignBlobRangeRequest req) {
|
|
|
|
|
ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(Reference<BlobWorkerData> bwData,
|
|
|
|
|
AssignBlobRangeRequest req) {
|
|
|
|
|
ASSERT(!req.continueAssignment);
|
|
|
|
|
state Transaction tr(bwData->db);
|
|
|
|
|
state Key lockKey = granuleLockKey(req.keyRange);
|
|
|
|
|
@ -1861,13 +1874,13 @@ ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(BlobWorkerData* bwD
|
|
|
|
|
info.previousDurableVersion = info.existingFiles.get().deltaFiles.back().version;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// for the non-splitting cases, this doesn't need to be 100% accurate, it just needs to be smaller
|
|
|
|
|
// than the next delta file write.
|
|
|
|
|
// for the non-splitting cases, this doesn't need to be 100% accurate, it just needs to be
|
|
|
|
|
// smaller than the next delta file write.
|
|
|
|
|
info.changeFeedStartVersion = info.previousDurableVersion;
|
|
|
|
|
} else {
|
|
|
|
|
// else we are first, no need to check for owner conflict
|
|
|
|
|
// FIXME: use actual 16 bytes of UID instead of converting it to 32 character string and then that
|
|
|
|
|
// to bytes
|
|
|
|
|
// FIXME: use actual 16 bytes of UID instead of converting it to 32 character string and then
|
|
|
|
|
// that to bytes
|
|
|
|
|
info.changeFeedId = deterministicRandom()->randomUniqueID();
|
|
|
|
|
wait(tr.registerChangeFeed(StringRef(info.changeFeedId.toString()), req.keyRange));
|
|
|
|
|
info.doSnapshot = true;
|
|
|
|
|
@ -1882,8 +1895,8 @@ ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(BlobWorkerData* bwD
|
|
|
|
|
Optional<Value> parentGranulesValue =
|
|
|
|
|
wait(tr.get(historyKey.getDataAsStandalone().withPrefix(blobGranuleHistoryKeys.begin)));
|
|
|
|
|
|
|
|
|
|
// If anything in previousGranules, need to do the handoff logic and set ret.previousChangeFeedId, and
|
|
|
|
|
// the previous durable version will come from the previous granules
|
|
|
|
|
// If anything in previousGranules, need to do the handoff logic and set ret.previousChangeFeedId,
|
|
|
|
|
// and the previous durable version will come from the previous granules
|
|
|
|
|
if (parentGranulesValue.present()) {
|
|
|
|
|
state Standalone<VectorRef<KeyRangeRef>> parentGranules =
|
|
|
|
|
decodeBlobGranuleHistoryValue(parentGranulesValue.get());
|
|
|
|
|
@ -1930,8 +1943,8 @@ ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(BlobWorkerData* bwD
|
|
|
|
|
req.keyRange,
|
|
|
|
|
info.prevChangeFeedId.get(),
|
|
|
|
|
BlobGranuleSplitState::Assigned));
|
|
|
|
|
// change feed was created as part of this transaction, changeFeedStartVersion will be set
|
|
|
|
|
// later
|
|
|
|
|
// change feed was created as part of this transaction, changeFeedStartVersion will be
|
|
|
|
|
// set later
|
|
|
|
|
} else {
|
|
|
|
|
ASSERT(false);
|
|
|
|
|
}
|
|
|
|
|
@ -1971,7 +1984,16 @@ ACTOR Future<GranuleChangeFeedInfo> persistAssignWorkerRange(BlobWorkerData* bwD
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
static GranuleRangeMetadata constructActiveBlobRange(BlobWorkerData* bwData,
|
|
|
|
|
ACTOR Future<Void> start(Reference<BlobWorkerData> bwData, GranuleRangeMetadata* meta, AssignBlobRangeRequest req) {
|
|
|
|
|
ASSERT(meta->activeMetadata.isValid());
|
|
|
|
|
meta->activeMetadata->originalReq = req;
|
|
|
|
|
meta->assignFuture = persistAssignWorkerRange(bwData, req);
|
|
|
|
|
meta->fileUpdaterFuture = blobGranuleUpdateFiles(bwData, meta->activeMetadata, meta->assignFuture);
|
|
|
|
|
wait(success(meta->assignFuture));
|
|
|
|
|
return Void();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
static GranuleRangeMetadata constructActiveBlobRange(Reference<BlobWorkerData> bwData,
|
|
|
|
|
KeyRange keyRange,
|
|
|
|
|
int64_t epoch,
|
|
|
|
|
int64_t seqno) {
|
|
|
|
|
@ -2013,13 +2035,13 @@ static bool newerRangeAssignment(GranuleRangeMetadata oldMetadata, int64_t epoch
|
|
|
|
|
// Returns future to wait on to ensure prior work of other granules is done before responding to the manager with a
|
|
|
|
|
// successful assignment And if the change produced a new granule that needs to start doing work, returns the new
|
|
|
|
|
// granule so that the caller can start() it with the appropriate starting state.
|
|
|
|
|
static std::pair<Future<Void>, Reference<GranuleMetadata>> changeBlobRange(BlobWorkerData* bwData,
|
|
|
|
|
KeyRange keyRange,
|
|
|
|
|
int64_t epoch,
|
|
|
|
|
int64_t seqno,
|
|
|
|
|
bool active,
|
|
|
|
|
bool disposeOnCleanup,
|
|
|
|
|
bool selfReassign) {
|
|
|
|
|
ACTOR Future<bool> changeBlobRange(Reference<BlobWorkerData> bwData,
|
|
|
|
|
KeyRange keyRange,
|
|
|
|
|
int64_t epoch,
|
|
|
|
|
int64_t seqno,
|
|
|
|
|
bool active,
|
|
|
|
|
bool disposeOnCleanup,
|
|
|
|
|
bool selfReassign) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("%s range for [%s - %s): %s @ (%lld, %lld)\n",
|
|
|
|
|
selfReassign ? "Re-assigning" : "Changing",
|
|
|
|
|
@ -2036,12 +2058,21 @@ static std::pair<Future<Void>, Reference<GranuleMetadata>> changeBlobRange(BlobW
|
|
|
|
|
// older range, cancel it if it is active. Insert the current range. Re-insert all newer ranges over the current
|
|
|
|
|
// range.
|
|
|
|
|
|
|
|
|
|
std::vector<Future<Void>> futures;
|
|
|
|
|
state std::vector<Future<Void>> futures;
|
|
|
|
|
|
|
|
|
|
std::vector<std::pair<KeyRange, GranuleRangeMetadata>> newerRanges;
|
|
|
|
|
state std::vector<std::pair<KeyRange, GranuleRangeMetadata>> newerRanges;
|
|
|
|
|
|
|
|
|
|
auto ranges = bwData->granuleMetadata.intersectingRanges(keyRange);
|
|
|
|
|
bool alreadyAssigned = false;
|
|
|
|
|
for (auto& r : ranges) {
|
|
|
|
|
if (!active) {
|
|
|
|
|
if (r.value().activeMetadata.isValid() && r.value().activeMetadata->cancelled.canBeSet()) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Cancelling activeMetadata\n");
|
|
|
|
|
}
|
|
|
|
|
r.value().activeMetadata->cancelled.send(Void());
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
bool thisAssignmentNewer = newerRangeAssignment(r.value(), epoch, seqno);
|
|
|
|
|
if (r.value().lastEpoch == epoch && r.value().lastSeqno == seqno) {
|
|
|
|
|
ASSERT(r.begin() == keyRange.begin);
|
|
|
|
|
@ -2050,12 +2081,13 @@ static std::pair<Future<Void>, Reference<GranuleMetadata>> changeBlobRange(BlobW
|
|
|
|
|
if (selfReassign) {
|
|
|
|
|
thisAssignmentNewer = true;
|
|
|
|
|
} else {
|
|
|
|
|
printf("same assignment\n");
|
|
|
|
|
// applied the same assignment twice, make idempotent
|
|
|
|
|
if (r.value().activeMetadata.isValid()) {
|
|
|
|
|
futures.push_back(success(r.value().activeMetadata->assignFuture));
|
|
|
|
|
futures.push_back(success(r.value().assignFuture));
|
|
|
|
|
}
|
|
|
|
|
return std::pair(waitForAll(futures),
|
|
|
|
|
Reference<GranuleMetadata>()); // already applied, nothing to do
|
|
|
|
|
alreadyAssigned = true;
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
@ -2068,7 +2100,6 @@ static std::pair<Future<Void>, Reference<GranuleMetadata>> changeBlobRange(BlobW
|
|
|
|
|
r.value().lastEpoch,
|
|
|
|
|
r.value().lastSeqno);
|
|
|
|
|
}
|
|
|
|
|
futures.push_back(r.value().activeMetadata->cancel(disposeOnCleanup));
|
|
|
|
|
r.value().activeMetadata.clear();
|
|
|
|
|
} else if (!thisAssignmentNewer) {
|
|
|
|
|
// this assignment is outdated, re-insert it over the current range
|
|
|
|
|
@ -2076,10 +2107,16 @@ static std::pair<Future<Void>, Reference<GranuleMetadata>> changeBlobRange(BlobW
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (alreadyAssigned) {
|
|
|
|
|
wait(waitForAll(futures)); // already applied, nothing to do
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// if range is active, and isn't surpassed by a newer range already, insert an active range
|
|
|
|
|
GranuleRangeMetadata newMetadata = (active && newerRanges.empty())
|
|
|
|
|
? constructActiveBlobRange(bwData, keyRange, epoch, seqno)
|
|
|
|
|
: constructInactiveBlobRange(epoch, seqno);
|
|
|
|
|
|
|
|
|
|
bwData->granuleMetadata.insert(keyRange, newMetadata);
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Inserting new range [%s - %s): %s @ (%lld, %lld)\n",
|
|
|
|
|
@ -2102,10 +2139,12 @@ static std::pair<Future<Void>, Reference<GranuleMetadata>> changeBlobRange(BlobW
|
|
|
|
|
bwData->granuleMetadata.insert(it.first, it.second);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return std::pair(waitForAll(futures), newMetadata.activeMetadata);
|
|
|
|
|
printf("returning from changeblobrange");
|
|
|
|
|
wait(waitForAll(futures));
|
|
|
|
|
return true;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
static bool resumeBlobRange(BlobWorkerData* bwData, KeyRange keyRange, int64_t epoch, int64_t seqno) {
|
|
|
|
|
static bool resumeBlobRange(Reference<BlobWorkerData> bwData, KeyRange keyRange, int64_t epoch, int64_t seqno) {
|
|
|
|
|
auto existingRange = bwData->granuleMetadata.rangeContaining(keyRange.begin);
|
|
|
|
|
// if range boundaries don't match, or this (epoch, seqno) is old or the granule is inactive, ignore
|
|
|
|
|
if (keyRange.begin != existingRange.begin() || keyRange.end != existingRange.end() ||
|
|
|
|
|
@ -2142,7 +2181,7 @@ static bool resumeBlobRange(BlobWorkerData* bwData, KeyRange keyRange, int64_t e
|
|
|
|
|
return true;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<Void> registerBlobWorker(BlobWorkerData* bwData, BlobWorkerInterface interf) {
|
|
|
|
|
ACTOR Future<Void> registerBlobWorker(Reference<BlobWorkerData> bwData, BlobWorkerInterface interf) {
|
|
|
|
|
state Reference<ReadYourWritesTransaction> tr = makeReference<ReadYourWritesTransaction>(bwData->db);
|
|
|
|
|
loop {
|
|
|
|
|
tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS);
|
|
|
|
|
@ -2167,18 +2206,20 @@ ACTOR Future<Void> registerBlobWorker(BlobWorkerData* bwData, BlobWorkerInterfac
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<Void> handleRangeAssign(BlobWorkerData* bwData, AssignBlobRangeRequest req, bool isSelfReassign) {
|
|
|
|
|
ACTOR Future<Void> handleRangeAssign(Reference<BlobWorkerData> bwData,
|
|
|
|
|
AssignBlobRangeRequest req,
|
|
|
|
|
bool isSelfReassign) {
|
|
|
|
|
try {
|
|
|
|
|
if (req.continueAssignment) {
|
|
|
|
|
resumeBlobRange(bwData, req.keyRange, req.managerEpoch, req.managerSeqno);
|
|
|
|
|
} else {
|
|
|
|
|
state std::pair<Future<Void>, Reference<GranuleMetadata>> futureAndNewGranule =
|
|
|
|
|
changeBlobRange(bwData, req.keyRange, req.managerEpoch, req.managerSeqno, true, false, isSelfReassign);
|
|
|
|
|
bool shouldStart = wait(
|
|
|
|
|
changeBlobRange(bwData, req.keyRange, req.managerEpoch, req.managerSeqno, true, false, isSelfReassign));
|
|
|
|
|
|
|
|
|
|
wait(futureAndNewGranule.first);
|
|
|
|
|
|
|
|
|
|
if (futureAndNewGranule.second.isValid()) {
|
|
|
|
|
wait(futureAndNewGranule.second->start(bwData, req));
|
|
|
|
|
if (shouldStart) {
|
|
|
|
|
auto m = bwData->granuleMetadata.rangeContaining(req.keyRange.begin);
|
|
|
|
|
ASSERT(m.begin() == req.keyRange.begin && m.end() == req.keyRange.end);
|
|
|
|
|
wait(start(bwData, &m.value(), req));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if (!isSelfReassign) {
|
|
|
|
|
@ -2199,14 +2240,15 @@ ACTOR Future<Void> handleRangeAssign(BlobWorkerData* bwData, AssignBlobRangeRequ
|
|
|
|
|
req.reply.sendError(e);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
throw;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
ACTOR Future<Void> handleRangeRevoke(BlobWorkerData* bwData, RevokeBlobRangeRequest req) {
|
|
|
|
|
ACTOR Future<Void> handleRangeRevoke(Reference<BlobWorkerData> bwData, RevokeBlobRangeRequest req) {
|
|
|
|
|
try {
|
|
|
|
|
wait(
|
|
|
|
|
changeBlobRange(bwData, req.keyRange, req.managerEpoch, req.managerSeqno, false, req.dispose, false).first);
|
|
|
|
|
bool _shouldStart =
|
|
|
|
|
wait(changeBlobRange(bwData, req.keyRange, req.managerEpoch, req.managerSeqno, false, req.dispose, false));
|
|
|
|
|
req.reply.send(AssignBlobRangeReply(true));
|
|
|
|
|
return Void();
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
@ -2229,7 +2271,7 @@ ACTOR Future<Void> handleRangeRevoke(BlobWorkerData* bwData, RevokeBlobRangeRequ
|
|
|
|
|
// uncommitted data. This means we must ensure the data is actually committed before "committing" those writes in
|
|
|
|
|
// the blob granule. The simplest way to do this is to have the blob worker do a periodic GRV, which is guaranteed
|
|
|
|
|
// to be an earlier committed version.
|
|
|
|
|
ACTOR Future<Void> runCommitVersionChecks(BlobWorkerData* bwData) {
|
|
|
|
|
ACTOR Future<Void> runCommitVersionChecks(Reference<BlobWorkerData> bwData) {
|
|
|
|
|
state Transaction tr(bwData->db);
|
|
|
|
|
loop {
|
|
|
|
|
// only do grvs to get committed version if we need it to persist delta files
|
|
|
|
|
@ -2262,9 +2304,12 @@ ACTOR Future<Void> runCommitVersionChecks(BlobWorkerData* bwData) {
|
|
|
|
|
ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
|
|
|
|
|
ReplyPromise<InitializeBlobWorkerReply> recruitReply,
|
|
|
|
|
Reference<AsyncVar<ServerDBInfo> const> dbInfo) {
|
|
|
|
|
state BlobWorkerData self(bwInterf.id(), openDBOnServer(dbInfo, TaskPriority::DefaultEndpoint, LockAware::True));
|
|
|
|
|
self.id = bwInterf.id();
|
|
|
|
|
self.locality = bwInterf.locality;
|
|
|
|
|
state Reference<BlobWorkerData> self(
|
|
|
|
|
new BlobWorkerData(bwInterf.id(), openDBOnServer(dbInfo, TaskPriority::DefaultEndpoint, LockAware::True)));
|
|
|
|
|
self->id = bwInterf.id();
|
|
|
|
|
self->locality = bwInterf.locality;
|
|
|
|
|
|
|
|
|
|
state Future<Void> collection = actorCollection(self->addActor.getFuture());
|
|
|
|
|
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Initializing blob worker s3 stuff\n");
|
|
|
|
|
@ -2275,19 +2320,19 @@ ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("BW constructing simulated backup container\n");
|
|
|
|
|
}
|
|
|
|
|
self.bstore = BackupContainerFileSystem::openContainerFS("file://fdbblob/");
|
|
|
|
|
self->bstore = BackupContainerFileSystem::openContainerFS("file://fdbblob/");
|
|
|
|
|
} else {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("BW constructing backup container from %s\n", SERVER_KNOBS->BG_URL.c_str());
|
|
|
|
|
}
|
|
|
|
|
self.bstore = BackupContainerFileSystem::openContainerFS(SERVER_KNOBS->BG_URL);
|
|
|
|
|
self->bstore = BackupContainerFileSystem::openContainerFS(SERVER_KNOBS->BG_URL);
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("BW constructed backup container\n");
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// register the blob worker to the system keyspace
|
|
|
|
|
wait(registerBlobWorker(&self, bwInterf));
|
|
|
|
|
wait(registerBlobWorker(self, bwInterf));
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("BW got backup container init error %s\n", e.name());
|
|
|
|
|
@ -2307,11 +2352,8 @@ ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
|
|
|
|
|
rep.interf = bwInterf;
|
|
|
|
|
recruitReply.send(rep);
|
|
|
|
|
|
|
|
|
|
state PromiseStream<Future<Void>> addActor;
|
|
|
|
|
state Future<Void> collection = actorCollection(addActor.getFuture());
|
|
|
|
|
|
|
|
|
|
addActor.send(waitFailureServer(bwInterf.waitFailure.getFuture()));
|
|
|
|
|
addActor.send(runCommitVersionChecks(&self));
|
|
|
|
|
self->addActor.send(waitFailureServer(bwInterf.waitFailure.getFuture()));
|
|
|
|
|
self->addActor.send(runCommitVersionChecks(self));
|
|
|
|
|
|
|
|
|
|
try {
|
|
|
|
|
loop choose {
|
|
|
|
|
@ -2319,25 +2361,28 @@ ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
|
|
|
|
|
/*printf("Got blob granule request [%s - %s)\n",
|
|
|
|
|
req.keyRange.begin.printable().c_str(),
|
|
|
|
|
req.keyRange.end.printable().c_str());*/
|
|
|
|
|
++self.stats.readRequests;
|
|
|
|
|
++self.stats.activeReadRequests;
|
|
|
|
|
addActor.send(handleBlobGranuleFileRequest(&self, req));
|
|
|
|
|
++self->stats.readRequests;
|
|
|
|
|
++self->stats.activeReadRequests;
|
|
|
|
|
self->addActor.send(handleBlobGranuleFileRequest(self, req));
|
|
|
|
|
}
|
|
|
|
|
when(GranuleStatusStreamRequest req = waitNext(bwInterf.granuleStatusStreamRequest.getFuture())) {
|
|
|
|
|
if (self.managerEpochOk(req.managerEpoch)) {
|
|
|
|
|
when(state GranuleStatusStreamRequest req = waitNext(bwInterf.granuleStatusStreamRequest.getFuture())) {
|
|
|
|
|
if (self->managerEpochOk(req.managerEpoch)) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Worker %s got new granule status endpoint\n", self.id.toString().c_str());
|
|
|
|
|
printf("Worker %s got new granule status endpoint\n", self->id.toString().c_str());
|
|
|
|
|
}
|
|
|
|
|
self.currentManagerStatusStream = req.reply;
|
|
|
|
|
// req.reply is marked const unless you mark req as `state`?!?!?
|
|
|
|
|
// TODO: pick a reasonable byte limit instead of just piggy-backing
|
|
|
|
|
req.reply.setByteLimit(SERVER_KNOBS->RANGESTREAM_LIMIT_BYTES);
|
|
|
|
|
self->currentManagerStatusStream.set(req.reply);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
when(AssignBlobRangeRequest _req = waitNext(bwInterf.assignBlobRangeRequest.getFuture())) {
|
|
|
|
|
++self.stats.rangeAssignmentRequests;
|
|
|
|
|
--self.stats.numRangesAssigned;
|
|
|
|
|
++self->stats.rangeAssignmentRequests;
|
|
|
|
|
--self->stats.numRangesAssigned;
|
|
|
|
|
state AssignBlobRangeRequest assignReq = _req;
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Worker %s assigned range [%s - %s) @ (%lld, %lld):\n continue=%s\n",
|
|
|
|
|
self.id.toString().c_str(),
|
|
|
|
|
self->id.toString().c_str(),
|
|
|
|
|
assignReq.keyRange.begin.printable().c_str(),
|
|
|
|
|
assignReq.keyRange.end.printable().c_str(),
|
|
|
|
|
assignReq.managerEpoch,
|
|
|
|
|
@ -2345,18 +2390,18 @@ ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
|
|
|
|
|
assignReq.continueAssignment ? "T" : "F");
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (self.managerEpochOk(assignReq.managerEpoch)) {
|
|
|
|
|
addActor.send(handleRangeAssign(&self, assignReq, false));
|
|
|
|
|
if (self->managerEpochOk(assignReq.managerEpoch)) {
|
|
|
|
|
self->addActor.send(handleRangeAssign(self, assignReq, false));
|
|
|
|
|
} else {
|
|
|
|
|
assignReq.reply.send(AssignBlobRangeReply(false));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
when(RevokeBlobRangeRequest _req = waitNext(bwInterf.revokeBlobRangeRequest.getFuture())) {
|
|
|
|
|
state RevokeBlobRangeRequest revokeReq = _req;
|
|
|
|
|
--self.stats.numRangesAssigned;
|
|
|
|
|
--self->stats.numRangesAssigned;
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Worker %s revoked range [%s - %s) @ (%lld, %lld):\n dispose=%s\n",
|
|
|
|
|
self.id.toString().c_str(),
|
|
|
|
|
self->id.toString().c_str(),
|
|
|
|
|
revokeReq.keyRange.begin.printable().c_str(),
|
|
|
|
|
revokeReq.keyRange.end.printable().c_str(),
|
|
|
|
|
revokeReq.managerEpoch,
|
|
|
|
|
@ -2364,30 +2409,37 @@ ACTOR Future<Void> blobWorker(BlobWorkerInterface bwInterf,
|
|
|
|
|
revokeReq.dispose ? "T" : "F");
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (self.managerEpochOk(revokeReq.managerEpoch)) {
|
|
|
|
|
addActor.send(handleRangeRevoke(&self, revokeReq));
|
|
|
|
|
if (self->managerEpochOk(revokeReq.managerEpoch)) {
|
|
|
|
|
self->addActor.send(handleRangeRevoke(self, revokeReq));
|
|
|
|
|
} else {
|
|
|
|
|
revokeReq.reply.send(AssignBlobRangeReply(false));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
when(AssignBlobRangeRequest granuleToReassign = waitNext(self.granuleUpdateErrors.getFuture())) {
|
|
|
|
|
addActor.send(handleRangeAssign(&self, granuleToReassign, true));
|
|
|
|
|
when(AssignBlobRangeRequest granuleToReassign = waitNext(self->granuleUpdateErrors.getFuture())) {
|
|
|
|
|
self->addActor.send(handleRangeAssign(self, granuleToReassign, true));
|
|
|
|
|
}
|
|
|
|
|
when(HaltBlobWorkerRequest req = waitNext(bwInterf.haltBlobWorker.getFuture())) {
|
|
|
|
|
req.reply.send(Void());
|
|
|
|
|
if (self->managerEpochOk(req.managerEpoch)) {
|
|
|
|
|
TraceEvent("BlobWorkerHalted", bwInterf.id()).detail("ReqID", req.requesterID);
|
|
|
|
|
printf("BW %s was halted\n", bwInterf.id().toString().c_str());
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
when(wait(collection)) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("BW actor collection returned, exiting\n");
|
|
|
|
|
}
|
|
|
|
|
TraceEvent("BlobWorkerActorCollectionError");
|
|
|
|
|
ASSERT(false);
|
|
|
|
|
throw internal_error();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
} catch (Error& e) {
|
|
|
|
|
if (BW_DEBUG) {
|
|
|
|
|
printf("Blob worker got error %s, exiting\n", e.name());
|
|
|
|
|
printf("Blob worker got error %s. Exiting...\n", e.name());
|
|
|
|
|
}
|
|
|
|
|
TraceEvent("BlobWorkerDied", self.id).error(e, true);
|
|
|
|
|
throw e;
|
|
|
|
|
TraceEvent("BlobWorkerDied", self->id).error(e, true);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return Void();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TODO add unit tests for assign/revoke range, especially version ordering
|
|
|
|
|
|