fix: replication of placeholder filesystems (#744)
fixes https://github.com/zrepl/zrepl/issues/742 Before this PR, when chaining replication from A => B => C, if B had placeholders and the `filesystems` included these placeholders, we'd incorrectly fail the planning phase with error `sender does not have any versions`. The non-placeholder child filesystems of these placeholders would then fail to replicate because of the initial-replication-dependency-tracking that we do, i.e., their parent failed to initially replication, hence they fail to replicate as well (`parent(s) failed during initial replication`). We can do better than that because we have the information whether a sender-side filesystem is a placeholder. This PR makes the planner act on that information. The outcome is that placeholders are replicated as placeholders (albeit the receiver remains in control of how these placeholders are created, i.e., `recv.placeholders`) The mechanism to do it is: 1. Don't plan any replication steps for filesystems that are placeholders on the sender. 2. Ensure that, if a receiving-side filesystem exists, it is indeed a placeholder. Check (2) may seem overly restrictive, but, the goal here is not just to mirror all non-placeholder filesystems, but also to mirror the hierarchy. Testing performed: - [x] confirm with issue reporter that this PR fixes their issue - [x] add a regression test that fails without the changes in this PR
This commit is contained in:
committed by
GitHub
parent
440b07443f
commit
5615f4929a
@@ -529,27 +529,31 @@ func (f *fs) do(ctx context.Context, pq *stepQueue, prev *fs) {
|
||||
// find the highest of the previously uncompleted steps for which we can also find a step
|
||||
// in our current plan
|
||||
prevUncompleted := prev.planned.steps[prev.planned.step:]
|
||||
var target struct{ prev, cur int }
|
||||
target.prev = -1
|
||||
target.cur = -1
|
||||
out:
|
||||
for p := len(prevUncompleted) - 1; p >= 0; p-- {
|
||||
for q := len(f.planned.steps) - 1; q >= 0; q-- {
|
||||
if prevUncompleted[p].step.TargetEquals(f.planned.steps[q].step) {
|
||||
target.prev = p
|
||||
target.cur = q
|
||||
break out
|
||||
if len(prevUncompleted) == 0 || len(f.planned.steps) == 0 {
|
||||
f.debug("no steps planned in previous attempt or this attempt, no correlation necessary len(prevUncompleted)=%d len(f.planned.steps)=%d", len(prevUncompleted), len(f.planned.steps))
|
||||
} else {
|
||||
var target struct{ prev, cur int }
|
||||
target.prev = -1
|
||||
target.cur = -1
|
||||
out:
|
||||
for p := len(prevUncompleted) - 1; p >= 0; p-- {
|
||||
for q := len(f.planned.steps) - 1; q >= 0; q-- {
|
||||
if prevUncompleted[p].step.TargetEquals(f.planned.steps[q].step) {
|
||||
target.prev = p
|
||||
target.cur = q
|
||||
break out
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if target.prev == -1 || target.cur == -1 {
|
||||
f.debug("no correlation possible between previous attempt and this attempt's plan")
|
||||
f.planning.err = newTimedError(fmt.Errorf("cannot correlate previously failed attempt to current plan"), time.Now())
|
||||
return
|
||||
}
|
||||
if target.prev == -1 || target.cur == -1 {
|
||||
f.debug("no correlation possible between previous attempt and this attempt's plan")
|
||||
f.planning.err = newTimedError(fmt.Errorf("cannot correlate previously failed attempt to current plan"), time.Now())
|
||||
return
|
||||
}
|
||||
|
||||
f.planned.steps = f.planned.steps[0:target.cur]
|
||||
f.debug("found correlation, new steps are len(fs.planned.steps) = %d", len(f.planned.steps))
|
||||
f.planned.steps = f.planned.steps[0:target.cur]
|
||||
f.debug("found correlation, new steps are len(fs.planned.steps) = %d", len(f.planned.steps))
|
||||
}
|
||||
} else {
|
||||
f.debug("previous attempt does not exist or did not finish planning, no correlation possible, taking this attempt's plan as is")
|
||||
}
|
||||
@@ -600,6 +604,8 @@ func (f *fs) do(ctx context.Context, pq *stepQueue, prev *fs) {
|
||||
f.debug("parentHasNoSteps=%v parentFirstStepIsIncremental=%v parentHasTakenAtLeastOneSuccessfulStep=%v",
|
||||
parentHasNoSteps, parentFirstStepIsIncremental, parentHasTakenAtLeastOneSuccessfulStep)
|
||||
|
||||
// If the parent is a placeholder on the sender, `parentHasNoSteps` is true because we plan no steps for sender placeholders.
|
||||
// The receiver will create the necessary placeholders when they start receiving the first non-placeholder child filesystem.
|
||||
parentPresentOnReceiver := parentHasNoSteps || parentFirstStepIsIncremental || parentHasTakenAtLeastOneSuccessfulStep
|
||||
|
||||
allParentsPresentOnReceiver = allParentsPresentOnReceiver && parentPresentOnReceiver // no shadow
|
||||
|
||||
@@ -333,6 +333,22 @@ func (fs *Filesystem) doPlanning(ctx context.Context) ([]*Step, error) {
|
||||
|
||||
log(ctx).Debug("assessing filesystem")
|
||||
|
||||
if fs.senderFS.IsPlaceholder {
|
||||
log(ctx).Debug("sender filesystem is placeholder")
|
||||
if fs.receiverFS != nil {
|
||||
if fs.receiverFS.IsPlaceholder {
|
||||
// all good, fall through
|
||||
log(ctx).Debug("receiver filesystem is placeholder")
|
||||
} else {
|
||||
err := fmt.Errorf("sender filesystem is placeholder, but receiver filesystem is not")
|
||||
log(ctx).Error(err.Error())
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
log(ctx).Debug("no steps required for replicating placeholders, the endpoint.Receiver will create a placeholder when we receive the first non-placeholder child filesystem")
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
sfsvsres, err := fs.sender.ListFilesystemVersions(ctx, &pdu.ListFilesystemVersionsReq{Filesystem: fs.Path})
|
||||
if err != nil {
|
||||
log(ctx).WithError(err).Error("cannot get remote filesystem versions")
|
||||
|
||||
Reference in New Issue
Block a user