fix(replication): recheck queued delete-marker creations under the lock

A delete-marker creation task carries the marker as it looked when it
was queued by the DELETE handler, a GET/HEAD/LIST heal, the scanner or
MRF. Tasks from several frontends serialize on the per-object
replication lock, so a task can run after the user has purged that
marker and the purge has already reached the targets. The target no
longer holds the marker, so the queued creation recreated it there with
the original VersionID and mtime. This is the late-create sequence seen
in the 2026-09-16 three-site runs: purge 204 on both targets, then about
85 ms later a replicated creation for the same VersionID.

Re-read the source version under the replication lock before sending a
creation. A missing version, a non-marker version, or a version under
purge makes the task stale; it is dropped without touching the source.
A read that cannot confirm either way is retried through MRF instead of
being treated as absence.

Creations already on the wire, replays from other sites, and cleanup of
minority residue after a crash are not covered by this check.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Signed-off-by: Feng Ruohang <rh@vonng.com>
This commit is contained in:
Feng Ruohang
2026-09-16 22:35:08 +08:00
parent eb4f5e5b31
commit 254b19ac07
2 changed files with 328 additions and 0 deletions
+64
View File
@@ -504,6 +504,35 @@ func replicateDelete(ctx context.Context, dobj DeletedObjectReplicationInfo, obj
ctx = lkctx.Context()
defer lk.Unlock(lkctx)
if !isPurge && dobj.DeleteMarkerVersionID != "" {
// A creation task carries the marker as it looked when it was queued
// by the DELETE handler, a GET/HEAD/LIST heal, the scanner or MRF.
// While it waited for this lock, another frontend may have purged the
// marker and replicated that purge. The targets no longer hold the
// marker, so the queued creation would recreate it from the stale
// snapshot. Confirm the source version under the lock first.
switch deleteMarkerCreationState(ctx, objectAPI, dobj) {
case creationStale:
return replicatedInfos{}
case creationUnverified:
dobj.RetryCount++
globalReplicationPool.Get().queueMRFSave(dobj.ToMRFEntry())
sendEvent(eventArgs{
BucketName: bucket,
Object: ObjectInfo{
Bucket: bucket,
Name: dobj.ObjectName,
VersionID: versionID,
DeleteMarker: dobj.DeleteMarker,
},
UserAgent: "Internal: [Replication]",
Host: globalLocalNodeName,
EventName: event.ObjectReplicationNotTracked,
})
return replicatedInfos{}
}
}
rinfos := replicatedInfos{Targets: make([]replicatedTargetInfo, 0, len(dsc.targetsMap))}
var wg sync.WaitGroup
var mu sync.Mutex
@@ -1955,6 +1984,41 @@ func (di DeletedObjectReplicationInfo) isVersionPurge() bool {
return di.VersionID != "" || di.DeleteMarkerVersionID != "" && !di.VersionPurgeStatus().Empty()
}
// creationState is the outcome of re-reading a queued delete-marker creation
// against the source.
type creationState int
const (
creationCurrent creationState = iota
creationStale
creationUnverified
)
// deleteMarkerCreationState re-reads the marker a queued creation task refers
// to. The version being absent, no longer a marker, or under a version purge
// makes the creation stale. A failed read is not absence: the caller retries
// later instead of guessing.
func deleteMarkerCreationState(ctx context.Context, objectAPI ObjectLayer, dobj DeletedObjectReplicationInfo) creationState {
oi, err := objectAPI.GetObjectInfo(ctx, dobj.Bucket, dobj.ObjectName, ObjectOptions{
VersionID: dobj.DeleteMarkerVersionID,
Versioned: globalBucketVersioningSys.PrefixEnabled(dobj.Bucket, dobj.ObjectName),
VersionSuspended: globalBucketVersioningSys.Suspended(dobj.Bucket),
})
switch {
case isErrObjectNotFound(err), isErrVersionNotFound(err):
return creationStale
case err != nil && !isErrMethodNotAllowed(err):
return creationUnverified
}
if !oi.DeleteMarker || oi.VersionID != dobj.DeleteMarkerVersionID {
return creationStale
}
if !oi.VersionPurgeStatus.Empty() || oi.VersionPurgeStatusInternal != "" {
return creationStale
}
return creationCurrent
}
// Purge metadata uses COMPLETE; operation statistics and audit use COMPLETED.
func purgeReplicationStatus(status VersionPurgeStatusType) replication.StatusType {
if replication.StatusType(status) == replication.CompletedLegacy {